From c061637d2118f302e91516b264f5b2de4e84ab37 Mon Sep 17 00:00:00 2001 From: AlmaLinux RelEng Bot Date: Tue, 14 Apr 2026 05:59:10 -0400 Subject: [PATCH] import CS qemu-kvm-10.1.0-16.el10 --- .gitignore | 2 +- 0004-Initial-redhat-build.patch | 58 +- 0005-Enable-disable-devices-for-RHEL.patch | 652 +++++---- ...Machine-type-related-general-changes.patch | 497 +++---- 0007-Add-aarch64-machine-types.patch | 430 ------ ...temporarily-disable-Wunused-function.patch | 36 + 0008-Add-s390x-machine-types.patch | 273 ---- ...machine-types-for-aarch64-s390x-and-.patch | 102 ++ ...rsioned-machine-type-macros-for-RHEL.patch | 195 +++ 0009-Add-x86_64-machine-types.patch | 920 ------------ 0010-Enable-make-check.patch | 231 --- ...ease-deletion-schedule-to-4-releases.patch | 37 + ...aarch64-versioned-virt-machine-types.patch | 399 +++++ ...390x-versioned-s390-ccw-virtio-machi.patch | 262 ++++ ...86_64-versioned-pc-q35-machine-types.patch | 517 +++++++ ...o-net-pci-romfile-loading-on-riscv64.patch | 58 + 0015-Add-upstream-compatibility-bits.patch | 145 -- ...temporarily-disable-Wunused-function.patch | 32 + 0016-Disable-FDC-devices.patch | 29 - 0016-Enable-make-check.patch | 365 +++++ 0017-Disable-vga-cirrus-device.patch | 24 - ...mber-of-devices-that-can-be-assigned.patch | 97 +- ...Add-support-statement-to-help-output.patch | 8 +- ...documentation-instead-of-qemu-system.patch | 6 +- ...on-warning-when-opening-v2-images-rw.patch | 6 +- ...le-posix-Define-DM_MPATH_PROBE_PATHS.patch | 46 + kvm-Enable-vhost-user-scmi-devices.patch | 50 - kvm-Enable-vhost-user-snd-pci-device.patch | 50 - ...vfio-pci-device-s-enable-migration-o.patch | 37 + ...Add-maintainers-for-mshv-accelerator.patch | 54 + ...rt-Use-ACPI-PCI-hotplug-by-default-f.patch | 81 + ...t-monitor-use-aio_co_reschedule_self.patch | 67 - ...and-config-support-for-MSHV-accelera.patch | 131 ++ kvm-accel-mshv-Add-accelerator-skeleton.patch | 291 ++++ ...Add-vCPU-creation-and-execution-loop.patch | 417 ++++++ kvm-accel-mshv-Add-vCPU-signal-handling.patch | 75 + ...mshv-Handle-overlapping-mem-mappings.patch | 695 +++++++++ kvm-accel-mshv-Initialize-VM-partition.patch | 1302 +++++++++++++++++ ...shv-Register-memory-region-listeners.patch | 172 +++ kvm-accel-mshv-initialize-thread-name.patch | 39 + ...-return-value-of-handle_pio_str_read.patch | 42 + ...n-about-iohandler_ctx-special-casing.patch | 64 - ...rhel-10.2-specific-virt-machine-type.patch | 50 + ...-rhel-9.8-specific-virt-machine-type.patch | 49 + ...rm-fix-oob-access-in-compat-handling.patch | 44 + ...vm-report-registers-we-failed-to-set.patch | 154 ++ ...bles-test-Allow-for-smmuv3-test-data.patch | 54 + ...xpose-block-limits-for-images-in-QMP.patch | 255 ++++ ...x-BDS-use-after-free-during-shutdown.patch | 64 + ...lock-Improve-comments-in-BlockLimits.patch | 98 ++ ...-BLOCK_IO_ERROR-with-action-stop-for.patch | 98 ++ ...names-only-when-explicitly-requested.patch | 252 ---- ...x-race-when-resuming-queued-requests.patch | 73 + ...-Take-reqs_lock-for-tracked_requests.patch | 66 + ...void-potentially-getting-stuck-after.patch | 167 +++ kvm-curl-Fix-coroutine-waking.patch | 172 +++ kvm-docs-Add-mshv-to-documentation.patch | 143 ++ ...cs-add-SCSI-migrate-pr-documentation.patch | 118 ++ ...rash-from-legacy-interrupt-firing-af.patch | 69 + ...e-suspended-dm-multipath-better-for-.patch | 123 ++ kvm-fix-pc_rhel_10_2_compat_len.patch | 36 + ...on-Check-SMMU-has-PCIe-Root-Complex-.patch | 131 ++ ...hw-arm-virt-Add-an-SMMU_IO_LEN-macro.patch | 61 + ...w-user-creatable-SMMUv3-dev-instanti.patch | 215 +++ ...or-out-common-SMMUV3-dt-bindings-cod.patch | 118 ++ ...ACPI-PCI-hotplug-by-default-from-10..patch | 66 + ...i-build-Re-arrange-SMMUv3-IORT-build.patch | 291 ++++ ...-build-Update-IORT-for-multiple-smmu.patch | 170 +++ ...ze-APIC-helper-names-from-kvm_-to-ac.patch | 386 +++++ ...ix-ACCEL_KERNEL_GSI_IRQFD_POSSIBLE-t.patch | 68 + ...-pci_setup_iommu_per_bus-for-per-bus.patch | 150 ++ ...ossible-crash-with-passed-through-vi.patch | 83 ++ ...-uefi-add-variable-digest-to-vmstate.patch | 89 ++ ...perv-Add-MSHV-ABI-header-definitions.patch | 1264 ++++++++++++++++ ...ter-free-in-websocket-handshake-code.patch | 189 +++ ...ock-resource-release-to-close-method.patch | 84 ++ ...uring-Resubmit-tails-of-short-writes.patch | 278 ++++ ...t-store-data-file-with-protocol-in-i.patch | 61 - ...t-store-data-file-with-json-prefix-i.patch | 64 - kvm-iotests-test-NBD-TLS-iothread.patch | 276 ---- ...-Put-all-parameters-into-qemu_laiocb.patch | 127 ++ ...Resubmit-tails-of-short-reads-writes.patch | 215 +++ ...io-add-IO_CMD_FDSYNC-command-support.patch | 126 -- ...ux-headers-Update-to-Linux-v6.17-rc1.patch | 941 ++++++++++++ ...ders-deal-with-counted_by-annotation.patch | 47 + ...nux-headers-linux-Add-mshv.h-headers.patch | 329 +++++ ...d-dirty-bitmap-writes-during-startup.patch | 162 ++ ...ze-query-mshv-info-mshv-to-query-acc.patch | 235 +++ ...024-7409-Avoid-use-after-free-when-c.patch | 101 -- ...024-7409-Cap-default-max-connections.patch | 184 --- ...024-7409-Close-stray-clients-at-serv.patch | 173 --- ...024-7409-Drop-non-negotiating-client.patch | 134 -- ...negotiation-functions-as-coroutine_f.patch | 330 ----- ...-Plumb-in-new-args-to-nbd_client_add.patch | 175 --- ...-not-poll-within-a-coroutine-context.patch | 208 --- ...ndle-all-offloads-in-a-single-struct.patch | 366 +++++ ...ement-UDP-tunnel-features-offloading.patch | 209 +++ kvm-net-implement-tunnel-probing.patch | 317 ++++ ...broken-MMIO-accesses-from-SR-IOV-VFs.patch | 177 +++ ...pcie_sriov_pf_exit-safe-on-non-SR-IO.patch | 73 + kvm-q35-increase-default-tseg-size.patch | 63 + ...cel-Allow-to-query-mshv-capabilities.patch | 169 +++ ...0x-add-QAPI-event-SCLP_CPI_INFO_AVAI.patch | 87 ++ ...n-t-open-data_file-with-BDRV_O_NO_IO.patch | 117 -- kvm-qcow2-Fix-cache_clean_timer.patch | 338 +++++ ...-initialize-lock-in-invalidate_cache.patch | 45 + kvm-qemu-img-info-Add-cache-mode-option.patch | 153 ++ ...mg-info-Optionally-show-block-limits.patch | 240 +++ ...ns.hx-Document-the-arm-smmuv3-device.patch | 53 + ...erit-follow_coroutine_ctx-across-TLS.patch | 130 -- ...o-features-map-to-support-extended-f.patch | 329 +++++ ...-not-run-bios-tables-test-on-aarch64.patch | 30 + ...s-test-Add-tests-for-legacy-smmuv3-a.patch | 160 ++ ...s-test-Update-tables-for-smmuv3-test.patch | 282 ++++ ...est-do-not-use-the-obsolete-pentium-.patch | 46 - ...utes-Unify-the-retrieval-of-the-bloc.patch | 47 + ...utes-fix-interaction-with-hugetlb-me.patch | 125 ++ ...-BH-CB-in-the-coroutine-s-AioContext.patch | 127 ++ ...hel9.8.0-and-rhel10.2.0-machine-type.patch | 99 ++ ...hat-allow-5-level-paging-for-TDX-VMs.patch | 43 + ...configs-enable-CONFIG_TDX-for-x86_64.patch | 37 + kvm-rh-enable-CONFIG_USB_STORAGE_BOT.patch | 62 + ...ne-type-compat-for-virtio-gpu-migrat.patch | 36 - ...remove-deprecated-rhel-machine-types.patch | 164 --- ...orrect-components-for-no-board-build.patch | 41 - ...si-add-error-reporting-to-scsi_SG_IO.patch | 131 ++ ...ze-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch | 117 ++ ...csi-save-load-SCSI-reservation-state.patch | 348 +++++ ...reservation-state-for-live-migration.patch | 331 +++++ ...s-x86-Remove-the-existing-deprecated.patch | 62 - ...compatibility-property-for-arch_capa.patch | 135 ++ ...compatibility-property-for-pdcm-feat.patch | 115 ++ ...ate-Allow-instruction-decoding-from-.patch | 150 ++ ...mshv-Add-CPU-create-and-remove-logic.patch | 76 + ...v-Add-x86-decoder-emu-implementation.patch | 431 ++++++ ...hv-Implement-mshv_arch_put_registers.patch | 319 ++++ ...mshv-Implement-mshv_get_special_regs.patch | 173 +++ ...shv-Implement-mshv_get_standard_regs.patch | 182 +++ ...-i386-mshv-Implement-mshv_store_regs.patch | 188 +++ ...et-i386-mshv-Implement-mshv_vcpu_run.patch | 489 +++++++ ...-Integrate-x86-instruction-decoder-e.patch | 283 ++++ ...shv-Register-CPUID-entries-with-MSHV.patch | 252 ++++ ...et-i386-mshv-Register-MSRs-with-MSHV.patch | 528 +++++++ ...-Set-local-interrupt-controller-stat.patch | 183 +++ ...shv-Use-preallocated-page-for-hvcall.patch | 193 +++ ...86-mshv-Write-MSRs-to-the-hypervisor.patch | 113 ++ ...-a-CONFIG-switch-to-disable-legacy-C.patch | 116 -- ...ert-the-old-s390x-CPU-model-disablem.patch | 66 - ..._models-Disable-everything-up-to-the.patch | 56 - ...ctional-add-tests-for-SCLP-event-CPI.patch | 76 + ...libqtest-add-qtest_has_cpu_model-api.patch | 162 -- ...check-for-availability-of-older-cpu-.patch | 359 ----- ...qemu_wait_io_event-qemu_wait_io_even.patch | 219 +++ ...-VFIO-migration-with-MultiFD-support.patch | 47 + ...region-info-cache-for-initial-region.patch | 83 ++ ...-rename-field-to-num_initial_regions.patch | 253 ++++ ...rt-for-negotiating-extended-features.patch | 292 ++++ ...-implement-extended-features-support.patch | 133 ++ ...-implement-extended-features-support.patch | 237 +++ ...ake-vhost_set_vring_file-synchronous.patch | 123 ++ ...rt-for-negotiating-extended-features.patch | 133 ++ kvm-virtio-gpu-fix-v2-migration.patch | 122 -- ...tio-introduce-extended-features-type.patch | 201 +++ ...-implement-extended-features-support.patch | 331 +++++ ...lement-support-for-extended-features.patch | 196 +++ ...io-serialize-extended-features-state.patch | 222 +++ ...e-cpu-models-that-do-not-support-x86.patch | 98 -- ...eprecation-string-to-match-lowest-un.patch | 38 - ...hel-10.2-specific-pc-q35-machine-typ.patch | 56 + ...hel-9.8-specific-pc-q35-machine-type.patch | 56 + qemu-ga.sysconfig | 2 +- qemu-kvm.spec | 1101 ++++++++++++-- sources | 2 +- 173 files changed, 25653 insertions(+), 6701 deletions(-) delete mode 100644 0007-Add-aarch64-machine-types.patch create mode 100644 0007-meson-temporarily-disable-Wunused-function.patch delete mode 100644 0008-Add-s390x-machine-types.patch create mode 100644 0008-Remove-upstream-machine-types-for-aarch64-s390x-and-.patch create mode 100644 0009-Adapt-versioned-machine-type-macros-for-RHEL.patch delete mode 100644 0009-Add-x86_64-machine-types.patch delete mode 100644 0010-Enable-make-check.patch create mode 100644 0010-Increase-deletion-schedule-to-4-releases.patch create mode 100644 0011-Add-downstream-aarch64-versioned-virt-machine-types.patch create mode 100644 0012-Add-downstream-s390x-versioned-s390-ccw-virtio-machi.patch create mode 100644 0013-Add-downstream-x86_64-versioned-pc-q35-machine-types.patch create mode 100644 0014-Disable-virtio-net-pci-romfile-loading-on-riscv64.patch delete mode 100644 0015-Add-upstream-compatibility-bits.patch create mode 100644 0015-Revert-meson-temporarily-disable-Wunused-function.patch delete mode 100644 0016-Disable-FDC-devices.patch create mode 100644 0016-Enable-make-check.patch delete mode 100644 0017-Disable-vga-cirrus-device.patch rename 0011-vfio-cap-number-of-devices-that-can-be-assigned.patch => 0017-vfio-cap-number-of-devices-that-can-be-assigned.patch (56%) rename 0012-Add-support-statement-to-help-output.patch => 0018-Add-support-statement-to-help-output.patch (85%) rename 0013-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch => 0019-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch (93%) rename 0014-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch => 0020-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch (94%) create mode 100644 0021-file-posix-Define-DM_MPATH_PROBE_PATHS.patch delete mode 100644 kvm-Enable-vhost-user-scmi-devices.patch delete mode 100644 kvm-Enable-vhost-user-snd-pci-device.patch create mode 100644 kvm-Fix-the-typo-of-vfio-pci-device-s-enable-migration-o.patch create mode 100644 kvm-MAINTAINERS-Add-maintainers-for-mshv-accelerator.patch create mode 100644 kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch delete mode 100644 kvm-Revert-monitor-use-aio_co_reschedule_self.patch create mode 100644 kvm-accel-Add-Meson-and-config-support-for-MSHV-accelera.patch create mode 100644 kvm-accel-mshv-Add-accelerator-skeleton.patch create mode 100644 kvm-accel-mshv-Add-vCPU-creation-and-execution-loop.patch create mode 100644 kvm-accel-mshv-Add-vCPU-signal-handling.patch create mode 100644 kvm-accel-mshv-Handle-overlapping-mem-mappings.patch create mode 100644 kvm-accel-mshv-Initialize-VM-partition.patch create mode 100644 kvm-accel-mshv-Register-memory-region-listeners.patch create mode 100644 kvm-accel-mshv-initialize-thread-name.patch create mode 100644 kvm-accel-mshv-use-return-value-of-handle_pio_str_read.patch delete mode 100644 kvm-aio-warn-about-iohandler_ctx-special-casing.patch create mode 100644 kvm-arm-create-new-rhel-10.2-specific-virt-machine-type.patch create mode 100644 kvm-arm-create-new-rhel-9.8-specific-virt-machine-type.patch create mode 100644 kvm-arm-fix-oob-access-in-compat-handling.patch create mode 100644 kvm-arm-kvm-report-registers-we-failed-to-set.patch create mode 100644 kvm-bios-tables-test-Allow-for-smmuv3-test-data.patch create mode 100644 kvm-block-Expose-block-limits-for-images-in-QMP.patch create mode 100644 kvm-block-Fix-BDS-use-after-free-during-shutdown.patch create mode 100644 kvm-block-Improve-comments-in-BlockLimits.patch create mode 100644 kvm-block-Never-drop-BLOCK_IO_ERROR-with-action-stop-for.patch delete mode 100644 kvm-block-Parse-filenames-only-when-explicitly-requested.patch create mode 100644 kvm-block-backend-Fix-race-when-resuming-queued-requests.patch create mode 100644 kvm-block-io-Take-reqs_lock-for-tracked_requests.patch create mode 100644 kvm-block-io_uring-avoid-potentially-getting-stuck-after.patch create mode 100644 kvm-curl-Fix-coroutine-waking.patch create mode 100644 kvm-docs-Add-mshv-to-documentation.patch create mode 100644 kvm-docs-add-SCSI-migrate-pr-documentation.patch create mode 100644 kvm-e1000e-Prevent-crash-from-legacy-interrupt-firing-af.patch create mode 100644 kvm-file-posix-Handle-suspended-dm-multipath-better-for-.patch create mode 100644 kvm-fix-pc_rhel_10_2_compat_len.patch create mode 100644 kvm-hw-arm-smmu-common-Check-SMMU-has-PCIe-Root-Complex-.patch create mode 100644 kvm-hw-arm-virt-Add-an-SMMU_IO_LEN-macro.patch create mode 100644 kvm-hw-arm-virt-Allow-user-creatable-SMMUv3-dev-instanti.patch create mode 100644 kvm-hw-arm-virt-Factor-out-common-SMMUV3-dt-bindings-cod.patch create mode 100644 kvm-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-from-10..patch create mode 100644 kvm-hw-arm-virt-acpi-build-Re-arrange-SMMUv3-IORT-build.patch create mode 100644 kvm-hw-arm-virt-acpi-build-Update-IORT-for-multiple-smmu.patch create mode 100644 kvm-hw-intc-Generalize-APIC-helper-names-from-kvm_-to-ac.patch create mode 100644 kvm-hw-intc-ioapic-Fix-ACCEL_KERNEL_GSI_IRQFD_POSSIBLE-t.patch create mode 100644 kvm-hw-pci-Introduce-pci_setup_iommu_per_bus-for-per-bus.patch create mode 100644 kvm-hw-s390x-Fix-a-possible-crash-with-passed-through-vi.patch create mode 100644 kvm-hw-uefi-add-variable-digest-to-vmstate.patch create mode 100644 kvm-include-hw-hyperv-Add-MSHV-ABI-header-definitions.patch create mode 100644 kvm-io-fix-use-after-free-in-websocket-handshake-code.patch create mode 100644 kvm-io-move-websock-resource-release-to-close-method.patch create mode 100644 kvm-io-uring-Resubmit-tails-of-short-writes.patch delete mode 100644 kvm-iotests-244-Don-t-store-data-file-with-protocol-in-i.patch delete mode 100644 kvm-iotests-270-Don-t-store-data-file-with-json-prefix-i.patch delete mode 100644 kvm-iotests-test-NBD-TLS-iothread.patch create mode 100644 kvm-linux-aio-Put-all-parameters-into-qemu_laiocb.patch create mode 100644 kvm-linux-aio-Resubmit-tails-of-short-reads-writes.patch delete mode 100644 kvm-linux-aio-add-IO_CMD_FDSYNC-command-support.patch create mode 100644 kvm-linux-headers-Update-to-Linux-v6.17-rc1.patch create mode 100644 kvm-linux-headers-deal-with-counted_by-annotation.patch create mode 100644 kvm-linux-headers-linux-Add-mshv.h-headers.patch create mode 100644 kvm-mirror-Fix-missed-dirty-bitmap-writes-during-startup.patch create mode 100644 kvm-monitor-generalize-query-mshv-info-mshv-to-query-acc.patch delete mode 100644 kvm-nbd-server-CVE-2024-7409-Avoid-use-after-free-when-c.patch delete mode 100644 kvm-nbd-server-CVE-2024-7409-Cap-default-max-connections.patch delete mode 100644 kvm-nbd-server-CVE-2024-7409-Close-stray-clients-at-serv.patch delete mode 100644 kvm-nbd-server-CVE-2024-7409-Drop-non-negotiating-client.patch delete mode 100644 kvm-nbd-server-Mark-negotiation-functions-as-coroutine_f.patch delete mode 100644 kvm-nbd-server-Plumb-in-new-args-to-nbd_client_add.patch delete mode 100644 kvm-nbd-server-do-not-poll-within-a-coroutine-context.patch create mode 100644 kvm-net-bundle-all-offloads-in-a-single-struct.patch create mode 100644 kvm-net-implement-UDP-tunnel-features-offloading.patch create mode 100644 kvm-net-implement-tunnel-probing.patch create mode 100644 kvm-pcie_sriov-Fix-broken-MMIO-accesses-from-SR-IOV-VFs.patch create mode 100644 kvm-pcie_sriov-make-pcie_sriov_pf_exit-safe-on-non-SR-IO.patch create mode 100644 kvm-q35-increase-default-tseg-size.patch create mode 100644 kvm-qapi-accel-Allow-to-query-mshv-capabilities.patch create mode 100644 kvm-qapi-machine-s390x-add-QAPI-event-SCLP_CPI_INFO_AVAI.patch delete mode 100644 kvm-qcow2-Don-t-open-data_file-with-BDRV_O_NO_IO.patch create mode 100644 kvm-qcow2-Fix-cache_clean_timer.patch create mode 100644 kvm-qcow2-Re-initialize-lock-in-invalidate_cache.patch create mode 100644 kvm-qemu-img-info-Add-cache-mode-option.patch create mode 100644 kvm-qemu-img-info-Optionally-show-block-limits.patch create mode 100644 kvm-qemu-options.hx-Document-the-arm-smmuv3-device.patch delete mode 100644 kvm-qio-Inherit-follow_coroutine_ctx-across-TLS.patch create mode 100644 kvm-qmp-update-virtio-features-map-to-support-extended-f.patch create mode 100644 kvm-qtest-Do-not-run-bios-tables-test-on-aarch64.patch create mode 100644 kvm-qtest-bios-tables-test-Add-tests-for-legacy-smmuv3-a.patch create mode 100644 kvm-qtest-bios-tables-test-Update-tables-for-smmuv3-test.patch delete mode 100644 kvm-qtest-x86-numa-test-do-not-use-the-obsolete-pentium-.patch create mode 100644 kvm-ram-block-attributes-Unify-the-retrieval-of-the-bloc.patch create mode 100644 kvm-ram-block-attributes-fix-interaction-with-hugetlb-me.patch create mode 100644 kvm-rbd-Run-co-BH-CB-in-the-coroutine-s-AioContext.patch create mode 100644 kvm-redhat-Add-new-rhel9.8.0-and-rhel10.2.0-machine-type.patch create mode 100644 kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch create mode 100644 kvm-rh-configs-enable-CONFIG_TDX-for-x86_64.patch create mode 100644 kvm-rh-enable-CONFIG_USB_STORAGE_BOT.patch delete mode 100644 kvm-rhel-9.4.0-machine-type-compat-for-virtio-gpu-migrat.patch delete mode 100644 kvm-s390x-remove-deprecated-rhel-machine-types.patch delete mode 100644 kvm-s390x-select-correct-components-for-no-board-build.patch create mode 100644 kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch create mode 100644 kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch create mode 100644 kvm-scsi-save-load-SCSI-reservation-state.patch create mode 100644 kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch delete mode 100644 kvm-target-cpu-models-x86-Remove-the-existing-deprecated.patch create mode 100644 kvm-target-i386-add-compatibility-property-for-arch_capa.patch create mode 100644 kvm-target-i386-add-compatibility-property-for-pdcm-feat.patch create mode 100644 kvm-target-i386-emulate-Allow-instruction-decoding-from-.patch create mode 100644 kvm-target-i386-mshv-Add-CPU-create-and-remove-logic.patch create mode 100644 kvm-target-i386-mshv-Add-x86-decoder-emu-implementation.patch create mode 100644 kvm-target-i386-mshv-Implement-mshv_arch_put_registers.patch create mode 100644 kvm-target-i386-mshv-Implement-mshv_get_special_regs.patch create mode 100644 kvm-target-i386-mshv-Implement-mshv_get_standard_regs.patch create mode 100644 kvm-target-i386-mshv-Implement-mshv_store_regs.patch create mode 100644 kvm-target-i386-mshv-Implement-mshv_vcpu_run.patch create mode 100644 kvm-target-i386-mshv-Integrate-x86-instruction-decoder-e.patch create mode 100644 kvm-target-i386-mshv-Register-CPUID-entries-with-MSHV.patch create mode 100644 kvm-target-i386-mshv-Register-MSRs-with-MSHV.patch create mode 100644 kvm-target-i386-mshv-Set-local-interrupt-controller-stat.patch create mode 100644 kvm-target-i386-mshv-Use-preallocated-page-for-hvcall.patch create mode 100644 kvm-target-i386-mshv-Write-MSRs-to-the-hypervisor.patch delete mode 100644 kvm-target-s390x-Add-a-CONFIG-switch-to-disable-legacy-C.patch delete mode 100644 kvm-target-s390x-Revert-the-old-s390x-CPU-model-disablem.patch delete mode 100644 kvm-target-s390x-cpu_models-Disable-everything-up-to-the.patch create mode 100644 kvm-tests-functional-add-tests-for-SCLP-event-CPI.patch delete mode 100644 kvm-tests-qtest-libqtest-add-qtest_has_cpu_model-api.patch delete mode 100644 kvm-tests-qtest-x86-check-for-availability-of-older-cpu-.patch create mode 100644 kvm-treewide-rename-qemu_wait_io_event-qemu_wait_io_even.patch create mode 100644 kvm-vfio-Disable-VFIO-migration-with-MultiFD-support.patch create mode 100644 kvm-vfio-only-check-region-info-cache-for-initial-region.patch create mode 100644 kvm-vfio-rename-field-to-num_initial_regions.patch create mode 100644 kvm-vhost-add-support-for-negotiating-extended-features.patch create mode 100644 kvm-vhost-backend-implement-extended-features-support.patch create mode 100644 kvm-vhost-net-implement-extended-features-support.patch create mode 100644 kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch create mode 100644 kvm-virtio-add-support-for-negotiating-extended-features.patch delete mode 100644 kvm-virtio-gpu-fix-v2-migration.patch create mode 100644 kvm-virtio-introduce-extended-features-type.patch create mode 100644 kvm-virtio-net-implement-extended-features-support.patch create mode 100644 kvm-virtio-pci-implement-support-for-extended-features.patch create mode 100644 kvm-virtio-serialize-extended-features-state.patch delete mode 100644 kvm-x86-cpu-deprecate-cpu-models-that-do-not-support-x86.patch delete mode 100644 kvm-x86-cpu-update-deprecation-string-to-match-lowest-un.patch create mode 100644 kvm-x86-create-new-rhel-10.2-specific-pc-q35-machine-typ.patch create mode 100644 kvm-x86-create-new-rhel-9.8-specific-pc-q35-machine-type.patch diff --git a/.gitignore b/.gitignore index c9ea5f4..0d37e27 100644 --- a/.gitignore +++ b/.gitignore @@ -1 +1 @@ -qemu-9.0.0.tar.xz +qemu-10.1.0.tar.xz diff --git a/0004-Initial-redhat-build.patch b/0004-Initial-redhat-build.patch index d17ded0..b2d2860 100644 --- a/0004-Initial-redhat-build.patch +++ b/0004-Initial-redhat-build.patch @@ -1,4 +1,4 @@ -From 91262ecfbd218a95dab8491e4226674f79debf5a Mon Sep 17 00:00:00 2001 +From 8a5eef9fcb74b2fa82ac6122caf3c3d38a26b195 Mon Sep 17 00:00:00 2001 From: Miroslav Rezanina Date: Wed, 26 May 2021 10:56:02 +0200 Subject: Initial redhat build @@ -13,25 +13,63 @@ several issues are fixed in QEMU tree: We disable make check due to issues with some of the tests. +We are rebasing from qemu-kvm-10.0.0-12.el10. + Signed-off-by: Miroslav Rezanina + +--- +Rebase notes (9.1.0): +- Remove --disable-block-migration and --disable-pvrdma configure options (upstream) +- Removed --disable-avx512f configure option +- Removed qemu-vsmr-helper (changed upstream) + +Rebase notes (10.0.0): +- Split --disable-sanitazers configure option (upstream change) +- Removed s390x-netboot.img (upstream) +- accel-tcg module no longer built (upstream) +- Removed new upstream npcm8xx board rom +- Not package hw-uefi-vars.so +- Not package pnv-pnor.bin on build +- Include riscv support + +Rebase notes (10.1.0 rc0): +- Remove avocado tests installation (removed upstream) +- dtb files installed in special directory (upstream change) +- Remove ast27x0_bootrom.bin +- Removed --disable-avx* configure options +- Removed 32bit archs conditionals +- Conditional qemu-kvm-block-rbd requirement +- Flip ipxe roms and seavgabios arch condition + +Merged patches (9.1.0): +- b206b8f7cb redhat: Remove the s390-netboot.img from the spec file +- 95605107f1 Require new dtrace package + +Merged patches (10.0.0): +- 07c8c9b9ff qemu-guest-agent: Update the logfile path of qga-fsfreeze-hook.log +- 1f54babd2a Recommend systemtap-client from qemu-tools +- 3e4d2a0fb8 Also recommend systemtap-devel from qemu-tools + +Merged patches (10.1.0 rc0): +- 72119e03ea distro: add an explicit valgrind-devel build dep --- .distro/Makefile | 101 ++ .distro/Makefile.common | 42 + .distro/README.tests | 39 + .distro/modules-load.conf | 4 + .distro/qemu-guest-agent.service | 1 - - .distro/qemu-kvm.spec.template | 1250 +++++++++++++++++++++++ + .distro/qemu-kvm.spec.template | 1795 +++++++++++++++++++++++ .distro/rpminspect.yaml | 6 +- .distro/scripts/extract_build_cmd.py | 12 + .distro/scripts/frh.py | 4 +- - .distro/scripts/process-patches.sh | 4 + + .distro/scripts/process-patches.sh | 6 +- .gitignore | 1 + README.systemtap | 43 + - scripts/qemu-guest-agent/fsfreeze-hook | 2 +- + scripts/qemu-guest-agent/fsfreeze-hook | 4 +- scripts/systemtap/conf.d/qemu_kvm.conf | 4 + scripts/systemtap/script.d/qemu_kvm.stp | 1 + ui/vnc-auth-sasl.c | 2 +- - 16 files changed, 1510 insertions(+), 6 deletions(-) + 16 files changed, 2057 insertions(+), 8 deletions(-) create mode 100644 .distro/Makefile create mode 100644 .distro/Makefile.common create mode 100644 .distro/README.tests @@ -91,14 +129,16 @@ index 0000000000..ad913fc990 +3. Translate the trace record to readable format. + # /usr/share/qemu-kvm/simpletrace.py --no-header /usr/share/qemu-kvm/trace-events /tmp/trace.log diff --git a/scripts/qemu-guest-agent/fsfreeze-hook b/scripts/qemu-guest-agent/fsfreeze-hook -index 13aafd4845..e9b84ec028 100755 +index c1feb6f5ce..d5d8d4daf8 100755 --- a/scripts/qemu-guest-agent/fsfreeze-hook +++ b/scripts/qemu-guest-agent/fsfreeze-hook -@@ -8,7 +8,7 @@ +@@ -7,8 +7,8 @@ + # "freeze" argument before the filesystem is frozen. And for fsfreeze-thaw # request, it is issued with "thaw" argument after filesystem is thawed. - LOGFILE=/var/log/qga-fsfreeze-hook.log +-LOGFILE=/var/log/qga-fsfreeze-hook.log -FSFREEZE_D=$(dirname -- "$0")/fsfreeze-hook.d ++LOGFILE=/var/log/qemu-ga/qga-fsfreeze-hook.log +FSFREEZE_D=$(dirname -- "$(realpath $0)")/fsfreeze-hook.d # Check whether file $1 is a backup or rpm-generated file and should be ignored @@ -121,7 +161,7 @@ index 0000000000..c04abf9449 @@ -0,0 +1 @@ +probe qemu.kvm.simpletrace.handle_qmp_command,qemu.kvm.simpletrace.monitor_protocol_*,qemu.kvm.simpletrace.migrate_set_state {} diff --git a/ui/vnc-auth-sasl.c b/ui/vnc-auth-sasl.c -index 47fdae5b21..2a950caa2a 100644 +index 3f4cfc471d..09dafba18d 100644 --- a/ui/vnc-auth-sasl.c +++ b/ui/vnc-auth-sasl.c @@ -42,7 +42,7 @@ diff --git a/0005-Enable-disable-devices-for-RHEL.patch b/0005-Enable-disable-devices-for-RHEL.patch index a748d94..222f261 100644 --- a/0005-Enable-disable-devices-for-RHEL.patch +++ b/0005-Enable-disable-devices-for-RHEL.patch @@ -1,4 +1,4 @@ -From 8e767ade83e18995692d3554b6b71c9e15b51d89 Mon Sep 17 00:00:00 2001 +From 03cf16ca98c4ed835c4da8c4424bfac5a9ae3aa6 Mon Sep 17 00:00:00 2001 From: Miroslav Rezanina Date: Wed, 7 Dec 2022 03:05:48 -0500 Subject: Enable/disable devices for RHEL @@ -6,56 +6,102 @@ Subject: Enable/disable devices for RHEL This commit adds all changes related to changes in supported devices. Signed-off-by: Miroslav Rezanina + --- - .distro/qemu-kvm.spec.template | 18 +-- - .../aarch64-softmmu/aarch64-rh-devices.mak | 42 +++++++ - .../ppc64-softmmu/ppc64-rh-devices.mak | 37 ++++++ +Rebase notes (9.1.0): +- Return value added for kvm_s390_apply_cpu_model +- Added new USB_HID and USB_HUB options +- Fixing valid_cpu_types preprocessing +- Moved x86 cpu deprecation from x86 machine type patch +- Removed unnecessary chunk in cirrus_vga.c +- Not needed hack removal of cpu-v7m.c from build +- Remove ppc64 device configuration +- Remove unnecessary chunks +- Removed CONFIG_VHOST_USER_SCMI and CONFIG_VHOST_USER_SND from some archs + +Rebase notes (10.0.0): +- Added CONFIG_PCI_BRIDGE for aarch64 and x86_64 (new upstream) +- Do not add deprecation_note member as it was added upstream (target/arm/cpu.h) +- Rename CONFIG_ARM_GICV3_TCG to CONFIG_ARM_GICV3 +- Remove deprecated line change for code commented out +- Do not change minimal revision for piix4 +- Remove YongFeng vcpu +- Add rebase devices changes +- Enable virtio-mem on s390x + +Rebase notes (10.1.0 rc0): +- Improved riscv cpu chunk re-added +- Comment out unused code + +Merged commits (9.1.0): +- f24c7a1fee Disable FDC devices +- fe8c6cb1ce Disable vga-cirrus device +- fccd117a12 Enable vhost-user-snd-pci device +- c0b40cc648 target/cpu-models/x86: Remove the existing deprecated CPU models on c10s +- ce42b3da0e x86/cpu: deprecate cpu models that do not support x86-64-v3 +- 01ffa96c3b target/s390x/cpu_models: Disable everything up to the z12 CPU model +- cd57d17e3c target/s390x: Revert the old s390x CPU model disablement code +- 42af7b3ad5 Enable vhost-user-scmi devices +- aa374ce5ea x86/cpu: update deprecation string to match lowest undeprecated model + +Merged commits (10.1.0 rc0): +- 312cdc116e Enable vhost-user-gpu-pci for RHIVOS +- be460986c1 Enable amd-iommu device + +Merged commits (10.1.0 rc2): +- 6306605028 Declare rtl8139 as deprecated + +Merged commits (10.1.0 rc3): +- f06f55a179 Enable uefi variable service for edk2 +--- + .distro/qemu-kvm.spec.template | 20 +-- + .../aarch64-softmmu/aarch64-rh-devices.mak | 49 ++++++++ configs/devices/rh-virtio.mak | 10 ++ - .../s390x-softmmu/s390x-rh-devices.mak | 19 +++ - .../x86_64-softmmu/x86_64-rh-devices.mak | 112 ++++++++++++++++++ - hw/arm/virt.c | 2 + - hw/block/fdc.c | 10 ++ - hw/cpu/meson.build | 3 +- + .../riscv64-softmmu/riscv64-rh-devices.mak | 39 ++++++ + .../s390x-softmmu/s390x-rh-devices.mak | 20 +++ + .../x86_64-softmmu/x86_64-rh-devices.mak | 118 ++++++++++++++++++ + hw/arm/virt.c | 4 + hw/cxl/meson.build | 3 +- - hw/display/cirrus_vga.c | 4 + hw/ide/piix.c | 5 +- hw/input/pckbd.c | 2 + hw/net/e1000.c | 2 + + hw/net/rtl8139.c | 4 + hw/usb/meson.build | 2 +- hw/virtio/meson.build | 6 +- target/arm/arm-qmp-cmds.c | 2 + - target/arm/cpu.c | 4 + - target/arm/cpu.h | 3 + - target/arm/cpu64.c | 12 +- + target/arm/cpu.h | 2 + + target/arm/cpu64.c | 7 +- target/arm/tcg/cpu32.c | 2 + target/arm/tcg/cpu64.c | 8 ++ target/arm/tcg/meson.build | 2 +- - target/s390x/cpu_models_sysemu.c | 3 + - target/s390x/kvm/kvm.c | 8 ++ + target/i386/cpu.c | 20 +++ + target/riscv/cpu.c | 6 + + target/s390x/cpu_models.c | 2 +- tests/qtest/arm-cpu-features.c | 4 + - 26 files changed, 309 insertions(+), 16 deletions(-) + 24 files changed, 323 insertions(+), 16 deletions(-) create mode 100644 configs/devices/aarch64-softmmu/aarch64-rh-devices.mak - create mode 100644 configs/devices/ppc64-softmmu/ppc64-rh-devices.mak create mode 100644 configs/devices/rh-virtio.mak + create mode 100644 configs/devices/riscv64-softmmu/riscv64-rh-devices.mak create mode 100644 configs/devices/s390x-softmmu/s390x-rh-devices.mak create mode 100644 configs/devices/x86_64-softmmu/x86_64-rh-devices.mak diff --git a/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak new file mode 100644 -index 0000000000..b0191d3c69 +index 0000000000..855278f70e --- /dev/null +++ b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak -@@ -0,0 +1,42 @@ +@@ -0,0 +1,49 @@ +include ../rh-virtio.mak + +CONFIG_ARM_GIC_KVM=y -+CONFIG_ARM_GICV3_TCG=y ++CONFIG_ARM_GICV3=y +CONFIG_ARM_GIC=y +CONFIG_ARM_SMMUV3=y +CONFIG_ARM_VIRT=y +CONFIG_CXL=y +CONFIG_CXL_MEM_DEVICE=y +CONFIG_EDID=y ++CONFIG_PCI_BRIDGE=y +CONFIG_PCIE_PORT=y +CONFIG_PCIE_PCI_BRIDGE=y +CONFIG_PCI_DEVICES=y @@ -68,6 +114,8 @@ index 0000000000..b0191d3c69 +CONFIG_USB_XHCI_PCI=y +CONFIG_USB_STORAGE_CORE=y +CONFIG_USB_STORAGE_CLASSIC=y ++CONFIG_USB_HUB=y ++CONFIG_USB_HID=y +CONFIG_VFIO=y +CONFIG_VFIO_PCI=y +CONFIG_VIRTIO_MMIO=y @@ -88,49 +136,10 @@ index 0000000000..b0191d3c69 +CONFIG_VHOST_USER_VSOCK=y +CONFIG_VHOST_USER_FS=y +CONFIG_IOMMUFD=y -diff --git a/configs/devices/ppc64-softmmu/ppc64-rh-devices.mak b/configs/devices/ppc64-softmmu/ppc64-rh-devices.mak -new file mode 100644 -index 0000000000..dbb7d30829 ---- /dev/null -+++ b/configs/devices/ppc64-softmmu/ppc64-rh-devices.mak -@@ -0,0 +1,37 @@ -+include ../rh-virtio.mak -+ -+CONFIG_DIMM=y -+CONFIG_MEM_DEVICE=y -+CONFIG_NVDIMM=y -+CONFIG_PCI=y -+CONFIG_PCI_DEVICES=y -+CONFIG_PCI_TESTDEV=y -+CONFIG_PCI_EXPRESS=y -+CONFIG_PSERIES=y -+CONFIG_SCSI=y -+CONFIG_SPAPR_VSCSI=y -+CONFIG_TEST_DEVICES=y -+CONFIG_USB=y -+CONFIG_USB_OHCI=y -+CONFIG_USB_OHCI_PCI=y -+CONFIG_USB_SMARTCARD=y -+CONFIG_USB_STORAGE_CORE=y -+CONFIG_USB_STORAGE_CLASSIC=y -+CONFIG_USB_XHCI=y -+CONFIG_USB_XHCI_NEC=y -+CONFIG_USB_XHCI_PCI=y -+CONFIG_VFIO=y -+CONFIG_VFIO_PCI=y -+CONFIG_VGA=y -+CONFIG_VGA_PCI=y -+CONFIG_VHOST_USER=y -+CONFIG_VIRTIO_PCI=y -+CONFIG_VIRTIO_VGA=y -+CONFIG_WDT_IB6300ESB=y -+CONFIG_XICS=y -+CONFIG_XIVE=y -+CONFIG_TPM=y -+CONFIG_TPM_SPAPR=y -+CONFIG_TPM_EMULATOR=y -+CONFIG_VHOST_VSOCK=y -+CONFIG_VHOST_USER_VSOCK=y ++CONFIG_VHOST_USER_SND=y ++CONFIG_VHOST_USER_SCMI=y ++CONFIG_VHOST_USER_GPU=y ++CONFIG_UEFI_VARS=y diff --git a/configs/devices/rh-virtio.mak b/configs/devices/rh-virtio.mak new file mode 100644 index 0000000000..94ede1b5f6 @@ -147,12 +156,57 @@ index 0000000000..94ede1b5f6 +CONFIG_VIRTIO_RNG=y +CONFIG_VIRTIO_SCSI=y +CONFIG_VIRTIO_SERIAL=y +diff --git a/configs/devices/riscv64-softmmu/riscv64-rh-devices.mak b/configs/devices/riscv64-softmmu/riscv64-rh-devices.mak +new file mode 100644 +index 0000000000..b5e55de916 +--- /dev/null ++++ b/configs/devices/riscv64-softmmu/riscv64-rh-devices.mak +@@ -0,0 +1,39 @@ ++include ../rh-virtio.mak ++ ++CONFIG_RISCV_VIRT=y ++CONFIG_CXL=y ++CONFIG_CXL_MEM_DEVICE=y ++CONFIG_EDID=y ++CONFIG_PCI_BRIDGE=y ++CONFIG_PCIE_PORT=y ++CONFIG_PCIE_PCI_BRIDGE=y ++CONFIG_PCI_DEVICES=y ++CONFIG_PCI_TESTDEV=y ++CONFIG_PFLASH_CFI01=y ++CONFIG_SCSI=y ++CONFIG_SEMIHOSTING=y ++CONFIG_USB=y ++CONFIG_USB_XHCI=y ++CONFIG_USB_XHCI_PCI=y ++CONFIG_USB_STORAGE_CORE=y ++CONFIG_USB_STORAGE_CLASSIC=y ++CONFIG_USB_HUB=y ++CONFIG_USB_HID=y ++CONFIG_VFIO=y ++CONFIG_VFIO_PCI=y ++CONFIG_VIRTIO_MMIO=y ++CONFIG_VIRTIO_PCI=y ++CONFIG_VIRTIO_IOMMU=y ++CONFIG_XIO3130=y ++CONFIG_ACPI_APEI=y ++CONFIG_TPM=y ++CONFIG_TPM_EMULATOR=y ++CONFIG_TPM_TIS_SYSBUS=y ++CONFIG_ARM_COMPATIBLE_SEMIHOSTING=y ++CONFIG_PVPANIC_PCI=y ++CONFIG_PXB=y ++CONFIG_VHOST_VSOCK=y ++CONFIG_VHOST_USER_VSOCK=y ++CONFIG_VHOST_USER_FS=y ++CONFIG_IOMMUFD=y ++CONFIG_VHOST_USER_SND=y diff --git a/configs/devices/s390x-softmmu/s390x-rh-devices.mak b/configs/devices/s390x-softmmu/s390x-rh-devices.mak new file mode 100644 -index 0000000000..24cf6dbd03 +index 0000000000..834281d872 --- /dev/null +++ b/configs/devices/s390x-softmmu/s390x-rh-devices.mak -@@ -0,0 +1,19 @@ +@@ -0,0 +1,20 @@ +include ../rh-virtio.mak + +CONFIG_PCI=y @@ -167,6 +221,7 @@ index 0000000000..24cf6dbd03 +CONFIG_VFIO_PCI=y +CONFIG_VHOST_USER=y +CONFIG_VIRTIO_CCW=y ++CONFIG_VIRTIO_MEM=y +CONFIG_WDT_DIAG288=y +CONFIG_VHOST_VSOCK=y +CONFIG_VHOST_USER_VSOCK=y @@ -174,10 +229,10 @@ index 0000000000..24cf6dbd03 +CONFIG_IOMMUFD=y diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak new file mode 100644 -index 0000000000..d60ff1bcfc +index 0000000000..828cb8aa6f --- /dev/null +++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -@@ -0,0 +1,112 @@ +@@ -0,0 +1,118 @@ +include ../rh-virtio.mak + +CONFIG_ACPI=y @@ -199,9 +254,9 @@ index 0000000000..d60ff1bcfc +CONFIG_E1000E_PCI_EXPRESS=y +CONFIG_E1000_PCI=y +CONFIG_EDU=y -+CONFIG_FDC=y -+CONFIG_FDC_SYSBUS=y -+CONFIG_FDC_ISA=y ++#CONFIG_FDC=y ++#CONFIG_FDC_SYSBUS=y ++#CONFIG_FDC_ISA=y +CONFIG_FW_CFG_DMA=y +CONFIG_HDA=y +CONFIG_HYPERV=y @@ -240,6 +295,7 @@ index 0000000000..d60ff1bcfc +CONFIG_PCSPK=y +CONFIG_PC_ACPI=y +CONFIG_PC_PCI=y ++CONFIG_PCI_BRIDGE=y +CONFIG_PCIE_PCI_BRIDGE=y +CONFIG_PFLASH_CFI01=y +CONFIG_PVPANIC_ISA=y @@ -264,10 +320,11 @@ index 0000000000..d60ff1bcfc +CONFIG_USB_XHCI=y +CONFIG_USB_XHCI_NEC=y +CONFIG_USB_XHCI_PCI=y ++CONFIG_USB_HUB=y ++CONFIG_USB_HID=y +CONFIG_VFIO=y +CONFIG_VFIO_PCI=y +CONFIG_VGA=y -+CONFIG_VGA_CIRRUS=y +CONFIG_VGA_PCI=y +CONFIG_VHOST_USER=y +CONFIG_VHOST_USER_BLK=y @@ -275,6 +332,7 @@ index 0000000000..d60ff1bcfc +CONFIG_VIRTIO_PCI=y +CONFIG_VIRTIO_VGA=y +CONFIG_VIRTIO_IOMMU=y ++CONFIG_AMD_IOMMU=y +CONFIG_VMMOUSE=y +CONFIG_VMPORT=y +CONFIG_VTD=y @@ -290,11 +348,14 @@ index 0000000000..d60ff1bcfc +CONFIG_VHOST_USER_VSOCK=y +CONFIG_VHOST_USER_FS=y +CONFIG_IOMMUFD=y ++CONFIG_VHOST_USER_SND=y ++CONFIG_VHOST_USER_GPU=y ++CONFIG_UEFI_VARS=y diff --git a/hw/arm/virt.c b/hw/arm/virt.c -index a9a913aead..6c6d155002 100644 +index ef6be3660f..b525e00365 100644 --- a/hw/arm/virt.c +++ b/hw/arm/virt.c -@@ -2954,6 +2954,7 @@ static void virt_machine_class_init(ObjectClass *oc, void *data) +@@ -3183,6 +3183,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) MachineClass *mc = MACHINE_CLASS(oc); HotplugHandlerClass *hc = HOTPLUG_HANDLER_CLASS(oc); static const char * const valid_cpu_types[] = { @@ -302,53 +363,18 @@ index a9a913aead..6c6d155002 100644 #ifdef CONFIG_TCG ARM_CPU_TYPE_NAME("cortex-a7"), ARM_CPU_TYPE_NAME("cortex-a15"), -@@ -2971,6 +2972,7 @@ static void virt_machine_class_init(ObjectClass *oc, void *data) +@@ -3198,8 +3199,11 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + ARM_CPU_TYPE_NAME("neoverse-n2"), + #endif /* TARGET_AARCH64 */ #endif /* CONFIG_TCG */ ++#endif /* disabled for RHEL */ #ifdef TARGET_AARCH64 ++#if 0 /* Disabled for Red Hat Enterprise Linux */ ARM_CPU_TYPE_NAME("cortex-a53"), +#endif /* disabled for RHEL */ ARM_CPU_TYPE_NAME("cortex-a57"), #if defined(CONFIG_KVM) || defined(CONFIG_HVF) ARM_CPU_TYPE_NAME("host"), -diff --git a/hw/block/fdc.c b/hw/block/fdc.c -index 6dd94e98bc..a05757fc9a 100644 ---- a/hw/block/fdc.c -+++ b/hw/block/fdc.c -@@ -49,6 +49,8 @@ - #include "qom/object.h" - #include "fdc-internal.h" - -+#include "hw/boards.h" -+ - /********************************************************/ - /* debug Floppy devices */ - -@@ -2346,6 +2348,14 @@ void fdctrl_realize_common(DeviceState *dev, FDCtrl *fdctrl, Error **errp) - FDrive *drive; - static int command_tables_inited = 0; - -+ /* Restricted for Red Hat Enterprise Linux: */ -+ MachineClass *mc = MACHINE_GET_CLASS(qdev_get_machine()); -+ if (!strstr(mc->name, "-rhel7.")) { -+ error_setg(errp, "Device %s is not supported with machine type %s", -+ object_get_typename(OBJECT(dev)), mc->name); -+ return; -+ } -+ - if (fdctrl->fallback == FLOPPY_DRIVE_TYPE_AUTO) { - error_setg(errp, "Cannot choose a fallback FDrive type of 'auto'"); - return; -diff --git a/hw/cpu/meson.build b/hw/cpu/meson.build -index 38cdcfbe57..e588ecfd42 100644 ---- a/hw/cpu/meson.build -+++ b/hw/cpu/meson.build -@@ -1,4 +1,5 @@ --system_ss.add(files('core.c', 'cluster.c')) -+#system_ss.add(files('core.c', 'cluster.c')) -+system_ss.add(files('core.c')) - - system_ss.add(when: 'CONFIG_ARM11MPCORE', if_true: files('arm11mpcore.c')) - system_ss.add(when: 'CONFIG_REALVIEW', if_true: files('realview_mpcore.c')) diff --git a/hw/cxl/meson.build b/hw/cxl/meson.build index 3e375f61a9..613adb3ebb 100644 --- a/hw/cxl/meson.build @@ -363,33 +389,11 @@ index 3e375f61a9..613adb3ebb 100644 ), if_false: files( 'cxl-host-stubs.c', -diff --git a/hw/display/cirrus_vga.c b/hw/display/cirrus_vga.c -index 150883a971..497365bd80 100644 ---- a/hw/display/cirrus_vga.c -+++ b/hw/display/cirrus_vga.c -@@ -36,6 +36,7 @@ - #include "qemu/module.h" - #include "qemu/units.h" - #include "qemu/log.h" -+#include "qemu/error-report.h" - #include "sysemu/reset.h" - #include "qapi/error.h" - #include "trace.h" -@@ -2946,6 +2947,9 @@ static void pci_cirrus_vga_realize(PCIDevice *dev, Error **errp) - PCIDeviceClass *pc = PCI_DEVICE_GET_CLASS(dev); - int16_t device_id = pc->device_id; - -+ warn_report("'cirrus-vga' is deprecated, " -+ "please use a different VGA card instead"); -+ - /* - * Follow real hardware, cirrus card emulated has 4 MB video memory. - * Also accept 8 MB/16 MB for backward compatibility. diff --git a/hw/ide/piix.c b/hw/ide/piix.c -index 80efc633d3..9cb82b8eea 100644 +index a0f2709c69..8c3962443b 100644 --- a/hw/ide/piix.c +++ b/hw/ide/piix.c -@@ -191,7 +191,8 @@ static void piix3_ide_class_init(ObjectClass *klass, void *data) +@@ -191,7 +191,8 @@ static void piix3_ide_class_init(ObjectClass *klass, const void *data) k->device_id = PCI_DEVICE_ID_INTEL_82371SB_1; k->class_id = PCI_CLASS_STORAGE_IDE; set_bit(DEVICE_CATEGORY_STORAGE, dc->categories); @@ -399,7 +403,7 @@ index 80efc633d3..9cb82b8eea 100644 } static const TypeInfo piix3_ide_info = { -@@ -215,6 +216,8 @@ static void piix4_ide_class_init(ObjectClass *klass, void *data) +@@ -215,6 +216,8 @@ static void piix4_ide_class_init(ObjectClass *klass, const void *data) k->class_id = PCI_CLASS_STORAGE_IDE; set_bit(DEVICE_CATEGORY_STORAGE, dc->categories); dc->hotpluggable = false; @@ -409,10 +413,10 @@ index 80efc633d3..9cb82b8eea 100644 static const TypeInfo piix4_ide_info = { diff --git a/hw/input/pckbd.c b/hw/input/pckbd.c -index 74f10b640f..2e85ecf476 100644 +index 71f5f976e9..794078ed84 100644 --- a/hw/input/pckbd.c +++ b/hw/input/pckbd.c -@@ -952,6 +952,8 @@ static void i8042_class_initfn(ObjectClass *klass, void *data) +@@ -950,6 +950,8 @@ static void i8042_class_initfn(ObjectClass *klass, const void *data) dc->vmsd = &vmstate_kbd_isa; adevc->build_dev_aml = i8042_build_aml; set_bit(DEVICE_CATEGORY_INPUT, dc->categories); @@ -422,10 +426,10 @@ index 74f10b640f..2e85ecf476 100644 static const TypeInfo i8042_info = { diff --git a/hw/net/e1000.c b/hw/net/e1000.c -index 43f3a4a701..267f182883 100644 +index a80a7b0cdb..7eb5a3b19d 100644 --- a/hw/net/e1000.c +++ b/hw/net/e1000.c -@@ -1746,6 +1746,7 @@ static const E1000Info e1000_devices[] = { +@@ -1732,6 +1732,7 @@ static const E1000Info e1000_devices[] = { .revision = 0x03, .phy_id2 = E1000_PHY_ID2_8254xx_DEFAULT, }, @@ -433,7 +437,7 @@ index 43f3a4a701..267f182883 100644 { .name = "e1000-82544gc", .device_id = E1000_DEV_ID_82544GC_COPPER, -@@ -1758,6 +1759,7 @@ static const E1000Info e1000_devices[] = { +@@ -1744,6 +1745,7 @@ static const E1000Info e1000_devices[] = { .revision = 0x03, .phy_id2 = E1000_PHY_ID2_8254xx_DEFAULT, }, @@ -441,11 +445,33 @@ index 43f3a4a701..267f182883 100644 }; static void e1000_register_types(void) +diff --git a/hw/net/rtl8139.c b/hw/net/rtl8139.c +index 324fb932aa..f4dd693abb 100644 +--- a/hw/net/rtl8139.c ++++ b/hw/net/rtl8139.c +@@ -57,6 +57,7 @@ + #include "system/dma.h" + #include "qemu/module.h" + #include "qemu/timer.h" ++#include "qemu/error-report.h" + #include "qemu/bswap.h" + #include "net/net.h" + #include "net/eth.h" +@@ -3363,6 +3364,9 @@ static void pci_rtl8139_realize(PCIDevice *dev, Error **errp) + DeviceState *d = DEVICE(dev); + uint8_t *pci_conf; + ++ warn_report("'rtl8139' is deprecated, " ++ "please use a different Network Interface Card"); ++ + pci_conf = dev->config; + pci_conf[PCI_INTERRUPT_PIN] = 1; /* interrupt pin A */ + /* TODO: start of capability list, but no capability diff --git a/hw/usb/meson.build b/hw/usb/meson.build -index aac3bb35f2..5411ff35df 100644 +index 17360a5b5a..3c4fdfc31d 100644 --- a/hw/usb/meson.build +++ b/hw/usb/meson.build -@@ -55,7 +55,7 @@ system_ss.add(when: 'CONFIG_USB_SMARTCARD', if_true: files('dev-smartcard-reader +@@ -53,7 +53,7 @@ system_ss.add(when: 'CONFIG_USB_SMARTCARD', if_true: files('dev-smartcard-reader if cacard.found() usbsmartcard_ss = ss.source_set() usbsmartcard_ss.add(when: 'CONFIG_USB_SMARTCARD', @@ -455,10 +481,10 @@ index aac3bb35f2..5411ff35df 100644 endif diff --git a/hw/virtio/meson.build b/hw/virtio/meson.build -index d7f18c96e6..aaabbb8b0b 100644 +index 3ea7b3cec8..3102d68bec 100644 --- a/hw/virtio/meson.build +++ b/hw/virtio/meson.build -@@ -20,7 +20,8 @@ if have_vhost +@@ -22,7 +22,8 @@ if have_vhost system_virtio_ss.add(files('vhost-user-base.c')) # MMIO Stubs @@ -468,7 +494,7 @@ index d7f18c96e6..aaabbb8b0b 100644 system_virtio_ss.add(when: 'CONFIG_VHOST_USER_GPIO', if_true: files('vhost-user-gpio.c')) system_virtio_ss.add(when: 'CONFIG_VHOST_USER_I2C', if_true: files('vhost-user-i2c.c')) system_virtio_ss.add(when: 'CONFIG_VHOST_USER_RNG', if_true: files('vhost-user-rng.c')) -@@ -28,7 +29,8 @@ if have_vhost +@@ -30,7 +31,8 @@ if have_vhost system_virtio_ss.add(when: 'CONFIG_VHOST_USER_INPUT', if_true: files('vhost-user-input.c')) # PCI Stubs @@ -479,10 +505,10 @@ index d7f18c96e6..aaabbb8b0b 100644 if_true: files('vhost-user-gpio-pci.c')) system_virtio_ss.add(when: ['CONFIG_VIRTIO_PCI', 'CONFIG_VHOST_USER_I2C'], diff --git a/target/arm/arm-qmp-cmds.c b/target/arm/arm-qmp-cmds.c -index 3cc8cc738b..6f21fea1f5 100644 +index d292c974c4..9bb68866e1 100644 --- a/target/arm/arm-qmp-cmds.c +++ b/target/arm/arm-qmp-cmds.c -@@ -223,6 +223,7 @@ CpuModelExpansionInfo *qmp_query_cpu_model_expansion(CpuModelExpansionType type, +@@ -225,6 +225,7 @@ CpuModelExpansionInfo *qmp_query_cpu_model_expansion(CpuModelExpansionType type, static void arm_cpu_add_definition(gpointer data, gpointer user_data) { ObjectClass *oc = data; @@ -490,7 +516,7 @@ index 3cc8cc738b..6f21fea1f5 100644 CpuDefinitionInfoList **cpu_list = user_data; CpuDefinitionInfo *info; const char *typename; -@@ -231,6 +232,7 @@ static void arm_cpu_add_definition(gpointer data, gpointer user_data) +@@ -233,6 +234,7 @@ static void arm_cpu_add_definition(gpointer data, gpointer user_data) info = g_malloc0(sizeof(*info)); info->name = cpu_model_from_type(typename); info->q_typename = g_strdup(typename); @@ -498,47 +524,24 @@ index 3cc8cc738b..6f21fea1f5 100644 QAPI_LIST_PREPEND(*cpu_list, info); } -diff --git a/target/arm/cpu.c b/target/arm/cpu.c -index ab8d007a86..e5dce20f19 100644 ---- a/target/arm/cpu.c -+++ b/target/arm/cpu.c -@@ -2546,6 +2546,10 @@ static void cpu_register_class_init(ObjectClass *oc, void *data) - - acc->info = data; - cc->gdb_core_xml_file = "arm-core.xml"; -+ -+ if (acc->info->deprecation_note) { -+ cc->deprecation_note = acc->info->deprecation_note; -+ } - } - - void arm_cpu_register(const ARMCPUInfo *info) diff --git a/target/arm/cpu.h b/target/arm/cpu.h -index bc0c84873f..e9472c8bb8 100644 +index dc9b6dce4c..dc0da8b0ae 100644 --- a/target/arm/cpu.h +++ b/target/arm/cpu.h -@@ -37,6 +37,8 @@ - #define KVM_HAVE_MCE_INJECTION 1 - #endif +@@ -34,6 +34,8 @@ + #include "target/arm/gtimer.h" + #include "target/arm/cpu-sysregs.h" +#define RHEL_CPU_DEPRECATION "use 'host' / 'max'" + #define EXCP_UDEF 1 /* undefined instruction */ #define EXCP_SWI 2 /* software interrupt */ #define EXCP_PREFETCH_ABORT 3 -@@ -1092,6 +1094,7 @@ typedef struct ARMCPUInfo { - const char *name; - void (*initfn)(Object *obj); - void (*class_init)(ObjectClass *oc, void *data); -+ const char *deprecation_note; - } ARMCPUInfo; - - /** diff --git a/target/arm/cpu64.c b/target/arm/cpu64.c -index 985b1efe16..46a4e80171 100644 +index 26cf7e6dfa..051d5d653b 100644 --- a/target/arm/cpu64.c +++ b/target/arm/cpu64.c -@@ -648,6 +648,7 @@ static void aarch64_a57_initfn(Object *obj) +@@ -698,6 +698,7 @@ static void aarch64_a57_initfn(Object *obj) define_cortex_a72_a57_a53_cp_reginfo(cpu); } @@ -546,7 +549,7 @@ index 985b1efe16..46a4e80171 100644 static void aarch64_a53_initfn(Object *obj) { ARMCPU *cpu = ARM_CPU(obj); -@@ -704,6 +705,7 @@ static void aarch64_a53_initfn(Object *obj) +@@ -759,6 +760,7 @@ static void aarch64_a53_initfn(Object *obj) cpu->gic_pribits = 5; define_cortex_a72_a57_a53_cp_reginfo(cpu); } @@ -554,7 +557,7 @@ index 985b1efe16..46a4e80171 100644 static void aarch64_host_initfn(Object *obj) { -@@ -742,8 +744,11 @@ static void aarch64_max_initfn(Object *obj) +@@ -797,8 +799,11 @@ static void aarch64_max_initfn(Object *obj) } static const ARMCPUInfo aarch64_cpus[] = { @@ -567,39 +570,25 @@ index 985b1efe16..46a4e80171 100644 { .name = "max", .initfn = aarch64_max_initfn }, #if defined(CONFIG_KVM) || defined(CONFIG_HVF) { .name = "host", .initfn = aarch64_host_initfn }, -@@ -814,8 +819,13 @@ static void aarch64_cpu_instance_init(Object *obj) - static void cpu_register_class_init(ObjectClass *oc, void *data) - { - ARMCPUClass *acc = ARM_CPU_CLASS(oc); -+ CPUClass *cc = CPU_CLASS(oc); - - acc->info = data; -+ -+ if (acc->info->deprecation_note) { -+ cc->deprecation_note = acc->info->deprecation_note; -+ } - } - - void aarch64_cpu_register(const ARMCPUInfo *info) diff --git a/target/arm/tcg/cpu32.c b/target/arm/tcg/cpu32.c -index de8f2be941..8896295ae3 100644 +index a2a23eae0d..c362759d65 100644 --- a/target/arm/tcg/cpu32.c +++ b/target/arm/tcg/cpu32.c -@@ -92,6 +92,7 @@ void aa32_max_features(ARMCPU *cpu) - cpu->isar.id_dfr1 = t; +@@ -115,6 +115,7 @@ void aa32_max_features(ARMCPU *cpu) + FIELD_DP32_IDREG(isar, ID_DFR1, HPMN0, 1); /* FEAT_HPMN0 */ } +#if 0 /* Disabled for Red Hat Enterprise Linux */ /* CPU models. These are not needed for the AArch64 linux-user build. */ #if !defined(CONFIG_USER_ONLY) || !defined(TARGET_AARCH64) -@@ -1037,3 +1038,4 @@ static void arm_tcg_cpu_register_types(void) +@@ -1084,3 +1085,4 @@ static void arm_tcg_cpu_register_types(void) type_init(arm_tcg_cpu_register_types) #endif /* !CONFIG_USER_ONLY || !TARGET_AARCH64 */ +#endif /* disabled for RHEL */ diff --git a/target/arm/tcg/cpu64.c b/target/arm/tcg/cpu64.c -index 9f7a9f3d2c..7ec6851c9c 100644 +index 35cddbafa4..c7c464a0af 100644 --- a/target/arm/tcg/cpu64.c +++ b/target/arm/tcg/cpu64.c @@ -29,6 +29,7 @@ @@ -607,10 +596,10 @@ index 9f7a9f3d2c..7ec6851c9c 100644 #include "cpregs.h" +#if 0 /* Disabled for Red Hat Enterprise Linux */ - static uint64_t make_ccsidr64(unsigned assoc, unsigned linesize, - unsigned cachesize) + static void aarch64_a35_initfn(Object *obj) { -@@ -134,6 +135,7 @@ static void aarch64_a35_initfn(Object *obj) + ARMCPU *cpu = ARM_CPU(obj); +@@ -113,6 +114,7 @@ static void aarch64_a35_initfn(Object *obj) /* These values are the same with A53/A57/A72. */ define_cortex_a72_a57_a53_cp_reginfo(cpu); } @@ -618,15 +607,15 @@ index 9f7a9f3d2c..7ec6851c9c 100644 static void cpu_max_get_sve_max_vq(Object *obj, Visitor *v, const char *name, void *opaque, Error **errp) -@@ -223,6 +225,7 @@ static void cpu_max_get_l0gptsz(Object *obj, Visitor *v, const char *name, - static Property arm_cpu_lpa2_property = +@@ -199,6 +201,7 @@ static void cpu_max_get_l0gptsz(Object *obj, Visitor *v, const char *name, + static const Property arm_cpu_lpa2_property = DEFINE_PROP_BOOL("lpa2", ARMCPU, prop_lpa2, true); +#if 0 /* Disabled for Red Hat Enterprise Linux */ static void aarch64_a55_initfn(Object *obj) { ARMCPU *cpu = ARM_CPU(obj); -@@ -1065,6 +1068,7 @@ static void aarch64_neoverse_n2_initfn(Object *obj) +@@ -1080,6 +1083,7 @@ static void aarch64_neoverse_n2_initfn(Object *obj) aarch64_add_pauth_properties(obj); aarch64_add_sve_properties(obj); } @@ -634,7 +623,7 @@ index 9f7a9f3d2c..7ec6851c9c 100644 /* * -cpu max: a CPU with as many features enabled as our emulation supports. -@@ -1271,6 +1275,7 @@ void aarch64_max_tcg_initfn(Object *obj) +@@ -1310,6 +1314,7 @@ void aarch64_max_tcg_initfn(Object *obj) qdev_property_add_static(DEVICE(obj), &arm_cpu_lpa2_property); } @@ -642,7 +631,7 @@ index 9f7a9f3d2c..7ec6851c9c 100644 static const ARMCPUInfo aarch64_cpus[] = { { .name = "cortex-a35", .initfn = aarch64_a35_initfn }, { .name = "cortex-a55", .initfn = aarch64_a55_initfn }, -@@ -1282,14 +1287,17 @@ static const ARMCPUInfo aarch64_cpus[] = { +@@ -1321,14 +1326,17 @@ static const ARMCPUInfo aarch64_cpus[] = { { .name = "neoverse-v1", .initfn = aarch64_neoverse_v1_initfn }, { .name = "neoverse-n2", .initfn = aarch64_neoverse_n2_initfn }, }; @@ -654,61 +643,230 @@ index 9f7a9f3d2c..7ec6851c9c 100644 size_t i; for (i = 0; i < ARRAY_SIZE(aarch64_cpus); ++i) { - aarch64_cpu_register(&aarch64_cpus[i]); + arm_cpu_register(&aarch64_cpus[i]); } +#endif } type_init(aarch64_cpu_register_types) diff --git a/target/arm/tcg/meson.build b/target/arm/tcg/meson.build -index 3b1a9f0fc5..6c95d99181 100644 +index 895facdc30..f1a9e01c51 100644 --- a/target/arm/tcg/meson.build +++ b/target/arm/tcg/meson.build -@@ -56,5 +56,5 @@ arm_system_ss.add(files( +@@ -53,7 +53,7 @@ arm_system_ss.add(files( 'psci.c', )) -arm_system_ss.add(when: 'CONFIG_ARM_V7M', if_true: files('cpu-v7m.c')) +#arm_system_ss.add(when: 'CONFIG_ARM_V7M', if_true: files('cpu-v7m.c')) arm_user_ss.add(when: 'TARGET_AARCH64', if_false: files('cpu-v7m.c')) -diff --git a/target/s390x/cpu_models_sysemu.c b/target/s390x/cpu_models_sysemu.c -index 2d99218069..0728bfcc20 100644 ---- a/target/s390x/cpu_models_sysemu.c -+++ b/target/s390x/cpu_models_sysemu.c -@@ -34,6 +34,9 @@ static void check_unavailable_features(const S390CPUModel *max_model, - (max_model->def->gen == model->def->gen && - max_model->def->ec_ga < model->def->ec_ga)) { - list_add_feat("type", unavailable); -+ } else if (model->def->gen < 11 && kvm_enabled()) { -+ /* Older CPU models are not supported on Red Hat Enterprise Linux */ -+ list_add_feat("type", unavailable); - } - /* detect missing features if any to properly report them */ -diff --git a/target/s390x/kvm/kvm.c b/target/s390x/kvm/kvm.c -index 4ce809c5d4..55fb4855b1 100644 ---- a/target/s390x/kvm/kvm.c -+++ b/target/s390x/kvm/kvm.c -@@ -2565,6 +2565,14 @@ void kvm_s390_apply_cpu_model(const S390CPUModel *model, Error **errp) - error_setg(errp, "KVM doesn't support CPU models"); - return; + arm_common_ss.add(zlib) +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 6d85149e6e..9c756a05f2 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -3163,6 +3163,7 @@ static const CPUCaches xeon_srf_cache_info = { + }, + }; + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static const CPUCaches yongfeng_cache_info = { + .l1d_cache = &(CPUCacheInfo) { + /* CPUID 0x4.0x0.EAX */ +@@ -3261,6 +3262,7 @@ static const CPUCaches yongfeng_cache_info = { + .share_level = CPU_TOPOLOGY_LEVEL_DIE, + }, + }; ++#endif + + /* The following VMX features are not supported by KVM and are left out in the + * CPU definitions: +@@ -3290,9 +3292,13 @@ static const CPUCaches yongfeng_cache_info = { + * PT in VMX operation + */ + ++#define RHEL_CPU_DEPRECATION \ ++ "use at least 'Haswell' / 'EPYC', or 'host' / 'max'" ++ + static const X86CPUDefinition builtin_x86_defs[] = { + { + .name = "qemu64", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 0xd, + .vendor = CPUID_VENDOR_AMD, + .family = 15, +@@ -3311,6 +3317,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + .xlevel = 0x8000000A, + .model_id = "QEMU Virtual CPU version " QEMU_HW_VERSION, + }, ++#if 0 // Deprecated CPU models are removed in RHEL-10 + { + .name = "phenom", + .level = 5, +@@ -3679,8 +3686,10 @@ static const X86CPUDefinition builtin_x86_defs[] = { + .xlevel = 0x80000008, + .model_id = "Intel Core 2 Duo P9xxx (Penryn Class Core 2)", + }, ++#endif // Removal of deprecated CPU models in RHEL-10 + { + .name = "Nehalem", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 11, + .vendor = CPUID_VENDOR_INTEL, + .family = 6, +@@ -3758,6 +3767,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + { + .name = "Westmere", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 11, + .vendor = CPUID_VENDOR_INTEL, + .family = 6, +@@ -3839,6 +3849,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + { + .name = "SandyBridge", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 0xd, + .vendor = CPUID_VENDOR_INTEL, + .family = 6, +@@ -3925,6 +3936,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + { + .name = "IvyBridge", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 0xd, + .vendor = CPUID_VENDOR_INTEL, + .family = 6, +@@ -5551,6 +5563,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + { + .name = "Denverton", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 21, + .vendor = CPUID_VENDOR_INTEL, + .family = 6, +@@ -5661,6 +5674,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + { + .name = "Snowridge", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 27, + .vendor = CPUID_VENDOR_INTEL, + .family = 6, +@@ -5842,6 +5856,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + .xlevel = 0x80000008, + .model_id = "Intel Xeon Phi Processor (Knights Mill)", + }, ++#if 0 // Deprecated CPU models are removed in RHEL-10 + { + .name = "Opteron_G1", + .level = 5, +@@ -5909,8 +5924,10 @@ static const X86CPUDefinition builtin_x86_defs[] = { + .xlevel = 0x80000008, + .model_id = "AMD Opteron 23xx (Gen 3 Class Opteron)", + }, ++#endif + { + .name = "Opteron_G4", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 0xd, + .vendor = CPUID_VENDOR_AMD, + .family = 21, +@@ -5943,6 +5960,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + { + .name = "Opteron_G5", ++ .deprecation_note = RHEL_CPU_DEPRECATION, + .level = 0xd, + .vendor = CPUID_VENDOR_AMD, + .family = 21, +@@ -6420,6 +6438,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + { /* end of list */ } + } + }, ++#if 0 // Disabled for Red Hat Enterprise Linux + { + .name = "YongFeng", + .level = 0x1F, +@@ -6565,6 +6584,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + { /* end of list */ } + } + }, ++#endif + { + .name = "EPYC-Turin", + .level = 0xd, +diff --git a/target/riscv/cpu.c b/target/riscv/cpu.c +index d055ddf462..bca50a39be 100644 +--- a/target/riscv/cpu.c ++++ b/target/riscv/cpu.c +@@ -2035,6 +2035,7 @@ static const PropertyInfo prop_marchid = { + .set = prop_marchid_set, + }; + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + /* + * RVA22U64 defines some 'named features' that are cache + * related: Za64rs, Zic64b, Ziccif, Ziccrse, Ziccamoa +@@ -2143,12 +2144,15 @@ static RISCVCPUProfile RVA23S64 = { + RISCV_PROFILE_EXT_LIST_END } -+ -+ /* Older CPU models are not supported on Red Hat Enterprise Linux */ -+ if (model->def->gen < 11) { -+ error_setg(errp, "KVM: Unsupported CPU type specified: %s", -+ MACHINE(qdev_get_machine())->cpu_type); -+ return; -+ } -+ - prop.cpuid = s390_cpuid_from_cpu_model(model); - prop.ibc = s390_ibc_from_cpu_model(model); - /* configure cpu features indicated via STFL(e) */ + }; ++#endif + + RISCVCPUProfile *riscv_profiles[] = { ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + &RVA22U64, + &RVA22S64, + &RVA23U64, + &RVA23S64, ++#endif + NULL, + }; + +@@ -2993,6 +2997,7 @@ static const TypeInfo riscv_cpu_type_infos[] = { + .cfg.pmp_regions = 8 + ), + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + #if defined(TARGET_RISCV32) || \ + (defined(TARGET_RISCV64) && !defined(CONFIG_USER_ONLY)) + DEFINE_RISCV_CPU(TYPE_RISCV_CPU_BASE32, TYPE_RISCV_DYNAMIC_CPU, +@@ -3287,6 +3292,7 @@ static const TypeInfo riscv_cpu_type_infos[] = { + DEFINE_PROFILE_CPU(TYPE_RISCV_CPU_RVA23U64, TYPE_RISCV_CPU_RV64I, RVA23U64), + DEFINE_PROFILE_CPU(TYPE_RISCV_CPU_RVA23S64, TYPE_RISCV_CPU_RV64I, RVA23S64), + #endif /* TARGET_RISCV64 */ ++#endif /* disabled for RHEL */ + }; + + DEFINE_TYPES(riscv_cpu_type_infos) +diff --git a/target/s390x/cpu_models.c b/target/s390x/cpu_models.c +index 954a7a99a9..fe29f5c5b7 100644 +--- a/target/s390x/cpu_models.c ++++ b/target/s390x/cpu_models.c +@@ -72,7 +72,6 @@ static S390CPUDef s390_cpu_defs[] = { + CPUDEF_INIT(0x2096, 9, 2, 40, 0x00000000U, "z9BC", "IBM System z9 BC GA1"), + CPUDEF_INIT(0x2094, 9, 3, 40, 0x00000000U, "z9EC.3", "IBM System z9 EC GA3"), + CPUDEF_INIT(0x2096, 9, 3, 40, 0x00000000U, "z9BC.2", "IBM System z9 BC GA2"), +-#endif + CPUDEF_INIT(0x2097, 10, 1, 43, 0x00000000U, "z10EC", "IBM System z10 EC GA1"), + CPUDEF_INIT(0x2097, 10, 2, 43, 0x00000000U, "z10EC.2", "IBM System z10 EC GA2"), + CPUDEF_INIT(0x2098, 10, 2, 43, 0x00000000U, "z10BC", "IBM System z10 BC GA1"), +@@ -81,6 +80,7 @@ static S390CPUDef s390_cpu_defs[] = { + CPUDEF_INIT(0x2817, 11, 1, 44, 0x08000000U, "z196", "IBM zEnterprise 196 GA1"), + CPUDEF_INIT(0x2817, 11, 2, 44, 0x08000000U, "z196.2", "IBM zEnterprise 196 GA2"), + CPUDEF_INIT(0x2818, 11, 2, 44, 0x08000000U, "z114", "IBM zEnterprise 114 GA1"), ++#endif + CPUDEF_INIT(0x2827, 12, 1, 44, 0x08000000U, "zEC12", "IBM zEnterprise EC12 GA1"), + CPUDEF_INIT(0x2827, 12, 2, 44, 0x08000000U, "zEC12.2", "IBM zEnterprise EC12 GA2"), + CPUDEF_INIT(0x2828, 12, 2, 44, 0x08000000U, "zBC12", "IBM zEnterprise BC12 GA1"), diff --git a/tests/qtest/arm-cpu-features.c b/tests/qtest/arm-cpu-features.c -index 9d6e6190d5..f822526acb 100644 +index eb8ddebffb..2d3304bb4a 100644 --- a/tests/qtest/arm-cpu-features.c +++ b/tests/qtest/arm-cpu-features.c -@@ -452,8 +452,10 @@ static void test_query_cpu_model_expansion(const void *data) +@@ -459,8 +459,10 @@ static void test_query_cpu_model_expansion(const void *data) assert_error(qts, "host", "The CPU type 'host' requires KVM", NULL); /* Test expected feature presence/absence for some cpu types */ @@ -719,7 +877,7 @@ index 9d6e6190d5..f822526acb 100644 /* Enabling and disabling pmu should always work. */ assert_has_feature_enabled(qts, "max", "pmu"); -@@ -470,6 +472,7 @@ static void test_query_cpu_model_expansion(const void *data) +@@ -477,6 +479,7 @@ static void test_query_cpu_model_expansion(const void *data) assert_has_feature_enabled(qts, "cortex-a57", "pmu"); assert_has_feature_enabled(qts, "cortex-a57", "aarch64"); @@ -727,7 +885,7 @@ index 9d6e6190d5..f822526acb 100644 assert_has_feature_enabled(qts, "a64fx", "pmu"); assert_has_feature_enabled(qts, "a64fx", "aarch64"); /* -@@ -482,6 +485,7 @@ static void test_query_cpu_model_expansion(const void *data) +@@ -489,6 +492,7 @@ static void test_query_cpu_model_expansion(const void *data) "{ 'sve384': true }"); assert_error(qts, "a64fx", "cannot enable sve640", "{ 'sve640': true }"); diff --git a/0006-Machine-type-related-general-changes.patch b/0006-Machine-type-related-general-changes.patch index d53eeb7..310fcfc 100644 --- a/0006-Machine-type-related-general-changes.patch +++ b/0006-Machine-type-related-general-changes.patch @@ -1,4 +1,4 @@ -From 802da738d5231ef56d25f4ffcfa6e7d97698ee72 Mon Sep 17 00:00:00 2001 +From dcaeeab5909a41372c9445e9f97282e8dd3d1d44 Mon Sep 17 00:00:00 2001 From: Miroslav Rezanina Date: Fri, 11 Jan 2019 09:54:45 +0100 Subject: Machine type related general changes @@ -8,24 +8,50 @@ split to allow easier review. It contains changes not related to any architecture. Signed-off-by: Miroslav Rezanina + +--- +Rebase notes (9.1.0): +- Upstream removed uuid_encoced argument on smbios_set_defaults + +Rebase notes (10.0.0 rc0): +- Solve conflicting xhci_pci_properties (downstream vs upstream) + +Rebase notes (10.0.0): +- Add riscv changes +- Added upstream compat changes + +Rebase notes (10.1.0 rc2): +- Remove downstream change to i8254 (review comment) + +Rebase notes (10.1.0): +- Added upstream compat changes from 10.1 + +Merged commits (9.1.0): +- 043ad5ce97 Add upstream compatibility bits (partial) +- bfbdab5824 rhel 9.4.0 machine type compat for virtio-gpu migration + +Merged commits (10.0.0 rc0): +- 03502faf70 Add upstream compatibility bits (partial) +- f53dbf7532 remove stale compat definitions (partial) +- d93fcb3940 virtio-net: disable USO for all RHEL9 (partial) --- hw/acpi/piix4.c | 2 +- hw/arm/virt.c | 2 +- - hw/core/machine.c | 269 +++++++++++++++++++++++++++++++++++ + hw/core/machine.c | 145 +++++++++++++++++++++++++++++++++++ hw/i386/fw_cfg.c | 3 +- hw/net/rtl8139.c | 4 +- - hw/smbios/smbios.c | 46 +++++- - hw/timer/i8254_common.c | 2 +- - hw/usb/hcd-xhci-pci.c | 59 ++++++-- + hw/riscv/virt.c | 4 +- + hw/smbios/smbios.c | 46 ++++++++++- + hw/usb/hcd-xhci-pci.c | 54 +++++++++---- hw/usb/hcd-xhci-pci.h | 1 + hw/virtio/virtio-mem.c | 3 +- - include/hw/boards.h | 40 ++++++ + include/hw/boards.h | 31 ++++++++ include/hw/firmware/smbios.h | 4 +- include/hw/i386/pc.h | 3 + - 13 files changed, 414 insertions(+), 24 deletions(-) + 13 files changed, 277 insertions(+), 25 deletions(-) diff --git a/hw/acpi/piix4.c b/hw/acpi/piix4.c -index debe1adb84..e8ddcd716e 100644 +index 7a18f18dda..2b3678b8c6 100644 --- a/hw/acpi/piix4.c +++ b/hw/acpi/piix4.c @@ -245,7 +245,7 @@ static bool vmstate_test_migrate_acpi_index(void *opaque, int version_id) @@ -38,25 +64,25 @@ index debe1adb84..e8ddcd716e 100644 .fields = (const VMStateField[]) { VMSTATE_PCI_DEVICE(parent_obj, PIIX4PMState), diff --git a/hw/arm/virt.c b/hw/arm/virt.c -index 6c6d155002..36e9b4b4e9 100644 +index b525e00365..1800981317 100644 --- a/hw/arm/virt.c +++ b/hw/arm/virt.c -@@ -1651,7 +1651,7 @@ static void virt_build_smbios(VirtMachineState *vms) +@@ -1755,7 +1755,7 @@ static void virt_build_smbios(VirtMachineState *vms) + product = "KVM Virtual Machine"; + } - smbios_set_defaults("QEMU", product, - vmc->smbios_old_sys_ver ? "1.0" : mc->name, -- true); -+ true, NULL, NULL); +- smbios_set_defaults("QEMU", product, mc->name); ++ smbios_set_defaults("QEMU", product, mc->name, NULL, NULL); /* build the array of physical mem area from base_memmap */ mem_array.address = vms->memmap[VIRT_MEM].base; diff --git a/hw/core/machine.c b/hw/core/machine.c -index 37ede0e7d4..695cb89a46 100644 +index bd47527479..2a1a42cebc 100644 --- a/hw/core/machine.c +++ b/hw/core/machine.c -@@ -296,6 +296,275 @@ GlobalProperty hw_compat_2_1[] = { +@@ -288,6 +288,151 @@ GlobalProperty hw_compat_2_6[] = { }; - const size_t hw_compat_2_1_len = G_N_ELEMENTS(hw_compat_2_1); + const size_t hw_compat_2_6_len = G_N_ELEMENTS(hw_compat_2_6); +/* + * RHEL only: machine types for previous major releases are deprecated @@ -64,6 +90,78 @@ index 37ede0e7d4..695cb89a46 100644 +const char *rhel_old_machine_deprecation = + "machine types for previous major releases are deprecated"; + ++GlobalProperty hw_compat_rhel_10_2[] = { ++ /* hw_compat_rhel_10_2 from hw_compat_10_0 */ ++ { "scsi-hd", "dpofua", "off" }, ++ /* hw_compat_rhel_10_2 from hw_compat_10_0 */ ++ { "vfio-pci", "x-migration-load-config-after-iter", "off" }, ++ /* hw_compat_rhel_10_2 from hw_compat_10_0 */ ++ { "ramfb", "use-legacy-x86-rom", "true"}, ++ /* hw_compat_rhel_10_2 from hw_compat_10_0 */ ++ { "vfio-pci-nohotplug", "use-legacy-x86-rom", "true" }, ++}; ++const size_t hw_compat_rhel_10_2_len = G_N_ELEMENTS(hw_compat_10_0); ++ ++GlobalProperty hw_compat_rhel_10_1[] = { ++ /* hw_compat_rhel_10_1 from hw_compat_9_1 */ ++ { TYPE_PCI_DEVICE, "x-pcie-ext-tag", "false" }, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ {"arm-cpu", "backcompat-pauth-default-use-qarma5", "true"}, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ { "virtio-balloon-pci", "vectors", "0" }, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ { "virtio-balloon-pci-transitional", "vectors", "0" }, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ { "virtio-balloon-pci-non-transitional", "vectors", "0" }, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ { "virtio-mem-pci", "vectors", "0" }, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ { "migration", "multifd-clean-tls-termination", "false" }, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ { "migration", "send-switchover-start", "off"}, ++ /* hw_compat_rhel_10_1 from hw_compat_9_2 */ ++ { "vfio-pci", "x-migration-multifd-transfer", "off" }, ++}; ++const size_t hw_compat_rhel_10_1_len = G_N_ELEMENTS(hw_compat_rhel_10_1); ++ ++ ++GlobalProperty hw_compat_rhel_10_0[] = { ++ /* hw_compat_rhel_10_0 from hw_compat_9_0 */ ++ {"arm-cpu", "backcompat-cntfrq", "true" }, ++ /* hw_compat_rhel_10_0 from hw_compat_9_0 */ ++ { "scsi-hd", "migrate-emulated-scsi-request", "false" }, ++ /* hw_compat_rhel_10_0 from hw_compat_9_0 */ ++ { "scsi-cd", "migrate-emulated-scsi-request", "false" }, ++ /* hw_compat_rhel_10_0 from hw_compat_9_0 */ ++ {"vfio-pci", "skip-vsc-check", "false" }, ++ /* hw_compat_rhel_10_0 from hw_compat_9_0 */ ++ { "virtio-pci", "x-pcie-pm-no-soft-reset", "off" }, ++ /* hw_compat_rhel_10_0 from hw_compat_9_0 */ ++ {"sd-card", "spec_version", "2" }, ++}; ++const size_t hw_compat_rhel_10_0_len = G_N_ELEMENTS(hw_compat_rhel_10_0); ++ ++/* Apply this to all RHEL9 boards going backward and forward */ ++GlobalProperty hw_compat_rhel_9[] = { ++ /* supported by userspace, but RHEL 9 *kernels* do not support USO. */ ++ { TYPE_VIRTIO_NET, "host_uso", "off"}, ++ { TYPE_VIRTIO_NET, "guest_uso4", "off"}, ++ { TYPE_VIRTIO_NET, "guest_uso6", "off"}, ++}; ++const size_t hw_compat_rhel_9_len = G_N_ELEMENTS(hw_compat_rhel_9); ++ ++GlobalProperty hw_compat_rhel_9_5[] = { ++ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ ++ { "migration", "zero-page-detection", "legacy"}, ++ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ ++ { TYPE_VIRTIO_IOMMU_PCI, "granule", "4k" }, ++ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ ++ { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "64" }, ++ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ ++ { "virtio-gpu-device", "x-scanout-vmstate-version", "1" }, ++}; ++const size_t hw_compat_rhel_9_5_len = G_N_ELEMENTS(hw_compat_rhel_9_5); ++ +GlobalProperty hw_compat_rhel_9_4[] = { + /* hw_compat_rhel_9_4 from hw_compat_8_0 */ + { TYPE_VIRTIO_NET, "host_uso", "off"}, @@ -130,225 +228,29 @@ index 37ede0e7d4..695cb89a46 100644 + { "PIIX4_PM", "x-not-migrate-acpi-index", "on"}, +}; +const size_t hw_compat_rhel_9_0_len = G_N_ELEMENTS(hw_compat_rhel_9_0); -+ -+GlobalProperty hw_compat_rhel_8_6[] = { -+ /* hw_compat_rhel_8_6 bz 2065589 */ -+ /* -+ * vhost-vsock device in RHEL 8 kernels doesn't support seqpacket, so -+ * we need do disable it downstream on the latest hw_compat_rhel_8. -+ */ -+ { "vhost-vsock-device", "seqpacket", "off" }, -+}; -+const size_t hw_compat_rhel_8_6_len = G_N_ELEMENTS(hw_compat_rhel_8_6); -+ -+/* -+ * Mostly the same as hw_compat_6_0 and hw_compat_6_1 -+ */ -+GlobalProperty hw_compat_rhel_8_5[] = { -+ /* hw_compat_rhel_8_5 from hw_compat_6_0 */ -+ { "gpex-pcihost", "allow-unmapped-accesses", "false" }, -+ /* hw_compat_rhel_8_5 from hw_compat_6_0 */ -+ { "i8042", "extended-state", "false"}, -+ /* hw_compat_rhel_8_5 from hw_compat_6_0 */ -+ { "nvme-ns", "eui64-default", "off"}, -+ /* hw_compat_rhel_8_5 from hw_compat_6_0 */ -+ { "e1000", "init-vet", "off" }, -+ /* hw_compat_rhel_8_5 from hw_compat_6_0 */ -+ { "e1000e", "init-vet", "off" }, -+ /* hw_compat_rhel_8_5 from hw_compat_6_0 */ -+ { "vhost-vsock-device", "seqpacket", "off" }, -+ /* hw_compat_rhel_8_5 from hw_compat_6_1 */ -+ { "vhost-user-vsock-device", "seqpacket", "off" }, -+ /* hw_compat_rhel_8_5 from hw_compat_6_1 */ -+ { "nvme-ns", "shared", "off" }, -+}; -+const size_t hw_compat_rhel_8_5_len = G_N_ELEMENTS(hw_compat_rhel_8_5); -+ -+/* -+ * Mostly the same as hw_compat_5_2 -+ */ -+GlobalProperty hw_compat_rhel_8_4[] = { -+ /* hw_compat_rhel_8_4 from hw_compat_5_2 */ -+ { "ICH9-LPC", "smm-compat", "on"}, -+ /* hw_compat_rhel_8_4 from hw_compat_5_2 */ -+ { "PIIX4_PM", "smm-compat", "on"}, -+ /* hw_compat_rhel_8_4 from hw_compat_5_2 */ -+ { "virtio-blk-device", "report-discard-granularity", "off" }, -+ /* hw_compat_rhel_8_4 from hw_compat_5_2 */ -+ /* -+ * Upstream incorrectly had "virtio-net-pci" instead of "virtio-net-pci-base", -+ * (https://bugzilla.redhat.com/show_bug.cgi?id=1999141) -+ */ -+ { "virtio-net-pci-base", "vectors", "3"}, -+}; -+const size_t hw_compat_rhel_8_4_len = G_N_ELEMENTS(hw_compat_rhel_8_4); -+ -+/* -+ * Mostly the same as hw_compat_5_1 -+ */ -+GlobalProperty hw_compat_rhel_8_3[] = { -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "vhost-scsi", "num_queues", "1"}, -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "vhost-user-blk", "num-queues", "1"}, -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "vhost-user-scsi", "num_queues", "1"}, -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "virtio-blk-device", "num-queues", "1"}, -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "virtio-scsi-device", "num_queues", "1"}, -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "nvme", "use-intel-id", "on"}, -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "pvpanic", "events", "1"}, /* PVPANIC_PANICKED */ -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "pl011", "migrate-clk", "off" }, -+ /* hw_compat_rhel_8_3 bz 1912846 */ -+ { "pci-xhci", "x-rh-late-msi-cap", "off" }, -+ /* hw_compat_rhel_8_3 from hw_compat_5_1 */ -+ { "virtio-pci", "x-ats-page-aligned", "off"}, -+}; -+const size_t hw_compat_rhel_8_3_len = G_N_ELEMENTS(hw_compat_rhel_8_3); -+ -+/* -+ * The same as hw_compat_4_2 + hw_compat_5_0 -+ */ -+GlobalProperty hw_compat_rhel_8_2[] = { -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "virtio-blk-device", "queue-size", "128"}, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "virtio-scsi-device", "virtqueue_size", "128"}, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "virtio-blk-device", "x-enable-wce-if-config-wce", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "virtio-blk-device", "seg-max-adjust", "off"}, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "virtio-scsi-device", "seg_max_adjust", "off"}, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "vhost-blk-device", "seg_max_adjust", "off"}, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "usb-host", "suppress-remote-wake", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "usb-redir", "suppress-remote-wake", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "qxl", "revision", "4" }, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "qxl-vga", "revision", "4" }, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "fw_cfg", "acpi-mr-restore", "false" }, -+ /* hw_compat_rhel_8_2 from hw_compat_4_2 */ -+ { "virtio-device", "use-disabled-flag", "false" }, -+ /* hw_compat_rhel_8_2 from hw_compat_5_0 */ -+ { "pci-host-bridge", "x-config-reg-migration-enabled", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_5_0 */ -+ { "virtio-balloon-device", "page-poison", "false" }, -+ /* hw_compat_rhel_8_2 from hw_compat_5_0 */ -+ { "vmport", "x-read-set-eax", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_5_0 */ -+ { "vmport", "x-signal-unsupported-cmd", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_5_0 */ -+ { "vmport", "x-report-vmx-type", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_5_0 */ -+ { "vmport", "x-cmds-v2", "off" }, -+ /* hw_compat_rhel_8_2 from hw_compat_5_0 */ -+ { "virtio-device", "x-disable-legacy-check", "true" }, -+}; -+const size_t hw_compat_rhel_8_2_len = G_N_ELEMENTS(hw_compat_rhel_8_2); -+ -+/* -+ * The same as hw_compat_4_1 -+ */ -+GlobalProperty hw_compat_rhel_8_1[] = { -+ /* hw_compat_rhel_8_1 from hw_compat_4_1 */ -+ { "virtio-pci", "x-pcie-flr-init", "off" }, -+}; -+const size_t hw_compat_rhel_8_1_len = G_N_ELEMENTS(hw_compat_rhel_8_1); -+ -+/* The same as hw_compat_3_1 -+ * format of array has been changed by: -+ * 6c36bddf5340 ("machine: Use shorter format for GlobalProperty arrays") -+ */ -+GlobalProperty hw_compat_rhel_8_0[] = { -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "pcie-root-port", "x-speed", "2_5" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "pcie-root-port", "x-width", "1" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "memory-backend-file", "x-use-canonical-path-for-ramblock-id", "true" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "memory-backend-memfd", "x-use-canonical-path-for-ramblock-id", "true" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "tpm-crb", "ppi", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "tpm-tis", "ppi", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "usb-kbd", "serial", "42" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "usb-mouse", "serial", "42" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "usb-tablet", "serial", "42" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "virtio-blk-device", "discard", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 */ -+ { "virtio-blk-device", "write-zeroes", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_4_0 */ -+ { "VGA", "edid", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_4_0 */ -+ { "secondary-vga", "edid", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_4_0 */ -+ { "bochs-display", "edid", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_4_0 */ -+ { "virtio-vga", "edid", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_4_0 */ -+ { "virtio-gpu-device", "edid", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_4_0 */ -+ { "virtio-device", "use-started", "false" }, -+ /* hw_compat_rhel_8_0 from hw_compat_3_1 - that was added in 4.1 */ -+ { "pcie-root-port-base", "disable-acs", "true" }, -+}; -+const size_t hw_compat_rhel_8_0_len = G_N_ELEMENTS(hw_compat_rhel_8_0); -+ -+/* The same as hw_compat_3_0 + hw_compat_2_12 -+ * except that -+ * there's nothing in 3_0 -+ * migration.decompress-error-check=off was in 7.5 from bz 1584139 -+ */ -+GlobalProperty hw_compat_rhel_7_6[] = { -+ /* hw_compat_rhel_7_6 from hw_compat_2_12 */ -+ { "hda-audio", "use-timer", "false" }, -+ /* hw_compat_rhel_7_6 from hw_compat_2_12 */ -+ { "cirrus-vga", "global-vmstate", "true" }, -+ /* hw_compat_rhel_7_6 from hw_compat_2_12 */ -+ { "VGA", "global-vmstate", "true" }, -+ /* hw_compat_rhel_7_6 from hw_compat_2_12 */ -+ { "vmware-svga", "global-vmstate", "true" }, -+ /* hw_compat_rhel_7_6 from hw_compat_2_12 */ -+ { "qxl-vga", "global-vmstate", "true" }, -+}; -+const size_t hw_compat_rhel_7_6_len = G_N_ELEMENTS(hw_compat_rhel_7_6); + MachineState *current_machine; static char *machine_get_kernel(Object *obj, Error **errp) diff --git a/hw/i386/fw_cfg.c b/hw/i386/fw_cfg.c -index d802d2787f..c7aa39a13e 100644 +index 5c0bcd5f8a..07df7281d2 100644 --- a/hw/i386/fw_cfg.c +++ b/hw/i386/fw_cfg.c -@@ -64,7 +64,8 @@ void fw_cfg_build_smbios(PCMachineState *pcms, FWCfgState *fw_cfg, +@@ -75,7 +75,8 @@ void fw_cfg_build_smbios(PCMachineState *pcms, FWCfgState *fw_cfg, + if (pcmc->smbios_defaults) { /* These values are guest ABI, do not change */ - smbios_set_defaults("QEMU", mc->desc, mc->name, -- pcmc->smbios_uuid_encoded); -+ pcmc->smbios_uuid_encoded, +- smbios_set_defaults("QEMU", mc->desc, mc->name); ++ smbios_set_defaults("QEMU", mc->desc, mc->name, + pcmc->smbios_stream_product, pcmc->smbios_stream_version); } /* tell smbios about cpuid version and features */ diff --git a/hw/net/rtl8139.c b/hw/net/rtl8139.c -index 897c86ec41..2d0db43f49 100644 +index f4dd693abb..ac3a7376ad 100644 --- a/hw/net/rtl8139.c +++ b/hw/net/rtl8139.c -@@ -3169,7 +3169,7 @@ static int rtl8139_pre_save(void *opaque) +@@ -3173,7 +3173,7 @@ static int rtl8139_pre_save(void *opaque) static const VMStateDescription vmstate_rtl8139 = { .name = "rtl8139", @@ -357,7 +259,7 @@ index 897c86ec41..2d0db43f49 100644 .minimum_version_id = 3, .post_load = rtl8139_post_load, .pre_save = rtl8139_pre_save, -@@ -3250,7 +3250,9 @@ static const VMStateDescription vmstate_rtl8139 = { +@@ -3254,7 +3254,9 @@ static const VMStateDescription vmstate_rtl8139 = { VMSTATE_UINT32(tally_counters.TxMCol, RTL8139State), VMSTATE_UINT64(tally_counters.RxOkPhy, RTL8139State), VMSTATE_UINT64(tally_counters.RxOkBrd, RTL8139State), @@ -367,8 +269,30 @@ index 897c86ec41..2d0db43f49 100644 VMSTATE_UINT16(tally_counters.TxAbt, RTL8139State), VMSTATE_UINT16(tally_counters.TxUndrn, RTL8139State), +diff --git a/hw/riscv/virt.c b/hw/riscv/virt.c +index 47e573f85a..ab5a9ec613 100644 +--- a/hw/riscv/virt.c ++++ b/hw/riscv/virt.c +@@ -1402,7 +1402,7 @@ static void virt_build_smbios(RISCVVirtState *s) + product = "KVM Virtual Machine"; + } + +- smbios_set_defaults("QEMU", product, mc->name); ++ smbios_set_defaults("QEMU", product, mc->name, NULL, NULL); + + if (riscv_is_32bit(&s->soc[0])) { + smbios_set_default_processor_family(0x200); +@@ -1920,7 +1920,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + mc->desc = "RISC-V VirtIO board"; + mc->init = virt_machine_init; + mc->max_cpus = VIRT_CPUS_MAX; +- mc->default_cpu_type = TYPE_RISCV_CPU_BASE; ++ mc->default_cpu_type = TYPE_RISCV_CPU_MAX; + mc->block_default_type = IF_VIRTIO; + mc->no_cdrom = 1; + mc->pci_allow_0_address = true; diff --git a/hw/smbios/smbios.c b/hw/smbios/smbios.c -index eed5787b15..68608a3403 100644 +index 1ac063cfb4..03f7a00ed1 100644 --- a/hw/smbios/smbios.c +++ b/hw/smbios/smbios.c @@ -39,6 +39,10 @@ size_t usr_blobs_len; @@ -382,7 +306,7 @@ index eed5787b15..68608a3403 100644 uint8_t *smbios_tables; size_t smbios_tables_len; unsigned smbios_table_max; -@@ -629,7 +633,7 @@ static void smbios_build_type_1_table(void) +@@ -627,7 +631,7 @@ static void smbios_build_type_1_table(void) static void smbios_build_type_2_table(void) { @@ -391,17 +315,16 @@ index eed5787b15..68608a3403 100644 SMBIOS_TABLE_SET_STR(2, manufacturer_str, type2.manufacturer); SMBIOS_TABLE_SET_STR(2, product_str, type2.product); -@@ -1018,16 +1022,52 @@ void smbios_set_default_processor_family(uint16_t processor_family) +@@ -1015,15 +1019,51 @@ void smbios_set_default_processor_family(uint16_t processor_family) + } void smbios_set_defaults(const char *manufacturer, const char *product, - const char *version, -- bool uuid_encoded) -+ bool uuid_encoded, +- const char *version) ++ const char *version, + const char *stream_product, + const char *stream_version) { smbios_have_defaults = true; - smbios_uuid_encoded = uuid_encoded; + /* + * If @stream_product & @stream_version are non-NULL, then @@ -446,24 +369,11 @@ index eed5787b15..68608a3403 100644 SMBIOS_SET_DEFAULT(type2.version, version); SMBIOS_SET_DEFAULT(type3.manufacturer, manufacturer); SMBIOS_SET_DEFAULT(type3.version, version); -diff --git a/hw/timer/i8254_common.c b/hw/timer/i8254_common.c -index 28fdabc321..bad13ec224 100644 ---- a/hw/timer/i8254_common.c -+++ b/hw/timer/i8254_common.c -@@ -229,7 +229,7 @@ static const VMStateDescription vmstate_pit_common = { - .pre_save = pit_dispatch_pre_save, - .post_load = pit_dispatch_post_load, - .fields = (const VMStateField[]) { -- VMSTATE_UINT32_V(channels[0].irq_disabled, PITCommonState, 3), -+ VMSTATE_UINT32(channels[0].irq_disabled, PITCommonState), /* qemu-kvm's v2 had 'flags' here */ - VMSTATE_STRUCT_ARRAY(channels, PITCommonState, 3, 2, - vmstate_pit_channel, PITChannelState), - VMSTATE_INT64(channels[0].next_transition_time, diff --git a/hw/usb/hcd-xhci-pci.c b/hw/usb/hcd-xhci-pci.c -index 4423983308..43b4b71fdf 100644 +index b93c80b09d..f4a2f1a1de 100644 --- a/hw/usb/hcd-xhci-pci.c +++ b/hw/usb/hcd-xhci-pci.c -@@ -104,6 +104,33 @@ static int xhci_pci_vmstate_post_load(void *opaque, int version_id) +@@ -120,6 +120,33 @@ static int xhci_pci_vmstate_post_load(void *opaque, int version_id) return 0; } @@ -497,7 +407,7 @@ index 4423983308..43b4b71fdf 100644 static void usb_xhci_pci_realize(struct PCIDevice *dev, Error **errp) { int ret; -@@ -125,23 +152,12 @@ static void usb_xhci_pci_realize(struct PCIDevice *dev, Error **errp) +@@ -144,23 +171,12 @@ static void usb_xhci_pci_realize(struct PCIDevice *dev, Error **errp) s->xhci.nec_quirks = true; } @@ -524,7 +434,7 @@ index 4423983308..43b4b71fdf 100644 } pci_register_bar(dev, 0, PCI_BASE_ADDRESS_SPACE_MEMORY | -@@ -154,6 +170,14 @@ static void usb_xhci_pci_realize(struct PCIDevice *dev, Error **errp) +@@ -172,6 +188,14 @@ static void usb_xhci_pci_realize(struct PCIDevice *dev, Error **errp) assert(ret > 0); } @@ -539,42 +449,32 @@ index 4423983308..43b4b71fdf 100644 if (s->msix != ON_OFF_AUTO_OFF) { /* TODO check for errors, and should fail when msix=on */ msix_init(dev, s->xhci.numintrs, -@@ -198,11 +222,18 @@ static void xhci_instance_init(Object *obj) - qdev_alias_all_properties(DEVICE(&s->xhci), obj); - } - -+static Property xhci_pci_properties[] = { +@@ -221,6 +245,8 @@ static const Property xhci_pci_properties[] = { + DEFINE_PROP_ON_OFF_AUTO("msix", XHCIPciState, msix, ON_OFF_AUTO_AUTO), + DEFINE_PROP_BOOL("conditional-intr-mapping", XHCIPciState, + conditional_intr_mapping, false), + /* RH bz 1912846 */ + DEFINE_PROP_BOOL("x-rh-late-msi-cap", XHCIPciState, rh_late_msi_cap, true), -+ DEFINE_PROP_END_OF_LIST() -+}; -+ - static void xhci_class_init(ObjectClass *klass, void *data) - { - PCIDeviceClass *k = PCI_DEVICE_CLASS(klass); - DeviceClass *dc = DEVICE_CLASS(klass); + }; -+ device_class_set_props(dc, xhci_pci_properties); - dc->reset = xhci_pci_reset; - dc->vmsd = &vmstate_xhci_pci; - set_bit(DEVICE_CATEGORY_USB, dc->categories); + static void xhci_class_init(ObjectClass *klass, const void *data) diff --git a/hw/usb/hcd-xhci-pci.h b/hw/usb/hcd-xhci-pci.h -index 08f70ce97c..1be7527c1b 100644 +index 5b61ae8455..3170db064b 100644 --- a/hw/usb/hcd-xhci-pci.h +++ b/hw/usb/hcd-xhci-pci.h -@@ -40,6 +40,7 @@ typedef struct XHCIPciState { - XHCIState xhci; +@@ -41,6 +41,7 @@ typedef struct XHCIPciState { OnOffAuto msi; OnOffAuto msix; + bool conditional_intr_mapping; + bool rh_late_msi_cap; /* bz 1912846 */ } XHCIPciState; #endif diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c -index ffd119ebac..0e2be2219c 100644 +index c46f6f9c3e..1805597879 100644 --- a/hw/virtio/virtio-mem.c +++ b/hw/virtio/virtio-mem.c -@@ -1694,8 +1694,9 @@ static Property virtio_mem_properties[] = { +@@ -1699,8 +1699,9 @@ static const Property virtio_mem_properties[] = { #endif DEFINE_PROP_BOOL(VIRTIO_MEM_EARLY_MIGRATION_PROP, VirtIOMEM, early_migration, true), @@ -582,17 +482,32 @@ index ffd119ebac..0e2be2219c 100644 DEFINE_PROP_BOOL(VIRTIO_MEM_DYNAMIC_MEMSLOTS_PROP, VirtIOMEM, - dynamic_memslots, false), + dynamic_memslots, true), - DEFINE_PROP_END_OF_LIST(), }; + static uint64_t virtio_mem_rdm_get_min_granularity(const RamDiscardManager *rdm, diff --git a/include/hw/boards.h b/include/hw/boards.h -index 8b8f6d5c00..0466f9d0f3 100644 +index f94713e6e2..a434b21909 100644 --- a/include/hw/boards.h +++ b/include/hw/boards.h -@@ -512,4 +512,44 @@ extern const size_t hw_compat_2_2_len; - extern GlobalProperty hw_compat_2_1[]; - extern const size_t hw_compat_2_1_len; +@@ -863,4 +863,35 @@ extern const size_t hw_compat_2_7_len; + extern GlobalProperty hw_compat_2_6[]; + extern const size_t hw_compat_2_6_len; ++extern GlobalProperty hw_compat_rhel_10_2[]; ++extern const size_t hw_compat_rhel_10_2_len; ++ ++extern GlobalProperty hw_compat_rhel_10_1[]; ++extern const size_t hw_compat_rhel_10_1_len; ++ ++extern GlobalProperty hw_compat_rhel_10_0[]; ++extern const size_t hw_compat_rhel_10_0_len; ++ ++extern GlobalProperty hw_compat_rhel_9[]; ++extern const size_t hw_compat_rhel_9_len; ++ ++extern GlobalProperty hw_compat_rhel_9_5[]; ++extern const size_t hw_compat_rhel_9_5_len; ++ +extern GlobalProperty hw_compat_rhel_9_4[]; +extern const size_t hw_compat_rhel_9_4_len; + @@ -608,54 +523,30 @@ index 8b8f6d5c00..0466f9d0f3 100644 +extern GlobalProperty hw_compat_rhel_9_0[]; +extern const size_t hw_compat_rhel_9_0_len; + -+extern GlobalProperty hw_compat_rhel_8_6[]; -+extern const size_t hw_compat_rhel_8_6_len; -+ -+extern GlobalProperty hw_compat_rhel_8_5[]; -+extern const size_t hw_compat_rhel_8_5_len; -+ -+extern GlobalProperty hw_compat_rhel_8_4[]; -+extern const size_t hw_compat_rhel_8_4_len; -+ -+extern GlobalProperty hw_compat_rhel_8_3[]; -+extern const size_t hw_compat_rhel_8_3_len; -+ -+extern GlobalProperty hw_compat_rhel_8_2[]; -+extern const size_t hw_compat_rhel_8_2_len; -+ -+extern GlobalProperty hw_compat_rhel_8_1[]; -+extern const size_t hw_compat_rhel_8_1_len; -+ -+extern GlobalProperty hw_compat_rhel_8_0[]; -+extern const size_t hw_compat_rhel_8_0_len; -+ -+extern GlobalProperty hw_compat_rhel_7_6[]; -+extern const size_t hw_compat_rhel_7_6_len; -+ +extern const char *rhel_old_machine_deprecation; #endif diff --git a/include/hw/firmware/smbios.h b/include/hw/firmware/smbios.h -index 8d3fb2fb3b..d9d6d7a169 100644 +index f066ab7262..e805d25fbe 100644 --- a/include/hw/firmware/smbios.h +++ b/include/hw/firmware/smbios.h -@@ -332,7 +332,9 @@ void smbios_entry_add(QemuOpts *opts, Error **errp); +@@ -331,7 +331,9 @@ void smbios_add_usr_blob_size(size_t size); + void smbios_entry_add(QemuOpts *opts, Error **errp); void smbios_set_cpuid(uint32_t version, uint32_t features); void smbios_set_defaults(const char *manufacturer, const char *product, - const char *version, -- bool uuid_encoded); -+ bool uuid_encoded, +- const char *version); ++ const char *version, + const char *stream_product, + const char *stream_version); void smbios_set_default_processor_family(uint16_t processor_family); uint8_t *smbios_get_table_legacy(size_t *length, Error **errp); void smbios_get_tables(MachineState *ms, diff --git a/include/hw/i386/pc.h b/include/hw/i386/pc.h -index 27a68071d7..ebd8f973f2 100644 +index 79b72c54dd..3b4ea24c20 100644 --- a/include/hw/i386/pc.h +++ b/include/hw/i386/pc.h -@@ -112,6 +112,9 @@ struct PCMachineClass { +@@ -103,6 +103,9 @@ struct PCMachineClass { + bool smbios_defaults; bool smbios_legacy_mode; - bool smbios_uuid_encoded; SmbiosEntryPointType default_smbios_ep_type; + /* New fields needed for Windows HardwareID-6 matching */ + const char *smbios_stream_product; diff --git a/0007-Add-aarch64-machine-types.patch b/0007-Add-aarch64-machine-types.patch deleted file mode 100644 index b92d07d..0000000 --- a/0007-Add-aarch64-machine-types.patch +++ /dev/null @@ -1,430 +0,0 @@ -From 3afc6e4cb6725d01b8f89207701bca199c9ecc9f Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Fri, 19 Oct 2018 12:53:31 +0200 -Subject: Add aarch64 machine types - -Adding changes to add RHEL machine types for aarch64 architecture. - -Signed-off-by: Miroslav Rezanina ---- - hw/arm/virt.c | 299 +++++++++++++++++++++++++++++++++++++++++- - include/hw/arm/virt.h | 8 ++ - 2 files changed, 306 insertions(+), 1 deletion(-) - -diff --git a/hw/arm/virt.c b/hw/arm/virt.c -index 36e9b4b4e9..22bc345137 100644 ---- a/hw/arm/virt.c -+++ b/hw/arm/virt.c -@@ -101,6 +101,7 @@ static void arm_virt_compat_set(MachineClass *mc) - arm_virt_compat_len); - } - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - #define DEFINE_VIRT_MACHINE_LATEST(major, minor, latest) \ - static void virt_##major##_##minor##_class_init(ObjectClass *oc, \ - void *data) \ -@@ -128,7 +129,63 @@ static void arm_virt_compat_set(MachineClass *mc) - DEFINE_VIRT_MACHINE_LATEST(major, minor, true) - #define DEFINE_VIRT_MACHINE(major, minor) \ - DEFINE_VIRT_MACHINE_LATEST(major, minor, false) -+#endif /* disabled for RHEL */ -+ -+/* -+ * This variable is for changes to properties that are RHEL specific, -+ * different to the current upstream and to be applied to the latest -+ * machine type. They may be overriden by older machine compats. -+ * -+ * virtio-net-pci variant romfiles are not needed because edk2 does -+ * fully support the pxe boot. Besides virtio romfiles are not shipped -+ * on rhel/aarch64. -+ */ -+GlobalProperty arm_rhel_compat[] = { -+ {"virtio-net-pci", "romfile", "" }, -+ {"virtio-net-pci-transitional", "romfile", "" }, -+ {"virtio-net-pci-non-transitional", "romfile", "" }, -+}; -+const size_t arm_rhel_compat_len = G_N_ELEMENTS(arm_rhel_compat); - -+/* -+ * This cannot be called from the rhel_virt_class_init() because -+ * TYPE_RHEL_MACHINE is abstract and mc->compat_props g_ptr_array_new() -+ * only is called on virt-rhelm.n.s non abstract class init. -+ */ -+static void arm_rhel_compat_set(MachineClass *mc) -+{ -+ compat_props_add(mc->compat_props, arm_rhel_compat, -+ arm_rhel_compat_len); -+} -+ -+#define DEFINE_RHEL_MACHINE_LATEST(m, n, s, latest) \ -+ static void rhel##m##n##s##_virt_class_init(ObjectClass *oc, \ -+ void *data) \ -+ { \ -+ MachineClass *mc = MACHINE_CLASS(oc); \ -+ arm_rhel_compat_set(mc); \ -+ rhel##m##n##s##_virt_options(mc); \ -+ mc->desc = "RHEL " # m "." # n "." # s " ARM Virtual Machine"; \ -+ if (latest) { \ -+ mc->alias = "virt"; \ -+ mc->is_default = 1; \ -+ } \ -+ } \ -+ static const TypeInfo rhel##m##n##s##_machvirt_info = { \ -+ .name = MACHINE_TYPE_NAME("virt-rhel" # m "." # n "." # s), \ -+ .parent = TYPE_RHEL_MACHINE, \ -+ .class_init = rhel##m##n##s##_virt_class_init, \ -+ }; \ -+ static void rhel##m##n##s##_machvirt_init(void) \ -+ { \ -+ type_register_static(&rhel##m##n##s##_machvirt_info); \ -+ } \ -+ type_init(rhel##m##n##s##_machvirt_init); -+ -+#define DEFINE_RHEL_MACHINE_AS_LATEST(major, minor, subminor) \ -+ DEFINE_RHEL_MACHINE_LATEST(major, minor, subminor, true) -+#define DEFINE_RHEL_MACHINE(major, minor, subminor) \ -+ DEFINE_RHEL_MACHINE_LATEST(major, minor, subminor, false) - - /* Number of external interrupt lines to configure the GIC with */ - #define NUM_IRQS 256 -@@ -2355,6 +2412,7 @@ static void machvirt_init(MachineState *machine) - qemu_add_machine_init_done_notifier(&vms->machine_done); - } - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static bool virt_get_secure(Object *obj, Error **errp) - { - VirtMachineState *vms = VIRT_MACHINE(obj); -@@ -2382,6 +2440,7 @@ static void virt_set_virt(Object *obj, bool value, Error **errp) - - vms->virt = value; - } -+#endif /* disabled for RHEL */ - - static bool virt_get_highmem(Object *obj, Error **errp) - { -@@ -2397,6 +2456,7 @@ static void virt_set_highmem(Object *obj, bool value, Error **errp) - vms->highmem = value; - } - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static bool virt_get_compact_highmem(Object *obj, Error **errp) - { - VirtMachineState *vms = VIRT_MACHINE(obj); -@@ -2410,6 +2470,7 @@ static void virt_set_compact_highmem(Object *obj, bool value, Error **errp) - - vms->highmem_compact = value; - } -+#endif /* disabled for RHEL */ - - static bool virt_get_highmem_redists(Object *obj, Error **errp) - { -@@ -2453,7 +2514,6 @@ static void virt_set_highmem_mmio(Object *obj, bool value, Error **errp) - vms->highmem_mmio = value; - } - -- - static bool virt_get_its(Object *obj, Error **errp) - { - VirtMachineState *vms = VIRT_MACHINE(obj); -@@ -2468,6 +2528,7 @@ static void virt_set_its(Object *obj, bool value, Error **errp) - vms->its = value; - } - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static bool virt_get_dtb_randomness(Object *obj, Error **errp) - { - VirtMachineState *vms = VIRT_MACHINE(obj); -@@ -2481,6 +2542,7 @@ static void virt_set_dtb_randomness(Object *obj, bool value, Error **errp) - - vms->dtb_randomness = value; - } -+#endif /* disabled for RHEL */ - - static char *virt_get_oem_id(Object *obj, Error **errp) - { -@@ -2564,6 +2626,7 @@ static void virt_set_ras(Object *obj, bool value, Error **errp) - vms->ras = value; - } - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static bool virt_get_mte(Object *obj, Error **errp) - { - VirtMachineState *vms = VIRT_MACHINE(obj); -@@ -2577,6 +2640,7 @@ static void virt_set_mte(Object *obj, bool value, Error **errp) - - vms->mte = value; - } -+#endif /* disabled for RHEL */ - - static char *virt_get_gic_version(Object *obj, Error **errp) - { -@@ -2949,6 +3013,7 @@ static int virt_kvm_type(MachineState *ms, const char *type_str) - return fixed_ipa ? 0 : requested_pa_size; - } - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static void virt_machine_class_init(ObjectClass *oc, void *data) - { - MachineClass *mc = MACHINE_CLASS(oc); -@@ -3463,3 +3528,235 @@ static void virt_machine_2_6_options(MachineClass *mc) - vmc->no_pmu = true; - } - DEFINE_VIRT_MACHINE(2, 6) -+#endif /* disabled for RHEL */ -+ -+static void rhel_machine_class_init(ObjectClass *oc, void *data) -+{ -+ MachineClass *mc = MACHINE_CLASS(oc); -+ HotplugHandlerClass *hc = HOTPLUG_HANDLER_CLASS(oc); -+ arm_virt_compat_set(mc); -+ -+ mc->family = "virt-rhel-Z"; -+ mc->init = machvirt_init; -+ /* Maximum supported VCPU count for all virt-rhel* machines */ -+ mc->max_cpus = 384; -+#ifdef CONFIG_TPM -+ machine_class_allow_dynamic_sysbus_dev(mc, TYPE_TPM_TIS_SYSBUS); -+#endif -+ mc->block_default_type = IF_VIRTIO; -+ mc->no_cdrom = 1; -+ mc->pci_allow_0_address = true; -+ /* We know we will never create a pre-ARMv7 CPU which needs 1K pages */ -+ mc->minimum_page_bits = 12; -+ mc->possible_cpu_arch_ids = virt_possible_cpu_arch_ids; -+ mc->cpu_index_to_instance_props = virt_cpu_index_to_props; -+ mc->default_cpu_type = ARM_CPU_TYPE_NAME("cortex-a57"); -+ mc->get_default_cpu_node_id = virt_get_default_cpu_node_id; -+ mc->kvm_type = virt_kvm_type; -+ assert(!mc->get_hotplug_handler); -+ mc->get_hotplug_handler = virt_machine_get_hotplug_handler; -+ hc->pre_plug = virt_machine_device_pre_plug_cb; -+ hc->plug = virt_machine_device_plug_cb; -+ hc->unplug_request = virt_machine_device_unplug_request_cb; -+ hc->unplug = virt_machine_device_unplug_cb; -+ mc->nvdimm_supported = true; -+ mc->smp_props.clusters_supported = true; -+ mc->auto_enable_numa_with_memhp = true; -+ mc->auto_enable_numa_with_memdev = true; -+ /* platform instead of architectural choice */ -+ mc->cpu_cluster_has_numa_boundary = true; -+ mc->default_ram_id = "mach-virt.ram"; -+ mc->default_nic = "virtio-net-pci"; -+ -+ object_class_property_add(oc, "acpi", "OnOffAuto", -+ virt_get_acpi, virt_set_acpi, -+ NULL, NULL); -+ object_class_property_set_description(oc, "acpi", -+ "Enable ACPI"); -+ -+ object_class_property_add_bool(oc, "highmem", virt_get_highmem, -+ virt_set_highmem); -+ object_class_property_set_description(oc, "highmem", -+ "Set on/off to enable/disable using " -+ "physical address space above 32 bits"); -+ -+ object_class_property_add_bool(oc, "highmem-redists", -+ virt_get_highmem_redists, -+ virt_set_highmem_redists); -+ object_class_property_set_description(oc, "highmem-redists", -+ "Set on/off to enable/disable high " -+ "memory region for GICv3 or GICv4 " -+ "redistributor"); -+ -+ object_class_property_add_bool(oc, "highmem-ecam", -+ virt_get_highmem_ecam, -+ virt_set_highmem_ecam); -+ object_class_property_set_description(oc, "highmem-ecam", -+ "Set on/off to enable/disable high " -+ "memory region for PCI ECAM"); -+ -+ object_class_property_add_bool(oc, "highmem-mmio", -+ virt_get_highmem_mmio, -+ virt_set_highmem_mmio); -+ object_class_property_set_description(oc, "highmem-mmio", -+ "Set on/off to enable/disable high " -+ "memory region for PCI MMIO"); -+ -+ object_class_property_add_str(oc, "gic-version", virt_get_gic_version, -+ virt_set_gic_version); -+ object_class_property_set_description(oc, "gic-version", -+ "Set GIC version. " -+ "Valid values are 2, 3, host and max"); -+ -+ object_class_property_add_str(oc, "iommu", virt_get_iommu, virt_set_iommu); -+ object_class_property_set_description(oc, "iommu", -+ "Set the IOMMU type. " -+ "Valid values are none and smmuv3"); -+ -+ object_class_property_add_bool(oc, "default-bus-bypass-iommu", -+ virt_get_default_bus_bypass_iommu, -+ virt_set_default_bus_bypass_iommu); -+ object_class_property_set_description(oc, "default-bus-bypass-iommu", -+ "Set on/off to enable/disable " -+ "bypass_iommu for default root bus"); -+ -+ object_class_property_add_bool(oc, "ras", virt_get_ras, -+ virt_set_ras); -+ object_class_property_set_description(oc, "ras", -+ "Set on/off to enable/disable reporting host memory errors " -+ "to a KVM guest using ACPI and guest external abort exceptions"); -+ -+ object_class_property_add_bool(oc, "its", virt_get_its, -+ virt_set_its); -+ object_class_property_set_description(oc, "its", -+ "Set on/off to enable/disable " -+ "ITS instantiation"); -+ -+ object_class_property_add_str(oc, "x-oem-id", -+ virt_get_oem_id, -+ virt_set_oem_id); -+ object_class_property_set_description(oc, "x-oem-id", -+ "Override the default value of field OEMID " -+ "in ACPI table header." -+ "The string may be up to 6 bytes in size"); -+ -+ -+ object_class_property_add_str(oc, "x-oem-table-id", -+ virt_get_oem_table_id, -+ virt_set_oem_table_id); -+ object_class_property_set_description(oc, "x-oem-table-id", -+ "Override the default value of field OEM Table ID " -+ "in ACPI table header." -+ "The string may be up to 8 bytes in size"); -+} -+ -+static void rhel_virt_instance_init(Object *obj) -+{ -+ VirtMachineState *vms = VIRT_MACHINE(obj); -+ VirtMachineClass *vmc = VIRT_MACHINE_GET_CLASS(vms); -+ -+ /* EL3 is disabled by default and non-configurable for RHEL */ -+ vms->secure = false; -+ -+ /* EL2 is disabled by default and non-configurable for RHEL */ -+ vms->virt = false; -+ -+ /* High memory is enabled by default */ -+ vms->highmem = true; -+ vms->highmem_compact = !vmc->no_highmem_compact; -+ vms->gic_version = VIRT_GIC_VERSION_NOSEL; -+ -+ vms->highmem_ecam = !vmc->no_highmem_ecam; -+ vms->highmem_mmio = true; -+ vms->highmem_redists = true; -+ -+ if (vmc->no_its) { -+ vms->its = false; -+ } else { -+ /* Default allows ITS instantiation */ -+ vms->its = true; -+ -+ if (vmc->no_tcg_its) { -+ vms->tcg_its = false; -+ } else { -+ vms->tcg_its = true; -+ } -+ } -+ -+ /* Default disallows iommu instantiation */ -+ vms->iommu = VIRT_IOMMU_NONE; -+ -+ /* The default root bus is attached to iommu by default */ -+ vms->default_bus_bypass_iommu = false; -+ -+ /* Default disallows RAS instantiation and is non-configurable for RHEL */ -+ vms->ras = false; -+ -+ /* MTE is disabled by default and non-configurable for RHEL */ -+ vms->mte = false; -+ -+ /* Supply kaslr-seed and rng-seed by default, non-configurable for RHEL */ -+ vms->dtb_randomness = true; -+ -+ vms->irqmap = a15irqmap; -+ -+ virt_flash_create(vms); -+ -+ vms->oem_id = g_strndup(ACPI_BUILD_APPNAME6, 6); -+ vms->oem_table_id = g_strndup(ACPI_BUILD_APPNAME8, 8); -+} -+ -+static const TypeInfo rhel_machine_info = { -+ .name = TYPE_RHEL_MACHINE, -+ .parent = TYPE_MACHINE, -+ .abstract = true, -+ .instance_size = sizeof(VirtMachineState), -+ .class_size = sizeof(VirtMachineClass), -+ .class_init = rhel_machine_class_init, -+ .instance_init = rhel_virt_instance_init, -+ .interfaces = (InterfaceInfo[]) { -+ { TYPE_HOTPLUG_HANDLER }, -+ { } -+ }, -+}; -+ -+static void rhel_machine_init(void) -+{ -+ type_register_static(&rhel_machine_info); -+} -+type_init(rhel_machine_init); -+ -+static void rhel940_virt_options(MachineClass *mc) -+{ -+} -+DEFINE_RHEL_MACHINE_AS_LATEST(9, 4, 0) -+ -+static void rhel920_virt_options(MachineClass *mc) -+{ -+ rhel940_virt_options(mc); -+ -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_4, hw_compat_rhel_9_4_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_3, hw_compat_rhel_9_3_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_2, hw_compat_rhel_9_2_len); -+ -+ /* RHEL 9.4 is the first supported release */ -+ mc->deprecation_reason = -+ "machine types for versions prior to 9.4 are deprecated"; -+} -+DEFINE_RHEL_MACHINE(9, 2, 0) -+ -+static void rhel900_virt_options(MachineClass *mc) -+{ -+ VirtMachineClass *vmc = VIRT_MACHINE_CLASS(OBJECT_CLASS(mc)); -+ -+ rhel920_virt_options(mc); -+ -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_1, hw_compat_rhel_9_1_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_0, hw_compat_rhel_9_0_len); -+ -+ /* Disable FEAT_LPA2 since old kernels (<= v5.12) don't boot with that feature */ -+ vmc->no_tcg_lpa2 = true; -+ /* Compact layout for high memory regions was introduced with 9.2.0 */ -+ vmc->no_highmem_compact = true; -+} -+DEFINE_RHEL_MACHINE(9, 0, 0) -diff --git a/include/hw/arm/virt.h b/include/hw/arm/virt.h -index bb486d36b1..237fc77bda 100644 ---- a/include/hw/arm/virt.h -+++ b/include/hw/arm/virt.h -@@ -179,9 +179,17 @@ struct VirtMachineState { - - #define VIRT_ECAM_ID(high) (high ? VIRT_HIGH_PCIE_ECAM : VIRT_PCIE_ECAM) - -+#if 0 /* disabled for Red Hat Enterprise Linux */ - #define TYPE_VIRT_MACHINE MACHINE_TYPE_NAME("virt") - OBJECT_DECLARE_TYPE(VirtMachineState, VirtMachineClass, VIRT_MACHINE) - -+#else -+#define TYPE_RHEL_MACHINE MACHINE_TYPE_NAME("virt-rhel") -+typedef struct VirtMachineClass VirtMachineClass; -+typedef struct VirtMachineState VirtMachineState; -+DECLARE_OBJ_CHECKERS(VirtMachineState, VirtMachineClass, VIRT_MACHINE, TYPE_RHEL_MACHINE) -+#endif -+ - void virt_acpi_setup(VirtMachineState *vms); - bool virt_is_acpi_enabled(VirtMachineState *vms); - --- -2.39.3 - diff --git a/0007-meson-temporarily-disable-Wunused-function.patch b/0007-meson-temporarily-disable-Wunused-function.patch new file mode 100644 index 0000000..c0313b5 --- /dev/null +++ b/0007-meson-temporarily-disable-Wunused-function.patch @@ -0,0 +1,36 @@ +From 7eff7b32584a50d73053b4e3e007621b14ebf766 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Wed, 3 Jul 2024 13:32:32 +0100 +Subject: meson: temporarily disable -Wunused-function +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +Deleting the upstream versioned machine types will leave some functions +unused until RHEL machine types are added once again. Temporarily +disable the -Wunused-function warning to preserve bisectability with +fine grained patch splits. + +Signed-off-by: Daniel P. Berrangé + +Rebase notes (9.1.0) +- New patch +--- + meson.build | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/meson.build b/meson.build +index 50c774a195..0e0120a613 100644 +--- a/meson.build ++++ b/meson.build +@@ -757,6 +757,7 @@ warn_flags = [ + '-Wno-string-plus-int', + '-Wno-tautological-type-limit-compare', + '-Wno-typedef-redefinition', ++ '-Wno-unused-function', + ] + + if host_os != 'darwin' +-- +2.39.3 + diff --git a/0008-Add-s390x-machine-types.patch b/0008-Add-s390x-machine-types.patch deleted file mode 100644 index ea9fe16..0000000 --- a/0008-Add-s390x-machine-types.patch +++ /dev/null @@ -1,273 +0,0 @@ -From fa1d70b9a9cfe020e7ebe7798ebb70314658ccf7 Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Fri, 19 Oct 2018 13:47:32 +0200 -Subject: Add s390x machine types - -Adding changes to add RHEL machine types for s390x architecture. - -Signed-off-by: Miroslav Rezanina ---- - hw/s390x/s390-virtio-ccw.c | 159 +++++++++++++++++++++++++++++++ - target/s390x/cpu_models.c | 11 +++ - target/s390x/cpu_models.h | 2 + - target/s390x/cpu_models_sysemu.c | 2 + - 4 files changed, 174 insertions(+) - -diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c -index b1dcb3857f..ff753a29e0 100644 ---- a/hw/s390x/s390-virtio-ccw.c -+++ b/hw/s390x/s390-virtio-ccw.c -@@ -859,6 +859,7 @@ bool css_migration_enabled(void) - } \ - type_init(ccw_machine_register_##suffix) - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static void ccw_machine_9_0_instance_options(MachineState *machine) - { - } -@@ -1272,6 +1273,164 @@ static void ccw_machine_2_4_class_options(MachineClass *mc) - compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); - } - DEFINE_CCW_MACHINE(2_4, "2.4", false); -+#endif -+ -+ -+static void ccw_machine_rhel940_instance_options(MachineState *machine) -+{ -+} -+ -+static void ccw_machine_rhel940_class_options(MachineClass *mc) -+{ -+} -+DEFINE_CCW_MACHINE(rhel940, "rhel9.4.0", true); -+ -+static void ccw_machine_rhel920_instance_options(MachineState *machine) -+{ -+ ccw_machine_rhel940_instance_options(machine); -+} -+ -+static void ccw_machine_rhel920_class_options(MachineClass *mc) -+{ -+ ccw_machine_rhel940_class_options(mc); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_4, hw_compat_rhel_9_4_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_3, hw_compat_rhel_9_3_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_2, hw_compat_rhel_9_2_len); -+ mc->smp_props.drawers_supported = false; /* from ccw_machine_8_1 */ -+ mc->smp_props.books_supported = false; /* from ccw_machine_8_1 */ -+} -+DEFINE_CCW_MACHINE(rhel920, "rhel9.2.0", false); -+ -+static void ccw_machine_rhel900_instance_options(MachineState *machine) -+{ -+ static const S390FeatInit qemu_cpu_feat = { S390_FEAT_LIST_QEMU_V6_2 }; -+ -+ ccw_machine_rhel920_instance_options(machine); -+ -+ s390_set_qemu_cpu_model(0x3906, 14, 2, qemu_cpu_feat); -+ s390_cpudef_featoff_greater(16, 1, S390_FEAT_PAIE); -+} -+ -+static void ccw_machine_rhel900_class_options(MachineClass *mc) -+{ -+ S390CcwMachineClass *s390mc = S390_CCW_MACHINE_CLASS(mc); -+ static GlobalProperty compat[] = { -+ { TYPE_S390_PCI_DEVICE, "interpret", "off", }, -+ { TYPE_S390_PCI_DEVICE, "forwarding-assist", "off", }, -+ }; -+ -+ ccw_machine_rhel920_class_options(mc); -+ -+ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_1, hw_compat_rhel_9_1_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_0, hw_compat_rhel_9_0_len); -+ s390mc->max_threads = S390_MAX_CPUS; -+} -+DEFINE_CCW_MACHINE(rhel900, "rhel9.0.0", false); -+ -+static void ccw_machine_rhel860_instance_options(MachineState *machine) -+{ -+ /* Note: The -rhel8.6.0 and -rhel9.0.0 machines are technically identical */ -+ ccw_machine_rhel900_instance_options(machine); -+} -+ -+static void ccw_machine_rhel860_class_options(MachineClass *mc) -+{ -+ static GlobalProperty compat[] = { -+ { TYPE_S390_PCI_DEVICE, "interpret", "on", }, -+ { TYPE_S390_PCI_DEVICE, "forwarding-assist", "on", }, -+ }; -+ -+ ccw_machine_rhel900_class_options(mc); -+ compat_props_add(mc->compat_props, hw_compat_rhel_8_6, hw_compat_rhel_8_6_len); -+ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); -+ -+ /* All RHEL machines for prior major releases are deprecated */ -+ mc->deprecation_reason = rhel_old_machine_deprecation; -+} -+DEFINE_CCW_MACHINE(rhel860, "rhel8.6.0", false); -+ -+static void ccw_machine_rhel850_instance_options(MachineState *machine) -+{ -+ static const S390FeatInit qemu_cpu_feat = { S390_FEAT_LIST_QEMU_V6_0 }; -+ -+ ccw_machine_rhel860_instance_options(machine); -+ -+ s390_set_qemu_cpu_model(0x2964, 13, 2, qemu_cpu_feat); -+ -+ s390_cpudef_featoff_greater(16, 1, S390_FEAT_NNPA); -+ s390_cpudef_featoff_greater(16, 1, S390_FEAT_VECTOR_PACKED_DECIMAL_ENH2); -+ s390_cpudef_featoff_greater(16, 1, S390_FEAT_BEAR_ENH); -+ s390_cpudef_featoff_greater(16, 1, S390_FEAT_RDP); -+ s390_cpudef_featoff_greater(16, 1, S390_FEAT_PAI); -+} -+ -+static void ccw_machine_rhel850_class_options(MachineClass *mc) -+{ -+ static GlobalProperty compat[] = { -+ { TYPE_S390_PCI_DEVICE, "interpret", "off", }, -+ { TYPE_S390_PCI_DEVICE, "forwarding-assist", "off", }, -+ }; -+ -+ ccw_machine_rhel860_class_options(mc); -+ compat_props_add(mc->compat_props, hw_compat_rhel_8_5, hw_compat_rhel_8_5_len); -+ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); -+ mc->smp_props.prefer_sockets = true; -+} -+DEFINE_CCW_MACHINE(rhel850, "rhel8.5.0", false); -+ -+static void ccw_machine_rhel840_instance_options(MachineState *machine) -+{ -+ ccw_machine_rhel850_instance_options(machine); -+} -+ -+static void ccw_machine_rhel840_class_options(MachineClass *mc) -+{ -+ ccw_machine_rhel850_class_options(mc); -+ compat_props_add(mc->compat_props, hw_compat_rhel_8_4, hw_compat_rhel_8_4_len); -+} -+DEFINE_CCW_MACHINE(rhel840, "rhel8.4.0", false); -+ -+static void ccw_machine_rhel820_instance_options(MachineState *machine) -+{ -+ ccw_machine_rhel840_instance_options(machine); -+} -+ -+static void ccw_machine_rhel820_class_options(MachineClass *mc) -+{ -+ ccw_machine_rhel840_class_options(mc); -+ mc->fixup_ram_size = s390_fixup_ram_size; -+ /* we did not publish a rhel8.3.0 machine */ -+ compat_props_add(mc->compat_props, hw_compat_rhel_8_3, hw_compat_rhel_8_3_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_8_2, hw_compat_rhel_8_2_len); -+} -+DEFINE_CCW_MACHINE(rhel820, "rhel8.2.0", false); -+ -+static void ccw_machine_rhel760_instance_options(MachineState *machine) -+{ -+ static const S390FeatInit qemu_cpu_feat = { S390_FEAT_LIST_QEMU_V3_1 }; -+ -+ ccw_machine_rhel820_instance_options(machine); -+ -+ s390_set_qemu_cpu_model(0x2827, 12, 2, qemu_cpu_feat); -+ -+ /* The multiple-epoch facility was not available with rhel7.6.0 on z14GA1 */ -+ s390_cpudef_featoff(14, 1, S390_FEAT_MULTIPLE_EPOCH); -+ s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_QSIE); -+ s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_QTOUE); -+ s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_STOE); -+ s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_STOUE); -+} -+ -+static void ccw_machine_rhel760_class_options(MachineClass *mc) -+{ -+ ccw_machine_rhel820_class_options(mc); -+ /* We never published the s390x version of RHEL-AV 8.0 and 8.1, so add this here */ -+ compat_props_add(mc->compat_props, hw_compat_rhel_8_1, hw_compat_rhel_8_1_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_8_0, hw_compat_rhel_8_0_len); -+ compat_props_add(mc->compat_props, hw_compat_rhel_7_6, hw_compat_rhel_7_6_len); -+} -+DEFINE_CCW_MACHINE(rhel760, "rhel7.6.0", false); - - static void ccw_machine_register_types(void) - { -diff --git a/target/s390x/cpu_models.c b/target/s390x/cpu_models.c -index 8ed3bb6a27..370b3b3065 100644 ---- a/target/s390x/cpu_models.c -+++ b/target/s390x/cpu_models.c -@@ -46,6 +46,9 @@ - * of a following release have been a superset of the previous release. With - * generation 15 one base feature and one optional feature have been deprecated. - */ -+ -+#define RHEL_CPU_DEPRECATION "use at least 'z14', or 'host' / 'qemu' / 'max'" -+ - static S390CPUDef s390_cpu_defs[] = { - CPUDEF_INIT(0x2064, 7, 1, 38, 0x00000000U, "z900", "IBM zSeries 900 GA1"), - CPUDEF_INIT(0x2064, 7, 2, 38, 0x00000000U, "z900.2", "IBM zSeries 900 GA2"), -@@ -866,22 +869,30 @@ static void s390_host_cpu_model_class_init(ObjectClass *oc, void *data) - static void s390_base_cpu_model_class_init(ObjectClass *oc, void *data) - { - S390CPUClass *xcc = S390_CPU_CLASS(oc); -+ CPUClass *cc = CPU_CLASS(oc); - - /* all base models are migration safe */ - xcc->cpu_def = (const S390CPUDef *) data; - xcc->is_migration_safe = true; - xcc->is_static = true; - xcc->desc = xcc->cpu_def->desc; -+ if (xcc->cpu_def->gen < 14) { -+ cc->deprecation_note = RHEL_CPU_DEPRECATION; -+ } - } - - static void s390_cpu_model_class_init(ObjectClass *oc, void *data) - { - S390CPUClass *xcc = S390_CPU_CLASS(oc); -+ CPUClass *cc = CPU_CLASS(oc); - - /* model that can change between QEMU versions */ - xcc->cpu_def = (const S390CPUDef *) data; - xcc->is_migration_safe = true; - xcc->desc = xcc->cpu_def->desc; -+ if (xcc->cpu_def->gen < 14) { -+ cc->deprecation_note = RHEL_CPU_DEPRECATION; -+ } - } - - static void s390_qemu_cpu_model_class_init(ObjectClass *oc, void *data) -diff --git a/target/s390x/cpu_models.h b/target/s390x/cpu_models.h -index d7b8912989..1a806a97c4 100644 ---- a/target/s390x/cpu_models.h -+++ b/target/s390x/cpu_models.h -@@ -38,6 +38,8 @@ typedef struct S390CPUDef { - S390FeatBitmap full_feat; - /* used to init full_feat from generated data */ - S390FeatInit full_init; -+ /* if deprecated, provides a suggestion */ -+ const char *deprecation_note; - } S390CPUDef; - - /* CPU model based on a CPU definition */ -diff --git a/target/s390x/cpu_models_sysemu.c b/target/s390x/cpu_models_sysemu.c -index 0728bfcc20..ca2e5d91e2 100644 ---- a/target/s390x/cpu_models_sysemu.c -+++ b/target/s390x/cpu_models_sysemu.c -@@ -59,6 +59,7 @@ static void create_cpu_model_list(ObjectClass *klass, void *opaque) - CpuDefinitionInfo *info; - char *name = g_strdup(object_class_get_name(klass)); - S390CPUClass *scc = S390_CPU_CLASS(klass); -+ CPUClass *cc = CPU_CLASS(klass); - - /* strip off the -s390x-cpu */ - g_strrstr(name, "-" TYPE_S390_CPU)[0] = 0; -@@ -68,6 +69,7 @@ static void create_cpu_model_list(ObjectClass *klass, void *opaque) - info->migration_safe = scc->is_migration_safe; - info->q_static = scc->is_static; - info->q_typename = g_strdup(object_class_get_name(klass)); -+ info->deprecated = !!cc->deprecation_note; - /* check for unavailable features */ - if (cpu_list_data->model) { - Object *obj; --- -2.39.3 - diff --git a/0008-Remove-upstream-machine-types-for-aarch64-s390x-and-.patch b/0008-Remove-upstream-machine-types-for-aarch64-s390x-and-.patch new file mode 100644 index 0000000..86d7669 --- /dev/null +++ b/0008-Remove-upstream-machine-types-for-aarch64-s390x-and-.patch @@ -0,0 +1,102 @@ +From fca16c8b4612edfa63d7882379897a9907dde738 Mon Sep 17 00:00:00 2001 +From: Miroslav Rezanina +Date: Wed, 10 Jul 2024 02:25:51 -0400 +Subject: Remove upstream machine types for aarch64, s390x and x86_64 + architectures +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +We will replace upstream machine types on supported architectures with RHEL +machine types. + +Signed-off-by: Daniel P. Berrangé +Signed-off-by: Miroslav Rezanina + +Rebase notes (9.1.0): +- Split off commits adding RHEL machine types +--- + hw/arm/virt.c | 2 ++ + hw/i386/pc_piix.c | 2 ++ + hw/i386/pc_q35.c | 2 ++ + hw/s390x/s390-virtio-ccw.c | 3 +++ + 4 files changed, 9 insertions(+) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index 1800981317..e6e98fef1c 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -3459,6 +3459,7 @@ static void machvirt_machine_init(void) + } + type_init(machvirt_machine_init); + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void virt_machine_10_1_options(MachineClass *mc) + { + } +@@ -3634,3 +3635,4 @@ static void virt_machine_4_1_options(MachineClass *mc) + mc->auto_enable_numa_with_memhp = false; + } + DEFINE_VIRT_MACHINE(4, 1) ++#endif /* disabled for RHEL */ +diff --git a/hw/i386/pc_piix.c b/hw/i386/pc_piix.c +index c03324281b..acf010e20f 100644 +--- a/hw/i386/pc_piix.c ++++ b/hw/i386/pc_piix.c +@@ -475,6 +475,7 @@ static void pc_i440fx_init(MachineState *machine) + #define DEFINE_I440FX_MACHINE_AS_LATEST(major, minor) \ + DEFINE_PC_VER_MACHINE(pc_i440fx, "pc-i440fx", pc_i440fx_init, true, "pc", major, minor); + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void pc_i440fx_machine_options(MachineClass *m) + { + PCMachineClass *pcmc = PC_MACHINE_CLASS(m); +@@ -802,6 +803,7 @@ static void pc_i440fx_machine_2_6_options(MachineClass *m) + } + + DEFINE_I440FX_MACHINE(2, 6); ++#endif /* Disabled for Red Hat Enterprise Linux */ + + #ifdef CONFIG_ISAPC + static void isapc_machine_options(MachineClass *m) +diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c +index b309b2b378..2203ffd67e 100644 +--- a/hw/i386/pc_q35.c ++++ b/hw/i386/pc_q35.c +@@ -374,6 +374,7 @@ static void pc_q35_machine_options(MachineClass *m) + pc_q35_compat_defaults, pc_q35_compat_defaults_len); + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void pc_q35_machine_10_1_options(MachineClass *m) + { + pc_q35_machine_options(m); +@@ -685,3 +686,4 @@ static void pc_q35_machine_2_6_options(MachineClass *m) + } + + DEFINE_Q35_MACHINE(2, 6); ++#endif /* Disabled for Red Hat Enterprise Linux */ +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index a79bd13275..2fca2bcf4d 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -911,6 +911,7 @@ static const TypeInfo ccw_machine_info = { + DEFINE_CCW_MACHINE_IMPL(false, major, minor) + + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void ccw_machine_10_1_instance_options(MachineState *machine) + { + } +@@ -1167,6 +1168,8 @@ static void ccw_machine_4_2_class_options(MachineClass *mc) + } + DEFINE_CCW_MACHINE(4, 2); + ++#endif /* disabled for RHEL */ ++ + static void ccw_machine_register_types(void) + { + type_register_static(&ccw_machine_info); +-- +2.39.3 + diff --git a/0009-Adapt-versioned-machine-type-macros-for-RHEL.patch b/0009-Adapt-versioned-machine-type-macros-for-RHEL.patch new file mode 100644 index 0000000..f0ccb30 --- /dev/null +++ b/0009-Adapt-versioned-machine-type-macros-for-RHEL.patch @@ -0,0 +1,195 @@ +From 0bc1a45f789d1aa6b75496739a77dbc77fc2492c Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Wed, 3 Jul 2024 15:27:03 +0100 +Subject: Adapt versioned machine type macros for RHEL +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +The versioned machine type macros are changed thus: + + * All symbol names get 'rhel' inserted eg 'virt_rhel_macine_9_4_0_' + * All machine type names get 'rhel' inserted eg 'virt-rhel9.4.0-machine' + * Lifecycle is changed to deprecate after 1 major RHEL release, + force non-registration (effectively deletion) after 2 major releases + * Custom message to explain RHEL deprecation/deletion policy + * Remove upstream logic that temporarily disabled deletion since + the upstream constraints in this area don't apply to RHEL + * For automatic deprecation/deletion, RHEL_VERSION is defined + +Signed-off-by: Daniel P. Berrangé +Signed-off-by: Miroslav Rezanina +--- +Rebase changes (10.1.0 rc0): +- Added deprecetion limits change to docs/conf.py +--- + .distro/Makefile | 2 +- + .distro/Makefile.common | 1 + + .distro/qemu-kvm.spec.template | 1 + + .distro/scripts/process-patches.sh | 3 +++ + docs/conf.py | 4 +-- + include/hw/boards.h | 39 ++++++++++++------------------ + meson.build | 1 + + meson_options.txt | 2 ++ + scripts/meson-buildoptions.sh | 2 ++ + 9 files changed, 29 insertions(+), 26 deletions(-) + +diff --git a/docs/conf.py b/docs/conf.py +index f892a6e1da..0b8861e4bf 100644 +--- a/docs/conf.py ++++ b/docs/conf.py +@@ -140,8 +140,8 @@ + # MACHINE_VER_DELETION_MAJOR & MACHINE_VER_DEPRECATION_MAJOR + # defined in include/hw/boards.h and the introductory text in + # docs/about/deprecated.rst +-ver_machine_deprecation_version = "%d.%d.0" % (major - 3, minor) +-ver_machine_deletion_version = "%d.%d.0" % (major - 6, minor) ++ver_machine_deprecation_version = "%d.%d.0" % (major - 1, minor) ++ver_machine_deletion_version = "%d.%d.0" % (major - 2, minor) + + # The language for content autogenerated by Sphinx. Refer to documentation + # for a list of supported languages. +diff --git a/include/hw/boards.h b/include/hw/boards.h +index a434b21909..da2fc92ce8 100644 +--- a/include/hw/boards.h ++++ b/include/hw/boards.h +@@ -577,16 +577,16 @@ struct MachineState { + * "{prefix}-{major}.{minor}.{micro}-{tag}" + */ + #define _MACHINE_VER_TYPE_NAME2(prefix, major, minor) \ +- prefix "-" #major "." #minor TYPE_MACHINE_SUFFIX ++ prefix "-rhel" #major "." #minor TYPE_MACHINE_SUFFIX + + #define _MACHINE_VER_TYPE_NAME3(prefix, major, minor, micro) \ +- prefix "-" #major "." #minor "." #micro TYPE_MACHINE_SUFFIX ++ prefix "-rhel" #major "." #minor "." #micro TYPE_MACHINE_SUFFIX + + #define _MACHINE_VER_TYPE_NAME4(prefix, major, minor, _unused_, tag) \ +- prefix "-" #major "." #minor "-" #tag TYPE_MACHINE_SUFFIX ++ prefix "-rhel" #major "." #minor "-" #tag TYPE_MACHINE_SUFFIX + + #define _MACHINE_VER_TYPE_NAME5(prefix, major, minor, micro, _unused_, tag) \ +- prefix "-" #major "." #minor "." #micro "-" #tag TYPE_MACHINE_SUFFIX ++ prefix "-rhel" #major "." #minor "." #micro "-" #tag TYPE_MACHINE_SUFFIX + + #define MACHINE_VER_TYPE_NAME(prefix, ...) \ + _MACHINE_VER_PICK(__VA_ARGS__, \ +@@ -614,16 +614,16 @@ struct MachineState { + * {prefix}_machine_{major}_{minor}_{micro}_{tag}_{sym} + */ + #define _MACHINE_VER_SYM2(sym, prefix, major, minor) \ +- prefix ## _machine_ ## major ## _ ## minor ## _ ## sym ++ prefix ## _rhel_machine_ ## major ## _ ## minor ## _ ## sym + + #define _MACHINE_VER_SYM3(sym, prefix, major, minor, micro) \ +- prefix ## _machine_ ## major ## _ ## minor ## _ ## micro ## _ ## sym ++ prefix ## _rhel_machine_ ## major ## _ ## minor ## _ ## micro ## _ ## sym + + #define _MACHINE_VER_SYM4(sym, prefix, major, minor, _unused_, tag) \ +- prefix ## _machine_ ## major ## _ ## minor ## _ ## tag ## _ ## sym ++ prefix ## _rhel_machine_ ## major ## _ ## minor ## _ ## tag ## _ ## sym + + #define _MACHINE_VER_SYM5(sym, prefix, major, minor, micro, _unused_, tag) \ +- prefix ## _machine_ ## major ## _ ## minor ## _ ## micro ## _ ## tag ## _ ## sym ++ prefix ## _rhel_machine_ ## major ## _ ## minor ## _ ## micro ## _ ## tag ## _ ## sym + + #define MACHINE_VER_SYM(sym, prefix, ...) \ + _MACHINE_VER_PICK(__VA_ARGS__, \ +@@ -642,17 +642,16 @@ struct MachineState { + * and ver_machine_deletion_version logic in docs/conf.py and + * the text in docs/about/deprecated.rst + */ +-#define MACHINE_VER_DELETION_MAJOR 6 +-#define MACHINE_VER_DEPRECATION_MAJOR 3 ++#define MACHINE_VER_DELETION_MAJOR 2 ++#define MACHINE_VER_DEPRECATION_MAJOR 1 + + /* + * Expands to a static string containing a deprecation + * message for a versioned machine type + */ + #define MACHINE_VER_DEPRECATION_MSG \ +- "machines more than " stringify(MACHINE_VER_DEPRECATION_MAJOR) \ +- " years old are subject to deletion after " \ +- stringify(MACHINE_VER_DELETION_MAJOR) " years" ++ "machines from the previous RHEL major release are " \ ++ "subject to deletion in the next RHEL major release" + + #define _MACHINE_VER_IS_CURRENT_EXPIRED(cutoff, major, minor) \ + (((QEMU_VERSION_MAJOR - major) > cutoff) || \ +@@ -683,12 +682,7 @@ struct MachineState { + * If this ever changes the logic below will need modifying.... + */ + #define _MACHINE_VER_IS_EXPIRED_IMPL(cutoff, major, minor) \ +- ((QEMU_VERSION_MICRO < 50 && \ +- _MACHINE_VER_IS_CURRENT_EXPIRED(cutoff, major, minor)) || \ +- (QEMU_VERSION_MICRO >= 50 && QEMU_VERSION_MINOR < 2 && \ +- _MACHINE_VER_IS_NEXT_MINOR_EXPIRED(cutoff, major, minor)) || \ +- (QEMU_VERSION_MICRO >= 50 && QEMU_VERSION_MINOR == 2 && \ +- _MACHINE_VER_IS_NEXT_MAJOR_EXPIRED(cutoff, major, minor))) ++ ((RHEL_VERSION - major) >= cutoff) + + #define _MACHINE_VER_IS_EXPIRED2(cutoff, major, minor) \ + _MACHINE_VER_IS_EXPIRED_IMPL(cutoff, major, minor) +@@ -750,10 +744,9 @@ struct MachineState { + * This must be unconditionally used in the register + * method for all machine types which support versioning. + * +- * Inijtially it will effectively be a no-op, but after a +- * suitable period of time has passed, it will cause +- * execution of the method to return, avoiding registration +- * of the machine ++ * It will automatically avoid registration of machines ++ * that should have been deleted at the start of this ++ * RHEL release + */ + #define MACHINE_VER_DELETION(...) \ + do { \ +diff --git a/meson.build b/meson.build +index 0e0120a613..23494666d9 100644 +--- a/meson.build ++++ b/meson.build +@@ -2636,6 +2636,7 @@ config_host_data.set('QEMU_VERSION', '"@0@"'.format(meson.project_version())) + config_host_data.set('QEMU_VERSION_MAJOR', meson.project_version().split('.')[0]) + config_host_data.set('QEMU_VERSION_MINOR', meson.project_version().split('.')[1]) + config_host_data.set('QEMU_VERSION_MICRO', meson.project_version().split('.')[2]) ++config_host_data.set('RHEL_VERSION', get_option('rhel_version').split('.')[0]) + + config_host_data.set_quoted('CONFIG_HOST_DSOSUF', host_dsosuf) + config_host_data.set('HAVE_HOST_BLOCK_DEVICE', have_host_block_device) +diff --git a/meson_options.txt b/meson_options.txt +index fff1521e58..f45d7ded45 100644 +--- a/meson_options.txt ++++ b/meson_options.txt +@@ -2,6 +2,8 @@ + # on the configure script command line. If you add more, list them in + # scripts/meson-buildoptions.py's SKIP_OPTIONS constant too. + ++option('rhel_version', type: 'string', value: '0.0', ++ description: 'RHEL major/minor version') + option('qemu_suffix', type : 'string', value: 'qemu', + description: 'Suffix for QEMU data/modules/config directories (can be empty)') + option('docdir', type : 'string', value : 'share/doc', +diff --git a/scripts/meson-buildoptions.sh b/scripts/meson-buildoptions.sh +index 0ebe6bc52a..4146dbc88d 100644 +--- a/scripts/meson-buildoptions.sh ++++ b/scripts/meson-buildoptions.sh +@@ -75,6 +75,7 @@ meson_options_help() { + printf "%s\n" ' [QEMU]' + printf "%s\n" ' --qemu-ga-version=VALUE version number for qemu-ga installer' + printf "%s\n" ' --rtsig-map=VALUE default value of QEMU_RTSIG_MAP [NULL]' ++ printf "%s\n" ' --rhel-version=VALUE RHEL major/minor version [0.0]' + printf "%s\n" ' --smbd=VALUE Path to smbd for slirp networking' + printf "%s\n" ' --sysconfdir=VALUE Sysconf data directory [etc]' + printf "%s\n" ' --tls-priority=VALUE Default TLS protocol/cipher priority string' +@@ -465,6 +466,7 @@ _meson_option_parse() { + --disable-relocatable) printf "%s" -Drelocatable=false ;; + --enable-replication) printf "%s" -Dreplication=enabled ;; + --disable-replication) printf "%s" -Dreplication=disabled ;; ++ --rhel-version=*) quote_sh "-Drhel_version=$2" ;; + --enable-rng-none) printf "%s" -Drng_none=true ;; + --disable-rng-none) printf "%s" -Drng_none=false ;; + --rtsig-map=*) quote_sh "-Drtsig_map=$2" ;; +-- +2.39.3 + diff --git a/0009-Add-x86_64-machine-types.patch b/0009-Add-x86_64-machine-types.patch deleted file mode 100644 index 4441c30..0000000 --- a/0009-Add-x86_64-machine-types.patch +++ /dev/null @@ -1,920 +0,0 @@ -From ec10588d2f5d748005e0dca42b299ae15868a900 Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Fri, 19 Oct 2018 13:10:31 +0200 -Subject: Add x86_64 machine types - -Adding changes to add RHEL machine types for x86_64 architecture. - -Signed-off-by: Miroslav Rezanina ---- - hw/i386/fw_cfg.c | 2 +- - hw/i386/pc.c | 159 ++++++++++++++++++++- - hw/i386/pc_piix.c | 109 ++++++++++++++ - hw/i386/pc_q35.c | 285 +++++++++++++++++++++++++++++++++++++ - include/hw/boards.h | 2 + - include/hw/i386/pc.h | 33 +++++ - target/i386/cpu.c | 21 +++ - target/i386/kvm/kvm-cpu.c | 1 + - target/i386/kvm/kvm.c | 4 + - tests/qtest/pvpanic-test.c | 5 +- - 10 files changed, 617 insertions(+), 4 deletions(-) - -diff --git a/hw/i386/fw_cfg.c b/hw/i386/fw_cfg.c -index c7aa39a13e..283c3f4c16 100644 ---- a/hw/i386/fw_cfg.c -+++ b/hw/i386/fw_cfg.c -@@ -63,7 +63,7 @@ void fw_cfg_build_smbios(PCMachineState *pcms, FWCfgState *fw_cfg, - - if (pcmc->smbios_defaults) { - /* These values are guest ABI, do not change */ -- smbios_set_defaults("QEMU", mc->desc, mc->name, -+ smbios_set_defaults("Red Hat", "KVM", mc->desc, - pcmc->smbios_uuid_encoded, - pcmc->smbios_stream_product, pcmc->smbios_stream_version); - } -diff --git a/hw/i386/pc.c b/hw/i386/pc.c -index 5c21b0c4db..4a154c1a9a 100644 ---- a/hw/i386/pc.c -+++ b/hw/i386/pc.c -@@ -326,6 +326,161 @@ GlobalProperty pc_compat_2_0[] = { - }; - const size_t pc_compat_2_0_len = G_N_ELEMENTS(pc_compat_2_0); - -+/* This macro is for changes to properties that are RHEL specific, -+ * different to the current upstream and to be applied to the latest -+ * machine type. -+ */ -+GlobalProperty pc_rhel_compat[] = { -+ /* we don't support s3/s4 suspend */ -+ { "PIIX4_PM", "disable_s3", "1" }, -+ { "PIIX4_PM", "disable_s4", "1" }, -+ { "ICH9-LPC", "disable_s3", "1" }, -+ { "ICH9-LPC", "disable_s4", "1" }, -+ -+ { TYPE_X86_CPU, "host-phys-bits", "on" }, -+ { TYPE_X86_CPU, "host-phys-bits-limit", "48" }, -+ { TYPE_X86_CPU, "vmx-entry-load-perf-global-ctrl", "off" }, -+ { TYPE_X86_CPU, "vmx-exit-load-perf-global-ctrl", "off" }, -+ /* bz 1508330 */ -+ { "vfio-pci", "x-no-geforce-quirks", "on" }, -+ /* bz 1941397 */ -+ { TYPE_X86_CPU, "kvm-asyncpf-int", "on" }, -+}; -+const size_t pc_rhel_compat_len = G_N_ELEMENTS(pc_rhel_compat); -+ -+GlobalProperty pc_rhel_9_3_compat[] = { -+ /* pc_rhel_9_3_compat from pc_compat_8_0 */ -+ { "virtio-mem", "unplugged-inaccessible", "auto" }, -+}; -+const size_t pc_rhel_9_3_compat_len = G_N_ELEMENTS(pc_rhel_9_3_compat); -+ -+GlobalProperty pc_rhel_9_2_compat[] = { -+ /* pc_rhel_9_2_compat from pc_compat_7_2 */ -+ { "ICH9-LPC", "noreboot", "true" }, -+}; -+const size_t pc_rhel_9_2_compat_len = G_N_ELEMENTS(pc_rhel_9_2_compat); -+ -+GlobalProperty pc_rhel_9_0_compat[] = { -+ /* pc_rhel_9_0_compat from pc_compat_6_2 */ -+ { "virtio-mem", "unplugged-inaccessible", "off" }, -+}; -+const size_t pc_rhel_9_0_compat_len = G_N_ELEMENTS(pc_rhel_9_0_compat); -+ -+GlobalProperty pc_rhel_8_5_compat[] = { -+ /* pc_rhel_8_5_compat from pc_compat_6_0 */ -+ { "qemu64" "-" TYPE_X86_CPU, "family", "6" }, -+ /* pc_rhel_8_5_compat from pc_compat_6_0 */ -+ { "qemu64" "-" TYPE_X86_CPU, "model", "6" }, -+ /* pc_rhel_8_5_compat from pc_compat_6_0 */ -+ { "qemu64" "-" TYPE_X86_CPU, "stepping", "3" }, -+ /* pc_rhel_8_5_compat from pc_compat_6_0 */ -+ { TYPE_X86_CPU, "x-vendor-cpuid-only", "off" }, -+ /* pc_rhel_8_5_compat from pc_compat_6_0 */ -+ { "ICH9-LPC", ACPI_PM_PROP_ACPI_PCIHP_BRIDGE, "off" }, -+ -+ /* pc_rhel_8_5_compat from pc_compat_6_1 */ -+ { TYPE_X86_CPU, "hv-version-id-build", "0x1bbc" }, -+ /* pc_rhel_8_5_compat from pc_compat_6_1 */ -+ { TYPE_X86_CPU, "hv-version-id-major", "0x0006" }, -+ /* pc_rhel_8_5_compat from pc_compat_6_1 */ -+ { TYPE_X86_CPU, "hv-version-id-minor", "0x0001" }, -+}; -+const size_t pc_rhel_8_5_compat_len = G_N_ELEMENTS(pc_rhel_8_5_compat); -+ -+GlobalProperty pc_rhel_8_4_compat[] = { -+ /* pc_rhel_8_4_compat from pc_compat_5_2 */ -+ { "ICH9-LPC", "x-smi-cpu-hotunplug", "off" }, -+ { TYPE_X86_CPU, "kvm-asyncpf-int", "off" }, -+}; -+const size_t pc_rhel_8_4_compat_len = G_N_ELEMENTS(pc_rhel_8_4_compat); -+ -+GlobalProperty pc_rhel_8_3_compat[] = { -+ /* pc_rhel_8_3_compat from pc_compat_5_1 */ -+ { "ICH9-LPC", "x-smi-cpu-hotplug", "off" }, -+}; -+const size_t pc_rhel_8_3_compat_len = G_N_ELEMENTS(pc_rhel_8_3_compat); -+ -+GlobalProperty pc_rhel_8_2_compat[] = { -+ /* pc_rhel_8_2_compat from pc_compat_4_2 */ -+ { "mch", "smbase-smram", "off" }, -+}; -+const size_t pc_rhel_8_2_compat_len = G_N_ELEMENTS(pc_rhel_8_2_compat); -+ -+/* pc_rhel_8_1_compat is empty since pc_4_1_compat is */ -+GlobalProperty pc_rhel_8_1_compat[] = { }; -+const size_t pc_rhel_8_1_compat_len = G_N_ELEMENTS(pc_rhel_8_1_compat); -+ -+GlobalProperty pc_rhel_8_0_compat[] = { -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "intel-iommu", "dma-drain", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Opteron_G3" "-" TYPE_X86_CPU, "rdtscp", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Opteron_G4" "-" TYPE_X86_CPU, "rdtscp", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Opteron_G4" "-" TYPE_X86_CPU, "npt", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Opteron_G4" "-" TYPE_X86_CPU, "nrip-save", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Opteron_G5" "-" TYPE_X86_CPU, "rdtscp", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Opteron_G5" "-" TYPE_X86_CPU, "npt", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Opteron_G5" "-" TYPE_X86_CPU, "nrip-save", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "EPYC" "-" TYPE_X86_CPU, "npt", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "EPYC" "-" TYPE_X86_CPU, "nrip-save", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "EPYC-IBPB" "-" TYPE_X86_CPU, "npt", "off" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "EPYC-IBPB" "-" TYPE_X86_CPU, "nrip-save", "off" }, -+ /** The mpx=on entries from pc_compat_3_1 are in pc_rhel_7_6_compat **/ -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { "Cascadelake-Server" "-" TYPE_X86_CPU, "stepping", "5" }, -+ /* pc_rhel_8_0_compat from pc_compat_3_1 */ -+ { TYPE_X86_CPU, "x-intel-pt-auto-level", "off" }, -+}; -+const size_t pc_rhel_8_0_compat_len = G_N_ELEMENTS(pc_rhel_8_0_compat); -+ -+/* Similar to PC_COMPAT_3_0 + PC_COMPAT_2_12, but: -+ * all of the 2_12 stuff was already in 7.6 from bz 1481253 -+ * x-migrate-smi-count comes from PC_COMPAT_2_11 but -+ * is really tied to kernel version so keep it off on 7.x -+ * machine types irrespective of host. -+ */ -+GlobalProperty pc_rhel_7_6_compat[] = { -+ /* pc_rhel_7_6_compat from pc_compat_3_0 */ -+ { TYPE_X86_CPU, "x-hv-synic-kvm-only", "on" }, -+ /* pc_rhel_7_6_compat from pc_compat_3_0 */ -+ { "Skylake-Server" "-" TYPE_X86_CPU, "pku", "off" }, -+ /* pc_rhel_7_6_compat from pc_compat_3_0 */ -+ { "Skylake-Server-IBRS" "-" TYPE_X86_CPU, "pku", "off" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { TYPE_X86_CPU, "x-migrate-smi-count", "off" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { "Skylake-Client" "-" TYPE_X86_CPU, "mpx", "on" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { "Skylake-Client-IBRS" "-" TYPE_X86_CPU, "mpx", "on" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { "Skylake-Server" "-" TYPE_X86_CPU, "mpx", "on" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { "Skylake-Server-IBRS" "-" TYPE_X86_CPU, "mpx", "on" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { "Cascadelake-Server" "-" TYPE_X86_CPU, "mpx", "on" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { "Icelake-Client" "-" TYPE_X86_CPU, "mpx", "on" }, -+ /* pc_rhel_7_6_compat from pc_compat_2_11 */ -+ { "Icelake-Server" "-" TYPE_X86_CPU, "mpx", "on" }, -+}; -+const size_t pc_rhel_7_6_compat_len = G_N_ELEMENTS(pc_rhel_7_6_compat); -+ -+/* -+ * The PC_RHEL_*_COMPAT serve the same purpose for RHEL-7 machine -+ * types as the PC_COMPAT_* do for upstream types. -+ * PC_RHEL_7_*_COMPAT apply both to i440fx and q35 types. -+ */ -+ - GSIState *pc_gsi_create(qemu_irq **irqs, bool pci_enabled) - { - GSIState *s; -@@ -1813,6 +1968,7 @@ static void pc_machine_class_init(ObjectClass *oc, void *data) - pcmc->resizable_acpi_blob = true; - x86mc->apic_xrupt_override = true; - assert(!mc->get_hotplug_handler); -+ mc->async_pf_vmexit_disable = false; - mc->get_hotplug_handler = pc_get_hotplug_handler; - mc->hotplug_allowed = pc_hotplug_allowed; - mc->cpu_index_to_instance_props = x86_cpu_index_to_props; -@@ -1823,7 +1979,8 @@ static void pc_machine_class_init(ObjectClass *oc, void *data) - mc->has_hotpluggable_cpus = true; - mc->default_boot_order = "cad"; - mc->block_default_type = IF_IDE; -- mc->max_cpus = 255; -+ /* 240: max CPU count for RHEL */ -+ mc->max_cpus = 240; - mc->reset = pc_machine_reset; - mc->wakeup = pc_machine_wakeup; - hc->pre_plug = pc_machine_device_pre_plug_cb; -diff --git a/hw/i386/pc_piix.c b/hw/i386/pc_piix.c -index 18ba076609..a647262d63 100644 ---- a/hw/i386/pc_piix.c -+++ b/hw/i386/pc_piix.c -@@ -52,6 +52,7 @@ - #include "qapi/error.h" - #include "qemu/error-report.h" - #include "sysemu/xen.h" -+#include "migration/migration.h" - #ifdef CONFIG_XEN - #include - #include "hw/xen/xen_pt.h" -@@ -422,6 +423,7 @@ static void pc_set_south_bridge(Object *obj, int value, Error **errp) - * hw_compat_*, pc_compat_*, or * pc_*_machine_options(). - */ - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static void pc_compat_2_3_fn(MachineState *machine) - { - X86MachineState *x86ms = X86_MACHINE(machine); -@@ -951,3 +953,110 @@ static void xenfv_3_1_machine_options(MachineClass *m) - DEFINE_PC_MACHINE(xenfv, "xenfv-3.1", pc_xen_hvm_init, - xenfv_3_1_machine_options); - #endif -+#endif /* Disabled for Red Hat Enterprise Linux */ -+ -+/* Red Hat Enterprise Linux machine types */ -+ -+/* Options for the latest rhel7 machine type */ -+static void pc_machine_rhel7_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ m->family = "pc_piix_Y"; -+ m->default_machine_opts = "firmware=bios-256k.bin,hpet=off"; -+ pcmc->pci_root_uid = 0; -+ pcmc->resizable_acpi_blob = true; -+ m->default_nic = "e1000"; -+ m->default_display = "std"; -+ m->no_parallel = 1; -+ m->numa_mem_supported = true; -+ m->auto_enable_numa_with_memdev = false; -+ machine_class_allow_dynamic_sysbus_dev(m, TYPE_RAMFB_DEVICE); -+ compat_props_add(m->compat_props, pc_rhel_compat, pc_rhel_compat_len); -+ m->alias = "pc"; -+ m->is_default = 1; -+ m->smp_props.prefer_sockets = true; -+} -+ -+static void pc_init_rhel760(MachineState *machine) -+{ -+ pc_init1(machine, TYPE_I440FX_PCI_DEVICE); -+} -+ -+static void pc_machine_rhel760_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ ObjectClass *oc = OBJECT_CLASS(m); -+ pc_machine_rhel7_options(m); -+ m->desc = "RHEL 7.6.0 PC (i440FX + PIIX, 1996)"; -+ m->async_pf_vmexit_disable = true; -+ m->smbus_no_migration_support = true; -+ -+ /* All RHEL machines for prior major releases are deprecated */ -+ m->deprecation_reason = rhel_old_machine_deprecation; -+ -+ pcmc->pvh_enabled = false; -+ pcmc->default_cpu_version = CPU_VERSION_LEGACY; -+ pcmc->kvmclock_create_always = false; -+ /* From pc_i440fx_5_1_machine_options() */ -+ pcmc->pci_root_uid = 1; -+ /* From pc_i440fx_7_0_machine_options() */ -+ pcmc->enforce_amd_1tb_hole = false; -+ /* From pc_i440fx_8_0_machine_options() */ -+ pcmc->default_smbios_ep_type = SMBIOS_ENTRY_POINT_TYPE_32; -+ /* From pc_i440fx_8_1_machine_options() */ -+ pcmc->broken_32bit_mem_addr_check = true; -+ /* Introduced in QEMU 8.2 */ -+ pcmc->default_south_bridge = TYPE_PIIX3_DEVICE; -+ -+ object_class_property_add_enum(oc, "x-south-bridge", "PCSouthBridgeOption", -+ &PCSouthBridgeOption_lookup, -+ pc_get_south_bridge, -+ pc_set_south_bridge); -+ object_class_property_set_description(oc, "x-south-bridge", -+ "Use a different south bridge than PIIX3"); -+ -+ -+ compat_props_add(m->compat_props, hw_compat_rhel_9_4, -+ hw_compat_rhel_9_4_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_9_3, -+ hw_compat_rhel_9_3_len); -+ compat_props_add(m->compat_props, pc_rhel_9_3_compat, -+ pc_rhel_9_3_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_9_2, -+ hw_compat_rhel_9_2_len); -+ compat_props_add(m->compat_props, pc_rhel_9_2_compat, -+ pc_rhel_9_2_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_9_1, -+ hw_compat_rhel_9_1_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_9_0, -+ hw_compat_rhel_9_0_len); -+ compat_props_add(m->compat_props, pc_rhel_9_0_compat, -+ pc_rhel_9_0_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_8_6, -+ hw_compat_rhel_8_6_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_8_5, -+ hw_compat_rhel_8_5_len); -+ compat_props_add(m->compat_props, pc_rhel_8_5_compat, -+ pc_rhel_8_5_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_8_4, -+ hw_compat_rhel_8_4_len); -+ compat_props_add(m->compat_props, pc_rhel_8_4_compat, -+ pc_rhel_8_4_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_8_3, -+ hw_compat_rhel_8_3_len); -+ compat_props_add(m->compat_props, pc_rhel_8_3_compat, -+ pc_rhel_8_3_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_8_2, -+ hw_compat_rhel_8_2_len); -+ compat_props_add(m->compat_props, pc_rhel_8_2_compat, -+ pc_rhel_8_2_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_8_1, hw_compat_rhel_8_1_len); -+ compat_props_add(m->compat_props, pc_rhel_8_1_compat, pc_rhel_8_1_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_8_0, hw_compat_rhel_8_0_len); -+ compat_props_add(m->compat_props, pc_rhel_8_0_compat, pc_rhel_8_0_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_7_6, hw_compat_rhel_7_6_len); -+ compat_props_add(m->compat_props, pc_rhel_7_6_compat, pc_rhel_7_6_compat_len); -+} -+ -+DEFINE_PC_MACHINE(rhel760, "pc-i440fx-rhel7.6.0", pc_init_rhel760, -+ pc_machine_rhel760_options); -diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c -index c7bc8a2041..e872dc7e46 100644 ---- a/hw/i386/pc_q35.c -+++ b/hw/i386/pc_q35.c -@@ -341,6 +341,7 @@ static void pc_q35_init(MachineState *machine) - DEFINE_PC_MACHINE(suffix, name, pc_init_##suffix, optionfn) - - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static void pc_q35_machine_options(MachineClass *m) - { - PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -@@ -693,3 +694,287 @@ static void pc_q35_2_4_machine_options(MachineClass *m) - - DEFINE_Q35_MACHINE(v2_4, "pc-q35-2.4", NULL, - pc_q35_2_4_machine_options); -+#endif /* Disabled for Red Hat Enterprise Linux */ -+ -+/* Red Hat Enterprise Linux machine types */ -+ -+/* Options for the latest rhel q35 machine type */ -+static void pc_q35_machine_rhel_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pcmc->pci_root_uid = 0; -+ m->default_nic = "e1000e"; -+ m->family = "pc_q35_Z"; -+ m->units_per_default_bus = 1; -+ m->default_machine_opts = "firmware=bios-256k.bin,hpet=off"; -+ m->default_display = "std"; -+ m->no_floppy = 1; -+ m->no_parallel = 1; -+ pcmc->default_cpu_version = 1; -+ machine_class_allow_dynamic_sysbus_dev(m, TYPE_AMD_IOMMU_DEVICE); -+ machine_class_allow_dynamic_sysbus_dev(m, TYPE_INTEL_IOMMU_DEVICE); -+ machine_class_allow_dynamic_sysbus_dev(m, TYPE_RAMFB_DEVICE); -+ m->alias = "q35"; -+ m->max_cpus = 710; -+ compat_props_add(m->compat_props, pc_rhel_compat, pc_rhel_compat_len); -+ compat_props_add(m->compat_props, -+ pc_q35_compat_defaults, pc_q35_compat_defaults_len); -+} -+ -+static void pc_q35_init_rhel940(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel940_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel_options(m); -+ m->desc = "RHEL-9.4.0 PC (Q35 + ICH9, 2009)"; -+ pcmc->smbios_stream_product = "RHEL"; -+ pcmc->smbios_stream_version = "9.4.0"; -+} -+ -+DEFINE_PC_MACHINE(q35_rhel940, "pc-q35-rhel9.4.0", pc_q35_init_rhel940, -+ pc_q35_machine_rhel940_options); -+ -+ -+static void pc_q35_init_rhel920(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel920_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel940_options(m); -+ m->desc = "RHEL-9.2.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ pcmc->smbios_stream_product = "RHEL"; -+ pcmc->smbios_stream_version = "9.2.0"; -+ -+ /* From pc_q35_8_0_machine_options() */ -+ pcmc->default_smbios_ep_type = SMBIOS_ENTRY_POINT_TYPE_32; -+ /* From pc_q35_8_1_machine_options() */ -+ pcmc->broken_32bit_mem_addr_check = true; -+ -+ compat_props_add(m->compat_props, hw_compat_rhel_9_4, -+ hw_compat_rhel_9_4_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_9_3, -+ hw_compat_rhel_9_3_len); -+ compat_props_add(m->compat_props, pc_rhel_9_3_compat, -+ pc_rhel_9_3_compat_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_9_2, -+ hw_compat_rhel_9_2_len); -+ compat_props_add(m->compat_props, pc_rhel_9_2_compat, -+ pc_rhel_9_2_compat_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel920, "pc-q35-rhel9.2.0", pc_q35_init_rhel920, -+ pc_q35_machine_rhel920_options); -+ -+static void pc_q35_init_rhel900(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel900_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel920_options(m); -+ m->desc = "RHEL-9.0.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ pcmc->smbios_stream_product = "RHEL"; -+ pcmc->smbios_stream_version = "9.0.0"; -+ pcmc->enforce_amd_1tb_hole = false; -+ compat_props_add(m->compat_props, hw_compat_rhel_9_1, -+ hw_compat_rhel_9_1_len); -+ compat_props_add(m->compat_props, hw_compat_rhel_9_0, -+ hw_compat_rhel_9_0_len); -+ compat_props_add(m->compat_props, pc_rhel_9_0_compat, -+ pc_rhel_9_0_compat_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel900, "pc-q35-rhel9.0.0", pc_q35_init_rhel900, -+ pc_q35_machine_rhel900_options); -+ -+static void pc_q35_init_rhel860(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel860_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel900_options(m); -+ m->desc = "RHEL-8.6.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ -+ /* All RHEL machines for prior major releases are deprecated */ -+ m->deprecation_reason = rhel_old_machine_deprecation; -+ -+ pcmc->smbios_stream_product = "RHEL-AV"; -+ pcmc->smbios_stream_version = "8.6.0"; -+ compat_props_add(m->compat_props, hw_compat_rhel_8_6, -+ hw_compat_rhel_8_6_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel860, "pc-q35-rhel8.6.0", pc_q35_init_rhel860, -+ pc_q35_machine_rhel860_options); -+ -+ -+static void pc_q35_init_rhel850(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel850_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel860_options(m); -+ m->desc = "RHEL-8.5.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ pcmc->smbios_stream_product = "RHEL-AV"; -+ pcmc->smbios_stream_version = "8.5.0"; -+ compat_props_add(m->compat_props, hw_compat_rhel_8_5, -+ hw_compat_rhel_8_5_len); -+ compat_props_add(m->compat_props, pc_rhel_8_5_compat, -+ pc_rhel_8_5_compat_len); -+ m->smp_props.prefer_sockets = true; -+} -+ -+DEFINE_PC_MACHINE(q35_rhel850, "pc-q35-rhel8.5.0", pc_q35_init_rhel850, -+ pc_q35_machine_rhel850_options); -+ -+ -+static void pc_q35_init_rhel840(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel840_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel850_options(m); -+ m->desc = "RHEL-8.4.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ pcmc->smbios_stream_product = "RHEL-AV"; -+ pcmc->smbios_stream_version = "8.4.0"; -+ compat_props_add(m->compat_props, hw_compat_rhel_8_4, -+ hw_compat_rhel_8_4_len); -+ compat_props_add(m->compat_props, pc_rhel_8_4_compat, -+ pc_rhel_8_4_compat_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel840, "pc-q35-rhel8.4.0", pc_q35_init_rhel840, -+ pc_q35_machine_rhel840_options); -+ -+ -+static void pc_q35_init_rhel830(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel830_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel840_options(m); -+ m->desc = "RHEL-8.3.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ pcmc->smbios_stream_product = "RHEL-AV"; -+ pcmc->smbios_stream_version = "8.3.0"; -+ compat_props_add(m->compat_props, hw_compat_rhel_8_3, -+ hw_compat_rhel_8_3_len); -+ compat_props_add(m->compat_props, pc_rhel_8_3_compat, -+ pc_rhel_8_3_compat_len); -+ /* From pc_q35_5_1_machine_options() */ -+ pcmc->kvmclock_create_always = false; -+ /* From pc_q35_5_1_machine_options() */ -+ pcmc->pci_root_uid = 1; -+} -+ -+DEFINE_PC_MACHINE(q35_rhel830, "pc-q35-rhel8.3.0", pc_q35_init_rhel830, -+ pc_q35_machine_rhel830_options); -+ -+static void pc_q35_init_rhel820(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel820_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel830_options(m); -+ m->desc = "RHEL-8.2.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ m->numa_mem_supported = true; -+ m->auto_enable_numa_with_memdev = false; -+ pcmc->smbios_stream_product = "RHEL-AV"; -+ pcmc->smbios_stream_version = "8.2.0"; -+ compat_props_add(m->compat_props, hw_compat_rhel_8_2, -+ hw_compat_rhel_8_2_len); -+ compat_props_add(m->compat_props, pc_rhel_8_2_compat, -+ pc_rhel_8_2_compat_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel820, "pc-q35-rhel8.2.0", pc_q35_init_rhel820, -+ pc_q35_machine_rhel820_options); -+ -+static void pc_q35_init_rhel810(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel810_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel820_options(m); -+ m->desc = "RHEL-8.1.0 PC (Q35 + ICH9, 2009)"; -+ m->alias = NULL; -+ pcmc->smbios_stream_product = NULL; -+ pcmc->smbios_stream_version = NULL; -+ compat_props_add(m->compat_props, hw_compat_rhel_8_1, hw_compat_rhel_8_1_len); -+ compat_props_add(m->compat_props, pc_rhel_8_1_compat, pc_rhel_8_1_compat_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel810, "pc-q35-rhel8.1.0", pc_q35_init_rhel810, -+ pc_q35_machine_rhel810_options); -+ -+static void pc_q35_init_rhel800(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel800_options(MachineClass *m) -+{ -+ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); -+ pc_q35_machine_rhel810_options(m); -+ m->desc = "RHEL-8.0.0 PC (Q35 + ICH9, 2009)"; -+ m->smbus_no_migration_support = true; -+ m->alias = NULL; -+ pcmc->pvh_enabled = false; -+ pcmc->default_cpu_version = CPU_VERSION_LEGACY; -+ compat_props_add(m->compat_props, hw_compat_rhel_8_0, hw_compat_rhel_8_0_len); -+ compat_props_add(m->compat_props, pc_rhel_8_0_compat, pc_rhel_8_0_compat_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel800, "pc-q35-rhel8.0.0", pc_q35_init_rhel800, -+ pc_q35_machine_rhel800_options); -+ -+static void pc_q35_init_rhel760(MachineState *machine) -+{ -+ pc_q35_init(machine); -+} -+ -+static void pc_q35_machine_rhel760_options(MachineClass *m) -+{ -+ pc_q35_machine_rhel800_options(m); -+ m->alias = NULL; -+ m->desc = "RHEL-7.6.0 PC (Q35 + ICH9, 2009)"; -+ m->async_pf_vmexit_disable = true; -+ compat_props_add(m->compat_props, hw_compat_rhel_7_6, hw_compat_rhel_7_6_len); -+ compat_props_add(m->compat_props, pc_rhel_7_6_compat, pc_rhel_7_6_compat_len); -+} -+ -+DEFINE_PC_MACHINE(q35_rhel760, "pc-q35-rhel7.6.0", pc_q35_init_rhel760, -+ pc_q35_machine_rhel760_options); -diff --git a/include/hw/boards.h b/include/hw/boards.h -index 0466f9d0f3..46b8725c41 100644 ---- a/include/hw/boards.h -+++ b/include/hw/boards.h -@@ -283,6 +283,8 @@ struct MachineClass { - strList *allowed_dynamic_sysbus_devices; - bool auto_enable_numa_with_memhp; - bool auto_enable_numa_with_memdev; -+ /* RHEL only */ -+ bool async_pf_vmexit_disable; - bool ignore_boot_device_suffixes; - bool smbus_no_migration_support; - bool nvdimm_supported; -diff --git a/include/hw/i386/pc.h b/include/hw/i386/pc.h -index ebd8f973f2..a984c951ad 100644 ---- a/include/hw/i386/pc.h -+++ b/include/hw/i386/pc.h -@@ -291,6 +291,39 @@ extern const size_t pc_compat_2_1_len; - extern GlobalProperty pc_compat_2_0[]; - extern const size_t pc_compat_2_0_len; - -+extern GlobalProperty pc_rhel_compat[]; -+extern const size_t pc_rhel_compat_len; -+ -+extern GlobalProperty pc_rhel_9_3_compat[]; -+extern const size_t pc_rhel_9_3_compat_len; -+ -+extern GlobalProperty pc_rhel_9_2_compat[]; -+extern const size_t pc_rhel_9_2_compat_len; -+ -+extern GlobalProperty pc_rhel_9_0_compat[]; -+extern const size_t pc_rhel_9_0_compat_len; -+ -+extern GlobalProperty pc_rhel_8_5_compat[]; -+extern const size_t pc_rhel_8_5_compat_len; -+ -+extern GlobalProperty pc_rhel_8_4_compat[]; -+extern const size_t pc_rhel_8_4_compat_len; -+ -+extern GlobalProperty pc_rhel_8_3_compat[]; -+extern const size_t pc_rhel_8_3_compat_len; -+ -+extern GlobalProperty pc_rhel_8_2_compat[]; -+extern const size_t pc_rhel_8_2_compat_len; -+ -+extern GlobalProperty pc_rhel_8_1_compat[]; -+extern const size_t pc_rhel_8_1_compat_len; -+ -+extern GlobalProperty pc_rhel_8_0_compat[]; -+extern const size_t pc_rhel_8_0_compat_len; -+ -+extern GlobalProperty pc_rhel_7_6_compat[]; -+extern const size_t pc_rhel_7_6_compat_len; -+ - #define DEFINE_PC_MACHINE(suffix, namestr, initfn, optsfn) \ - static void pc_machine_##suffix##_class_init(ObjectClass *oc, void *data) \ - { \ -diff --git a/target/i386/cpu.c b/target/i386/cpu.c -index 33760a2ee1..be7b0663cd 100644 ---- a/target/i386/cpu.c -+++ b/target/i386/cpu.c -@@ -2190,9 +2190,13 @@ static const CPUCaches epyc_genoa_cache_info = { - * PT in VMX operation - */ - -+#define RHEL_CPU_DEPRECATION \ -+ "use at least 'Nehalem' / 'Opteron_G4', or 'host' / 'max'" -+ - static const X86CPUDefinition builtin_x86_defs[] = { - { - .name = "qemu64", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 0xd, - .vendor = CPUID_VENDOR_AMD, - .family = 15, -@@ -2213,6 +2217,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "phenom", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 5, - .vendor = CPUID_VENDOR_AMD, - .family = 16, -@@ -2245,6 +2250,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "core2duo", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 10, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2287,6 +2293,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "kvm64", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 0xd, - .vendor = CPUID_VENDOR_INTEL, - .family = 15, -@@ -2328,6 +2335,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "qemu32", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 4, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2342,6 +2350,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "kvm32", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 5, - .vendor = CPUID_VENDOR_INTEL, - .family = 15, -@@ -2372,6 +2381,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "coreduo", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 10, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2405,6 +2415,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "486", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 1, - .vendor = CPUID_VENDOR_INTEL, - .family = 4, -@@ -2417,6 +2428,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "pentium", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 1, - .vendor = CPUID_VENDOR_INTEL, - .family = 5, -@@ -2429,6 +2441,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "pentium2", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 2, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2441,6 +2454,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "pentium3", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 3, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2453,6 +2467,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "athlon", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 2, - .vendor = CPUID_VENDOR_AMD, - .family = 6, -@@ -2468,6 +2483,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "n270", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 10, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2493,6 +2509,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Conroe", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 10, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2533,6 +2550,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Penryn", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 10, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -4394,6 +4412,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Opteron_G1", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 5, - .vendor = CPUID_VENDOR_AMD, - .family = 15, -@@ -4414,6 +4433,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Opteron_G2", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 5, - .vendor = CPUID_VENDOR_AMD, - .family = 15, -@@ -4436,6 +4456,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Opteron_G3", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 5, - .vendor = CPUID_VENDOR_AMD, - .family = 16, -diff --git a/target/i386/kvm/kvm-cpu.c b/target/i386/kvm/kvm-cpu.c -index 9c791b7b05..b91af5051f 100644 ---- a/target/i386/kvm/kvm-cpu.c -+++ b/target/i386/kvm/kvm-cpu.c -@@ -138,6 +138,7 @@ static PropValue kvm_default_props[] = { - { "acpi", "off" }, - { "monitor", "off" }, - { "svm", "off" }, -+ { "kvm-pv-unhalt", "on" }, - { NULL, NULL }, - }; - -diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c -index e68cbe9293..739f33db47 100644 ---- a/target/i386/kvm/kvm.c -+++ b/target/i386/kvm/kvm.c -@@ -3715,6 +3715,7 @@ static int kvm_get_msrs(X86CPU *cpu) - struct kvm_msr_entry *msrs = cpu->kvm_msr_buf->entries; - int ret, i; - uint64_t mtrr_top_bits; -+ MachineClass *mc = MACHINE_GET_CLASS(qdev_get_machine()); - - kvm_msr_buf_reset(cpu); - -@@ -4069,6 +4070,9 @@ static int kvm_get_msrs(X86CPU *cpu) - break; - case MSR_KVM_ASYNC_PF_EN: - env->async_pf_en_msr = msrs[i].data; -+ if (mc->async_pf_vmexit_disable) { -+ env->async_pf_en_msr &= ~(1ULL << 2); -+ } - break; - case MSR_KVM_ASYNC_PF_INT: - env->async_pf_int_msr = msrs[i].data; -diff --git a/tests/qtest/pvpanic-test.c b/tests/qtest/pvpanic-test.c -index 78f1cf8186..ac954c9b06 100644 ---- a/tests/qtest/pvpanic-test.c -+++ b/tests/qtest/pvpanic-test.c -@@ -17,7 +17,7 @@ static void test_panic_nopause(void) - QDict *response, *data; - QTestState *qts; - -- qts = qtest_init("-device pvpanic -action panic=none"); -+ qts = qtest_init("-M q35 -device pvpanic -action panic=none"); - - val = qtest_inb(qts, 0x505); - g_assert_cmpuint(val, ==, 3); -@@ -40,7 +40,8 @@ static void test_panic(void) - QDict *response, *data; - QTestState *qts; - -- qts = qtest_init("-device pvpanic -action panic=pause"); -+ /* RHEL: Use q35 */ -+ qts = qtest_init("-M q35 -device pvpanic -action panic=pause"); - - val = qtest_inb(qts, 0x505); - g_assert_cmpuint(val, ==, 3); --- -2.39.3 - diff --git a/0010-Enable-make-check.patch b/0010-Enable-make-check.patch deleted file mode 100644 index 8d99bf9..0000000 --- a/0010-Enable-make-check.patch +++ /dev/null @@ -1,231 +0,0 @@ -From 241ad69d849fce983685fc754fc0572c5b737cbe Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Wed, 2 Sep 2020 09:39:41 +0200 -Subject: Enable make check - -Fixing tests after device disabling and machine types changes and enabling -make check run during build. - -Signed-off-by: Miroslav Rezanina ---- - .distro/qemu-kvm.spec.template | 4 ++-- - tests/avocado/replay_kernel.py | 2 +- - tests/avocado/reverse_debugging.py | 2 +- - tests/avocado/tcg_plugins.py | 6 ++--- - tests/qemu-iotests/meson.build | 34 ++++++++++++++--------------- - tests/qemu-iotests/testenv.py | 3 +++ - tests/qtest/fuzz-e1000e-test.c | 2 +- - tests/qtest/fuzz-virtio-scsi-test.c | 2 +- - tests/qtest/intel-hda-test.c | 2 +- - tests/qtest/libqos/meson.build | 2 +- - tests/qtest/lpc-ich9-test.c | 2 +- - tests/qtest/meson.build | 1 - - tests/qtest/virtio-net-failover.c | 1 + - 13 files changed, 33 insertions(+), 30 deletions(-) - -diff --git a/tests/avocado/replay_kernel.py b/tests/avocado/replay_kernel.py -index 10d99403a4..c3422ea1e4 100644 ---- a/tests/avocado/replay_kernel.py -+++ b/tests/avocado/replay_kernel.py -@@ -166,7 +166,7 @@ def test_aarch64_virt(self): - """ - :avocado: tags=arch:aarch64 - :avocado: tags=machine:virt -- :avocado: tags=cpu:cortex-a53 -+ :avocado: tags=cpu:cortex-a57 - """ - kernel_url = ('https://archives.fedoraproject.org/pub/archive/fedora' - '/linux/releases/29/Everything/aarch64/os/images/pxeboot' -diff --git a/tests/avocado/reverse_debugging.py b/tests/avocado/reverse_debugging.py -index 92855a02a5..87822074b6 100644 ---- a/tests/avocado/reverse_debugging.py -+++ b/tests/avocado/reverse_debugging.py -@@ -230,7 +230,7 @@ def test_aarch64_virt(self): - """ - :avocado: tags=arch:aarch64 - :avocado: tags=machine:virt -- :avocado: tags=cpu:cortex-a53 -+ :avocado: tags=cpu:cortex-a57 - """ - kernel_url = ('https://archives.fedoraproject.org/pub/archive/fedora' - '/linux/releases/29/Everything/aarch64/os/images/pxeboot' -diff --git a/tests/avocado/tcg_plugins.py b/tests/avocado/tcg_plugins.py -index 15fd87b2c1..f0d9d89c93 100644 ---- a/tests/avocado/tcg_plugins.py -+++ b/tests/avocado/tcg_plugins.py -@@ -66,7 +66,7 @@ def test_aarch64_virt_insn(self): - :avocado: tags=accel:tcg - :avocado: tags=arch:aarch64 - :avocado: tags=machine:virt -- :avocado: tags=cpu:cortex-a53 -+ :avocado: tags=cpu:cortex-a57 - """ - kernel_path = self._grab_aarch64_kernel() - kernel_command_line = (self.KERNEL_COMMON_COMMAND_LINE + -@@ -96,7 +96,7 @@ def test_aarch64_virt_insn_icount(self): - :avocado: tags=accel:tcg - :avocado: tags=arch:aarch64 - :avocado: tags=machine:virt -- :avocado: tags=cpu:cortex-a53 -+ :avocado: tags=cpu:cortex-a57 - """ - kernel_path = self._grab_aarch64_kernel() - kernel_command_line = (self.KERNEL_COMMON_COMMAND_LINE + -@@ -126,7 +126,7 @@ def test_aarch64_virt_mem_icount(self): - :avocado: tags=accel:tcg - :avocado: tags=arch:aarch64 - :avocado: tags=machine:virt -- :avocado: tags=cpu:cortex-a53 -+ :avocado: tags=cpu:cortex-a57 - """ - kernel_path = self._grab_aarch64_kernel() - kernel_command_line = (self.KERNEL_COMMON_COMMAND_LINE + -diff --git a/tests/qemu-iotests/meson.build b/tests/qemu-iotests/meson.build -index fad340ad59..3c0d5241f6 100644 ---- a/tests/qemu-iotests/meson.build -+++ b/tests/qemu-iotests/meson.build -@@ -51,21 +51,21 @@ foreach format, speed: qemu_iotests_formats - check: true, - ) - -- foreach item: rc.stdout().strip().split() -- args = [qemu_iotests_check_cmd, -- '-tap', '-' + format, item, -- '--source-dir', meson.current_source_dir(), -- '--build-dir', meson.current_build_dir()] -- # Some individual tests take as long as 45 seconds -- # Bump the timeout to 3 minutes for some headroom -- # on slow machines to minimize spurious failures -- test('io-' + format + '-' + item, -- python, -- args: args, -- depends: qemu_iotests_binaries, -- env: qemu_iotests_env, -- protocol: 'tap', -- timeout: 180, -- suite: suites) -- endforeach -+# foreach item: rc.stdout().strip().split() -+# args = [qemu_iotests_check_cmd, -+# '-tap', '-' + format, item, -+# '--source-dir', meson.current_source_dir(), -+# '--build-dir', meson.current_build_dir()] -+# # Some individual tests take as long as 45 seconds -+# # Bump the timeout to 3 minutes for some headroom -+# # on slow machines to minimize spurious failures -+# test('io-' + format + '-' + item, -+# python, -+# args: args, -+# depends: qemu_iotests_binaries, -+# env: qemu_iotests_env, -+# protocol: 'tap', -+# timeout: 180, -+# suite: suites) -+# endforeach - endforeach -diff --git a/tests/qemu-iotests/testenv.py b/tests/qemu-iotests/testenv.py -index 588f30a4f1..3929a3634f 100644 ---- a/tests/qemu-iotests/testenv.py -+++ b/tests/qemu-iotests/testenv.py -@@ -244,6 +244,9 @@ def __init__(self, source_dir: str, build_dir: str, - if self.qemu_prog.endswith(f'qemu-system-{suffix}'): - self.qemu_options += f' -machine {machine}' - -+ if self.qemu_prog.endswith('qemu-system-x86_64'): -+ self.qemu_options += ' -cpu Nehalem' -+ - # QEMU_DEFAULT_MACHINE - self.qemu_default_machine = get_default_machine(self.qemu_prog) - -diff --git a/tests/qtest/fuzz-e1000e-test.c b/tests/qtest/fuzz-e1000e-test.c -index 5052883fb6..8242190170 100644 ---- a/tests/qtest/fuzz-e1000e-test.c -+++ b/tests/qtest/fuzz-e1000e-test.c -@@ -17,7 +17,7 @@ static void test_lp1879531_eth_get_rss_ex_dst_addr(void) - { - QTestState *s; - -- s = qtest_init("-nographic -monitor none -serial none -M pc-q35-5.0"); -+ s = qtest_init("-nographic -monitor none -serial none -M pc-q35-rhel9.4.0"); - - qtest_outl(s, 0xcf8, 0x80001010); - qtest_outl(s, 0xcfc, 0xe1020000); -diff --git a/tests/qtest/fuzz-virtio-scsi-test.c b/tests/qtest/fuzz-virtio-scsi-test.c -index e37b48b2cc..9f1965b530 100644 ---- a/tests/qtest/fuzz-virtio-scsi-test.c -+++ b/tests/qtest/fuzz-virtio-scsi-test.c -@@ -19,7 +19,7 @@ static void test_mmio_oob_from_memory_region_cache(void) - { - QTestState *s; - -- s = qtest_init("-M pc-q35-5.2 -m 512M " -+ s = qtest_init("-M pc-q35-rhel9.4.0 -m 512M " - "-device virtio-scsi,num_queues=8,addr=03.0 "); - - qtest_outl(s, 0xcf8, 0x80001811); -diff --git a/tests/qtest/intel-hda-test.c b/tests/qtest/intel-hda-test.c -index 663bb6c485..2efc43e3f7 100644 ---- a/tests/qtest/intel-hda-test.c -+++ b/tests/qtest/intel-hda-test.c -@@ -42,7 +42,7 @@ static void test_issue542_ich6(void) - { - QTestState *s; - -- s = qtest_init("-nographic -nodefaults -M pc-q35-6.2 " -+ s = qtest_init("-nographic -nodefaults -M pc-q35-rhel9.0.0 " - AUDIODEV - "-device intel-hda,id=" HDA_ID CODEC_DEVICES); - -diff --git a/tests/qtest/libqos/meson.build b/tests/qtest/libqos/meson.build -index 3aed6efcb8..119613237e 100644 ---- a/tests/qtest/libqos/meson.build -+++ b/tests/qtest/libqos/meson.build -@@ -44,7 +44,7 @@ libqos_srcs = files( - 'virtio-rng.c', - 'virtio-scsi.c', - 'virtio-serial.c', -- 'virtio-iommu.c', -+# 'virtio-iommu.c', - 'virtio-gpio.c', - 'virtio-scmi.c', - 'generic-pcihost.c', -diff --git a/tests/qtest/lpc-ich9-test.c b/tests/qtest/lpc-ich9-test.c -index 8ac95b89f7..0e118b76eb 100644 ---- a/tests/qtest/lpc-ich9-test.c -+++ b/tests/qtest/lpc-ich9-test.c -@@ -15,7 +15,7 @@ static void test_lp1878642_pci_bus_get_irq_level_assert(void) - { - QTestState *s; - -- s = qtest_init("-M pc-q35-5.0 " -+ s = qtest_init("-M pc-q35-rhel9.4.0 " - "-nographic -monitor none -serial none"); - - qtest_outl(s, 0xcf8, 0x8000f840); /* PMBASE */ -diff --git a/tests/qtest/meson.build b/tests/qtest/meson.build -index 36c5c13a7b..a2887d6057 100644 ---- a/tests/qtest/meson.build -+++ b/tests/qtest/meson.build -@@ -101,7 +101,6 @@ qtests_i386 = \ - 'drive_del-test', - 'tco-test', - 'cpu-plug-test', -- 'q35-test', - 'vmgenid-test', - 'migration-test', - 'test-x86-cpuid-compat', -diff --git a/tests/qtest/virtio-net-failover.c b/tests/qtest/virtio-net-failover.c -index 73dfabc272..a9dd304781 100644 ---- a/tests/qtest/virtio-net-failover.c -+++ b/tests/qtest/virtio-net-failover.c -@@ -26,6 +26,7 @@ - #define PCI_SEL_BASE 0x0010 - - #define BASE_MACHINE "-M q35 -nodefaults " \ -+ "-global ICH9-LPC.acpi-pci-hotplug-with-bridge-support=on " \ - "-device pcie-root-port,id=root0,addr=0x1,bus=pcie.0,chassis=1 " \ - "-device pcie-root-port,id=root1,addr=0x2,bus=pcie.0,chassis=2 " - --- -2.39.3 - diff --git a/0010-Increase-deletion-schedule-to-4-releases.patch b/0010-Increase-deletion-schedule-to-4-releases.patch new file mode 100644 index 0000000..51b1741 --- /dev/null +++ b/0010-Increase-deletion-schedule-to-4-releases.patch @@ -0,0 +1,37 @@ +From 7c4955a929940701a7613fe3db516adbcc97576b Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Wed, 3 Jul 2024 18:45:58 +0100 +Subject: Increase deletion schedule to 4 releases +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +Until RHEL 10 pc machine type is introduced, we have to keep +7.6.0 machine types as a special exception to our normal rule of +deleting machine types after 2 releases due to being a default +machine type. + +Signed-off-by: Daniel P. Berrangé + +Rebase notes (9.1.0) + - New patch +--- + include/hw/boards.h | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/include/hw/boards.h b/include/hw/boards.h +index da2fc92ce8..aca254ea18 100644 +--- a/include/hw/boards.h ++++ b/include/hw/boards.h +@@ -642,7 +642,7 @@ struct MachineState { + * and ver_machine_deletion_version logic in docs/conf.py and + * the text in docs/about/deprecated.rst + */ +-#define MACHINE_VER_DELETION_MAJOR 2 ++#define MACHINE_VER_DELETION_MAJOR 4 + #define MACHINE_VER_DEPRECATION_MAJOR 1 + + /* +-- +2.39.3 + diff --git a/0011-Add-downstream-aarch64-versioned-virt-machine-types.patch b/0011-Add-downstream-aarch64-versioned-virt-machine-types.patch new file mode 100644 index 0000000..ce712b7 --- /dev/null +++ b/0011-Add-downstream-aarch64-versioned-virt-machine-types.patch @@ -0,0 +1,399 @@ +From 59b7559dfa658e325e8c1e23f13f48cf6952ac2e Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Wed, 3 Jul 2024 13:25:47 +0100 +Subject: Add downstream aarch64 versioned 'virt' machine types + +Adding changes to add RHEL machine types for aarch64 architecture. + +Signed-off-by: Miroslav Rezanina +--- +Rebase notes (9.1.0): +- Merge copy+pasted base machine definition back with upstream + base machine definition to reduce RHEL delta, as is done with + other targets +- Convert to new DEFINE_VIRT_MACHINE macros +- do not remove cpu validation (review comment) +- use ifdef instead of removal for disabling unwanted upstream code + +Rebase notes (10.0.0) +- Removed unwanted sysbus dev +- Set no_nested_smmu +- Use upstream compat + +Rebase notes (10.1.0): +- Use rebase compat + +Merged patches (9.1.0): +- 043ad5ce97 Add upstream compatibility bits (partial) + +Merged patches (10.0.0 rc0): +- 03502faf70 Add upstream compatibility bits +- 17c3bccf2f arm: ensure compatibility of virt-rhel9* +- 12e5b038be arm: create new virt machine type for rhel 9.6 +- fb1bc2766e arm: create virt machine type for rhel10 +- 727307e5ed hw/arm/virt: Fix Manufacturer and Product Name in emulated SMBIOS mode +- d93fcb3940 virtio-net: disable USO for all RHEL9 (partial) +- 0440f3d003 arm: disable pauth for virt-rhel9* in RHEL10 +--- + hw/arm/virt.c | 154 +++++++++++++++++++++++++++++++++++++----- + include/hw/arm/virt.h | 1 + + 2 files changed, 137 insertions(+), 18 deletions(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index e6e98fef1c..d37d1bb3cf 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -97,6 +97,32 @@ static GlobalProperty arm_virt_compat[] = { + }; + static const size_t arm_virt_compat_len = G_N_ELEMENTS(arm_virt_compat); + ++/* ++ * RHEL9 kernels have pauth disabled while RHEL10 has it enabled, ++ * since qemu will setup the VM with pauth when KVM supports it we ++ * have to disable it for virt-rhel9* to support upgrades / migration. ++ */ ++GlobalProperty arm_rhel9_compat[] = { ++ {TYPE_ARM_CPU, "pauth", "off", .optional = true}, ++}; ++const size_t arm_rhel9_compat_len = G_N_ELEMENTS(arm_rhel9_compat); ++ ++/* ++ * This variable is for changes to properties that are RHEL specific, ++ * different to the current upstream and to be applied to the latest ++ * machine type. They may be overriden by older machine compats. ++ * ++ * virtio-net-pci variant romfiles are not needed because edk2 does ++ * fully support the pxe boot. Besides virtio romfiles are not shipped ++ * on rhel/aarch64. ++ */ ++GlobalProperty arm_rhel_compat[] = { ++ {"virtio-net-pci", "romfile", "" }, ++ {"virtio-net-pci-transitional", "romfile", "" }, ++ {"virtio-net-pci-non-transitional", "romfile", "" }, ++}; ++const size_t arm_rhel_compat_len = G_N_ELEMENTS(arm_rhel_compat); ++ + /* + * This cannot be called from the virt_machine_class_init() because + * TYPE_VIRT_MACHINE is abstract and mc->compat_props g_ptr_array_new() +@@ -106,6 +132,8 @@ static void arm_virt_compat_set(MachineClass *mc) + { + compat_props_add(mc->compat_props, arm_virt_compat, + arm_virt_compat_len); ++ compat_props_add(mc->compat_props, arm_rhel_compat, ++ arm_rhel_compat_len); + } + + #define DEFINE_VIRT_MACHINE_IMPL(latest, ...) \ +@@ -116,10 +144,11 @@ static void arm_virt_compat_set(MachineClass *mc) + MachineClass *mc = MACHINE_CLASS(oc); \ + arm_virt_compat_set(mc); \ + MACHINE_VER_SYM(options, virt, __VA_ARGS__)(mc); \ +- mc->desc = "QEMU " MACHINE_VER_STR(__VA_ARGS__) " ARM Virtual Machine"; \ ++ mc->desc = "RHEL " MACHINE_VER_STR(__VA_ARGS__) " ARM Virtual Machine"; \ + MACHINE_VER_DEPRECATION(__VA_ARGS__); \ + if (latest) { \ + mc->alias = "virt"; \ ++ mc->is_default = 1; \ + } \ + } \ + static const TypeInfo MACHINE_VER_SYM(info, virt, __VA_ARGS__) = \ +@@ -135,10 +164,10 @@ static void arm_virt_compat_set(MachineClass *mc) + } \ + type_init(MACHINE_VER_SYM(register, virt, __VA_ARGS__)); + +-#define DEFINE_VIRT_MACHINE_AS_LATEST(major, minor) \ +- DEFINE_VIRT_MACHINE_IMPL(true, major, minor) +-#define DEFINE_VIRT_MACHINE(major, minor) \ +- DEFINE_VIRT_MACHINE_IMPL(false, major, minor) ++#define DEFINE_VIRT_MACHINE_AS_LATEST(major, minor, micro) \ ++ DEFINE_VIRT_MACHINE_IMPL(true, major, minor, micro) ++#define DEFINE_VIRT_MACHINE(major, minor, micro) \ ++ DEFINE_VIRT_MACHINE_IMPL(false, major, minor, micro) + + + /* Number of external interrupt lines to configure the GIC with */ +@@ -1746,16 +1775,26 @@ static void virt_build_smbios(VirtMachineState *vms) + { + MachineClass *mc = MACHINE_GET_CLASS(vms); + MachineState *ms = MACHINE(vms); ++ VirtMachineClass *vmc = VIRT_MACHINE_GET_CLASS(vms); + uint8_t *smbios_tables, *smbios_anchor; + size_t smbios_tables_len, smbios_anchor_len; + struct smbios_phys_mem_area mem_array; ++ const char *manufacturer = "QEMU"; + const char *product = "QEMU Virtual Machine"; ++ const char *version = mc->name; + + if (kvm_enabled()) { + product = "KVM Virtual Machine"; + } + +- smbios_set_defaults("QEMU", product, mc->name, NULL, NULL); ++ if (!vmc->manufacturer_product_compat) { ++ manufacturer = "Red Hat"; ++ product = "KVM"; ++ version = mc->desc; ++ } ++ ++ smbios_set_defaults(manufacturer, product, version, ++ NULL, NULL); + + /* build the array of physical mem area from base_memmap */ + mem_array.address = vms->memmap[VIRT_MEM].base; +@@ -2517,6 +2556,7 @@ static void machvirt_init(MachineState *machine) + qemu_add_machine_init_done_notifier(&vms->machine_done); + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static bool virt_get_secure(Object *obj, Error **errp) + { + VirtMachineState *vms = VIRT_MACHINE(obj); +@@ -2544,6 +2584,7 @@ static void virt_set_virt(Object *obj, bool value, Error **errp) + + vms->virt = value; + } ++#endif /* disabled for RHEL */ + + static bool virt_get_highmem(Object *obj, Error **errp) + { +@@ -2559,6 +2600,7 @@ static void virt_set_highmem(Object *obj, bool value, Error **errp) + vms->highmem = value; + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static bool virt_get_compact_highmem(Object *obj, Error **errp) + { + VirtMachineState *vms = VIRT_MACHINE(obj); +@@ -2572,6 +2614,7 @@ static void virt_set_compact_highmem(Object *obj, bool value, Error **errp) + + vms->highmem_compact = value; + } ++#endif /* disabled for RHEL */ + + static bool virt_get_highmem_redists(Object *obj, Error **errp) + { +@@ -2664,6 +2707,7 @@ static void virt_set_its(Object *obj, bool value, Error **errp) + vms->its = value; + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static bool virt_get_dtb_randomness(Object *obj, Error **errp) + { + VirtMachineState *vms = VIRT_MACHINE(obj); +@@ -2677,6 +2721,7 @@ static void virt_set_dtb_randomness(Object *obj, bool value, Error **errp) + + vms->dtb_randomness = value; + } ++#endif /* disabled for RHEL */ + + static char *virt_get_oem_id(Object *obj, Error **errp) + { +@@ -2760,6 +2805,7 @@ static void virt_set_ras(Object *obj, bool value, Error **errp) + vms->ras = value; + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static bool virt_get_mte(Object *obj, Error **errp) + { + VirtMachineState *vms = VIRT_MACHINE(obj); +@@ -2773,6 +2819,7 @@ static void virt_set_mte(Object *obj, bool value, Error **errp) + + vms->mte = value; + } ++#endif /* disabled for RHEL */ + + static char *virt_get_gic_version(Object *obj, Error **errp) + { +@@ -3213,16 +3260,16 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + NULL + }; + ++ mc->family = "virt-rhel-Z"; + mc->init = machvirt_init; +- /* Start with max_cpus set to 512, which is the maximum supported by KVM. +- * The value may be reduced later when we have more information about the +- * configuration of the particular instance. +- */ +- mc->max_cpus = 512; ++ /* Maximum supported VCPU count for all virt-rhel* machines */ ++ mc->max_cpus = 384; ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_VFIO_CALXEDA_XGMAC); + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_VFIO_AMD_XGBE); +- machine_class_allow_dynamic_sysbus_dev(mc, TYPE_RAMFB_DEVICE); + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_VFIO_PLATFORM); ++#endif ++ machine_class_allow_dynamic_sysbus_dev(mc, TYPE_RAMFB_DEVICE); + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_UEFI_VARS_SYSBUS); + #ifdef CONFIG_TPM + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_TPM_TIS_SYSBUS); +@@ -3234,11 +3281,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + mc->minimum_page_bits = 12; + mc->possible_cpu_arch_ids = virt_possible_cpu_arch_ids; + mc->cpu_index_to_instance_props = virt_cpu_index_to_props; +-#ifdef CONFIG_TCG +- mc->default_cpu_type = ARM_CPU_TYPE_NAME("cortex-a15"); +-#else +- mc->default_cpu_type = ARM_CPU_TYPE_NAME("max"); +-#endif ++ mc->default_cpu_type = ARM_CPU_TYPE_NAME("cortex-a57"); + mc->valid_cpu_types = valid_cpu_types; + mc->get_default_cpu_node_id = virt_get_default_cpu_node_id; + mc->kvm_type = virt_kvm_type; +@@ -3263,6 +3306,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + NULL, NULL); + object_class_property_set_description(oc, "acpi", + "Enable ACPI"); ++#if 0 /* disabled for RHEL */ + object_class_property_add_bool(oc, "secure", virt_get_secure, + virt_set_secure); + object_class_property_set_description(oc, "secure", +@@ -3275,6 +3319,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + "Set on/off to enable/disable emulating a " + "guest CPU which implements the ARM " + "Virtualization Extensions"); ++#endif /* disabled for RHEL */ + + object_class_property_add_bool(oc, "highmem", virt_get_highmem, + virt_set_highmem); +@@ -3282,12 +3327,14 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + "Set on/off to enable/disable using " + "physical address space above 32 bits"); + ++#if 0 /* disabled for RHEL */ + object_class_property_add_bool(oc, "compact-highmem", + virt_get_compact_highmem, + virt_set_compact_highmem); + object_class_property_set_description(oc, "compact-highmem", + "Set on/off to enable/disable compact " + "layout for high memory regions"); ++#endif /* disabled for RHEL */ + + object_class_property_add_bool(oc, "highmem-redists", + virt_get_highmem_redists, +@@ -3323,7 +3370,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + virt_set_gic_version); + object_class_property_set_description(oc, "gic-version", + "Set GIC version. " +- "Valid values are 2, 3, 4, host and max"); ++ "Valid values are 2, 3, host and max"); + + object_class_property_add_str(oc, "iommu", virt_get_iommu, virt_set_iommu); + object_class_property_set_description(oc, "iommu", +@@ -3343,11 +3390,13 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + "Set on/off to enable/disable reporting host memory errors " + "to a KVM guest using ACPI and guest external abort exceptions"); + ++#if 0 /* disabled for RHEL */ + object_class_property_add_bool(oc, "mte", virt_get_mte, virt_set_mte); + object_class_property_set_description(oc, "mte", + "Set on/off to enable/disable emulating a " + "guest CPU which implements the ARM " + "Memory Tagging Extension"); ++#endif /* disabled for RHEL */ + + object_class_property_add_bool(oc, "its", virt_get_its, + virt_set_its); +@@ -3355,6 +3404,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + "Set on/off to enable/disable " + "ITS instantiation"); + ++#if 0 /* disabled for RHEL */ + object_class_property_add_bool(oc, "dtb-randomness", + virt_get_dtb_randomness, + virt_set_dtb_randomness); +@@ -3367,6 +3417,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + virt_set_dtb_randomness); + object_class_property_set_description(oc, "dtb-kaslr-seed", + "Deprecated synonym of dtb-randomness"); ++#endif /* disabled for RHEL */ + + object_class_property_add_str(oc, "x-oem-id", + virt_get_oem_id, +@@ -3636,3 +3687,70 @@ static void virt_machine_4_1_options(MachineClass *mc) + } + DEFINE_VIRT_MACHINE(4, 1) + #endif /* disabled for RHEL */ ++ ++static void virt_rhel_machine_10_0_0_options(MachineClass *mc) ++{ ++ VirtMachineClass *vmc = VIRT_MACHINE_CLASS(OBJECT_CLASS(mc)); ++ ++ /* QEMU 9.1 and earlier have only a stage-1 SMMU, not a nested s1+2 one */ ++ vmc->no_nested_smmu = true; ++ compat_props_add(mc->compat_props, hw_compat_rhel_10_2, hw_compat_rhel_10_2_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_10_1, hw_compat_rhel_10_1_len); ++} ++DEFINE_VIRT_MACHINE_AS_LATEST(10, 0, 0) ++ ++static void virt_rhel_machine_9_6_0_options(MachineClass *mc) ++{ ++ virt_rhel_machine_10_0_0_options(mc); ++ ++ compat_props_add(mc->compat_props, arm_rhel9_compat, arm_rhel9_compat_len); ++ /* NB: remember to move this line to the *latest* RHEL 9 machine */ ++ compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); ++} ++DEFINE_VIRT_MACHINE(9, 6, 0) ++ ++static void virt_rhel_machine_9_4_0_options(MachineClass *mc) ++{ ++ VirtMachineClass *vmc = VIRT_MACHINE_CLASS(OBJECT_CLASS(mc)); ++ ++ virt_rhel_machine_9_6_0_options(mc); ++ ++ /* From virt_machine_9_0_options() */ ++ mc->smbios_memory_device_size = 16 * GiB; ++ ++ compat_props_add(mc->compat_props, hw_compat_rhel_10_0, hw_compat_rhel_10_0_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_5, hw_compat_rhel_9_5_len); ++ ++ vmc->manufacturer_product_compat = true; ++} ++DEFINE_VIRT_MACHINE(9, 4, 0) ++ ++static void virt_rhel_machine_9_2_0_options(MachineClass *mc) ++{ ++ virt_rhel_machine_9_4_0_options(mc); ++ ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_4, hw_compat_rhel_9_4_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_3, hw_compat_rhel_9_3_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_2, hw_compat_rhel_9_2_len); ++ ++ /* RHEL 9.4 is the first supported release */ ++ mc->deprecation_reason = ++ "machine types for versions prior to 9.4 are deprecated"; ++} ++DEFINE_VIRT_MACHINE(9, 2, 0) ++ ++static void virt_rhel_machine_9_0_0_options(MachineClass *mc) ++{ ++ VirtMachineClass *vmc = VIRT_MACHINE_CLASS(OBJECT_CLASS(mc)); ++ ++ virt_rhel_machine_9_2_0_options(mc); ++ ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_1, hw_compat_rhel_9_1_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_0, hw_compat_rhel_9_0_len); ++ ++ /* Disable FEAT_LPA2 since old kernels (<= v5.12) don't boot with that feature */ ++ vmc->no_tcg_lpa2 = true; ++ /* Compact layout for high memory regions was introduced with 9.2.0 */ ++ vmc->no_highmem_compact = true; ++} ++DEFINE_VIRT_MACHINE(9, 0, 0) +diff --git a/include/hw/arm/virt.h b/include/hw/arm/virt.h +index 365a28b082..94c79d6c6d 100644 +--- a/include/hw/arm/virt.h ++++ b/include/hw/arm/virt.h +@@ -132,6 +132,7 @@ struct VirtMachineClass { + bool no_tcg_lpa2; + bool no_ns_el2_virt_timer_irq; + bool no_nested_smmu; ++ bool manufacturer_product_compat; + }; + + struct VirtMachineState { +-- +2.39.3 + diff --git a/0012-Add-downstream-s390x-versioned-s390-ccw-virtio-machi.patch b/0012-Add-downstream-s390x-versioned-s390-ccw-virtio-machi.patch new file mode 100644 index 0000000..1f3a1b0 --- /dev/null +++ b/0012-Add-downstream-s390x-versioned-s390-ccw-virtio-machi.patch @@ -0,0 +1,262 @@ +From 0976e78ca34a38fbc21c71c2f05e884c0264b62e Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Wed, 3 Jul 2024 13:44:36 +0100 +Subject: Add downstream s390x versioned 's390-ccw-virtio' machine types + +Adding changes to add RHEL machine types for s390x architecture. + +Signed-off-by: Miroslav Rezanina +-- +Rebase notes(9.1.0): +- Convert to new DEFINE_CCW_MACHINE macros + +Rebase notes (10.0.0): +- Use upstream compat +- Disabled relaxed-translation for older types + +Rebase notes (10.1.0): +- Use rebase compat + +Merged patches (9.1.0): +- 043ad5ce97 Add upstream compatibility bits (partial) +- 04596b496e s390x: remove deprecated rhel machine types + +Merged patches (10.0.0 rc0): +- 03502faf70 Add upstream compatibility bits (partial) +- d27437e5ba redhat: Add QEMU 9.1 compat handling to the s390x machine types +- 926a9d0ca2 redhat: Add rhel9.6.0 and rhel10.0.0 machine types +- d93fcb3940 virtio-net: disable USO for all RHEL9 (partial) +--- + hw/s390x/s390-virtio-ccw.c | 106 +++++++++++++++++++++++++++++-- + target/s390x/cpu_models.c | 11 ++++ + target/s390x/cpu_models.h | 2 + + target/s390x/cpu_models_system.c | 2 + + 4 files changed, 116 insertions(+), 5 deletions(-) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 2fca2bcf4d..9be423858d 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -716,6 +716,7 @@ static void s390_nmi(NMIState *n, int cpu_index, Error **errp) + s390_cpu_restart(S390_CPU(cs)); + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static ram_addr_t s390_fixup_ram_size(ram_addr_t sz) + { + /* same logic as in sclp.c */ +@@ -735,6 +736,7 @@ static ram_addr_t s390_fixup_ram_size(ram_addr_t sz) + } + return newsz; + } ++#endif /* disabled for RHEL */ + + static inline bool machine_get_aes_key_wrap(Object *obj, Error **errp) + { +@@ -883,7 +885,7 @@ static const TypeInfo ccw_machine_info = { + { \ + MachineClass *mc = MACHINE_CLASS(oc); \ + MACHINE_VER_SYM(class_options, ccw, __VA_ARGS__)(mc); \ +- mc->desc = "Virtual s390x machine (version " MACHINE_VER_STR(__VA_ARGS__) ")"; \ ++ mc->desc = "Virtual s390x machine (version rhel" MACHINE_VER_STR(__VA_ARGS__) ")"; \ + mc->init = MACHINE_VER_SYM(mach_init, ccw, __VA_ARGS__); \ + MACHINE_VER_DEPRECATION(__VA_ARGS__); \ + if (latest) { \ +@@ -904,11 +906,11 @@ static const TypeInfo ccw_machine_info = { + } \ + type_init(MACHINE_VER_SYM(register, ccw, __VA_ARGS__)) + +-#define DEFINE_CCW_MACHINE_AS_LATEST(major, minor) \ +- DEFINE_CCW_MACHINE_IMPL(true, major, minor) ++#define DEFINE_CCW_MACHINE_AS_LATEST(major, minor, micro) \ ++ DEFINE_CCW_MACHINE_IMPL(true, major, minor, micro) + +-#define DEFINE_CCW_MACHINE(major, minor) \ +- DEFINE_CCW_MACHINE_IMPL(false, major, minor) ++#define DEFINE_CCW_MACHINE(major, minor, micro) \ ++ DEFINE_CCW_MACHINE_IMPL(false, major, minor, micro) + + + #if 0 /* Disabled for Red Hat Enterprise Linux */ +@@ -1170,6 +1172,100 @@ DEFINE_CCW_MACHINE(4, 2); + + #endif /* disabled for RHEL */ + ++static void ccw_rhel_machine_10_0_0_instance_options(MachineState *machine) ++{ ++} ++ ++static void ccw_rhel_machine_10_0_0_class_options(MachineClass *mc) ++{ ++ S390CcwMachineClass *s390mc = S390_CCW_MACHINE_CLASS(mc); ++ static GlobalProperty compat[] = { ++ { TYPE_S390_PCI_DEVICE, "relaxed-translation", "off", }, ++ }; ++ ++ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); ++ compat_props_add(mc->compat_props, hw_compat_rhel_10_2, hw_compat_rhel_10_2_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_10_1, hw_compat_rhel_10_1_len); ++ s390mc->use_cpi = false; ++} ++DEFINE_CCW_MACHINE_AS_LATEST(10, 0, 0); ++ ++static void ccw_rhel_machine_9_6_0_instance_options(MachineState *machine) ++{ ++ ccw_rhel_machine_10_0_0_instance_options(machine); ++} ++ ++static void ccw_rhel_machine_9_6_0_class_options(MachineClass *mc) ++{ ++ ccw_rhel_machine_10_0_0_class_options(mc); ++ ++ /* NB: remember to move this line to the *latest* RHEL 9 machine */ ++ compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); ++} ++DEFINE_CCW_MACHINE(9, 6, 0); ++ ++static void ccw_rhel_machine_9_4_0_instance_options(MachineState *machine) ++{ ++ ccw_rhel_machine_9_6_0_instance_options(machine); ++} ++ ++static void ccw_rhel_machine_9_4_0_class_options(MachineClass *mc) ++{ ++ static GlobalProperty compat[] = { ++ { TYPE_QEMU_S390_FLIC, "migrate-all-state", "off", }, ++ }; ++ ++ ccw_rhel_machine_9_6_0_class_options(mc); ++ ++ compat_props_add(mc->compat_props, hw_compat_rhel_10_0, hw_compat_rhel_10_0_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_5, hw_compat_rhel_9_5_len); ++ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); ++} ++DEFINE_CCW_MACHINE(9, 4, 0); ++ ++static void ccw_rhel_machine_9_2_0_instance_options(MachineState *machine) ++{ ++ ccw_rhel_machine_9_4_0_instance_options(machine); ++} ++ ++static void ccw_rhel_machine_9_2_0_class_options(MachineClass *mc) ++{ ++ ccw_rhel_machine_9_4_0_class_options(mc); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_4, hw_compat_rhel_9_4_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_3, hw_compat_rhel_9_3_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_2, hw_compat_rhel_9_2_len); ++ mc->smp_props.drawers_supported = false; /* from ccw_machine_8_1 */ ++ mc->smp_props.books_supported = false; /* from ccw_machine_8_1 */ ++} ++DEFINE_CCW_MACHINE(9, 2, 0); ++ ++static void ccw_rhel_machine_9_0_0_instance_options(MachineState *machine) ++{ ++ static const S390FeatInit qemu_cpu_feat = { S390_FEAT_LIST_QEMU_V6_2 }; ++ ++ ccw_rhel_machine_9_2_0_instance_options(machine); ++ ++ s390_set_qemu_cpu_model(0x3906, 14, 2, qemu_cpu_feat); ++ s390_cpudef_featoff_greater(16, 1, S390_FEAT_PAIE); ++} ++ ++static void ccw_rhel_machine_9_0_0_class_options(MachineClass *mc) ++{ ++ S390CcwMachineClass *s390mc = S390_CCW_MACHINE_CLASS(mc); ++ static GlobalProperty compat[] = { ++ { TYPE_S390_PCI_DEVICE, "interpret", "off", }, ++ { TYPE_S390_PCI_DEVICE, "forwarding-assist", "off", }, ++ }; ++ ++ ccw_rhel_machine_9_2_0_class_options(mc); ++ ++ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_1, hw_compat_rhel_9_1_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9_0, hw_compat_rhel_9_0_len); ++ s390mc->max_threads = S390_MAX_CPUS; ++} ++DEFINE_CCW_MACHINE(9, 0, 0); ++ + static void ccw_machine_register_types(void) + { + type_register_static(&ccw_machine_info); +diff --git a/target/s390x/cpu_models.c b/target/s390x/cpu_models.c +index fe29f5c5b7..2a7fc949a4 100644 +--- a/target/s390x/cpu_models.c ++++ b/target/s390x/cpu_models.c +@@ -47,6 +47,9 @@ + * of a following release have been a superset of the previous release. With + * generation 15 one base feature and one optional feature have been deprecated. + */ ++ ++#define RHEL_CPU_DEPRECATION "use at least 'z14', or 'host' / 'qemu' / 'max'" ++ + static S390CPUDef s390_cpu_defs[] = { + /* + * Linux requires at least z10 nowadays, and IBM only supports recent CPUs +@@ -931,22 +934,30 @@ static void s390_host_cpu_model_class_init(ObjectClass *oc, const void *data) + static void s390_base_cpu_model_class_init(ObjectClass *oc, const void *data) + { + S390CPUClass *xcc = S390_CPU_CLASS(oc); ++ CPUClass *cc = CPU_CLASS(oc); + + /* all base models are migration safe */ + xcc->cpu_def = (const S390CPUDef *) data; + xcc->is_migration_safe = true; + xcc->is_static = true; + xcc->desc = xcc->cpu_def->desc; ++ if (xcc->cpu_def->gen < 14) { ++ cc->deprecation_note = RHEL_CPU_DEPRECATION; ++ } + } + + static void s390_cpu_model_class_init(ObjectClass *oc, const void *data) + { + S390CPUClass *xcc = S390_CPU_CLASS(oc); ++ CPUClass *cc = CPU_CLASS(oc); + + /* model that can change between QEMU versions */ + xcc->cpu_def = (const S390CPUDef *) data; + xcc->is_migration_safe = true; + xcc->desc = xcc->cpu_def->desc; ++ if (xcc->cpu_def->gen < 14) { ++ cc->deprecation_note = RHEL_CPU_DEPRECATION; ++ } + } + + static void s390_qemu_cpu_model_class_init(ObjectClass *oc, const void *data) +diff --git a/target/s390x/cpu_models.h b/target/s390x/cpu_models.h +index f701bc0b53..670a567c67 100644 +--- a/target/s390x/cpu_models.h ++++ b/target/s390x/cpu_models.h +@@ -38,6 +38,8 @@ typedef struct S390CPUDef { + S390FeatBitmap full_feat; + /* used to init full_feat from generated data */ + S390FeatInit full_init; ++ /* if deprecated, provides a suggestion */ ++ const char *deprecation_note; + } S390CPUDef; + + /* CPU model based on a CPU definition */ +diff --git a/target/s390x/cpu_models_system.c b/target/s390x/cpu_models_system.c +index 5b84604867..d715bdc870 100644 +--- a/target/s390x/cpu_models_system.c ++++ b/target/s390x/cpu_models_system.c +@@ -56,6 +56,7 @@ static void create_cpu_model_list(ObjectClass *klass, void *opaque) + CpuDefinitionInfo *info; + char *name = g_strdup(object_class_get_name(klass)); + S390CPUClass *scc = S390_CPU_CLASS(klass); ++ CPUClass *cc = CPU_CLASS(klass); + + /* strip off the -s390x-cpu */ + g_strrstr(name, "-" TYPE_S390_CPU)[0] = 0; +@@ -65,6 +66,7 @@ static void create_cpu_model_list(ObjectClass *klass, void *opaque) + info->migration_safe = scc->is_migration_safe; + info->q_static = scc->is_static; + info->q_typename = g_strdup(object_class_get_name(klass)); ++ info->deprecated = !!cc->deprecation_note; + /* check for unavailable features */ + if (cpu_list_data->model) { + Object *obj; +-- +2.39.3 + diff --git a/0013-Add-downstream-x86_64-versioned-pc-q35-machine-types.patch b/0013-Add-downstream-x86_64-versioned-pc-q35-machine-types.patch new file mode 100644 index 0000000..3e1945f --- /dev/null +++ b/0013-Add-downstream-x86_64-versioned-pc-q35-machine-types.patch @@ -0,0 +1,517 @@ +From f7424ca0a529f1f3a76a4044b535d1ecd7b5b114 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Wed, 3 Jul 2024 13:44:41 +0100 +Subject: Add downstream x86_64 versioned 'pc' & 'q35' machine types + +Adding changes to add RHEL machine types for x86_64 architecture. + +Signed-off-by: Miroslav Rezanina +--- +Rebase notes (9.1.0): +- Merged pc_q35_machine_rhel_options back into + pc_q35_machine_options to reduce delta to upstream +- Convert to new DEFINE_(I440FX|Q35)_MACHINE macros +- Moved x86 cpu deprecation note to device disable patch + +Rebase notes (10.0.0 rc0): +- Do not use bugfix macro for q35 +- Use upstream compat +- Add downstream specific compat +- Fixing rhel-9.4 compat issue + +Rebase notes (10.1.0): +- Use rebase compat + +Merged patches (9.1.0): +- 043ad5ce97 Add upstream compatibility bits (partial) + +Merged patches (10.0.0 rc0): +- 03502faf70 Add upstream compatibility bits +- 6be70c681b x86: ensure compatibility of pc-q35-rhel9* +- d6b6ae511c x86: create new pc-q35 machine type for rhel 9.6 +- fcf4da60bd x86: create pc-i440fx machine type for rhel10 +- 4acb295fa2 x86: create pc-q35 machine type for rhel10 +- 0489497df8 x86: remove deprecated rhel machine types +- f53dbf7532 remove stale compat definitions (partial) +- 379e14ba88 pc: q35: Bump max_cpus to 4096 vcpus +- d93fcb3940 virtio-net: disable USO for all RHEL9 (partial) +--- + hw/i386/fw_cfg.c | 2 +- + hw/i386/pc.c | 70 ++++++++++++++++++++- + hw/i386/pc_piix.c | 55 +++++++++++++++-- + hw/i386/pc_q35.c | 123 ++++++++++++++++++++++++++++++++++--- + include/hw/boards.h | 2 + + include/hw/i386/pc.h | 21 +++++++ + target/i386/kvm/kvm-cpu.c | 1 + + target/i386/kvm/kvm.c | 4 ++ + tests/qtest/meson.build | 2 +- + tests/qtest/pvpanic-test.c | 5 +- + 10 files changed, 267 insertions(+), 18 deletions(-) + +diff --git a/hw/i386/fw_cfg.c b/hw/i386/fw_cfg.c +index 07df7281d2..8009f5f31f 100644 +--- a/hw/i386/fw_cfg.c ++++ b/hw/i386/fw_cfg.c +@@ -75,7 +75,7 @@ void fw_cfg_build_smbios(PCMachineState *pcms, FWCfgState *fw_cfg, + + if (pcmc->smbios_defaults) { + /* These values are guest ABI, do not change */ +- smbios_set_defaults("QEMU", mc->desc, mc->name, ++ smbios_set_defaults("Red Hat", "KVM", mc->desc, + pcmc->smbios_stream_product, pcmc->smbios_stream_version); + } + +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index 2f58e73d33..439abe8f46 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -273,6 +273,72 @@ const size_t pc_compat_2_6_len = G_N_ELEMENTS(pc_compat_2_6); + */ + #define PC_FW_DATA (0x20000 + 0x8000) + ++/* This macro is for changes to properties that are RHEL specific, ++ * different to the current upstream and to be applied to the latest ++ * machine type. ++ */ ++GlobalProperty pc_rhel_compat[] = { ++ /* we don't support s3/s4 suspend */ ++ { "PIIX4_PM", "disable_s3", "1" }, ++ { "PIIX4_PM", "disable_s4", "1" }, ++ { "ICH9-LPC", "disable_s3", "1" }, ++ { "ICH9-LPC", "disable_s4", "1" }, ++ ++ { TYPE_X86_CPU, "host-phys-bits", "on" }, ++ { TYPE_X86_CPU, "host-phys-bits-limit", "48" }, ++ { TYPE_X86_CPU, "vmx-entry-load-perf-global-ctrl", "off" }, ++ { TYPE_X86_CPU, "vmx-exit-load-perf-global-ctrl", "off" }, ++ /* bz 1508330 */ ++ { "vfio-pci", "x-no-geforce-quirks", "on" }, ++ /* bz 1941397 */ ++ { TYPE_X86_CPU, "kvm-asyncpf-int", "on" }, ++}; ++const size_t pc_rhel_compat_len = G_N_ELEMENTS(pc_rhel_compat); ++ ++GlobalProperty pc_rhel_10_2_compat[] = { ++ /* pc_rhel_10_2_compat from pc_compat_10_0 */ ++ { TYPE_X86_CPU, "x-consistent-cache", "false" }, ++ { TYPE_X86_CPU, "x-vendor-cpuid-only-v2", "false" }, ++}; ++const size_t pc_rhel_10_2_compat_len = G_N_ELEMENTS(pc_compat_10_0); ++ ++GlobalProperty pc_rhel_10_1_compat[] = { ++ /* pc_rhel_10_1_compat from pc_compat_9_1 */ ++ { "ICH9-LPC", "x-smi-swsmi-timer", "off" }, ++ { "ICH9-LPC", "x-smi-periodic-timer", "off" }, ++ { TYPE_INTEL_IOMMU_DEVICE, "stale-tm", "on" }, ++ { TYPE_INTEL_IOMMU_DEVICE, "aw-bits", "39" }, ++}; ++const size_t pc_rhel_10_1_compat_len = G_N_ELEMENTS(pc_rhel_10_1_compat); ++ ++GlobalProperty pc_rhel_10_0_compat[] = { ++ /* pc_rhel_10_0_compat from pc_compat_9_0 */ ++ { TYPE_X86_CPU, "x-amd-topoext-features-only", "false" }, ++ { TYPE_X86_CPU, "x-l1-cache-per-thread", "false" }, ++ { TYPE_X86_CPU, "guest-phys-bits", "0" }, ++ { "sev-guest", "legacy-vm-type", "on" }, ++ { TYPE_X86_CPU, "legacy-multi-node", "on" }, ++}; ++const size_t pc_rhel_10_0_compat_len = G_N_ELEMENTS(pc_rhel_10_0_compat); ++ ++GlobalProperty pc_rhel_9_3_compat[] = { ++ /* pc_rhel_9_3_compat from pc_compat_8_0 */ ++ { "virtio-mem", "unplugged-inaccessible", "auto" }, ++}; ++const size_t pc_rhel_9_3_compat_len = G_N_ELEMENTS(pc_rhel_9_3_compat); ++ ++GlobalProperty pc_rhel_9_2_compat[] = { ++ /* pc_rhel_9_2_compat from pc_compat_7_2 */ ++ { "ICH9-LPC", "noreboot", "true" }, ++}; ++const size_t pc_rhel_9_2_compat_len = G_N_ELEMENTS(pc_rhel_9_2_compat); ++ ++GlobalProperty pc_rhel_9_0_compat[] = { ++ /* pc_rhel_9_0_compat from pc_compat_6_2 */ ++ { "virtio-mem", "unplugged-inaccessible", "off" }, ++}; ++const size_t pc_rhel_9_0_compat_len = G_N_ELEMENTS(pc_rhel_9_0_compat); ++ + GSIState *pc_gsi_create(qemu_irq **irqs, bool pci_enabled) + { + GSIState *s; +@@ -1754,6 +1820,7 @@ static void pc_machine_class_init(ObjectClass *oc, const void *data) + pcmc->kvmclock_create_always = true; + x86mc->apic_xrupt_override = true; + assert(!mc->get_hotplug_handler); ++ mc->async_pf_vmexit_disable = false; + mc->get_hotplug_handler = pc_get_hotplug_handler; + mc->hotplug_allowed = pc_hotplug_allowed; + mc->auto_enable_numa_with_memhp = true; +@@ -1761,7 +1828,8 @@ static void pc_machine_class_init(ObjectClass *oc, const void *data) + mc->has_hotpluggable_cpus = true; + mc->default_boot_order = "cad"; + mc->block_default_type = IF_IDE; +- mc->max_cpus = 255; ++ /* 240: max CPU count for RHEL */ ++ mc->max_cpus = 240; + mc->reset = pc_machine_reset; + mc->wakeup = pc_machine_wakeup; + hc->pre_plug = pc_machine_device_pre_plug_cb; +diff --git a/hw/i386/pc_piix.c b/hw/i386/pc_piix.c +index acf010e20f..d546c4a8a9 100644 +--- a/hw/i386/pc_piix.c ++++ b/hw/i386/pc_piix.c +@@ -53,6 +53,7 @@ + #include "qapi/error.h" + #include "qemu/error-report.h" + #include "system/xen.h" ++#include "migration/migration.h" + #ifdef CONFIG_XEN + #include + #include "hw/xen/xen_pt.h" +@@ -469,11 +470,11 @@ static void pc_i440fx_init(MachineState *machine) + pc_init1(machine, TYPE_I440FX_PCI_DEVICE); + } + +-#define DEFINE_I440FX_MACHINE(major, minor) \ +- DEFINE_PC_VER_MACHINE(pc_i440fx, "pc-i440fx", pc_i440fx_init, false, NULL, major, minor); ++#define DEFINE_I440FX_MACHINE(major, minor, micro) \ ++ DEFINE_PC_VER_MACHINE(pc_i440fx, "pc-i440fx", pc_i440fx_init, false, NULL, major, minor, micro); + +-#define DEFINE_I440FX_MACHINE_AS_LATEST(major, minor) \ +- DEFINE_PC_VER_MACHINE(pc_i440fx, "pc-i440fx", pc_i440fx_init, true, "pc", major, minor); ++#define DEFINE_I440FX_MACHINE_AS_LATEST(major, minor, micro) \ ++ DEFINE_PC_VER_MACHINE(pc_i440fx, "pc-i440fx", pc_i440fx_init, true, "pc", major, minor, micro); + + #if 0 /* Disabled for Red Hat Enterprise Linux */ + static void pc_i440fx_machine_options(MachineClass *m) +@@ -853,3 +854,49 @@ static void xenfv_machine_3_1_options(MachineClass *m) + DEFINE_PC_MACHINE(xenfv, "xenfv-3.1", pc_xen_hvm_init, + xenfv_machine_3_1_options); + #endif ++ ++/* Red Hat Enterprise Linux machine types */ ++ ++static void pc_machine_rhel10_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ ObjectClass *oc = OBJECT_CLASS(m); ++ pcmc->default_south_bridge = TYPE_PIIX3_DEVICE; ++ pcmc->pci_root_uid = 0; ++ pcmc->default_cpu_version = 1; ++ ++ m->family = "pc_piix_Y"; ++ m->default_machine_opts = "firmware=bios-256k.bin"; ++ m->default_display = "std"; ++ m->default_nic = "e1000"; ++ m->no_parallel = 1; ++ m->no_floppy = 1; ++ machine_class_allow_dynamic_sysbus_dev(m, TYPE_RAMFB_DEVICE); ++ ++ object_class_property_add_enum(oc, "x-south-bridge", "PCSouthBridgeOption", ++ &PCSouthBridgeOption_lookup, ++ pc_get_south_bridge, ++ pc_set_south_bridge); ++ object_class_property_set_description(oc, "x-south-bridge", ++ "Use a different south bridge than PIIX3"); ++ compat_props_add(m->compat_props, ++ pc_piix_compat_defaults, pc_piix_compat_defaults_len); ++} ++ ++static void pc_i440fx_rhel_machine_10_0_0_options(MachineClass *m) ++{ ++ pc_machine_rhel10_options(m); ++ ++ m->desc = "RHEL 10.0.0 PC (i440FX + PIIX, 1996)"; ++ m->deprecation_reason = rhel_old_machine_deprecation; ++ ++ compat_props_add(m->compat_props, hw_compat_rhel_10_2, ++ hw_compat_rhel_10_2_len); ++ compat_props_add(m->compat_props, hw_compat_rhel_10_1, ++ hw_compat_rhel_10_1_len); ++ compat_props_add(m->compat_props, pc_rhel_10_2_compat, ++ pc_rhel_10_2_compat_len); ++ compat_props_add(m->compat_props, pc_rhel_10_1_compat, ++ pc_rhel_10_1_compat_len); ++} ++DEFINE_I440FX_MACHINE_AS_LATEST(10, 0, 0); +diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c +index 2203ffd67e..e5d10c0335 100644 +--- a/hw/i386/pc_q35.c ++++ b/hw/i386/pc_q35.c +@@ -340,11 +340,11 @@ static void pc_q35_init(MachineState *machine) + #endif + } + +-#define DEFINE_Q35_MACHINE(major, minor) \ +- DEFINE_PC_VER_MACHINE(pc_q35, "pc-q35", pc_q35_init, false, NULL, major, minor); ++#define DEFINE_Q35_MACHINE(major, minor, micro) \ ++ DEFINE_PC_VER_MACHINE(pc_q35, "pc-q35", pc_q35_init, false, NULL, major, minor, micro); + +-#define DEFINE_Q35_MACHINE_AS_LATEST(major, minor) \ +- DEFINE_PC_VER_MACHINE(pc_q35, "pc-q35", pc_q35_init, false, "q35", major, minor); ++#define DEFINE_Q35_MACHINE_AS_LATEST(major, minor, micro) \ ++ DEFINE_PC_VER_MACHINE(pc_q35, "pc-q35", pc_q35_init, false, "q35", major, minor, micro); + + #define DEFINE_Q35_MACHINE_BUGFIX(major, minor, micro) \ + DEFINE_PC_VER_MACHINE(pc_q35, "pc-q35", pc_q35_init, false, NULL, major, minor, micro); +@@ -355,21 +355,21 @@ static void pc_q35_machine_options(MachineClass *m) + pcmc->pci_root_uid = 0; + pcmc->default_cpu_version = 1; + +- m->family = "pc_q35"; +- m->desc = "Standard PC (Q35 + ICH9, 2009)"; ++ m->family = "pc_q35_Z"; + m->units_per_default_bus = 1; +- m->default_machine_opts = "firmware=bios-256k.bin"; ++ m->default_machine_opts = "firmware=bios-256k.bin,hpet=off"; + m->default_display = "std"; + m->default_nic = "e1000e"; +- m->default_kernel_irqchip_split = false; + m->no_floppy = 1; + m->max_cpus = 4096; +- m->no_parallel = !module_object_class_by_name(TYPE_ISA_PARALLEL); ++ m->no_parallel = 1; ++ m->alias = "q35"; + machine_class_allow_dynamic_sysbus_dev(m, TYPE_AMD_IOMMU_DEVICE); + machine_class_allow_dynamic_sysbus_dev(m, TYPE_INTEL_IOMMU_DEVICE); + machine_class_allow_dynamic_sysbus_dev(m, TYPE_RAMFB_DEVICE); + machine_class_allow_dynamic_sysbus_dev(m, TYPE_VMBUS_BRIDGE); + machine_class_allow_dynamic_sysbus_dev(m, TYPE_UEFI_VARS_X64); ++ compat_props_add(m->compat_props, pc_rhel_compat, pc_rhel_compat_len); + compat_props_add(m->compat_props, + pc_q35_compat_defaults, pc_q35_compat_defaults_len); + } +@@ -687,3 +687,108 @@ static void pc_q35_machine_2_6_options(MachineClass *m) + + DEFINE_Q35_MACHINE(2, 6); + #endif /* Disabled for Red Hat Enterprise Linux */ ++ ++/* Red Hat Enterprise Linux machine types */ ++ ++static void pc_q35_rhel_machine_10_0_0_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ pc_q35_machine_options(m); ++ m->desc = "RHEL-10.0.0 PC (Q35 + ICH9, 2009)"; ++ pcmc->smbios_stream_product = "RHEL"; ++ pcmc->smbios_stream_version = "10.0.0"; ++ ++ compat_props_add(m->compat_props, hw_compat_rhel_10_2, ++ hw_compat_rhel_10_2_len); ++ compat_props_add(m->compat_props, hw_compat_rhel_10_1, ++ hw_compat_rhel_10_1_len); ++ compat_props_add(m->compat_props, pc_rhel_10_2_compat, ++ pc_rhel_10_2_compat_len); ++ compat_props_add(m->compat_props, pc_rhel_10_1_compat, ++ pc_rhel_10_1_compat_len); ++} ++DEFINE_Q35_MACHINE_AS_LATEST(10, 0, 0); ++ ++static void pc_q35_rhel_machine_9_6_0_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ pc_q35_rhel_machine_10_0_0_options(m); ++ m->desc = "RHEL-9.6.0 PC (Q35 + ICH9, 2009)"; ++ pcmc->smbios_stream_product = "RHEL"; ++ pcmc->smbios_stream_version = "9.6.0"; ++ ++ /* NB: remember to move this line to the *latest* RHEL 9 machine */ ++ compat_props_add(m->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); ++} ++ ++DEFINE_Q35_MACHINE(9, 6, 0); ++ ++static void pc_q35_rhel_machine_9_4_0_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ pc_q35_rhel_machine_9_6_0_options(m); ++ ++ /* older RHEL machines continue to support 710 vcpus */ ++ m->max_cpus = 710; ++ m->desc = "RHEL-9.4.0 PC (Q35 + ICH9, 2009)"; ++ pcmc->smbios_stream_product = "RHEL"; ++ pcmc->smbios_stream_version = "9.4.0"; ++ ++ /* From pc_q35_machine_9_0_options() */ ++ pcmc->isa_bios_alias = false; ++ m->smbios_memory_device_size = 16 * GiB; ++ ++ compat_props_add(m->compat_props, hw_compat_rhel_10_0, ++ hw_compat_rhel_10_0_len); ++ compat_props_add(m->compat_props, hw_compat_rhel_9_5, ++ hw_compat_rhel_9_5_len); ++ compat_props_add(m->compat_props, pc_rhel_10_0_compat, ++ pc_rhel_10_0_compat_len); ++} ++ ++DEFINE_Q35_MACHINE(9, 4, 0); ++ ++static void pc_q35_rhel_machine_9_2_0_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ pc_q35_rhel_machine_9_4_0_options(m); ++ m->desc = "RHEL-9.2.0 PC (Q35 + ICH9, 2009)"; ++ pcmc->smbios_stream_product = "RHEL"; ++ pcmc->smbios_stream_version = "9.2.0"; ++ ++ /* From pc_q35_8_0_machine_options() */ ++ pcmc->default_smbios_ep_type = SMBIOS_ENTRY_POINT_TYPE_32; ++ /* From pc_q35_8_1_machine_options() */ ++ pcmc->broken_32bit_mem_addr_check = true; ++ ++ compat_props_add(m->compat_props, hw_compat_rhel_9_4, ++ hw_compat_rhel_9_4_len); ++ compat_props_add(m->compat_props, hw_compat_rhel_9_3, ++ hw_compat_rhel_9_3_len); ++ compat_props_add(m->compat_props, pc_rhel_9_3_compat, ++ pc_rhel_9_3_compat_len); ++ compat_props_add(m->compat_props, hw_compat_rhel_9_2, ++ hw_compat_rhel_9_2_len); ++ compat_props_add(m->compat_props, pc_rhel_9_2_compat, ++ pc_rhel_9_2_compat_len); ++} ++ ++DEFINE_Q35_MACHINE(9, 2, 0); ++ ++static void pc_q35_rhel_machine_9_0_0_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ pc_q35_rhel_machine_9_2_0_options(m); ++ m->desc = "RHEL-9.0.0 PC (Q35 + ICH9, 2009)"; ++ pcmc->smbios_stream_product = "RHEL"; ++ pcmc->smbios_stream_version = "9.0.0"; ++ pcmc->enforce_amd_1tb_hole = false; ++ compat_props_add(m->compat_props, hw_compat_rhel_9_1, ++ hw_compat_rhel_9_1_len); ++ compat_props_add(m->compat_props, hw_compat_rhel_9_0, ++ hw_compat_rhel_9_0_len); ++ compat_props_add(m->compat_props, pc_rhel_9_0_compat, ++ pc_rhel_9_0_compat_len); ++} ++ ++DEFINE_Q35_MACHINE(9, 0, 0); +diff --git a/include/hw/boards.h b/include/hw/boards.h +index aca254ea18..22c3abd51e 100644 +--- a/include/hw/boards.h ++++ b/include/hw/boards.h +@@ -308,6 +308,8 @@ struct MachineClass { + strList *allowed_dynamic_sysbus_devices; + bool auto_enable_numa_with_memhp; + bool auto_enable_numa_with_memdev; ++ /* RHEL only */ ++ bool async_pf_vmexit_disable; + bool ignore_boot_device_suffixes; + bool smbus_no_migration_support; + bool nvdimm_supported; +diff --git a/include/hw/i386/pc.h b/include/hw/i386/pc.h +index 3b4ea24c20..633df2fcf8 100644 +--- a/include/hw/i386/pc.h ++++ b/include/hw/i386/pc.h +@@ -301,6 +301,27 @@ extern const size_t pc_compat_2_7_len; + extern GlobalProperty pc_compat_2_6[]; + extern const size_t pc_compat_2_6_len; + ++extern GlobalProperty pc_rhel_compat[]; ++extern const size_t pc_rhel_compat_len; ++ ++extern GlobalProperty pc_rhel_10_2_compat[]; ++extern const size_t pc_rhel_10_2_compat_len; ++ ++extern GlobalProperty pc_rhel_10_1_compat[]; ++extern const size_t pc_rhel_10_1_compat_len; ++ ++extern GlobalProperty pc_rhel_10_0_compat[]; ++extern const size_t pc_rhel_10_0_compat_len; ++ ++extern GlobalProperty pc_rhel_9_3_compat[]; ++extern const size_t pc_rhel_9_3_compat_len; ++ ++extern GlobalProperty pc_rhel_9_2_compat[]; ++extern const size_t pc_rhel_9_2_compat_len; ++ ++extern GlobalProperty pc_rhel_9_0_compat[]; ++extern const size_t pc_rhel_9_0_compat_len; ++ + #define DEFINE_PC_MACHINE(suffix, namestr, initfn, optsfn) \ + static void pc_machine_##suffix##_class_init(ObjectClass *oc, \ + const void *data) \ +diff --git a/target/i386/kvm/kvm-cpu.c b/target/i386/kvm/kvm-cpu.c +index 89a7953659..74c0b036e3 100644 +--- a/target/i386/kvm/kvm-cpu.c ++++ b/target/i386/kvm/kvm-cpu.c +@@ -175,6 +175,7 @@ static PropValue kvm_default_props[] = { + { "acpi", "off" }, + { "monitor", "off" }, + { "svm", "off" }, ++ { "kvm-pv-unhalt", "on" }, + { NULL, NULL }, + }; + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 369626f8c8..0eb39d22d6 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -4389,6 +4389,7 @@ static int kvm_get_msrs(X86CPU *cpu) + struct kvm_msr_entry *msrs = cpu->kvm_msr_buf->entries; + int ret, i; + uint64_t mtrr_top_bits; ++ MachineClass *mc = MACHINE_GET_CLASS(qdev_get_machine()); + + kvm_msr_buf_reset(cpu); + +@@ -4786,6 +4787,9 @@ static int kvm_get_msrs(X86CPU *cpu) + break; + case MSR_KVM_ASYNC_PF_EN: + env->async_pf_en_msr = msrs[i].data; ++ if (mc->async_pf_vmexit_disable) { ++ env->async_pf_en_msr &= ~(1ULL << 2); ++ } + break; + case MSR_KVM_ASYNC_PF_INT: + env->async_pf_int_msr = msrs[i].data; +diff --git a/tests/qtest/meson.build b/tests/qtest/meson.build +index 669d07c06b..b96aa06084 100644 +--- a/tests/qtest/meson.build ++++ b/tests/qtest/meson.build +@@ -49,6 +49,7 @@ qtests_filter = \ + (get_option('default_devices') and host_os != 'windows' ? ['test-filter-mirror'] : []) + \ + (get_option('default_devices') and host_os != 'windows' ? ['test-filter-redirector'] : []) + ++# RHEL: Removed intel-iommu-test as it's not working with 10.0 machine type + qtests_i386 = \ + (slirp.found() ? ['pxe-test'] : []) + \ + qtests_filter + \ +@@ -94,7 +95,6 @@ qtests_i386 = \ + (config_all_devices.has_key('CONFIG_SB16') ? ['fuzz-sb16-test'] : []) + \ + (config_all_devices.has_key('CONFIG_SDHCI_PCI') ? ['fuzz-sdcard-test'] : []) + \ + (config_all_devices.has_key('CONFIG_ESP_PCI') ? ['am53c974-test'] : []) + \ +- (config_all_devices.has_key('CONFIG_VTD') ? ['intel-iommu-test'] : []) + \ + (host_os != 'windows' and \ + config_all_devices.has_key('CONFIG_ACPI_ERST') ? ['erst-test'] : []) + \ + (config_all_devices.has_key('CONFIG_PCIE_PORT') and \ +diff --git a/tests/qtest/pvpanic-test.c b/tests/qtest/pvpanic-test.c +index 5606baf47b..094c56b0cd 100644 +--- a/tests/qtest/pvpanic-test.c ++++ b/tests/qtest/pvpanic-test.c +@@ -18,7 +18,7 @@ static void test_panic_nopause(void) + QDict *response, *data; + QTestState *qts; + +- qts = qtest_init("-device pvpanic -action panic=none"); ++ qts = qtest_init("-M q35 -device pvpanic -action panic=none"); + + val = qtest_inb(qts, 0x505); + g_assert_cmpuint(val, ==, PVPANIC_EVENTS); +@@ -41,7 +41,8 @@ static void test_panic(void) + QDict *response, *data; + QTestState *qts; + +- qts = qtest_init("-device pvpanic -action panic=pause"); ++ /* RHEL: Use q35 */ ++ qts = qtest_init("-M q35 -device pvpanic -action panic=pause"); + + val = qtest_inb(qts, 0x505); + g_assert_cmpuint(val, ==, PVPANIC_EVENTS); +-- +2.39.3 + diff --git a/0014-Disable-virtio-net-pci-romfile-loading-on-riscv64.patch b/0014-Disable-virtio-net-pci-romfile-loading-on-riscv64.patch new file mode 100644 index 0000000..21b840f --- /dev/null +++ b/0014-Disable-virtio-net-pci-romfile-loading-on-riscv64.patch @@ -0,0 +1,58 @@ +From 7bc17dffbc537e8546249c7c2d19e426ad50e61f Mon Sep 17 00:00:00 2001 +From: Andrea Bolognani +Date: Tue, 10 Jun 2025 14:27:29 +0200 +Subject: Disable virtio-net-pci romfile loading on riscv64 + +RH-Author: Andrea Bolognani +RH-MergeRequest: 373: Various small fixes +RH-Jira: RHEL-96057 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [4/4] b490ef3c3ab6a47f90c67016f685e19a65d97100 (abologna/centos-stream-qemu-kvm) + +Same motivation for disabling it as on aarch64. + +Signed-off-by: Andrea Bolognani + +Patch-name: kvm-Disable-virtio-net-pci-romfile-loading-on-riscv64.patch +Patch-id: 51 +Patch-present-in-specfile: True +--- + hw/riscv/virt.c | 15 +++++++++++++++ + 1 file changed, 15 insertions(+) + +diff --git a/hw/riscv/virt.c b/hw/riscv/virt.c +index ab5a9ec613..3b187d6c98 100644 +--- a/hw/riscv/virt.c ++++ b/hw/riscv/virt.c +@@ -59,6 +59,18 @@ + #include "hw/virtio/virtio-iommu.h" + #include "hw/uefi/var-service-api.h" + ++/* ++ * virtio-net-pci variant romfiles are not needed because edk2 does ++ * fully support the pxe boot. Besides virtio romfiles are not shipped ++ * on rhel/riscv64. ++ */ ++static GlobalProperty riscv_virt_compat[] = { ++ {"virtio-net-pci", "romfile", "" }, ++ {"virtio-net-pci-transitional", "romfile", "" }, ++ {"virtio-net-pci-non-transitional", "romfile", "" }, ++}; ++const size_t riscv_virt_compat_len = G_N_ELEMENTS(riscv_virt_compat); ++ + /* KVM AIA only supports APLIC MSI. APLIC Wired is always emulated by QEMU. */ + static bool virt_use_kvm_aia_aplic_imsic(RISCVVirtAIAType aia_type) + { +@@ -1978,6 +1990,9 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + NULL, NULL); + object_class_property_set_description(oc, "iommu-sys", + "Enable IOMMU platform device"); ++ ++ compat_props_add(mc->compat_props, riscv_virt_compat, ++ riscv_virt_compat_len); + } + + static const TypeInfo virt_machine_typeinfo = { +-- +2.39.3 + diff --git a/0015-Add-upstream-compatibility-bits.patch b/0015-Add-upstream-compatibility-bits.patch deleted file mode 100644 index de8b72f..0000000 --- a/0015-Add-upstream-compatibility-bits.patch +++ /dev/null @@ -1,145 +0,0 @@ -From 043ad5ce9789dbbfe1a888de58f6039ea7ae47a4 Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Wed, 20 Mar 2024 05:34:32 -0400 -Subject: Add upstream compatibility bits - -Adding new compats structure for changes introduced during rebase to QEMU 9.0.0. - -Signed-off-by: Miroslav Rezanina - ---- - -Rebase notes (9.0.0 rc2): -- Add aw-bits setting for aarch compat record (overwritten for 9.4 and older) ---- - hw/arm/virt.c | 6 ++++-- - hw/core/machine.c | 10 ++++++++++ - hw/i386/pc_piix.c | 3 ++- - hw/i386/pc_q35.c | 3 +++ - hw/s390x/s390-virtio-ccw.c | 1 + - include/hw/boards.h | 3 +++ - 6 files changed, 23 insertions(+), 3 deletions(-) - -diff --git a/hw/arm/virt.c b/hw/arm/virt.c -index 22bc345137..3f0496cdb9 100644 ---- a/hw/arm/virt.c -+++ b/hw/arm/virt.c -@@ -85,6 +85,7 @@ - #include "hw/char/pl011.h" - #include "qemu/guest-random.h" - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static GlobalProperty arm_virt_compat[] = { - { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "48" }, - }; -@@ -101,7 +102,6 @@ static void arm_virt_compat_set(MachineClass *mc) - arm_virt_compat_len); - } - --#if 0 /* Disabled for Red Hat Enterprise Linux */ - #define DEFINE_VIRT_MACHINE_LATEST(major, minor, latest) \ - static void virt_##major##_##minor##_class_init(ObjectClass *oc, \ - void *data) \ -@@ -144,6 +144,8 @@ GlobalProperty arm_rhel_compat[] = { - {"virtio-net-pci", "romfile", "" }, - {"virtio-net-pci-transitional", "romfile", "" }, - {"virtio-net-pci-non-transitional", "romfile", "" }, -+ /* arm_rhel_compat from arm_virt_compat, added for 9.0.0 rebase */ -+ { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "48" }, - }; - const size_t arm_rhel_compat_len = G_N_ELEMENTS(arm_rhel_compat); - -@@ -3534,7 +3536,6 @@ static void rhel_machine_class_init(ObjectClass *oc, void *data) - { - MachineClass *mc = MACHINE_CLASS(oc); - HotplugHandlerClass *hc = HOTPLUG_HANDLER_CLASS(oc); -- arm_virt_compat_set(mc); - - mc->family = "virt-rhel-Z"; - mc->init = machvirt_init; -@@ -3728,6 +3729,7 @@ type_init(rhel_machine_init); - - static void rhel940_virt_options(MachineClass *mc) - { -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_5, hw_compat_rhel_9_5_len); - } - DEFINE_RHEL_MACHINE_AS_LATEST(9, 4, 0) - -diff --git a/hw/core/machine.c b/hw/core/machine.c -index 695cb89a46..0f256d9633 100644 ---- a/hw/core/machine.c -+++ b/hw/core/machine.c -@@ -302,6 +302,16 @@ const size_t hw_compat_2_1_len = G_N_ELEMENTS(hw_compat_2_1); - const char *rhel_old_machine_deprecation = - "machine types for previous major releases are deprecated"; - -+GlobalProperty hw_compat_rhel_9_5[] = { -+ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ -+ { "migration", "zero-page-detection", "legacy"}, -+ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ -+ { TYPE_VIRTIO_IOMMU_PCI, "granule", "4k" }, -+ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ -+ { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "64" }, -+}; -+const size_t hw_compat_rhel_9_5_len = G_N_ELEMENTS(hw_compat_rhel_9_5); -+ - GlobalProperty hw_compat_rhel_9_4[] = { - /* hw_compat_rhel_9_4 from hw_compat_8_0 */ - { TYPE_VIRTIO_NET, "host_uso", "off"}, -diff --git a/hw/i386/pc_piix.c b/hw/i386/pc_piix.c -index a647262d63..6b260682eb 100644 ---- a/hw/i386/pc_piix.c -+++ b/hw/i386/pc_piix.c -@@ -1015,7 +1015,8 @@ static void pc_machine_rhel760_options(MachineClass *m) - object_class_property_set_description(oc, "x-south-bridge", - "Use a different south bridge than PIIX3"); - -- -+ compat_props_add(m->compat_props, hw_compat_rhel_9_5, -+ hw_compat_rhel_9_5_len); - compat_props_add(m->compat_props, hw_compat_rhel_9_4, - hw_compat_rhel_9_4_len); - compat_props_add(m->compat_props, hw_compat_rhel_9_3, -diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c -index e872dc7e46..2b54944c0f 100644 ---- a/hw/i386/pc_q35.c -+++ b/hw/i386/pc_q35.c -@@ -733,6 +733,9 @@ static void pc_q35_machine_rhel940_options(MachineClass *m) - m->desc = "RHEL-9.4.0 PC (Q35 + ICH9, 2009)"; - pcmc->smbios_stream_product = "RHEL"; - pcmc->smbios_stream_version = "9.4.0"; -+ -+ compat_props_add(m->compat_props, hw_compat_rhel_9_5, -+ hw_compat_rhel_9_5_len); - } - - DEFINE_PC_MACHINE(q35_rhel940, "pc-q35-rhel9.4.0", pc_q35_init_rhel940, -diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c -index ff753a29e0..9ad54682c6 100644 ---- a/hw/s390x/s390-virtio-ccw.c -+++ b/hw/s390x/s390-virtio-ccw.c -@@ -1282,6 +1282,7 @@ static void ccw_machine_rhel940_instance_options(MachineState *machine) - - static void ccw_machine_rhel940_class_options(MachineClass *mc) - { -+ compat_props_add(mc->compat_props, hw_compat_rhel_9_5, hw_compat_rhel_9_5_len); - } - DEFINE_CCW_MACHINE(rhel940, "rhel9.4.0", true); - -diff --git a/include/hw/boards.h b/include/hw/boards.h -index 46b8725c41..cca62f906b 100644 ---- a/include/hw/boards.h -+++ b/include/hw/boards.h -@@ -514,6 +514,9 @@ extern const size_t hw_compat_2_2_len; - extern GlobalProperty hw_compat_2_1[]; - extern const size_t hw_compat_2_1_len; - -+extern GlobalProperty hw_compat_rhel_9_5[]; -+extern const size_t hw_compat_rhel_9_5_len; -+ - extern GlobalProperty hw_compat_rhel_9_4[]; - extern const size_t hw_compat_rhel_9_4_len; - --- -2.39.3 - diff --git a/0015-Revert-meson-temporarily-disable-Wunused-function.patch b/0015-Revert-meson-temporarily-disable-Wunused-function.patch new file mode 100644 index 0000000..e3bec6c --- /dev/null +++ b/0015-Revert-meson-temporarily-disable-Wunused-function.patch @@ -0,0 +1,32 @@ +From d132184ec50656d9ed675801695a66f620fe0821 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Wed, 3 Jul 2024 13:47:04 +0100 +Subject: Revert "meson: temporarily disable -Wunused-function" +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +This reverts commit c682111eaa73d9b985187b8be330338f50b78a7a. + +No longer needed after introduction of downstream machines. + +Signed-off-by: Daniel P. Berrangé +--- + meson.build | 1 - + 1 file changed, 1 deletion(-) + +diff --git a/meson.build b/meson.build +index 23494666d9..ef2e5be6e2 100644 +--- a/meson.build ++++ b/meson.build +@@ -757,7 +757,6 @@ warn_flags = [ + '-Wno-string-plus-int', + '-Wno-tautological-type-limit-compare', + '-Wno-typedef-redefinition', +- '-Wno-unused-function', + ] + + if host_os != 'darwin' +-- +2.39.3 + diff --git a/0016-Disable-FDC-devices.patch b/0016-Disable-FDC-devices.patch deleted file mode 100644 index 23133f7..0000000 --- a/0016-Disable-FDC-devices.patch +++ /dev/null @@ -1,29 +0,0 @@ -From f24c7a1feef2a6f153582c06f10871b78a014bf1 Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Fri, 26 Apr 2024 05:58:31 -0400 -Subject: Disable FDC devices - ---- - configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 6 +++--- - 1 file changed, 3 insertions(+), 3 deletions(-) - -diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -index d60ff1bcfc..ee75bb4c21 100644 ---- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -+++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -@@ -19,9 +19,9 @@ CONFIG_DIMM=y - CONFIG_E1000E_PCI_EXPRESS=y - CONFIG_E1000_PCI=y - CONFIG_EDU=y --CONFIG_FDC=y --CONFIG_FDC_SYSBUS=y --CONFIG_FDC_ISA=y -+#CONFIG_FDC=y -+#CONFIG_FDC_SYSBUS=y -+#CONFIG_FDC_ISA=y - CONFIG_FW_CFG_DMA=y - CONFIG_HDA=y - CONFIG_HYPERV=y --- -2.39.3 - diff --git a/0016-Enable-make-check.patch b/0016-Enable-make-check.patch new file mode 100644 index 0000000..809be58 --- /dev/null +++ b/0016-Enable-make-check.patch @@ -0,0 +1,365 @@ +From 0f4d74ce6ff137291962909aaebc1dbf3dd27508 Mon Sep 17 00:00:00 2001 +From: Miroslav Rezanina +Date: Wed, 2 Sep 2020 09:39:41 +0200 +Subject: Enable make check + +Fixing tests after device disabling and machine types changes and enabling +make check run during build. + +Signed-off-by: Miroslav Rezanina + +--- +Rebase notes (9.1.0): +- Disable fdc-test +- Use q35 machine type for new pvpanic test + +Rebase notes (10.0.0 rc0) +- Disable mem_addr_space functional test +- Updated removal of q35 test (upstream change) + +Rebase notes (10.0.0): +- Add riscv changes + +Rebase notes (10.1.0 rc0): +- Comment out unused code +--- + .distro/qemu-kvm.spec.template | 4 +-- + tests/functional/meson.build | 2 +- + tests/functional/test_aarch64_tcg_plugins.py | 4 +-- + tests/qemu-iotests/meson.build | 34 ++++++++++---------- + tests/qemu-iotests/testenv.py | 3 ++ + tests/qtest/bios-tables-test.c | 10 ++++++ + tests/qtest/fuzz-e1000e-test.c | 2 +- + tests/qtest/fuzz-virtio-scsi-test.c | 2 +- + tests/qtest/intel-hda-test.c | 2 +- + tests/qtest/libqos/meson.build | 2 +- + tests/qtest/lpc-ich9-test.c | 2 +- + tests/qtest/machine-none-test.c | 2 +- + tests/qtest/meson.build | 1 - + tests/qtest/pvpanic-test.c | 2 +- + tests/qtest/riscv-csr-test.c | 4 +++ + tests/qtest/virtio-net-failover.c | 1 + + 16 files changed, 47 insertions(+), 30 deletions(-) + +diff --git a/tests/functional/meson.build b/tests/functional/meson.build +index 311c6f1806..c7f9051c90 100644 +--- a/tests/functional/meson.build ++++ b/tests/functional/meson.build +@@ -319,7 +319,7 @@ tests_sparc64_system_thorough = [ + + tests_x86_64_system_quick = [ + 'cpu_queries', +- 'mem_addr_space', ++# 'mem_addr_space', + 'migration', + 'pc_cpu_hotplug_props', + 'virtio_version', +diff --git a/tests/functional/test_aarch64_tcg_plugins.py b/tests/functional/test_aarch64_tcg_plugins.py +index cb7e9298fb..9efa826b01 100755 +--- a/tests/functional/test_aarch64_tcg_plugins.py ++++ b/tests/functional/test_aarch64_tcg_plugins.py +@@ -64,7 +64,7 @@ class PluginKernelNormal(PluginKernelBase): + + def test_aarch64_virt_insn(self): + self.set_machine('virt') +- self.cpu='cortex-a53' ++ self.cpu='cortex-a57' + kernel_path = self.ASSET_KERNEL.fetch() + kernel_command_line = (self.KERNEL_COMMON_COMMAND_LINE + + 'console=ttyAMA0') +@@ -90,7 +90,7 @@ def test_aarch64_virt_insn(self): + + def test_aarch64_virt_insn_icount(self): + self.set_machine('virt') +- self.cpu='cortex-a53' ++ self.cpu='cortex-a57' + kernel_path = self.ASSET_KERNEL.fetch() + kernel_command_line = (self.KERNEL_COMMON_COMMAND_LINE + + 'console=ttyAMA0') +diff --git a/tests/qemu-iotests/meson.build b/tests/qemu-iotests/meson.build +index fad340ad59..3c0d5241f6 100644 +--- a/tests/qemu-iotests/meson.build ++++ b/tests/qemu-iotests/meson.build +@@ -51,21 +51,21 @@ foreach format, speed: qemu_iotests_formats + check: true, + ) + +- foreach item: rc.stdout().strip().split() +- args = [qemu_iotests_check_cmd, +- '-tap', '-' + format, item, +- '--source-dir', meson.current_source_dir(), +- '--build-dir', meson.current_build_dir()] +- # Some individual tests take as long as 45 seconds +- # Bump the timeout to 3 minutes for some headroom +- # on slow machines to minimize spurious failures +- test('io-' + format + '-' + item, +- python, +- args: args, +- depends: qemu_iotests_binaries, +- env: qemu_iotests_env, +- protocol: 'tap', +- timeout: 180, +- suite: suites) +- endforeach ++# foreach item: rc.stdout().strip().split() ++# args = [qemu_iotests_check_cmd, ++# '-tap', '-' + format, item, ++# '--source-dir', meson.current_source_dir(), ++# '--build-dir', meson.current_build_dir()] ++# # Some individual tests take as long as 45 seconds ++# # Bump the timeout to 3 minutes for some headroom ++# # on slow machines to minimize spurious failures ++# test('io-' + format + '-' + item, ++# python, ++# args: args, ++# depends: qemu_iotests_binaries, ++# env: qemu_iotests_env, ++# protocol: 'tap', ++# timeout: 180, ++# suite: suites) ++# endforeach + endforeach +diff --git a/tests/qemu-iotests/testenv.py b/tests/qemu-iotests/testenv.py +index 6326e46b7b..bc849ae9cf 100644 +--- a/tests/qemu-iotests/testenv.py ++++ b/tests/qemu-iotests/testenv.py +@@ -252,6 +252,9 @@ def __init__(self, source_dir: str, build_dir: str, + if self.qemu_prog.endswith(f'qemu-system-{suffix}'): + self.qemu_options += f' -machine {machine}' + ++ if self.qemu_prog.endswith('qemu-system-x86_64'): ++ self.qemu_options += ' -cpu Nehalem' ++ + # QEMU_DEFAULT_MACHINE + self.qemu_default_machine = get_default_machine(self.qemu_prog) + +diff --git a/tests/qtest/bios-tables-test.c b/tests/qtest/bios-tables-test.c +index e7e6926c81..386196edc8 100644 +--- a/tests/qtest/bios-tables-test.c ++++ b/tests/qtest/bios-tables-test.c +@@ -1755,6 +1755,7 @@ static void test_acpi_microvm_ioapic2_tcg(void) + free_test_data(&data); + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void test_acpi_riscv64_virt_tcg_numamem(void) + { + test_data data = { +@@ -1780,6 +1781,7 @@ static void test_acpi_riscv64_virt_tcg_numamem(void) + &data); + free_test_data(&data); + } ++#endif /* disabled for RHEL */ + + static void test_acpi_aarch64_virt_tcg_numamem(void) + { +@@ -1856,6 +1858,7 @@ static void test_acpi_aarch64_virt_tcg_acpi_spcr(void) + free_test_data(&data); + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void test_acpi_riscv64_virt_tcg_acpi_spcr(void) + { + test_data data = { +@@ -1874,6 +1877,7 @@ static void test_acpi_riscv64_virt_tcg_acpi_spcr(void) + "-machine spcr=off", &data); + free_test_data(&data); + } ++#endif + + static void test_acpi_tcg_acpi_hmat(const char *machine, const char *arch) + { +@@ -2171,6 +2175,7 @@ static void test_acpi_microvm_acpi_erst(void) + } + #endif /* CONFIG_POSIX */ + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void test_acpi_riscv64_virt_tcg(void) + { + test_data data = { +@@ -2192,6 +2197,7 @@ static void test_acpi_riscv64_virt_tcg(void) + test_acpi_one("-cpu rva22s64 ", &data); + free_test_data(&data); + } ++#endif /* disabled for RHEL */ + + static void test_acpi_aarch64_virt_tcg(void) + { +@@ -2526,6 +2532,7 @@ static void test_acpi_aarch64_virt_oem_fields(void) + g_free(args); + } + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + #define LOONGARCH64_INIT_TEST_DATA(data) \ + test_data data = { \ + .machine = "virt", \ +@@ -2594,6 +2601,7 @@ static void test_acpi_loongarch64_virt_oem_fields(void) + free_test_data(&data); + g_free(args); + } ++#endif + + int main(int argc, char *argv[]) + { +@@ -2769,6 +2777,7 @@ int main(int argc, char *argv[]) + qtest_add_func("acpi/virt/viot", test_acpi_aarch64_virt_viot); + } + } ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + } else if (strcmp(arch, "riscv64") == 0) { + if (has_tcg && qtest_has_device("virtio-blk-pci")) { + qtest_add_func("acpi/virt", test_acpi_riscv64_virt_tcg); +@@ -2788,6 +2797,7 @@ int main(int argc, char *argv[]) + qtest_add_func("acpi/virt/oem-fields", + test_acpi_loongarch64_virt_oem_fields); + } ++#endif /* disabled for RHEL */ + } + ret = g_test_run(); + boot_sector_cleanup(disk); +diff --git a/tests/qtest/fuzz-e1000e-test.c b/tests/qtest/fuzz-e1000e-test.c +index 5052883fb6..8242190170 100644 +--- a/tests/qtest/fuzz-e1000e-test.c ++++ b/tests/qtest/fuzz-e1000e-test.c +@@ -17,7 +17,7 @@ static void test_lp1879531_eth_get_rss_ex_dst_addr(void) + { + QTestState *s; + +- s = qtest_init("-nographic -monitor none -serial none -M pc-q35-5.0"); ++ s = qtest_init("-nographic -monitor none -serial none -M pc-q35-rhel9.4.0"); + + qtest_outl(s, 0xcf8, 0x80001010); + qtest_outl(s, 0xcfc, 0xe1020000); +diff --git a/tests/qtest/fuzz-virtio-scsi-test.c b/tests/qtest/fuzz-virtio-scsi-test.c +index e37b48b2cc..9f1965b530 100644 +--- a/tests/qtest/fuzz-virtio-scsi-test.c ++++ b/tests/qtest/fuzz-virtio-scsi-test.c +@@ -19,7 +19,7 @@ static void test_mmio_oob_from_memory_region_cache(void) + { + QTestState *s; + +- s = qtest_init("-M pc-q35-5.2 -m 512M " ++ s = qtest_init("-M pc-q35-rhel9.4.0 -m 512M " + "-device virtio-scsi,num_queues=8,addr=03.0 "); + + qtest_outl(s, 0xcf8, 0x80001811); +diff --git a/tests/qtest/intel-hda-test.c b/tests/qtest/intel-hda-test.c +index 663bb6c485..2efc43e3f7 100644 +--- a/tests/qtest/intel-hda-test.c ++++ b/tests/qtest/intel-hda-test.c +@@ -42,7 +42,7 @@ static void test_issue542_ich6(void) + { + QTestState *s; + +- s = qtest_init("-nographic -nodefaults -M pc-q35-6.2 " ++ s = qtest_init("-nographic -nodefaults -M pc-q35-rhel9.0.0 " + AUDIODEV + "-device intel-hda,id=" HDA_ID CODEC_DEVICES); + +diff --git a/tests/qtest/libqos/meson.build b/tests/qtest/libqos/meson.build +index 1ddaf7b095..1cb403e90d 100644 +--- a/tests/qtest/libqos/meson.build ++++ b/tests/qtest/libqos/meson.build +@@ -43,7 +43,7 @@ libqos_srcs = files( + 'virtio-rng.c', + 'virtio-scsi.c', + 'virtio-serial.c', +- 'virtio-iommu.c', ++# 'virtio-iommu.c', + 'virtio-gpio.c', + 'virtio-scmi.c', + 'generic-pcihost.c', +diff --git a/tests/qtest/lpc-ich9-test.c b/tests/qtest/lpc-ich9-test.c +index 8ac95b89f7..0e118b76eb 100644 +--- a/tests/qtest/lpc-ich9-test.c ++++ b/tests/qtest/lpc-ich9-test.c +@@ -15,7 +15,7 @@ static void test_lp1878642_pci_bus_get_irq_level_assert(void) + { + QTestState *s; + +- s = qtest_init("-M pc-q35-5.0 " ++ s = qtest_init("-M pc-q35-rhel9.4.0 " + "-nographic -monitor none -serial none"); + + qtest_outl(s, 0xcf8, 0x8000f840); /* PMBASE */ +diff --git a/tests/qtest/machine-none-test.c b/tests/qtest/machine-none-test.c +index b6a87d27ed..423ba12159 100644 +--- a/tests/qtest/machine-none-test.c ++++ b/tests/qtest/machine-none-test.c +@@ -49,7 +49,7 @@ static struct arch2cpu cpus_map[] = { + { "xtensa", "dc233c" }, + { "xtensaeb", "fsf" }, + { "hppa", "hppa" }, +- { "riscv64", "rv64" }, ++ { "riscv64", "max" }, + { "riscv32", "rv32" }, + { "rx", "rx62n" }, + { "loongarch64", "la464"}, +diff --git a/tests/qtest/meson.build b/tests/qtest/meson.build +index b96aa06084..ef44ffaf78 100644 +--- a/tests/qtest/meson.build ++++ b/tests/qtest/meson.build +@@ -91,7 +91,6 @@ qtests_i386 = \ + (config_all_devices.has_key('CONFIG_LSI_SCSI_PCI') ? ['fuzz-lsi53c895a-test'] : []) + \ + (config_all_devices.has_key('CONFIG_VIRTIO_SCSI') ? ['fuzz-virtio-scsi-test'] : []) + \ + (config_all_devices.has_key('CONFIG_VIRTIO_BALLOON') ? ['virtio-balloon-test'] : []) + \ +- (config_all_devices.has_key('CONFIG_Q35') ? ['q35-test'] : []) + \ + (config_all_devices.has_key('CONFIG_SB16') ? ['fuzz-sb16-test'] : []) + \ + (config_all_devices.has_key('CONFIG_SDHCI_PCI') ? ['fuzz-sdcard-test'] : []) + \ + (config_all_devices.has_key('CONFIG_ESP_PCI') ? ['am53c974-test'] : []) + \ +diff --git a/tests/qtest/pvpanic-test.c b/tests/qtest/pvpanic-test.c +index 094c56b0cd..338f94dcd9 100644 +--- a/tests/qtest/pvpanic-test.c ++++ b/tests/qtest/pvpanic-test.c +@@ -65,7 +65,7 @@ static void test_pvshutdown(void) + QDict *response, *data; + QTestState *qts; + +- qts = qtest_init("-device pvpanic"); ++ qts = qtest_init("-M q35 -device pvpanic"); + + val = qtest_inb(qts, 0x505); + g_assert_cmpuint(val, ==, PVPANIC_EVENTS); +diff --git a/tests/qtest/riscv-csr-test.c b/tests/qtest/riscv-csr-test.c +index ff5c29e6c6..cc3b08a976 100644 +--- a/tests/qtest/riscv-csr-test.c ++++ b/tests/qtest/riscv-csr-test.c +@@ -20,6 +20,7 @@ + #define CSR_MVENDORID 0xf11 + #define CSR_MISELECT 0x350 + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + static void run_test_csr(void) + { + uint64_t res; +@@ -45,12 +46,15 @@ static void run_test_csr(void) + + qtest_quit(qts); + } ++#endif /* disabled for RHEL */ + + int main(int argc, char **argv) + { + g_test_init(&argc, &argv, NULL); + ++#if 0 /* Disabled for Red Hat Enterprise Linux */ + qtest_add_func("/cpu/csr", run_test_csr); ++#endif /* disabled for RHEL */ + + return g_test_run(); + } +diff --git a/tests/qtest/virtio-net-failover.c b/tests/qtest/virtio-net-failover.c +index 5baf81c3e6..aa87bf5698 100644 +--- a/tests/qtest/virtio-net-failover.c ++++ b/tests/qtest/virtio-net-failover.c +@@ -27,6 +27,7 @@ + #define PCI_SEL_BASE 0x0010 + + #define BASE_MACHINE "-M q35 -nodefaults " \ ++ "-global ICH9-LPC.acpi-pci-hotplug-with-bridge-support=on " \ + "-device pcie-root-port,id=root0,addr=0x1,bus=pcie.0,chassis=1 " \ + "-device pcie-root-port,id=root1,addr=0x2,bus=pcie.0,chassis=2 " + +-- +2.39.3 + diff --git a/0017-Disable-vga-cirrus-device.patch b/0017-Disable-vga-cirrus-device.patch deleted file mode 100644 index 3de3e10..0000000 --- a/0017-Disable-vga-cirrus-device.patch +++ /dev/null @@ -1,24 +0,0 @@ -From fe8c6cb1cecb3cde16871c4ec7368e4d004fa42a Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Fri, 26 Apr 2024 05:59:53 -0400 -Subject: Disable vga-cirrus device - ---- - configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 1 - - 1 file changed, 1 deletion(-) - -diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -index ee75bb4c21..fe69f04ead 100644 ---- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -+++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -@@ -87,7 +87,6 @@ CONFIG_USB_XHCI_PCI=y - CONFIG_VFIO=y - CONFIG_VFIO_PCI=y - CONFIG_VGA=y --CONFIG_VGA_CIRRUS=y - CONFIG_VGA_PCI=y - CONFIG_VHOST_USER=y - CONFIG_VHOST_USER_BLK=y --- -2.39.3 - diff --git a/0011-vfio-cap-number-of-devices-that-can-be-assigned.patch b/0017-vfio-cap-number-of-devices-that-can-be-assigned.patch similarity index 56% rename from 0011-vfio-cap-number-of-devices-that-can-be-assigned.patch rename to 0017-vfio-cap-number-of-devices-that-can-be-assigned.patch index bc52cd2..03adb71 100644 --- a/0011-vfio-cap-number-of-devices-that-can-be-assigned.patch +++ b/0017-vfio-cap-number-of-devices-that-can-be-assigned.patch @@ -1,8 +1,17 @@ -From 8ba1a6d1a432e2ae82ae532253c2b254e6ce82a7 Mon Sep 17 00:00:00 2001 +From 0f17fef61fc05f7617e47fcbc6a6c13efa5435d8 Mon Sep 17 00:00:00 2001 From: Bandan Das Date: Tue, 3 Dec 2013 20:05:13 +0100 Subject: vfio: cap number of devices that can be assigned +RH-Author: Bandan Das +Message-id: <1386101113-31560-3-git-send-email-bsd@redhat.com> +Patchwork-id: 55984 +O-Subject: [PATCH RHEL7 qemu-kvm v2 2/2] vfio: cap number of devices that can be assigned +Bugzilla: 678368 +RH-Acked-by: Alex Williamson +RH-Acked-by: Marcelo Tosatti +RH-Acked-by: Michael S. Tsirkin + Go through all groups to get count of total number of devices active to enforce limit @@ -17,16 +26,50 @@ Count of slots increased to 509 later so we could increase limit to 64 as some usecases require more than 32 devices. Signed-off-by: Bandan Das ---- - hw/vfio/pci.c | 31 ++++++++++++++++++++++++++++++- - hw/vfio/pci.h | 1 + - 2 files changed, 31 insertions(+), 1 deletion(-) +Rebase changes (8.2.0): +- Update to upstream changes + +Rebased notes (10.1.0) +- Update to upstream changes +- Introduced vfio_device_count() +--- + hw/vfio/container.c | 14 ++++++++++++++ + hw/vfio/pci.c | 21 +++++++++++++++++++++ + hw/vfio/pci.h | 1 + + include/hw/vfio/vfio-device.h | 1 + + 4 files changed, 37 insertions(+) + +diff --git a/hw/vfio/container.c b/hw/vfio/container.c +index 3e13feaa74..b912b9396b 100644 +--- a/hw/vfio/container.c ++++ b/hw/vfio/container.c +@@ -44,6 +44,20 @@ typedef QLIST_HEAD(VFIOGroupList, VFIOGroup) VFIOGroupList; + static VFIOGroupList vfio_group_list = + QLIST_HEAD_INITIALIZER(vfio_group_list); + ++int vfio_device_count(void) ++{ ++ int i = 0; ++ VFIOGroup *group; ++ VFIODevice *vbasedev_iter; ++ ++ QLIST_FOREACH(group, &vfio_group_list, next) { ++ QLIST_FOREACH(vbasedev_iter, &group->device_list, next) { ++ i++; ++ } ++ } ++ return i; ++} ++ + static int vfio_ram_block_discard_disable(VFIOContainer *container, bool state) + { + switch (container->iommu_type) { diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c -index 64780d1b79..57ac63c10c 100644 +index 07257d0fa0..48da233cb2 100644 --- a/hw/vfio/pci.c +++ b/hw/vfio/pci.c -@@ -50,6 +50,9 @@ +@@ -52,6 +52,9 @@ /* Protected by BQL */ static KVMRouteChange vfio_route_change; @@ -36,19 +79,9 @@ index 64780d1b79..57ac63c10c 100644 static void vfio_disable_interrupts(VFIOPCIDevice *vdev); static void vfio_mmap_set_enabled(VFIOPCIDevice *vdev, bool enabled); static void vfio_msi_disable_common(VFIOPCIDevice *vdev); -@@ -2946,13 +2949,36 @@ static void vfio_realize(PCIDevice *pdev, Error **errp) - ERRP_GUARD(); - VFIOPCIDevice *vdev = VFIO_PCI(pdev); - VFIODevice *vbasedev = &vdev->vbasedev; -+ VFIODevice *vbasedev_iter; -+ VFIOGroup *group; - char *tmp, *subsys; - Error *err = NULL; -- int i, ret; -+ int ret, i = 0; - bool is_mdev; +@@ -3355,6 +3358,21 @@ static void vfio_pci_realize(PCIDevice *pdev, Error **errp) char uuid[UUID_STR_LEN]; - char *name; + g_autofree char *name = NULL; + if (device_limit && device_limit != vdev->assigned_device_limit) { + error_setg(errp, "Assigned device limit has been redefined. " @@ -59,13 +92,7 @@ index 64780d1b79..57ac63c10c 100644 + device_limit = vdev->assigned_device_limit; + } + -+ QLIST_FOREACH(group, &vfio_group_list, next) { -+ QLIST_FOREACH(vbasedev_iter, &group->device_list, next) { -+ i++; -+ } -+ } -+ -+ if (i >= vdev->assigned_device_limit) { ++ if (vfio_device_count() >= vdev->assigned_device_limit) { + error_setg(errp, "Maximum supported vfio devices (%d) " + "already attached", vdev->assigned_device_limit); + return; @@ -74,7 +101,7 @@ index 64780d1b79..57ac63c10c 100644 if (vbasedev->fd < 0 && !vbasedev->sysfsdev) { if (!(~vdev->host.domain || ~vdev->host.bus || ~vdev->host.slot || ~vdev->host.function)) { -@@ -3370,6 +3396,9 @@ static Property vfio_pci_dev_properties[] = { +@@ -3687,6 +3705,9 @@ static const Property vfio_pci_dev_properties[] = { DEFINE_PROP_BOOL("x-no-kvm-msix", VFIOPCIDevice, no_kvm_msix, false), DEFINE_PROP_BOOL("x-no-geforce-quirks", VFIOPCIDevice, no_geforce_quirks, false), @@ -85,10 +112,10 @@ index 64780d1b79..57ac63c10c 100644 false), DEFINE_PROP_BOOL("x-no-vfio-ioeventfd", VFIOPCIDevice, no_vfio_ioeventfd, diff --git a/hw/vfio/pci.h b/hw/vfio/pci.h -index 6e64a2654e..b7de39c010 100644 +index 810a842f4a..81555d8774 100644 --- a/hw/vfio/pci.h +++ b/hw/vfio/pci.h -@@ -142,6 +142,7 @@ struct VFIOPCIDevice { +@@ -145,6 +145,7 @@ struct VFIOPCIDevice { EventNotifier err_notifier; EventNotifier req_notifier; int (*resetfn)(struct VFIOPCIDevice *); @@ -96,6 +123,18 @@ index 6e64a2654e..b7de39c010 100644 uint32_t vendor_id; uint32_t device_id; uint32_t sub_vendor_id; +diff --git a/include/hw/vfio/vfio-device.h b/include/hw/vfio/vfio-device.h +index 6e4d5ccdac..9290774299 100644 +--- a/include/hw/vfio/vfio-device.h ++++ b/include/hw/vfio/vfio-device.h +@@ -140,6 +140,7 @@ struct VFIODeviceOps { + #define strwriteerror(ret) \ + (ret < 0 ? strerror(-ret) : "short write") + ++int vfio_device_count(void); + void vfio_device_irq_disable(VFIODevice *vbasedev, int index); + void vfio_device_irq_unmask(VFIODevice *vbasedev, int index); + void vfio_device_irq_mask(VFIODevice *vbasedev, int index); -- 2.39.3 diff --git a/0012-Add-support-statement-to-help-output.patch b/0018-Add-support-statement-to-help-output.patch similarity index 85% rename from 0012-Add-support-statement-to-help-output.patch rename to 0018-Add-support-statement-to-help-output.patch index cac0eb7..884c4ce 100644 --- a/0012-Add-support-statement-to-help-output.patch +++ b/0018-Add-support-statement-to-help-output.patch @@ -1,4 +1,4 @@ -From 7bc7a2d39bb2c00bcc8e573f05e629f5f21edc35 Mon Sep 17 00:00:00 2001 +From 74a65ebea34ca36e3f510bc80274ce039df4e383 Mon Sep 17 00:00:00 2001 From: Eduardo Habkost Date: Wed, 4 Dec 2013 18:53:17 +0100 Subject: Add support statement to -help output @@ -12,10 +12,10 @@ Signed-off-by: Eduardo Habkost 1 file changed, 9 insertions(+) diff --git a/system/vl.c b/system/vl.c -index c644222982..03c3b0aa94 100644 +index 3b7057e6c6..d3e6158753 100644 --- a/system/vl.c +++ b/system/vl.c -@@ -869,9 +869,17 @@ static void version(void) +@@ -872,9 +872,17 @@ static void version(void) QEMU_COPYRIGHT "\n"); } @@ -33,7 +33,7 @@ index c644222982..03c3b0aa94 100644 printf("usage: %s [options] [disk_image]\n\n" "'disk_image' is a raw hard disk image for IDE hard disk 0\n\n", g_get_prgname()); -@@ -897,6 +905,7 @@ static void help(int exitcode) +@@ -900,6 +908,7 @@ static void help(int exitcode) "\n" QEMU_HELP_BOTTOM "\n"); diff --git a/0013-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch b/0019-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch similarity index 93% rename from 0013-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch rename to 0019-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch index b59920d..a7371c5 100644 --- a/0013-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch +++ b/0019-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch @@ -1,4 +1,4 @@ -From ec651d300d350a37219b09f5baab827ae6891006 Mon Sep 17 00:00:00 2001 +From 50b5abd584d9157677d694260a797be793fcf985 Mon Sep 17 00:00:00 2001 From: Miroslav Rezanina Date: Wed, 8 Jul 2020 08:35:50 +0200 Subject: Use qemu-kvm in documentation instead of qemu-system- @@ -27,10 +27,10 @@ index 52d6454b93..d74dbdeca9 100644 .. |I2C| replace:: I\ :sup:`2`\ C .. |I2S| replace:: I\ :sup:`2`\ S diff --git a/qemu-options.hx b/qemu-options.hx -index 8ce85d4559..4fc27ee2e2 100644 +index ab23f14d21..3837456a61 100644 --- a/qemu-options.hx +++ b/qemu-options.hx -@@ -3493,11 +3493,11 @@ SRST +@@ -3858,11 +3858,11 @@ SRST :: diff --git a/0014-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch b/0020-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch similarity index 94% rename from 0014-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch rename to 0020-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch index bc006b9..d264a5f 100644 --- a/0014-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch +++ b/0020-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch @@ -1,4 +1,4 @@ -From 080f22d8fb8ca63996f1b6ecb3637033529d8016 Mon Sep 17 00:00:00 2001 +From f1ec21d5adafcd06563a6c5404c5d631f470ab0c Mon Sep 17 00:00:00 2001 From: Kevin Wolf Date: Fri, 20 Aug 2021 18:25:12 +0200 Subject: qcow2: Deprecation warning when opening v2 images rw @@ -25,7 +25,7 @@ Signed-off-by: Kevin Wolf 2 files changed, 7 insertions(+) diff --git a/block/qcow2.c b/block/qcow2.c -index 956128b409..0e8b2f7518 100644 +index 4aa9f9e068..6df65aab93 100644 --- a/block/qcow2.c +++ b/block/qcow2.c @@ -1358,6 +1358,12 @@ qcow2_do_open(BlockDriverState *bs, QDict *options, int flags, @@ -42,7 +42,7 @@ index 956128b409..0e8b2f7518 100644 s->qcow_version = header.version; diff --git a/tests/qemu-iotests/common.filter b/tests/qemu-iotests/common.filter -index 2846c83808..83472953a2 100644 +index 511a55b1e8..35c0fc0d20 100644 --- a/tests/qemu-iotests/common.filter +++ b/tests/qemu-iotests/common.filter @@ -83,6 +83,7 @@ _filter_qemu() diff --git a/0021-file-posix-Define-DM_MPATH_PROBE_PATHS.patch b/0021-file-posix-Define-DM_MPATH_PROBE_PATHS.patch new file mode 100644 index 0000000..e329fe0 --- /dev/null +++ b/0021-file-posix-Define-DM_MPATH_PROBE_PATHS.patch @@ -0,0 +1,46 @@ +From 90041be5316257fc98eb62af1e8a927e53d2d612 Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Tue, 29 Apr 2025 17:05:41 +0200 +Subject: file-posix: Define DM_MPATH_PROBE_PATHS + +RH-Author: Kevin Wolf +RH-MergeRequest: 370: file-posix: Fix multipath failover with SCSI passthrough +RH-Jira: RHEL-65852 +RH-Acked-by: Hanna Czenczek +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [1/2] 6680f4a3f15a768b1ca51aafea452c3d0886f12e (kmwolf/centos-qemu-kvm) + +While the kernel side isn't merged yet and we're still using old kernel +headers, just define DM_MPATH_PROBE_PATHS manually. + +This is a downstream-only patch that can be removed after the next minor +release. + +Signed-off-by: Kevin Wolf + +Patch-name: kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch +Patch-id: 41 +Patch-present-in-specfile: True +--- + block/file-posix.c | 5 +++++ + 1 file changed, 5 insertions(+) + +diff --git a/block/file-posix.c b/block/file-posix.c +index 8c738674ce..cb2e94d7db 100644 +--- a/block/file-posix.c ++++ b/block/file-posix.c +@@ -156,6 +156,11 @@ + */ + #define SG_IO_MAX_RETRIES 8 + ++/* TODO Remove this when the kernel side is merged */ ++#if !defined(DM_MPATH_PROBE_PATHS) && defined(DM_GET_TARGET_VERSION) ++#define DM_MPATH_PROBE_PATHS _IO(DM_IOCTL, DM_GET_TARGET_VERSION_CMD + 1) ++#endif ++ + typedef struct BDRVRawState { + int fd; + bool use_lock; +-- +2.39.3 + diff --git a/kvm-Enable-vhost-user-scmi-devices.patch b/kvm-Enable-vhost-user-scmi-devices.patch deleted file mode 100644 index 20afbf2..0000000 --- a/kvm-Enable-vhost-user-scmi-devices.patch +++ /dev/null @@ -1,50 +0,0 @@ -From ca89f2eb9588bfebe2796a579a563bd974dadf72 Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Wed, 24 Jul 2024 07:31:12 -0400 -Subject: [PATCH] Enable vhost-user-scmi devices - -RH-Author: Miroslav Rezanina -RH-MergeRequest: 258: Enable vhost-user-scmi devices -RH-Jira: RHEL-50165 -RH-Acked-by: Sandro Bonazzola -RH-Commit: [1/1] edf95ef0fab99eb079beb16409fdab2a3cb0b94b (mrezanin/centos-src-qemu-kvm) - -Enabling vhost-user-scmi and vhost-user-scmi-pci devices for qemu-kvm. - -Signed-off-by: Miroslav Rezanina ---- - configs/devices/aarch64-softmmu/aarch64-rh-devices.mak | 1 + - configs/devices/s390x-softmmu/s390x-rh-devices.mak | 1 + - configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 1 + - 3 files changed, 3 insertions(+) - -diff --git a/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak -index 0a95438e25..4495d033e5 100644 ---- a/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak -+++ b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak -@@ -41,3 +41,4 @@ CONFIG_VHOST_USER_VSOCK=y - CONFIG_VHOST_USER_FS=y - CONFIG_IOMMUFD=y - CONFIG_VHOST_USER_SND=y -+CONFIG_VHOST_USER_SCMI=y -diff --git a/configs/devices/s390x-softmmu/s390x-rh-devices.mak b/configs/devices/s390x-softmmu/s390x-rh-devices.mak -index 719f802565..963ec43b6c 100644 ---- a/configs/devices/s390x-softmmu/s390x-rh-devices.mak -+++ b/configs/devices/s390x-softmmu/s390x-rh-devices.mak -@@ -18,3 +18,4 @@ CONFIG_VHOST_USER_VSOCK=y - CONFIG_VHOST_USER_FS=y - CONFIG_IOMMUFD=y - CONFIG_VHOST_USER_SND=y -+CONFIG_VHOST_USER_SCMI=y -diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -index b85bb1fe53..276397f3be 100644 ---- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -+++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -@@ -110,3 +110,4 @@ CONFIG_VHOST_USER_VSOCK=y - CONFIG_VHOST_USER_FS=y - CONFIG_IOMMUFD=y - CONFIG_VHOST_USER_SND=y -+CONFIG_VHOST_USER_SCMI=y --- -2.39.3 - diff --git a/kvm-Enable-vhost-user-snd-pci-device.patch b/kvm-Enable-vhost-user-snd-pci-device.patch deleted file mode 100644 index fc05aa4..0000000 --- a/kvm-Enable-vhost-user-snd-pci-device.patch +++ /dev/null @@ -1,50 +0,0 @@ -From d7256c0d15a3ae142c80462c66e0d68120ebd001 Mon Sep 17 00:00:00 2001 -From: Miroslav Rezanina -Date: Wed, 22 May 2024 03:56:55 -0400 -Subject: [PATCH] Enable vhost-user-snd-pci device - -RH-Author: Miroslav Rezanina -RH-MergeRequest: 242: Enable vhost-user-snd-pci device -RH-Jira: RHEL-37563 -RH-Acked-by: Sandro Bonazzola -RH-Commit: [1/1] 014f47770fc9f7d4bd0e7fac9a072911325f3283 (mrezanin/centos-src-qemu-kvm) - -RHIVOS requires vhost-user-snd-pci device. Enabling it for aarch64 and x86_64 only. - -Signed-off-by: Miroslav Rezanina ---- - configs/devices/aarch64-softmmu/aarch64-rh-devices.mak | 1 + - configs/devices/s390x-softmmu/s390x-rh-devices.mak | 1 + - configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 1 + - 3 files changed, 3 insertions(+) - -diff --git a/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak -index b0191d3c69..0a95438e25 100644 ---- a/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak -+++ b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak -@@ -40,3 +40,4 @@ CONFIG_VHOST_VSOCK=y - CONFIG_VHOST_USER_VSOCK=y - CONFIG_VHOST_USER_FS=y - CONFIG_IOMMUFD=y -+CONFIG_VHOST_USER_SND=y -diff --git a/configs/devices/s390x-softmmu/s390x-rh-devices.mak b/configs/devices/s390x-softmmu/s390x-rh-devices.mak -index 24cf6dbd03..719f802565 100644 ---- a/configs/devices/s390x-softmmu/s390x-rh-devices.mak -+++ b/configs/devices/s390x-softmmu/s390x-rh-devices.mak -@@ -17,3 +17,4 @@ CONFIG_VHOST_VSOCK=y - CONFIG_VHOST_USER_VSOCK=y - CONFIG_VHOST_USER_FS=y - CONFIG_IOMMUFD=y -+CONFIG_VHOST_USER_SND=y -diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -index fe69f04ead..b85bb1fe53 100644 ---- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -+++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak -@@ -109,3 +109,4 @@ CONFIG_VHOST_VSOCK=y - CONFIG_VHOST_USER_VSOCK=y - CONFIG_VHOST_USER_FS=y - CONFIG_IOMMUFD=y -+CONFIG_VHOST_USER_SND=y --- -2.39.3 - diff --git a/kvm-Fix-the-typo-of-vfio-pci-device-s-enable-migration-o.patch b/kvm-Fix-the-typo-of-vfio-pci-device-s-enable-migration-o.patch new file mode 100644 index 0000000..2416ca0 --- /dev/null +++ b/kvm-Fix-the-typo-of-vfio-pci-device-s-enable-migration-o.patch @@ -0,0 +1,37 @@ +From 0509eb94d2f166cfd3a24aa5fd14e1af76af8dea Mon Sep 17 00:00:00 2001 +From: Yanghang Liu +Date: Mon, 24 Nov 2025 23:02:37 +0800 +Subject: [PATCH 4/4] Fix the typo of vfio-pci device's enable-migration option + +RH-Author: YangHang Liu +RH-MergeRequest: 428: RHEL10: Fix the typo of vfio-pci device's enable-migration option +RH-Jira: RHEL-130704 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/1] bcf8b2089f687d02b5f0a48dd1e3dfc4664946e0 (yanghliu/qemu-kvm) + +Signed-off-by: Yanghang Liu +Reported-by: Mario Casquero +Reviewed-by: Michael Tokarev +Signed-off-by: Michael Tokarev +(cherry picked from commit 5f9ac963735598c1efbdce9c4a09f0a64e13d613) +Signed-off-by: Yanghang Liu +--- + hw/vfio/pci.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c +index 83ecffb535..7c057ee2f9 100644 +--- a/hw/vfio/pci.c ++++ b/hw/vfio/pci.c +@@ -3847,7 +3847,7 @@ static void vfio_pci_dev_class_init(ObjectClass *klass, const void *data) + "(DEBUG)"); + object_class_property_set_description(klass, /* 5.2, 8.0 non-experimetal */ + "enable-migration", +- "Enale device migration. Also requires a host VFIO PCI " ++ "Enable device migration. Also requires a host VFIO PCI " + "variant or mdev driver with migration support enabled"); + object_class_property_set_description(klass, /* 8.1 */ + "vf-token", +-- +2.47.3 + diff --git a/kvm-MAINTAINERS-Add-maintainers-for-mshv-accelerator.patch b/kvm-MAINTAINERS-Add-maintainers-for-mshv-accelerator.patch new file mode 100644 index 0000000..0db0143 --- /dev/null +++ b/kvm-MAINTAINERS-Add-maintainers-for-mshv-accelerator.patch @@ -0,0 +1,54 @@ +From 5807bd68ad5c038a57ffb2f665f732dfc51d8868 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:47 +0200 +Subject: [PATCH 30/32] MAINTAINERS: Add maintainers for mshv accelerator + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [28/30] 0178a9a61285bd708b5f589888c36b74efd4a4bf + +Adding Magnus Kulke and Wei Liu to the maintainers file for the +respective folders/files. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-28-magnuskulke@linux.microsoft.com +[Rename "MAHV CPUs" to mention x86. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 1872bd9f2dad5113b0c27d352ec49683e0953c7f) +Signed-off-by: Igor Mammedov +--- + MAINTAINERS | 15 +++++++++++++++ + 1 file changed, 15 insertions(+) + +diff --git a/MAINTAINERS b/MAINTAINERS +index a07086ed76..d4696f00d7 100644 +--- a/MAINTAINERS ++++ b/MAINTAINERS +@@ -546,6 +546,21 @@ F: target/i386/whpx/ + F: accel/stubs/whpx-stub.c + F: include/system/whpx.h + ++MSHV ++M: Magnus Kulke ++R: Wei Liu ++S: Supported ++F: accel/mshv/ ++F: include/system/mshv.h ++F: include/hw/hyperv/hvgdk*.h ++F: include/hw/hyperv/hvhdk*.h ++ ++X86 MSHV CPUs ++M: Magnus Kulke ++R: Wei Liu ++S: Supported ++F: target/i386/mshv/ ++ + X86 Instruction Emulator + M: Cameron Esfahani + M: Roman Bolshakov +-- +2.47.3 + diff --git a/kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch b/kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch new file mode 100644 index 0000000..cd7c009 --- /dev/null +++ b/kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch @@ -0,0 +1,81 @@ +From eee1f8abab9cbcb64ab690737f1a8db293d87c05 Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Wed, 11 Feb 2026 09:57:37 -0500 +Subject: [PATCH 7/7] Revert "hw/arm/virt: Use ACPI PCI hotplug by default from + 10.2 onwards" + +RH-Author: Eric Auger +RH-MergeRequest: 465: Revert "hw/arm/virt: Use ACPI PCI hotplug by default from 10.2 onwards" +RH-Jira: RHEL-134989 RHEL-146584 +RH-Acked-by: Sebastian Ott +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Cornelia Huck +RH-Commit: [1/1] e22612dc81813762f8d7f4bc9f75df0b9f2135e0 (eauger1/centos-qemu-kvm) + +JIRA: https://issues.redhat.com/browse/RHEL-134989 +JIRA: https://issues.redhat.com/browse/RHEL-146584 +UPSTREAM: RHEL-only + +This reverts commit 58cba97a715fa3f506234e718191fcc34286f333. + +Conflicts: small contextual conflict when reverting changes in +hw/arm/virt.c due to subsequent fix by commit +fa0a758781fc ("arm: fix oob access in compat handling") + +Unfortunately the change of the default for the PCI hotplug method +introduced some regressions that cannot be fixed in 10.2 cycle. An +example is hotplugging a virtio-net-pci device with page-per-vq=true. +This induces an increase in the BAR size which is larger than the +default size the FW accomodates. At the moment we do not have any +workaround for those devices with large BARs, ie. we noticed +pcie-root-port pref64-reserve does not work as on x86 and we do not +have any way to opt-in for legacy PCIe hotplug at libvirt +level. So let's revert the change until we get all those stuff +properly fixed. + +Signed-off-by: Eric Auger +--- + hw/arm/virt.c | 10 ---------- + 1 file changed, 10 deletions(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index 1cfb386f64..752dc08720 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -95,16 +95,9 @@ + + static GlobalProperty arm_virt_compat[] = { + { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "48" }, +- { TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "on" }, + }; + static const size_t arm_virt_compat_len = G_N_ELEMENTS(arm_virt_compat); + +-GlobalProperty arm_acpi_pci_hp_disabled_compat[] = { +- { TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "off" }, +-}; +-static const size_t arm_acpi_pci_hp_disabled_compat_len = +- G_N_ELEMENTS(arm_acpi_pci_hp_disabled_compat); +- + /* + * RHEL9 kernels have pauth disabled while RHEL10 has it enabled, + * since qemu will setup the VM with pauth when KVM supports it we +@@ -112,7 +105,6 @@ static const size_t arm_acpi_pci_hp_disabled_compat_len = + */ + GlobalProperty arm_rhel9_compat[] = { + {TYPE_ARM_CPU, "pauth", "off", .optional = true}, +- {TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "off" }, + }; + const size_t arm_rhel9_compat_len = G_N_ELEMENTS(arm_rhel9_compat); + +@@ -3768,8 +3760,6 @@ static void virt_rhel_machine_10_0_0_options(MachineClass *mc) + + /* QEMU 9.1 and earlier have only a stage-1 SMMU, not a nested s1+2 one */ + vmc->no_nested_smmu = true; +- compat_props_add(mc->compat_props, arm_acpi_pci_hp_disabled_compat, +- arm_acpi_pci_hp_disabled_compat_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_2, hw_compat_rhel_10_2_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_1, hw_compat_rhel_10_1_len); + } +-- +2.47.3 + diff --git a/kvm-Revert-monitor-use-aio_co_reschedule_self.patch b/kvm-Revert-monitor-use-aio_co_reschedule_self.patch deleted file mode 100644 index c0dcc12..0000000 --- a/kvm-Revert-monitor-use-aio_co_reschedule_self.patch +++ /dev/null @@ -1,67 +0,0 @@ -From 53cc7daf2b6356f236a493cbe63d01afc5636fd3 Mon Sep 17 00:00:00 2001 -From: Stefan Hajnoczi -Date: Mon, 6 May 2024 15:06:21 -0400 -Subject: [PATCH 13/14] Revert "monitor: use aio_co_reschedule_self()" - -RH-Author: Kevin Wolf -RH-MergeRequest: 253: Revert "monitor: use aio_co_reschedule_self()" -RH-Jira: RHEL-43409 RHEL-43410 -RH-Acked-by: Miroslav Rezanina -RH-Acked-by: Hanna Czenczek -RH-Commit: [1/2] 772eccc9da09e6c1793d46ab6cf9ee6615812154 (kmwolf/centos-qemu-kvm) - -Commit 1f25c172f837 ("monitor: use aio_co_reschedule_self()") was a code -cleanup that uses aio_co_reschedule_self() instead of open coding -coroutine rescheduling. - -Bug RHEL-34618 was reported and Kevin Wolf identified -the root cause. I missed that aio_co_reschedule_self() -> -qemu_get_current_aio_context() only knows about -qemu_aio_context/IOThread AioContexts and not about iohandler_ctx. It -does not function correctly when going back from the iohandler_ctx to -qemu_aio_context. - -Go back to open coding the AioContext transitions to avoid this bug. - -This reverts commit 1f25c172f83704e350c0829438d832384084a74d. - -Cc: qemu-stable@nongnu.org -Buglink: https://issues.redhat.com/browse/RHEL-34618 -Signed-off-by: Stefan Hajnoczi -Message-ID: <20240506190622.56095-2-stefanha@redhat.com> -Reviewed-by: Kevin Wolf -Signed-off-by: Kevin Wolf -(cherry picked from commit 719c6819ed9a9838520fa732f9861918dc693bda) -Signed-off-by: Kevin Wolf ---- - qapi/qmp-dispatch.c | 7 +++++-- - 1 file changed, 5 insertions(+), 2 deletions(-) - -diff --git a/qapi/qmp-dispatch.c b/qapi/qmp-dispatch.c -index f3488afeef..176b549473 100644 ---- a/qapi/qmp-dispatch.c -+++ b/qapi/qmp-dispatch.c -@@ -212,7 +212,8 @@ QDict *coroutine_mixed_fn qmp_dispatch(const QmpCommandList *cmds, QObject *requ - * executing the command handler so that it can make progress if it - * involves an AIO_WAIT_WHILE(). - */ -- aio_co_reschedule_self(qemu_get_aio_context()); -+ aio_co_schedule(qemu_get_aio_context(), qemu_coroutine_self()); -+ qemu_coroutine_yield(); - } - - monitor_set_cur(qemu_coroutine_self(), cur_mon); -@@ -226,7 +227,9 @@ QDict *coroutine_mixed_fn qmp_dispatch(const QmpCommandList *cmds, QObject *requ - * Move back to iohandler_ctx so that nested event loops for - * qemu_aio_context don't start new monitor commands. - */ -- aio_co_reschedule_self(iohandler_get_aio_context()); -+ aio_co_schedule(iohandler_get_aio_context(), -+ qemu_coroutine_self()); -+ qemu_coroutine_yield(); - } - } else { - /* --- -2.39.3 - diff --git a/kvm-accel-Add-Meson-and-config-support-for-MSHV-accelera.patch b/kvm-accel-Add-Meson-and-config-support-for-MSHV-accelera.patch new file mode 100644 index 0000000..3c906b7 --- /dev/null +++ b/kvm-accel-Add-Meson-and-config-support-for-MSHV-accelera.patch @@ -0,0 +1,131 @@ +From 502e461dabe1eecc22f6a7d22755b8407ee74683 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:21 +0200 +Subject: [PATCH 03/32] accel: Add Meson and config support for MSHV + accelerator +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/30] 04ab9685b09f4d837715bf624d87566f472cb6c2 + +Introduce a Meson feature option and default-config entry to allow +building QEMU with MSHV (Microsoft Hypervisor) acceleration support. + +This is the first step toward implementing an MSHV backend in QEMU. + +Signed-off-by: Magnus Kulke +Reviewed-by: Daniel P. Berrangé +Link: https://lore.kernel.org/r/20250916164847.77883-2-magnuskulke@linux.microsoft.com +[Add error for unavailable accelerator. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 37e12da5df8eb74042f11e9e7bec8a50b8090adb) +Signed-off-by: Igor Mammedov +--- + accel/Kconfig | 3 +++ + meson.build | 13 +++++++++++++ + meson_options.txt | 2 ++ + scripts/meson-buildoptions.sh | 3 +++ + 4 files changed, 21 insertions(+) + +diff --git a/accel/Kconfig b/accel/Kconfig +index 4263cab722..a60f114923 100644 +--- a/accel/Kconfig ++++ b/accel/Kconfig +@@ -13,6 +13,9 @@ config TCG + config KVM + bool + ++config MSHV ++ bool ++ + config XEN + bool + select FSDEV_9P if VIRTFS +diff --git a/meson.build b/meson.build +index ef2e5be6e2..96254f8075 100644 +--- a/meson.build ++++ b/meson.build +@@ -334,6 +334,7 @@ elif cpu == 'x86_64' + 'CONFIG_HVF': ['x86_64-softmmu'], + 'CONFIG_NVMM': ['i386-softmmu', 'x86_64-softmmu'], + 'CONFIG_WHPX': ['i386-softmmu', 'x86_64-softmmu'], ++ 'CONFIG_MSHV': ['x86_64-softmmu'], + } + endif + +@@ -884,6 +885,14 @@ accelerators = [] + if get_option('kvm').allowed() and host_os == 'linux' + accelerators += 'CONFIG_KVM' + endif ++ ++if get_option('mshv').allowed() and host_os == 'linux' ++ if get_option('mshv').enabled() and host_machine.cpu() != 'x86_64' ++ error('mshv accelerator requires x64_64 host') ++ endif ++ accelerators += 'CONFIG_MSHV' ++endif ++ + if get_option('whpx').allowed() and host_os == 'windows' + if get_option('whpx').enabled() and host_machine.cpu() != 'x86_64' + error('WHPX requires 64-bit host') +@@ -953,6 +962,9 @@ endif + if 'CONFIG_WHPX' not in accelerators and get_option('whpx').enabled() + error('WHPX not available on this platform') + endif ++if 'CONFIG_MSHV' not in accelerators and get_option('mshv').enabled() ++ error('mshv not available on this platform') ++endif + + xen = not_found + if get_option('xen').enabled() or (get_option('xen').auto() and have_system) +@@ -4821,6 +4833,7 @@ if have_system + summary_info += {'HVF support': config_all_accel.has_key('CONFIG_HVF')} + summary_info += {'WHPX support': config_all_accel.has_key('CONFIG_WHPX')} + summary_info += {'NVMM support': config_all_accel.has_key('CONFIG_NVMM')} ++ summary_info += {'MSHV support': config_all_accel.has_key('CONFIG_MSHV')} + summary_info += {'Xen support': xen.found()} + if xen.found() + summary_info += {'xen ctrl version': xen.version()} +diff --git a/meson_options.txt b/meson_options.txt +index f45d7ded45..2267dee8a0 100644 +--- a/meson_options.txt ++++ b/meson_options.txt +@@ -73,6 +73,8 @@ option('malloc', type : 'combo', choices : ['system', 'tcmalloc', 'jemalloc'], + + option('kvm', type: 'feature', value: 'auto', + description: 'KVM acceleration support') ++option('mshv', type: 'feature', value: 'auto', ++ description: 'MSHV acceleration support') + option('whpx', type: 'feature', value: 'auto', + description: 'WHPX acceleration support') + option('hvf', type: 'feature', value: 'auto', +diff --git a/scripts/meson-buildoptions.sh b/scripts/meson-buildoptions.sh +index 4146dbc88d..9aff126c28 100644 +--- a/scripts/meson-buildoptions.sh ++++ b/scripts/meson-buildoptions.sh +@@ -155,6 +155,7 @@ meson_options_help() { + printf "%s\n" ' membarrier membarrier system call (for Linux 4.14+ or Windows' + printf "%s\n" ' modules modules support (non Windows)' + printf "%s\n" ' mpath Multipath persistent reservation passthrough' ++ printf "%s\n" ' mshv MSHV acceleration support' + printf "%s\n" ' multiprocess Out of process device emulation support' + printf "%s\n" ' netmap netmap network backend support' + printf "%s\n" ' nettle nettle cryptography support' +@@ -409,6 +410,8 @@ _meson_option_parse() { + --disable-modules) printf "%s" -Dmodules=disabled ;; + --enable-mpath) printf "%s" -Dmpath=enabled ;; + --disable-mpath) printf "%s" -Dmpath=disabled ;; ++ --enable-mshv) printf "%s" -Dmshv=enabled ;; ++ --disable-mshv) printf "%s" -Dmshv=disabled ;; + --enable-multiprocess) printf "%s" -Dmultiprocess=enabled ;; + --disable-multiprocess) printf "%s" -Dmultiprocess=disabled ;; + --enable-netmap) printf "%s" -Dnetmap=enabled ;; +-- +2.47.3 + diff --git a/kvm-accel-mshv-Add-accelerator-skeleton.patch b/kvm-accel-mshv-Add-accelerator-skeleton.patch new file mode 100644 index 0000000..e14da46 --- /dev/null +++ b/kvm-accel-mshv-Add-accelerator-skeleton.patch @@ -0,0 +1,291 @@ +From a491ddf980ed1c8a54ba06db007d770817554b91 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Thu, 2 Oct 2025 18:25:02 +0200 +Subject: [PATCH 09/32] accel/mshv: Add accelerator skeleton + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [7/30] 2bf55db3c9fbc6ce5964090f192be3699ae2ea00 + +Introduce the initial scaffold for the MSHV (Microsoft Hypervisor) +accelerator backend. This includes the basic directory structure and +stub implementations needed to integrate with QEMU's accelerator +framework. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-8-magnuskulke@linux.microsoft.com +[Move include of linux/mshv.h in the per-target section; create + include/system/mshv_int.h. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit d0d2918f968c55628e17e2733b799fcefb50f16b) +Signed-off-by: Igor Mammedov +--- + accel/meson.build | 1 + + accel/mshv/meson.build | 6 ++ + accel/mshv/mshv-all.c | 144 ++++++++++++++++++++++++++++++++++++++ + include/system/mshv.h | 12 ++++ + include/system/mshv_int.h | 41 +++++++++++ + 5 files changed, 204 insertions(+) + create mode 100644 accel/mshv/meson.build + create mode 100644 accel/mshv/mshv-all.c + create mode 100644 include/system/mshv_int.h + +diff --git a/accel/meson.build b/accel/meson.build +index 6349efe682..983dfd0bd5 100644 +--- a/accel/meson.build ++++ b/accel/meson.build +@@ -10,6 +10,7 @@ if have_system + subdir('kvm') + subdir('xen') + subdir('stubs') ++ subdir('mshv') + endif + + # qtest +diff --git a/accel/mshv/meson.build b/accel/mshv/meson.build +new file mode 100644 +index 0000000000..4c03ac7921 +--- /dev/null ++++ b/accel/mshv/meson.build +@@ -0,0 +1,6 @@ ++mshv_ss = ss.source_set() ++mshv_ss.add(if_true: files( ++ 'mshv-all.c' ++)) ++ ++specific_ss.add_all(when: 'CONFIG_MSHV', if_true: mshv_ss) +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +new file mode 100644 +index 0000000000..ae12f0f58b +--- /dev/null ++++ b/accel/mshv/mshv-all.c +@@ -0,0 +1,144 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: ++ * Ziqiao Zhou ++ * Magnus Kulke ++ * Jinank Jain ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ * ++ */ ++ ++#include "qemu/osdep.h" ++#include "qapi/error.h" ++#include "qemu/error-report.h" ++#include "qemu/event_notifier.h" ++#include "qemu/module.h" ++#include "qemu/main-loop.h" ++#include "hw/boards.h" ++ ++#include "hw/hyperv/hvhdk.h" ++#include "hw/hyperv/hvhdk_mini.h" ++#include "hw/hyperv/hvgdk.h" ++#include "linux/mshv.h" ++ ++#include "qemu/accel.h" ++#include "qemu/guest-random.h" ++#include "accel/accel-ops.h" ++#include "accel/accel-cpu-ops.h" ++#include "system/cpus.h" ++#include "system/runstate.h" ++#include "system/accel-blocker.h" ++#include "system/address-spaces.h" ++#include "system/mshv.h" ++#include "system/mshv_int.h" ++#include "system/reset.h" ++#include "trace.h" ++#include ++#include ++#include ++ ++#define TYPE_MSHV_ACCEL ACCEL_CLASS_NAME("mshv") ++ ++DECLARE_INSTANCE_CHECKER(MshvState, MSHV_STATE, TYPE_MSHV_ACCEL) ++ ++bool mshv_allowed; ++ ++MshvState *mshv_state; ++ ++static int mshv_init(AccelState *as, MachineState *ms) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++static void mshv_start_vcpu_thread(CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++static void mshv_cpu_synchronize_post_init(CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++static void mshv_cpu_synchronize_post_reset(CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++static void mshv_cpu_synchronize_pre_loadvm(CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++static void mshv_cpu_synchronize(CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++static bool mshv_cpus_are_resettable(void) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++static void mshv_accel_class_init(ObjectClass *oc, const void *data) ++{ ++ AccelClass *ac = ACCEL_CLASS(oc); ++ ++ ac->name = "MSHV"; ++ ac->init_machine = mshv_init; ++ ac->allowed = &mshv_allowed; ++} ++ ++static void mshv_accel_instance_init(Object *obj) ++{ ++ MshvState *s = MSHV_STATE(obj); ++ ++ s->vm = 0; ++} ++ ++static const TypeInfo mshv_accel_type = { ++ .name = TYPE_MSHV_ACCEL, ++ .parent = TYPE_ACCEL, ++ .instance_init = mshv_accel_instance_init, ++ .class_init = mshv_accel_class_init, ++ .instance_size = sizeof(MshvState), ++}; ++ ++static void mshv_accel_ops_class_init(ObjectClass *oc, const void *data) ++{ ++ AccelOpsClass *ops = ACCEL_OPS_CLASS(oc); ++ ++ ops->create_vcpu_thread = mshv_start_vcpu_thread; ++ ops->synchronize_post_init = mshv_cpu_synchronize_post_init; ++ ops->synchronize_post_reset = mshv_cpu_synchronize_post_reset; ++ ops->synchronize_state = mshv_cpu_synchronize; ++ ops->synchronize_pre_loadvm = mshv_cpu_synchronize_pre_loadvm; ++ ops->cpus_are_resettable = mshv_cpus_are_resettable; ++ ops->handle_interrupt = generic_handle_interrupt; ++} ++ ++static const TypeInfo mshv_accel_ops_type = { ++ .name = ACCEL_OPS_NAME("mshv"), ++ .parent = TYPE_ACCEL_OPS, ++ .class_init = mshv_accel_ops_class_init, ++ .abstract = true, ++}; ++ ++static void mshv_type_init(void) ++{ ++ type_register_static(&mshv_accel_type); ++ type_register_static(&mshv_accel_ops_type); ++} ++ ++type_init(mshv_type_init); +diff --git a/include/system/mshv.h b/include/system/mshv.h +index 2a504ed81f..434ea9682e 100644 +--- a/include/system/mshv.h ++++ b/include/system/mshv.h +@@ -14,8 +14,17 @@ + #ifndef QEMU_MSHV_H + #define QEMU_MSHV_H + ++#include "qemu/osdep.h" ++#include "qemu/accel.h" ++#include "hw/hyperv/hyperv-proto.h" ++#include "hw/hyperv/hvhdk.h" ++#include "qapi/qapi-types-common.h" ++#include "system/memory.h" ++#include "accel/accel-ops.h" ++ + #ifdef COMPILING_PER_TARGET + #ifdef CONFIG_MSHV ++#include + #define CONFIG_MSHV_IS_POSSIBLE + #endif + #else +@@ -30,6 +39,9 @@ extern bool mshv_allowed; + #endif + #define mshv_msi_via_irqfd_enabled() false + ++typedef struct MshvState MshvState; ++extern MshvState *mshv_state; ++ + /* interrupt */ + int mshv_irqchip_add_msi_route(int vector, PCIDevice *dev); + int mshv_irqchip_update_msi_route(int virq, MSIMessage msg, PCIDevice *dev); +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +new file mode 100644 +index 0000000000..132491b599 +--- /dev/null ++++ b/include/system/mshv_int.h +@@ -0,0 +1,41 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Ziqiao Zhou ++ * Magnus Kulke ++ * Jinank Jain ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ * ++ */ ++ ++#ifndef QEMU_MSHV_INT_H ++#define QEMU_MSHV_INT_H ++ ++struct AccelCPUState { ++ int cpufd; ++ bool dirty; ++}; ++ ++typedef struct MshvMemoryListener { ++ MemoryListener listener; ++ int as_id; ++} MshvMemoryListener; ++ ++typedef struct MshvAddressSpace { ++ MshvMemoryListener *ml; ++ AddressSpace *as; ++} MshvAddressSpace; ++ ++struct MshvState { ++ AccelState parent_obj; ++ int vm; ++ MshvMemoryListener memory_listener; ++ /* number of listeners */ ++ int nr_as; ++ MshvAddressSpace *as; ++}; ++ ++#endif +-- +2.47.3 + diff --git a/kvm-accel-mshv-Add-vCPU-creation-and-execution-loop.patch b/kvm-accel-mshv-Add-vCPU-creation-and-execution-loop.patch new file mode 100644 index 0000000..a74a8cb --- /dev/null +++ b/kvm-accel-mshv-Add-vCPU-creation-and-execution-loop.patch @@ -0,0 +1,417 @@ +From da65137c39b3662f0288b5149b14c82d2cfa4000 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:30 +0200 +Subject: [PATCH 13/32] accel/mshv: Add vCPU creation and execution loop + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [11/30] e235ebda329666192a3bf95fb4fee018f5f5f10e + +Create MSHV vCPUs using MSHV_CREATE_VP and initialize their state. +Register the MSHV CPU execution loop loop with the QEMU accelerator +framework to enable guest code execution. + +The target/i386 functionality is still mostly stubbed out and will be +populated in a later commit in this series. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-11-magnuskulke@linux.microsoft.com +[Fix g_free/g_clear_pointer confusion; rename qemu_wait_io_event; + mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 4dc5d4257259764b6fcd035870517fc4140d8962) +Signed-off-by: Igor Mammedov +--- + accel/mshv/mshv-all.c | 186 +++++++++++++++++++++++++++++++++--- + accel/mshv/trace-events | 2 + + include/system/mshv.h | 2 +- + include/system/mshv_int.h | 20 ++++ + target/i386/mshv/mshv-cpu.c | 63 ++++++++++++ + 5 files changed, 260 insertions(+), 13 deletions(-) + +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +index 653195c57c..e02421d79d 100644 +--- a/accel/mshv/mshv-all.c ++++ b/accel/mshv/mshv-all.c +@@ -393,6 +393,24 @@ int mshv_hvcall(int fd, const struct mshv_root_hvcall *args) + return ret; + } + ++static int mshv_init_vcpu(CPUState *cpu) ++{ ++ int vm_fd = mshv_state->vm; ++ uint8_t vp_index = cpu->cpu_index; ++ int ret; ++ ++ mshv_arch_init_vcpu(cpu); ++ cpu->accel = g_new0(AccelCPUState, 1); ++ ++ ret = mshv_create_vcpu(vm_fd, vp_index, &cpu->accel->cpufd); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ cpu->accel->dirty = true; ++ ++ return 0; ++} + + static int mshv_init(AccelState *as, MachineState *ms) + { +@@ -415,6 +433,8 @@ static int mshv_init(AccelState *as, MachineState *ms) + return -1; + } + ++ mshv_init_mmio_emu(); ++ + mshv_init_msicontrol(); + + ret = create_vm(mshv_fd, &vm_fd); +@@ -444,40 +464,182 @@ static int mshv_init(AccelState *as, MachineState *ms) + return 0; + } + ++static int mshv_destroy_vcpu(CPUState *cpu) ++{ ++ int cpu_fd = mshv_vcpufd(cpu); ++ int vm_fd = mshv_state->vm; ++ ++ mshv_remove_vcpu(vm_fd, cpu_fd); ++ mshv_vcpufd(cpu) = 0; ++ ++ mshv_arch_destroy_vcpu(cpu); ++ g_clear_pointer(&cpu->accel, g_free); ++ return 0; ++} ++ ++static int mshv_cpu_exec(CPUState *cpu) ++{ ++ hv_message mshv_msg; ++ enum MshvVmExit exit_reason; ++ int ret = 0; ++ ++ bql_unlock(); ++ cpu_exec_start(cpu); ++ ++ do { ++ if (cpu->accel->dirty) { ++ ret = mshv_arch_put_registers(cpu); ++ if (ret) { ++ error_report("Failed to put registers after init: %s", ++ strerror(-ret)); ++ ret = -1; ++ break; ++ } ++ cpu->accel->dirty = false; ++ } ++ ++ ret = mshv_run_vcpu(mshv_state->vm, cpu, &mshv_msg, &exit_reason); ++ if (ret < 0) { ++ error_report("Failed to run on vcpu %d", cpu->cpu_index); ++ abort(); ++ } ++ ++ switch (exit_reason) { ++ case MshvVmExitIgnore: ++ break; ++ default: ++ ret = EXCP_INTERRUPT; ++ break; ++ } ++ } while (ret == 0); ++ ++ cpu_exec_end(cpu); ++ bql_lock(); ++ ++ if (ret < 0) { ++ cpu_dump_state(cpu, stderr, CPU_DUMP_CODE); ++ vm_stop(RUN_STATE_INTERNAL_ERROR); ++ } ++ ++ return ret; ++} ++ ++static void *mshv_vcpu_thread(void *arg) ++{ ++ CPUState *cpu = arg; ++ int ret; ++ ++ rcu_register_thread(); ++ ++ bql_lock(); ++ qemu_thread_get_self(cpu->thread); ++ cpu->thread_id = qemu_get_thread_id(); ++ current_cpu = cpu; ++ ret = mshv_init_vcpu(cpu); ++ if (ret < 0) { ++ error_report("Failed to init vcpu %d", cpu->cpu_index); ++ goto cleanup; ++ } ++ ++ /* signal CPU creation */ ++ cpu_thread_signal_created(cpu); ++ qemu_guest_random_seed_thread_part2(cpu->random_seed); ++ ++ do { ++ qemu_process_cpu_events(cpu); ++ if (cpu_can_run(cpu)) { ++ mshv_cpu_exec(cpu); ++ } ++ } while (!cpu->unplug || cpu_can_run(cpu)); ++ ++ mshv_destroy_vcpu(cpu); ++cleanup: ++ cpu_thread_signal_destroyed(cpu); ++ bql_unlock(); ++ rcu_unregister_thread(); ++ return NULL; ++} ++ + static void mshv_start_vcpu_thread(CPUState *cpu) + { +- error_report("unimplemented"); +- abort(); ++ char thread_name[VCPU_THREAD_NAME_SIZE]; ++ ++ cpu->thread = g_malloc0(sizeof(QemuThread)); ++ cpu->halt_cond = g_malloc0(sizeof(QemuCond)); ++ ++ qemu_cond_init(cpu->halt_cond); ++ ++ trace_mshv_start_vcpu_thread(thread_name, cpu->cpu_index); ++ qemu_thread_create(cpu->thread, thread_name, mshv_vcpu_thread, cpu, ++ QEMU_THREAD_JOINABLE); ++} ++ ++static void do_mshv_cpu_synchronize_post_init(CPUState *cpu, ++ run_on_cpu_data arg) ++{ ++ int ret = mshv_arch_put_registers(cpu); ++ if (ret < 0) { ++ error_report("Failed to put registers after init: %s", strerror(-ret)); ++ abort(); ++ } ++ ++ cpu->accel->dirty = false; + } + + static void mshv_cpu_synchronize_post_init(CPUState *cpu) + { +- error_report("unimplemented"); +- abort(); ++ run_on_cpu(cpu, do_mshv_cpu_synchronize_post_init, RUN_ON_CPU_NULL); + } + + static void mshv_cpu_synchronize_post_reset(CPUState *cpu) + { +- error_report("unimplemented"); +- abort(); ++ int ret = mshv_arch_put_registers(cpu); ++ if (ret) { ++ error_report("Failed to put registers after reset: %s", ++ strerror(-ret)); ++ cpu_dump_state(cpu, stderr, CPU_DUMP_CODE); ++ vm_stop(RUN_STATE_INTERNAL_ERROR); ++ } ++ cpu->accel->dirty = false; ++} ++ ++static void do_mshv_cpu_synchronize_pre_loadvm(CPUState *cpu, ++ run_on_cpu_data arg) ++{ ++ cpu->accel->dirty = true; + } + + static void mshv_cpu_synchronize_pre_loadvm(CPUState *cpu) + { +- error_report("unimplemented"); +- abort(); ++ run_on_cpu(cpu, do_mshv_cpu_synchronize_pre_loadvm, RUN_ON_CPU_NULL); ++} ++ ++static void do_mshv_cpu_synchronize(CPUState *cpu, run_on_cpu_data arg) ++{ ++ if (!cpu->accel->dirty) { ++ int ret = mshv_load_regs(cpu); ++ if (ret < 0) { ++ error_report("Failed to load registers for vcpu %d", ++ cpu->cpu_index); ++ ++ cpu_dump_state(cpu, stderr, CPU_DUMP_CODE); ++ vm_stop(RUN_STATE_INTERNAL_ERROR); ++ } ++ ++ cpu->accel->dirty = true; ++ } + } + + static void mshv_cpu_synchronize(CPUState *cpu) + { +- error_report("unimplemented"); +- abort(); ++ if (!cpu->accel->dirty) { ++ run_on_cpu(cpu, do_mshv_cpu_synchronize, RUN_ON_CPU_NULL); ++ } + } + + static bool mshv_cpus_are_resettable(void) + { +- error_report("unimplemented"); +- abort(); ++ return false; + } + + static void mshv_accel_class_init(ObjectClass *oc, const void *data) +diff --git a/accel/mshv/trace-events b/accel/mshv/trace-events +index 6130c4abf8..a4dffeb24a 100644 +--- a/accel/mshv/trace-events ++++ b/accel/mshv/trace-events +@@ -3,6 +3,8 @@ + # + # SPDX-License-Identifier: GPL-2.0-or-later + ++mshv_start_vcpu_thread(const char* thread, uint32_t cpu) "thread=%s cpu_index=%d" ++ + mshv_set_memory(bool add, uint64_t gpa, uint64_t size, uint64_t user_addr, bool readonly, int ret) "add=%d gpa=0x%" PRIx64 " size=0x%" PRIx64 " user=0x%" PRIx64 " readonly=%d result=%d" + mshv_mem_ioeventfd_add(uint64_t addr, uint32_t size, uint32_t data) "addr=0x%" PRIx64 " size=%d data=0x%x" + mshv_mem_ioeventfd_del(uint64_t addr, uint32_t size, uint32_t data) "addr=0x%" PRIx64 " size=%d data=0x%x" +diff --git a/include/system/mshv.h b/include/system/mshv.h +index 1011e81df4..bbc42f4dc3 100644 +--- a/include/system/mshv.h ++++ b/include/system/mshv.h +@@ -41,7 +41,7 @@ extern bool mshv_allowed; + #define mshv_msi_via_irqfd_enabled() mshv_enabled() + #else /* CONFIG_MSHV_IS_POSSIBLE */ + #define mshv_enabled() false +-#define mshv_msi_via_irqfd_enabled() false ++#define mshv_msi_via_irqfd_enabled() mshv_enabled() + #endif + + typedef struct MshvState MshvState; +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index b36124a0ea..fb80f69772 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -14,6 +14,8 @@ + #ifndef QEMU_MSHV_INT_H + #define QEMU_MSHV_INT_H + ++typedef struct hyperv_message hv_message; ++ + struct AccelCPUState { + int cpufd; + bool dirty; +@@ -44,6 +46,24 @@ typedef struct MshvMsiControl { + GHashTable *gsi_routes; + } MshvMsiControl; + ++#define mshv_vcpufd(cpu) (cpu->accel->cpufd) ++ ++/* cpu */ ++typedef enum MshvVmExit { ++ MshvVmExitIgnore = 0, ++ MshvVmExitShutdown = 1, ++ MshvVmExitSpecial = 2, ++} MshvVmExit; ++ ++void mshv_init_mmio_emu(void); ++int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd); ++void mshv_remove_vcpu(int vm_fd, int cpu_fd); ++int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit); ++int mshv_load_regs(CPUState *cpu); ++int mshv_store_regs(CPUState *cpu); ++int mshv_arch_put_registers(const CPUState *cpu); ++void mshv_arch_init_vcpu(CPUState *cpu); ++void mshv_arch_destroy_vcpu(CPUState *cpu); + void mshv_arch_amend_proc_features( + union hv_partition_synthetic_processor_features *features); + int mshv_arch_post_init_vm(int vm_fd); +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index de0c26bc6c..02d71ebc14 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -22,15 +22,78 @@ + #include "hw/hyperv/hvgdk_mini.h" + #include "hw/hyperv/hvhdk_mini.h" + ++#include "cpu.h" ++#include "emulate/x86_decode.h" ++#include "emulate/x86_emu.h" ++#include "emulate/x86_flags.h" ++ + #include "trace-accel_mshv.h" + #include "trace.h" + ++int mshv_store_regs(CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++int mshv_load_regs(CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++int mshv_arch_put_registers(const CPUState *cpu) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ + void mshv_arch_amend_proc_features( + union hv_partition_synthetic_processor_features *features) + { + features->access_guest_idle_reg = 1; + } + ++int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++void mshv_remove_vcpu(int vm_fd, int cpu_fd) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++void mshv_init_mmio_emu(void) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ ++void mshv_arch_init_vcpu(CPUState *cpu) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ ++ env->emu_mmio_buf = g_new(char, 4096); ++} ++ ++void mshv_arch_destroy_vcpu(CPUState *cpu) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ ++ g_clear_pointer(&env->emu_mmio_buf, g_free); ++} ++ + /* + * Default Microsoft Hypervisor behavior for unimplemented MSR is to send a + * fault to the guest if it tries to access it. It is possible to override +-- +2.47.3 + diff --git a/kvm-accel-mshv-Add-vCPU-signal-handling.patch b/kvm-accel-mshv-Add-vCPU-signal-handling.patch new file mode 100644 index 0000000..918e2b4 --- /dev/null +++ b/kvm-accel-mshv-Add-vCPU-signal-handling.patch @@ -0,0 +1,75 @@ +From 38112aed4e07ec8f43b552f887bc2077e443ba4b Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:31 +0200 +Subject: [PATCH 14/32] accel/mshv: Add vCPU signal handling + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [12/30] c3c9c5d9ad4d335a12c9d54499e4a2902b523441 + +Implement signal handling for MSHV vCPUs to support asynchronous +interrupts from the main thread. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-12-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 575df4df54e060661db45e6f80293b29ea8901c1) +Signed-off-by: Igor Mammedov +--- + accel/mshv/mshv-all.c | 30 ++++++++++++++++++++++++++++++ + 1 file changed, 30 insertions(+) + +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +index e02421d79d..fa1f8f35bd 100644 +--- a/accel/mshv/mshv-all.c ++++ b/accel/mshv/mshv-all.c +@@ -524,6 +524,35 @@ static int mshv_cpu_exec(CPUState *cpu) + return ret; + } + ++/* ++ * The signal handler is triggered when QEMU's main thread receives a SIG_IPI ++ * (SIGUSR1). This signal causes the current CPU thread to be kicked, forcing a ++ * VM exit on the CPU. The VM exit generates an exit reason that breaks the loop ++ * (see mshv_cpu_exec). If the exit is due to a Ctrl+A+x command, the system ++ * will shut down. For other cases, the system will continue running. ++ */ ++static void sa_ipi_handler(int sig) ++{ ++ /* TODO: call IOCTL to set_immediate_exit, once implemented. */ ++ ++ qemu_cpu_kick_self(); ++} ++ ++static void init_signal(CPUState *cpu) ++{ ++ /* init cpu signals */ ++ struct sigaction sigact; ++ sigset_t set; ++ ++ memset(&sigact, 0, sizeof(sigact)); ++ sigact.sa_handler = sa_ipi_handler; ++ sigaction(SIG_IPI, &sigact, NULL); ++ ++ pthread_sigmask(SIG_BLOCK, NULL, &set); ++ sigdelset(&set, SIG_IPI); ++ pthread_sigmask(SIG_SETMASK, &set, NULL); ++} ++ + static void *mshv_vcpu_thread(void *arg) + { + CPUState *cpu = arg; +@@ -540,6 +569,7 @@ static void *mshv_vcpu_thread(void *arg) + error_report("Failed to init vcpu %d", cpu->cpu_index); + goto cleanup; + } ++ init_signal(cpu); + + /* signal CPU creation */ + cpu_thread_signal_created(cpu); +-- +2.47.3 + diff --git a/kvm-accel-mshv-Handle-overlapping-mem-mappings.patch b/kvm-accel-mshv-Handle-overlapping-mem-mappings.patch new file mode 100644 index 0000000..92181b5 --- /dev/null +++ b/kvm-accel-mshv-Handle-overlapping-mem-mappings.patch @@ -0,0 +1,695 @@ +From c6e4e7657d407507a5e8cc6daadc6f309e8a42e9 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:43 +0200 +Subject: [PATCH 26/32] accel/mshv: Handle overlapping mem mappings + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [24/30] abe9045eadeabccae811682dccd8679f41c32cc0 + +QEMU maps certain regions into the guest multiple times, as seen in the +trace below. Currently the MSHV kernel driver will reject those +mappings. To workaround this, a record is kept (a static global list of +"slots", inspired by what the HVF accelerator has implemented). An +overlapping region is not registered at the hypervisor, and marked as +mapped=false. If there is an UNMAPPED_GPA exit, we can look for a slot +that is unmapped and would cover the GPA. In this case we map out the +conflicting slot and map in the requested region. + +mshv_set_phys_mem add=1 name=pc.bios +mshv_map_memory => u_a=7ffff4e00000 gpa=00fffc0000 size=00040000 +mshv_set_phys_mem add=1 name=ioapic +mshv_set_phys_mem add=1 name=hpet +mshv_set_phys_mem add=0 name=pc.ram +mshv_unmap_memory u_a=7fff67e00000 gpa=0000000000 size=80000000 +mshv_set_phys_mem add=1 name=pc.ram +mshv_map_memory u_a=7fff67e00000 gpa=0000000000 size=000c0000 +mshv_set_phys_mem add=1 name=pc.rom +mshv_map_memory u_a=7ffff4c00000 gpa=00000c0000 size=00020000 +mshv_set_phys_mem add=1 name=pc.bios +mshv_remap_attempt => u_a=7ffff4e20000 gpa=00000e0000 size=00020000 + +The mapping table is guarded by a mutex for concurrent modification and +RCU mechanisms for concurrent reads. Writes occur rarely, but we'll have +to verify whether an unmapped region exist for each UNMAPPED_GPA exit, +which happens frequently. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-24-magnuskulke@linux.microsoft.com +[Fix format strings for trace-events; mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit efc4093358511a58846a409b965213aa1bb9f31a) +Signed-off-by: Igor Mammedov +--- + accel/mshv/mem.c | 406 +++++++++++++++++++++++++++++++++--- + accel/mshv/mshv-all.c | 2 + + accel/mshv/trace-events | 5 + + include/system/mshv_int.h | 22 +- + target/i386/mshv/mshv-cpu.c | 43 ++++ + 5 files changed, 448 insertions(+), 30 deletions(-) + +diff --git a/accel/mshv/mem.c b/accel/mshv/mem.c +index e55c38d4db..0e2164af3e 100644 +--- a/accel/mshv/mem.c ++++ b/accel/mshv/mem.c +@@ -11,7 +11,9 @@ + */ + + #include "qemu/osdep.h" ++#include "qemu/lockable.h" + #include "qemu/error-report.h" ++#include "qemu/rcu.h" + #include "linux/mshv.h" + #include "system/address-spaces.h" + #include "system/mshv.h" +@@ -20,6 +22,137 @@ + #include + #include "trace.h" + ++typedef struct SlotsRCUReclaim { ++ struct rcu_head rcu; ++ GList *old_head; ++ MshvMemorySlot *removed_slot; ++} SlotsRCUReclaim; ++ ++static void rcu_reclaim_slotlist(struct rcu_head *rcu) ++{ ++ SlotsRCUReclaim *r = container_of(rcu, SlotsRCUReclaim, rcu); ++ g_list_free(r->old_head); ++ g_free(r->removed_slot); ++ g_free(r); ++} ++ ++static void publish_slots(GList *new_head, GList *old_head, ++ MshvMemorySlot *removed_slot) ++{ ++ MshvMemorySlotManager *manager = &mshv_state->msm; ++ ++ assert(manager); ++ qatomic_store_release(&manager->slots, new_head); ++ ++ SlotsRCUReclaim *r = g_new(SlotsRCUReclaim, 1); ++ r->old_head = old_head; ++ r->removed_slot = removed_slot; ++ ++ call_rcu1(&r->rcu, rcu_reclaim_slotlist); ++} ++ ++/* Needs to be called with mshv_state->msm.mutex held */ ++static int remove_slot(MshvMemorySlot *slot) ++{ ++ GList *old_head, *new_head; ++ MshvMemorySlotManager *manager = &mshv_state->msm; ++ ++ assert(manager); ++ old_head = qatomic_load_acquire(&manager->slots); ++ ++ if (!g_list_find(old_head, slot)) { ++ error_report("slot requested for removal not found"); ++ return -1; ++ } ++ ++ new_head = g_list_copy(old_head); ++ new_head = g_list_remove(new_head, slot); ++ manager->n_slots--; ++ ++ publish_slots(new_head, old_head, slot); ++ ++ return 0; ++} ++ ++/* Needs to be called with mshv_state->msm.mutex held */ ++static MshvMemorySlot *append_slot(uint64_t gpa, uint64_t userspace_addr, ++ uint64_t size, bool readonly) ++{ ++ GList *old_head, *new_head; ++ MshvMemorySlot *slot; ++ MshvMemorySlotManager *manager = &mshv_state->msm; ++ ++ assert(manager); ++ ++ old_head = qatomic_load_acquire(&manager->slots); ++ ++ if (manager->n_slots >= MSHV_MAX_MEM_SLOTS) { ++ error_report("no free memory slots available"); ++ return NULL; ++ } ++ ++ slot = g_new0(MshvMemorySlot, 1); ++ slot->guest_phys_addr = gpa; ++ slot->userspace_addr = userspace_addr; ++ slot->memory_size = size; ++ slot->readonly = readonly; ++ ++ new_head = g_list_copy(old_head); ++ new_head = g_list_append(new_head, slot); ++ manager->n_slots++; ++ ++ publish_slots(new_head, old_head, NULL); ++ ++ return slot; ++} ++ ++static int slot_overlaps(const MshvMemorySlot *slot1, ++ const MshvMemorySlot *slot2) ++{ ++ uint64_t start_1 = slot1->userspace_addr, ++ start_2 = slot2->userspace_addr; ++ size_t len_1 = slot1->memory_size, ++ len_2 = slot2->memory_size; ++ ++ if (slot1 == slot2) { ++ return -1; ++ } ++ ++ return ranges_overlap(start_1, len_1, start_2, len_2) ? 0 : -1; ++} ++ ++static bool is_mapped(MshvMemorySlot *slot) ++{ ++ /* Subsequent reads of mapped field see a fully-initialized slot */ ++ return qatomic_load_acquire(&slot->mapped); ++} ++ ++/* ++ * Find slot that is: ++ * - overlapping in userspace ++ * - currently mapped in the guest ++ * ++ * Needs to be called with mshv_state->msm.mutex or RCU read lock held. ++ */ ++static MshvMemorySlot *find_overlap_mem_slot(GList *head, MshvMemorySlot *slot) ++{ ++ GList *found; ++ MshvMemorySlot *overlap_slot; ++ ++ found = g_list_find_custom(head, slot, (GCompareFunc) slot_overlaps); ++ ++ if (!found) { ++ return NULL; ++ } ++ ++ overlap_slot = found->data; ++ if (!overlap_slot || !is_mapped(overlap_slot)) { ++ return NULL; ++ } ++ ++ return overlap_slot; ++} ++ + static int set_guest_memory(int vm_fd, + const struct mshv_user_mem_region *region) + { +@@ -27,38 +160,169 @@ static int set_guest_memory(int vm_fd, + + ret = ioctl(vm_fd, MSHV_SET_GUEST_MEMORY, region); + if (ret < 0) { +- error_report("failed to set guest memory"); +- return -errno; ++ error_report("failed to set guest memory: %s", strerror(errno)); ++ return -1; + } + + return 0; + } + +-static int map_or_unmap(int vm_fd, const MshvMemoryRegion *mr, bool map) ++static int map_or_unmap(int vm_fd, const MshvMemorySlot *slot, bool map) + { + struct mshv_user_mem_region region = {0}; + +- region.guest_pfn = mr->guest_phys_addr >> MSHV_PAGE_SHIFT; +- region.size = mr->memory_size; +- region.userspace_addr = mr->userspace_addr; ++ region.guest_pfn = slot->guest_phys_addr >> MSHV_PAGE_SHIFT; ++ region.size = slot->memory_size; ++ region.userspace_addr = slot->userspace_addr; + + if (!map) { + region.flags |= (1 << MSHV_SET_MEM_BIT_UNMAP); +- trace_mshv_unmap_memory(mr->userspace_addr, mr->guest_phys_addr, +- mr->memory_size); ++ trace_mshv_unmap_memory(slot->userspace_addr, slot->guest_phys_addr, ++ slot->memory_size); + return set_guest_memory(vm_fd, ®ion); + } + + region.flags = BIT(MSHV_SET_MEM_BIT_EXECUTABLE); +- if (!mr->readonly) { ++ if (!slot->readonly) { + region.flags |= BIT(MSHV_SET_MEM_BIT_WRITABLE); + } + +- trace_mshv_map_memory(mr->userspace_addr, mr->guest_phys_addr, +- mr->memory_size); ++ trace_mshv_map_memory(slot->userspace_addr, slot->guest_phys_addr, ++ slot->memory_size); + return set_guest_memory(vm_fd, ®ion); + } + ++static int slot_matches_region(const MshvMemorySlot *slot1, ++ const MshvMemorySlot *slot2) ++{ ++ return (slot1->guest_phys_addr == slot2->guest_phys_addr && ++ slot1->userspace_addr == slot2->userspace_addr && ++ slot1->memory_size == slot2->memory_size) ? 0 : -1; ++} ++ ++/* Needs to be called with mshv_state->msm.mutex held */ ++static MshvMemorySlot *find_mem_slot_by_region(uint64_t gpa, uint64_t size, ++ uint64_t userspace_addr) ++{ ++ MshvMemorySlot ref_slot = { ++ .guest_phys_addr = gpa, ++ .userspace_addr = userspace_addr, ++ .memory_size = size, ++ }; ++ GList *found; ++ MshvMemorySlotManager *manager = &mshv_state->msm; ++ ++ assert(manager); ++ found = g_list_find_custom(manager->slots, &ref_slot, ++ (GCompareFunc) slot_matches_region); ++ ++ return found ? found->data : NULL; ++} ++ ++static int slot_covers_gpa(const MshvMemorySlot *slot, uint64_t *gpa_p) ++{ ++ uint64_t gpa_offset, gpa = *gpa_p; ++ ++ gpa_offset = gpa - slot->guest_phys_addr; ++ return (slot->guest_phys_addr <= gpa && gpa_offset < slot->memory_size) ++ ? 0 : -1; ++} ++ ++/* Needs to be called with mshv_state->msm.mutex or RCU read lock held */ ++static MshvMemorySlot *find_mem_slot_by_gpa(GList *head, uint64_t gpa) ++{ ++ GList *found; ++ MshvMemorySlot *slot; ++ ++ trace_mshv_find_slot_by_gpa(gpa); ++ ++ found = g_list_find_custom(head, &gpa, (GCompareFunc) slot_covers_gpa); ++ if (found) { ++ slot = found->data; ++ trace_mshv_found_slot(slot->userspace_addr, slot->guest_phys_addr, ++ slot->memory_size); ++ return slot; ++ } ++ ++ return NULL; ++} ++ ++/* Needs to be called with mshv_state->msm.mutex held */ ++static void set_mapped(MshvMemorySlot *slot, bool mapped) ++{ ++ /* prior writes to mapped field becomes visible before readers see slot */ ++ qatomic_store_release(&slot->mapped, mapped); ++} ++ ++MshvRemapResult mshv_remap_overlap_region(int vm_fd, uint64_t gpa) ++{ ++ MshvMemorySlot *gpa_slot, *overlap_slot; ++ GList *head; ++ int ret; ++ MshvMemorySlotManager *manager = &mshv_state->msm; ++ ++ /* fast path, called often by unmapped_gpa vm exit */ ++ WITH_RCU_READ_LOCK_GUARD() { ++ assert(manager); ++ head = qatomic_load_acquire(&manager->slots); ++ /* return early if no slot is found */ ++ gpa_slot = find_mem_slot_by_gpa(head, gpa); ++ if (gpa_slot == NULL) { ++ return MshvRemapNoMapping; ++ } ++ ++ /* return early if no overlapping slot is found */ ++ overlap_slot = find_overlap_mem_slot(head, gpa_slot); ++ if (overlap_slot == NULL) { ++ return MshvRemapNoOverlap; ++ } ++ } ++ ++ /* ++ * We'll modify the mapping list, so we need to upgrade to mutex and ++ * recheck. ++ */ ++ assert(manager); ++ QEMU_LOCK_GUARD(&manager->mutex); ++ ++ /* return early if no slot is found */ ++ gpa_slot = find_mem_slot_by_gpa(manager->slots, gpa); ++ if (gpa_slot == NULL) { ++ return MshvRemapNoMapping; ++ } ++ ++ /* return early if no overlapping slot is found */ ++ overlap_slot = find_overlap_mem_slot(manager->slots, gpa_slot); ++ if (overlap_slot == NULL) { ++ return MshvRemapNoOverlap; ++ } ++ ++ /* unmap overlapping slot */ ++ ret = map_or_unmap(vm_fd, overlap_slot, false); ++ if (ret < 0) { ++ error_report("failed to unmap overlap region"); ++ abort(); ++ } ++ set_mapped(overlap_slot, false); ++ warn_report("mapped out userspace_addr=0x%016lx gpa=0x%010lx size=0x%lx", ++ overlap_slot->userspace_addr, ++ overlap_slot->guest_phys_addr, ++ overlap_slot->memory_size); ++ ++ /* map region for gpa */ ++ ret = map_or_unmap(vm_fd, gpa_slot, true); ++ if (ret < 0) { ++ error_report("failed to map new region"); ++ abort(); ++ } ++ set_mapped(gpa_slot, true); ++ warn_report("mapped in userspace_addr=0x%016lx gpa=0x%010lx size=0x%lx", ++ gpa_slot->userspace_addr, gpa_slot->guest_phys_addr, ++ gpa_slot->memory_size); ++ ++ return MshvRemapOk; ++} ++ + static int handle_unmapped_mmio_region_read(uint64_t gpa, uint64_t size, + uint8_t *data) + { +@@ -124,20 +388,97 @@ int mshv_guest_mem_write(uint64_t gpa, const uint8_t *data, uintptr_t size, + return -1; + } + +-static int set_memory(const MshvMemoryRegion *mshv_mr, bool add) ++static int tracked_unmap(int vm_fd, uint64_t gpa, uint64_t size, ++ uint64_t userspace_addr) + { +- int ret = 0; ++ int ret; ++ MshvMemorySlot *slot; ++ MshvMemorySlotManager *manager = &mshv_state->msm; ++ ++ assert(manager); ++ ++ QEMU_LOCK_GUARD(&manager->mutex); ++ ++ slot = find_mem_slot_by_region(gpa, size, userspace_addr); ++ if (!slot) { ++ trace_mshv_skip_unset_mem(userspace_addr, gpa, size); ++ /* no work to do */ ++ return 0; ++ } ++ ++ if (!is_mapped(slot)) { ++ /* remove slot, no need to unmap */ ++ return remove_slot(slot); ++ } + +- if (!mshv_mr) { +- error_report("Invalid mshv_mr"); ++ ret = map_or_unmap(vm_fd, slot, false); ++ if (ret < 0) { ++ error_report("failed to unmap memory region"); ++ return ret; ++ } ++ return remove_slot(slot); ++} ++ ++static int tracked_map(int vm_fd, uint64_t gpa, uint64_t size, bool readonly, ++ uint64_t userspace_addr) ++{ ++ MshvMemorySlot *slot, *overlap_slot; ++ int ret; ++ MshvMemorySlotManager *manager = &mshv_state->msm; ++ ++ assert(manager); ++ ++ QEMU_LOCK_GUARD(&manager->mutex); ++ ++ slot = find_mem_slot_by_region(gpa, size, userspace_addr); ++ if (slot) { ++ error_report("memory region already mapped at gpa=0x%lx, " ++ "userspace_addr=0x%lx, size=0x%lx", ++ slot->guest_phys_addr, slot->userspace_addr, ++ slot->memory_size); + return -1; + } + +- trace_mshv_set_memory(add, mshv_mr->guest_phys_addr, +- mshv_mr->memory_size, +- mshv_mr->userspace_addr, mshv_mr->readonly, +- ret); +- return map_or_unmap(mshv_state->vm, mshv_mr, add); ++ slot = append_slot(gpa, userspace_addr, size, readonly); ++ ++ overlap_slot = find_overlap_mem_slot(manager->slots, slot); ++ if (overlap_slot) { ++ trace_mshv_remap_attempt(slot->userspace_addr, ++ slot->guest_phys_addr, ++ slot->memory_size); ++ warn_report("attempt to map region [0x%lx-0x%lx], while " ++ "[0x%lx-0x%lx] is already mapped in the guest", ++ userspace_addr, userspace_addr + size - 1, ++ overlap_slot->userspace_addr, ++ overlap_slot->userspace_addr + ++ overlap_slot->memory_size - 1); ++ ++ /* do not register mem slot in hv, but record for later swap-in */ ++ set_mapped(slot, false); ++ ++ return 0; ++ } ++ ++ ret = map_or_unmap(vm_fd, slot, true); ++ if (ret < 0) { ++ error_report("failed to map memory region"); ++ return -1; ++ } ++ set_mapped(slot, true); ++ ++ return 0; ++} ++ ++static int set_memory(uint64_t gpa, uint64_t size, bool readonly, ++ uint64_t userspace_addr, bool add) ++{ ++ int vm_fd = mshv_state->vm; ++ ++ if (add) { ++ return tracked_map(vm_fd, gpa, size, readonly, userspace_addr); ++ } ++ ++ return tracked_unmap(vm_fd, gpa, size, userspace_addr); + } + + /* +@@ -173,7 +514,9 @@ void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, + bool writable = !area->readonly && !area->rom_device; + hwaddr start_addr, mr_offset, size; + void *ram; +- MshvMemoryRegion mshv_mr = {0}; ++ ++ size = align_section(section, &start_addr); ++ trace_mshv_set_phys_mem(add, section->mr->name, start_addr); + + size = align_section(section, &start_addr); + trace_mshv_set_phys_mem(add, section->mr->name, start_addr); +@@ -200,14 +543,21 @@ void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, + + ram = memory_region_get_ram_ptr(area) + mr_offset; + +- mshv_mr.guest_phys_addr = start_addr; +- mshv_mr.memory_size = size; +- mshv_mr.readonly = !writable; +- mshv_mr.userspace_addr = (uint64_t)ram; +- +- ret = set_memory(&mshv_mr, add); ++ ret = set_memory(start_addr, size, !writable, (uint64_t)ram, add); + if (ret < 0) { +- error_report("Failed to set memory region"); ++ error_report("failed to set memory region"); + abort(); + } + } ++ ++void mshv_init_memory_slot_manager(MshvState *mshv_state) ++{ ++ MshvMemorySlotManager *manager; ++ ++ assert(mshv_state); ++ manager = &mshv_state->msm; ++ ++ manager->n_slots = 0; ++ manager->slots = NULL; ++ qemu_mutex_init(&manager->mutex); ++} +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +index fa1f8f35bd..5edfcbad9d 100644 +--- a/accel/mshv/mshv-all.c ++++ b/accel/mshv/mshv-all.c +@@ -437,6 +437,8 @@ static int mshv_init(AccelState *as, MachineState *ms) + + mshv_init_msicontrol(); + ++ mshv_init_memory_slot_manager(s); ++ + ret = create_vm(mshv_fd, &vm_fd); + if (ret < 0) { + close(mshv_fd); +diff --git a/accel/mshv/trace-events b/accel/mshv/trace-events +index a4dffeb24a..36f0d59b38 100644 +--- a/accel/mshv/trace-events ++++ b/accel/mshv/trace-events +@@ -26,3 +26,8 @@ mshv_map_memory(uint64_t userspace_addr, uint64_t gpa, uint64_t size) "\tu_a=0x% + mshv_unmap_memory(uint64_t userspace_addr, uint64_t gpa, uint64_t size) "\tu_a=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%08" PRIx64 + mshv_set_phys_mem(bool add, const char *name, uint64_t gpa) "\tadd=%d name=%s gpa=0x%010" PRIx64 + mshv_handle_mmio(uint64_t gva, uint64_t gpa, uint64_t size, uint8_t access_type) "\tgva=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%" PRIx64 " access_type=%d" ++ ++mshv_found_slot(uint64_t userspace_addr, uint64_t gpa, uint64_t size) "\tu_a=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%08" PRIx64 ++mshv_skip_unset_mem(uint64_t userspace_addr, uint64_t gpa, uint64_t size) "\tu_a=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%08" PRIx64 ++mshv_remap_attempt(uint64_t userspace_addr, uint64_t gpa, uint64_t size) "\tu_a=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%08" PRIx64 ++mshv_find_slot_by_gpa(uint64_t gpa) "\tgpa=0x%010" PRIx64 +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index b29d39911d..6350c69e9d 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -16,6 +16,8 @@ + + #define MSHV_MSR_ENTRIES_COUNT 64 + ++#define MSHV_MAX_MEM_SLOTS 32 ++ + typedef struct hyperv_message hv_message; + + struct AccelCPUState { +@@ -33,6 +35,12 @@ typedef struct MshvAddressSpace { + AddressSpace *as; + } MshvAddressSpace; + ++typedef struct MshvMemorySlotManager { ++ size_t n_slots; ++ GList *slots; ++ QemuMutex mutex; ++} MshvMemorySlotManager; ++ + struct MshvState { + AccelState parent_obj; + int vm; +@@ -41,6 +49,7 @@ struct MshvState { + int nr_as; + MshvAddressSpace *as; + int fd; ++ MshvMemorySlotManager msm; + }; + + typedef struct MshvMsiControl { +@@ -71,6 +80,12 @@ typedef enum MshvVmExit { + MshvVmExitSpecial = 2, + } MshvVmExit; + ++typedef enum MshvRemapResult { ++ MshvRemapOk = 0, ++ MshvRemapNoMapping = 1, ++ MshvRemapNoOverlap = 2, ++} MshvRemapResult; ++ + void mshv_init_mmio_emu(void); + int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd); + void mshv_remove_vcpu(int vm_fd, int cpu_fd); +@@ -94,19 +109,22 @@ int mshv_hvcall(int fd, const struct mshv_root_hvcall *args); + #endif + + /* memory */ +-typedef struct MshvMemoryRegion { ++typedef struct MshvMemorySlot { + uint64_t guest_phys_addr; + uint64_t memory_size; + uint64_t userspace_addr; + bool readonly; +-} MshvMemoryRegion; ++ bool mapped; ++} MshvMemorySlot; + ++MshvRemapResult mshv_remap_overlap_region(int vm_fd, uint64_t gpa); + int mshv_guest_mem_read(uint64_t gpa, uint8_t *data, uintptr_t size, + bool is_secure_mode, bool instruction_fetch); + int mshv_guest_mem_write(uint64_t gpa, const uint8_t *data, uintptr_t size, + bool is_secure_mode); + void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, + bool add); ++void mshv_init_memory_slot_manager(MshvState *mshv_state); + + /* msr */ + typedef struct MshvMsrEntry { +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 7edc032cea..de87142bff 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -1170,6 +1170,43 @@ static int handle_mmio(CPUState *cpu, const struct hyperv_message *msg, + return 0; + } + ++static int handle_unmapped_mem(int vm_fd, CPUState *cpu, ++ const struct hyperv_message *msg, ++ MshvVmExit *exit_reason) ++{ ++ struct hv_x64_memory_intercept_message info = { 0 }; ++ uint64_t gpa; ++ int ret; ++ enum MshvRemapResult remap_result; ++ ++ ret = set_memory_info(msg, &info); ++ if (ret < 0) { ++ error_report("failed to convert message to memory info"); ++ return -1; ++ } ++ ++ gpa = info.guest_physical_address; ++ ++ /* attempt to remap the region, in case of overlapping userspace mappings */ ++ remap_result = mshv_remap_overlap_region(vm_fd, gpa); ++ *exit_reason = MshvVmExitIgnore; ++ ++ switch (remap_result) { ++ case MshvRemapNoMapping: ++ /* if we didn't find a mapping, it is probably mmio */ ++ return handle_mmio(cpu, msg, exit_reason); ++ case MshvRemapOk: ++ break; ++ case MshvRemapNoOverlap: ++ /* This should not happen, but we are forgiving it */ ++ warn_report("found no overlap for unmapped region"); ++ *exit_reason = MshvVmExitSpecial; ++ break; ++ } ++ ++ return 0; ++} ++ + static int set_ioport_info(const struct hyperv_message *msg, + hv_x64_io_port_intercept_message *info) + { +@@ -1507,6 +1544,12 @@ int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit) + case HVMSG_UNRECOVERABLE_EXCEPTION: + return MshvVmExitShutdown; + case HVMSG_UNMAPPED_GPA: ++ ret = handle_unmapped_mem(vm_fd, cpu, msg, &exit_reason); ++ if (ret < 0) { ++ error_report("failed to handle unmapped memory"); ++ return -1; ++ } ++ return exit_reason; + case HVMSG_GPA_INTERCEPT: + ret = handle_mmio(cpu, msg, &exit_reason); + if (ret < 0) { +-- +2.47.3 + diff --git a/kvm-accel-mshv-Initialize-VM-partition.patch b/kvm-accel-mshv-Initialize-VM-partition.patch new file mode 100644 index 0000000..da24863 --- /dev/null +++ b/kvm-accel-mshv-Initialize-VM-partition.patch @@ -0,0 +1,1302 @@ +From f2010418608cbbfe00095d628d74006d69e83138 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Thu, 2 Oct 2025 18:28:16 +0200 +Subject: [PATCH 11/32] accel/mshv: Initialize VM partition + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [9/30] 7ef17e116fd86fc05b94f7f1e9bbbd4ae1299fa5 + +Create the MSHV virtual machine by opening a partition and issuing +the necessary ioctl to initialize it. This sets up the basic VM +structure and initial configuration used by MSHV to manage guest state. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-10-magnuskulke@linux.microsoft.com +[Add stubs; fix format strings for trace-events; make mshv_hvcall + available only in per-target files; mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit c5f23bccde416aab52fe14dec6f6716a85f0cd04) +Signed-off-by: Igor Mammedov +--- + accel/mshv/irq.c | 399 +++++++++++++++++++++++++++++++++++ + accel/mshv/mem.c | 129 ++++++++++- + accel/mshv/meson.build | 1 + + accel/mshv/mshv-all.c | 326 ++++++++++++++++++++++++++++ + accel/mshv/trace-events | 26 +++ + accel/mshv/trace.h | 14 ++ + accel/stubs/meson.build | 1 + + accel/stubs/mshv-stub.c | 44 ++++ + hw/intc/apic.c | 8 + + include/system/mshv.h | 11 +- + include/system/mshv_int.h | 25 +++ + meson.build | 1 + + target/i386/mshv/meson.build | 1 + + target/i386/mshv/mshv-cpu.c | 72 +++++++ + 14 files changed, 1054 insertions(+), 4 deletions(-) + create mode 100644 accel/mshv/irq.c + create mode 100644 accel/mshv/trace-events + create mode 100644 accel/mshv/trace.h + create mode 100644 accel/stubs/mshv-stub.c + create mode 100644 target/i386/mshv/mshv-cpu.c + +diff --git a/accel/mshv/irq.c b/accel/mshv/irq.c +new file mode 100644 +index 0000000000..adf8f337d9 +--- /dev/null ++++ b/accel/mshv/irq.c +@@ -0,0 +1,399 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Ziqiao Zhou ++ * Magnus Kulke ++ * Stanislav Kinsburskii ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "linux/mshv.h" ++#include "qemu/osdep.h" ++#include "qemu/error-report.h" ++#include "hw/hyperv/hvhdk_mini.h" ++#include "hw/hyperv/hvgdk_mini.h" ++#include "hw/intc/ioapic.h" ++#include "hw/pci/msi.h" ++#include "system/mshv.h" ++#include "system/mshv_int.h" ++#include "trace.h" ++#include ++#include ++ ++#define MSHV_IRQFD_RESAMPLE_FLAG (1 << MSHV_IRQFD_BIT_RESAMPLE) ++#define MSHV_IRQFD_BIT_DEASSIGN_FLAG (1 << MSHV_IRQFD_BIT_DEASSIGN) ++ ++static MshvMsiControl *msi_control; ++static QemuMutex msi_control_mutex; ++ ++void mshv_init_msicontrol(void) ++{ ++ qemu_mutex_init(&msi_control_mutex); ++ msi_control = g_new0(MshvMsiControl, 1); ++ msi_control->gsi_routes = g_hash_table_new(g_direct_hash, g_direct_equal); ++ msi_control->updated = false; ++} ++ ++static int set_msi_routing(uint32_t gsi, uint64_t addr, uint32_t data) ++{ ++ struct mshv_user_irq_entry *entry; ++ uint32_t high_addr = addr >> 32; ++ uint32_t low_addr = addr & 0xFFFFFFFF; ++ GHashTable *gsi_routes; ++ ++ trace_mshv_set_msi_routing(gsi, addr, data); ++ ++ if (gsi >= MSHV_MAX_MSI_ROUTES) { ++ error_report("gsi >= MSHV_MAX_MSI_ROUTES"); ++ return -1; ++ } ++ ++ assert(msi_control); ++ ++ WITH_QEMU_LOCK_GUARD(&msi_control_mutex) { ++ gsi_routes = msi_control->gsi_routes; ++ entry = g_hash_table_lookup(gsi_routes, GINT_TO_POINTER(gsi)); ++ ++ if (entry ++ && entry->address_hi == high_addr ++ && entry->address_lo == low_addr ++ && entry->data == data) ++ { ++ /* nothing to update */ ++ return 0; ++ } ++ ++ /* free old entry */ ++ g_free(entry); ++ ++ /* create new entry */ ++ entry = g_new0(struct mshv_user_irq_entry, 1); ++ entry->gsi = gsi; ++ entry->address_hi = high_addr; ++ entry->address_lo = low_addr; ++ entry->data = data; ++ ++ g_hash_table_insert(gsi_routes, GINT_TO_POINTER(gsi), entry); ++ msi_control->updated = true; ++ } ++ ++ return 0; ++} ++ ++static int add_msi_routing(uint64_t addr, uint32_t data) ++{ ++ struct mshv_user_irq_entry *route_entry; ++ uint32_t high_addr = addr >> 32; ++ uint32_t low_addr = addr & 0xFFFFFFFF; ++ int gsi; ++ GHashTable *gsi_routes; ++ ++ trace_mshv_add_msi_routing(addr, data); ++ ++ assert(msi_control); ++ ++ WITH_QEMU_LOCK_GUARD(&msi_control_mutex) { ++ /* find an empty slot */ ++ gsi = 0; ++ gsi_routes = msi_control->gsi_routes; ++ while (gsi < MSHV_MAX_MSI_ROUTES) { ++ route_entry = g_hash_table_lookup(gsi_routes, GINT_TO_POINTER(gsi)); ++ if (!route_entry) { ++ break; ++ } ++ gsi++; ++ } ++ if (gsi >= MSHV_MAX_MSI_ROUTES) { ++ error_report("No empty gsi slot available"); ++ return -1; ++ } ++ ++ /* create new entry */ ++ route_entry = g_new0(struct mshv_user_irq_entry, 1); ++ route_entry->gsi = gsi; ++ route_entry->address_hi = high_addr; ++ route_entry->address_lo = low_addr; ++ route_entry->data = data; ++ ++ g_hash_table_insert(gsi_routes, GINT_TO_POINTER(gsi), route_entry); ++ msi_control->updated = true; ++ } ++ ++ return gsi; ++} ++ ++static int commit_msi_routing_table(int vm_fd) ++{ ++ guint len; ++ int i, ret; ++ size_t table_size; ++ struct mshv_user_irq_table *table; ++ GHashTableIter iter; ++ gpointer key, value; ++ ++ assert(msi_control); ++ ++ WITH_QEMU_LOCK_GUARD(&msi_control_mutex) { ++ if (!msi_control->updated) { ++ /* nothing to update */ ++ return 0; ++ } ++ ++ /* Calculate the size of the table */ ++ len = g_hash_table_size(msi_control->gsi_routes); ++ table_size = sizeof(struct mshv_user_irq_table) ++ + len * sizeof(struct mshv_user_irq_entry); ++ table = g_malloc0(table_size); ++ ++ g_hash_table_iter_init(&iter, msi_control->gsi_routes); ++ i = 0; ++ while (g_hash_table_iter_next(&iter, &key, &value)) { ++ struct mshv_user_irq_entry *entry = value; ++ table->entries[i] = *entry; ++ i++; ++ } ++ table->nr = i; ++ ++ trace_mshv_commit_msi_routing_table(vm_fd, len); ++ ++ ret = ioctl(vm_fd, MSHV_SET_MSI_ROUTING, table); ++ g_free(table); ++ if (ret < 0) { ++ error_report("Failed to commit msi routing table"); ++ return -1; ++ } ++ msi_control->updated = false; ++ } ++ return 0; ++} ++ ++static int remove_msi_routing(uint32_t gsi) ++{ ++ struct mshv_user_irq_entry *route_entry; ++ GHashTable *gsi_routes; ++ ++ trace_mshv_remove_msi_routing(gsi); ++ ++ if (gsi >= MSHV_MAX_MSI_ROUTES) { ++ error_report("Invalid GSI: %u", gsi); ++ return -1; ++ } ++ ++ assert(msi_control); ++ ++ WITH_QEMU_LOCK_GUARD(&msi_control_mutex) { ++ gsi_routes = msi_control->gsi_routes; ++ route_entry = g_hash_table_lookup(gsi_routes, GINT_TO_POINTER(gsi)); ++ if (route_entry) { ++ g_hash_table_remove(gsi_routes, GINT_TO_POINTER(gsi)); ++ g_free(route_entry); ++ msi_control->updated = true; ++ } ++ } ++ ++ return 0; ++} ++ ++/* Pass an eventfd which is to be used for injecting interrupts from userland */ ++static int irqfd(int vm_fd, int fd, int resample_fd, uint32_t gsi, ++ uint32_t flags) ++{ ++ int ret; ++ struct mshv_user_irqfd arg = { ++ .fd = fd, ++ .resamplefd = resample_fd, ++ .gsi = gsi, ++ .flags = flags, ++ }; ++ ++ ret = ioctl(vm_fd, MSHV_IRQFD, &arg); ++ if (ret < 0) { ++ error_report("Failed to set irqfd: gsi=%u, fd=%d", gsi, fd); ++ return -1; ++ } ++ return ret; ++} ++ ++static int register_irqfd(int vm_fd, int event_fd, uint32_t gsi) ++{ ++ int ret; ++ ++ trace_mshv_register_irqfd(vm_fd, event_fd, gsi); ++ ++ ret = irqfd(vm_fd, event_fd, 0, gsi, 0); ++ if (ret < 0) { ++ error_report("Failed to register irqfd: gsi=%u", gsi); ++ return -1; ++ } ++ return 0; ++} ++ ++static int register_irqfd_with_resample(int vm_fd, int event_fd, ++ int resample_fd, uint32_t gsi) ++{ ++ int ret; ++ uint32_t flags = MSHV_IRQFD_RESAMPLE_FLAG; ++ ++ ret = irqfd(vm_fd, event_fd, resample_fd, gsi, flags); ++ if (ret < 0) { ++ error_report("Failed to register irqfd with resample: gsi=%u", gsi); ++ return -errno; ++ } ++ return 0; ++} ++ ++static int unregister_irqfd(int vm_fd, int event_fd, uint32_t gsi) ++{ ++ int ret; ++ uint32_t flags = MSHV_IRQFD_BIT_DEASSIGN_FLAG; ++ ++ ret = irqfd(vm_fd, event_fd, 0, gsi, flags); ++ if (ret < 0) { ++ error_report("Failed to unregister irqfd: gsi=%u", gsi); ++ return -errno; ++ } ++ return 0; ++} ++ ++static int irqchip_update_irqfd_notifier_gsi(const EventNotifier *event, ++ const EventNotifier *resample, ++ int virq, bool add) ++{ ++ int fd = event_notifier_get_fd(event); ++ int rfd = resample ? event_notifier_get_fd(resample) : -1; ++ int vm_fd = mshv_state->vm; ++ ++ trace_mshv_irqchip_update_irqfd_notifier_gsi(fd, rfd, virq, add); ++ ++ if (!add) { ++ return unregister_irqfd(vm_fd, fd, virq); ++ } ++ ++ if (rfd > 0) { ++ return register_irqfd_with_resample(vm_fd, fd, rfd, virq); ++ } ++ ++ return register_irqfd(vm_fd, fd, virq); ++} ++ ++ ++int mshv_irqchip_add_msi_route(int vector, PCIDevice *dev) ++{ ++ MSIMessage msg = { 0, 0 }; ++ int virq = 0; ++ ++ if (pci_available && dev) { ++ msg = pci_get_msi_message(dev, vector); ++ virq = add_msi_routing(msg.address, le32_to_cpu(msg.data)); ++ } ++ ++ return virq; ++} ++ ++void mshv_irqchip_release_virq(int virq) ++{ ++ remove_msi_routing(virq); ++} ++ ++int mshv_irqchip_update_msi_route(int virq, MSIMessage msg, PCIDevice *dev) ++{ ++ int ret; ++ ++ ret = set_msi_routing(virq, msg.address, le32_to_cpu(msg.data)); ++ if (ret < 0) { ++ error_report("Failed to set msi routing"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++int mshv_request_interrupt(MshvState *mshv_state, uint32_t interrupt_type, uint32_t vector, ++ uint32_t vp_index, bool logical_dest_mode, ++ bool level_triggered) ++{ ++ int ret; ++ int vm_fd = mshv_state->vm; ++ ++ if (vector == 0) { ++ warn_report("Ignoring request for interrupt vector 0"); ++ return 0; ++ } ++ ++ union hv_interrupt_control control = { ++ .interrupt_type = interrupt_type, ++ .level_triggered = level_triggered, ++ .logical_dest_mode = logical_dest_mode, ++ .rsvd = 0, ++ }; ++ ++ struct hv_input_assert_virtual_interrupt arg = {0}; ++ arg.control = control; ++ arg.dest_addr = (uint64_t)vp_index; ++ arg.vector = vector; ++ ++ struct mshv_root_hvcall args = {0}; ++ args.code = HVCALL_ASSERT_VIRTUAL_INTERRUPT; ++ args.in_sz = sizeof(arg); ++ args.in_ptr = (uint64_t)&arg; ++ ++ ret = mshv_hvcall(vm_fd, &args); ++ if (ret < 0) { ++ error_report("Failed to request interrupt"); ++ return -errno; ++ } ++ return 0; ++} ++ ++void mshv_irqchip_commit_routes(void) ++{ ++ int ret; ++ int vm_fd = mshv_state->vm; ++ ++ ret = commit_msi_routing_table(vm_fd); ++ if (ret < 0) { ++ error_report("Failed to commit msi routing table"); ++ abort(); ++ } ++} ++ ++int mshv_irqchip_add_irqfd_notifier_gsi(const EventNotifier *event, ++ const EventNotifier *resample, ++ int virq) ++{ ++ return irqchip_update_irqfd_notifier_gsi(event, resample, virq, true); ++} ++ ++int mshv_irqchip_remove_irqfd_notifier_gsi(const EventNotifier *event, ++ int virq) ++{ ++ return irqchip_update_irqfd_notifier_gsi(event, NULL, virq, false); ++} ++ ++int mshv_reserve_ioapic_msi_routes(int vm_fd) ++{ ++ int ret, gsi; ++ ++ /* ++ * Reserve GSI 0-23 for IOAPIC pins, to avoid conflicts of legacy ++ * peripherals with MSI-X devices ++ */ ++ for (gsi = 0; gsi < IOAPIC_NUM_PINS; gsi++) { ++ ret = add_msi_routing(0, 0); ++ if (ret < 0) { ++ error_report("Failed to reserve GSI %d", gsi); ++ return -1; ++ } ++ } ++ ++ ret = commit_msi_routing_table(vm_fd); ++ if (ret < 0) { ++ error_report("Failed to commit reserved IOAPIC MSI routes"); ++ return -1; ++ } ++ ++ return 0; ++} +diff --git a/accel/mshv/mem.c b/accel/mshv/mem.c +index 9889918c31..a0a40eb333 100644 +--- a/accel/mshv/mem.c ++++ b/accel/mshv/mem.c +@@ -12,14 +12,137 @@ + + #include "qemu/osdep.h" + #include "qemu/error-report.h" ++#include "linux/mshv.h" + #include "system/address-spaces.h" + #include "system/mshv.h" + #include "system/mshv_int.h" ++#include "exec/memattrs.h" ++#include ++#include "trace.h" ++ ++static int set_guest_memory(int vm_fd, ++ const struct mshv_user_mem_region *region) ++{ ++ int ret; ++ ++ ret = ioctl(vm_fd, MSHV_SET_GUEST_MEMORY, region); ++ if (ret < 0) { ++ error_report("failed to set guest memory"); ++ return -errno; ++ } ++ ++ return 0; ++} ++ ++static int map_or_unmap(int vm_fd, const MshvMemoryRegion *mr, bool map) ++{ ++ struct mshv_user_mem_region region = {0}; ++ ++ region.guest_pfn = mr->guest_phys_addr >> MSHV_PAGE_SHIFT; ++ region.size = mr->memory_size; ++ region.userspace_addr = mr->userspace_addr; ++ ++ if (!map) { ++ region.flags |= (1 << MSHV_SET_MEM_BIT_UNMAP); ++ trace_mshv_unmap_memory(mr->userspace_addr, mr->guest_phys_addr, ++ mr->memory_size); ++ return set_guest_memory(vm_fd, ®ion); ++ } ++ ++ region.flags = BIT(MSHV_SET_MEM_BIT_EXECUTABLE); ++ if (!mr->readonly) { ++ region.flags |= BIT(MSHV_SET_MEM_BIT_WRITABLE); ++ } ++ ++ trace_mshv_map_memory(mr->userspace_addr, mr->guest_phys_addr, ++ mr->memory_size); ++ return set_guest_memory(vm_fd, ®ion); ++} ++ ++static int set_memory(const MshvMemoryRegion *mshv_mr, bool add) ++{ ++ int ret = 0; ++ ++ if (!mshv_mr) { ++ error_report("Invalid mshv_mr"); ++ return -1; ++ } ++ ++ trace_mshv_set_memory(add, mshv_mr->guest_phys_addr, ++ mshv_mr->memory_size, ++ mshv_mr->userspace_addr, mshv_mr->readonly, ++ ret); ++ return map_or_unmap(mshv_state->vm, mshv_mr, add); ++} ++ ++/* ++ * Calculate and align the start address and the size of the section. ++ * Return the size. If the size is 0, the aligned section is empty. ++ */ ++static hwaddr align_section(MemoryRegionSection *section, hwaddr *start) ++{ ++ hwaddr size = int128_get64(section->size); ++ hwaddr delta, aligned; ++ ++ /* ++ * works in page size chunks, but the function may be called ++ * with sub-page size and unaligned start address. Pad the start ++ * address to next and truncate size to previous page boundary. ++ */ ++ aligned = ROUND_UP(section->offset_within_address_space, ++ qemu_real_host_page_size()); ++ delta = aligned - section->offset_within_address_space; ++ *start = aligned; ++ if (delta > size) { ++ return 0; ++ } ++ ++ return (size - delta) & qemu_real_host_page_mask(); ++} + + void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, + bool add) + { +- error_report("unimplemented"); +- abort(); +-} ++ int ret = 0; ++ MemoryRegion *area = section->mr; ++ bool writable = !area->readonly && !area->rom_device; ++ hwaddr start_addr, mr_offset, size; ++ void *ram; ++ MshvMemoryRegion mshv_mr = {0}; ++ ++ size = align_section(section, &start_addr); ++ trace_mshv_set_phys_mem(add, section->mr->name, start_addr); ++ ++ /* ++ * If the memory device is a writable non-ram area, we do not ++ * want to map it into the guest memory. If it is not a ROM device, ++ * we want to remove mshv memory mapping, so accesses will trap. ++ */ ++ if (!memory_region_is_ram(area)) { ++ if (writable) { ++ return; ++ } else if (!area->romd_mode) { ++ add = false; ++ } ++ } ++ ++ if (!size) { ++ return; ++ } + ++ mr_offset = section->offset_within_region + start_addr - ++ section->offset_within_address_space; ++ ++ ram = memory_region_get_ram_ptr(area) + mr_offset; ++ ++ mshv_mr.guest_phys_addr = start_addr; ++ mshv_mr.memory_size = size; ++ mshv_mr.readonly = !writable; ++ mshv_mr.userspace_addr = (uint64_t)ram; ++ ++ ret = set_memory(&mshv_mr, add); ++ if (ret < 0) { ++ error_report("Failed to set memory region"); ++ abort(); ++ } ++} +diff --git a/accel/mshv/meson.build b/accel/mshv/meson.build +index 8a6beb3fb1..f88fc8678c 100644 +--- a/accel/mshv/meson.build ++++ b/accel/mshv/meson.build +@@ -1,5 +1,6 @@ + mshv_ss = ss.source_set() + mshv_ss.add(if_true: files( ++ 'irq.c', + 'mem.c', + 'mshv-all.c' + )) +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +index a684a36677..653195c57c 100644 +--- a/accel/mshv/mshv-all.c ++++ b/accel/mshv/mshv-all.c +@@ -7,6 +7,7 @@ + * Ziqiao Zhou + * Magnus Kulke + * Jinank Jain ++ * Wei Liu + * + * SPDX-License-Identifier: GPL-2.0-or-later + * +@@ -23,6 +24,7 @@ + #include "hw/hyperv/hvhdk.h" + #include "hw/hyperv/hvhdk_mini.h" + #include "hw/hyperv/hvgdk.h" ++#include "hw/hyperv/hvgdk_mini.h" + #include "linux/mshv.h" + + #include "qemu/accel.h" +@@ -49,6 +51,175 @@ bool mshv_allowed; + + MshvState *mshv_state; + ++static int init_mshv(int *mshv_fd) ++{ ++ int fd = open("/dev/mshv", O_RDWR | O_CLOEXEC); ++ if (fd < 0) { ++ error_report("Failed to open /dev/mshv: %s", strerror(errno)); ++ return -1; ++ } ++ *mshv_fd = fd; ++ return 0; ++} ++ ++/* freeze 1 to pause, 0 to resume */ ++static int set_time_freeze(int vm_fd, int freeze) ++{ ++ int ret; ++ struct hv_input_set_partition_property in = {0}; ++ in.property_code = HV_PARTITION_PROPERTY_TIME_FREEZE; ++ in.property_value = freeze; ++ ++ struct mshv_root_hvcall args = {0}; ++ args.code = HVCALL_SET_PARTITION_PROPERTY; ++ args.in_sz = sizeof(in); ++ args.in_ptr = (uint64_t)∈ ++ ++ ret = mshv_hvcall(vm_fd, &args); ++ if (ret < 0) { ++ error_report("Failed to set time freeze"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int pause_vm(int vm_fd) ++{ ++ int ret; ++ ++ ret = set_time_freeze(vm_fd, 1); ++ if (ret < 0) { ++ error_report("Failed to pause partition: %s", strerror(errno)); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int resume_vm(int vm_fd) ++{ ++ int ret; ++ ++ ret = set_time_freeze(vm_fd, 0); ++ if (ret < 0) { ++ error_report("Failed to resume partition: %s", strerror(errno)); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int create_partition(int mshv_fd, int *vm_fd) ++{ ++ int ret; ++ struct mshv_create_partition args = {0}; ++ ++ /* Initialize pt_flags with the desired features */ ++ uint64_t pt_flags = (1ULL << MSHV_PT_BIT_LAPIC) | ++ (1ULL << MSHV_PT_BIT_X2APIC) | ++ (1ULL << MSHV_PT_BIT_GPA_SUPER_PAGES); ++ ++ /* Set default isolation type */ ++ uint64_t pt_isolation = MSHV_PT_ISOLATION_NONE; ++ ++ args.pt_flags = pt_flags; ++ args.pt_isolation = pt_isolation; ++ ++ ret = ioctl(mshv_fd, MSHV_CREATE_PARTITION, &args); ++ if (ret < 0) { ++ error_report("Failed to create partition: %s", strerror(errno)); ++ return -1; ++ } ++ ++ *vm_fd = ret; ++ return 0; ++} ++ ++static int set_synthetic_proc_features(int vm_fd) ++{ ++ int ret; ++ struct hv_input_set_partition_property in = {0}; ++ union hv_partition_synthetic_processor_features features = {0}; ++ ++ /* Access the bitfield and set the desired features */ ++ features.hypervisor_present = 1; ++ features.hv1 = 1; ++ features.access_partition_reference_counter = 1; ++ features.access_synic_regs = 1; ++ features.access_synthetic_timer_regs = 1; ++ features.access_partition_reference_tsc = 1; ++ features.access_frequency_regs = 1; ++ features.access_intr_ctrl_regs = 1; ++ features.access_vp_index = 1; ++ features.access_hypercall_regs = 1; ++ features.tb_flush_hypercalls = 1; ++ features.synthetic_cluster_ipi = 1; ++ features.direct_synthetic_timers = 1; ++ ++ mshv_arch_amend_proc_features(&features); ++ ++ in.property_code = HV_PARTITION_PROPERTY_SYNTHETIC_PROC_FEATURES; ++ in.property_value = features.as_uint64[0]; ++ ++ struct mshv_root_hvcall args = {0}; ++ args.code = HVCALL_SET_PARTITION_PROPERTY; ++ args.in_sz = sizeof(in); ++ args.in_ptr = (uint64_t)∈ ++ ++ trace_mshv_hvcall_args("synthetic_proc_features", args.code, args.in_sz); ++ ++ ret = mshv_hvcall(vm_fd, &args); ++ if (ret < 0) { ++ error_report("Failed to set synthethic proc features"); ++ return -errno; ++ } ++ return 0; ++} ++ ++static int initialize_vm(int vm_fd) ++{ ++ int ret = ioctl(vm_fd, MSHV_INITIALIZE_PARTITION); ++ if (ret < 0) { ++ error_report("Failed to initialize partition: %s", strerror(errno)); ++ return -1; ++ } ++ return 0; ++} ++ ++static int create_vm(int mshv_fd, int *vm_fd) ++{ ++ int ret = create_partition(mshv_fd, vm_fd); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ ret = set_synthetic_proc_features(*vm_fd); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ ret = initialize_vm(*vm_fd); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ ret = mshv_reserve_ioapic_msi_routes(*vm_fd); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ ret = mshv_arch_post_init_vm(*vm_fd); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ /* Always create a frozen partition */ ++ pause_vm(*vm_fd); ++ ++ return 0; ++} ++ + static void mem_region_add(MemoryListener *listener, + MemoryRegionSection *section) + { +@@ -67,11 +238,124 @@ static void mem_region_del(MemoryListener *listener, + memory_region_unref(section->mr); + } + ++typedef enum { ++ DATAMATCH_NONE, ++ DATAMATCH_U32, ++ DATAMATCH_U64, ++} DatamatchTag; ++ ++typedef struct { ++ DatamatchTag tag; ++ union { ++ uint32_t u32; ++ uint64_t u64; ++ } value; ++} Datamatch; ++ ++/* flags: determine whether to de/assign */ ++static int ioeventfd(int vm_fd, int event_fd, uint64_t addr, Datamatch dm, ++ uint32_t flags) ++{ ++ struct mshv_user_ioeventfd args = {0}; ++ args.fd = event_fd; ++ args.addr = addr; ++ args.flags = flags; ++ ++ if (dm.tag == DATAMATCH_NONE) { ++ args.datamatch = 0; ++ } else { ++ flags |= BIT(MSHV_IOEVENTFD_BIT_DATAMATCH); ++ args.flags = flags; ++ if (dm.tag == DATAMATCH_U64) { ++ args.len = sizeof(uint64_t); ++ args.datamatch = dm.value.u64; ++ } else { ++ args.len = sizeof(uint32_t); ++ args.datamatch = dm.value.u32; ++ } ++ } ++ ++ return ioctl(vm_fd, MSHV_IOEVENTFD, &args); ++} ++ ++static int unregister_ioevent(int vm_fd, int event_fd, uint64_t mmio_addr) ++{ ++ uint32_t flags = 0; ++ Datamatch dm = {0}; ++ ++ flags |= BIT(MSHV_IOEVENTFD_BIT_DEASSIGN); ++ dm.tag = DATAMATCH_NONE; ++ ++ return ioeventfd(vm_fd, event_fd, mmio_addr, dm, flags); ++} ++ ++static int register_ioevent(int vm_fd, int event_fd, uint64_t mmio_addr, ++ uint64_t val, bool is_64bit, bool is_datamatch) ++{ ++ uint32_t flags = 0; ++ Datamatch dm = {0}; ++ ++ if (!is_datamatch) { ++ dm.tag = DATAMATCH_NONE; ++ } else if (is_64bit) { ++ dm.tag = DATAMATCH_U64; ++ dm.value.u64 = val; ++ } else { ++ dm.tag = DATAMATCH_U32; ++ dm.value.u32 = val; ++ } ++ ++ return ioeventfd(vm_fd, event_fd, mmio_addr, dm, flags); ++} ++ ++static void mem_ioeventfd_add(MemoryListener *listener, ++ MemoryRegionSection *section, ++ bool match_data, uint64_t data, ++ EventNotifier *e) ++{ ++ int fd = event_notifier_get_fd(e); ++ int ret; ++ bool is_64 = int128_get64(section->size) == 8; ++ uint64_t addr = section->offset_within_address_space; ++ ++ trace_mshv_mem_ioeventfd_add(addr, int128_get64(section->size), data); ++ ++ ret = register_ioevent(mshv_state->vm, fd, addr, data, is_64, match_data); ++ ++ if (ret < 0) { ++ error_report("Failed to register ioeventfd: %s (%d)", strerror(-ret), ++ -ret); ++ abort(); ++ } ++} ++ ++static void mem_ioeventfd_del(MemoryListener *listener, ++ MemoryRegionSection *section, ++ bool match_data, uint64_t data, ++ EventNotifier *e) ++{ ++ int fd = event_notifier_get_fd(e); ++ int ret; ++ uint64_t addr = section->offset_within_address_space; ++ ++ trace_mshv_mem_ioeventfd_del(section->offset_within_address_space, ++ int128_get64(section->size), data); ++ ++ ret = unregister_ioevent(mshv_state->vm, fd, addr); ++ if (ret < 0) { ++ error_report("Failed to unregister ioeventfd: %s (%d)", strerror(-ret), ++ -ret); ++ abort(); ++ } ++} ++ + static MemoryListener mshv_memory_listener = { + .name = "mshv", + .priority = MEMORY_LISTENER_PRIORITY_ACCEL, + .region_add = mem_region_add, + .region_del = mem_region_del, ++ .eventfd_add = mem_ioeventfd_add, ++ .eventfd_del = mem_ioeventfd_del, + }; + + static MemoryListener mshv_io_listener = { +@@ -97,15 +381,57 @@ static void register_mshv_memory_listener(MshvState *s, MshvMemoryListener *mml, + } + } + ++int mshv_hvcall(int fd, const struct mshv_root_hvcall *args) ++{ ++ int ret = 0; ++ ++ ret = ioctl(fd, MSHV_ROOT_HVCALL, args); ++ if (ret < 0) { ++ error_report("Failed to perform hvcall: %s", strerror(errno)); ++ return -1; ++ } ++ return ret; ++} ++ ++ + static int mshv_init(AccelState *as, MachineState *ms) + { + MshvState *s; ++ int mshv_fd, vm_fd, ret; ++ ++ if (mshv_state) { ++ warn_report("MSHV accelerator already initialized"); ++ return 0; ++ } ++ + s = MSHV_STATE(as); + + accel_blocker_init(); + + s->vm = 0; + ++ ret = init_mshv(&mshv_fd); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ mshv_init_msicontrol(); ++ ++ ret = create_vm(mshv_fd, &vm_fd); ++ if (ret < 0) { ++ close(mshv_fd); ++ return -1; ++ } ++ ++ ret = resume_vm(vm_fd); ++ if (ret < 0) { ++ close(mshv_fd); ++ close(vm_fd); ++ return -1; ++ } ++ ++ s->vm = vm_fd; ++ s->fd = mshv_fd; + s->nr_as = 1; + s->as = g_new0(MshvAddressSpace, s->nr_as); + +diff --git a/accel/mshv/trace-events b/accel/mshv/trace-events +new file mode 100644 +index 0000000000..6130c4abf8 +--- /dev/null ++++ b/accel/mshv/trace-events +@@ -0,0 +1,26 @@ ++# Authors: Ziqiao Zhou ++# Magnus Kulke ++# ++# SPDX-License-Identifier: GPL-2.0-or-later ++ ++mshv_set_memory(bool add, uint64_t gpa, uint64_t size, uint64_t user_addr, bool readonly, int ret) "add=%d gpa=0x%" PRIx64 " size=0x%" PRIx64 " user=0x%" PRIx64 " readonly=%d result=%d" ++mshv_mem_ioeventfd_add(uint64_t addr, uint32_t size, uint32_t data) "addr=0x%" PRIx64 " size=%d data=0x%x" ++mshv_mem_ioeventfd_del(uint64_t addr, uint32_t size, uint32_t data) "addr=0x%" PRIx64 " size=%d data=0x%x" ++ ++mshv_hvcall_args(const char* hvcall, uint16_t code, uint16_t in_sz) "built args for '%s' code: %d in_sz: %d" ++ ++mshv_handle_interrupt(uint32_t cpu, int mask) "cpu_index=%d mask=0x%x" ++mshv_set_msi_routing(uint32_t gsi, uint64_t addr, uint32_t data) "gsi=%d addr=0x%" PRIx64 " data=0x%x" ++mshv_remove_msi_routing(uint32_t gsi) "gsi=%d" ++mshv_add_msi_routing(uint64_t addr, uint32_t data) "addr=0x%" PRIx64 " data=0x%x" ++mshv_commit_msi_routing_table(int vm_fd, int len) "vm_fd=%d table_size=%d" ++mshv_register_irqfd(int vm_fd, int event_fd, uint32_t gsi) "vm_fd=%d event_fd=%d gsi=%d" ++mshv_irqchip_update_irqfd_notifier_gsi(int event_fd, int resample_fd, int virq, bool add) "event_fd=%d resample_fd=%d virq=%d add=%d" ++ ++mshv_insn_fetch(uint64_t addr, size_t size) "gpa=0x%" PRIx64 " size=%zu" ++mshv_mem_write(uint64_t addr, size_t size) "\tgpa=0x%" PRIx64 " size=%zu" ++mshv_mem_read(uint64_t addr, size_t size) "\tgpa=0x%" PRIx64 " size=%zu" ++mshv_map_memory(uint64_t userspace_addr, uint64_t gpa, uint64_t size) "\tu_a=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%08" PRIx64 ++mshv_unmap_memory(uint64_t userspace_addr, uint64_t gpa, uint64_t size) "\tu_a=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%08" PRIx64 ++mshv_set_phys_mem(bool add, const char *name, uint64_t gpa) "\tadd=%d name=%s gpa=0x%010" PRIx64 ++mshv_handle_mmio(uint64_t gva, uint64_t gpa, uint64_t size, uint8_t access_type) "\tgva=0x%" PRIx64 " gpa=0x%010" PRIx64 " size=0x%" PRIx64 " access_type=%d" +diff --git a/accel/mshv/trace.h b/accel/mshv/trace.h +new file mode 100644 +index 0000000000..0dca48f917 +--- /dev/null ++++ b/accel/mshv/trace.h +@@ -0,0 +1,14 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: ++ * Ziqiao Zhou ++ * Magnus Kulke ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ * ++ */ ++ ++#include "trace/trace-accel_mshv.h" +diff --git a/accel/stubs/meson.build b/accel/stubs/meson.build +index 9dfc4f9dda..48eccd1b86 100644 +--- a/accel/stubs/meson.build ++++ b/accel/stubs/meson.build +@@ -5,5 +5,6 @@ system_stubs_ss.add(when: 'CONFIG_TCG', if_false: files('tcg-stub.c')) + system_stubs_ss.add(when: 'CONFIG_HVF', if_false: files('hvf-stub.c')) + system_stubs_ss.add(when: 'CONFIG_NVMM', if_false: files('nvmm-stub.c')) + system_stubs_ss.add(when: 'CONFIG_WHPX', if_false: files('whpx-stub.c')) ++system_stubs_ss.add(when: 'CONFIG_MSHV', if_false: files('mshv-stub.c')) + + specific_ss.add_all(when: ['CONFIG_SYSTEM_ONLY'], if_true: system_stubs_ss) +diff --git a/accel/stubs/mshv-stub.c b/accel/stubs/mshv-stub.c +new file mode 100644 +index 0000000000..e499b199d9 +--- /dev/null ++++ b/accel/stubs/mshv-stub.c +@@ -0,0 +1,44 @@ ++/* ++ * QEMU MSHV stub ++ * ++ * Copyright Red Hat, Inc. 2025 ++ * ++ * Author: Paolo Bonzini ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "hw/pci/msi.h" ++#include "system/mshv.h" ++ ++bool mshv_allowed; ++ ++int mshv_irqchip_add_msi_route(int vector, PCIDevice *dev) ++{ ++ return -ENOSYS; ++} ++ ++void mshv_irqchip_release_virq(int virq) ++{ ++} ++ ++int mshv_irqchip_update_msi_route(int virq, MSIMessage msg, PCIDevice *dev) ++{ ++ return -ENOSYS; ++} ++ ++void mshv_irqchip_commit_routes(void) ++{ ++} ++ ++int mshv_irqchip_add_irqfd_notifier_gsi(const EventNotifier *n, ++ const EventNotifier *rn, int virq) ++{ ++ return -ENOSYS; ++} ++ ++int mshv_irqchip_remove_irqfd_notifier_gsi(const EventNotifier *n, int virq) ++{ ++ return -ENOSYS; ++} +diff --git a/hw/intc/apic.c b/hw/intc/apic.c +index bcb103560c..6d7859640c 100644 +--- a/hw/intc/apic.c ++++ b/hw/intc/apic.c +@@ -27,6 +27,7 @@ + #include "hw/pci/msi.h" + #include "qemu/host-utils.h" + #include "system/kvm.h" ++#include "system/mshv.h" + #include "trace.h" + #include "hw/i386/apic-msidef.h" + #include "qapi/error.h" +@@ -932,6 +933,13 @@ static void apic_send_msi(MSIMessage *msi) + uint8_t trigger_mode = (data >> MSI_DATA_TRIGGER_SHIFT) & 0x1; + uint8_t delivery = (data >> MSI_DATA_DELIVERY_MODE_SHIFT) & 0x7; + /* XXX: Ignore redirection hint. */ ++#ifdef CONFIG_MSHV ++ if (mshv_enabled()) { ++ mshv_request_interrupt(mshv_state, delivery, vector, dest, ++ dest_mode, trigger_mode); ++ return; ++ } ++#endif + apic_deliver_irq(dest, dest_mode, delivery, vector, trigger_mode); + } + +diff --git a/include/system/mshv.h b/include/system/mshv.h +index 434ea9682e..1011e81df4 100644 +--- a/include/system/mshv.h ++++ b/include/system/mshv.h +@@ -31,18 +31,27 @@ + #define CONFIG_MSHV_IS_POSSIBLE + #endif + ++#define MSHV_MAX_MSI_ROUTES 4096 ++ ++#define MSHV_PAGE_SHIFT 12 ++ + #ifdef CONFIG_MSHV_IS_POSSIBLE + extern bool mshv_allowed; + #define mshv_enabled() (mshv_allowed) ++#define mshv_msi_via_irqfd_enabled() mshv_enabled() + #else /* CONFIG_MSHV_IS_POSSIBLE */ + #define mshv_enabled() false +-#endif + #define mshv_msi_via_irqfd_enabled() false ++#endif + + typedef struct MshvState MshvState; + extern MshvState *mshv_state; + + /* interrupt */ ++int mshv_request_interrupt(MshvState *mshv_state, uint32_t interrupt_type, uint32_t vector, ++ uint32_t vp_index, bool logical_destination_mode, ++ bool level_triggered); ++ + int mshv_irqchip_add_msi_route(int vector, PCIDevice *dev); + int mshv_irqchip_update_msi_route(int virq, MSIMessage msg, PCIDevice *dev); + void mshv_irqchip_commit_routes(void); +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index cfa177ff72..b36124a0ea 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -36,10 +36,35 @@ struct MshvState { + /* number of listeners */ + int nr_as; + MshvAddressSpace *as; ++ int fd; + }; + ++typedef struct MshvMsiControl { ++ bool updated; ++ GHashTable *gsi_routes; ++} MshvMsiControl; ++ ++void mshv_arch_amend_proc_features( ++ union hv_partition_synthetic_processor_features *features); ++int mshv_arch_post_init_vm(int vm_fd); ++ ++#if defined COMPILING_PER_TARGET && defined CONFIG_MSHV_IS_POSSIBLE ++int mshv_hvcall(int fd, const struct mshv_root_hvcall *args); ++#endif ++ + /* memory */ ++typedef struct MshvMemoryRegion { ++ uint64_t guest_phys_addr; ++ uint64_t memory_size; ++ uint64_t userspace_addr; ++ bool readonly; ++} MshvMemoryRegion; ++ + void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, + bool add); + ++/* interrupt */ ++void mshv_init_msicontrol(void); ++int mshv_reserve_ioapic_msi_routes(int vm_fd); ++ + #endif +diff --git a/meson.build b/meson.build +index 96254f8075..f7b21c818d 100644 +--- a/meson.build ++++ b/meson.build +@@ -3667,6 +3667,7 @@ if have_system + trace_events_subdirs += [ + 'accel/hvf', + 'accel/kvm', ++ 'accel/mshv', + 'audio', + 'backends', + 'backends/tpm', +diff --git a/target/i386/mshv/meson.build b/target/i386/mshv/meson.build +index 8ddaa7c11d..647e5dafb7 100644 +--- a/target/i386/mshv/meson.build ++++ b/target/i386/mshv/meson.build +@@ -1,6 +1,7 @@ + i386_mshv_ss = ss.source_set() + + i386_mshv_ss.add(files( ++ 'mshv-cpu.c', + 'x86.c', + )) + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +new file mode 100644 +index 0000000000..de0c26bc6c +--- /dev/null ++++ b/target/i386/mshv/mshv-cpu.c +@@ -0,0 +1,72 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Ziqiao Zhou ++ * Magnus Kulke ++ * Jinank Jain ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "qemu/error-report.h" ++#include "qemu/typedefs.h" ++ ++#include "system/mshv.h" ++#include "system/mshv_int.h" ++#include "system/address-spaces.h" ++#include "linux/mshv.h" ++#include "hw/hyperv/hvgdk.h" ++#include "hw/hyperv/hvgdk_mini.h" ++#include "hw/hyperv/hvhdk_mini.h" ++ ++#include "trace-accel_mshv.h" ++#include "trace.h" ++ ++void mshv_arch_amend_proc_features( ++ union hv_partition_synthetic_processor_features *features) ++{ ++ features->access_guest_idle_reg = 1; ++} ++ ++/* ++ * Default Microsoft Hypervisor behavior for unimplemented MSR is to send a ++ * fault to the guest if it tries to access it. It is possible to override ++ * this behavior with a more suitable option i.e., ignore writes from the guest ++ * and return zero in attempt to read unimplemented. ++ */ ++static int set_unimplemented_msr_action(int vm_fd) ++{ ++ struct hv_input_set_partition_property in = {0}; ++ struct mshv_root_hvcall args = {0}; ++ ++ in.property_code = HV_PARTITION_PROPERTY_UNIMPLEMENTED_MSR_ACTION; ++ in.property_value = HV_UNIMPLEMENTED_MSR_ACTION_IGNORE_WRITE_READ_ZERO; ++ ++ args.code = HVCALL_SET_PARTITION_PROPERTY; ++ args.in_sz = sizeof(in); ++ args.in_ptr = (uint64_t)∈ ++ ++ trace_mshv_hvcall_args("unimplemented_msr_action", args.code, args.in_sz); ++ ++ int ret = mshv_hvcall(vm_fd, &args); ++ if (ret < 0) { ++ error_report("Failed to set unimplemented MSR action"); ++ return -1; ++ } ++ return 0; ++} ++ ++int mshv_arch_post_init_vm(int vm_fd) ++{ ++ int ret; ++ ++ ret = set_unimplemented_msr_action(vm_fd); ++ if (ret < 0) { ++ error_report("Failed to set unimplemented MSR action"); ++ } ++ ++ return ret; ++} +-- +2.47.3 + diff --git a/kvm-accel-mshv-Register-memory-region-listeners.patch b/kvm-accel-mshv-Register-memory-region-listeners.patch new file mode 100644 index 0000000..044407a --- /dev/null +++ b/kvm-accel-mshv-Register-memory-region-listeners.patch @@ -0,0 +1,172 @@ +From 1c84d5e352e60196218f657292e7f26226ab9647 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:28 +0200 +Subject: [PATCH 10/32] accel/mshv: Register memory region listeners + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [8/30] ecd0d05d8f04c400ad1d9c428f89b23b12889968 + +Add memory listener hooks for the MSHV accelerator to track guest +memory regions. This enables the backend to respond to region +additions, removals and will be used to manage guest memory mappings +inside the hypervisor. + +Actually registering physical memory in the hypervisor is still stubbed +out. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-9-magnuskulke@linux.microsoft.com +[mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 5006ea1344d134356a9b2e1afd521cf8df0c6a85) +Signed-off-by: Igor Mammedov +--- + accel/mshv/mem.c | 25 +++++++++++++++ + accel/mshv/meson.build | 1 + + accel/mshv/mshv-all.c | 67 +++++++++++++++++++++++++++++++++++++-- + include/system/mshv_int.h | 4 +++ + 4 files changed, 95 insertions(+), 2 deletions(-) + create mode 100644 accel/mshv/mem.c + +diff --git a/accel/mshv/mem.c b/accel/mshv/mem.c +new file mode 100644 +index 0000000000..9889918c31 +--- /dev/null ++++ b/accel/mshv/mem.c +@@ -0,0 +1,25 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: ++ * Magnus Kulke ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ * ++ */ ++ ++#include "qemu/osdep.h" ++#include "qemu/error-report.h" ++#include "system/address-spaces.h" ++#include "system/mshv.h" ++#include "system/mshv_int.h" ++ ++void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, ++ bool add) ++{ ++ error_report("unimplemented"); ++ abort(); ++} ++ +diff --git a/accel/mshv/meson.build b/accel/mshv/meson.build +index 4c03ac7921..8a6beb3fb1 100644 +--- a/accel/mshv/meson.build ++++ b/accel/mshv/meson.build +@@ -1,5 +1,6 @@ + mshv_ss = ss.source_set() + mshv_ss.add(if_true: files( ++ 'mem.c', + 'mshv-all.c' + )) + +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +index ae12f0f58b..a684a36677 100644 +--- a/accel/mshv/mshv-all.c ++++ b/accel/mshv/mshv-all.c +@@ -49,10 +49,73 @@ bool mshv_allowed; + + MshvState *mshv_state; + ++static void mem_region_add(MemoryListener *listener, ++ MemoryRegionSection *section) ++{ ++ MshvMemoryListener *mml; ++ mml = container_of(listener, MshvMemoryListener, listener); ++ memory_region_ref(section->mr); ++ mshv_set_phys_mem(mml, section, true); ++} ++ ++static void mem_region_del(MemoryListener *listener, ++ MemoryRegionSection *section) ++{ ++ MshvMemoryListener *mml; ++ mml = container_of(listener, MshvMemoryListener, listener); ++ mshv_set_phys_mem(mml, section, false); ++ memory_region_unref(section->mr); ++} ++ ++static MemoryListener mshv_memory_listener = { ++ .name = "mshv", ++ .priority = MEMORY_LISTENER_PRIORITY_ACCEL, ++ .region_add = mem_region_add, ++ .region_del = mem_region_del, ++}; ++ ++static MemoryListener mshv_io_listener = { ++ .name = "mshv", .priority = MEMORY_LISTENER_PRIORITY_DEV_BACKEND, ++ /* MSHV does not support PIO eventfd */ ++}; ++ ++static void register_mshv_memory_listener(MshvState *s, MshvMemoryListener *mml, ++ AddressSpace *as, int as_id, ++ const char *name) ++{ ++ int i; ++ ++ mml->listener = mshv_memory_listener; ++ mml->listener.name = name; ++ memory_listener_register(&mml->listener, as); ++ for (i = 0; i < s->nr_as; ++i) { ++ if (!s->as[i].as) { ++ s->as[i].as = as; ++ s->as[i].ml = mml; ++ break; ++ } ++ } ++} ++ + static int mshv_init(AccelState *as, MachineState *ms) + { +- error_report("unimplemented"); +- abort(); ++ MshvState *s; ++ s = MSHV_STATE(as); ++ ++ accel_blocker_init(); ++ ++ s->vm = 0; ++ ++ s->nr_as = 1; ++ s->as = g_new0(MshvAddressSpace, s->nr_as); ++ ++ mshv_state = s; ++ ++ register_mshv_memory_listener(s, &s->memory_listener, &address_space_memory, ++ 0, "mshv-memory"); ++ memory_listener_register(&mshv_io_listener, &address_space_io); ++ ++ return 0; + } + + static void mshv_start_vcpu_thread(CPUState *cpu) +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index 132491b599..cfa177ff72 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -38,4 +38,8 @@ struct MshvState { + MshvAddressSpace *as; + }; + ++/* memory */ ++void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, ++ bool add); ++ + #endif +-- +2.47.3 + diff --git a/kvm-accel-mshv-initialize-thread-name.patch b/kvm-accel-mshv-initialize-thread-name.patch new file mode 100644 index 0000000..71554c0 --- /dev/null +++ b/kvm-accel-mshv-initialize-thread-name.patch @@ -0,0 +1,39 @@ +From f102098ed05c9157f2cffb4ffef8324c4f83f005 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Wed, 22 Oct 2025 14:52:48 +0200 +Subject: [PATCH 31/32] accel/mshv: initialize thread name + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [29/30] 3e7286070c5ce162001fd407ebb1fa3920e300fa + +The initialization was dropped when the code was copied from existing +accelerators. Coverity knows (CID 1641400). Fix it. + +Signed-off-by: Paolo Bonzini +(cherry picked from commit 2cd3c1d35a06a62b25afa80c5baf381ce4b56805) +Signed-off-by: Igor Mammedov +--- + accel/mshv/mshv-all.c | 3 +++ + 1 file changed, 3 insertions(+) + +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +index 45174f7c4e..80428d130d 100644 +--- a/accel/mshv/mshv-all.c ++++ b/accel/mshv/mshv-all.c +@@ -596,6 +596,9 @@ static void mshv_start_vcpu_thread(CPUState *cpu) + { + char thread_name[VCPU_THREAD_NAME_SIZE]; + ++ snprintf(thread_name, VCPU_THREAD_NAME_SIZE, "CPU %d/MSHV", ++ cpu->cpu_index); ++ + cpu->thread = g_malloc0(sizeof(QemuThread)); + cpu->halt_cond = g_malloc0(sizeof(QemuCond)); + +-- +2.47.3 + diff --git a/kvm-accel-mshv-use-return-value-of-handle_pio_str_read.patch b/kvm-accel-mshv-use-return-value-of-handle_pio_str_read.patch new file mode 100644 index 0000000..02b98ee --- /dev/null +++ b/kvm-accel-mshv-use-return-value-of-handle_pio_str_read.patch @@ -0,0 +1,42 @@ +From 257a0cba46705253cb708f8b74ceb736b80c2c27 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Wed, 22 Oct 2025 14:54:30 +0200 +Subject: [PATCH 32/32] accel/mshv: use return value of handle_pio_str_read + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [30/30] 60404a4d9211cf711fd4581d9f586510c3897bd4 + +Coverity complains because we assign to ret here but +then never read it again before we overwrite it with +the call to set_x64_registers(). + +Analyzed-by: Peter Maydell +Signed-off-by: Paolo Bonzini +(cherry picked from commit 1557adc82698416bc68033765cfffb0a0b91c6bf) +Signed-off-by: Igor Mammedov +--- + target/i386/mshv/mshv-cpu.c | 4 ++++ + 1 file changed, 4 insertions(+) + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 1f7b9cb37e..1c3db02188 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -1489,6 +1489,10 @@ static int handle_pio_str(CPUState *cpu, hv_x64_io_port_intercept_message *info) + reg_values[0] = info->rsi; + } else { + ret = handle_pio_str_read(cpu, info, repeat, port, direction_flag); ++ if (ret < 0) { ++ error_report("Failed to handle pio str read"); ++ return -1; ++ } + reg_names[0] = HV_X64_REGISTER_RDI; + reg_values[0] = info->rdi; + } +-- +2.47.3 + diff --git a/kvm-aio-warn-about-iohandler_ctx-special-casing.patch b/kvm-aio-warn-about-iohandler_ctx-special-casing.patch deleted file mode 100644 index eeafb8b..0000000 --- a/kvm-aio-warn-about-iohandler_ctx-special-casing.patch +++ /dev/null @@ -1,64 +0,0 @@ -From 6c8da957fd534b3546354a8b8252c01cf9ee3511 Mon Sep 17 00:00:00 2001 -From: Stefan Hajnoczi -Date: Mon, 6 May 2024 15:06:22 -0400 -Subject: [PATCH 14/14] aio: warn about iohandler_ctx special casing - -RH-Author: Kevin Wolf -RH-MergeRequest: 253: Revert "monitor: use aio_co_reschedule_self()" -RH-Jira: RHEL-43409 RHEL-43410 -RH-Acked-by: Miroslav Rezanina -RH-Acked-by: Hanna Czenczek -RH-Commit: [2/2] 895231553731f09f51275c1abbf50c3440fe977f (kmwolf/centos-qemu-kvm) - -The main loop has two AioContexts: qemu_aio_context and iohandler_ctx. -The main loop runs them both, but nested aio_poll() calls on -qemu_aio_context exclude iohandler_ctx. - -Which one should qemu_get_current_aio_context() return when called from -the main loop? Document that it's always qemu_aio_context. - -This has subtle effects on functions that use -qemu_get_current_aio_context(). For example, aio_co_reschedule_self() -does not work when moving from iohandler_ctx to qemu_aio_context because -qemu_get_current_aio_context() does not differentiate these two -AioContexts. - -Document this in order to reduce the chance of future bugs. - -Signed-off-by: Stefan Hajnoczi -Message-ID: <20240506190622.56095-3-stefanha@redhat.com> -Reviewed-by: Kevin Wolf -Signed-off-by: Kevin Wolf -(cherry picked from commit e669e800fc9ef8806af5c5578249ab758a4f8a5a) -Signed-off-by: Kevin Wolf ---- - include/block/aio.h | 6 ++++++ - 1 file changed, 6 insertions(+) - -diff --git a/include/block/aio.h b/include/block/aio.h -index 8378553eb9..4ee81936ed 100644 ---- a/include/block/aio.h -+++ b/include/block/aio.h -@@ -629,6 +629,9 @@ void aio_co_schedule(AioContext *ctx, Coroutine *co); - * - * Move the currently running coroutine to new_ctx. If the coroutine is already - * running in new_ctx, do nothing. -+ * -+ * Note that this function cannot reschedule from iohandler_ctx to -+ * qemu_aio_context. - */ - void coroutine_fn aio_co_reschedule_self(AioContext *new_ctx); - -@@ -661,6 +664,9 @@ void aio_co_enter(AioContext *ctx, Coroutine *co); - * If called from an IOThread this will be the IOThread's AioContext. If - * called from the main thread or with the "big QEMU lock" taken it - * will be the main loop AioContext. -+ * -+ * Note that the return value is never the main loop's iohandler_ctx and the -+ * return value is the main loop AioContext instead. - */ - AioContext *qemu_get_current_aio_context(void); - --- -2.39.3 - diff --git a/kvm-arm-create-new-rhel-10.2-specific-virt-machine-type.patch b/kvm-arm-create-new-rhel-10.2-specific-virt-machine-type.patch new file mode 100644 index 0000000..1d7b92c --- /dev/null +++ b/kvm-arm-create-new-rhel-10.2-specific-virt-machine-type.patch @@ -0,0 +1,50 @@ +From 1e99cb806fb7d6a341bf8f7c238aa7261af7dbd0 Mon Sep 17 00:00:00 2001 +From: Sebastian Ott +Date: Wed, 29 Oct 2025 15:53:41 +0100 +Subject: [PATCH 06/10] arm: create new rhel 10.2 specific virt machine type + +RH-Author: Sebastian Ott +RH-MergeRequest: 416: x86, arm: create new rhel 9.8, 10.2 specific machine types +RH-Jira: RHEL-105826 RHEL-105828 +RH-Acked-by: Eric Auger +RH-Acked-by: Gavin Shan +RH-Acked-by: Cornelia Huck +RH-Commit: [1/4] 829eed93aefddcc4684ddacff3cd6ff3f991f442 (seott1/cos-qemu-kvm) + +Signed-off-by: Sebastian Ott +--- + hw/arm/virt.c | 9 ++++++++- + 1 file changed, 8 insertions(+), 1 deletion(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index d37d1bb3cf..e68979d2c5 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -3688,16 +3688,23 @@ static void virt_machine_4_1_options(MachineClass *mc) + DEFINE_VIRT_MACHINE(4, 1) + #endif /* disabled for RHEL */ + ++static void virt_rhel_machine_10_2_0_options(MachineClass *mc) ++{ ++} ++DEFINE_VIRT_MACHINE_AS_LATEST(10, 2, 0) ++ + static void virt_rhel_machine_10_0_0_options(MachineClass *mc) + { + VirtMachineClass *vmc = VIRT_MACHINE_CLASS(OBJECT_CLASS(mc)); + ++ virt_rhel_machine_10_2_0_options(mc); ++ + /* QEMU 9.1 and earlier have only a stage-1 SMMU, not a nested s1+2 one */ + vmc->no_nested_smmu = true; + compat_props_add(mc->compat_props, hw_compat_rhel_10_2, hw_compat_rhel_10_2_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_1, hw_compat_rhel_10_1_len); + } +-DEFINE_VIRT_MACHINE_AS_LATEST(10, 0, 0) ++DEFINE_VIRT_MACHINE(10, 0, 0) + + static void virt_rhel_machine_9_6_0_options(MachineClass *mc) + { +-- +2.47.3 + diff --git a/kvm-arm-create-new-rhel-9.8-specific-virt-machine-type.patch b/kvm-arm-create-new-rhel-9.8-specific-virt-machine-type.patch new file mode 100644 index 0000000..2327e80 --- /dev/null +++ b/kvm-arm-create-new-rhel-9.8-specific-virt-machine-type.patch @@ -0,0 +1,49 @@ +From c6b6b0d2969f8baf5f09bc1d4953df5e7445820c Mon Sep 17 00:00:00 2001 +From: Sebastian Ott +Date: Wed, 29 Oct 2025 16:05:12 +0100 +Subject: [PATCH 07/10] arm: create new rhel 9.8 specific virt machine type + +RH-Author: Sebastian Ott +RH-MergeRequest: 416: x86, arm: create new rhel 9.8, 10.2 specific machine types +RH-Jira: RHEL-105826 RHEL-105828 +RH-Acked-by: Eric Auger +RH-Acked-by: Gavin Shan +RH-Acked-by: Cornelia Huck +RH-Commit: [2/4] fb3fc4decafd0c6e94ef31c0120b843512089573 (seott1/cos-qemu-kvm) + +Signed-off-by: Sebastian Ott +--- + hw/arm/virt.c | 13 +++++++++---- + 1 file changed, 9 insertions(+), 4 deletions(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index e68979d2c5..dcdd53043e 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -3706,14 +3706,19 @@ static void virt_rhel_machine_10_0_0_options(MachineClass *mc) + } + DEFINE_VIRT_MACHINE(10, 0, 0) + +-static void virt_rhel_machine_9_6_0_options(MachineClass *mc) ++static void virt_rhel_machine_9_8_0_options(MachineClass *mc) + { +- virt_rhel_machine_10_0_0_options(mc); +- ++ /* NB: remember to move these lines to the *latest* RHEL 9 machine */ + compat_props_add(mc->compat_props, arm_rhel9_compat, arm_rhel9_compat_len); +- /* NB: remember to move this line to the *latest* RHEL 9 machine */ + compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } ++DEFINE_VIRT_MACHINE(9, 8, 0) ++ ++static void virt_rhel_machine_9_6_0_options(MachineClass *mc) ++{ ++ virt_rhel_machine_10_0_0_options(mc); ++ virt_rhel_machine_9_8_0_options(mc); ++} + DEFINE_VIRT_MACHINE(9, 6, 0) + + static void virt_rhel_machine_9_4_0_options(MachineClass *mc) +-- +2.47.3 + diff --git a/kvm-arm-fix-oob-access-in-compat-handling.patch b/kvm-arm-fix-oob-access-in-compat-handling.patch new file mode 100644 index 0000000..f27f68b --- /dev/null +++ b/kvm-arm-fix-oob-access-in-compat-handling.patch @@ -0,0 +1,44 @@ +From c2c080f94d9107273650ebdaf50ca78e8cb2f866 Mon Sep 17 00:00:00 2001 +From: Sebastian Ott +Date: Tue, 25 Nov 2025 09:19:30 +0100 +Subject: [PATCH 2/2] arm: fix oob access in compat handling + +RH-Author: Sebastian Ott +RH-MergeRequest: 430: arm: fix oob access in compat handling +RH-Jira: RHEL-130478 +RH-Acked-by: Cornelia Huck +RH-Acked-by: Eric Auger +RH-Acked-by: Gavin Shan +RH-Acked-by: Igor Mammedov +RH-Commit: [1/1] 5d6ab0821b0ffce732862969f275aa50f6917bf1 (seott1/cos-qemu-kvm) + +Upstream: RHEL only +Fixes: 58cba97a "hw/arm/virt: Use ACPI PCI hotplug by default from 10.2 onwards" + +Due to incorrect length information arm_acpi_pci_hp_disabled_compat[] +was accessed out of bounds. This led to the weird side effect that the +1st member of arm_rhel9_compat[] was applied to the virt-rhel10.0 +machine type causing migration failures. Fix the length. + +Signed-off-by: Sebastian Ott +--- + hw/arm/virt.c | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index e8e64fe7fe..1cfb386f64 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -102,7 +102,8 @@ static const size_t arm_virt_compat_len = G_N_ELEMENTS(arm_virt_compat); + GlobalProperty arm_acpi_pci_hp_disabled_compat[] = { + { TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "off" }, + }; +-static const size_t arm_acpi_pci_hp_disabled_compat_len = G_N_ELEMENTS(arm_virt_compat); ++static const size_t arm_acpi_pci_hp_disabled_compat_len = ++ G_N_ELEMENTS(arm_acpi_pci_hp_disabled_compat); + + /* + * RHEL9 kernels have pauth disabled while RHEL10 has it enabled, +-- +2.47.3 + diff --git a/kvm-arm-kvm-report-registers-we-failed-to-set.patch b/kvm-arm-kvm-report-registers-we-failed-to-set.patch new file mode 100644 index 0000000..9f956e9 --- /dev/null +++ b/kvm-arm-kvm-report-registers-we-failed-to-set.patch @@ -0,0 +1,154 @@ +From d635b553683b9a057d7a1a4b7e3348c88dcab6d6 Mon Sep 17 00:00:00 2001 +From: Cornelia Huck +Date: Thu, 11 Sep 2025 17:41:59 +0200 +Subject: [PATCH 1/4] arm/kvm: report registers we failed to set + +RH-Author: Eric Auger +RH-MergeRequest: 410: arm/kvm: report registers we failed to set +RH-Jira: RHEL-119368 +RH-Acked-by: Cornelia Huck +RH-Acked-by: Sebastian Ott +RH-Acked-by: Gavin Shan +RH-Acked-by: Donald Dutile +RH-Commit: [1/1] 82b4496284ff0a4dd2dd0eae7bb1cf114dded61e (eauger1/centos-qemu-kvm) + +If we fail migration because of a mismatch of some registers between +source and destination, the error message is not very informative: + +qemu-system-aarch64: error while loading state for instance 0x0 ofdevice 'cpu' +qemu-system-aarch64: Failed to put registers after init: Invalid argument + +At least try to give the user a hint which registers had a problem, +even if they cannot really do anything about it right now. + +Sample output: + +Could not set register op0:3 op1:0 crn:0 crm:0 op2:0 to c00fac31 (is 413fd0c1) + +We could be even more helpful once we support writable ID registers, +at which point the user might actually be able to configure something +that is migratable. + +Suggested-by: Eric Auger +Reviewed-by: Sebastian Ott +Signed-off-by: Cornelia Huck +Message-id: 20250911154159.158046-1-cohuck@redhat.com +Signed-off-by: Peter Maydell +(cherry picked from commit 19f6dcfe6b8b2a3523362812fc696ab83050d316) +Signed-off-by: Eric Auger +--- + target/arm/kvm.c | 86 ++++++++++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 86 insertions(+) + +diff --git a/target/arm/kvm.c b/target/arm/kvm.c +index 6672344855..c1ec6654ca 100644 +--- a/target/arm/kvm.c ++++ b/target/arm/kvm.c +@@ -900,6 +900,58 @@ bool write_kvmstate_to_list(ARMCPU *cpu) + return ok; + } + ++/* pretty-print a KVM register */ ++#define CP_REG_ARM64_SYSREG_OP(_reg, _op) \ ++ ((uint8_t)((_reg & CP_REG_ARM64_SYSREG_ ## _op ## _MASK) >> \ ++ CP_REG_ARM64_SYSREG_ ## _op ## _SHIFT)) ++ ++static gchar *kvm_print_sve_register_name(uint64_t regidx) ++{ ++ uint16_t sve_reg = regidx & 0x000000000000ffff; ++ ++ if (regidx == KVM_REG_ARM64_SVE_VLS) { ++ return g_strdup_printf("SVE VLS"); ++ } ++ /* zreg, preg, ffr */ ++ switch (sve_reg & 0xfc00) { ++ case 0: ++ return g_strdup_printf("SVE zreg n:%d slice:%d", ++ (sve_reg & 0x03e0) >> 5, sve_reg & 0x001f); ++ case 0x04: ++ return g_strdup_printf("SVE preg n:%d slice:%d", ++ (sve_reg & 0x01e0) >> 5, sve_reg & 0x001f); ++ case 0x06: ++ return g_strdup_printf("SVE ffr slice:%d", sve_reg & 0x001f); ++ default: ++ return g_strdup_printf("SVE ???"); ++ } ++} ++ ++static gchar *kvm_print_register_name(uint64_t regidx) ++{ ++ switch ((regidx & KVM_REG_ARM_COPROC_MASK)) { ++ case KVM_REG_ARM_CORE: ++ return g_strdup_printf("core reg %"PRIx64, regidx); ++ case KVM_REG_ARM_DEMUX: ++ return g_strdup_printf("demuxed reg %"PRIx64, regidx); ++ case KVM_REG_ARM64_SYSREG: ++ return g_strdup_printf("op0:%d op1:%d crn:%d crm:%d op2:%d", ++ CP_REG_ARM64_SYSREG_OP(regidx, OP0), ++ CP_REG_ARM64_SYSREG_OP(regidx, OP1), ++ CP_REG_ARM64_SYSREG_OP(regidx, CRN), ++ CP_REG_ARM64_SYSREG_OP(regidx, CRM), ++ CP_REG_ARM64_SYSREG_OP(regidx, OP2)); ++ case KVM_REG_ARM_FW: ++ return g_strdup_printf("fw reg %d", (int)(regidx & 0xffff)); ++ case KVM_REG_ARM64_SVE: ++ return kvm_print_sve_register_name(regidx); ++ case KVM_REG_ARM_FW_FEAT_BMAP: ++ return g_strdup_printf("fw feat reg %d", (int)(regidx & 0xffff)); ++ default: ++ return g_strdup_printf("%"PRIx64, regidx); ++ } ++} ++ + bool write_list_to_kvmstate(ARMCPU *cpu, int level) + { + CPUState *cs = CPU(cpu); +@@ -927,11 +979,45 @@ bool write_list_to_kvmstate(ARMCPU *cpu, int level) + g_assert_not_reached(); + } + if (ret) { ++ gchar *reg_str = kvm_print_register_name(regidx); ++ + /* We might fail for "unknown register" and also for + * "you tried to set a register which is constant with + * a different value from what it actually contains". + */ + ok = false; ++ switch (ret) { ++ case -ENOENT: ++ error_report("Could not set register %s: unknown to KVM", ++ reg_str); ++ break; ++ case -EINVAL: ++ if ((regidx & KVM_REG_SIZE_MASK) == KVM_REG_SIZE_U32) { ++ if (!kvm_get_one_reg(cs, regidx, &v32)) { ++ error_report("Could not set register %s to %x (is %x)", ++ reg_str, (uint32_t)cpu->cpreg_values[i], ++ v32); ++ } else { ++ error_report("Could not set register %s to %x", ++ reg_str, (uint32_t)cpu->cpreg_values[i]); ++ } ++ } else /* U64 */ { ++ uint64_t v64; ++ ++ if (!kvm_get_one_reg(cs, regidx, &v64)) { ++ error_report("Could not set register %s to %"PRIx64" (is %"PRIx64")", ++ reg_str, cpu->cpreg_values[i], v64); ++ } else { ++ error_report("Could not set register %s to %"PRIx64, ++ reg_str, cpu->cpreg_values[i]); ++ } ++ } ++ break; ++ default: ++ error_report("Could not set register %s: %s", ++ reg_str, strerror(-ret)); ++ } ++ g_free(reg_str); + } + } + return ok; +-- +2.47.3 + diff --git a/kvm-bios-tables-test-Allow-for-smmuv3-test-data.patch b/kvm-bios-tables-test-Allow-for-smmuv3-test-data.patch new file mode 100644 index 0000000..72136a7 --- /dev/null +++ b/kvm-bios-tables-test-Allow-for-smmuv3-test-data.patch @@ -0,0 +1,54 @@ +From b4eeed1e8633df76598de0fe6ca5df4be359222c Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:31 +0100 +Subject: [PATCH 13/16] bios-tables-test: Allow for smmuv3 test data. + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [9/11] cf98e2e7589b794775c1d9c4f564e3cd536b886e (eauger1/centos-qemu-kvm) + +The tests to be added exercise both legacy(iommu=smmuv3) and new +-device arm-smmuv3,.. cases. + +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Reviewed-by: Nicolin Chen +Message-id: 20250829082543.7680-10-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit c69520c13d6ea45a69a7a49361806fa05b19046d) +Signed-off-by: Eric Auger +--- + tests/data/acpi/aarch64/virt/DSDT.smmuv3-dev | 0 + tests/data/acpi/aarch64/virt/DSDT.smmuv3-legacy | 0 + tests/data/acpi/aarch64/virt/IORT.smmuv3-dev | 0 + tests/data/acpi/aarch64/virt/IORT.smmuv3-legacy | 0 + tests/qtest/bios-tables-test-allowed-diff.h | 4 ++++ + 5 files changed, 4 insertions(+) + create mode 100644 tests/data/acpi/aarch64/virt/DSDT.smmuv3-dev + create mode 100644 tests/data/acpi/aarch64/virt/DSDT.smmuv3-legacy + create mode 100644 tests/data/acpi/aarch64/virt/IORT.smmuv3-dev + create mode 100644 tests/data/acpi/aarch64/virt/IORT.smmuv3-legacy + +diff --git a/tests/qtest/bios-tables-test-allowed-diff.h b/tests/qtest/bios-tables-test-allowed-diff.h +index dfb8523c8b..2e3e3ccdce 100644 +--- a/tests/qtest/bios-tables-test-allowed-diff.h ++++ b/tests/qtest/bios-tables-test-allowed-diff.h +@@ -1 +1,5 @@ + /* List of comma-separated changed AML files to ignore */ ++"tests/data/acpi/aarch64/virt/DSDT.smmuv3-legacy", ++"tests/data/acpi/aarch64/virt/DSDT.smmuv3-dev", ++"tests/data/acpi/aarch64/virt/IORT.smmuv3-legacy", ++"tests/data/acpi/aarch64/virt/IORT.smmuv3-dev", +-- +2.47.3 + diff --git a/kvm-block-Expose-block-limits-for-images-in-QMP.patch b/kvm-block-Expose-block-limits-for-images-in-QMP.patch new file mode 100644 index 0000000..4c67a90 --- /dev/null +++ b/kvm-block-Expose-block-limits-for-images-in-QMP.patch @@ -0,0 +1,255 @@ +From c8b07a9c5c6926a63c2db56c3de8046b2160417d Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Fri, 24 Oct 2025 14:30:38 +0200 +Subject: [PATCH 3/6] block: Expose block limits for images in QMP + +RH-Author: Kevin Wolf +RH-MergeRequest: 439: block: Expose block limits in monitor and qemu-img info +RH-Jira: RHEL-110003 +RH-Acked-by: Hanna Czenczek +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/4] a2d45f05abd5bedbad2f10ae4dc7a5cce909a2fe (kmwolf/centos-qemu-kvm) + +This information can be useful both for debugging and for management +tools trying to configure guest devices with the optimal limits +(possibly across multiple hosts). There is no reason not to make it +available, so just add it to BlockNodeInfo. + +Signed-off-by: Kevin Wolf +Reviewed-by: Eric Blake +Reviewed-by: Hanna Czenczek +Message-ID: <20251024123041.51254-3-kwolf@redhat.com> +Signed-off-by: Kevin Wolf +(cherry picked from commit d2634e18286a5772f04cc724e64f8b16a2124587) +Signed-off-by: Kevin Wolf +--- + block/qapi.c | 34 ++++++++++++++-- + qapi/block-core.json | 66 ++++++++++++++++++++++++++++++++ + tests/qemu-iotests/184 | 5 ++- + tests/qemu-iotests/184.out | 8 ---- + tests/qemu-iotests/common.filter | 3 +- + 5 files changed, 102 insertions(+), 14 deletions(-) + +diff --git a/block/qapi.c b/block/qapi.c +index 12fbf8d1b7..54521d0a68 100644 +--- a/block/qapi.c ++++ b/block/qapi.c +@@ -235,7 +235,8 @@ int bdrv_query_snapshot_info_list(BlockDriverState *bs, + * in @info, setting @errp on error. + */ + static void GRAPH_RDLOCK +-bdrv_do_query_node_info(BlockDriverState *bs, BlockNodeInfo *info, Error **errp) ++bdrv_do_query_node_info(BlockDriverState *bs, BlockNodeInfo *info, bool limits, ++ Error **errp) + { + int64_t size; + const char *backing_filename; +@@ -269,6 +270,33 @@ bdrv_do_query_node_info(BlockDriverState *bs, BlockNodeInfo *info, Error **errp) + info->dirty_flag = bdi.is_dirty; + info->has_dirty_flag = true; + } ++ ++ if (limits) { ++ info->limits = g_new(BlockLimitsInfo, 1); ++ *info->limits = (BlockLimitsInfo) { ++ .request_alignment = bs->bl.request_alignment, ++ .has_max_discard = bs->bl.max_pdiscard != 0, ++ .max_discard = bs->bl.max_pdiscard, ++ .has_discard_alignment = bs->bl.pdiscard_alignment != 0, ++ .discard_alignment = bs->bl.pdiscard_alignment, ++ .has_max_write_zeroes = bs->bl.max_pwrite_zeroes != 0, ++ .max_write_zeroes = bs->bl.max_pwrite_zeroes, ++ .has_write_zeroes_alignment = bs->bl.pwrite_zeroes_alignment != 0, ++ .write_zeroes_alignment = bs->bl.pwrite_zeroes_alignment, ++ .has_opt_transfer = bs->bl.opt_transfer != 0, ++ .opt_transfer = bs->bl.opt_transfer, ++ .has_max_transfer = bs->bl.max_transfer != 0, ++ .max_transfer = bs->bl.max_transfer, ++ .has_max_hw_transfer = bs->bl.max_hw_transfer != 0, ++ .max_hw_transfer = bs->bl.max_hw_transfer, ++ .max_iov = bs->bl.max_iov, ++ .has_max_hw_iov = bs->bl.max_hw_iov != 0, ++ .max_hw_iov = bs->bl.max_hw_iov, ++ .min_mem_alignment = bs->bl.min_mem_alignment, ++ .opt_mem_alignment = bs->bl.opt_mem_alignment, ++ }; ++ } ++ + info->format_specific = bdrv_get_specific_info(bs, &err); + if (err) { + error_propagate(errp, err); +@@ -343,7 +371,7 @@ void bdrv_query_image_info(BlockDriverState *bs, + ImageInfo *info; + + info = g_new0(ImageInfo, 1); +- bdrv_do_query_node_info(bs, qapi_ImageInfo_base(info), errp); ++ bdrv_do_query_node_info(bs, qapi_ImageInfo_base(info), true, errp); + if (*errp) { + goto fail; + } +@@ -397,7 +425,7 @@ void bdrv_query_block_graph_info(BlockDriverState *bs, + BdrvChild *c; + + info = g_new0(BlockGraphInfo, 1); +- bdrv_do_query_node_info(bs, qapi_BlockGraphInfo_base(info), errp); ++ bdrv_do_query_node_info(bs, qapi_BlockGraphInfo_base(info), false, errp); + if (*errp) { + goto fail; + } +diff --git a/qapi/block-core.json b/qapi/block-core.json +index dc6eb4ae23..2c037183f0 100644 +--- a/qapi/block-core.json ++++ b/qapi/block-core.json +@@ -275,6 +275,69 @@ + 'file': 'ImageInfoSpecificFileWrapper' + } } + ++## ++# @BlockLimitsInfo: ++# ++# @request-alignment: Alignment requirement, in bytes, for ++# offset/length of I/O requests. ++# ++# @max-discard: Maximum number of bytes that can be discarded at once. ++# If not present, there is no specific maximum. ++# ++# @discard-alignment: Optimal alignment for discard requests in bytes. ++# Note that this doesn't have to be a power of two. If not ++# present, discards don't have a alignment requirement different ++# from @request-alignment. ++# ++# @max-write-zeroes: Maximum number of bytes that can be zeroed out at ++# once. If not present, there is no specific maximum. ++# ++# @write-zeroes-alignment: Optimal alignment for write zeroes requests ++# in bytes. Note that this doesn't have to be a power of two. If ++# not present, write_zeroes doesn't have a alignment requirement ++# different from @request-alignment. ++# ++# @opt-transfer: Optimal transfer length in bytes. If not present, ++# there is no preferred size. ++# ++# @max-transfer: Maximal transfer length in bytes. If not present, ++# there is no specific maximum. ++# ++# @max-hw-transfer: Maximal hardware transfer length in bytes. ++# Applies whenever transfers to the device bypass the kernel I/O ++# scheduler, for example with SG_IO. If not present, there is no ++# specific maximum. ++# ++# @max-iov: Maximum number of scatter/gather elements ++# ++# @max-hw-iov: Maximum number of scatter/gather elements allowed by ++# the hardware. Applies whenever transfers to the device bypass ++# the kernel I/O scheduler, for example with SG_IO. If not ++# present, the hardware limits is unknown and @max-iov is always ++# used. ++# ++# @min-mem-alignment: Minimal required memory alignment in bytes for ++# zero-copy I/O to succeed. For unaligned requests, a bounce ++# buffer will be used. ++# ++# @opt-mem-alignment: Optimal memory alignment in bytes. This is the ++# alignment used for any buffer allocations QEMU performs ++# internally. ++## ++{ 'struct': 'BlockLimitsInfo', ++ 'data': { 'request-alignment': 'uint32', ++ '*max-discard': 'uint64', ++ '*discard-alignment': 'uint32', ++ '*max-write-zeroes': 'uint64', ++ '*write-zeroes-alignment': 'uint32', ++ '*opt-transfer': 'uint32', ++ '*max-transfer': 'uint32', ++ '*max-hw-transfer': 'uint32', ++ 'max-iov': 'int', ++ '*max-hw-iov': 'int', ++ 'min-mem-alignment': 'size', ++ 'opt-mem-alignment': 'size' } } ++ + ## + # @BlockNodeInfo: + # +@@ -304,6 +367,8 @@ + # + # @snapshots: list of VM snapshots + # ++# @limits: block limits that are used for I/O on the node (Since 10.2) ++# + # @format-specific: structure supplying additional format-specific + # information (since 1.7) + # +@@ -315,6 +380,7 @@ + '*cluster-size': 'int', '*encrypted': 'bool', '*compressed': 'bool', + '*backing-filename': 'str', '*full-backing-filename': 'str', + '*backing-filename-format': 'str', '*snapshots': ['SnapshotInfo'], ++ '*limits': 'BlockLimitsInfo', + '*format-specific': 'ImageInfoSpecific' } } + + ## +diff --git a/tests/qemu-iotests/184 b/tests/qemu-iotests/184 +index e4cbcd8634..6d0afe9d38 100755 +--- a/tests/qemu-iotests/184 ++++ b/tests/qemu-iotests/184 +@@ -45,8 +45,9 @@ do_run_qemu() + + run_qemu() + { +- do_run_qemu "$@" 2>&1 | _filter_testdir | _filter_qemu | _filter_qmp\ +- | _filter_qemu_io | _filter_generated_node_ids ++ do_run_qemu "$@" 2>&1 | _filter_testdir | _filter_qemu | _filter_qmp \ ++ | _filter_qemu_io | _filter_generated_node_ids \ ++ | _filter_img_info + } + + test_throttle=$($QEMU_IMG --help|grep throttle) +diff --git a/tests/qemu-iotests/184.out b/tests/qemu-iotests/184.out +index ef99bb2e9a..52692b6b3b 100644 +--- a/tests/qemu-iotests/184.out ++++ b/tests/qemu-iotests/184.out +@@ -41,12 +41,6 @@ Testing: + }, + "iops_wr": 0, + "ro": false, +- "children": [ +- { +- "node-name": "disk0", +- "child": "file" +- } +- ], + "node-name": "throttle0", + "backing_file_depth": 1, + "drv": "throttle", +@@ -75,8 +69,6 @@ Testing: + }, + "iops_wr": 0, + "ro": false, +- "children": [ +- ], + "node-name": "disk0", + "backing_file_depth": 0, + "drv": "null-co", +diff --git a/tests/qemu-iotests/common.filter b/tests/qemu-iotests/common.filter +index 35c0fc0d20..cd8b506dba 100644 +--- a/tests/qemu-iotests/common.filter ++++ b/tests/qemu-iotests/common.filter +@@ -230,6 +230,7 @@ _filter_img_info() + discard=0 + regex_json_spec_start='^ *"format-specific": \{' + regex_json_child_start='^ *"children": \[' ++ regex_json_limit_start='^ *"limits": \{' + gsed -e "s#$REMOTE_TEST_DIR#TEST_DIR#g" \ + -e "s#$IMGPROTO:$TEST_DIR#TEST_DIR#g" \ + -e "s#$TEST_DIR#TEST_DIR#g" \ +@@ -262,7 +263,7 @@ _filter_img_info() + discard=1 + elif [[ $line =~ "Child node '/" ]]; then + discard=1 +- elif [[ $line =~ $regex_json_spec_start ]]; then ++ elif [[ $line =~ $regex_json_spec_start || $line =~ $regex_json_limit_start ]]; then + discard=2 + regex_json_end="^${line%%[^ ]*}\\},? *$" + elif [[ $line =~ $regex_json_child_start ]]; then +-- +2.47.3 + diff --git a/kvm-block-Fix-BDS-use-after-free-during-shutdown.patch b/kvm-block-Fix-BDS-use-after-free-during-shutdown.patch new file mode 100644 index 0000000..0bc5baf --- /dev/null +++ b/kvm-block-Fix-BDS-use-after-free-during-shutdown.patch @@ -0,0 +1,64 @@ +From 8f3dbb64c2217e7c46df964311f210c6a3c9e8be Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Mon, 15 Dec 2025 16:07:14 +0100 +Subject: [PATCH] block: Fix BDS use after free during shutdown + +RH-Author: Thomas Huth +RH-MergeRequest: 446: Fix crash that happens when powering-off a guest during migration +RH-Jira: RHEL-108142 +RH-Acked-by: Cornelia Huck +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/1] af2687210d89fef5c4229a84371236c2b9bbdf5b (thuth/qemu-kvm-cs) + +JIRA: https://issues.redhat.com/browse/RHEL-108142 + +During shutdown, blockdev_close_all_bdrv_states() drops any block node +references that are still owned by the monitor (i.e. the user). However, +in doing so, it forgot to also remove the node from monitor_bdrv_states +(which qmp_blockdev_del() correctly does), which means that later calls +of bdrv_first()/bdrv_next() will still return the (now stale) pointer to +the node. + +Usually there is no such call after this point, but in some cases it can +happen. In the reported case, there was an ongoing migration, and the +migration thread wasn't shut down yet: migration_shutdown() called by +qemu_cleanup() doesn't actually wait for the migration to be shut down, +but may just move it to MIGRATION_STATUS_CANCELLING. The next time +migration_iteration_finish() runs, it sees the status and tries to +re-activate all block devices that migration may have previously +inactivated. This is where bdrv_first()/bdrv_next() get called and the +access to the already freed node happens. + +It is debatable if migration_shutdown() should really return before +migration has settled, but leaving a dangling pointer in the list of +monitor-owned block nodes is clearly a bug either way and fixing it +solves the immediate problem, so fix it. + +Reported-by: Thomas Huth +Signed-off-by: Kevin Wolf +Message-ID: <20251215150714.130214-1-kwolf@redhat.com> +Reviewed-by: Thomas Huth +Tested-by: Thomas Huth +Reviewed-by: Stefan Hajnoczi +Signed-off-by: Kevin Wolf +(cherry picked from commit 307bc43095b8ab1765fd66c26003d5da06681c05) +Signed-off-by: Thomas Huth +--- + blockdev.c | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/blockdev.c b/blockdev.c +index b451fee6e1..76c8dd0573 100644 +--- a/blockdev.c ++++ b/blockdev.c +@@ -685,6 +685,7 @@ void blockdev_close_all_bdrv_states(void) + + GLOBAL_STATE_CODE(); + QTAILQ_FOREACH_SAFE(bs, &monitor_bdrv_states, monitor_list, next_bs) { ++ QTAILQ_REMOVE(&monitor_bdrv_states, bs, monitor_list); + bdrv_unref(bs); + } + } +-- +2.47.3 + diff --git a/kvm-block-Improve-comments-in-BlockLimits.patch b/kvm-block-Improve-comments-in-BlockLimits.patch new file mode 100644 index 0000000..1e2aeda --- /dev/null +++ b/kvm-block-Improve-comments-in-BlockLimits.patch @@ -0,0 +1,98 @@ +From e5745092ebc473f359cb2eef67ff375a1e942a74 Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Fri, 24 Oct 2025 14:30:37 +0200 +Subject: [PATCH 2/6] block: Improve comments in BlockLimits + +RH-Author: Kevin Wolf +RH-MergeRequest: 439: block: Expose block limits in monitor and qemu-img info +RH-Jira: RHEL-110003 +RH-Acked-by: Hanna Czenczek +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/4] 06c55e52ff339d2bc4559b109b0a6dd625caa7f0 (kmwolf/centos-qemu-kvm) + +Patches to expose the limits in QAPI have made clear that the existing +documentation of BlockLimits could be improved: The meaning of +min_mem_alignment and opt_mem_alignment could be clearer, and talking +about better alignment values isn't helpful when we only detect these +values and never choose them. + +Make the changes in the BlockLimits documentation now, so that the +patches exposing the fields in QAPI can use descriptions consistent with +it. + +Signed-off-by: Kevin Wolf +Message-ID: <20251024123041.51254-2-kwolf@redhat.com> +Reviewed-by: Eric Blake +Signed-off-by: Kevin Wolf +(cherry picked from commit 46dd683d56b1328cb2bc923914bfd7ac590064f7) +Signed-off-by: Kevin Wolf +--- + include/block/block_int-common.h | 30 +++++++++++++++++------------- + 1 file changed, 17 insertions(+), 13 deletions(-) + +diff --git a/include/block/block_int-common.h b/include/block/block_int-common.h +index 034c0634c8..5206f32d53 100644 +--- a/include/block/block_int-common.h ++++ b/include/block/block_int-common.h +@@ -817,10 +817,10 @@ typedef struct BlockLimits { + int64_t max_pdiscard; + + /* +- * Optimal alignment for discard requests in bytes. A power of 2 +- * is best but not mandatory. Must be a multiple of +- * bl.request_alignment, and must be less than max_pdiscard if +- * that is set. May be 0 if bl.request_alignment is good enough ++ * Optimal alignment for discard requests in bytes. Note that this doesn't ++ * have to be a power of two. Must be a multiple of bl.request_alignment, ++ * and must be less than max_pdiscard if that is set. May be 0 if ++ * bl.request_alignment is good enough. + */ + uint32_t pdiscard_alignment; + +@@ -831,11 +831,10 @@ typedef struct BlockLimits { + int64_t max_pwrite_zeroes; + + /* +- * Optimal alignment for write zeroes requests in bytes. A power +- * of 2 is best but not mandatory. Must be a multiple of +- * bl.request_alignment, and must be less than max_pwrite_zeroes +- * if that is set. May be 0 if bl.request_alignment is good +- * enough ++ * Optimal alignment for write zeroes requests in bytes. Note that this ++ * doesn't have to be a power of two. Must be a multiple of ++ * bl.request_alignment, and must be less than max_pwrite_zeroes if that is ++ * set. May be 0 if bl.request_alignment is good enough. + */ + uint32_t pwrite_zeroes_alignment; + +@@ -863,18 +862,23 @@ typedef struct BlockLimits { + uint64_t max_hw_transfer; + + /* +- * Maximal number of scatter/gather elements allowed by the hardware. ++ * Maximum number of scatter/gather elements allowed by the hardware. + * Applies whenever transfers to the device bypass the kernel I/O + * scheduler, for example with SG_IO. If larger than max_iov + * or if zero, blk_get_max_hw_iov will fall back to max_iov. + */ + int max_hw_iov; + +- +- /* memory alignment, in bytes so that no bounce buffer is needed */ ++ /* ++ * Minimal required memory alignment in bytes for zero-copy I/O to succeed. ++ * For unaligned requests, a bounce buffer will be used. ++ */ + size_t min_mem_alignment; + +- /* memory alignment, in bytes, for bounce buffer */ ++ /* ++ * Optimal memory alignment in bytes. This is the alignment used for any ++ * buffer allocations QEMU performs internally. ++ */ + size_t opt_mem_alignment; + + /* maximum number of iovec elements */ +-- +2.47.3 + diff --git a/kvm-block-Never-drop-BLOCK_IO_ERROR-with-action-stop-for.patch b/kvm-block-Never-drop-BLOCK_IO_ERROR-with-action-stop-for.patch new file mode 100644 index 0000000..ecd825d --- /dev/null +++ b/kvm-block-Never-drop-BLOCK_IO_ERROR-with-action-stop-for.patch @@ -0,0 +1,98 @@ +From 2704bca029bc3c7a3e430d3af8b7696d7d2b1e37 Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Wed, 4 Mar 2026 13:28:00 +0100 +Subject: [PATCH 2/2] block: Never drop BLOCK_IO_ERROR with action=stop for + rate limiting + +RH-Author: Kevin Wolf +RH-MergeRequest: 472: block: Never drop BLOCK_IO_ERROR with action=stop for rate limiting +RH-Jira: RHEL-144004 +RH-Acked-by: Hanna Czenczek +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [1/1] 96b29a65a4a49fb159970892124e5b4bbdcdfeb7 (kmwolf/centos-qemu-kvm) + +Commit 2155d2dd introduced rate limiting for BLOCK_IO_ERROR to emit an +event only once a second. This makes sense for cases in which the guest +keeps running and can submit more requests that would possibly also fail +because there is a problem with the backend. + +However, if the error policy is configured so that the VM is stopped on +errors, this is both unnecessary because stopping the VM means that the +guest can't issue more requests and in fact harmful because stopping the +VM is an important state change that management tools need to keep track +of even if it happens more than once in a given second. If an event is +dropped, the management tool would see a VM randomly going to paused +state without an associated error, so it has a hard time deciding how to +handle the situation. + +This patch disables rate limiting for action=stop by not relying on the +event type alone any more in monitor_qapi_event_queue_no_reenter(), but +checking action for BLOCK_IO_ERROR, too. If the error is reported to the +guest or ignored, the rate limiting stays in place. + +Fixes: 2155d2dd7f73 ('block-backend: per-device throttling of BLOCK_IO_ERROR reports') +Signed-off-by: Kevin Wolf +Message-ID: <20260304122800.51923-1-kwolf@redhat.com> +Signed-off-by: Kevin Wolf +(cherry picked from commit 544ddbb6373d61292a0e2dc269809cd6bd5edec6) +Signed-off-by: Kevin Wolf +--- + monitor/monitor.c | 21 ++++++++++++++++++++- + qapi/block-core.json | 2 +- + 2 files changed, 21 insertions(+), 2 deletions(-) + +diff --git a/monitor/monitor.c b/monitor/monitor.c +index c5a5d30877..ae7cf64de0 100644 +--- a/monitor/monitor.c ++++ b/monitor/monitor.c +@@ -363,14 +363,33 @@ monitor_qapi_event_queue_no_reenter(QAPIEvent event, QDict *qdict) + { + MonitorQAPIEventConf *evconf; + MonitorQAPIEventState *evstate; ++ bool throttled; + + assert(event < QAPI_EVENT__MAX); + evconf = &monitor_qapi_event_conf[event]; + trace_monitor_protocol_event_queue(event, qdict, evconf->rate); ++ throttled = evconf->rate; ++ ++ /* ++ * Rate limit BLOCK_IO_ERROR only for action != "stop". ++ * ++ * If the VM is stopped after an I/O error, this is important information ++ * for the management tool to keep track of the state of QEMU and we can't ++ * merge any events. At the same time, stopping the VM means that the guest ++ * can't send additional requests and the number of events is already ++ * limited, so we can do without rate limiting. ++ */ ++ if (event == QAPI_EVENT_BLOCK_IO_ERROR) { ++ QDict *data = qobject_to(QDict, qdict_get(qdict, "data")); ++ const char *action = qdict_get_str(data, "action"); ++ if (!strcmp(action, "stop")) { ++ throttled = false; ++ } ++ } + + QEMU_LOCK_GUARD(&monitor_lock); + +- if (!evconf->rate) { ++ if (!throttled) { + /* Unthrottled event */ + monitor_qapi_event_emit(event, qdict); + } else { +diff --git a/qapi/block-core.json b/qapi/block-core.json +index 2c037183f0..0236936139 100644 +--- a/qapi/block-core.json ++++ b/qapi/block-core.json +@@ -5783,7 +5783,7 @@ + # .. note:: If action is "stop", a `STOP` event will eventually follow + # the `BLOCK_IO_ERROR` event. + # +-# .. note:: This event is rate-limited. ++# .. note:: This event is rate-limited, except if action is "stop". + # + # Since: 0.13 + # +-- +2.47.3 + diff --git a/kvm-block-Parse-filenames-only-when-explicitly-requested.patch b/kvm-block-Parse-filenames-only-when-explicitly-requested.patch deleted file mode 100644 index 13d8c9c..0000000 --- a/kvm-block-Parse-filenames-only-when-explicitly-requested.patch +++ /dev/null @@ -1,252 +0,0 @@ -From 53153ebcf066e962cd73d7fcfeca53039be2a945 Mon Sep 17 00:00:00 2001 -From: Kevin Wolf -Date: Thu, 25 Apr 2024 14:56:02 +0200 -Subject: [PATCH 4/4] block: Parse filenames only when explicitly requested - -RH-Author: Hana Czenczek -RH-MergeRequest: 1: CVE 2024-4467 (PRDSC) -RH-Jira: RHEL-46239 -RH-CVE: CVE-2024-4467 -RH-Acked-by: Kevin Wolf -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Eric Blake -RH-Commit: [4/4] f44c2941d4419e60f16dea3e9adca164e75aa78d - -When handling image filenames from legacy options such as -drive or from -tools, these filenames are parsed for protocol prefixes, including for -the json:{} pseudo-protocol. - -This behaviour is intended for filenames that come directly from the -command line and for backing files, which may come from the image file -itself. Higher level management tools generally take care to verify that -untrusted images don't contain a bad (or any) backing file reference; -'qemu-img info' is a suitable tool for this. - -However, for other files that can be referenced in images, such as -qcow2 data files or VMDK extents, the string from the image file is -usually not verified by management tools - and 'qemu-img info' wouldn't -be suitable because in contrast to backing files, it already opens these -other referenced files. So here the string should be interpreted as a -literal local filename. More complex configurations need to be specified -explicitly on the command line or in QMP. - -This patch changes bdrv_open_inherit() so that it only parses filenames -if a new parameter parse_filename is true. It is set for the top level -in bdrv_open(), for the file child and for the backing file child. All -other callers pass false and disable filename parsing this way. - -Signed-off-by: Kevin Wolf -Reviewed-by: Eric Blake -Reviewed-by: Stefan Hajnoczi -Reviewed-by: Hanna Czenczek -Upstream: N/A, embargoed -Signed-off-by: Hanna Czenczek ---- - block.c | 90 ++++++++++++++++++++++++++++++++++++--------------------- - 1 file changed, 57 insertions(+), 33 deletions(-) - -diff --git a/block.c b/block.c -index 468cf5e67d..50bdd197b7 100644 ---- a/block.c -+++ b/block.c -@@ -86,6 +86,7 @@ static BlockDriverState *bdrv_open_inherit(const char *filename, - BlockDriverState *parent, - const BdrvChildClass *child_class, - BdrvChildRole child_role, -+ bool parse_filename, - Error **errp); - - static bool bdrv_recurse_has_child(BlockDriverState *bs, -@@ -2058,7 +2059,8 @@ static void parse_json_protocol(QDict *options, const char **pfilename, - * block driver has been specified explicitly. - */ - static int bdrv_fill_options(QDict **options, const char *filename, -- int *flags, Error **errp) -+ int *flags, bool allow_parse_filename, -+ Error **errp) - { - const char *drvname; - bool protocol = *flags & BDRV_O_PROTOCOL; -@@ -2100,7 +2102,7 @@ static int bdrv_fill_options(QDict **options, const char *filename, - if (protocol && filename) { - if (!qdict_haskey(*options, "filename")) { - qdict_put_str(*options, "filename", filename); -- parse_filename = true; -+ parse_filename = allow_parse_filename; - } else { - error_setg(errp, "Can't specify 'file' and 'filename' options at " - "the same time"); -@@ -3663,7 +3665,8 @@ int bdrv_open_backing_file(BlockDriverState *bs, QDict *parent_options, - } - - backing_hd = bdrv_open_inherit(backing_filename, reference, options, 0, bs, -- &child_of_bds, bdrv_backing_role(bs), errp); -+ &child_of_bds, bdrv_backing_role(bs), true, -+ errp); - if (!backing_hd) { - bs->open_flags |= BDRV_O_NO_BACKING; - error_prepend(errp, "Could not open backing file: "); -@@ -3697,7 +3700,8 @@ free_exit: - static BlockDriverState * - bdrv_open_child_bs(const char *filename, QDict *options, const char *bdref_key, - BlockDriverState *parent, const BdrvChildClass *child_class, -- BdrvChildRole child_role, bool allow_none, Error **errp) -+ BdrvChildRole child_role, bool allow_none, -+ bool parse_filename, Error **errp) - { - BlockDriverState *bs = NULL; - QDict *image_options; -@@ -3728,7 +3732,8 @@ bdrv_open_child_bs(const char *filename, QDict *options, const char *bdref_key, - } - - bs = bdrv_open_inherit(filename, reference, image_options, 0, -- parent, child_class, child_role, errp); -+ parent, child_class, child_role, parse_filename, -+ errp); - if (!bs) { - goto done; - } -@@ -3738,6 +3743,33 @@ done: - return bs; - } - -+static BdrvChild *bdrv_open_child_common(const char *filename, -+ QDict *options, const char *bdref_key, -+ BlockDriverState *parent, -+ const BdrvChildClass *child_class, -+ BdrvChildRole child_role, -+ bool allow_none, bool parse_filename, -+ Error **errp) -+{ -+ BlockDriverState *bs; -+ BdrvChild *child; -+ -+ GLOBAL_STATE_CODE(); -+ -+ bs = bdrv_open_child_bs(filename, options, bdref_key, parent, child_class, -+ child_role, allow_none, parse_filename, errp); -+ if (bs == NULL) { -+ return NULL; -+ } -+ -+ bdrv_graph_wrlock(); -+ child = bdrv_attach_child(parent, bs, bdref_key, child_class, child_role, -+ errp); -+ bdrv_graph_wrunlock(); -+ -+ return child; -+} -+ - /* - * Opens a disk image whose options are given as BlockdevRef in another block - * device's options. -@@ -3761,27 +3793,15 @@ BdrvChild *bdrv_open_child(const char *filename, - BdrvChildRole child_role, - bool allow_none, Error **errp) - { -- BlockDriverState *bs; -- BdrvChild *child; -- -- GLOBAL_STATE_CODE(); -- -- bs = bdrv_open_child_bs(filename, options, bdref_key, parent, child_class, -- child_role, allow_none, errp); -- if (bs == NULL) { -- return NULL; -- } -- -- bdrv_graph_wrlock(); -- child = bdrv_attach_child(parent, bs, bdref_key, child_class, child_role, -- errp); -- bdrv_graph_wrunlock(); -- -- return child; -+ return bdrv_open_child_common(filename, options, bdref_key, parent, -+ child_class, child_role, allow_none, false, -+ errp); - } - - /* -- * Wrapper on bdrv_open_child() for most popular case: open primary child of bs. -+ * This does mostly the same as bdrv_open_child(), but for opening the primary -+ * child of a node. A notable difference from bdrv_open_child() is that it -+ * enables filename parsing for protocol names (including json:). - * - * @parent can move to a different AioContext in this function. - */ -@@ -3796,8 +3816,8 @@ int bdrv_open_file_child(const char *filename, - role = parent->drv->is_filter ? - (BDRV_CHILD_FILTERED | BDRV_CHILD_PRIMARY) : BDRV_CHILD_IMAGE; - -- if (!bdrv_open_child(filename, options, bdref_key, parent, -- &child_of_bds, role, false, errp)) -+ if (!bdrv_open_child_common(filename, options, bdref_key, parent, -+ &child_of_bds, role, false, true, errp)) - { - return -EINVAL; - } -@@ -3842,7 +3862,8 @@ BlockDriverState *bdrv_open_blockdev_ref(BlockdevRef *ref, Error **errp) - - } - -- bs = bdrv_open_inherit(NULL, reference, qdict, 0, NULL, NULL, 0, errp); -+ bs = bdrv_open_inherit(NULL, reference, qdict, 0, NULL, NULL, 0, false, -+ errp); - obj = NULL; - qobject_unref(obj); - visit_free(v); -@@ -3932,7 +3953,7 @@ static BlockDriverState * no_coroutine_fn - bdrv_open_inherit(const char *filename, const char *reference, QDict *options, - int flags, BlockDriverState *parent, - const BdrvChildClass *child_class, BdrvChildRole child_role, -- Error **errp) -+ bool parse_filename, Error **errp) - { - int ret; - BlockBackend *file = NULL; -@@ -3980,9 +4001,11 @@ bdrv_open_inherit(const char *filename, const char *reference, QDict *options, - } - - /* json: syntax counts as explicit options, as if in the QDict */ -- parse_json_protocol(options, &filename, &local_err); -- if (local_err) { -- goto fail; -+ if (parse_filename) { -+ parse_json_protocol(options, &filename, &local_err); -+ if (local_err) { -+ goto fail; -+ } - } - - bs->explicit_options = qdict_clone_shallow(options); -@@ -4007,7 +4030,8 @@ bdrv_open_inherit(const char *filename, const char *reference, QDict *options, - parent->open_flags, parent->options); - } - -- ret = bdrv_fill_options(&options, filename, &flags, &local_err); -+ ret = bdrv_fill_options(&options, filename, &flags, parse_filename, -+ &local_err); - if (ret < 0) { - goto fail; - } -@@ -4076,7 +4100,7 @@ bdrv_open_inherit(const char *filename, const char *reference, QDict *options, - - file_bs = bdrv_open_child_bs(filename, options, "file", bs, - &child_of_bds, BDRV_CHILD_IMAGE, -- true, &local_err); -+ true, true, &local_err); - if (local_err) { - goto fail; - } -@@ -4225,7 +4249,7 @@ BlockDriverState *bdrv_open(const char *filename, const char *reference, - GLOBAL_STATE_CODE(); - - return bdrv_open_inherit(filename, reference, options, flags, NULL, -- NULL, 0, errp); -+ NULL, 0, true, errp); - } - - /* Return true if the NULL-terminated @list contains @str */ --- -2.39.3 - diff --git a/kvm-block-backend-Fix-race-when-resuming-queued-requests.patch b/kvm-block-backend-Fix-race-when-resuming-queued-requests.patch new file mode 100644 index 0000000..a9d881b --- /dev/null +++ b/kvm-block-backend-Fix-race-when-resuming-queued-requests.patch @@ -0,0 +1,73 @@ +From 1393ba1b8f4ee2b7ec9dcb12317ec2d53c7d793e Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Wed, 19 Nov 2025 18:27:20 +0100 +Subject: [PATCH 01/32] block-backend: Fix race when resuming queued requests + +RH-Author: Kevin Wolf +RH-MergeRequest: 432: block-backend: Fix race when resuming queued requests [c10s] +RH-Jira: RHEL-129540 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Hanna Czenczek +RH-Commit: [1/1] fa7ead79ae44d89054ed574deb3fbd196536fef3 (kmwolf/centos-qemu-kvm) + +When new requests arrive at a BlockBackend that is currently drained, +these requests are queued until the drain section ends. + +There is a race window between blk_root_drained_end() waking up a queued +request in an iothread from the main thread and blk_wait_while_drained() +actually being woken up in the iothread and calling blk_inc_in_flight(). +If the BlockBackend is drained again during this window, drain won't +wait for this request and it will sneak in when the BlockBackend is +already supposed to be quiesced. This causes assertion failures in +bdrv_drain_all_begin() and can have other unintended consequences. + +Fix this by increasing the in_flight counter immediately when scheduling +the request to be resumed so that the next drain will wait for it to +complete. + +Cc: qemu-stable@nongnu.org +Reported-by: Andrey Drobyshev +Signed-off-by: Kevin Wolf +Message-ID: <20251119172720.135424-1-kwolf@redhat.com> +Reviewed-by: Hanna Czenczek +Tested-by: Andrey Drobyshev +Reviewed-by: Fiona Ebner +Signed-off-by: Kevin Wolf +(cherry picked from commit 8eeaa706ba73251063cb80d87ae838d2d5b08e9a) +Signed-off-by: Kevin Wolf +--- + block/block-backend.c | 8 +++++--- + 1 file changed, 5 insertions(+), 3 deletions(-) + +diff --git a/block/block-backend.c b/block/block-backend.c +index f8d6ba65c1..d6df369188 100644 +--- a/block/block-backend.c ++++ b/block/block-backend.c +@@ -1318,9 +1318,9 @@ static void coroutine_fn blk_wait_while_drained(BlockBackend *blk) + * section. + */ + qemu_mutex_lock(&blk->queued_requests_lock); ++ /* blk_root_drained_end() has the corresponding blk_inc_in_flight() */ + blk_dec_in_flight(blk); + qemu_co_queue_wait(&blk->queued_requests, &blk->queued_requests_lock); +- blk_inc_in_flight(blk); + qemu_mutex_unlock(&blk->queued_requests_lock); + } + } +@@ -2767,9 +2767,11 @@ static void blk_root_drained_end(BdrvChild *child) + blk->dev_ops->drained_end(blk->dev_opaque); + } + qemu_mutex_lock(&blk->queued_requests_lock); +- while (qemu_co_enter_next(&blk->queued_requests, +- &blk->queued_requests_lock)) { ++ while (!qemu_co_queue_empty(&blk->queued_requests)) { + /* Resume all queued requests */ ++ blk_inc_in_flight(blk); ++ qemu_co_enter_next(&blk->queued_requests, ++ &blk->queued_requests_lock); + } + qemu_mutex_unlock(&blk->queued_requests_lock); + } +-- +2.47.3 + diff --git a/kvm-block-io-Take-reqs_lock-for-tracked_requests.patch b/kvm-block-io-Take-reqs_lock-for-tracked_requests.patch new file mode 100644 index 0000000..d5a053a --- /dev/null +++ b/kvm-block-io-Take-reqs_lock-for-tracked_requests.patch @@ -0,0 +1,66 @@ +From c2e109dd6504a8fb06fddf3b5e86956a4699bdef Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Mon, 10 Nov 2025 16:48:45 +0100 +Subject: [PATCH 03/19] block/io: Take reqs_lock for tracked_requests + +RH-Author: Hanna Czenczek +RH-MergeRequest: 454: Multithreading fixes for rbd, curl, qcow2 +RH-Jira: RHEL-79118 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/5] 91512b6569546b6e82eca0034608fb5c7b426916 (hreitz/qemu-kvm-c-9-s) + +bdrv_co_get_self_request() does not take a lock around iterating through +bs->tracked_requests. With multiqueue, it may thus iterate over a list +that is in the process of being modified, producing an assertion +failure: + +../block/file-posix.c:3702: raw_do_pwrite_zeroes: Assertion `req' failed. + +[0] abort() at /lib64/libc.so.6 +[1] __assert_fail_base.cold() at /lib64/libc.so.6 +[2] raw_do_pwrite_zeroes() at ../block/file-posix.c:3702 +[3] bdrv_co_do_pwrite_zeroes() at ../block/io.c:1910 +[4] bdrv_aligned_pwritev() at ../block/io.c:2109 +[5] bdrv_co_do_zero_pwritev() at ../block/io.c:2192 +[6] bdrv_co_pwritev_part() at ../block/io.c:2292 +[7] bdrv_co_pwritev() at ../block/io.c:2225 +[8] handle_alloc_space() at ../block/qcow2.c:2573 +[9] qcow2_co_pwritev_task() at ../block/qcow2.c:2625 + +Fix this by taking reqs_lock. + +Cc: qemu-stable@nongnu.org +Signed-off-by: Hanna Czenczek +Message-ID: <20251110154854.151484-11-hreitz@redhat.com> +Reviewed-by: Stefan Hajnoczi +Reviewed-by: Kevin Wolf +Signed-off-by: Kevin Wolf +(cherry picked from commit 9b9ee60c07f52009f9bb659f54c42afae95c1d94) +Signed-off-by: Hanna Czenczek +--- + block/io.c | 3 +++ + 1 file changed, 3 insertions(+) + +diff --git a/block/io.c b/block/io.c +index 9bd8ba8431..37df1e0253 100644 +--- a/block/io.c ++++ b/block/io.c +@@ -721,11 +721,14 @@ BdrvTrackedRequest *coroutine_fn bdrv_co_get_self_request(BlockDriverState *bs) + Coroutine *self = qemu_coroutine_self(); + IO_CODE(); + ++ qemu_mutex_lock(&bs->reqs_lock); + QLIST_FOREACH(req, &bs->tracked_requests, list) { + if (req->co == self) { ++ qemu_mutex_unlock(&bs->reqs_lock); + return req; + } + } ++ qemu_mutex_unlock(&bs->reqs_lock); + + return NULL; + } +-- +2.47.3 + diff --git a/kvm-block-io_uring-avoid-potentially-getting-stuck-after.patch b/kvm-block-io_uring-avoid-potentially-getting-stuck-after.patch new file mode 100644 index 0000000..918862f --- /dev/null +++ b/kvm-block-io_uring-avoid-potentially-getting-stuck-after.patch @@ -0,0 +1,167 @@ +From 4ad89aa6a9efcbc0420e332acab2dd06e55be2fa Mon Sep 17 00:00:00 2001 +From: Fiona Ebner +Date: Tue, 25 Nov 2025 14:31:03 +0100 +Subject: [PATCH 3/4] block/io_uring: avoid potentially getting stuck after + resubmit at the end of ioq_submit() + +RH-Author: Hanna Czenczek +RH-MergeRequest: 479: linux-aio/io-uring: Resubmit tails of short requests +RH-Jira: RHEL-158224 +RH-Acked-by: Kevin Wolf +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [3/4] fd599da3ffcd6b37ceff35587ae9dbc1698b0f57 (hreitz/qemu-kvm-c-9-s) + +Note that this issue seems already fixed as a consequence of the large +io_uring rework with 047dabef97 ("block/io_uring: use aio_add_sqe()") +in current master, so this is purely for QEMU stable branches. + +At the end of ioq_submit(), there is an opportunistic call to +luring_process_completions(). This is the single caller of +luring_process_completions() that doesn't use the +luring_process_completions_and_submit() wrapper. + +Other callers use the wrapper, because luring_process_completions() +might require a subsequent call to ioq_submit() after resubmitting a +request. As noted for luring_resubmit(): + +> Resubmit a request by appending it to submit_queue. The caller must ensure +> that ioq_submit() is called later so that submit_queue requests are started. + +So the caller at the end of ioq_submit() violates the contract and can +in fact be problematic if no other requests come in later. In such a +case, the request intended to be resubmitted will never be actually be +submitted via io_uring_submit(). + +A reproducer exposing this issue is [0], which is based on user +reports from [1]. Another reproducer is iotest 109 with '-i io_uring'. + +I had the most success to trigger the issue with [0] when using a +BTRFS RAID 1 storage. With tmpfs, it can take quite a few iterations, +but also triggers eventually on my machine. With iotest 109 with '-i +io_uring' the issue triggers reliably on my ext4 file system. + +Have ioq_submit() submit any resubmitted requests after calling +luring_process_completions(). The return value from io_uring_submit() +is checked to be non-negative before the opportunistic processing of +completions and going for the new resubmit logic, to ensure that a +failure of io_uring_submit() is not missed. Also note that the return +value already was not necessarily the total number of submissions, +since the loop might've been iterated more than once even before the +current change. + +Only trigger the resubmission logic if it is actually necessary to +avoid changing behavior more than necessary. For example iotest 109 +would produce more 'mirror ready' events if always resubmitting after +luring_process_completions() at the end of ioq_submit(). + +Note iotest 109 still does not pass as is when run with '-i io_uring', +because of two offset values for BLOCK_JOB_COMPLETED events being zero +instead of non-zero as in the expected output. Note that the two +affected test cases are expected failures and still fail, so they just +fail "faster". The test cases are actually not triggering the resubmit +logic, so the reason seems to be different ordering of requests and +completions of the current aio=io_uring implementation versus +aio=threads. + +[0]: + +> #!/bin/bash -e +> #file=/mnt/btrfs/disk.raw +> file=/tmp/disk.raw +> filesize=256 +> readsize=512 +> rm -f $file +> truncate -s $filesize $file +> ./qemu-system-x86_64 --trace '*uring*' --qmp stdio \ +> --blockdev raw,node-name=node0,file.driver=file,file.cache.direct=off,file.filename=$file,file.aio=io_uring \ +> < {"execute": "qmp_capabilities"} +> {"execute": "human-monitor-command", "arguments": { "command-line": "qemu-io node0 \"read 0 $readsize \"" }} +> {"execute": "quit"} +> EOF + +[1]: https://forum.proxmox.com/threads/170045/ + +Cc: qemu-stable@nongnu.org +Signed-off-by: Fiona Ebner +Reviewed-by: Stefan Hajnoczi +Signed-off-by: Michael Tokarev +(cherry picked from commit 2bb0153cd806b8f6b4f82b353bd0113cd1c488a5) +Signed-off-by: Hanna Czenczek +--- + block/io_uring.c | 16 +++++++++++++--- + 1 file changed, 13 insertions(+), 3 deletions(-) + +diff --git a/block/io_uring.c b/block/io_uring.c +index dd4f304910..5dbafc8f7b 100644 +--- a/block/io_uring.c ++++ b/block/io_uring.c +@@ -120,11 +120,14 @@ static void luring_resubmit_short_read(LuringState *s, LuringAIOCB *luringcb, + * event loop. When there are no events left to complete the BH is being + * canceled. + * ++ * Returns whether ioq_submit() must be called again afterwards since requests ++ * were resubmitted via luring_resubmit(). + */ +-static void luring_process_completions(LuringState *s) ++static bool luring_process_completions(LuringState *s) + { + struct io_uring_cqe *cqes; + int total_bytes; ++ bool resubmit = false; + + defer_call_begin(); + +@@ -182,6 +185,7 @@ static void luring_process_completions(LuringState *s) + */ + if (ret == -EINTR || ret == -EAGAIN) { + luring_resubmit(s, luringcb); ++ resubmit = true; + continue; + } + } else if (!luringcb->qiov) { +@@ -194,6 +198,7 @@ static void luring_process_completions(LuringState *s) + if (luringcb->is_read) { + if (ret > 0) { + luring_resubmit_short_read(s, luringcb, ret); ++ resubmit = true; + continue; + } else { + /* Pad with zeroes */ +@@ -224,6 +229,8 @@ end: + qemu_bh_cancel(s->completion_bh); + + defer_call_end(); ++ ++ return resubmit; + } + + static int ioq_submit(LuringState *s) +@@ -231,6 +238,7 @@ static int ioq_submit(LuringState *s) + int ret = 0; + LuringAIOCB *luringcb, *luringcb_next; + ++resubmit: + while (s->io_q.in_queue > 0) { + /* + * Try to fetch sqes from the ring for requests waiting in +@@ -260,12 +268,14 @@ static int ioq_submit(LuringState *s) + } + s->io_q.blocked = (s->io_q.in_queue > 0); + +- if (s->io_q.in_flight) { ++ if (ret >= 0 && s->io_q.in_flight) { + /* + * We can try to complete something just right away if there are + * still requests in-flight. + */ +- luring_process_completions(s); ++ if (luring_process_completions(s)) { ++ goto resubmit; ++ } + } + return ret; + } +-- +2.47.3 + diff --git a/kvm-curl-Fix-coroutine-waking.patch b/kvm-curl-Fix-coroutine-waking.patch new file mode 100644 index 0000000..5dbf7ce --- /dev/null +++ b/kvm-curl-Fix-coroutine-waking.patch @@ -0,0 +1,172 @@ +From 14d1e6aa70f97aa75c8f3f78e3c730e286e3b683 Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Mon, 10 Nov 2025 16:48:40 +0100 +Subject: [PATCH 02/19] curl: Fix coroutine waking +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Hanna Czenczek +RH-MergeRequest: 454: Multithreading fixes for rbd, curl, qcow2 +RH-Jira: RHEL-79118 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/5] 20268ed1de88de45a7525d01dbb64899b8c4e443 (hreitz/qemu-kvm-c-9-s) + +If we wake a coroutine from a different context, we must ensure that it +will yield exactly once (now or later), awaiting that wake. + +curl’s current .ret == -EINPROGRESS loop may lead to the coroutine not +yielding if the request finishes before the loop gets run. To fix it, +we must drop the loop and yield exactly once, if we need to yield. + +Finding out that latter part ("if we need to yield") makes it a bit +complicated: Requests may be served from a cache internal to the curl +block driver, or fail before being submitted. In these cases, we must +not yield. However, if we find a matching but still ongoing request in +the cache, we will have to await that, i.e. still yield. + +To address this, move the yield inside of the respective functions: +- Inside of curl_find_buf() when awaiting ongoing concurrent requests, +- Inside of curl_setup_preadv() when having created a new request. + +Rename curl_setup_preadv() to curl_do_preadv() to reflect this. + +(Can be reproduced with multiqueue by adding a usleep(100000) before the +`while (acb.ret == -EINPROGRESS)` loop.) + +Also, add a comment why aio_co_wake() is safe regardless of whether the +coroutine and curl_multi_check_completion() run in the same context. + +Cc: qemu-stable@nongnu.org +Signed-off-by: Hanna Czenczek +Message-ID: <20251110154854.151484-6-hreitz@redhat.com> +Reviewed-by: Kevin Wolf +Signed-off-by: Kevin Wolf +(cherry picked from commit 53d5c7ffac7bd4e0d12174432ebb2b3e88614b15) +Signed-off-by: Hanna Czenczek +--- + block/curl.c | 45 +++++++++++++++++++++++++++++++-------------- + 1 file changed, 31 insertions(+), 14 deletions(-) + +diff --git a/block/curl.c b/block/curl.c +index 5467678024..d69bcdff79 100644 +--- a/block/curl.c ++++ b/block/curl.c +@@ -262,8 +262,8 @@ read_end: + } + + /* Called with s->mutex held. */ +-static bool curl_find_buf(BDRVCURLState *s, uint64_t start, uint64_t len, +- CURLAIOCB *acb) ++static bool coroutine_fn ++curl_find_buf(BDRVCURLState *s, uint64_t start, uint64_t len, CURLAIOCB *acb) + { + int i; + uint64_t end = start + len; +@@ -311,6 +311,10 @@ static bool curl_find_buf(BDRVCURLState *s, uint64_t start, uint64_t len, + for (j=0; jacb[j]) { + state->acb[j] = acb; ++ /* Await ongoing request */ ++ qemu_mutex_unlock(&s->mutex); ++ qemu_coroutine_yield(); ++ qemu_mutex_lock(&s->mutex); + return true; + } + } +@@ -382,6 +386,16 @@ static void curl_multi_check_completion(BDRVCURLState *s) + acb->ret = error ? -EIO : 0; + state->acb[i] = NULL; + qemu_mutex_unlock(&s->mutex); ++ /* ++ * Current AioContext is the BDS context, which may or may not ++ * be the request (coroutine) context. ++ * - If it is, the coroutine must have yielded or the FD handler ++ * (curl_multi_do()/curl_multi_timeout_do()) could not have ++ * been called and we would not be here ++ * - If it is not, it doesn't matter whether it has already ++ * yielded or not; it will be scheduled once it does yield ++ * So aio_co_wake() is safe to call. ++ */ + aio_co_wake(acb->co); + qemu_mutex_lock(&s->mutex); + } +@@ -882,7 +896,7 @@ out_noclean: + return -EINVAL; + } + +-static void coroutine_fn curl_setup_preadv(BlockDriverState *bs, CURLAIOCB *acb) ++static void coroutine_fn curl_do_preadv(BlockDriverState *bs, CURLAIOCB *acb) + { + CURLState *state; + int running; +@@ -894,10 +908,13 @@ static void coroutine_fn curl_setup_preadv(BlockDriverState *bs, CURLAIOCB *acb) + + qemu_mutex_lock(&s->mutex); + +- // In case we have the requested data already (e.g. read-ahead), +- // we can just call the callback and be done. ++ /* ++ * In case we have the requested data already (e.g. read-ahead), ++ * we can just call the callback and be done. This may have to ++ * await an ongoing request, in which case it itself will yield. ++ */ + if (curl_find_buf(s, start, acb->bytes, acb)) { +- goto out; ++ goto dont_yield; + } + + // No cache found, so let's start a new request +@@ -912,7 +929,7 @@ static void coroutine_fn curl_setup_preadv(BlockDriverState *bs, CURLAIOCB *acb) + if (curl_init_state(s, state) < 0) { + curl_clean_state(state); + acb->ret = -EIO; +- goto out; ++ goto dont_yield; + } + + acb->start = 0; +@@ -927,7 +944,7 @@ static void coroutine_fn curl_setup_preadv(BlockDriverState *bs, CURLAIOCB *acb) + if (state->buf_len && state->orig_buf == NULL) { + curl_clean_state(state); + acb->ret = -ENOMEM; +- goto out; ++ goto dont_yield; + } + state->acb[0] = acb; + +@@ -939,13 +956,16 @@ static void coroutine_fn curl_setup_preadv(BlockDriverState *bs, CURLAIOCB *acb) + acb->ret = -EIO; + + curl_clean_state(state); +- goto out; ++ goto dont_yield; + } + + /* Tell curl it needs to kick things off */ + curl_multi_socket_action(s->multi, CURL_SOCKET_TIMEOUT, 0, &running); ++ qemu_mutex_unlock(&s->mutex); ++ qemu_coroutine_yield(); ++ return; + +-out: ++dont_yield: + qemu_mutex_unlock(&s->mutex); + } + +@@ -961,10 +981,7 @@ static int coroutine_fn curl_co_preadv(BlockDriverState *bs, + .bytes = bytes + }; + +- curl_setup_preadv(bs, &acb); +- while (acb.ret == -EINPROGRESS) { +- qemu_coroutine_yield(); +- } ++ curl_do_preadv(bs, &acb); + return acb.ret; + } + +-- +2.47.3 + diff --git a/kvm-docs-Add-mshv-to-documentation.patch b/kvm-docs-Add-mshv-to-documentation.patch new file mode 100644 index 0000000..5de3b56 --- /dev/null +++ b/kvm-docs-Add-mshv-to-documentation.patch @@ -0,0 +1,143 @@ +From a42a299252fd1c53600387a81924adee6fa761f6 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:46 +0200 +Subject: [PATCH 29/32] docs: Add mshv to documentation + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [27/30] 6450fc04dede6796c6a6383904792da34fb11de1 + +Added mshv to the list of accelerators in doc text. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-27-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 3af71a1a6a7a497df7d3026239d6136b56e3d5ab) +Signed-off-by: Igor Mammedov +--- + docs/about/build-platforms.rst | 2 +- + docs/devel/codebase.rst | 2 +- + docs/glossary.rst | 7 +++---- + docs/system/introduction.rst | 3 +++ + qemu-options.hx | 16 ++++++++-------- + 5 files changed, 16 insertions(+), 14 deletions(-) + +diff --git a/docs/about/build-platforms.rst b/docs/about/build-platforms.rst +index 8671c3be9c..06ba0ddc9a 100644 +--- a/docs/about/build-platforms.rst ++++ b/docs/about/build-platforms.rst +@@ -55,7 +55,7 @@ Those hosts are officially supported, with various accelerators: + * - SPARC + - tcg + * - x86 +- - hvf (64 bit only), kvm, nvmm, tcg, whpx (64 bit only), xen ++ - hvf (64 bit only), mshv (64 bit only), kvm, nvmm, tcg, whpx (64 bit only), xen + + Other host architectures are not supported. It is possible to build QEMU system + emulation on an unsupported host architecture using the configure +diff --git a/docs/devel/codebase.rst b/docs/devel/codebase.rst +index 2a3143787a..69d8827117 100644 +--- a/docs/devel/codebase.rst ++++ b/docs/devel/codebase.rst +@@ -48,7 +48,7 @@ yet, so sometimes the source code is all you have. + * `accel `_: + Infrastructure and architecture agnostic code related to the various + `accelerators ` supported by QEMU +- (TCG, KVM, hvf, whpx, xen, nvmm). ++ (TCG, KVM, hvf, whpx, xen, nvmm, mshv). + Contains interfaces for operations that will be implemented per + `target `_. + * `audio `_: +diff --git a/docs/glossary.rst b/docs/glossary.rst +index 4fa044bfb6..2857731bc4 100644 +--- a/docs/glossary.rst ++++ b/docs/glossary.rst +@@ -12,7 +12,7 @@ Accelerator + + A specific API used to accelerate execution of guest instructions. It can be + hardware-based, through a virtualization API provided by the host OS (kvm, hvf, +-whpx, ...), or software-based (tcg). See this description of `supported ++whpx, mshv, ...), or software-based (tcg). See this description of `supported + accelerators`. + + Board +@@ -101,9 +101,8 @@ manage a virtual machine. QEMU is a virtualizer, that interacts with various + hypervisors. + + In the context of QEMU, an hypervisor is an API, provided by the Host OS, +-allowing to execute virtual machines. Linux implementation is KVM (and supports +-Xen as well). For MacOS, it's HVF. Windows defines WHPX. And NetBSD provides +-NVMM. ++allowing to execute virtual machines. Linux provides a choice of KVM, Xen ++or MSHV; MacOS provides HVF; Windows provides WHPX; NetBSD provides NVMM. + + .. _machine: + +diff --git a/docs/system/introduction.rst b/docs/system/introduction.rst +index 4cd46b5b8f..9c57523b6c 100644 +--- a/docs/system/introduction.rst ++++ b/docs/system/introduction.rst +@@ -23,6 +23,9 @@ Tiny Code Generator (TCG) capable of emulating many CPUs. + * - Xen + - Linux (as dom0) + - Arm, x86 ++ * - MSHV ++ - Linux (as dom0) ++ - x86 + * - Hypervisor Framework (hvf) + - MacOS + - x86 (64 bit only), Arm (64 bit only) +diff --git a/qemu-options.hx b/qemu-options.hx +index 5f146c1860..8eca7faa94 100644 +--- a/qemu-options.hx ++++ b/qemu-options.hx +@@ -28,7 +28,7 @@ DEF("machine", HAS_ARG, QEMU_OPTION_machine, \ + "-machine [type=]name[,prop[=value][,...]]\n" + " selects emulated machine ('-machine help' for list)\n" + " property accel=accel1[:accel2[:...]] selects accelerator\n" +- " supported accelerators are kvm, xen, hvf, nvmm, whpx or tcg (default: tcg)\n" ++ " supported accelerators are kvm, xen, hvf, nvmm, whpx, mshv or tcg (default: tcg)\n" + " vmport=on|off|auto controls emulation of vmport (default: auto)\n" + " dump-guest-core=on|off include guest memory in a core dump (default=on)\n" + " mem-merge=on|off controls memory merge support (default: on)\n" +@@ -66,10 +66,10 @@ SRST + + ``accel=accels1[:accels2[:...]]`` + This is used to enable an accelerator. Depending on the target +- architecture, kvm, xen, hvf, nvmm, whpx or tcg can be available. +- By default, tcg is used. If there is more than one accelerator +- specified, the next one is used if the previous one fails to +- initialize. ++ architecture, kvm, xen, hvf, nvmm, whpx, mshv or tcg can be ++ available. By default, tcg is used. If there is more than one ++ accelerator specified, the next one is used if the previous one ++ fails to initialize. + + ``vmport=on|off|auto`` + Enables emulation of VMWare IO port, for vmmouse etc. auto says +@@ -226,7 +226,7 @@ ERST + + DEF("accel", HAS_ARG, QEMU_OPTION_accel, + "-accel [accel=]accelerator[,prop[=value][,...]]\n" +- " select accelerator (kvm, xen, hvf, nvmm, whpx or tcg; use 'help' for a list)\n" ++ " select accelerator (kvm, xen, hvf, nvmm, whpx, mshv or tcg; use 'help' for a list)\n" + " igd-passthru=on|off (enable Xen integrated Intel graphics passthrough, default=off)\n" + " kernel-irqchip=on|off|split controls accelerated irqchip support (default=on)\n" + " kvm-shadow-mem=size of KVM shadow MMU in bytes\n" +@@ -241,8 +241,8 @@ DEF("accel", HAS_ARG, QEMU_OPTION_accel, + SRST + ``-accel name[,prop=value[,...]]`` + This is used to enable an accelerator. Depending on the target +- architecture, kvm, xen, hvf, nvmm, whpx or tcg can be available. By +- default, tcg is used. If there is more than one accelerator ++ architecture, kvm, xen, hvf, nvmm, whpx, mshv or tcg can be available. ++ By default, tcg is used. If there is more than one accelerator + specified, the next one is used if the previous one fails to + initialize. + +-- +2.47.3 + diff --git a/kvm-docs-add-SCSI-migrate-pr-documentation.patch b/kvm-docs-add-SCSI-migrate-pr-documentation.patch new file mode 100644 index 0000000..507524b --- /dev/null +++ b/kvm-docs-add-SCSI-migrate-pr-documentation.patch @@ -0,0 +1,118 @@ +From a47cd8b532de2235c3be76a79f42d40c32a4fa58 Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:35 -0500 +Subject: [PATCH 6/7] docs: add SCSI migrate-pr documentation + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [5/5] e081d8f5e12d1228cd3de395964c0cb4d483559e (stefanha/centos-stream-qemu-kvm) + +Suggested-by: Paolo Bonzini +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-6-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit a67819adb2212977360e9290bd005badb07dd2e4) +Signed-off-by: Stefan Hajnoczi +--- + docs/system/device-emulation.rst | 1 + + docs/system/devices/scsi/index.rst | 10 +++++ + docs/system/devices/scsi/migrate-pr.rst | 54 +++++++++++++++++++++++++ + 3 files changed, 65 insertions(+) + create mode 100644 docs/system/devices/scsi/index.rst + create mode 100644 docs/system/devices/scsi/migrate-pr.rst + +diff --git a/docs/system/device-emulation.rst b/docs/system/device-emulation.rst +index 911381643f..72a9dbd54d 100644 +--- a/docs/system/device-emulation.rst ++++ b/docs/system/device-emulation.rst +@@ -91,6 +91,7 @@ Emulated Devices + devices/keyboard.rst + devices/net.rst + devices/nvme.rst ++ devices/scsi/index.rst + devices/usb.rst + devices/vhost-user.rst + devices/virtio-gpu.rst +diff --git a/docs/system/devices/scsi/index.rst b/docs/system/devices/scsi/index.rst +new file mode 100644 +index 0000000000..4f0929b0ca +--- /dev/null ++++ b/docs/system/devices/scsi/index.rst +@@ -0,0 +1,10 @@ ++SCSI Devices ++============ ++ ++Several SCSI devices are available in QEMU. They are primarily used for block ++storage. ++ ++.. toctree:: ++ :maxdepth: 1 ++ ++ migrate-pr.rst +diff --git a/docs/system/devices/scsi/migrate-pr.rst b/docs/system/devices/scsi/migrate-pr.rst +new file mode 100644 +index 0000000000..a8f2790a86 +--- /dev/null ++++ b/docs/system/devices/scsi/migrate-pr.rst +@@ -0,0 +1,54 @@ ++.. ++ SPDX-License-Identifier: GPL-2.0-or-later ++ ++.. _scsi_migrate_pr: ++ ++SCSI Persistent Reservation Live Migration ++========================================== ++ ++This document explains how to live migrate SCSI Persistent Reservations. ++ ++The ``scsi-block`` device migrates SCSI Persistent Reservations when the ++``migrate-pr=on`` parameter is given. Migration is enabled by default in ++versioned machine types since QEMU 11.0. It is disabled by default on older ++machine types and needs to be explicitly enabled with ``--device ++scsi-block,migrate-pr=on,...``. ++ ++When migration is enabled, QEMU snoops PERSISTENT RESERVATION OUT commands and ++tracks the reservation key registered by the guest as well as reservations that ++the guest acquires. This information is migrated along with the guest and the ++destination QEMU submits a PERSISTENT RESERVATION OUT command with the PREEMPT ++service action to atomically transfer the reservation to the destination before ++the guest starts running on the destination. ++ ++The following persistent reservation capabilities reported by the PERSISTENT ++RESERVATION IN command with the REPORT CAPABILITIES service action are masked ++from the guest by QEMU when migration is enabled: ++ ++ * Specify Initiator Ports Capable (SIP_C) ++ * All Target Ports Capable (ATC_C) ++ ++When migration is disabled, the ``scsi-block`` device is live migrated but ++reservations remain in place on the source. Usually this is not the intended ++behavior unless there is another mechanism to update reservations during ++migration. The PERSISTENT RESERVATION IN command also does not mask ++capabilities reported to the guest when migration is disabled. ++ ++Limitations ++----------- ++ ++QEMU does not remember snooped reservation details across restart, so software ++inside the guest must acquire the reservation after boot in order for live ++migration to work. Similarly, if the reservation is acquired outside the guest ++then it will not live migrate along with the guest. ++ ++Snooping only considers the PERSISTENT RESERVATION OUT commands from the guest ++and does not track reservation changes made by other SCSI initiators. QEMU's ++snooped reservation details can become stale if another SCSI initiator ++makes changes to the reservation. ++ ++Guests running on the same host share a single SCSI initiator identity unless ++Fibre Channel N_Port ID Virtualization is configured. As a consequence, ++multiple guests on the same hosts may observe unexpected behavior if they use ++the same physical LUN. From the LUN's perspective all guests are the same ++initiator and there is no way to distinguish between guests. +-- +2.47.3 + diff --git a/kvm-e1000e-Prevent-crash-from-legacy-interrupt-firing-af.patch b/kvm-e1000e-Prevent-crash-from-legacy-interrupt-firing-af.patch new file mode 100644 index 0000000..d1e4d00 --- /dev/null +++ b/kvm-e1000e-Prevent-crash-from-legacy-interrupt-firing-af.patch @@ -0,0 +1,69 @@ +From a2f30bafa346ef50932c359eaf71574ed3c1239d Mon Sep 17 00:00:00 2001 +From: Laurent Vivier +Date: Thu, 7 Aug 2025 13:08:06 +0200 +Subject: [PATCH] e1000e: Prevent crash from legacy interrupt firing after + MSI-X enable +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Laurent Vivier +RH-MergeRequest: 403: e1000e: Prevent crash from legacy interrupt firing after MSI-X enable +RH-Jira: RHEL-112882 +RH-Acked-by: Cindy Lu +RH-Acked-by: Jason Wang +RH-Commit: [1/1] 8241a58b76307f27ad3d3b3b2106e00b153b7b53 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-112882 + +A race condition between guest driver actions and QEMU timers can lead +to an assertion failure when the guest switches the e1000e from legacy +interrupt mode to MSI-X. If a legacy interrupt delay timer (TIDV or +RDTR) is active, but the guest enables MSI-X before the timer fires, +the pending interrupt cause can trigger an assert in +e1000e_intmgr_collect_delayed_causes(). + +This patch removes the assertion and executes the code that clears the +pending legacy causes. This change is safe and introduces no unintended +behavioral side effects, as it only alters a state that previously led +to termination. + +- when core->delayed_causes == 0 the function was already a no-op and + remains so. + +- when core->delayed_causes != 0 the function would previously + crash due to the assertion failure. The patch now defines a safe + outcome by clearing the cause and returning. Since behavior after + the assertion never existed, this simply corrects the crash. + +Resolves: https://gitlab.com/qemu-project/qemu/-/issues/1863 +Suggested-by: Akihiko Odaki +Signed-off-by: Laurent Vivier +Acked-by: Jason Wang +Reviewed-by: Akihiko Odaki +Message-ID: <20250807110806.409065-1-lvivier@redhat.com> +Signed-off-by: Philippe Mathieu-Daudé +(cherry picked from commit 8e4649cac9bcddc050d2df07908075e9e69bccc7) +--- + hw/net/e1000e_core.c | 5 ----- + 1 file changed, 5 deletions(-) + +diff --git a/hw/net/e1000e_core.c b/hw/net/e1000e_core.c +index 2413858790..06657bb3ac 100644 +--- a/hw/net/e1000e_core.c ++++ b/hw/net/e1000e_core.c +@@ -341,11 +341,6 @@ e1000e_intmgr_collect_delayed_causes(E1000ECore *core) + { + uint32_t res; + +- if (msix_enabled(core->owner)) { +- assert(core->delayed_causes == 0); +- return 0; +- } +- + res = core->delayed_causes; + core->delayed_causes = 0; + +-- +2.47.3 + diff --git a/kvm-file-posix-Handle-suspended-dm-multipath-better-for-.patch b/kvm-file-posix-Handle-suspended-dm-multipath-better-for-.patch new file mode 100644 index 0000000..b4dd6ee --- /dev/null +++ b/kvm-file-posix-Handle-suspended-dm-multipath-better-for-.patch @@ -0,0 +1,123 @@ +From 56703080fdf6630d6d844c736f3195b3b7b80f74 Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Fri, 28 Nov 2025 23:14:40 +0100 +Subject: [PATCH 02/32] file-posix: Handle suspended dm-multipath better for + SG_IO + +RH-Author: Kevin Wolf +RH-MergeRequest: 436: file-posix: Handle suspended dm-multipath better for SG_IO +RH-Jira: RHEL-121543 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Hanna Czenczek +RH-Commit: [1/1] f90d36aba5b374dcb9d5986f968c2dbde4bdd18d (kmwolf/centos-qemu-kvm) + +When introducing DM_MPATH_PROBE_PATHS, we already anticipated that +dm-multipath devices might be suspended for a short time when the DM +tables are reloaded and that they return -EAGAIN in this case. We then +wait for a millisecond and retry. + +However, meanwhile it has also turned out that libmpathpersist (which is +used by qemu-pr-helper) may need to perform more complex recovery +operations to get reservations back to expected state if a path failure +happened in the middle of a PR operation. In this case, the device is +suspended for a longer time compared to the case we originally expected. + +This patch changes hdev_co_ioctl() to treat -EAGAIN separately so that +it doesn't result in an immediate failure if the device is suspended for +more than 1ms, and moves to incremental backoff to cover both quick and +slow cases without excessive delays. + +Buglink: https://issues.redhat.com/browse/RHEL-121543 +Signed-off-by: Kevin Wolf +Message-ID: <20251128221440.89125-1-kwolf@redhat.com> +Signed-off-by: Kevin Wolf +(cherry picked from commit 2c3165a1a61c299b4a3ae30899e1cc738d20e004) +Signed-off-by: Kevin Wolf +--- + block/file-posix.c | 56 ++++++++++++++++++++++++++++------------------ + 1 file changed, 34 insertions(+), 22 deletions(-) + +diff --git a/block/file-posix.c b/block/file-posix.c +index cb2e94d7db..ffca37130b 100644 +--- a/block/file-posix.c ++++ b/block/file-posix.c +@@ -4289,25 +4289,8 @@ hdev_open_Mac_error: + static bool coroutine_fn sgio_path_error(int ret, sg_io_hdr_t *io_hdr) + { + if (ret < 0) { +- switch (ret) { +- case -ENODEV: +- return true; +- case -EAGAIN: +- /* +- * The device is probably suspended. This happens while the dm table +- * is reloaded, e.g. because a path is added or removed. This is an +- * operation that should complete within 1ms, so just wait a bit and +- * retry. +- * +- * If the device was suspended for another reason, we'll wait and +- * retry SG_IO_MAX_RETRIES times. This is a tolerable delay before +- * we return an error and potentially stop the VM. +- */ +- qemu_co_sleep_ns(QEMU_CLOCK_REALTIME, 1000000); +- return true; +- default: +- return false; +- } ++ /* Path errors sometimes result in -ENODEV */ ++ return ret == -ENODEV; + } + + if (io_hdr->host_status != SCSI_HOST_OK) { +@@ -4376,6 +4359,7 @@ hdev_co_ioctl(BlockDriverState *bs, unsigned long int req, void *buf) + { + BDRVRawState *s = bs->opaque; + RawPosixAIOData acb; ++ uint64_t eagain_sleep_ns = 1 * SCALE_MS; + int retries = SG_IO_MAX_RETRIES; + int ret; + +@@ -4404,9 +4388,37 @@ hdev_co_ioctl(BlockDriverState *bs, unsigned long int req, void *buf) + }, + }; + +- do { +- ret = raw_thread_pool_submit(handle_aiocb_ioctl, &acb); +- } while (req == SG_IO && retries-- && hdev_co_ioctl_sgio_retry(&acb, ret)); ++retry: ++ ret = raw_thread_pool_submit(handle_aiocb_ioctl, &acb); ++ if (req == SG_IO && s->use_mpath) { ++ if (ret == -EAGAIN && eagain_sleep_ns < NANOSECONDS_PER_SECOND) { ++ /* ++ * If this is a multipath device, it is probably suspended. ++ * ++ * This can happen while the dm table is reloaded, e.g. because a ++ * path is added or removed. This is an operation that should ++ * complete within 1ms, so just wait a bit and retry. ++ * ++ * There are also some cases in which libmpathpersist must recover ++ * from path failure during its operation, which can leave the ++ * device suspended for a bit longer while the library brings back ++ * reservations into the expected state. ++ * ++ * Use increasing delays to cover both cases without waiting ++ * excessively, and stop after a bit more than a second (1023 ms). ++ * This is a tolerable delay before we return an error and ++ * potentially stop the VM. ++ */ ++ qemu_co_sleep_ns(QEMU_CLOCK_REALTIME, eagain_sleep_ns); ++ eagain_sleep_ns *= 2; ++ goto retry; ++ } ++ ++ /* Even for ret == 0, the SG_IO header can contain an error */ ++ if (retries-- && hdev_co_ioctl_sgio_retry(&acb, ret)) { ++ goto retry; ++ } ++ } + + return ret; + } +-- +2.47.3 + diff --git a/kvm-fix-pc_rhel_10_2_compat_len.patch b/kvm-fix-pc_rhel_10_2_compat_len.patch new file mode 100644 index 0000000..b93defd --- /dev/null +++ b/kvm-fix-pc_rhel_10_2_compat_len.patch @@ -0,0 +1,36 @@ +From c4415936b6033aff4b2e38b1c470c920e14fa35a Mon Sep 17 00:00:00 2001 +From: Gerd Hoffmann +Date: Mon, 12 Jan 2026 09:19:07 +0100 +Subject: [PATCH 1/4] fix pc_rhel_10_2_compat_len + +RH-Author: Gerd Hoffmann +RH-MergeRequest: 447: q35: increase default tseg size +RH-Jira: RHEL-126707 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/2] b9291186e031c0c018dfdc87960b04817421ecec (kraxel.rh/centos-src-qemu-kvm) + +There is an (apparently) cut+paste error in the definition +pc_rhel_10_2_compat_len variable, it calculates the length +of the wrong array. Fix it. + +Signed-off-by: Gerd Hoffmann +--- + hw/i386/pc.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index 446d4a7c93..394a84eb8a 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -304,7 +304,7 @@ GlobalProperty pc_rhel_10_2_compat[] = { + { TYPE_X86_CPU, "x-arch-cap-always-on", "true" }, + { TYPE_X86_CPU, "x-pdcm-on-even-without-pmu", "true" }, + }; +-const size_t pc_rhel_10_2_compat_len = G_N_ELEMENTS(pc_compat_10_0); ++const size_t pc_rhel_10_2_compat_len = G_N_ELEMENTS(pc_rhel_10_2_compat); + + GlobalProperty pc_rhel_10_1_compat[] = { + /* pc_rhel_10_1_compat from pc_compat_9_1 */ +-- +2.47.3 + diff --git a/kvm-hw-arm-smmu-common-Check-SMMU-has-PCIe-Root-Complex-.patch b/kvm-hw-arm-smmu-common-Check-SMMU-has-PCIe-Root-Complex-.patch new file mode 100644 index 0000000..1f0fe4a --- /dev/null +++ b/kvm-hw-arm-smmu-common-Check-SMMU-has-PCIe-Root-Complex-.patch @@ -0,0 +1,131 @@ +From ad929c3b2e90eeb1f81a3f7074cdaaa922b073b9 Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:23 +0100 +Subject: [PATCH 05/16] hw/arm/smmu-common: Check SMMU has PCIe Root Complex + association + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [1/11] 9e7a87070ebfef643848d31fe66f5b4e82bfe0cf (eauger1/centos-qemu-kvm) + +We only allow default PCIe Root Complex(pcie.0) or pxb-pcie based extra +root complexes to be associated with SMMU. + +Although this change does not affect functionality at present, it is +required when we add support for user-creatable SMMUv3 devices in +future patches. + +Note: Added a specific check to identify pxb-pcie to avoid matching +pxb-cxl host bridges, which are also of type PCI_HOST_BRIDGE. This +restriction can be relaxed once support for CXL devices on arm/virt +is added and validated with SMMUv3. + +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Nathan Chen +Tested-by: Eric Auger +Reviewed-by: Nicolin Chen +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Message-id: 20250829082543.7680-2-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit d9e6b8424fd2523a0361972d5dd841471879479c) +Signed-off-by: Eric Auger +--- + hw/arm/smmu-common.c | 31 ++++++++++++++++++++++++++--- + hw/pci-bridge/pci_expander_bridge.c | 1 - + include/hw/pci/pci_bridge.h | 1 + + 3 files changed, 29 insertions(+), 4 deletions(-) + +diff --git a/hw/arm/smmu-common.c b/hw/arm/smmu-common.c +index 0dcaf2f589..7f64ea48d0 100644 +--- a/hw/arm/smmu-common.c ++++ b/hw/arm/smmu-common.c +@@ -20,6 +20,7 @@ + #include "trace.h" + #include "exec/target_page.h" + #include "hw/core/cpu.h" ++#include "hw/pci/pci_bridge.h" + #include "hw/qdev-properties.h" + #include "qapi/error.h" + #include "qemu/jhash.h" +@@ -925,6 +926,7 @@ static void smmu_base_realize(DeviceState *dev, Error **errp) + { + SMMUState *s = ARM_SMMU(dev); + SMMUBaseClass *sbc = ARM_SMMU_GET_CLASS(dev); ++ PCIBus *pci_bus = s->primary_bus; + Error *local_err = NULL; + + sbc->parent_realize(dev, &local_err); +@@ -937,11 +939,34 @@ static void smmu_base_realize(DeviceState *dev, Error **errp) + g_free, g_free); + s->smmu_pcibus_by_busptr = g_hash_table_new(NULL, NULL); + +- if (s->primary_bus) { +- pci_setup_iommu(s->primary_bus, &smmu_ops, s); +- } else { ++ if (!pci_bus) { + error_setg(errp, "SMMU is not attached to any PCI bus!"); ++ return; ++ } ++ ++ /* ++ * We only allow default PCIe Root Complex(pcie.0) or pxb-pcie based extra ++ * root complexes to be associated with SMMU. ++ */ ++ if (pci_bus_is_express(pci_bus) && pci_bus_is_root(pci_bus) && ++ object_dynamic_cast(OBJECT(pci_bus)->parent, TYPE_PCI_HOST_BRIDGE)) { ++ /* ++ * This condition matches either the default pcie.0, pxb-pcie, or ++ * pxb-cxl. For both pxb-pcie and pxb-cxl, parent_dev will be set. ++ * Currently, we don't allow pxb-cxl as it requires further ++ * verification. Therefore, make sure this is indeed pxb-pcie. ++ */ ++ if (pci_bus->parent_dev) { ++ if (!object_dynamic_cast(OBJECT(pci_bus), TYPE_PXB_PCIE_BUS)) { ++ goto out_err; ++ } ++ } ++ pci_setup_iommu(pci_bus, &smmu_ops, s); ++ return; + } ++out_err: ++ error_setg(errp, "SMMU should be attached to a default PCIe root complex" ++ "(pcie.0) or a pxb-pcie based root complex"); + } + + /* +diff --git a/hw/pci-bridge/pci_expander_bridge.c b/hw/pci-bridge/pci_expander_bridge.c +index 3a29dfefc2..1bcceddbc4 100644 +--- a/hw/pci-bridge/pci_expander_bridge.c ++++ b/hw/pci-bridge/pci_expander_bridge.c +@@ -34,7 +34,6 @@ typedef struct PXBBus PXBBus; + DECLARE_INSTANCE_CHECKER(PXBBus, PXB_BUS, + TYPE_PXB_BUS) + +-#define TYPE_PXB_PCIE_BUS "pxb-pcie-bus" + DECLARE_INSTANCE_CHECKER(PXBBus, PXB_PCIE_BUS, + TYPE_PXB_PCIE_BUS) + +diff --git a/include/hw/pci/pci_bridge.h b/include/hw/pci/pci_bridge.h +index 8cdacbc4e1..a055fd8d32 100644 +--- a/include/hw/pci/pci_bridge.h ++++ b/include/hw/pci/pci_bridge.h +@@ -104,6 +104,7 @@ typedef struct PXBPCIEDev { + PXBDev parent_obj; + } PXBPCIEDev; + ++#define TYPE_PXB_PCIE_BUS "pxb-pcie-bus" + #define TYPE_PXB_CXL_BUS "pxb-cxl-bus" + #define TYPE_PXB_DEV "pxb" + OBJECT_DECLARE_SIMPLE_TYPE(PXBDev, PXB_DEV) +-- +2.47.3 + diff --git a/kvm-hw-arm-virt-Add-an-SMMU_IO_LEN-macro.patch b/kvm-hw-arm-virt-Add-an-SMMU_IO_LEN-macro.patch new file mode 100644 index 0000000..f511d7c --- /dev/null +++ b/kvm-hw-arm-virt-Add-an-SMMU_IO_LEN-macro.patch @@ -0,0 +1,61 @@ +From c62e5defde6f02bdd316b772169571d0de5d2d83 Mon Sep 17 00:00:00 2001 +From: Nicolin Chen +Date: Fri, 29 Aug 2025 09:25:27 +0100 +Subject: [PATCH 09/16] hw/arm/virt: Add an SMMU_IO_LEN macro + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [5/11] 72c82e228bb256db07fbe28728ad47dbd8b04dc3 (eauger1/centos-qemu-kvm) + +This is useful as the subsequent support for new SMMUv3 dev will also +use the same. + +Signed-off-by: Nicolin Chen +Reviewed-by: Donald Dutile +Reviewed-by: Eric Auger +Tested-by: Nathan Chen +Reviewed-by: Jonathan Cameron +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Nicolin Chen +Message-id: 20250829082543.7680-6-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 466197fc7a25658f9187d538c26887f5738d1ac9) +Signed-off-by: Eric Auger +--- + hw/arm/virt.c | 5 ++++- + 1 file changed, 4 insertions(+), 1 deletion(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index 9b95a7c9a9..b435efafe1 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -186,6 +186,9 @@ static void arm_virt_compat_set(MachineClass *mc) + #define LEGACY_RAMLIMIT_GB 255 + #define LEGACY_RAMLIMIT_BYTES (LEGACY_RAMLIMIT_GB * GiB) + ++/* MMIO region size for SMMUv3 */ ++#define SMMU_IO_LEN 0x20000 ++ + /* Addresses and sizes of our components. + * 0..128MB is space for a flash device so we can run bootrom code such as UEFI. + * 128MB..256MB is used for miscellaneous device I/O. +@@ -217,7 +220,7 @@ static const MemMapEntry base_memmap[] = { + [VIRT_FW_CFG] = { 0x09020000, 0x00000018 }, + [VIRT_GPIO] = { 0x09030000, 0x00001000 }, + [VIRT_UART1] = { 0x09040000, 0x00001000 }, +- [VIRT_SMMU] = { 0x09050000, 0x00020000 }, ++ [VIRT_SMMU] = { 0x09050000, SMMU_IO_LEN }, + [VIRT_PCDIMM_ACPI] = { 0x09070000, MEMORY_HOTPLUG_IO_LEN }, + [VIRT_ACPI_GED] = { 0x09080000, ACPI_GED_EVT_SEL_LEN }, + [VIRT_NVDIMM_ACPI] = { 0x09090000, NVDIMM_ACPI_IO_LEN}, +-- +2.47.3 + diff --git a/kvm-hw-arm-virt-Allow-user-creatable-SMMUv3-dev-instanti.patch b/kvm-hw-arm-virt-Allow-user-creatable-SMMUv3-dev-instanti.patch new file mode 100644 index 0000000..41e280e --- /dev/null +++ b/kvm-hw-arm-virt-Allow-user-creatable-SMMUv3-dev-instanti.patch @@ -0,0 +1,215 @@ +From 20b24c8ae68ff5059392188762c8d8b24c3dfa28 Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:29 +0100 +Subject: [PATCH 11/16] hw/arm/virt: Allow user-creatable SMMUv3 dev + instantiation +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [7/11] 8f4a03c34d5c699023b3916f4919caf669f7a87c (eauger1/centos-qemu-kvm) + +Allow cold-plugging of an SMMUv3 device on the virt machine when no +global (legacy) SMMUv3 is present or when a virtio-iommu is specified. + +This user-created SMMUv3 device is tied to a specific PCI bus provided +by the user, so ensure the IOMMU ops are configured accordingly. + +Due to current limitations in QEMU’s device tree support, specifically +its inability to properly present pxb-pcie based root complexes and +their devices, the device tree support for the new SMMUv3 device is +limited to cases where it is attached to the default pcie.0 root complex. + +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Nathan Chen +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Reviewed-by: Nicolin Chen +Message-id: 20250829082543.7680-8-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 66d2f665e163cf1afccd171e3c16f8d3acb3d94a) +Signed-off-by: Eric Auger +--- + hw/arm/smmu-common.c | 8 +++++- + hw/arm/smmuv3.c | 2 ++ + hw/arm/virt.c | 51 ++++++++++++++++++++++++++++++++++++ + hw/core/sysbus-fdt.c | 3 +++ + include/hw/arm/smmu-common.h | 1 + + 5 files changed, 64 insertions(+), 1 deletion(-) + +diff --git a/hw/arm/smmu-common.c b/hw/arm/smmu-common.c +index 7f64ea48d0..62a7612184 100644 +--- a/hw/arm/smmu-common.c ++++ b/hw/arm/smmu-common.c +@@ -961,7 +961,12 @@ static void smmu_base_realize(DeviceState *dev, Error **errp) + goto out_err; + } + } +- pci_setup_iommu(pci_bus, &smmu_ops, s); ++ ++ if (s->smmu_per_bus) { ++ pci_setup_iommu_per_bus(pci_bus, &smmu_ops, s); ++ } else { ++ pci_setup_iommu(pci_bus, &smmu_ops, s); ++ } + return; + } + out_err: +@@ -986,6 +991,7 @@ static void smmu_base_reset_exit(Object *obj, ResetType type) + + static const Property smmu_dev_properties[] = { + DEFINE_PROP_UINT8("bus_num", SMMUState, bus_num, 0), ++ DEFINE_PROP_BOOL("smmu_per_bus", SMMUState, smmu_per_bus, false), + DEFINE_PROP_LINK("primary-bus", SMMUState, primary_bus, + TYPE_PCI_BUS, PCIBus *), + }; +diff --git a/hw/arm/smmuv3.c b/hw/arm/smmuv3.c +index ab67972353..bcf8af8dc7 100644 +--- a/hw/arm/smmuv3.c ++++ b/hw/arm/smmuv3.c +@@ -1996,6 +1996,8 @@ static void smmuv3_class_init(ObjectClass *klass, const void *data) + device_class_set_parent_realize(dc, smmu_realize, + &c->parent_realize); + device_class_set_props(dc, smmuv3_properties); ++ dc->hotpluggable = false; ++ dc->user_creatable = true; + } + + static int smmuv3_notify_flag_changed(IOMMUMemoryRegion *iommu, +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index b435efafe1..e8e64fe7fe 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -56,6 +56,7 @@ + #include "qemu/cutils.h" + #include "qemu/error-report.h" + #include "qemu/module.h" ++#include "hw/pci/pci_bus.h" + #include "hw/pci-host/gpex.h" + #include "hw/pci-bridge/pci_expander_bridge.h" + #include "hw/virtio/virtio-pci.h" +@@ -1510,6 +1511,29 @@ static void create_smmuv3_dt_bindings(const VirtMachineState *vms, hwaddr base, + g_free(node); + } + ++static void create_smmuv3_dev_dtb(VirtMachineState *vms, ++ DeviceState *dev, PCIBus *bus) ++{ ++ PlatformBusDevice *pbus = PLATFORM_BUS_DEVICE(vms->platform_bus_dev); ++ SysBusDevice *sbdev = SYS_BUS_DEVICE(dev); ++ int irq = platform_bus_get_irqn(pbus, sbdev, 0); ++ hwaddr base = platform_bus_get_mmio_addr(pbus, sbdev, 0); ++ MachineState *ms = MACHINE(vms); ++ ++ if (!(vms->bootinfo.firmware_loaded && virt_is_acpi_enabled(vms)) && ++ strcmp("pcie.0", bus->qbus.name)) { ++ warn_report("SMMUv3 device only supported with pcie.0 for DT"); ++ return; ++ } ++ base += vms->memmap[VIRT_PLATFORM_BUS].base; ++ irq += vms->irqmap[VIRT_PLATFORM_BUS]; ++ ++ vms->iommu_phandle = qemu_fdt_alloc_phandle(ms->fdt); ++ create_smmuv3_dt_bindings(vms, base, SMMU_IO_LEN, irq); ++ qemu_fdt_setprop_cells(ms->fdt, vms->pciehb_nodename, "iommu-map", ++ 0x0, vms->iommu_phandle, 0x0, 0x10000); ++} ++ + static void create_smmu(const VirtMachineState *vms, + PCIBus *bus) + { +@@ -3057,6 +3081,16 @@ static void virt_machine_device_pre_plug_cb(HotplugHandler *hotplug_dev, + qlist_append_str(reserved_regions, resv_prop_str); + qdev_prop_set_array(dev, "reserved-regions", reserved_regions); + g_free(resv_prop_str); ++ } else if (object_dynamic_cast(OBJECT(dev), TYPE_ARM_SMMUV3)) { ++ if (vms->legacy_smmuv3_present || vms->iommu == VIRT_IOMMU_VIRTIO) { ++ error_setg(errp, "virt machine already has %s set. " ++ "Doesn't support incompatible iommus", ++ (vms->legacy_smmuv3_present) ? ++ "iommu=smmuv3" : "virtio-iommu"); ++ } else if (vms->iommu == VIRT_IOMMU_NONE) { ++ /* The new SMMUv3 device is specific to the PCI bus */ ++ object_property_set_bool(OBJECT(dev), "smmu_per_bus", true, NULL); ++ } + } + } + +@@ -3080,6 +3114,22 @@ static void virt_machine_device_plug_cb(HotplugHandler *hotplug_dev, + virtio_md_pci_plug(VIRTIO_MD_PCI(dev), MACHINE(hotplug_dev), errp); + } + ++ if (object_dynamic_cast(OBJECT(dev), TYPE_ARM_SMMUV3)) { ++ if (!vms->legacy_smmuv3_present && vms->platform_bus_dev) { ++ PCIBus *bus; ++ ++ bus = PCI_BUS(object_property_get_link(OBJECT(dev), "primary-bus", ++ &error_abort)); ++ if (pci_bus_bypass_iommu(bus)) { ++ error_setg(errp, "Bypass option cannot be set for SMMUv3 " ++ "associated PCIe RC"); ++ return; ++ } ++ ++ create_smmuv3_dev_dtb(vms, dev, bus); ++ } ++ } ++ + if (object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_IOMMU_PCI)) { + PCIDevice *pdev = PCI_DEVICE(dev); + +@@ -3286,6 +3336,7 @@ static void virt_machine_class_init(ObjectClass *oc, const void *data) + #endif + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_RAMFB_DEVICE); + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_UEFI_VARS_SYSBUS); ++ machine_class_allow_dynamic_sysbus_dev(mc, TYPE_ARM_SMMUV3); + #ifdef CONFIG_TPM + machine_class_allow_dynamic_sysbus_dev(mc, TYPE_TPM_TIS_SYSBUS); + #endif +diff --git a/hw/core/sysbus-fdt.c b/hw/core/sysbus-fdt.c +index 1e1966813f..673e083d31 100644 +--- a/hw/core/sysbus-fdt.c ++++ b/hw/core/sysbus-fdt.c +@@ -31,6 +31,7 @@ + #include "qemu/error-report.h" + #include "system/device_tree.h" + #include "system/tpm.h" ++#include "hw/arm/smmuv3.h" + #include "hw/platform-bus.h" + #include "hw/vfio/vfio-platform.h" + #include "hw/vfio/vfio-calxeda-xgmac.h" +@@ -518,6 +519,8 @@ static const BindingEntry bindings[] = { + #ifdef CONFIG_TPM + TYPE_BINDING(TYPE_TPM_TIS_SYSBUS, add_tpm_tis_fdt_node), + #endif ++ /* No generic DT support for smmuv3 dev. Support added for arm virt only */ ++ TYPE_BINDING(TYPE_ARM_SMMUV3, no_fdt_node), + TYPE_BINDING(TYPE_RAMFB_DEVICE, no_fdt_node), + TYPE_BINDING(TYPE_UEFI_VARS_SYSBUS, add_uefi_vars_node), + TYPE_BINDING("", NULL), /* last element */ +diff --git a/include/hw/arm/smmu-common.h b/include/hw/arm/smmu-common.h +index e5e2d09294..80d0fecfde 100644 +--- a/include/hw/arm/smmu-common.h ++++ b/include/hw/arm/smmu-common.h +@@ -161,6 +161,7 @@ struct SMMUState { + QLIST_HEAD(, SMMUDevice) devices_with_notifiers; + uint8_t bus_num; + PCIBus *primary_bus; ++ bool smmu_per_bus; /* SMMU is specific to the primary_bus */ + }; + + struct SMMUBaseClass { +-- +2.47.3 + diff --git a/kvm-hw-arm-virt-Factor-out-common-SMMUV3-dt-bindings-cod.patch b/kvm-hw-arm-virt-Factor-out-common-SMMUV3-dt-bindings-cod.patch new file mode 100644 index 0000000..5935080 --- /dev/null +++ b/kvm-hw-arm-virt-Factor-out-common-SMMUV3-dt-bindings-cod.patch @@ -0,0 +1,118 @@ +From 1b3c413355ee5f3917e8e39dbf7a281f8e31a0f5 Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:26 +0100 +Subject: [PATCH 08/16] hw/arm/virt: Factor out common SMMUV3 dt bindings code + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [4/11] db5d2a44f4cd1583c839b93ae551a2ddbd68b83b (eauger1/centos-qemu-kvm) + +No functional changes intended. This will be useful when we +add support for user-creatable smmuv3 device. + +Reviewed-by: Nicolin Chen +Reviewed-by: Eric Auger +Tested-by: Nathan Chen +Reviewed-by: Jonathan Cameron +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Message-id: 20250829082543.7680-5-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 7a276b7570266ec39611f9d91089741ec7e9295b) +Signed-off-by: Eric Auger +--- + hw/arm/virt.c | 54 +++++++++++++++++++++++++++------------------------ + 1 file changed, 29 insertions(+), 25 deletions(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index 0cc9e5f068..9b95a7c9a9 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -1479,19 +1479,43 @@ static void create_pcie_irq_map(const MachineState *ms, + 0x7 /* PCI irq */); + } + ++static void create_smmuv3_dt_bindings(const VirtMachineState *vms, hwaddr base, ++ hwaddr size, int irq) ++{ ++ char *node; ++ const char compat[] = "arm,smmu-v3"; ++ const char irq_names[] = "eventq\0priq\0cmdq-sync\0gerror"; ++ MachineState *ms = MACHINE(vms); ++ ++ node = g_strdup_printf("/smmuv3@%" PRIx64, base); ++ qemu_fdt_add_subnode(ms->fdt, node); ++ qemu_fdt_setprop(ms->fdt, node, "compatible", compat, sizeof(compat)); ++ qemu_fdt_setprop_sized_cells(ms->fdt, node, "reg", 2, base, 2, size); ++ ++ qemu_fdt_setprop_cells(ms->fdt, node, "interrupts", ++ GIC_FDT_IRQ_TYPE_SPI, irq , GIC_FDT_IRQ_FLAGS_EDGE_LO_HI, ++ GIC_FDT_IRQ_TYPE_SPI, irq + 1, GIC_FDT_IRQ_FLAGS_EDGE_LO_HI, ++ GIC_FDT_IRQ_TYPE_SPI, irq + 2, GIC_FDT_IRQ_FLAGS_EDGE_LO_HI, ++ GIC_FDT_IRQ_TYPE_SPI, irq + 3, GIC_FDT_IRQ_FLAGS_EDGE_LO_HI); ++ ++ qemu_fdt_setprop(ms->fdt, node, "interrupt-names", irq_names, ++ sizeof(irq_names)); ++ ++ qemu_fdt_setprop(ms->fdt, node, "dma-coherent", NULL, 0); ++ qemu_fdt_setprop_cell(ms->fdt, node, "#iommu-cells", 1); ++ qemu_fdt_setprop_cell(ms->fdt, node, "phandle", vms->iommu_phandle); ++ g_free(node); ++} ++ + static void create_smmu(const VirtMachineState *vms, + PCIBus *bus) + { + VirtMachineClass *vmc = VIRT_MACHINE_GET_CLASS(vms); +- char *node; +- const char compat[] = "arm,smmu-v3"; + int irq = vms->irqmap[VIRT_SMMU]; + int i; + hwaddr base = vms->memmap[VIRT_SMMU].base; + hwaddr size = vms->memmap[VIRT_SMMU].size; +- const char irq_names[] = "eventq\0priq\0cmdq-sync\0gerror"; + DeviceState *dev; +- MachineState *ms = MACHINE(vms); + + if (vms->iommu != VIRT_IOMMU_SMMUV3 || !vms->iommu_phandle) { + return; +@@ -1510,27 +1534,7 @@ static void create_smmu(const VirtMachineState *vms, + sysbus_connect_irq(SYS_BUS_DEVICE(dev), i, + qdev_get_gpio_in(vms->gic, irq + i)); + } +- +- node = g_strdup_printf("/smmuv3@%" PRIx64, base); +- qemu_fdt_add_subnode(ms->fdt, node); +- qemu_fdt_setprop(ms->fdt, node, "compatible", compat, sizeof(compat)); +- qemu_fdt_setprop_sized_cells(ms->fdt, node, "reg", 2, base, 2, size); +- +- qemu_fdt_setprop_cells(ms->fdt, node, "interrupts", +- GIC_FDT_IRQ_TYPE_SPI, irq , GIC_FDT_IRQ_FLAGS_EDGE_LO_HI, +- GIC_FDT_IRQ_TYPE_SPI, irq + 1, GIC_FDT_IRQ_FLAGS_EDGE_LO_HI, +- GIC_FDT_IRQ_TYPE_SPI, irq + 2, GIC_FDT_IRQ_FLAGS_EDGE_LO_HI, +- GIC_FDT_IRQ_TYPE_SPI, irq + 3, GIC_FDT_IRQ_FLAGS_EDGE_LO_HI); +- +- qemu_fdt_setprop(ms->fdt, node, "interrupt-names", irq_names, +- sizeof(irq_names)); +- +- qemu_fdt_setprop(ms->fdt, node, "dma-coherent", NULL, 0); +- +- qemu_fdt_setprop_cell(ms->fdt, node, "#iommu-cells", 1); +- +- qemu_fdt_setprop_cell(ms->fdt, node, "phandle", vms->iommu_phandle); +- g_free(node); ++ create_smmuv3_dt_bindings(vms, base, size, irq); + } + + static void create_virtio_iommu_dt_bindings(VirtMachineState *vms) +-- +2.47.3 + diff --git a/kvm-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-from-10..patch b/kvm-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-from-10..patch new file mode 100644 index 0000000..a30d175 --- /dev/null +++ b/kvm-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-from-10..patch @@ -0,0 +1,66 @@ +From 5264d9ea8c029dab0663a3da82f4d8241ad0f1b9 Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Fri, 7 Nov 2025 05:23:16 -0500 +Subject: [PATCH 04/16] hw/arm/virt: Use ACPI PCI hotplug by default from 10.2 + onwards + +RH-Author: Eric Auger +RH-MergeRequest: 422: hw/arm/virt: Use ACPI PCI hotplug by default from 10.2 onwards +RH-Jira: RHEL-67323 +RH-Acked-by: Sebastian Ott +RH-Acked-by: Cornelia Huck +RH-Acked-by: Gavin Shan +RH-Acked-by: Igor Mammedov +RH-Commit: [1/1] 4539ba6526fef80adb9893a643eb001449397447 (eauger1/centos-qemu-kvm) + +UPSTREAM: RHEL-only + +Use ACPI PCI hotplug by default from 10.2 onwards. For older +rhel10 machine types and all rhel9 machine types ACPI PCI hotplug +is kept disabled. + +Signed-off-by: Eric Auger +--- + hw/arm/virt.c | 9 +++++++++ + 1 file changed, 9 insertions(+) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index dcdd53043e..542d702513 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -94,9 +94,15 @@ + + static GlobalProperty arm_virt_compat[] = { + { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "48" }, ++ { TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "on" }, + }; + static const size_t arm_virt_compat_len = G_N_ELEMENTS(arm_virt_compat); + ++GlobalProperty arm_acpi_pci_hp_disabled_compat[] = { ++ { TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "off" }, ++}; ++static const size_t arm_acpi_pci_hp_disabled_compat_len = G_N_ELEMENTS(arm_virt_compat); ++ + /* + * RHEL9 kernels have pauth disabled while RHEL10 has it enabled, + * since qemu will setup the VM with pauth when KVM supports it we +@@ -104,6 +110,7 @@ static const size_t arm_virt_compat_len = G_N_ELEMENTS(arm_virt_compat); + */ + GlobalProperty arm_rhel9_compat[] = { + {TYPE_ARM_CPU, "pauth", "off", .optional = true}, ++ {TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "off" }, + }; + const size_t arm_rhel9_compat_len = G_N_ELEMENTS(arm_rhel9_compat); + +@@ -3701,6 +3708,8 @@ static void virt_rhel_machine_10_0_0_options(MachineClass *mc) + + /* QEMU 9.1 and earlier have only a stage-1 SMMU, not a nested s1+2 one */ + vmc->no_nested_smmu = true; ++ compat_props_add(mc->compat_props, arm_acpi_pci_hp_disabled_compat, ++ arm_acpi_pci_hp_disabled_compat_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_2, hw_compat_rhel_10_2_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_1, hw_compat_rhel_10_1_len); + } +-- +2.47.3 + diff --git a/kvm-hw-arm-virt-acpi-build-Re-arrange-SMMUv3-IORT-build.patch b/kvm-hw-arm-virt-acpi-build-Re-arrange-SMMUv3-IORT-build.patch new file mode 100644 index 0000000..e12a15d --- /dev/null +++ b/kvm-hw-arm-virt-acpi-build-Re-arrange-SMMUv3-IORT-build.patch @@ -0,0 +1,291 @@ +From 221e12accdd5e699d727cd862760829e973a7b2a Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:24 +0100 +Subject: [PATCH 06/16] hw/arm/virt-acpi-build: Re-arrange SMMUv3 IORT build + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [2/11] 73e2dd4f48ffaf614c79241bc73cbb0457849131 (eauger1/centos-qemu-kvm) + +Introduce a new struct AcpiIortSMMUv3Dev to hold all the information +required for SMMUv3 IORT node and use that for populating the node. + +The current machine wide SMMUv3 is named as legacy SMMUv3 as we will +soon add support for user-creatable SMMUv3 devices. These changes will +be useful to have common code paths when we add that support. + +Tested-by: Nathan Chen +Reviewed-by: Nicolin Chen +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Message-id: 20250829082543.7680-3-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 0e6a5bfb0eb17f57fb923b7905bd1435204bdd62) +Signed-off-by: Eric Auger +--- + hw/arm/virt-acpi-build.c | 137 ++++++++++++++++++++++++++------------- + hw/arm/virt.c | 1 + + include/hw/arm/virt.h | 1 + + 3 files changed, 94 insertions(+), 45 deletions(-) + +diff --git a/hw/arm/virt-acpi-build.c b/hw/arm/virt-acpi-build.c +index b01fc4f8ef..bef4fabe56 100644 +--- a/hw/arm/virt-acpi-build.c ++++ b/hw/arm/virt-acpi-build.c +@@ -305,29 +305,65 @@ static int iort_idmap_compare(gconstpointer a, gconstpointer b) + return idmap_a->input_base - idmap_b->input_base; + } + ++typedef struct AcpiIortSMMUv3Dev { ++ int irq; ++ hwaddr base; ++ GArray *rc_smmu_idmaps; ++ /* Offset of the SMMUv3 IORT Node relative to the start of the IORT */ ++ size_t offset; ++} AcpiIortSMMUv3Dev; ++ ++/* ++ * Populate the struct AcpiIortSMMUv3Dev for the legacy SMMUv3 and ++ * return the total number of associated idmaps. ++ */ ++static int populate_smmuv3_legacy_dev(GArray *sdev_blob) ++{ ++ VirtMachineState *vms = VIRT_MACHINE(qdev_get_machine()); ++ AcpiIortSMMUv3Dev sdev; ++ ++ sdev.rc_smmu_idmaps = g_array_new(false, true, sizeof(AcpiIortIdMapping)); ++ object_child_foreach_recursive(object_get_root(), iort_host_bridges, ++ sdev.rc_smmu_idmaps); ++ /* ++ * There can be only one legacy SMMUv3("iommu=smmuv3") as it is a machine ++ * wide one. Since it may cover multiple PCIe RCs(based on "bypass_iommu" ++ * property), may have multiple SMMUv3 idmaps. Sort it by input_base. ++ */ ++ g_array_sort(sdev.rc_smmu_idmaps, iort_idmap_compare); ++ ++ sdev.base = vms->memmap[VIRT_SMMU].base; ++ sdev.irq = vms->irqmap[VIRT_SMMU] + ARM_SPI_BASE; ++ g_array_append_val(sdev_blob, sdev); ++ return sdev.rc_smmu_idmaps->len; ++} ++ + /* Compute ID ranges (RIDs) from RC that are directed to the ITS Group node */ +-static void create_rc_its_idmaps(GArray *its_idmaps, GArray *smmu_idmaps) ++static void create_rc_its_idmaps(GArray *its_idmaps, GArray *smmuv3_devs) + { + AcpiIortIdMapping *idmap; + AcpiIortIdMapping next_range = {0}; ++ AcpiIortSMMUv3Dev *sdev; + +- /* +- * Based on the RID ranges that are directed to the SMMU, determine the +- * bypassed RID ranges, i.e., the ones that are directed to the ITS Group +- * node and do not pass through the SMMU, by subtracting the SMMU-bound +- * ranges from the full RID range (0x0000–0xFFFF). +- */ +- for (int i = 0; i < smmu_idmaps->len; i++) { +- idmap = &g_array_index(smmu_idmaps, AcpiIortIdMapping, i); ++ for (int i = 0; i < smmuv3_devs->len; i++) { ++ sdev = &g_array_index(smmuv3_devs, AcpiIortSMMUv3Dev, i); ++ /* ++ * Based on the RID ranges that are directed to the SMMU, determine the ++ * bypassed RID ranges, i.e., the ones that are directed to the ITS ++ * Group node and do not pass through the SMMU, by subtracting the ++ * SMMU-bound ranges from the full RID range (0x0000–0xFFFF). ++ */ ++ for (int j = 0; j < sdev->rc_smmu_idmaps->len; j++) { ++ idmap = &g_array_index(sdev->rc_smmu_idmaps, AcpiIortIdMapping, j); + +- if (next_range.input_base < idmap->input_base) { +- next_range.id_count = idmap->input_base - next_range.input_base; +- g_array_append_val(its_idmaps, next_range); +- } ++ if (next_range.input_base < idmap->input_base) { ++ next_range.id_count = idmap->input_base - next_range.input_base; ++ g_array_append_val(its_idmaps, next_range); ++ } + +- next_range.input_base = idmap->input_base + idmap->id_count; ++ next_range.input_base = idmap->input_base + idmap->id_count; ++ } + } +- + /* + * Append the last RC -> ITS ID mapping. + * +@@ -341,7 +377,6 @@ static void create_rc_its_idmaps(GArray *its_idmaps, GArray *smmu_idmaps) + } + } + +- + /* + * Input Output Remapping Table (IORT) + * Conforms to "IO Remapping Table System Software on ARM Platforms", +@@ -351,9 +386,12 @@ static void + build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + { + int i, nb_nodes, rc_mapping_count; +- size_t node_size, smmu_offset = 0; ++ AcpiIortSMMUv3Dev *sdev; ++ size_t node_size; ++ int num_smmus = 0; + uint32_t id = 0; +- GArray *rc_smmu_idmaps = g_array_new(false, true, sizeof(AcpiIortIdMapping)); ++ int rc_smmu_idmaps_len = 0; ++ GArray *smmuv3_devs = g_array_new(false, true, sizeof(AcpiIortSMMUv3Dev)); + GArray *rc_its_idmaps = g_array_new(false, true, sizeof(AcpiIortIdMapping)); + + AcpiTable table = { .sig = "IORT", .rev = 3, .oem_id = vms->oem_id, +@@ -361,22 +399,21 @@ build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + /* Table 2 The IORT */ + acpi_table_begin(&table, table_data); + +- if (vms->iommu == VIRT_IOMMU_SMMUV3) { +- object_child_foreach_recursive(object_get_root(), +- iort_host_bridges, rc_smmu_idmaps); +- +- /* Sort the smmu idmap by input_base */ +- g_array_sort(rc_smmu_idmaps, iort_idmap_compare); ++ if (vms->legacy_smmuv3_present) { ++ rc_smmu_idmaps_len = populate_smmuv3_legacy_dev(smmuv3_devs); ++ } + +- nb_nodes = 2; /* RC and SMMUv3 */ +- rc_mapping_count = rc_smmu_idmaps->len; ++ num_smmus = smmuv3_devs->len; ++ if (num_smmus) { ++ nb_nodes = num_smmus + 1; /* RC and SMMUv3 */ ++ rc_mapping_count = rc_smmu_idmaps_len; + + if (vms->its) { + /* + * Knowing the ID ranges from the RC to the SMMU, it's possible to + * determine the ID ranges from RC that go directly to ITS. + */ +- create_rc_its_idmaps(rc_its_idmaps, rc_smmu_idmaps); ++ create_rc_its_idmaps(rc_its_idmaps, smmuv3_devs); + + nb_nodes++; /* ITS */ + rc_mapping_count += rc_its_idmaps->len; +@@ -411,9 +448,10 @@ build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + build_append_int_noprefix(table_data, 0 /* MADT translation_id */, 4); + } + +- if (vms->iommu == VIRT_IOMMU_SMMUV3) { +- int irq = vms->irqmap[VIRT_SMMU] + ARM_SPI_BASE; ++ for (i = 0; i < num_smmus; i++) { ++ sdev = &g_array_index(smmuv3_devs, AcpiIortSMMUv3Dev, i); + int smmu_mapping_count, offset_to_id_array; ++ int irq = sdev->irq; + + if (vms->its) { + smmu_mapping_count = 1; /* ITS Group node */ +@@ -422,7 +460,7 @@ build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + smmu_mapping_count = 0; /* No ID mappings */ + offset_to_id_array = 0; /* No ID mappings array */ + } +- smmu_offset = table_data->len - table.table_offset; ++ sdev->offset = table_data->len - table.table_offset; + /* Table 9 SMMUv3 Format */ + build_append_int_noprefix(table_data, 4 /* SMMUv3 */, 1); /* Type */ + node_size = SMMU_V3_ENTRY_SIZE + +@@ -435,7 +473,7 @@ build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + /* Reference to ID Array */ + build_append_int_noprefix(table_data, offset_to_id_array, 4); + /* Base address */ +- build_append_int_noprefix(table_data, vms->memmap[VIRT_SMMU].base, 8); ++ build_append_int_noprefix(table_data, sdev->base, 8); + /* Flags */ + build_append_int_noprefix(table_data, 1 /* COHACC Override */, 4); + build_append_int_noprefix(table_data, 0, 4); /* Reserved */ +@@ -486,21 +524,26 @@ build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + build_append_int_noprefix(table_data, 0, 3); /* Reserved */ + + /* Output Reference */ +- if (vms->iommu == VIRT_IOMMU_SMMUV3) { ++ if (num_smmus) { + AcpiIortIdMapping *range; + +- /* +- * Map RIDs (input) from RC to SMMUv3 nodes: RC -> SMMUv3. +- * +- * N.B.: The mapping from SMMUv3 to ITS Group node (SMMUv3 -> ITS) is +- * defined in the SMMUv3 table, where all SMMUv3 IDs are mapped to the +- * ITS Group node, if ITS is available. +- */ +- for (i = 0; i < rc_smmu_idmaps->len; i++) { +- range = &g_array_index(rc_smmu_idmaps, AcpiIortIdMapping, i); +- /* Output IORT node is the SMMUv3 node. */ +- build_iort_id_mapping(table_data, range->input_base, +- range->id_count, smmu_offset); ++ for (i = 0; i < num_smmus; i++) { ++ sdev = &g_array_index(smmuv3_devs, AcpiIortSMMUv3Dev, i); ++ ++ /* ++ * Map RIDs (input) from RC to SMMUv3 nodes: RC -> SMMUv3. ++ * ++ * N.B.: The mapping from SMMUv3 to ITS Group node (SMMUv3 -> ITS) ++ * is defined in the SMMUv3 table, where all SMMUv3 IDs are mapped ++ * to the ITS Group node, if ITS is available. ++ */ ++ for (int j = 0; j < sdev->rc_smmu_idmaps->len; j++) { ++ range = &g_array_index(sdev->rc_smmu_idmaps, ++ AcpiIortIdMapping, j); ++ /* Output IORT node is the SMMUv3 node. */ ++ build_iort_id_mapping(table_data, range->input_base, ++ range->id_count, sdev->offset); ++ } + } + + if (vms->its) { +@@ -525,8 +568,12 @@ build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + } + + acpi_table_end(linker, &table); +- g_array_free(rc_smmu_idmaps, true); + g_array_free(rc_its_idmaps, true); ++ for (i = 0; i < num_smmus; i++) { ++ sdev = &g_array_index(smmuv3_devs, AcpiIortSMMUv3Dev, i); ++ g_array_free(sdev->rc_smmu_idmaps, true); ++ } ++ g_array_free(smmuv3_devs, true); + } + + /* +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index 542d702513..0cc9e5f068 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -1686,6 +1686,7 @@ static void create_pcie(VirtMachineState *vms) + qemu_fdt_setprop_cells(ms->fdt, nodename, "iommu-map", + 0x0, vms->iommu_phandle, 0x0, 0x10000); + } ++ vms->legacy_smmuv3_present = true; + break; + default: + g_assert_not_reached(); +diff --git a/include/hw/arm/virt.h b/include/hw/arm/virt.h +index 94c79d6c6d..98b877c8b9 100644 +--- a/include/hw/arm/virt.h ++++ b/include/hw/arm/virt.h +@@ -180,6 +180,7 @@ struct VirtMachineState { + char *oem_table_id; + bool ns_el2_virt_timer_irq; + CXLState cxl_devices_state; ++ bool legacy_smmuv3_present; + }; + + #define VIRT_ECAM_ID(high) (high ? VIRT_HIGH_PCIE_ECAM : VIRT_PCIE_ECAM) +-- +2.47.3 + diff --git a/kvm-hw-arm-virt-acpi-build-Update-IORT-for-multiple-smmu.patch b/kvm-hw-arm-virt-acpi-build-Update-IORT-for-multiple-smmu.patch new file mode 100644 index 0000000..7e4267a --- /dev/null +++ b/kvm-hw-arm-virt-acpi-build-Update-IORT-for-multiple-smmu.patch @@ -0,0 +1,170 @@ +From f89d89a3758ebd8725e677431f1e7493c65381c2 Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:25 +0100 +Subject: [PATCH 07/16] hw/arm/virt-acpi-build: Update IORT for multiple smmuv3 + devices + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [3/11] 9cb15768a319676af16cd2cdef8b8fabfa7b6f13 (eauger1/centos-qemu-kvm) + +With the soon to be introduced user-creatable SMMUv3 devices for +virt, it is possible to have multiple SMMUv3 devices associated +with different PCIe root complexes. + +Update IORT nodes accordingly. + +An example IORT Id mappings for a Qemu virt machine with two +PCIe Root Complexes each assocaited with a SMMUv3 will +be something like below, + + -device arm-smmuv3,primary-bus=pcie.0,id=smmuv3.0 + -device arm-smmuv3,primary-bus=pcie.1,id=smmuv3.1 + ... + + +--------------------+ +--------------------+ + | Root Complex 0 | | Root Complex 1 | + | | | | + | Requestor IDs | | Requestor IDs | + | 0x0000 - 0x00FF | | 0x0100 - 0x01FF | + +---------+----------+ +---------+----------+ + | | + | | + | Stream ID Mapping | + v v + +--------------------+ +--------------------+ + | SMMUv3 Node 0 | | SMMUv3 Node 1 | + | | | | + | Stream IDs 0x0000- | | Stream IDs 0x0100- | + | 0x00FF mapped from | | 0x01FF mapped from | + | RC0 Requestor IDs | | RC1 Requestor IDs | + +--------------------+ +--------------------+ + | | + | | + +----------------+---------------+ + | + |Device ID Mapping + v + +----------------------------+ + | ITS Node 0 | + | | + | Device IDs: | + | 0x0000 - 0x00FF (from RC0) | + | 0x0100 - 0x01FF (from RC1) | + | 0x0200 - 0xFFFF (No SMMU) | + +----------------------------+ + +Tested-by: Nathan Chen +Reviewed-by: Nicolin Chen +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Message-id: 20250829082543.7680-4-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 01e9a18730e6f56f713ed074603a8b0f2982ed26) +Signed-off-by: Eric Auger +--- + hw/arm/virt-acpi-build.c | 64 ++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 64 insertions(+) + +diff --git a/hw/arm/virt-acpi-build.c b/hw/arm/virt-acpi-build.c +index bef4fabe56..96830f7c4e 100644 +--- a/hw/arm/virt-acpi-build.c ++++ b/hw/arm/virt-acpi-build.c +@@ -45,6 +45,7 @@ + #include "hw/acpi/generic_event_device.h" + #include "hw/acpi/tpm.h" + #include "hw/acpi/hmat.h" ++#include "hw/arm/smmuv3.h" + #include "hw/cxl/cxl.h" + #include "hw/pci/pcie_host.h" + #include "hw/pci/pci.h" +@@ -338,6 +339,67 @@ static int populate_smmuv3_legacy_dev(GArray *sdev_blob) + return sdev.rc_smmu_idmaps->len; + } + ++static int smmuv3_dev_idmap_compare(gconstpointer a, gconstpointer b) ++{ ++ AcpiIortSMMUv3Dev *sdev_a = (AcpiIortSMMUv3Dev *)a; ++ AcpiIortSMMUv3Dev *sdev_b = (AcpiIortSMMUv3Dev *)b; ++ AcpiIortIdMapping *map_a = &g_array_index(sdev_a->rc_smmu_idmaps, ++ AcpiIortIdMapping, 0); ++ AcpiIortIdMapping *map_b = &g_array_index(sdev_b->rc_smmu_idmaps, ++ AcpiIortIdMapping, 0); ++ return map_a->input_base - map_b->input_base; ++} ++ ++static int iort_smmuv3_devices(Object *obj, void *opaque) ++{ ++ VirtMachineState *vms = VIRT_MACHINE(qdev_get_machine()); ++ GArray *sdev_blob = opaque; ++ AcpiIortIdMapping idmap; ++ PlatformBusDevice *pbus; ++ AcpiIortSMMUv3Dev sdev; ++ int min_bus, max_bus; ++ SysBusDevice *sbdev; ++ PCIBus *bus; ++ ++ if (!object_dynamic_cast(obj, TYPE_ARM_SMMUV3)) { ++ return 0; ++ } ++ ++ bus = PCI_BUS(object_property_get_link(obj, "primary-bus", &error_abort)); ++ pbus = PLATFORM_BUS_DEVICE(vms->platform_bus_dev); ++ sbdev = SYS_BUS_DEVICE(obj); ++ sdev.base = platform_bus_get_mmio_addr(pbus, sbdev, 0); ++ sdev.base += vms->memmap[VIRT_PLATFORM_BUS].base; ++ sdev.irq = platform_bus_get_irqn(pbus, sbdev, 0); ++ sdev.irq += vms->irqmap[VIRT_PLATFORM_BUS]; ++ sdev.irq += ARM_SPI_BASE; ++ ++ pci_bus_range(bus, &min_bus, &max_bus); ++ sdev.rc_smmu_idmaps = g_array_new(false, true, sizeof(AcpiIortIdMapping)); ++ idmap.input_base = min_bus << 8, ++ idmap.id_count = (max_bus - min_bus + 1) << 8, ++ g_array_append_val(sdev.rc_smmu_idmaps, idmap); ++ g_array_append_val(sdev_blob, sdev); ++ return 0; ++} ++ ++/* ++ * Populate the struct AcpiIortSMMUv3Dev for all SMMUv3 devices and ++ * return the total number of idmaps. ++ */ ++static int populate_smmuv3_dev(GArray *sdev_blob) ++{ ++ object_child_foreach_recursive(object_get_root(), ++ iort_smmuv3_devices, sdev_blob); ++ /* Sort the smmuv3 devices(if any) by smmu idmap input_base */ ++ g_array_sort(sdev_blob, smmuv3_dev_idmap_compare); ++ /* ++ * Since each SMMUv3 dev is assocaited with specific host bridge, ++ * total number of idmaps equals to total number of smmuv3 devices. ++ */ ++ return sdev_blob->len; ++} ++ + /* Compute ID ranges (RIDs) from RC that are directed to the ITS Group node */ + static void create_rc_its_idmaps(GArray *its_idmaps, GArray *smmuv3_devs) + { +@@ -401,6 +463,8 @@ build_iort(GArray *table_data, BIOSLinker *linker, VirtMachineState *vms) + + if (vms->legacy_smmuv3_present) { + rc_smmu_idmaps_len = populate_smmuv3_legacy_dev(smmuv3_devs); ++ } else { ++ rc_smmu_idmaps_len = populate_smmuv3_dev(smmuv3_devs); + } + + num_smmus = smmuv3_devs->len; +-- +2.47.3 + diff --git a/kvm-hw-intc-Generalize-APIC-helper-names-from-kvm_-to-ac.patch b/kvm-hw-intc-Generalize-APIC-helper-names-from-kvm_-to-ac.patch new file mode 100644 index 0000000..09dbaef --- /dev/null +++ b/kvm-hw-intc-Generalize-APIC-helper-names-from-kvm_-to-ac.patch @@ -0,0 +1,386 @@ +From ea903da8f0546ded20f84a3c4f46139a8206609e Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:24 +0200 +Subject: [PATCH 06/32] hw/intc: Generalize APIC helper names from kvm_* to + accel_* + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [4/30] 2622233d15ed5138cb143d2a5b54fb05c8574654 + +Rename APIC helper functions to use an accel_* prefix instead of kvm_* +to support use by accelerators other than KVM. This is a preparatory +step for integrating MSHV support with common APIC logic. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-5-magnuskulke@linux.microsoft.com +[Remove dead definition of mshv_msi_via_irqfd_enabled. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 638ac1c78457dd93ccd795b9c6c2673af8c7dd21) +Signed-off-by: Igor Mammedov +--- + accel/accel-irq.c | 106 +++++++++++++++++++++++++++++++++++++ + accel/meson.build | 2 +- + hw/intc/ioapic.c | 20 ++++--- + hw/virtio/virtio-pci.c | 21 ++++---- + include/system/accel-irq.h | 37 +++++++++++++ + include/system/mshv.h | 17 ++++++ + 6 files changed, 185 insertions(+), 18 deletions(-) + create mode 100644 accel/accel-irq.c + create mode 100644 include/system/accel-irq.h + +diff --git a/accel/accel-irq.c b/accel/accel-irq.c +new file mode 100644 +index 0000000000..7f864e35c4 +--- /dev/null ++++ b/accel/accel-irq.c +@@ -0,0 +1,106 @@ ++/* ++ * Accelerated irqchip abstraction ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Ziqiao Zhou ++ * Magnus Kulke ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "hw/pci/msi.h" ++ ++#include "system/kvm.h" ++#include "system/mshv.h" ++#include "system/accel-irq.h" ++ ++int accel_irqchip_add_msi_route(KVMRouteChange *c, int vector, PCIDevice *dev) ++{ ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++ if (mshv_msi_via_irqfd_enabled()) { ++ return mshv_irqchip_add_msi_route(vector, dev); ++ } ++#endif ++ if (kvm_enabled()) { ++ return kvm_irqchip_add_msi_route(c, vector, dev); ++ } ++ return -ENOSYS; ++} ++ ++int accel_irqchip_update_msi_route(int vector, MSIMessage msg, PCIDevice *dev) ++{ ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++ if (mshv_msi_via_irqfd_enabled()) { ++ return mshv_irqchip_update_msi_route(vector, msg, dev); ++ } ++#endif ++ if (kvm_enabled()) { ++ return kvm_irqchip_update_msi_route(kvm_state, vector, msg, dev); ++ } ++ return -ENOSYS; ++} ++ ++void accel_irqchip_commit_route_changes(KVMRouteChange *c) ++{ ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++ if (mshv_msi_via_irqfd_enabled()) { ++ mshv_irqchip_commit_routes(); ++ } ++#endif ++ if (kvm_enabled()) { ++ kvm_irqchip_commit_route_changes(c); ++ } ++} ++ ++void accel_irqchip_commit_routes(void) ++{ ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++ if (mshv_msi_via_irqfd_enabled()) { ++ mshv_irqchip_commit_routes(); ++ } ++#endif ++ if (kvm_enabled()) { ++ kvm_irqchip_commit_routes(kvm_state); ++ } ++} ++ ++void accel_irqchip_release_virq(int virq) ++{ ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++ if (mshv_msi_via_irqfd_enabled()) { ++ mshv_irqchip_release_virq(virq); ++ } ++#endif ++ if (kvm_enabled()) { ++ kvm_irqchip_release_virq(kvm_state, virq); ++ } ++} ++ ++int accel_irqchip_add_irqfd_notifier_gsi(EventNotifier *n, EventNotifier *rn, ++ int virq) ++{ ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++ if (mshv_msi_via_irqfd_enabled()) { ++ return mshv_irqchip_add_irqfd_notifier_gsi(n, rn, virq); ++ } ++#endif ++ if (kvm_enabled()) { ++ return kvm_irqchip_add_irqfd_notifier_gsi(kvm_state, n, rn, virq); ++ } ++ return -ENOSYS; ++} ++ ++int accel_irqchip_remove_irqfd_notifier_gsi(EventNotifier *n, int virq) ++{ ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++ if (mshv_msi_via_irqfd_enabled()) { ++ return mshv_irqchip_remove_irqfd_notifier_gsi(n, virq); ++ } ++#endif ++ if (kvm_enabled()) { ++ return kvm_irqchip_remove_irqfd_notifier_gsi(kvm_state, n, virq); ++ } ++ return -ENOSYS; ++} +diff --git a/accel/meson.build b/accel/meson.build +index 25b0f100b5..6349efe682 100644 +--- a/accel/meson.build ++++ b/accel/meson.build +@@ -1,6 +1,6 @@ + common_ss.add(files('accel-common.c')) + specific_ss.add(files('accel-target.c')) +-system_ss.add(files('accel-system.c', 'accel-blocker.c', 'accel-qmp.c')) ++system_ss.add(files('accel-system.c', 'accel-blocker.c', 'accel-qmp.c', 'accel-irq.c')) + user_ss.add(files('accel-user.c')) + + subdir('tcg') +diff --git a/hw/intc/ioapic.c b/hw/intc/ioapic.c +index 133bef852d..e431d00311 100644 +--- a/hw/intc/ioapic.c ++++ b/hw/intc/ioapic.c +@@ -30,12 +30,18 @@ + #include "hw/intc/ioapic_internal.h" + #include "hw/pci/msi.h" + #include "hw/qdev-properties.h" ++#include "system/accel-irq.h" + #include "system/kvm.h" + #include "system/system.h" + #include "hw/i386/apic-msidef.h" + #include "hw/i386/x86-iommu.h" + #include "trace.h" + ++ ++#if defined(CONFIG_KVM) || defined(CONFIG_MSHV) ++#define ACCEL_GSI_IRQFD_POSSIBLE ++#endif ++ + #define APIC_DELIVERY_MODE_SHIFT 8 + #define APIC_POLARITY_SHIFT 14 + #define APIC_TRIG_MODE_SHIFT 15 +@@ -191,10 +197,10 @@ static void ioapic_set_irq(void *opaque, int vector, int level) + + static void ioapic_update_kvm_routes(IOAPICCommonState *s) + { +-#ifdef CONFIG_KVM ++#ifdef ACCEL_GSI_IRQFD_POSSIBLE + int i; + +- if (kvm_irqchip_is_split()) { ++ if (accel_irqchip_is_split()) { + for (i = 0; i < IOAPIC_NUM_PINS; i++) { + MSIMessage msg; + struct ioapic_entry_info info; +@@ -202,15 +208,15 @@ static void ioapic_update_kvm_routes(IOAPICCommonState *s) + if (!info.masked) { + msg.address = info.addr; + msg.data = info.data; +- kvm_irqchip_update_msi_route(kvm_state, i, msg, NULL); ++ accel_irqchip_update_msi_route(i, msg, NULL); + } + } +- kvm_irqchip_commit_routes(kvm_state); ++ accel_irqchip_commit_routes(); + } + #endif + } + +-#ifdef CONFIG_KVM ++#ifdef ACCEL_KERNEL_GSI_IRQFD_POSSIBLE + static void ioapic_iec_notifier(void *private, bool global, + uint32_t index, uint32_t mask) + { +@@ -428,11 +434,11 @@ static const MemoryRegionOps ioapic_io_ops = { + + static void ioapic_machine_done_notify(Notifier *notifier, void *data) + { +-#ifdef CONFIG_KVM ++#ifdef ACCEL_KERNEL_GSI_IRQFD_POSSIBLE + IOAPICCommonState *s = container_of(notifier, IOAPICCommonState, + machine_done); + +- if (kvm_irqchip_is_split()) { ++ if (accel_irqchip_is_split()) { + X86IOMMUState *iommu = x86_iommu_get_default(); + if (iommu) { + /* Register this IOAPIC with IOMMU IEC notifier, so that +diff --git a/hw/virtio/virtio-pci.c b/hw/virtio/virtio-pci.c +index 767216d795..0cdc16217f 100644 +--- a/hw/virtio/virtio-pci.c ++++ b/hw/virtio/virtio-pci.c +@@ -34,6 +34,7 @@ + #include "hw/pci/msi.h" + #include "hw/pci/msix.h" + #include "hw/loader.h" ++#include "system/accel-irq.h" + #include "system/kvm.h" + #include "hw/virtio/virtio-pci.h" + #include "qemu/range.h" +@@ -825,11 +826,11 @@ static int kvm_virtio_pci_vq_vector_use(VirtIOPCIProxy *proxy, + + if (irqfd->users == 0) { + KVMRouteChange c = kvm_irqchip_begin_route_changes(kvm_state); +- ret = kvm_irqchip_add_msi_route(&c, vector, &proxy->pci_dev); ++ ret = accel_irqchip_add_msi_route(&c, vector, &proxy->pci_dev); + if (ret < 0) { + return ret; + } +- kvm_irqchip_commit_route_changes(&c); ++ accel_irqchip_commit_route_changes(&c); + irqfd->virq = ret; + } + irqfd->users++; +@@ -841,7 +842,7 @@ static void kvm_virtio_pci_vq_vector_release(VirtIOPCIProxy *proxy, + { + VirtIOIRQFD *irqfd = &proxy->vector_irqfd[vector]; + if (--irqfd->users == 0) { +- kvm_irqchip_release_virq(kvm_state, irqfd->virq); ++ accel_irqchip_release_virq(irqfd->virq); + } + } + +@@ -850,7 +851,7 @@ static int kvm_virtio_pci_irqfd_use(VirtIOPCIProxy *proxy, + unsigned int vector) + { + VirtIOIRQFD *irqfd = &proxy->vector_irqfd[vector]; +- return kvm_irqchip_add_irqfd_notifier_gsi(kvm_state, n, NULL, irqfd->virq); ++ return accel_irqchip_add_irqfd_notifier_gsi(n, NULL, irqfd->virq); + } + + static void kvm_virtio_pci_irqfd_release(VirtIOPCIProxy *proxy, +@@ -860,7 +861,7 @@ static void kvm_virtio_pci_irqfd_release(VirtIOPCIProxy *proxy, + VirtIOIRQFD *irqfd = &proxy->vector_irqfd[vector]; + int ret; + +- ret = kvm_irqchip_remove_irqfd_notifier_gsi(kvm_state, n, irqfd->virq); ++ ret = accel_irqchip_remove_irqfd_notifier_gsi(n, irqfd->virq); + assert(ret == 0); + } + static int virtio_pci_get_notifier(VirtIOPCIProxy *proxy, int queue_no, +@@ -995,12 +996,12 @@ static int virtio_pci_one_vector_unmask(VirtIOPCIProxy *proxy, + if (proxy->vector_irqfd) { + irqfd = &proxy->vector_irqfd[vector]; + if (irqfd->msg.data != msg.data || irqfd->msg.address != msg.address) { +- ret = kvm_irqchip_update_msi_route(kvm_state, irqfd->virq, msg, +- &proxy->pci_dev); ++ ret = accel_irqchip_update_msi_route(irqfd->virq, msg, ++ &proxy->pci_dev); + if (ret < 0) { + return ret; + } +- kvm_irqchip_commit_routes(kvm_state); ++ accel_irqchip_commit_routes(); + } + } + +@@ -1229,7 +1230,7 @@ static int virtio_pci_set_guest_notifiers(DeviceState *d, int nvqs, bool assign) + VirtioDeviceClass *k = VIRTIO_DEVICE_GET_CLASS(vdev); + int r, n; + bool with_irqfd = msix_enabled(&proxy->pci_dev) && +- kvm_msi_via_irqfd_enabled(); ++ accel_msi_via_irqfd_enabled() ; + + nvqs = MIN(nvqs, VIRTIO_QUEUE_MAX); + +@@ -1433,7 +1434,7 @@ static void virtio_pci_set_vector(VirtIODevice *vdev, + uint16_t new_vector) + { + bool kvm_irqfd = (vdev->status & VIRTIO_CONFIG_S_DRIVER_OK) && +- msix_enabled(&proxy->pci_dev) && kvm_msi_via_irqfd_enabled(); ++ msix_enabled(&proxy->pci_dev) && accel_msi_via_irqfd_enabled(); + + if (new_vector == old_vector) { + return; +diff --git a/include/system/accel-irq.h b/include/system/accel-irq.h +new file mode 100644 +index 0000000000..671fb7dfdb +--- /dev/null ++++ b/include/system/accel-irq.h +@@ -0,0 +1,37 @@ ++/* ++ * Accelerated irqchip abstraction ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Ziqiao Zhou ++ * Magnus Kulke ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef SYSTEM_ACCEL_IRQ_H ++#define SYSTEM_ACCEL_IRQ_H ++#include "hw/pci/msi.h" ++#include "qemu/osdep.h" ++#include "system/kvm.h" ++#include "system/mshv.h" ++ ++static inline bool accel_msi_via_irqfd_enabled(void) ++{ ++ return mshv_msi_via_irqfd_enabled() || kvm_msi_via_irqfd_enabled(); ++} ++ ++static inline bool accel_irqchip_is_split(void) ++{ ++ return mshv_msi_via_irqfd_enabled() || kvm_irqchip_is_split(); ++} ++ ++int accel_irqchip_add_msi_route(KVMRouteChange *c, int vector, PCIDevice *dev); ++int accel_irqchip_update_msi_route(int vector, MSIMessage msg, PCIDevice *dev); ++void accel_irqchip_commit_route_changes(KVMRouteChange *c); ++void accel_irqchip_commit_routes(void); ++void accel_irqchip_release_virq(int virq); ++int accel_irqchip_add_irqfd_notifier_gsi(EventNotifier *n, EventNotifier *rn, ++ int virq); ++int accel_irqchip_remove_irqfd_notifier_gsi(EventNotifier *n, int virq); ++#endif +diff --git a/include/system/mshv.h b/include/system/mshv.h +index 342f1ef6a9..2a504ed81f 100644 +--- a/include/system/mshv.h ++++ b/include/system/mshv.h +@@ -22,4 +22,21 @@ + #define CONFIG_MSHV_IS_POSSIBLE + #endif + ++#ifdef CONFIG_MSHV_IS_POSSIBLE ++extern bool mshv_allowed; ++#define mshv_enabled() (mshv_allowed) ++#else /* CONFIG_MSHV_IS_POSSIBLE */ ++#define mshv_enabled() false ++#endif ++#define mshv_msi_via_irqfd_enabled() false ++ ++/* interrupt */ ++int mshv_irqchip_add_msi_route(int vector, PCIDevice *dev); ++int mshv_irqchip_update_msi_route(int virq, MSIMessage msg, PCIDevice *dev); ++void mshv_irqchip_commit_routes(void); ++void mshv_irqchip_release_virq(int virq); ++int mshv_irqchip_add_irqfd_notifier_gsi(const EventNotifier *n, ++ const EventNotifier *rn, int virq); ++int mshv_irqchip_remove_irqfd_notifier_gsi(const EventNotifier *n, int virq); ++ + #endif +-- +2.47.3 + diff --git a/kvm-hw-intc-ioapic-Fix-ACCEL_KERNEL_GSI_IRQFD_POSSIBLE-t.patch b/kvm-hw-intc-ioapic-Fix-ACCEL_KERNEL_GSI_IRQFD_POSSIBLE-t.patch new file mode 100644 index 0000000..fc1797d --- /dev/null +++ b/kvm-hw-intc-ioapic-Fix-ACCEL_KERNEL_GSI_IRQFD_POSSIBLE-t.patch @@ -0,0 +1,68 @@ +From 7ecba7856c08f410391ccb12620aeb304d800815 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?C=C3=A9dric=20Le=20Goater?= +Date: Thu, 6 Nov 2025 11:51:48 +0100 +Subject: [PATCH 3/4] hw/intc/ioapic: Fix ACCEL_KERNEL_GSI_IRQFD_POSSIBLE typo +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Cédric Le Goater +RH-MergeRequest: 449: hw/intc/ioapic: Fix ACCEL_KERNEL_GSI_IRQFD_POSSIBLE typo +RH-Jira: RHEL-139028 +RH-Acked-by: Igor Mammedov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/1] 61c1a3f68e0a5403a90f1d1f2cb793cc83e0cea9 (clegoate/qemu-kvm-centos) + +Commit 638ac1c78457 introduced a regression in interrupt remapping +when running a VM configured with an intel-iommu device and an +assigned PCI VF. During boot, Linux reports repeated messages : + + [ 15.416794] __common_interrupt: 2.37 No irq handler for vector + [ 15.417266] __common_interrupt: 2.37 No irq handler for vector + [ 15.417733] __common_interrupt: 2.37 No irq handler for vector + [ 15.418202] __common_interrupt: 2.37 No irq handler for vector + [ 15.418670] __common_interrupt: 2.37 No irq handler for vector + +and may eventually hang. + +The issue is caused by the incorrect use of the macro +ACCEL_KERNEL_GSI_IRQFD_POSSIBLE, which should instead be +ACCEL_GSI_IRQFD_POSSIBLE. + +Fixes: 638ac1c78457 ("hw/intc: Generalize APIC helper names from kvm_* to accel_*") +Cc: Magnus Kulke +Signed-off-by: Cédric Le Goater +Reviewed-by: Philippe Mathieu-Daudé +Message-ID: <20251106105148.737093-1-clg@redhat.com> +Signed-off-by: Philippe Mathieu-Daudé +(cherry picked from commit 3abfbb571143ba865488b6c11f8ad75dda97d1a3) +Signed-off-by: Cédric Le Goater +--- + hw/intc/ioapic.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) + +diff --git a/hw/intc/ioapic.c b/hw/intc/ioapic.c +index e431d00311..38e4384648 100644 +--- a/hw/intc/ioapic.c ++++ b/hw/intc/ioapic.c +@@ -216,7 +216,7 @@ static void ioapic_update_kvm_routes(IOAPICCommonState *s) + #endif + } + +-#ifdef ACCEL_KERNEL_GSI_IRQFD_POSSIBLE ++#ifdef ACCEL_GSI_IRQFD_POSSIBLE + static void ioapic_iec_notifier(void *private, bool global, + uint32_t index, uint32_t mask) + { +@@ -434,7 +434,7 @@ static const MemoryRegionOps ioapic_io_ops = { + + static void ioapic_machine_done_notify(Notifier *notifier, void *data) + { +-#ifdef ACCEL_KERNEL_GSI_IRQFD_POSSIBLE ++#ifdef ACCEL_GSI_IRQFD_POSSIBLE + IOAPICCommonState *s = container_of(notifier, IOAPICCommonState, + machine_done); + +-- +2.47.3 + diff --git a/kvm-hw-pci-Introduce-pci_setup_iommu_per_bus-for-per-bus.patch b/kvm-hw-pci-Introduce-pci_setup_iommu_per_bus-for-per-bus.patch new file mode 100644 index 0000000..b22c39f --- /dev/null +++ b/kvm-hw-pci-Introduce-pci_setup_iommu_per_bus-for-per-bus.patch @@ -0,0 +1,150 @@ +From 34d06db7ea02cd3a0a07082fef93e08bfbf0b06a Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:28 +0100 +Subject: [PATCH 10/16] hw/pci: Introduce pci_setup_iommu_per_bus() for per-bus + IOMMU ops retrieval + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [6/11] 0c41f77254cd66a3648c14c5d4ba2dfdbd396665 (eauger1/centos-qemu-kvm) + +Currently, pci_setup_iommu() registers IOMMU ops for a given PCIBus. +However, when retrieving IOMMU ops for a device using +pci_device_get_iommu_bus_devfn(), the function checks the parent_dev +and fetches IOMMU ops from the parent device, even if the current +bus does not have any associated IOMMU ops. + +This behavior works for now because QEMU's IOMMU implementations are +globally scoped, and host bridges rely on the bypass_iommu property +to skip IOMMU translation when needed. + +However, this model will break with the soon to be introduced +arm-smmuv3 device, which allows users to associate the IOMMU +with a specific PCIe root complex (e.g., the default pcie.0 +or a pxb-pcie root complex). + +For example, consider the following setup with multiple root +complexes: + +-device arm-smmuv3,primary-bus=pcie.0,id=smmuv3.0 \ +... +-device pxb-pcie,id=pcie.1,bus_nr=8,bus=pcie.0 \ +-device pcie-root-port,id=pcie.port1,bus=pcie.1 \ +-device virtio-net-pci,bus=pcie.port1 + +In Qemu, pxb-pcie acts as a special root complex whose parent is +effectively the default root complex(pcie.0). Hence, though pcie.1 +has no associated SMMUv3 as per above, pci_device_get_iommu_bus_devfn() +will incorrectly return the IOMMU ops from pcie.0 due to the fallback +via parent_dev. + +To fix this, introduce a new helper pci_setup_iommu_per_bus() that +explicitly sets the new iommu_per_bus field in the PCIBus structure. +This helper will be used in a subsequent patch that adds support for +the new arm-smmuv3 device. + +Update pci_device_get_iommu_bus_devfn() to use iommu_per_bus when +determining the correct IOMMU ops, ensuring accurate behavior for +per-bus IOMMUs. + +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Nathan Chen +Tested-by: Eric Auger +Reviewed-by: Nicolin Chen +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Message-id: 20250829082543.7680-7-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 951bc76fb669eab96cc60e38a50097ad4435163e) +Signed-off-by: Eric Auger +--- + hw/pci/pci.c | 31 +++++++++++++++++++++++++++++++ + include/hw/pci/pci.h | 2 ++ + include/hw/pci/pci_bus.h | 1 + + 3 files changed, 34 insertions(+) + +diff --git a/hw/pci/pci.c b/hw/pci/pci.c +index c70b5ceeba..0012cc12e7 100644 +--- a/hw/pci/pci.c ++++ b/hw/pci/pci.c +@@ -2909,6 +2909,19 @@ static void pci_device_get_iommu_bus_devfn(PCIDevice *dev, + } + } + ++ /* ++ * When multiple PCI Express Root Buses are defined using pxb-pcie, ++ * the IOMMU configuration may be specific to each root bus. However, ++ * pxb-pcie acts as a special root complex whose parent is effectively ++ * the default root complex(pcie.0). Ensure that we retrieve the ++ * correct IOMMU ops(if any) in such cases. ++ */ ++ if (pci_bus_is_express(iommu_bus) && pci_bus_is_root(iommu_bus)) { ++ if (parent_bus->iommu_per_bus) { ++ break; ++ } ++ } ++ + iommu_bus = parent_bus; + } + +@@ -3169,6 +3182,24 @@ void pci_setup_iommu(PCIBus *bus, const PCIIOMMUOps *ops, void *opaque) + bus->iommu_opaque = opaque; + } + ++/* ++ * Similar to pci_setup_iommu(), but sets iommu_per_bus to true, ++ * indicating that the IOMMU is specific to this bus. This is used by ++ * IOMMU implementations that are tied to a specific PCIe root complex. ++ * ++ * In QEMU, pxb-pcie behaves as a special root complex whose parent is ++ * effectively the default root complex (pcie.0). The iommu_per_bus ++ * is checked in pci_device_get_iommu_bus_devfn() to ensure the correct ++ * IOMMU ops are returned, avoiding the use of the parent’s IOMMU when ++ * it's not appropriate. ++ */ ++void pci_setup_iommu_per_bus(PCIBus *bus, const PCIIOMMUOps *ops, ++ void *opaque) ++{ ++ pci_setup_iommu(bus, ops, opaque); ++ bus->iommu_per_bus = true; ++} ++ + static void pci_dev_get_w64(PCIBus *b, PCIDevice *dev, void *opaque) + { + Range *range = opaque; +diff --git a/include/hw/pci/pci.h b/include/hw/pci/pci.h +index 6b7d3ac8a3..6bccb25ac2 100644 +--- a/include/hw/pci/pci.h ++++ b/include/hw/pci/pci.h +@@ -773,6 +773,8 @@ int pci_iommu_unregister_iotlb_notifier(PCIDevice *dev, uint32_t pasid, + */ + void pci_setup_iommu(PCIBus *bus, const PCIIOMMUOps *ops, void *opaque); + ++void pci_setup_iommu_per_bus(PCIBus *bus, const PCIIOMMUOps *ops, void *opaque); ++ + pcibus_t pci_bar_address(PCIDevice *d, + int reg, uint8_t type, pcibus_t size); + +diff --git a/include/hw/pci/pci_bus.h b/include/hw/pci/pci_bus.h +index 2261312546..c738446788 100644 +--- a/include/hw/pci/pci_bus.h ++++ b/include/hw/pci/pci_bus.h +@@ -35,6 +35,7 @@ struct PCIBus { + enum PCIBusFlags flags; + const PCIIOMMUOps *iommu_ops; + void *iommu_opaque; ++ bool iommu_per_bus; + uint8_t devfn_min; + uint32_t slot_reserved_mask; + pci_set_irq_fn set_irq; +-- +2.47.3 + diff --git a/kvm-hw-s390x-Fix-a-possible-crash-with-passed-through-vi.patch b/kvm-hw-s390x-Fix-a-possible-crash-with-passed-through-vi.patch new file mode 100644 index 0000000..709d857 --- /dev/null +++ b/kvm-hw-s390x-Fix-a-possible-crash-with-passed-through-vi.patch @@ -0,0 +1,83 @@ +From f607a40a84b80b2cb33ef3bb42b60b84af596cc9 Mon Sep 17 00:00:00 2001 +From: Thomas Huth +Date: Tue, 18 Nov 2025 18:40:47 +0100 +Subject: [PATCH 3/4] hw/s390x: Fix a possible crash with passed-through virtio + devices +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 427: s390x: Fix a possible crash with passed-through virtio devices +RH-Jira: RHEL-128085 +RH-Acked-by: Cornelia Huck +RH-Acked-by: Cédric Le Goater +RH-Commit: [1/1] a368a65a46a8d85f7ae83cfb5af23b1a341ea9d4 (thuth/qemu-kvm-cs) + +JIRA: https://issues.redhat.com/browse/RHEL-128085 + +Consider the following nested setup: An L1 host uses some virtio device +(e.g. virtio-keyboard) for the L2 guest, and this L2 guest passes this +device through to the L3 guest. Since the L3 guest sees a virtio device, +it might send virtio notifications to the QEMU in L2 for that device. +But since the QEMU in L2 defined this device as vfio-ccw, the function +handle_virtio_ccw_notify() cannot handle this and crashes: It calls +virtio_ccw_get_vdev() that casts sch->driver_data into a VirtioCcwDevice, +but since "sch" belongs to a vfio-ccw device, that driver_data rather +points to a CcwDevice instead. So as soon as QEMU tries to use some +VirtioCcwDevice specific data from that device, we've lost. + +We must not take virtio notifications for such devices. Thus fix the +issue by adding a check to the handle_virtio_ccw_notify() handler to +refuse all devices that are not our own virtio devices. Like in the +other branches that detect wrong settings, we return -EINVAL from the +function, which will later be placed in GPR2 to inform the guest about +the error. + +Reviewed-by: Halil Pasic +Reviewed-by: Eric Farman +Tested-by: Eric Farman +Reviewed-by: Cornelia Huck +Acked-by: Christian Borntraeger +Signed-off-by: Thomas Huth +Message-ID: <20251118174047.73103-1-thuth@redhat.com> +(cherry picked from commit e5cb62e7b6f99d45a42f0cd358d76d6ee2cef5cd) +--- + hw/s390x/s390-hypercall.c | 14 ++++++++++++++ + 1 file changed, 14 insertions(+) + +diff --git a/hw/s390x/s390-hypercall.c b/hw/s390x/s390-hypercall.c +index ac1b08b2cd..508dd97ca0 100644 +--- a/hw/s390x/s390-hypercall.c ++++ b/hw/s390x/s390-hypercall.c +@@ -10,6 +10,7 @@ + */ + + #include "qemu/osdep.h" ++#include "qemu/error-report.h" + #include "cpu.h" + #include "hw/s390x/s390-virtio-ccw.h" + #include "hw/s390x/s390-hypercall.h" +@@ -42,6 +43,19 @@ static int handle_virtio_ccw_notify(uint64_t subch_id, uint64_t data) + if (!sch || !css_subch_visible(sch)) { + return -EINVAL; + } ++ if (sch->id.cu_type != VIRTIO_CCW_CU_TYPE) { ++ /* ++ * This might happen in nested setups: If the L1 host defined the ++ * L2 guest with a virtio device (e.g. virtio-keyboard), and the ++ * L2 guest passes this device through to the L3 guest, the L3 guest ++ * might send virtio notifications to the QEMU in L2 for that device. ++ * But since the QEMU in L2 defined this device as vfio-ccw, it's not ++ * a VirtIODevice that we can handle here! ++ */ ++ warn_report_once("Got virtio notification for unsupported device " ++ "on subchannel %02x.%1x.%04x!", cssid, ssid, schid); ++ return -EINVAL; ++ } + + vdev = virtio_ccw_get_vdev(sch); + if (vq_idx >= VIRTIO_QUEUE_MAX || !virtio_queue_get_num(vdev, vq_idx)) { +-- +2.47.3 + diff --git a/kvm-hw-uefi-add-variable-digest-to-vmstate.patch b/kvm-hw-uefi-add-variable-digest-to-vmstate.patch new file mode 100644 index 0000000..d9c8a13 --- /dev/null +++ b/kvm-hw-uefi-add-variable-digest-to-vmstate.patch @@ -0,0 +1,89 @@ +From b12eac2c22066ee4ff568a4108b38d7bdcd958cc Mon Sep 17 00:00:00 2001 +From: Gerd Hoffmann +Date: Wed, 4 Mar 2026 08:05:34 +0100 +Subject: [PATCH 1/2] hw/uefi: add variable digest to vmstate +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Gerd Hoffmann +RH-MergeRequest: 471: hw/uefi: add variable digest to vmstate +RH-Jira: RHEL-153058 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/1] c300b30db6bafaacd11b8383531f04873bc8b428 (kraxel.rh/centos-src-qemu-kvm) + +Add digest to vmstate if needed. Also clear digest before loading +to make sure it is initialized. + +Fixes: db1ecfb473ac ("hw/uefi: add var-service-vars.c") +Signed-off-by: Gerd Hoffmann +Reviewed-by: Philippe Mathieu-Daudé +Message-ID: <20260304075954.584423-1-kraxel@redhat.com> +Signed-off-by: Philippe Mathieu-Daudé +(cherry picked from commit b28c3ad1d63c2fe167b6f93fad1616ecd769e599) + +Resolves: RHEL-153058 +--- + hw/uefi/var-service-vars.c | 36 ++++++++++++++++++++++++++++++++++++ + 1 file changed, 36 insertions(+) + +diff --git a/hw/uefi/var-service-vars.c b/hw/uefi/var-service-vars.c +index 8533533ea5..ed4e0a6494 100644 +--- a/hw/uefi/var-service-vars.c ++++ b/hw/uefi/var-service-vars.c +@@ -37,8 +37,40 @@ const VMStateDescription vmstate_uefi_time = { + }, + }; + ++static int uefi_vars_pre_load(void *opaque) ++{ ++ uefi_variable *var = opaque; ++ ++ /* clear digest which is optional in the live migration data stream */ ++ var->digest = NULL; ++ var->digest_size = 0; ++ return 0; ++} ++ ++static bool uefi_vars_digest_is_needed(void *opaque) ++{ ++ uefi_variable *var = opaque; ++ ++ if ((var->attributes & EFI_VARIABLE_TIME_BASED_AUTHENTICATED_WRITE_ACCESS) && ++ !uefi_vars_is_sb_any(var)) { ++ return true; ++ } ++ return false; ++} ++ ++const VMStateDescription vmstate_uefi_variable_digest = { ++ .name = "uefi-variable-digest", ++ .needed = uefi_vars_digest_is_needed, ++ .fields = (VMStateField[]) { ++ VMSTATE_UINT32(digest_size, uefi_variable), ++ VMSTATE_VBUFFER_ALLOC_UINT32(digest, uefi_variable, 0, NULL, digest_size), ++ VMSTATE_END_OF_LIST() ++ }, ++}; ++ + const VMStateDescription vmstate_uefi_variable = { + .name = "uefi-variable", ++ .pre_load = uefi_vars_pre_load, + .fields = (VMStateField[]) { + VMSTATE_UINT8_ARRAY_V(guid.data, uefi_variable, sizeof(QemuUUID), 0), + VMSTATE_UINT32(name_size, uefi_variable), +@@ -49,6 +81,10 @@ const VMStateDescription vmstate_uefi_variable = { + VMSTATE_STRUCT(time, uefi_variable, 0, vmstate_uefi_time, efi_time), + VMSTATE_END_OF_LIST() + }, ++ .subsections = (const VMStateDescription * const []) { ++ &vmstate_uefi_variable_digest, ++ NULL ++ } + }; + + uefi_variable *uefi_vars_find_variable(uefi_vars_state *uv, QemuUUID guid, +-- +2.47.3 + diff --git a/kvm-include-hw-hyperv-Add-MSHV-ABI-header-definitions.patch b/kvm-include-hw-hyperv-Add-MSHV-ABI-header-definitions.patch new file mode 100644 index 0000000..a7351ea --- /dev/null +++ b/kvm-include-hw-hyperv-Add-MSHV-ABI-header-definitions.patch @@ -0,0 +1,1264 @@ +From 28c5559bcd0f99ca2f361705f7cada968c83eb49 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:25 +0200 +Subject: [PATCH 07/32] include/hw/hyperv: Add MSHV ABI header definitions + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [5/30] 61968a1149ddbc713982cac62fa09e327590fb47 + +Introduce headers for the Microsoft Hypervisor (MSHV) userspace ABI, +including IOCTLs and structures used to interface with the hypervisor. + +These definitions are based on the upstream Linux MSHV interface and +will be used by the MSHV accelerator backend in later patches. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-6-magnuskulke@linux.microsoft.com +[Do not use __uN types. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 7db6086287502ad5a015ccc31e85ec9cac7e2716) +Signed-off-by: Igor Mammedov +--- + include/hw/hyperv/hvgdk.h | 20 + + include/hw/hyperv/hvgdk_mini.h | 817 ++++++++++++++++++++++++++++++++ + include/hw/hyperv/hvhdk.h | 249 ++++++++++ + include/hw/hyperv/hvhdk_mini.h | 102 ++++ + scripts/update-linux-headers.sh | 2 +- + 5 files changed, 1189 insertions(+), 1 deletion(-) + create mode 100644 include/hw/hyperv/hvgdk.h + create mode 100644 include/hw/hyperv/hvgdk_mini.h + create mode 100644 include/hw/hyperv/hvhdk.h + create mode 100644 include/hw/hyperv/hvhdk_mini.h + +diff --git a/include/hw/hyperv/hvgdk.h b/include/hw/hyperv/hvgdk.h +new file mode 100644 +index 0000000000..71161f477c +--- /dev/null ++++ b/include/hw/hyperv/hvgdk.h +@@ -0,0 +1,20 @@ ++/* ++ * Type definitions for the mshv guest interface. ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef HW_HYPERV_HVGDK_H ++#define HW_HYPERV_HVGDK_H ++ ++#define HVGDK_H_VERSION (25125) ++ ++enum hv_unimplemented_msr_action { ++ HV_UNIMPLEMENTED_MSR_ACTION_FAULT = 0, ++ HV_UNIMPLEMENTED_MSR_ACTION_IGNORE_WRITE_READ_ZERO = 1, ++ HV_UNIMPLEMENTED_MSR_ACTION_COUNT = 2, ++}; ++ ++#endif /* HW_HYPERV_HVGDK_H */ +diff --git a/include/hw/hyperv/hvgdk_mini.h b/include/hw/hyperv/hvgdk_mini.h +new file mode 100644 +index 0000000000..d89315f545 +--- /dev/null ++++ b/include/hw/hyperv/hvgdk_mini.h +@@ -0,0 +1,817 @@ ++/* ++ * Userspace interfaces for /dev/mshv* devices and derived fds ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef HW_HYPERV_HVGDK_MINI_H ++#define HW_HYPERV_HVGDK_MINI_H ++ ++#define MSHV_IOCTL 0xB8 ++ ++typedef enum hv_register_name { ++ /* Pending Interruption Register */ ++ HV_REGISTER_PENDING_INTERRUPTION = 0x00010002, ++ ++ /* X64 User-Mode Registers */ ++ HV_X64_REGISTER_RAX = 0x00020000, ++ HV_X64_REGISTER_RCX = 0x00020001, ++ HV_X64_REGISTER_RDX = 0x00020002, ++ HV_X64_REGISTER_RBX = 0x00020003, ++ HV_X64_REGISTER_RSP = 0x00020004, ++ HV_X64_REGISTER_RBP = 0x00020005, ++ HV_X64_REGISTER_RSI = 0x00020006, ++ HV_X64_REGISTER_RDI = 0x00020007, ++ HV_X64_REGISTER_R8 = 0x00020008, ++ HV_X64_REGISTER_R9 = 0x00020009, ++ HV_X64_REGISTER_R10 = 0x0002000A, ++ HV_X64_REGISTER_R11 = 0x0002000B, ++ HV_X64_REGISTER_R12 = 0x0002000C, ++ HV_X64_REGISTER_R13 = 0x0002000D, ++ HV_X64_REGISTER_R14 = 0x0002000E, ++ HV_X64_REGISTER_R15 = 0x0002000F, ++ HV_X64_REGISTER_RIP = 0x00020010, ++ HV_X64_REGISTER_RFLAGS = 0x00020011, ++ ++ /* X64 Floating Point and Vector Registers */ ++ HV_X64_REGISTER_XMM0 = 0x00030000, ++ HV_X64_REGISTER_XMM1 = 0x00030001, ++ HV_X64_REGISTER_XMM2 = 0x00030002, ++ HV_X64_REGISTER_XMM3 = 0x00030003, ++ HV_X64_REGISTER_XMM4 = 0x00030004, ++ HV_X64_REGISTER_XMM5 = 0x00030005, ++ HV_X64_REGISTER_XMM6 = 0x00030006, ++ HV_X64_REGISTER_XMM7 = 0x00030007, ++ HV_X64_REGISTER_XMM8 = 0x00030008, ++ HV_X64_REGISTER_XMM9 = 0x00030009, ++ HV_X64_REGISTER_XMM10 = 0x0003000A, ++ HV_X64_REGISTER_XMM11 = 0x0003000B, ++ HV_X64_REGISTER_XMM12 = 0x0003000C, ++ HV_X64_REGISTER_XMM13 = 0x0003000D, ++ HV_X64_REGISTER_XMM14 = 0x0003000E, ++ HV_X64_REGISTER_XMM15 = 0x0003000F, ++ HV_X64_REGISTER_FP_MMX0 = 0x00030010, ++ HV_X64_REGISTER_FP_MMX1 = 0x00030011, ++ HV_X64_REGISTER_FP_MMX2 = 0x00030012, ++ HV_X64_REGISTER_FP_MMX3 = 0x00030013, ++ HV_X64_REGISTER_FP_MMX4 = 0x00030014, ++ HV_X64_REGISTER_FP_MMX5 = 0x00030015, ++ HV_X64_REGISTER_FP_MMX6 = 0x00030016, ++ HV_X64_REGISTER_FP_MMX7 = 0x00030017, ++ HV_X64_REGISTER_FP_CONTROL_STATUS = 0x00030018, ++ HV_X64_REGISTER_XMM_CONTROL_STATUS = 0x00030019, ++ ++ /* X64 Control Registers */ ++ HV_X64_REGISTER_CR0 = 0x00040000, ++ HV_X64_REGISTER_CR2 = 0x00040001, ++ HV_X64_REGISTER_CR3 = 0x00040002, ++ HV_X64_REGISTER_CR4 = 0x00040003, ++ HV_X64_REGISTER_CR8 = 0x00040004, ++ HV_X64_REGISTER_XFEM = 0x00040005, ++ ++ /* X64 Segment Registers */ ++ HV_X64_REGISTER_ES = 0x00060000, ++ HV_X64_REGISTER_CS = 0x00060001, ++ HV_X64_REGISTER_SS = 0x00060002, ++ HV_X64_REGISTER_DS = 0x00060003, ++ HV_X64_REGISTER_FS = 0x00060004, ++ HV_X64_REGISTER_GS = 0x00060005, ++ HV_X64_REGISTER_LDTR = 0x00060006, ++ HV_X64_REGISTER_TR = 0x00060007, ++ ++ /* X64 Table Registers */ ++ HV_X64_REGISTER_IDTR = 0x00070000, ++ HV_X64_REGISTER_GDTR = 0x00070001, ++ ++ /* X64 Virtualized MSRs */ ++ HV_X64_REGISTER_TSC = 0x00080000, ++ HV_X64_REGISTER_EFER = 0x00080001, ++ HV_X64_REGISTER_KERNEL_GS_BASE = 0x00080002, ++ HV_X64_REGISTER_APIC_BASE = 0x00080003, ++ HV_X64_REGISTER_PAT = 0x00080004, ++ HV_X64_REGISTER_SYSENTER_CS = 0x00080005, ++ HV_X64_REGISTER_SYSENTER_EIP = 0x00080006, ++ HV_X64_REGISTER_SYSENTER_ESP = 0x00080007, ++ HV_X64_REGISTER_STAR = 0x00080008, ++ HV_X64_REGISTER_LSTAR = 0x00080009, ++ HV_X64_REGISTER_CSTAR = 0x0008000A, ++ HV_X64_REGISTER_SFMASK = 0x0008000B, ++ HV_X64_REGISTER_INITIAL_APIC_ID = 0x0008000C, ++ ++ /* X64 Cache control MSRs */ ++ HV_X64_REGISTER_MSR_MTRR_CAP = 0x0008000D, ++ HV_X64_REGISTER_MSR_MTRR_DEF_TYPE = 0x0008000E, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE0 = 0x00080010, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE1 = 0x00080011, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE2 = 0x00080012, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE3 = 0x00080013, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE4 = 0x00080014, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE5 = 0x00080015, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE6 = 0x00080016, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE7 = 0x00080017, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE8 = 0x00080018, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASE9 = 0x00080019, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASEA = 0x0008001A, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASEB = 0x0008001B, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASEC = 0x0008001C, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASED = 0x0008001D, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASEE = 0x0008001E, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_BASEF = 0x0008001F, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK0 = 0x00080040, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK1 = 0x00080041, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK2 = 0x00080042, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK3 = 0x00080043, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK4 = 0x00080044, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK5 = 0x00080045, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK6 = 0x00080046, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK7 = 0x00080047, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK8 = 0x00080048, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASK9 = 0x00080049, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASKA = 0x0008004A, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASKB = 0x0008004B, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASKC = 0x0008004C, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASKD = 0x0008004D, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASKE = 0x0008004E, ++ HV_X64_REGISTER_MSR_MTRR_PHYS_MASKF = 0x0008004F, ++ HV_X64_REGISTER_MSR_MTRR_FIX64K00000 = 0x00080070, ++ HV_X64_REGISTER_MSR_MTRR_FIX16K80000 = 0x00080071, ++ HV_X64_REGISTER_MSR_MTRR_FIX16KA0000 = 0x00080072, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KC0000 = 0x00080073, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KC8000 = 0x00080074, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KD0000 = 0x00080075, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KD8000 = 0x00080076, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KE0000 = 0x00080077, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KE8000 = 0x00080078, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KF0000 = 0x00080079, ++ HV_X64_REGISTER_MSR_MTRR_FIX4KF8000 = 0x0008007A, ++ ++ HV_X64_REGISTER_TSC_AUX = 0x0008007B, ++ HV_X64_REGISTER_BNDCFGS = 0x0008007C, ++ HV_X64_REGISTER_DEBUG_CTL = 0x0008007D, ++ ++ /* Available */ ++ ++ HV_X64_REGISTER_SPEC_CTRL = 0x00080084, ++ HV_X64_REGISTER_TSC_ADJUST = 0x00080096, ++ ++ /* Other MSRs */ ++ HV_X64_REGISTER_MSR_IA32_MISC_ENABLE = 0x000800A0, ++ ++ /* Misc */ ++ HV_REGISTER_GUEST_OS_ID = 0x00090002, ++ HV_REGISTER_REFERENCE_TSC = 0x00090017, ++ ++ /* Hypervisor-defined Registers (Synic) */ ++ HV_REGISTER_SINT0 = 0x000A0000, ++ HV_REGISTER_SINT1 = 0x000A0001, ++ HV_REGISTER_SINT2 = 0x000A0002, ++ HV_REGISTER_SINT3 = 0x000A0003, ++ HV_REGISTER_SINT4 = 0x000A0004, ++ HV_REGISTER_SINT5 = 0x000A0005, ++ HV_REGISTER_SINT6 = 0x000A0006, ++ HV_REGISTER_SINT7 = 0x000A0007, ++ HV_REGISTER_SINT8 = 0x000A0008, ++ HV_REGISTER_SINT9 = 0x000A0009, ++ HV_REGISTER_SINT10 = 0x000A000A, ++ HV_REGISTER_SINT11 = 0x000A000B, ++ HV_REGISTER_SINT12 = 0x000A000C, ++ HV_REGISTER_SINT13 = 0x000A000D, ++ HV_REGISTER_SINT14 = 0x000A000E, ++ HV_REGISTER_SINT15 = 0x000A000F, ++ HV_REGISTER_SCONTROL = 0x000A0010, ++ HV_REGISTER_SVERSION = 0x000A0011, ++ HV_REGISTER_SIEFP = 0x000A0012, ++ HV_REGISTER_SIMP = 0x000A0013, ++ HV_REGISTER_EOM = 0x000A0014, ++ HV_REGISTER_SIRBP = 0x000A0015, ++} hv_register_name; ++ ++enum hv_intercept_type { ++ HV_INTERCEPT_TYPE_X64_IO_PORT = 0X00000000, ++ HV_INTERCEPT_TYPE_X64_MSR = 0X00000001, ++ HV_INTERCEPT_TYPE_X64_CPUID = 0X00000002, ++ HV_INTERCEPT_TYPE_EXCEPTION = 0X00000003, ++ ++ /* Used to be HV_INTERCEPT_TYPE_REGISTER */ ++ HV_INTERCEPT_TYPE_RESERVED0 = 0X00000004, ++ HV_INTERCEPT_TYPE_MMIO = 0X00000005, ++ HV_INTERCEPT_TYPE_X64_GLOBAL_CPUID = 0X00000006, ++ HV_INTERCEPT_TYPE_X64_APIC_SMI = 0X00000007, ++ HV_INTERCEPT_TYPE_HYPERCALL = 0X00000008, ++ ++ HV_INTERCEPT_TYPE_X64_APIC_INIT_SIPI = 0X00000009, ++ HV_INTERCEPT_MC_UPDATE_PATCH_LEVEL_MSR_READ = 0X0000000A, ++ ++ HV_INTERCEPT_TYPE_X64_APIC_WRITE = 0X0000000B, ++ HV_INTERCEPT_TYPE_X64_MSR_INDEX = 0X0000000C, ++ HV_INTERCEPT_TYPE_MAX, ++ HV_INTERCEPT_TYPE_INVALID = 0XFFFFFFFF, ++}; ++ ++struct hv_u128 { ++ uint64_t low_part; ++ uint64_t high_part; ++}; ++ ++union hv_x64_xmm_control_status_register { ++ struct hv_u128 as_uint128; ++ struct { ++ union { ++ /* long mode */ ++ uint64_t last_fp_rdp; ++ /* 32 bit mode */ ++ struct { ++ uint32_t last_fp_dp; ++ uint16_t last_fp_ds; ++ uint16_t padding; ++ }; ++ }; ++ uint32_t xmm_status_control; ++ uint32_t xmm_status_control_mask; ++ }; ++}; ++ ++union hv_x64_fp_register { ++ struct hv_u128 as_uint128; ++ struct { ++ uint64_t mantissa; ++ uint64_t biased_exponent:15; ++ uint64_t sign:1; ++ uint64_t reserved:48; ++ }; ++}; ++ ++union hv_x64_pending_exception_event { ++ uint64_t as_uint64[2]; ++ struct { ++ uint32_t event_pending:1; ++ uint32_t event_type:3; ++ uint32_t reserved0:4; ++ uint32_t deliver_error_code:1; ++ uint32_t reserved1:7; ++ uint32_t vector:16; ++ uint32_t error_code; ++ uint64_t exception_parameter; ++ }; ++}; ++ ++union hv_x64_pending_virtualization_fault_event { ++ uint64_t as_uint64[2]; ++ struct { ++ uint32_t event_pending:1; ++ uint32_t event_type:3; ++ uint32_t reserved0:4; ++ uint32_t reserved1:8; ++ uint32_t parameter0:16; ++ uint32_t code; ++ uint64_t parameter1; ++ }; ++}; ++ ++union hv_x64_pending_interruption_register { ++ uint64_t as_uint64; ++ struct { ++ uint32_t interruption_pending:1; ++ uint32_t interruption_type:3; ++ uint32_t deliver_error_code:1; ++ uint32_t instruction_length:4; ++ uint32_t nested_event:1; ++ uint32_t reserved:6; ++ uint32_t interruption_vector:16; ++ uint32_t error_code; ++ }; ++}; ++ ++union hv_x64_register_sev_control { ++ uint64_t as_uint64; ++ struct { ++ uint64_t enable_encrypted_state:1; ++ uint64_t reserved_z:11; ++ uint64_t vmsa_gpa_page_number:52; ++ }; ++}; ++ ++union hv_x64_msr_npiep_config_contents { ++ uint64_t as_uint64; ++ struct { ++ /* ++ * These bits enable instruction execution prevention for ++ * specific instructions. ++ */ ++ uint64_t prevents_gdt:1; ++ uint64_t prevents_idt:1; ++ uint64_t prevents_ldt:1; ++ uint64_t prevents_tr:1; ++ ++ /* The reserved bits must always be 0. */ ++ uint64_t reserved:60; ++ }; ++}; ++ ++typedef struct hv_x64_segment_register { ++ uint64_t base; ++ uint32_t limit; ++ uint16_t selector; ++ union { ++ struct { ++ uint16_t segment_type:4; ++ uint16_t non_system_segment:1; ++ uint16_t descriptor_privilege_level:2; ++ uint16_t present:1; ++ uint16_t reserved:4; ++ uint16_t available:1; ++ uint16_t _long:1; ++ uint16_t _default:1; ++ uint16_t granularity:1; ++ }; ++ uint16_t attributes; ++ }; ++} hv_x64_segment_register; ++ ++typedef struct hv_x64_table_register { ++ uint16_t pad[3]; ++ uint16_t limit; ++ uint64_t base; ++} hv_x64_table_register; ++ ++union hv_x64_fp_control_status_register { ++ struct hv_u128 as_uint128; ++ struct { ++ uint16_t fp_control; ++ uint16_t fp_status; ++ uint8_t fp_tag; ++ uint8_t reserved; ++ uint16_t last_fp_op; ++ union { ++ /* long mode */ ++ uint64_t last_fp_rip; ++ /* 32 bit mode */ ++ struct { ++ uint32_t last_fp_eip; ++ uint16_t last_fp_cs; ++ uint16_t padding; ++ }; ++ }; ++ }; ++}; ++ ++/* General Hypervisor Register Content Definitions */ ++ ++union hv_explicit_suspend_register { ++ uint64_t as_uint64; ++ struct { ++ uint64_t suspended:1; ++ uint64_t reserved:63; ++ }; ++}; ++ ++union hv_internal_activity_register { ++ uint64_t as_uint64; ++ ++ struct { ++ uint64_t startup_suspend:1; ++ uint64_t halt_suspend:1; ++ uint64_t idle_suspend:1; ++ uint64_t rsvd_z:61; ++ }; ++}; ++ ++union hv_x64_interrupt_state_register { ++ uint64_t as_uint64; ++ struct { ++ uint64_t interrupt_shadow:1; ++ uint64_t nmi_masked:1; ++ uint64_t reserved:62; ++ }; ++}; ++ ++union hv_intercept_suspend_register { ++ uint64_t as_uint64; ++ struct { ++ uint64_t suspended:1; ++ uint64_t reserved:63; ++ }; ++}; ++ ++typedef union hv_register_value { ++ struct hv_u128 reg128; ++ uint64_t reg64; ++ uint32_t reg32; ++ uint16_t reg16; ++ uint8_t reg8; ++ union hv_x64_fp_register fp; ++ union hv_x64_fp_control_status_register fp_control_status; ++ union hv_x64_xmm_control_status_register xmm_control_status; ++ struct hv_x64_segment_register segment; ++ struct hv_x64_table_register table; ++ union hv_explicit_suspend_register explicit_suspend; ++ union hv_intercept_suspend_register intercept_suspend; ++ union hv_internal_activity_register internal_activity; ++ union hv_x64_interrupt_state_register interrupt_state; ++ union hv_x64_pending_interruption_register pending_interruption; ++ union hv_x64_msr_npiep_config_contents npiep_config; ++ union hv_x64_pending_exception_event pending_exception_event; ++ union hv_x64_pending_virtualization_fault_event ++ pending_virtualization_fault_event; ++ union hv_x64_register_sev_control sev_control; ++} hv_register_value; ++ ++typedef struct hv_register_assoc { ++ uint32_t name; /* enum hv_register_name */ ++ uint32_t reserved1; ++ uint64_t reserved2; ++ union hv_register_value value; ++} hv_register_assoc; ++ ++union hv_input_vtl { ++ uint8_t as_uint8; ++ struct { ++ uint8_t target_vtl:4; ++ uint8_t use_target_vtl:1; ++ uint8_t reserved_z:3; ++ }; ++}; ++ ++typedef struct hv_input_get_vp_registers { ++ uint64_t partition_id; ++ uint32_t vp_index; ++ union hv_input_vtl input_vtl; ++ uint8_t rsvd_z8; ++ uint16_t rsvd_z16; ++ uint32_t names[]; ++} hv_input_get_vp_registers; ++ ++typedef struct hv_input_set_vp_registers { ++ uint64_t partition_id; ++ uint32_t vp_index; ++ union hv_input_vtl input_vtl; ++ uint8_t rsvd_z8; ++ uint16_t rsvd_z16; ++ struct hv_register_assoc elements[]; ++} hv_input_set_vp_registers; ++ ++#define MSHV_VP_MAX_REGISTERS 128 ++ ++struct mshv_vp_registers { ++ int count; /* at most MSHV_VP_MAX_REGISTERS */ ++ struct hv_register_assoc *regs; ++}; ++ ++union hv_interrupt_control { ++ uint64_t as_uint64; ++ struct { ++ uint32_t interrupt_type; /* enum hv_interrupt type */ ++ uint32_t level_triggered:1; ++ uint32_t logical_dest_mode:1; ++ uint32_t rsvd:30; ++ }; ++}; ++ ++struct hv_input_assert_virtual_interrupt { ++ uint64_t partition_id; ++ union hv_interrupt_control control; ++ uint64_t dest_addr; /* cpu's apic id */ ++ uint32_t vector; ++ uint8_t target_vtl; ++ uint8_t rsvd_z0; ++ uint16_t rsvd_z1; ++}; ++ ++/* /dev/mshv */ ++#define MSHV_CREATE_PARTITION _IOW(MSHV_IOCTL, 0x00, struct mshv_create_partition) ++#define MSHV_CREATE_VP _IOW(MSHV_IOCTL, 0x01, struct mshv_create_vp) ++ ++/* Partition fds created with MSHV_CREATE_PARTITION */ ++#define MSHV_INITIALIZE_PARTITION _IO(MSHV_IOCTL, 0x00) ++#define MSHV_SET_GUEST_MEMORY _IOW(MSHV_IOCTL, 0x02, struct mshv_user_mem_region) ++#define MSHV_IRQFD _IOW(MSHV_IOCTL, 0x03, struct mshv_user_irqfd) ++#define MSHV_IOEVENTFD _IOW(MSHV_IOCTL, 0x04, struct mshv_user_ioeventfd) ++#define MSHV_SET_MSI_ROUTING _IOW(MSHV_IOCTL, 0x05, struct mshv_user_irq_table) ++ ++/* ++ ******************************** ++ * VP APIs for child partitions * ++ ******************************** ++ */ ++ ++struct hv_local_interrupt_controller_state { ++ /* HV_X64_INTERRUPT_CONTROLLER_STATE */ ++ uint32_t apic_id; ++ uint32_t apic_version; ++ uint32_t apic_ldr; ++ uint32_t apic_dfr; ++ uint32_t apic_spurious; ++ uint32_t apic_isr[8]; ++ uint32_t apic_tmr[8]; ++ uint32_t apic_irr[8]; ++ uint32_t apic_esr; ++ uint32_t apic_icr_high; ++ uint32_t apic_icr_low; ++ uint32_t apic_lvt_timer; ++ uint32_t apic_lvt_thermal; ++ uint32_t apic_lvt_perfmon; ++ uint32_t apic_lvt_lint0; ++ uint32_t apic_lvt_lint1; ++ uint32_t apic_lvt_error; ++ uint32_t apic_lvt_cmci; ++ uint32_t apic_error_status; ++ uint32_t apic_initial_count; ++ uint32_t apic_counter_value; ++ uint32_t apic_divide_configuration; ++ uint32_t apic_remote_read; ++}; ++ ++/* Generic hypercall */ ++#define MSHV_ROOT_HVCALL _IOWR(MSHV_IOCTL, 0x07, struct mshv_root_hvcall) ++ ++/* From hvgdk_mini.h */ ++ ++#define HV_X64_MSR_GUEST_OS_ID 0x40000000 ++#define HV_X64_MSR_SINT0 0x40000090 ++#define HV_X64_MSR_SINT1 0x40000091 ++#define HV_X64_MSR_SINT2 0x40000092 ++#define HV_X64_MSR_SINT3 0x40000093 ++#define HV_X64_MSR_SINT4 0x40000094 ++#define HV_X64_MSR_SINT5 0x40000095 ++#define HV_X64_MSR_SINT6 0x40000096 ++#define HV_X64_MSR_SINT7 0x40000097 ++#define HV_X64_MSR_SINT8 0x40000098 ++#define HV_X64_MSR_SINT9 0x40000099 ++#define HV_X64_MSR_SINT10 0x4000009A ++#define HV_X64_MSR_SINT11 0x4000009B ++#define HV_X64_MSR_SINT12 0x4000009C ++#define HV_X64_MSR_SINT13 0x4000009D ++#define HV_X64_MSR_SINT14 0x4000009E ++#define HV_X64_MSR_SINT15 0x4000009F ++#define HV_X64_MSR_SCONTROL 0x40000080 ++#define HV_X64_MSR_SIEFP 0x40000082 ++#define HV_X64_MSR_SIMP 0x40000083 ++#define HV_X64_MSR_REFERENCE_TSC 0x40000021 ++#define HV_X64_MSR_EOM 0x40000084 ++ ++/* Define port identifier type. */ ++union hv_port_id { ++ uint32_t asuint32_t; ++ struct { ++ uint32_t id:24; ++ uint32_t reserved:8; ++ }; ++}; ++ ++#define HV_MESSAGE_SIZE (256) ++#define HV_MESSAGE_PAYLOAD_BYTE_COUNT (240) ++#define HV_MESSAGE_PAYLOAD_QWORD_COUNT (30) ++ ++/* Define hypervisor message types. */ ++enum hv_message_type { ++ HVMSG_NONE = 0x00000000, ++ ++ /* Memory access messages. */ ++ HVMSG_UNMAPPED_GPA = 0x80000000, ++ HVMSG_GPA_INTERCEPT = 0x80000001, ++ HVMSG_UNACCEPTED_GPA = 0x80000003, ++ HVMSG_GPA_ATTRIBUTE_INTERCEPT = 0x80000004, ++ ++ /* Timer notification messages. */ ++ HVMSG_TIMER_EXPIRED = 0x80000010, ++ ++ /* Error messages. */ ++ HVMSG_INVALID_VP_REGISTER_VALUE = 0x80000020, ++ HVMSG_UNRECOVERABLE_EXCEPTION = 0x80000021, ++ HVMSG_UNSUPPORTED_FEATURE = 0x80000022, ++ ++ /* ++ * Opaque intercept message. The original intercept message is only ++ * accessible from the mapped intercept message page. ++ */ ++ HVMSG_OPAQUE_INTERCEPT = 0x8000003F, ++ ++ /* Trace buffer complete messages. */ ++ HVMSG_EVENTLOG_BUFFERCOMPLETE = 0x80000040, ++ ++ /* Hypercall intercept */ ++ HVMSG_HYPERCALL_INTERCEPT = 0x80000050, ++ ++ /* SynIC intercepts */ ++ HVMSG_SYNIC_EVENT_INTERCEPT = 0x80000060, ++ HVMSG_SYNIC_SINT_INTERCEPT = 0x80000061, ++ HVMSG_SYNIC_SINT_DELIVERABLE = 0x80000062, ++ ++ /* Async call completion intercept */ ++ HVMSG_ASYNC_CALL_COMPLETION = 0x80000070, ++ ++ /* Root scheduler messages */ ++ HVMSG_SCHEDULER_VP_SIGNAL_BITSE = 0x80000100, ++ HVMSG_SCHEDULER_VP_SIGNAL_PAIR = 0x80000101, ++ ++ /* Platform-specific processor intercept messages. */ ++ HVMSG_X64_IO_PORT_INTERCEPT = 0x80010000, ++ HVMSG_X64_MSR_INTERCEPT = 0x80010001, ++ HVMSG_X64_CPUID_INTERCEPT = 0x80010002, ++ HVMSG_X64_EXCEPTION_INTERCEPT = 0x80010003, ++ HVMSG_X64_APIC_EOI = 0x80010004, ++ HVMSG_X64_LEGACY_FP_ERROR = 0x80010005, ++ HVMSG_X64_IOMMU_PRQ = 0x80010006, ++ HVMSG_X64_HALT = 0x80010007, ++ HVMSG_X64_INTERRUPTION_DELIVERABLE = 0x80010008, ++ HVMSG_X64_SIPI_INTERCEPT = 0x80010009, ++ HVMSG_X64_SEV_VMGEXIT_INTERCEPT = 0x80010013, ++}; ++ ++union hv_x64_vp_execution_state { ++ uint16_t as_uint16; ++ struct { ++ uint16_t cpl:2; ++ uint16_t cr0_pe:1; ++ uint16_t cr0_am:1; ++ uint16_t efer_lma:1; ++ uint16_t debug_active:1; ++ uint16_t interruption_pending:1; ++ uint16_t vtl:4; ++ uint16_t enclave_mode:1; ++ uint16_t interrupt_shadow:1; ++ uint16_t virtualization_fault_active:1; ++ uint16_t reserved:2; ++ }; ++}; ++ ++/* From openvmm::hvdef */ ++enum hv_x64_intercept_access_type { ++ HV_X64_INTERCEPT_ACCESS_TYPE_READ = 0, ++ HV_X64_INTERCEPT_ACCESS_TYPE_WRITE = 1, ++ HV_X64_INTERCEPT_ACCESS_TYPE_EXECUTE = 2, ++}; ++ ++struct hv_x64_intercept_message_header { ++ uint32_t vp_index; ++ uint8_t instruction_length:4; ++ uint8_t cr8:4; /* Only set for exo partitions */ ++ uint8_t intercept_access_type; ++ union hv_x64_vp_execution_state execution_state; ++ struct hv_x64_segment_register cs_segment; ++ uint64_t rip; ++ uint64_t rflags; ++}; ++ ++union hv_x64_io_port_access_info { ++ uint8_t as_uint8; ++ struct { ++ uint8_t access_size:3; ++ uint8_t string_op:1; ++ uint8_t rep_prefix:1; ++ uint8_t reserved:3; ++ }; ++}; ++ ++typedef struct hv_x64_io_port_intercept_message { ++ struct hv_x64_intercept_message_header header; ++ uint16_t port_number; ++ union hv_x64_io_port_access_info access_info; ++ uint8_t instruction_byte_count; ++ uint32_t reserved; ++ uint64_t rax; ++ uint8_t instruction_bytes[16]; ++ struct hv_x64_segment_register ds_segment; ++ struct hv_x64_segment_register es_segment; ++ uint64_t rcx; ++ uint64_t rsi; ++ uint64_t rdi; ++} hv_x64_io_port_intercept_message; ++ ++union hv_x64_memory_access_info { ++ uint8_t as_uint8; ++ struct { ++ uint8_t gva_valid:1; ++ uint8_t gva_gpa_valid:1; ++ uint8_t hypercall_output_pending:1; ++ uint8_t tlb_locked_no_overlay:1; ++ uint8_t reserved:4; ++ }; ++}; ++ ++struct hv_x64_memory_intercept_message { ++ struct hv_x64_intercept_message_header header; ++ uint32_t cache_type; /* enum hv_cache_type */ ++ uint8_t instruction_byte_count; ++ union hv_x64_memory_access_info memory_access_info; ++ uint8_t tpr_priority; ++ uint8_t reserved1; ++ uint64_t guest_virtual_address; ++ uint64_t guest_physical_address; ++ uint8_t instruction_bytes[16]; ++}; ++ ++union hv_message_flags { ++ uint8_t asu8; ++ struct { ++ uint8_t msg_pending:1; ++ uint8_t reserved:7; ++ }; ++}; ++ ++struct hv_message_header { ++ uint32_t message_type; ++ uint8_t payload_size; ++ union hv_message_flags message_flags; ++ uint8_t reserved[2]; ++ union { ++ uint64_t sender; ++ union hv_port_id port; ++ }; ++}; ++ ++struct hv_message { ++ struct hv_message_header header; ++ union { ++ uint64_t payload[HV_MESSAGE_PAYLOAD_QWORD_COUNT]; ++ } u; ++}; ++ ++/* From github.com/rust-vmm/mshv-bindings/src/x86_64/regs.rs */ ++ ++struct hv_cpuid_entry { ++ uint32_t function; ++ uint32_t index; ++ uint32_t flags; ++ uint32_t eax; ++ uint32_t ebx; ++ uint32_t ecx; ++ uint32_t edx; ++ uint32_t padding[3]; ++}; ++ ++struct hv_cpuid { ++ uint32_t nent; ++ uint32_t padding; ++ struct hv_cpuid_entry entries[0]; ++}; ++ ++#define IA32_MSR_TSC 0x00000010 ++#define IA32_MSR_EFER 0xC0000080 ++#define IA32_MSR_KERNEL_GS_BASE 0xC0000102 ++#define IA32_MSR_APIC_BASE 0x0000001B ++#define IA32_MSR_PAT 0x0277 ++#define IA32_MSR_SYSENTER_CS 0x00000174 ++#define IA32_MSR_SYSENTER_ESP 0x00000175 ++#define IA32_MSR_SYSENTER_EIP 0x00000176 ++#define IA32_MSR_STAR 0xC0000081 ++#define IA32_MSR_LSTAR 0xC0000082 ++#define IA32_MSR_CSTAR 0xC0000083 ++#define IA32_MSR_SFMASK 0xC0000084 ++ ++#define IA32_MSR_MTRR_CAP 0x00FE ++#define IA32_MSR_MTRR_DEF_TYPE 0x02FF ++#define IA32_MSR_MTRR_PHYSBASE0 0x0200 ++#define IA32_MSR_MTRR_PHYSMASK0 0x0201 ++#define IA32_MSR_MTRR_PHYSBASE1 0x0202 ++#define IA32_MSR_MTRR_PHYSMASK1 0x0203 ++#define IA32_MSR_MTRR_PHYSBASE2 0x0204 ++#define IA32_MSR_MTRR_PHYSMASK2 0x0205 ++#define IA32_MSR_MTRR_PHYSBASE3 0x0206 ++#define IA32_MSR_MTRR_PHYSMASK3 0x0207 ++#define IA32_MSR_MTRR_PHYSBASE4 0x0208 ++#define IA32_MSR_MTRR_PHYSMASK4 0x0209 ++#define IA32_MSR_MTRR_PHYSBASE5 0x020A ++#define IA32_MSR_MTRR_PHYSMASK5 0x020B ++#define IA32_MSR_MTRR_PHYSBASE6 0x020C ++#define IA32_MSR_MTRR_PHYSMASK6 0x020D ++#define IA32_MSR_MTRR_PHYSBASE7 0x020E ++#define IA32_MSR_MTRR_PHYSMASK7 0x020F ++ ++#define IA32_MSR_MTRR_FIX64K_00000 0x0250 ++#define IA32_MSR_MTRR_FIX16K_80000 0x0258 ++#define IA32_MSR_MTRR_FIX16K_A0000 0x0259 ++#define IA32_MSR_MTRR_FIX4K_C0000 0x0268 ++#define IA32_MSR_MTRR_FIX4K_C8000 0x0269 ++#define IA32_MSR_MTRR_FIX4K_D0000 0x026A ++#define IA32_MSR_MTRR_FIX4K_D8000 0x026B ++#define IA32_MSR_MTRR_FIX4K_E0000 0x026C ++#define IA32_MSR_MTRR_FIX4K_E8000 0x026D ++#define IA32_MSR_MTRR_FIX4K_F0000 0x026E ++#define IA32_MSR_MTRR_FIX4K_F8000 0x026F ++ ++#define IA32_MSR_TSC_AUX 0xC0000103 ++#define IA32_MSR_BNDCFGS 0x00000d90 ++#define IA32_MSR_DEBUG_CTL 0x1D9 ++#define IA32_MSR_SPEC_CTRL 0x00000048 ++#define IA32_MSR_TSC_ADJUST 0x0000003b ++ ++#define IA32_MSR_MISC_ENABLE 0x000001a0 ++ ++#define HV_TRANSLATE_GVA_VALIDATE_READ (0x0001) ++#define HV_TRANSLATE_GVA_VALIDATE_WRITE (0x0002) ++#define HV_TRANSLATE_GVA_VALIDATE_EXECUTE (0x0004) ++ ++#define HV_HYP_PAGE_SHIFT 12 ++#define HV_HYP_PAGE_SIZE BIT(HV_HYP_PAGE_SHIFT) ++#define HV_HYP_PAGE_MASK (~(HV_HYP_PAGE_SIZE - 1)) ++ ++#define HVCALL_GET_PARTITION_PROPERTY 0x0044 ++#define HVCALL_SET_PARTITION_PROPERTY 0x0045 ++#define HVCALL_GET_VP_REGISTERS 0x0050 ++#define HVCALL_SET_VP_REGISTERS 0x0051 ++#define HVCALL_TRANSLATE_VIRTUAL_ADDRESS 0x0052 ++#define HVCALL_REGISTER_INTERCEPT_RESULT 0x0091 ++#define HVCALL_ASSERT_VIRTUAL_INTERRUPT 0x0094 ++ ++#endif /* HW_HYPERV_HVGDK_MINI_H */ +diff --git a/include/hw/hyperv/hvhdk.h b/include/hw/hyperv/hvhdk.h +new file mode 100644 +index 0000000000..866c8211bf +--- /dev/null ++++ b/include/hw/hyperv/hvhdk.h +@@ -0,0 +1,249 @@ ++/* ++ * Type definitions for the mshv host. ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef HW_HYPERV_HVHDK_H ++#define HW_HYPERV_HVHDK_H ++ ++#define HV_PARTITION_SYNTHETIC_PROCESSOR_FEATURES_BANKS 1 ++ ++struct hv_input_set_partition_property { ++ uint64_t partition_id; ++ uint32_t property_code; /* enum hv_partition_property_code */ ++ uint32_t padding; ++ uint64_t property_value; ++}; ++ ++union hv_partition_synthetic_processor_features { ++ uint64_t as_uint64[HV_PARTITION_SYNTHETIC_PROCESSOR_FEATURES_BANKS]; ++ ++ struct { ++ /* ++ * Report a hypervisor is present. CPUID leaves ++ * 0x40000000 and 0x40000001 are supported. ++ */ ++ uint64_t hypervisor_present:1; ++ ++ /* ++ * Features associated with HV#1: ++ */ ++ ++ /* Report support for Hv1 (CPUID leaves 0x40000000 - 0x40000006). */ ++ uint64_t hv1:1; ++ ++ /* ++ * Access to HV_X64_MSR_VP_RUNTIME. ++ * Corresponds to access_vp_run_time_reg privilege. ++ */ ++ uint64_t access_vp_run_time_reg:1; ++ ++ /* ++ * Access to HV_X64_MSR_TIME_REF_COUNT. ++ * Corresponds to access_partition_reference_counter privilege. ++ */ ++ uint64_t access_partition_reference_counter:1; ++ ++ /* ++ * Access to SINT-related registers (HV_X64_MSR_SCONTROL through ++ * HV_X64_MSR_EOM and HV_X64_MSR_SINT0 through HV_X64_MSR_SINT15). ++ * Corresponds to access_synic_regs privilege. ++ */ ++ uint64_t access_synic_regs:1; ++ ++ /* ++ * Access to synthetic timers and associated MSRs ++ * (HV_X64_MSR_STIMER0_CONFIG through HV_X64_MSR_STIMER3_COUNT). ++ * Corresponds to access_synthetic_timer_regs privilege. ++ */ ++ uint64_t access_synthetic_timer_regs:1; ++ ++ /* ++ * Access to APIC MSRs (HV_X64_MSR_EOI, HV_X64_MSR_ICR and ++ * HV_X64_MSR_TPR) as well as the VP assist page. ++ * Corresponds to access_intr_ctrl_regs privilege. ++ */ ++ uint64_t access_intr_ctrl_regs:1; ++ ++ /* ++ * Access to registers associated with hypercalls ++ * (HV_X64_MSR_GUEST_OS_ID and HV_X64_MSR_HYPERCALL). ++ * Corresponds to access_hypercall_msrs privilege. ++ */ ++ uint64_t access_hypercall_regs:1; ++ ++ /* VP index can be queried. corresponds to access_vp_index privilege. */ ++ uint64_t access_vp_index:1; ++ ++ /* ++ * Access to the reference TSC. Corresponds to ++ * access_partition_reference_tsc privilege. ++ */ ++ uint64_t access_partition_reference_tsc:1; ++ ++ /* ++ * Partition has access to the guest idle reg. Corresponds to ++ * access_guest_idle_reg privilege. ++ */ ++ uint64_t access_guest_idle_reg:1; ++ ++ /* ++ * Partition has access to frequency regs. corresponds to ++ * access_frequency_regs privilege. ++ */ ++ uint64_t access_frequency_regs:1; ++ ++ uint64_t reserved_z12:1; /* Reserved for access_reenlightenment_controls */ ++ uint64_t reserved_z13:1; /* Reserved for access_root_scheduler_reg */ ++ uint64_t reserved_z14:1; /* Reserved for access_tsc_invariant_controls */ ++ ++ /* ++ * Extended GVA ranges for HvCallFlushVirtualAddressList hypercall. ++ * Corresponds to privilege. ++ */ ++ uint64_t enable_extended_gva_ranges_for_flush_virtual_address_list:1; ++ ++ uint64_t reserved_z16:1; /* Reserved for access_vsm. */ ++ uint64_t reserved_z17:1; /* Reserved for access_vp_registers. */ ++ ++ /* Use fast hypercall output. Corresponds to privilege. */ ++ uint64_t fast_hypercall_output:1; ++ ++ uint64_t reserved_z19:1; /* Reserved for enable_extended_hypercalls. */ ++ ++ /* ++ * HvStartVirtualProcessor can be used to start virtual processors. ++ * Corresponds to privilege. ++ */ ++ uint64_t start_virtual_processor:1; ++ ++ uint64_t reserved_z21:1; /* Reserved for Isolation. */ ++ ++ /* Synthetic timers in direct mode. */ ++ uint64_t direct_synthetic_timers:1; ++ ++ uint64_t reserved_z23:1; /* Reserved for synthetic time unhalted timer */ ++ ++ /* Use extended processor masks. */ ++ uint64_t extended_processor_masks:1; ++ ++ /* ++ * HvCallFlushVirtualAddressSpace / HvCallFlushVirtualAddressList are ++ * supported. ++ */ ++ uint64_t tb_flush_hypercalls:1; ++ ++ /* HvCallSendSyntheticClusterIpi is supported. */ ++ uint64_t synthetic_cluster_ipi:1; ++ ++ /* HvCallNotifyLongSpinWait is supported. */ ++ uint64_t notify_long_spin_wait:1; ++ ++ /* HvCallQueryNumaDistance is supported. */ ++ uint64_t query_numa_distance:1; ++ ++ /* HvCallSignalEvent is supported. Corresponds to privilege. */ ++ uint64_t signal_events:1; ++ ++ /* HvCallRetargetDeviceInterrupt is supported. */ ++ uint64_t retarget_device_interrupt:1; ++ ++ /* HvCallRestorePartitionTime is supported. */ ++ uint64_t restore_time:1; ++ ++ /* EnlightenedVmcs nested enlightenment is supported. */ ++ uint64_t enlightened_vmcs:1; ++ ++ uint64_t reserved:30; ++ }; ++}; ++ ++enum hv_translate_gva_result_code { ++ HV_TRANSLATE_GVA_SUCCESS = 0, ++ ++ /* Translation failures. */ ++ HV_TRANSLATE_GVA_PAGE_NOT_PRESENT = 1, ++ HV_TRANSLATE_GVA_PRIVILEGE_VIOLATION = 2, ++ HV_TRANSLATE_GVA_INVALIDE_PAGE_TABLE_FLAGS = 3, ++ ++ /* GPA access failures. */ ++ HV_TRANSLATE_GVA_GPA_UNMAPPED = 4, ++ HV_TRANSLATE_GVA_GPA_NO_READ_ACCESS = 5, ++ HV_TRANSLATE_GVA_GPA_NO_WRITE_ACCESS = 6, ++ HV_TRANSLATE_GVA_GPA_ILLEGAL_OVERLAY_ACCESS = 7, ++ ++ /* ++ * Intercept for memory access by either ++ * - a higher VTL ++ * - a nested hypervisor (due to a violation of the nested page table) ++ */ ++ HV_TRANSLATE_GVA_INTERCEPT = 8, ++ ++ HV_TRANSLATE_GVA_GPA_UNACCEPTED = 9, ++}; ++ ++union hv_translate_gva_result { ++ uint64_t as_uint64; ++ struct { ++ uint32_t result_code; /* enum hv_translate_hva_result_code */ ++ uint32_t cache_type:8; ++ uint32_t overlay_page:1; ++ uint32_t reserved:23; ++ }; ++}; ++ ++typedef struct hv_input_translate_virtual_address { ++ uint64_t partition_id; ++ uint32_t vp_index; ++ uint32_t padding; ++ uint64_t control_flags; ++ uint64_t gva_page; ++} hv_input_translate_virtual_address; ++ ++typedef struct hv_output_translate_virtual_address { ++ union hv_translate_gva_result translation_result; ++ uint64_t gpa_page; ++} hv_output_translate_virtual_address; ++ ++typedef struct hv_register_x64_cpuid_result_parameters { ++ struct { ++ uint32_t eax; ++ uint32_t ecx; ++ uint8_t subleaf_specific; ++ uint8_t always_override; ++ uint16_t padding; ++ } input; ++ struct { ++ uint32_t eax; ++ uint32_t eax_mask; ++ uint32_t ebx; ++ uint32_t ebx_mask; ++ uint32_t ecx; ++ uint32_t ecx_mask; ++ uint32_t edx; ++ uint32_t edx_mask; ++ } result; ++} hv_register_x64_cpuid_result_parameters; ++ ++typedef struct hv_register_x64_msr_result_parameters { ++ uint32_t msr_index; ++ uint32_t access_type; ++ uint32_t action; /* enum hv_unimplemented_msr_action */ ++} hv_register_x64_msr_result_parameters; ++ ++union hv_register_intercept_result_parameters { ++ struct hv_register_x64_cpuid_result_parameters cpuid; ++ struct hv_register_x64_msr_result_parameters msr; ++}; ++ ++typedef struct hv_input_register_intercept_result { ++ uint64_t partition_id; ++ uint32_t vp_index; ++ uint32_t intercept_type; /* enum hv_intercept_type */ ++ union hv_register_intercept_result_parameters parameters; ++} hv_input_register_intercept_result; ++ ++#endif /* HW_HYPERV_HVHDK_H */ +diff --git a/include/hw/hyperv/hvhdk_mini.h b/include/hw/hyperv/hvhdk_mini.h +new file mode 100644 +index 0000000000..9c2f3cf5ae +--- /dev/null ++++ b/include/hw/hyperv/hvhdk_mini.h +@@ -0,0 +1,102 @@ ++/* ++ * Type definitions for the mshv host interface. ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef HW_HYPERV_HVHDK_MINI_H ++#define HW_HYPERV_HVHDK_MINI_H ++ ++#define HVHVK_MINI_VERSION (25294) ++ ++/* Each generic set contains 64 elements */ ++#define HV_GENERIC_SET_SHIFT (6) ++#define HV_GENERIC_SET_MASK (63) ++ ++enum hv_generic_set_format { ++ HV_GENERIC_SET_SPARSE_4K, ++ HV_GENERIC_SET_ALL, ++}; ++ ++enum hv_partition_property_code { ++ /* Privilege properties */ ++ HV_PARTITION_PROPERTY_PRIVILEGE_FLAGS = 0x00010000, ++ HV_PARTITION_PROPERTY_SYNTHETIC_PROC_FEATURES = 0x00010001, ++ ++ /* Scheduling properties */ ++ HV_PARTITION_PROPERTY_SUSPEND = 0x00020000, ++ HV_PARTITION_PROPERTY_CPU_RESERVE = 0x00020001, ++ HV_PARTITION_PROPERTY_CPU_CAP = 0x00020002, ++ HV_PARTITION_PROPERTY_CPU_WEIGHT = 0x00020003, ++ HV_PARTITION_PROPERTY_CPU_GROUP_ID = 0x00020004, ++ ++ /* Time properties */ ++ HV_PARTITION_PROPERTY_TIME_FREEZE = 0x00030003, ++ HV_PARTITION_PROPERTY_REFERENCE_TIME = 0x00030005, ++ ++ /* Debugging properties */ ++ HV_PARTITION_PROPERTY_DEBUG_CHANNEL_ID = 0x00040000, ++ ++ /* Resource properties */ ++ HV_PARTITION_PROPERTY_VIRTUAL_TLB_PAGE_COUNT = 0x00050000, ++ HV_PARTITION_PROPERTY_VSM_CONFIG = 0x00050001, ++ HV_PARTITION_PROPERTY_ZERO_MEMORY_ON_RESET = 0x00050002, ++ HV_PARTITION_PROPERTY_PROCESSORS_PER_SOCKET = 0x00050003, ++ HV_PARTITION_PROPERTY_NESTED_TLB_SIZE = 0x00050004, ++ HV_PARTITION_PROPERTY_GPA_PAGE_ACCESS_TRACKING = 0x00050005, ++ HV_PARTITION_PROPERTY_VSM_PERMISSIONS_DIRTY_SINCE_LAST_QUERY = 0x00050006, ++ HV_PARTITION_PROPERTY_SGX_LAUNCH_CONTROL_CONFIG = 0x00050007, ++ HV_PARTITION_PROPERTY_DEFAULT_SGX_LAUNCH_CONTROL0 = 0x00050008, ++ HV_PARTITION_PROPERTY_DEFAULT_SGX_LAUNCH_CONTROL1 = 0x00050009, ++ HV_PARTITION_PROPERTY_DEFAULT_SGX_LAUNCH_CONTROL2 = 0x0005000a, ++ HV_PARTITION_PROPERTY_DEFAULT_SGX_LAUNCH_CONTROL3 = 0x0005000b, ++ HV_PARTITION_PROPERTY_ISOLATION_STATE = 0x0005000c, ++ HV_PARTITION_PROPERTY_ISOLATION_CONTROL = 0x0005000d, ++ HV_PARTITION_PROPERTY_ALLOCATION_ID = 0x0005000e, ++ HV_PARTITION_PROPERTY_MONITORING_ID = 0x0005000f, ++ HV_PARTITION_PROPERTY_IMPLEMENTED_PHYSICAL_ADDRESS_BITS = 0x00050010, ++ HV_PARTITION_PROPERTY_NON_ARCHITECTURAL_CORE_SHARING = 0x00050011, ++ HV_PARTITION_PROPERTY_HYPERCALL_DOORBELL_PAGE = 0x00050012, ++ HV_PARTITION_PROPERTY_ISOLATION_POLICY = 0x00050014, ++ HV_PARTITION_PROPERTY_UNIMPLEMENTED_MSR_ACTION = 0x00050017, ++ HV_PARTITION_PROPERTY_SEV_VMGEXIT_OFFLOADS = 0x00050022, ++ ++ /* Compatibility properties */ ++ HV_PARTITION_PROPERTY_PROCESSOR_VENDOR = 0x00060000, ++ HV_PARTITION_PROPERTY_PROCESSOR_FEATURES_DEPRECATED = 0x00060001, ++ HV_PARTITION_PROPERTY_PROCESSOR_XSAVE_FEATURES = 0x00060002, ++ HV_PARTITION_PROPERTY_PROCESSOR_CL_FLUSH_SIZE = 0x00060003, ++ HV_PARTITION_PROPERTY_ENLIGHTENMENT_MODIFICATIONS = 0x00060004, ++ HV_PARTITION_PROPERTY_COMPATIBILITY_VERSION = 0x00060005, ++ HV_PARTITION_PROPERTY_PHYSICAL_ADDRESS_WIDTH = 0x00060006, ++ HV_PARTITION_PROPERTY_XSAVE_STATES = 0x00060007, ++ HV_PARTITION_PROPERTY_MAX_XSAVE_DATA_SIZE = 0x00060008, ++ HV_PARTITION_PROPERTY_PROCESSOR_CLOCK_FREQUENCY = 0x00060009, ++ HV_PARTITION_PROPERTY_PROCESSOR_FEATURES0 = 0x0006000a, ++ HV_PARTITION_PROPERTY_PROCESSOR_FEATURES1 = 0x0006000b, ++ ++ /* Guest software properties */ ++ HV_PARTITION_PROPERTY_GUEST_OS_ID = 0x00070000, ++ ++ /* Nested virtualization properties */ ++ HV_PARTITION_PROPERTY_PROCESSOR_VIRTUALIZATION_FEATURES = 0x00080000, ++}; ++ ++/* HV Map GPA (Guest Physical Address) Flags */ ++#define HV_MAP_GPA_PERMISSIONS_NONE 0x0 ++#define HV_MAP_GPA_READABLE 0x1 ++#define HV_MAP_GPA_WRITABLE 0x2 ++#define HV_MAP_GPA_KERNEL_EXECUTABLE 0x4 ++#define HV_MAP_GPA_USER_EXECUTABLE 0x8 ++#define HV_MAP_GPA_EXECUTABLE 0xC ++#define HV_MAP_GPA_PERMISSIONS_MASK 0xF ++#define HV_MAP_GPA_ADJUSTABLE 0x8000 ++#define HV_MAP_GPA_NO_ACCESS 0x10000 ++#define HV_MAP_GPA_NOT_CACHED 0x200000 ++#define HV_MAP_GPA_LARGE_PAGE 0x80000000 ++ ++#define HV_PFN_RNG_PAGEBITS 24 /* HV_SPA_PAGE_RANGE_ADDITIONAL_PAGES_BITS */ ++ ++#endif /* HW_HYPERV_HVHDK_MINI_H */ +diff --git a/scripts/update-linux-headers.sh b/scripts/update-linux-headers.sh +index 717c379f9e..828a7809f7 100755 +--- a/scripts/update-linux-headers.sh ++++ b/scripts/update-linux-headers.sh +@@ -195,7 +195,7 @@ rm -rf "$output/linux-headers/linux" + mkdir -p "$output/linux-headers/linux" + for header in const.h stddef.h kvm.h vfio.h vfio_ccw.h vfio_zdev.h vhost.h \ + psci.h psp-sev.h userfaultfd.h memfd.h mman.h nvme_ioctl.h \ +- vduse.h iommufd.h bits.h; do ++ vduse.h iommufd.h bits.h mshv.h; do + cp "$hdrdir/include/linux/$header" "$output/linux-headers/linux" + done + +-- +2.47.3 + diff --git a/kvm-io-fix-use-after-free-in-websocket-handshake-code.patch b/kvm-io-fix-use-after-free-in-websocket-handshake-code.patch new file mode 100644 index 0000000..5c59ad0 --- /dev/null +++ b/kvm-io-fix-use-after-free-in-websocket-handshake-code.patch @@ -0,0 +1,189 @@ +From 728cf99416aaaae2cc0fca6ee88f28ccec33d697 Mon Sep 17 00:00:00 2001 +From: Jon Maloy +Date: Tue, 4 Nov 2025 17:28:47 -0500 +Subject: [PATCH 02/16] io: fix use after free in websocket handshake code +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Jon Maloy +RH-MergeRequest: 419: io: move websock resource release to close method +RH-Jira: RHEL-120116 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/2] acdb5414387815a8b2f0a84a151990875947e855 (jmaloy/jmaloy-qemu-kvm-2) + +JIRA: https://issues.redhat.com/browse/RHEL-120116 +CVE: CVE-2025-11234 + +commit b7a1f2ca45c7865b9e98e02ae605a65fc9458ae9 +Author: Daniel P. Berrangé +Date: Tue Sep 30 12:03:15 2025 +0100 + + io: fix use after free in websocket handshake code + + If the QIOChannelWebsock object is freed while it is waiting to + complete a handshake, a GSource is leaked. This can lead to the + callback firing later on and triggering a use-after-free in the + use of the channel. This was observed in the VNC server with the + following trace from valgrind: + + ==2523108== Invalid read of size 4 + ==2523108== at 0x4054A24: vnc_disconnect_start (vnc.c:1296) + ==2523108== by 0x4054A24: vnc_client_error (vnc.c:1392) + ==2523108== by 0x4068A09: vncws_handshake_done (vnc-ws.c:105) + ==2523108== by 0x44863B4: qio_task_complete (task.c:197) + ==2523108== by 0x448343D: qio_channel_websock_handshake_io (channel-websock.c:588) + ==2523108== by 0x6EDB862: UnknownInlinedFun (gmain.c:3398) + ==2523108== by 0x6EDB862: g_main_context_dispatch_unlocked.lto_priv.0 (gmain.c:4249) + ==2523108== by 0x6EDBAE4: g_main_context_dispatch (gmain.c:4237) + ==2523108== by 0x45EC79F: glib_pollfds_poll (main-loop.c:287) + ==2523108== by 0x45EC79F: os_host_main_loop_wait (main-loop.c:310) + ==2523108== by 0x45EC79F: main_loop_wait (main-loop.c:589) + ==2523108== by 0x423A56D: qemu_main_loop (runstate.c:835) + ==2523108== by 0x454F300: qemu_default_main (main.c:37) + ==2523108== by 0x73D6574: (below main) (libc_start_call_main.h:58) + ==2523108== Address 0x57a6e0dc is 28 bytes inside a block of size 103,608 free'd + ==2523108== at 0x5F2FE43: free (vg_replace_malloc.c:989) + ==2523108== by 0x6EDC444: g_free (gmem.c:208) + ==2523108== by 0x4053F23: vnc_update_client (vnc.c:1153) + ==2523108== by 0x4053F23: vnc_refresh (vnc.c:3225) + ==2523108== by 0x4042881: dpy_refresh (console.c:880) + ==2523108== by 0x4042881: gui_update (console.c:90) + ==2523108== by 0x45EFA1B: timerlist_run_timers.part.0 (qemu-timer.c:562) + ==2523108== by 0x45EFC8F: timerlist_run_timers (qemu-timer.c:495) + ==2523108== by 0x45EFC8F: qemu_clock_run_timers (qemu-timer.c:576) + ==2523108== by 0x45EFC8F: qemu_clock_run_all_timers (qemu-timer.c:663) + ==2523108== by 0x45EC765: main_loop_wait (main-loop.c:600) + ==2523108== by 0x423A56D: qemu_main_loop (runstate.c:835) + ==2523108== by 0x454F300: qemu_default_main (main.c:37) + ==2523108== by 0x73D6574: (below main) (libc_start_call_main.h:58) + ==2523108== Block was alloc'd at + ==2523108== at 0x5F343F3: calloc (vg_replace_malloc.c:1675) + ==2523108== by 0x6EE2F81: g_malloc0 (gmem.c:133) + ==2523108== by 0x4057DA3: vnc_connect (vnc.c:3245) + ==2523108== by 0x448591B: qio_net_listener_channel_func (net-listener.c:54) + ==2523108== by 0x6EDB862: UnknownInlinedFun (gmain.c:3398) + ==2523108== by 0x6EDB862: g_main_context_dispatch_unlocked.lto_priv.0 (gmain.c:4249) + ==2523108== by 0x6EDBAE4: g_main_context_dispatch (gmain.c:4237) + ==2523108== by 0x45EC79F: glib_pollfds_poll (main-loop.c:287) + ==2523108== by 0x45EC79F: os_host_main_loop_wait (main-loop.c:310) + ==2523108== by 0x45EC79F: main_loop_wait (main-loop.c:589) + ==2523108== by 0x423A56D: qemu_main_loop (runstate.c:835) + ==2523108== by 0x454F300: qemu_default_main (main.c:37) + ==2523108== by 0x73D6574: (below main) (libc_start_call_main.h:58) + ==2523108== + + The above can be reproduced by launching QEMU with + + $ qemu-system-x86_64 -vnc localhost:0,websocket=5700 + + and then repeatedly running: + + for i in {1..100}; do + (echo -n "GET / HTTP/1.1" && sleep 0.05) | nc -w 1 localhost 5700 & + done + + CVE-2025-11234 + Reported-by: Grant Millar | Cylo + Reviewed-by: Eric Blake + Signed-off-by: Daniel P. Berrangé + +Signed-off-by: Jon Maloy +--- + include/io/channel-websock.h | 3 ++- + io/channel-websock.c | 22 ++++++++++++++++------ + 2 files changed, 18 insertions(+), 7 deletions(-) + +diff --git a/include/io/channel-websock.h b/include/io/channel-websock.h +index e180827c57..6700cf8946 100644 +--- a/include/io/channel-websock.h ++++ b/include/io/channel-websock.h +@@ -61,7 +61,8 @@ struct QIOChannelWebsock { + size_t payload_remain; + size_t pong_remain; + QIOChannelWebsockMask mask; +- guint io_tag; ++ guint hs_io_tag; /* tracking handshake task */ ++ guint io_tag; /* tracking watch task */ + Error *io_err; + gboolean io_eof; + uint8_t opcode; +diff --git a/io/channel-websock.c b/io/channel-websock.c +index a19b902ff9..ec5e09f9ab 100644 +--- a/io/channel-websock.c ++++ b/io/channel-websock.c +@@ -545,6 +545,7 @@ static gboolean qio_channel_websock_handshake_send(QIOChannel *ioc, + trace_qio_channel_websock_handshake_fail(ioc, error_get_pretty(err)); + qio_task_set_error(task, err); + qio_task_complete(task); ++ wioc->hs_io_tag = 0; + return FALSE; + } + +@@ -560,6 +561,7 @@ static gboolean qio_channel_websock_handshake_send(QIOChannel *ioc, + trace_qio_channel_websock_handshake_complete(ioc); + qio_task_complete(task); + } ++ wioc->hs_io_tag = 0; + return FALSE; + } + trace_qio_channel_websock_handshake_pending(ioc, G_IO_OUT); +@@ -586,6 +588,7 @@ static gboolean qio_channel_websock_handshake_io(QIOChannel *ioc, + trace_qio_channel_websock_handshake_fail(ioc, error_get_pretty(err)); + qio_task_set_error(task, err); + qio_task_complete(task); ++ wioc->hs_io_tag = 0; + return FALSE; + } + if (ret == 0) { +@@ -597,7 +600,7 @@ static gboolean qio_channel_websock_handshake_io(QIOChannel *ioc, + error_propagate(&wioc->io_err, err); + + trace_qio_channel_websock_handshake_reply(ioc); +- qio_channel_add_watch( ++ wioc->hs_io_tag = qio_channel_add_watch( + wioc->master, + G_IO_OUT, + qio_channel_websock_handshake_send, +@@ -907,11 +910,12 @@ void qio_channel_websock_handshake(QIOChannelWebsock *ioc, + + trace_qio_channel_websock_handshake_start(ioc); + trace_qio_channel_websock_handshake_pending(ioc, G_IO_IN); +- qio_channel_add_watch(ioc->master, +- G_IO_IN, +- qio_channel_websock_handshake_io, +- task, +- NULL); ++ ioc->hs_io_tag = qio_channel_add_watch( ++ ioc->master, ++ G_IO_IN, ++ qio_channel_websock_handshake_io, ++ task, ++ NULL); + } + + +@@ -922,6 +926,9 @@ static void qio_channel_websock_finalize(Object *obj) + buffer_free(&ioc->encinput); + buffer_free(&ioc->encoutput); + buffer_free(&ioc->rawinput); ++ if (ioc->hs_io_tag) { ++ g_source_remove(ioc->hs_io_tag); ++ } + if (ioc->io_tag) { + g_source_remove(ioc->io_tag); + } +@@ -1222,6 +1229,9 @@ static int qio_channel_websock_close(QIOChannel *ioc, + buffer_free(&wioc->encinput); + buffer_free(&wioc->encoutput); + buffer_free(&wioc->rawinput); ++ if (wioc->hs_io_tag) { ++ g_clear_handle_id(&wioc->hs_io_tag, g_source_remove); ++ } + if (wioc->io_tag) { + g_clear_handle_id(&wioc->io_tag, g_source_remove); + } +-- +2.47.3 + diff --git a/kvm-io-move-websock-resource-release-to-close-method.patch b/kvm-io-move-websock-resource-release-to-close-method.patch new file mode 100644 index 0000000..052a53e --- /dev/null +++ b/kvm-io-move-websock-resource-release-to-close-method.patch @@ -0,0 +1,84 @@ +From 9aaede253bb55035f0a1171fb1c4eda847ca9493 Mon Sep 17 00:00:00 2001 +From: Jon Maloy +Date: Tue, 4 Nov 2025 17:23:29 -0500 +Subject: [PATCH 01/16] io: move websock resource release to close method +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Jon Maloy +RH-MergeRequest: 419: io: move websock resource release to close method +RH-Jira: RHEL-120116 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/2] ca3067b2afed8d770626436b77fdd90bd5cb22e7 (jmaloy/jmaloy-qemu-kvm-2) + +JIRA: https://issues.redhat.com/browse/RHEL-120116 +CVE: CVE-2025-11234 + +commit 322c3c4f3abee616a18b3bfe563ec29dd67eae63 +Author: Daniel P. Berrangé +Date: Tue Sep 30 11:58:35 2025 +0100 + + io: move websock resource release to close method + + The QIOChannelWebsock object releases all its resources in the + finalize callback. This is later than desired, as callers expect + to be able to call qio_channel_close() to fully close a channel + and release resources related to I/O. + + The logic in the finalize method is at most a failsafe to handle + cases where a consumer forgets to call qio_channel_close. + + This adds equivalent logic to the close method to release the + resources, using g_clear_handle_id/g_clear_pointer to be robust + against repeated invocations. The finalize method is tweaked + so that the GSource is removed before releasing the underlying + channel. + + Reviewed-by: Eric Blake + Signed-off-by: Daniel P. Berrangé + +Signed-off-by: Jon Maloy +--- + io/channel-websock.c | 11 ++++++++++- + 1 file changed, 10 insertions(+), 1 deletion(-) + +diff --git a/io/channel-websock.c b/io/channel-websock.c +index 08ddb274f0..a19b902ff9 100644 +--- a/io/channel-websock.c ++++ b/io/channel-websock.c +@@ -922,13 +922,13 @@ static void qio_channel_websock_finalize(Object *obj) + buffer_free(&ioc->encinput); + buffer_free(&ioc->encoutput); + buffer_free(&ioc->rawinput); +- object_unref(OBJECT(ioc->master)); + if (ioc->io_tag) { + g_source_remove(ioc->io_tag); + } + if (ioc->io_err) { + error_free(ioc->io_err); + } ++ object_unref(OBJECT(ioc->master)); + } + + +@@ -1219,6 +1219,15 @@ static int qio_channel_websock_close(QIOChannel *ioc, + QIOChannelWebsock *wioc = QIO_CHANNEL_WEBSOCK(ioc); + + trace_qio_channel_websock_close(ioc); ++ buffer_free(&wioc->encinput); ++ buffer_free(&wioc->encoutput); ++ buffer_free(&wioc->rawinput); ++ if (wioc->io_tag) { ++ g_clear_handle_id(&wioc->io_tag, g_source_remove); ++ } ++ if (wioc->io_err) { ++ g_clear_pointer(&wioc->io_err, error_free); ++ } + return qio_channel_close(wioc->master, errp); + } + +-- +2.47.3 + diff --git a/kvm-io-uring-Resubmit-tails-of-short-writes.patch b/kvm-io-uring-Resubmit-tails-of-short-writes.patch new file mode 100644 index 0000000..378e5af --- /dev/null +++ b/kvm-io-uring-Resubmit-tails-of-short-writes.patch @@ -0,0 +1,278 @@ +From 1bdac2a8c4ac77133cb0c2b4d40819bff1a35fc4 Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Tue, 24 Mar 2026 09:43:36 +0100 +Subject: [PATCH 4/4] io-uring: Resubmit tails of short writes + +RH-Author: Hanna Czenczek +RH-MergeRequest: 479: linux-aio/io-uring: Resubmit tails of short requests +RH-Jira: RHEL-158224 +RH-Acked-by: Kevin Wolf +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [4/4] 7ff66622acbb5dbdbacafe71bffd0a277ac919d9 (hreitz/qemu-kvm-c-9-s) + +Short writes can happen, too, not just short reads. The difference to +aio=native is that the kernel will actually retry the tail of short +requests internally already -- so it is harder to reproduce. But if the +tail of a short request returns an error to the kernel, we will see it +in userspace still. To reproduce this, apply the following patch on top +of the one shown in HEAD^ (again %s/escaped // to apply): + +escaped diff --git a/block/export/fuse.c b/block/export/fuse.c +escaped index 67dc50a412..2b98489a32 100644 +escaped --- a/block/export/fuse.c +escaped +++ b/block/export/fuse.c +@@ -1059,8 +1059,15 @@ fuse_co_read(FuseExport *exp, void **bufptr, uint64_t offset, uint32_t size) + int64_t blk_len; + void *buf; + int ret; ++ static uint32_t error_size; + +- size = MIN(size, 4096); ++ if (error_size == size) { ++ error_size = 0; ++ return -EIO; ++ } else if (size > 4096) { ++ error_size = size - 4096; ++ size = 4096; ++ } + + /* Limited by max_read, should not happen */ + if (size > FUSE_MAX_READ_BYTES) { +@@ -1111,8 +1118,15 @@ fuse_co_write(FuseExport *exp, struct fuse_write_out *out, + { + int64_t blk_len; + int ret; ++ static uint32_t error_size; + +- size = MIN(size, 4096); ++ if (error_size == size) { ++ error_size = 0; ++ return -EIO; ++ } else if (size > 4096) { ++ error_size = size - 4096; ++ size = 4096; ++ } + + QEMU_BUILD_BUG_ON(FUSE_MAX_WRITE_BYTES > BDRV_REQUEST_MAX_BYTES); + /* Limited by max_write, should not happen */ + +I know this is a bit artificial because to produce this, there must be +an I/O error somewhere anyway, but if it does happen, qemu will +understand it to mean ENOSPC for short writes, which is incorrect. So I +believe we need to resubmit the tail to maybe have it succeed now, or at +least get the correct error code. + +Reproducer as before: +$ ./qemu-img create -f raw test.raw 8k +Formatting 'test.raw', fmt=raw size=8192 +$ ./qemu-io -f raw -c 'write -P 42 0 8k' test.raw +wrote 8192/8192 bytes at offset 0 +8 KiB, 1 ops; 00.00 sec (64.804 MiB/sec and 8294.9003 ops/sec) +$ hexdump -C test.raw +00000000 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a |****************| +* +00002000 +$ storage-daemon/qemu-storage-daemon \ + --blockdev file,node-name=test,filename=test.raw \ + --export fuse,id=exp,node-name=test,mountpoint=test.raw,writable=true + +$ ./qemu-io --image-opts -c 'read -P 23 0 8k' \ + driver=file,filename=test.raw,cache.direct=on,aio=io_uring +read 8192/8192 bytes at offset 0 +8 KiB, 1 ops; 00.00 sec (58.481 MiB/sec and 7485.5342 ops/sec) +$ ./qemu-io --image-opts -c 'write -P 23 0 8k' \ + driver=file,filename=test.raw,cache.direct=on,aio=io_uring +write failed: No space left on device +$ hexdump -C test.raw +00000000 17 17 17 17 17 17 17 17 17 17 17 17 17 17 17 17 |................| +* +00001000 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a |****************| +* +00002000 + +So short reads already work (because there is code for that), but short +writes incorrectly produce ENOSPC. This patch fixes that by +resubmitting not only the tail of short reads but short writes also. + +(And this patch uses the opportunity to make it so qemu_iovec_destroy() +is called only if req->resubmit_qiov.iov is non-NULL. Functionally a +non-op, but this is how the code generally checks whether the +resubmit_qiov has been set up or not.) + +Reviewed-by: Kevin Wolf +Signed-off-by: Hanna Czenczek +Message-ID: <20260324084338.37453-4-hreitz@redhat.com> +Signed-off-by: Kevin Wolf +(cherry picked from commit cf9cdaea6e24d13dfdf8402f6829d2ca4dca864b) + +Conflicts: +- block/io_uring.c, block/trace-events + Missing 047dabef97bd0c4af3c3dc453b19e20345de3602 ("block/io_uring: use + aio_add_sqe()") downstream, which changed quite a few things about the + io-uring code. Backporting it seems excessive, though, especially + given that it would pull in even more dependencies (general aio and + aio-posix changes that come right before 047dabef). + +Signed-off-by: Hanna Czenczek +--- + block/io_uring.c | 77 +++++++++++++++++++++++++--------------------- + block/trace-events | 2 +- + 2 files changed, 43 insertions(+), 36 deletions(-) + +diff --git a/block/io_uring.c b/block/io_uring.c +index 5dbafc8f7b..582550f8e9 100644 +--- a/block/io_uring.c ++++ b/block/io_uring.c +@@ -31,14 +31,14 @@ typedef struct LuringAIOCB { + struct io_uring_sqe sqeq; + ssize_t ret; + QEMUIOVector *qiov; +- bool is_read; ++ int type; + QSIMPLEQ_ENTRY(LuringAIOCB) next; + + /* +- * Buffered reads may require resubmission, see +- * luring_resubmit_short_read(). ++ * Short reads/writes require resubmission, see ++ * luring_resubmit_short_io(). + */ +- int total_read; ++ int total_done; + QEMUIOVector resubmit_qiov; + } LuringAIOCB; + +@@ -73,22 +73,27 @@ static void luring_resubmit(LuringState *s, LuringAIOCB *luringcb) + } + + /** +- * luring_resubmit_short_read: ++ * luring_resubmit_short_io: + * +- * Short reads are rare but may occur. The remaining read request needs to be +- * resubmitted. ++ * Short reads and writes are rare but may occur. The remaining request needs ++ * to be resubmitted. ++ * ++ * For example, short reads can be reproduced by a FUSE export deliberately ++ * executing short reads. The tail of short writes is generally resubmitted by ++ * io-uring in the kernel, but if that resubmission encounters an I/O error, the ++ * already submitted portion will be returned as a short write. + */ +-static void luring_resubmit_short_read(LuringState *s, LuringAIOCB *luringcb, +- int nread) ++static void luring_resubmit_short_io(LuringState *s, LuringAIOCB *luringcb, ++ int ndone) + { + QEMUIOVector *resubmit_qiov; + size_t remaining; + +- trace_luring_resubmit_short_read(s, luringcb, nread); ++ trace_luring_resubmit_short_io(s, luringcb, ndone); + +- /* Update read position */ +- luringcb->total_read += nread; +- remaining = luringcb->qiov->size - luringcb->total_read; ++ /* Update I/O position */ ++ luringcb->total_done += ndone; ++ remaining = luringcb->qiov->size - luringcb->total_done; + + /* Shorten qiov */ + resubmit_qiov = &luringcb->resubmit_qiov; +@@ -97,11 +102,11 @@ static void luring_resubmit_short_read(LuringState *s, LuringAIOCB *luringcb, + } else { + qemu_iovec_reset(resubmit_qiov); + } +- qemu_iovec_concat(resubmit_qiov, luringcb->qiov, luringcb->total_read, ++ qemu_iovec_concat(resubmit_qiov, luringcb->qiov, luringcb->total_done, + remaining); + + /* Update sqe */ +- luringcb->sqeq.off += nread; ++ luringcb->sqeq.off += ndone; + luringcb->sqeq.addr = (uintptr_t)luringcb->resubmit_qiov.iov; + luringcb->sqeq.len = luringcb->resubmit_qiov.niov; + +@@ -165,8 +170,8 @@ static bool luring_process_completions(LuringState *s) + s->io_q.in_flight--; + trace_luring_process_completion(s, luringcb, ret); + +- /* total_read is non-zero only for resubmitted read requests */ +- total_bytes = ret + luringcb->total_read; ++ /* total_done is non-zero only for resubmitted requests */ ++ total_bytes = ret + luringcb->total_done; + + if (ret < 0) { + /* +@@ -192,27 +197,29 @@ static bool luring_process_completions(LuringState *s) + goto end; + } else if (total_bytes == luringcb->qiov->size) { + ret = 0; +- /* Only read/write */ ++ } else if (ret > 0 && (luringcb->type == QEMU_AIO_READ || ++ luringcb->type == QEMU_AIO_WRITE)) { ++ luring_resubmit_short_io(s, luringcb, ret); ++ resubmit = true; ++ continue; ++ } else if (luringcb->type == QEMU_AIO_READ) { ++ /* Read ret == 0: EOF, pad with zeroes */ ++ qemu_iovec_memset(luringcb->qiov, total_bytes, 0, ++ luringcb->qiov->size - total_bytes); ++ ret = 0; + } else { +- /* Short Read/Write */ +- if (luringcb->is_read) { +- if (ret > 0) { +- luring_resubmit_short_read(s, luringcb, ret); +- resubmit = true; +- continue; +- } else { +- /* Pad with zeroes */ +- qemu_iovec_memset(luringcb->qiov, total_bytes, 0, +- luringcb->qiov->size - total_bytes); +- ret = 0; +- } +- } else { +- ret = -ENOSPC; +- } ++ /* ++ * Normal write ret == 0 means ENOSPC. ++ * For zone-append, we treat any 0 <= ret < qiov->size as ENOSPC, ++ * too, because resubmitting the tail seems a little unsafe. ++ */ ++ ret = -ENOSPC; + } + end: + luringcb->ret = ret; +- qemu_iovec_destroy(&luringcb->resubmit_qiov); ++ if (luringcb->resubmit_qiov.iov) { ++ qemu_iovec_destroy(&luringcb->resubmit_qiov); ++ } + + /* + * If the coroutine is already entered it must be in ioq_submit() +@@ -409,7 +416,7 @@ int coroutine_fn luring_co_submit(BlockDriverState *bs, int fd, uint64_t offset, + .co = qemu_coroutine_self(), + .ret = -EINPROGRESS, + .qiov = qiov, +- .is_read = (type == QEMU_AIO_READ), ++ .type = type, + }; + trace_luring_co_submit(bs, s, &luringcb, fd, offset, qiov ? qiov->size : 0, + type); +diff --git a/block/trace-events b/block/trace-events +index 8e789e1f12..99b8c12bc8 100644 +--- a/block/trace-events ++++ b/block/trace-events +@@ -70,7 +70,7 @@ luring_do_submit_done(void *s, int ret) "LuringState %p submitted to kernel %d" + luring_co_submit(void *bs, void *s, void *luringcb, int fd, uint64_t offset, size_t nbytes, int type) "bs %p s %p luringcb %p fd %d offset %" PRId64 " nbytes %zd type %d" + luring_process_completion(void *s, void *aiocb, int ret) "LuringState %p luringcb %p ret %d" + luring_io_uring_submit(void *s, int ret) "LuringState %p ret %d" +-luring_resubmit_short_read(void *s, void *luringcb, int nread) "LuringState %p luringcb %p nread %d" ++luring_resubmit_short_io(void *s, void *luringcb, int ndone) "LuringState %p luringcb %p ndone %d" + + # qcow2.c + qcow2_add_task(void *co, void *bs, void *pool, const char *action, int cluster_type, uint64_t host_offset, uint64_t offset, uint64_t bytes, void *qiov, size_t qiov_offset) "co %p bs %p pool %p: %s: cluster_type %d file_cluster_offset %" PRIu64 " offset %" PRIu64 " bytes %" PRIu64 " qiov %p qiov_offset %zu" +-- +2.47.3 + diff --git a/kvm-iotests-244-Don-t-store-data-file-with-protocol-in-i.patch b/kvm-iotests-244-Don-t-store-data-file-with-protocol-in-i.patch deleted file mode 100644 index efb9f25..0000000 --- a/kvm-iotests-244-Don-t-store-data-file-with-protocol-in-i.patch +++ /dev/null @@ -1,61 +0,0 @@ -From 80e197ac72a4b0c810f69833e1f9e552a415e82a Mon Sep 17 00:00:00 2001 -From: Kevin Wolf -Date: Thu, 25 Apr 2024 14:49:40 +0200 -Subject: [PATCH 2/4] iotests/244: Don't store data-file with protocol in image - -RH-Author: Hana Czenczek -RH-MergeRequest: 1: CVE 2024-4467 (PRDSC) -RH-Jira: RHEL-46239 -RH-CVE: CVE-2024-4467 -RH-Acked-by: Kevin Wolf -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Eric Blake -RH-Commit: [2/4] 92e00dab8be1570b13172353d77d2af44cb4e22b - -We want to disable filename parsing for data files because it's too easy -to abuse in malicious image files. Make the test ready for the change by -passing the data file explicitly in command line options. - -Signed-off-by: Kevin Wolf -Reviewed-by: Eric Blake -Reviewed-by: Stefan Hajnoczi -Reviewed-by: Hanna Czenczek -Upstream: N/A, embargoed -Signed-off-by: Hanna Czenczek ---- - tests/qemu-iotests/244 | 19 ++++++++++++++++--- - 1 file changed, 16 insertions(+), 3 deletions(-) - -diff --git a/tests/qemu-iotests/244 b/tests/qemu-iotests/244 -index 3e61fa25bb..bb9cc6512f 100755 ---- a/tests/qemu-iotests/244 -+++ b/tests/qemu-iotests/244 -@@ -215,9 +215,22 @@ $QEMU_IMG convert -f $IMGFMT -O $IMGFMT -n -C "$TEST_IMG.src" "$TEST_IMG" - $QEMU_IMG compare -f $IMGFMT -F $IMGFMT "$TEST_IMG.src" "$TEST_IMG" - - # blkdebug doesn't support copy offloading, so this tests the error path --$QEMU_IMG amend -f $IMGFMT -o "data_file=blkdebug::$TEST_IMG.data" "$TEST_IMG" --$QEMU_IMG convert -f $IMGFMT -O $IMGFMT -n -C "$TEST_IMG.src" "$TEST_IMG" --$QEMU_IMG compare -f $IMGFMT -F $IMGFMT "$TEST_IMG.src" "$TEST_IMG" -+test_img_with_blkdebug="json:{ -+ 'driver': 'qcow2', -+ 'file': { -+ 'driver': 'file', -+ 'filename': '$TEST_IMG' -+ }, -+ 'data-file': { -+ 'driver': 'blkdebug', -+ 'image': { -+ 'driver': 'file', -+ 'filename': '$TEST_IMG.data' -+ } -+ } -+}" -+$QEMU_IMG convert -f $IMGFMT -O $IMGFMT -n -C "$TEST_IMG.src" "$test_img_with_blkdebug" -+$QEMU_IMG compare -f $IMGFMT -F $IMGFMT "$TEST_IMG.src" "$test_img_with_blkdebug" - - echo - echo "=== Flushing should flush the data file ===" --- -2.39.3 - diff --git a/kvm-iotests-270-Don-t-store-data-file-with-json-prefix-i.patch b/kvm-iotests-270-Don-t-store-data-file-with-json-prefix-i.patch deleted file mode 100644 index 4f31988..0000000 --- a/kvm-iotests-270-Don-t-store-data-file-with-json-prefix-i.patch +++ /dev/null @@ -1,64 +0,0 @@ -From bf01c03b0120f5ed8e54c2a30b7830901b22b893 Mon Sep 17 00:00:00 2001 -From: Kevin Wolf -Date: Thu, 25 Apr 2024 14:49:40 +0200 -Subject: [PATCH 3/4] iotests/270: Don't store data-file with json: prefix in - image - -RH-Author: Hana Czenczek -RH-MergeRequest: 1: CVE 2024-4467 (PRDSC) -RH-Jira: RHEL-46239 -RH-CVE: CVE-2024-4467 -RH-Acked-by: Kevin Wolf -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Eric Blake -RH-Commit: [3/4] 705bcc2819ce8e0f8b9d660a93bc48de26413aec - -We want to disable filename parsing for data files because it's too easy -to abuse in malicious image files. Make the test ready for the change by -passing the data file explicitly in command line options. - -Signed-off-by: Kevin Wolf -Reviewed-by: Eric Blake -Reviewed-by: Stefan Hajnoczi -Reviewed-by: Hanna Czenczek -Upstream: N/A, embargoed -Signed-off-by: Hanna Czenczek ---- - tests/qemu-iotests/270 | 14 +++++++++++--- - 1 file changed, 11 insertions(+), 3 deletions(-) - -diff --git a/tests/qemu-iotests/270 b/tests/qemu-iotests/270 -index 74352342db..c37b674aa2 100755 ---- a/tests/qemu-iotests/270 -+++ b/tests/qemu-iotests/270 -@@ -60,8 +60,16 @@ _make_test_img -o cluster_size=2M,data_file="$TEST_IMG.orig" \ - # "write" 2G of data without using any space. - # (qemu-img create does not like it, though, because null-co does not - # support image creation.) --$QEMU_IMG amend -o data_file="json:{'driver':'null-co',,'size':'4294967296'}" \ -- "$TEST_IMG" -+test_img_with_null_data="json:{ -+ 'driver': '$IMGFMT', -+ 'file': { -+ 'filename': '$TEST_IMG' -+ }, -+ 'data-file': { -+ 'driver': 'null-co', -+ 'size':'4294967296' -+ } -+}" - - # This gives us a range of: - # 2^31 - 512 + 768 - 1 = 2^31 + 255 > 2^31 -@@ -74,7 +82,7 @@ $QEMU_IMG amend -o data_file="json:{'driver':'null-co',,'size':'4294967296'}" \ - # on L2 boundaries, we need large L2 tables; hence the cluster size of - # 2 MB. (Anything from 256 kB should work, though, because then one L2 - # table covers 8 GB.) --$QEMU_IO -c "write 768 $((2 ** 31 - 512))" "$TEST_IMG" | _filter_qemu_io -+$QEMU_IO -c "write 768 $((2 ** 31 - 512))" "$test_img_with_null_data" | _filter_qemu_io - - _check_test_img - --- -2.39.3 - diff --git a/kvm-iotests-test-NBD-TLS-iothread.patch b/kvm-iotests-test-NBD-TLS-iothread.patch deleted file mode 100644 index c34ed04..0000000 --- a/kvm-iotests-test-NBD-TLS-iothread.patch +++ /dev/null @@ -1,276 +0,0 @@ -From 2f12be8abfc90dc383a221441f60bdaae6b617d2 Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Fri, 17 May 2024 21:50:15 -0500 -Subject: [PATCH 4/4] iotests: test NBD+TLS+iothread -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Eric Blake -RH-MergeRequest: 257: nbd/server: fix TLS negotiation across coroutine context -RH-Jira: RHEL-40959 -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Miroslav Rezanina -RH-Commit: [4/4] 39a37bf3ae6e7046577de151ef2f6fd1fd694e62 (ebblake/centos-qemu-kvm) - -Prevent regressions when using NBD with TLS in the presence of -iothreads, adding coverage the fix to qio channels made in the -previous patch. - -The shell function pick_unused_port() was copied from -nbdkit.git/tests/functions.sh.in, where it had all authors from Red -Hat, agreeing to the resulting relicensing from 2-clause BSD to GPLv2. - -CC: qemu-stable@nongnu.org -CC: "Richard W.M. Jones" -Signed-off-by: Eric Blake -Message-ID: <20240531180639.1392905-6-eblake@redhat.com> -Reviewed-by: Daniel P. Berrangé - -(cherry picked from commit a73c99378022ebb785481e84cfe1e81097546268) -Jira: https://issues.redhat.com/browse/RHEL-40959 -Signed-off-by: Eric Blake ---- - tests/qemu-iotests/tests/nbd-tls-iothread | 168 ++++++++++++++++++ - tests/qemu-iotests/tests/nbd-tls-iothread.out | 54 ++++++ - 2 files changed, 222 insertions(+) - create mode 100755 tests/qemu-iotests/tests/nbd-tls-iothread - create mode 100644 tests/qemu-iotests/tests/nbd-tls-iothread.out - -diff --git a/tests/qemu-iotests/tests/nbd-tls-iothread b/tests/qemu-iotests/tests/nbd-tls-iothread -new file mode 100755 -index 0000000000..a2fb07206e ---- /dev/null -+++ b/tests/qemu-iotests/tests/nbd-tls-iothread -@@ -0,0 +1,168 @@ -+#!/usr/bin/env bash -+# group: rw quick -+# -+# Test of NBD+TLS+iothread -+# -+# Copyright (C) 2024 Red Hat, Inc. -+# -+# This program is free software; you can redistribute it and/or modify -+# it under the terms of the GNU General Public License as published by -+# the Free Software Foundation; either version 2 of the License, or -+# (at your option) any later version. -+# -+# This program is distributed in the hope that it will be useful, -+# but WITHOUT ANY WARRANTY; without even the implied warranty of -+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -+# GNU General Public License for more details. -+# -+# You should have received a copy of the GNU General Public License -+# along with this program. If not, see . -+# -+ -+# creator -+owner=eblake@redhat.com -+ -+seq=`basename $0` -+echo "QA output created by $seq" -+ -+status=1 # failure is the default! -+ -+_cleanup() -+{ -+ _cleanup_qemu -+ _cleanup_test_img -+ rm -f "$dst_image" -+ tls_x509_cleanup -+} -+trap "_cleanup; exit \$status" 0 1 2 3 15 -+ -+# get standard environment, filters and checks -+cd .. -+. ./common.rc -+. ./common.filter -+. ./common.qemu -+. ./common.tls -+. ./common.nbd -+ -+_supported_fmt qcow2 # Hardcoded to qcow2 command line and QMP below -+_supported_proto file -+ -+# pick_unused_port -+# -+# Picks and returns an "unused" port, setting the global variable -+# $port. -+# -+# This is inherently racy, but we need it because qemu does not currently -+# permit NBD+TLS over a Unix domain socket -+pick_unused_port () -+{ -+ if ! (ss --version) >/dev/null 2>&1; then -+ _notrun "ss utility required, skipped this test" -+ fi -+ -+ # Start at a random port to make it less likely that two parallel -+ # tests will conflict. -+ port=$(( 50000 + (RANDOM%15000) )) -+ while ss -ltn | grep -sqE ":$port\b"; do -+ ((port++)) -+ if [ $port -eq 65000 ]; then port=50000; fi -+ done -+ echo picked unused port -+} -+ -+tls_x509_init -+ -+size=1G -+DST_IMG="$TEST_DIR/dst.qcow2" -+ -+echo -+echo "== preparing TLS creds and spare port ==" -+ -+pick_unused_port -+tls_x509_create_root_ca "ca1" -+tls_x509_create_server "ca1" "server1" -+tls_x509_create_client "ca1" "client1" -+tls_obj_base=tls-creds-x509,id=tls0,verify-peer=true,dir="${tls_dir}" -+ -+echo -+echo "== preparing image ==" -+ -+_make_test_img $size -+$QEMU_IMG create -f qcow2 "$DST_IMG" $size | _filter_img_create -+ -+echo -+echo === Starting Src QEMU === -+echo -+ -+_launch_qemu -machine q35 \ -+ -object iothread,id=iothread0 \ -+ -object "${tls_obj_base}"/client1,endpoint=client \ -+ -device '{"driver":"pcie-root-port", "id":"root0", "multifunction":true, -+ "bus":"pcie.0"}' \ -+ -device '{"driver":"virtio-scsi-pci", "id":"virtio_scsi_pci0", -+ "bus":"root0", "iothread":"iothread0"}' \ -+ -device '{"driver":"scsi-hd", "id":"image1", "drive":"drive_image1", -+ "bus":"virtio_scsi_pci0.0"}' \ -+ -blockdev '{"driver":"file", "cache":{"direct":true, "no-flush":false}, -+ "filename":"'"$TEST_IMG"'", "node-name":"drive_sys1"}' \ -+ -blockdev '{"driver":"qcow2", "node-name":"drive_image1", -+ "file":"drive_sys1"}' -+h1=$QEMU_HANDLE -+_send_qemu_cmd $h1 '{"execute": "qmp_capabilities"}' 'return' -+ -+echo -+echo === Starting Dst VM2 === -+echo -+ -+_launch_qemu -machine q35 \ -+ -object iothread,id=iothread0 \ -+ -object "${tls_obj_base}"/server1,endpoint=server \ -+ -device '{"driver":"pcie-root-port", "id":"root0", "multifunction":true, -+ "bus":"pcie.0"}' \ -+ -device '{"driver":"virtio-scsi-pci", "id":"virtio_scsi_pci0", -+ "bus":"root0", "iothread":"iothread0"}' \ -+ -device '{"driver":"scsi-hd", "id":"image1", "drive":"drive_image1", -+ "bus":"virtio_scsi_pci0.0"}' \ -+ -blockdev '{"driver":"file", "cache":{"direct":true, "no-flush":false}, -+ "filename":"'"$DST_IMG"'", "node-name":"drive_sys1"}' \ -+ -blockdev '{"driver":"qcow2", "node-name":"drive_image1", -+ "file":"drive_sys1"}' \ -+ -incoming defer -+h2=$QEMU_HANDLE -+_send_qemu_cmd $h2 '{"execute": "qmp_capabilities"}' 'return' -+ -+echo -+echo === Dst VM: Enable NBD server for incoming storage migration === -+echo -+ -+_send_qemu_cmd $h2 '{"execute": "nbd-server-start", "arguments": -+ {"addr": {"type": "inet", "data": {"host": "127.0.0.1", "port": "'$port'"}}, -+ "tls-creds": "tls0"}}' '{"return": {}}' | sed "s/\"$port\"/PORT/g" -+_send_qemu_cmd $h2 '{"execute": "block-export-add", "arguments": -+ {"node-name": "drive_image1", "type": "nbd", "writable": true, -+ "id": "drive_image1"}}' '{"return": {}}' -+ -+echo -+echo === Src VM: Mirror to dst NBD for outgoing storage migration === -+echo -+ -+_send_qemu_cmd $h1 '{"execute": "blockdev-add", "arguments": -+ {"node-name": "mirror", "driver": "nbd", -+ "server": {"type": "inet", "host": "127.0.0.1", "port": "'$port'"}, -+ "export": "drive_image1", "tls-creds": "tls0", -+ "tls-hostname": "127.0.0.1"}}' '{"return": {}}' | sed "s/\"$port\"/PORT/g" -+_send_qemu_cmd $h1 '{"execute": "blockdev-mirror", "arguments": -+ {"sync": "full", "device": "drive_image1", "target": "mirror", -+ "job-id": "drive_image1_53"}}' '{"return": {}}' -+_timed_wait_for $h1 '"ready"' -+ -+echo -+echo === Cleaning up === -+echo -+ -+_send_qemu_cmd $h1 '{"execute":"quit"}' '' -+_send_qemu_cmd $h2 '{"execute":"quit"}' '' -+ -+echo "*** done" -+rm -f $seq.full -+status=0 -diff --git a/tests/qemu-iotests/tests/nbd-tls-iothread.out b/tests/qemu-iotests/tests/nbd-tls-iothread.out -new file mode 100644 -index 0000000000..1d83d4f903 ---- /dev/null -+++ b/tests/qemu-iotests/tests/nbd-tls-iothread.out -@@ -0,0 +1,54 @@ -+QA output created by nbd-tls-iothread -+ -+== preparing TLS creds and spare port == -+picked unused port -+Generating a self signed certificate... -+Generating a signed certificate... -+Generating a signed certificate... -+ -+== preparing image == -+Formatting 'TEST_DIR/t.IMGFMT', fmt=IMGFMT size=1073741824 -+Formatting 'TEST_DIR/dst.IMGFMT', fmt=IMGFMT size=1073741824 -+ -+=== Starting Src QEMU === -+ -+{"execute": "qmp_capabilities"} -+{"return": {}} -+ -+=== Starting Dst VM2 === -+ -+{"execute": "qmp_capabilities"} -+{"return": {}} -+ -+=== Dst VM: Enable NBD server for incoming storage migration === -+ -+{"execute": "nbd-server-start", "arguments": -+ {"addr": {"type": "inet", "data": {"host": "127.0.0.1", "port": PORT}}, -+ "tls-creds": "tls0"}} -+{"return": {}} -+{"execute": "block-export-add", "arguments": -+ {"node-name": "drive_image1", "type": "nbd", "writable": true, -+ "id": "drive_image1"}} -+{"return": {}} -+ -+=== Src VM: Mirror to dst NBD for outgoing storage migration === -+ -+{"execute": "blockdev-add", "arguments": -+ {"node-name": "mirror", "driver": "nbd", -+ "server": {"type": "inet", "host": "127.0.0.1", "port": PORT}, -+ "export": "drive_image1", "tls-creds": "tls0", -+ "tls-hostname": "127.0.0.1"}} -+{"return": {}} -+{"execute": "blockdev-mirror", "arguments": -+ {"sync": "full", "device": "drive_image1", "target": "mirror", -+ "job-id": "drive_image1_53"}} -+{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "drive_image1_53"}} -+{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "drive_image1_53"}} -+{"return": {}} -+{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "drive_image1_53"}} -+ -+=== Cleaning up === -+ -+{"execute":"quit"} -+{"execute":"quit"} -+*** done --- -2.39.3 - diff --git a/kvm-linux-aio-Put-all-parameters-into-qemu_laiocb.patch b/kvm-linux-aio-Put-all-parameters-into-qemu_laiocb.patch new file mode 100644 index 0000000..b458e78 --- /dev/null +++ b/kvm-linux-aio-Put-all-parameters-into-qemu_laiocb.patch @@ -0,0 +1,127 @@ +From 30c3962bbef6ce083f326988f35741dd21aaf09d Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Tue, 24 Mar 2026 09:43:34 +0100 +Subject: [PATCH 1/4] linux-aio: Put all parameters into qemu_laiocb + +RH-Author: Hanna Czenczek +RH-MergeRequest: 479: linux-aio/io-uring: Resubmit tails of short requests +RH-Jira: RHEL-158224 +RH-Acked-by: Kevin Wolf +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [1/4] f449ea0c49a093bbec59b24fb44308e7f58c9ed2 (hreitz/qemu-kvm-c-9-s) + +Put all request parameters into the qemu_laiocb struct, which will allow +re-submitting the tail of short reads/writes. + +Reviewed-by: Kevin Wolf +Signed-off-by: Hanna Czenczek +Message-ID: <20260324084338.37453-2-hreitz@redhat.com> +Signed-off-by: Kevin Wolf +(cherry picked from commit cc03b62df47a09c507e199cc043f57bdc941cc67) +Signed-off-by: Hanna Czenczek +--- + block/linux-aio.c | 34 ++++++++++++++++++++++------------ + 1 file changed, 22 insertions(+), 12 deletions(-) + +diff --git a/block/linux-aio.c b/block/linux-aio.c +index c200e7ad20..c2c5e11946 100644 +--- a/block/linux-aio.c ++++ b/block/linux-aio.c +@@ -41,9 +41,15 @@ struct qemu_laiocb { + LinuxAioState *ctx; + struct iocb iocb; + ssize_t ret; ++ off_t offset; + size_t nbytes; + QEMUIOVector *qiov; +- bool is_read; ++ ++ int fd; ++ int type; ++ BdrvRequestFlags flags; ++ ++ uint64_t dev_max_batch; + QSIMPLEQ_ENTRY(qemu_laiocb) next; + }; + +@@ -87,7 +93,7 @@ static void qemu_laio_process_completion(struct qemu_laiocb *laiocb) + ret = 0; + } else if (ret >= 0) { + /* Short reads mean EOF, pad with zeros. */ +- if (laiocb->is_read) { ++ if (laiocb->type == QEMU_AIO_READ) { + qemu_iovec_memset(laiocb->qiov, ret, 0, + laiocb->qiov->size - ret); + } else { +@@ -367,23 +373,23 @@ static void laio_deferred_fn(void *opaque) + } + } + +-static int laio_do_submit(int fd, struct qemu_laiocb *laiocb, off_t offset, +- int type, BdrvRequestFlags flags, +- uint64_t dev_max_batch) ++static int laio_do_submit(struct qemu_laiocb *laiocb) + { + LinuxAioState *s = laiocb->ctx; + struct iocb *iocbs = &laiocb->iocb; + QEMUIOVector *qiov = laiocb->qiov; ++ int fd = laiocb->fd; ++ off_t offset = laiocb->offset; + +- switch (type) { ++ switch (laiocb->type) { + case QEMU_AIO_WRITE: + #ifdef HAVE_IO_PREP_PWRITEV2 + { +- int laio_flags = (flags & BDRV_REQ_FUA) ? RWF_DSYNC : 0; ++ int laio_flags = (laiocb->flags & BDRV_REQ_FUA) ? RWF_DSYNC : 0; + io_prep_pwritev2(iocbs, fd, qiov->iov, qiov->niov, offset, laio_flags); + } + #else +- assert(flags == 0); ++ assert(laiocb->flags == 0); + io_prep_pwritev(iocbs, fd, qiov->iov, qiov->niov, offset); + #endif + break; +@@ -399,7 +405,7 @@ static int laio_do_submit(int fd, struct qemu_laiocb *laiocb, off_t offset, + /* Currently Linux kernel does not support other operations */ + default: + fprintf(stderr, "%s: invalid AIO request type 0x%x.\n", +- __func__, type); ++ __func__, laiocb->type); + return -EIO; + } + io_set_eventfd(&laiocb->iocb, event_notifier_get_fd(&s->e)); +@@ -407,7 +413,7 @@ static int laio_do_submit(int fd, struct qemu_laiocb *laiocb, off_t offset, + QSIMPLEQ_INSERT_TAIL(&s->io_q.pending, laiocb, next); + s->io_q.in_queue++; + if (!s->io_q.blocked) { +- if (s->io_q.in_queue >= laio_max_batch(s, dev_max_batch)) { ++ if (s->io_q.in_queue >= laio_max_batch(s, laiocb->dev_max_batch)) { + ioq_submit(s); + } else { + defer_call(laio_deferred_fn, s); +@@ -425,14 +431,18 @@ int coroutine_fn laio_co_submit(int fd, uint64_t offset, QEMUIOVector *qiov, + AioContext *ctx = qemu_get_current_aio_context(); + struct qemu_laiocb laiocb = { + .co = qemu_coroutine_self(), ++ .offset = offset, + .nbytes = qiov ? qiov->size : 0, + .ctx = aio_get_linux_aio(ctx), + .ret = -EINPROGRESS, +- .is_read = (type == QEMU_AIO_READ), + .qiov = qiov, ++ .fd = fd, ++ .type = type, ++ .flags = flags, ++ .dev_max_batch = dev_max_batch, + }; + +- ret = laio_do_submit(fd, &laiocb, offset, type, flags, dev_max_batch); ++ ret = laio_do_submit(&laiocb); + if (ret < 0) { + return ret; + } +-- +2.47.3 + diff --git a/kvm-linux-aio-Resubmit-tails-of-short-reads-writes.patch b/kvm-linux-aio-Resubmit-tails-of-short-reads-writes.patch new file mode 100644 index 0000000..2da3b76 --- /dev/null +++ b/kvm-linux-aio-Resubmit-tails-of-short-reads-writes.patch @@ -0,0 +1,215 @@ +From ad3a6c9b3487226d9622120eaea8218bd6050a04 Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Tue, 24 Mar 2026 09:43:35 +0100 +Subject: [PATCH 2/4] linux-aio: Resubmit tails of short reads/writes + +RH-Author: Hanna Czenczek +RH-MergeRequest: 479: linux-aio/io-uring: Resubmit tails of short requests +RH-Jira: RHEL-158224 +RH-Acked-by: Kevin Wolf +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [2/4] f97271e609a150acc04a15ffc85c40e7bcb00060 (hreitz/qemu-kvm-c-9-s) + +Short reads/writes can happen. One way to reproduce them is via our +FUSE export, with the following diff applied (%s/escaped // to apply -- +if you put plain diffs in commit messages, git-am will apply them, and I +would rather avoid breaking FUSE accidentally via this patch): + +escaped diff --git a/block/export/fuse.c b/block/export/fuse.c +escaped index a2a478d293..67dc50a412 100644 +escaped --- a/block/export/fuse.c +escaped +++ b/block/export/fuse.c +@@ -828,7 +828,7 @@ static ssize_t coroutine_fn GRAPH_RDLOCK + fuse_co_init(FuseExport *exp, struct fuse_init_out *out, + const struct fuse_init_in_compat *in) + { +- const uint32_t supported_flags = FUSE_ASYNC_READ | FUSE_ASYNC_DIO; ++ const uint32_t supported_flags = FUSE_ASYNC_READ; + + if (in->major != 7) { + error_report("FUSE major version mismatch: We have 7, but kernel has %" +@@ -1060,6 +1060,8 @@ fuse_co_read(FuseExport *exp, void **bufptr, uint64_t offset, uint32_t size) + void *buf; + int ret; + ++ size = MIN(size, 4096); ++ + /* Limited by max_read, should not happen */ + if (size > FUSE_MAX_READ_BYTES) { + return -EINVAL; +@@ -1110,6 +1112,8 @@ fuse_co_write(FuseExport *exp, struct fuse_write_out *out, + int64_t blk_len; + int ret; + ++ size = MIN(size, 4096); ++ + QEMU_BUILD_BUG_ON(FUSE_MAX_WRITE_BYTES > BDRV_REQUEST_MAX_BYTES); + /* Limited by max_write, should not happen */ + if (size > FUSE_MAX_WRITE_BYTES) { + +Then: +$ ./qemu-img create -f raw test.raw 8k +Formatting 'test.raw', fmt=raw size=8192 +$ ./qemu-io -f raw -c 'write -P 42 0 8k' test.raw +wrote 8192/8192 bytes at offset 0 +8 KiB, 1 ops; 00.00 sec (64.804 MiB/sec and 8294.9003 ops/sec) +$ hexdump -C test.raw +00000000 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a |****************| +* +00002000 + +With aio=threads, short I/O works: +$ storage-daemon/qemu-storage-daemon \ + --blockdev file,node-name=test,filename=test.raw \ + --export fuse,id=exp,node-name=test,mountpoint=test.raw,writable=true + +Other shell: +$ ./qemu-io --image-opts -c 'read -P 42 0 8k' \ + driver=file,filename=test.raw,cache.direct=on,aio=threads +read 8192/8192 bytes at offset 0 +8 KiB, 1 ops; 00.00 sec (36.563 MiB/sec and 4680.0923 ops/sec) +$ ./qemu-io --image-opts -c 'write -P 23 0 8k' \ + driver=file,filename=test.raw,cache.direct=on,aio=threads +wrote 8192/8192 bytes at offset 0 +8 KiB, 1 ops; 00.00 sec (35.995 MiB/sec and 4607.2970 ops/sec) +$ hexdump -C test.raw +00000000 17 17 17 17 17 17 17 17 17 17 17 17 17 17 17 17 |................| +* +00002000 + +But with aio=native, it does not: +$ ./qemu-io --image-opts -c 'read -P 23 0 8k' \ + driver=file,filename=test.raw,cache.direct=on,aio=native +Pattern verification failed at offset 0, 8192 bytes +read 8192/8192 bytes at offset 0 +8 KiB, 1 ops; 00.00 sec (86.155 MiB/sec and 11027.7900 ops/sec) +$ ./qemu-io --image-opts -c 'write -P 42 0 8k' \ + driver=file,filename=test.raw,cache.direct=on,aio=native +write failed: No space left on device +$ hexdump -C test.raw +00000000 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a 2a |****************| +* +00001000 17 17 17 17 17 17 17 17 17 17 17 17 17 17 17 17 |................| +* +00002000 + +This patch fixes that. + +Reviewed-by: Kevin Wolf +Signed-off-by: Hanna Czenczek +Message-ID: <20260324084338.37453-3-hreitz@redhat.com> +Signed-off-by: Kevin Wolf +(cherry picked from commit 7eca3d4883be8d328377001a9ea7ae9882b00f3c) +Signed-off-by: Hanna Czenczek +--- + block/linux-aio.c | 56 ++++++++++++++++++++++++++++++++++++++++++----- + 1 file changed, 50 insertions(+), 6 deletions(-) + +diff --git a/block/linux-aio.c b/block/linux-aio.c +index c2c5e11946..84397de54c 100644 +--- a/block/linux-aio.c ++++ b/block/linux-aio.c +@@ -45,6 +45,10 @@ struct qemu_laiocb { + size_t nbytes; + QEMUIOVector *qiov; + ++ /* For handling short reads/writes */ ++ size_t total_done; ++ QEMUIOVector resubmit_qiov; ++ + int fd; + int type; + BdrvRequestFlags flags; +@@ -74,28 +78,61 @@ struct LinuxAioState { + }; + + static void ioq_submit(LinuxAioState *s); ++static int laio_do_submit(struct qemu_laiocb *laiocb); + + static inline ssize_t io_event_ret(struct io_event *ev) + { + return (ssize_t)(((uint64_t)ev->res2 << 32) | ev->res); + } + ++/** ++ * Retry tail of short requests. ++ */ ++static int laio_resubmit_short_io(struct qemu_laiocb *laiocb, size_t done) ++{ ++ QEMUIOVector *resubmit_qiov = &laiocb->resubmit_qiov; ++ ++ laiocb->total_done += done; ++ ++ if (!resubmit_qiov->iov) { ++ qemu_iovec_init(resubmit_qiov, laiocb->qiov->niov); ++ } else { ++ qemu_iovec_reset(resubmit_qiov); ++ } ++ qemu_iovec_concat(resubmit_qiov, laiocb->qiov, ++ laiocb->total_done, laiocb->nbytes - laiocb->total_done); ++ ++ return laio_do_submit(laiocb); ++} ++ + /* + * Completes an AIO request. + */ + static void qemu_laio_process_completion(struct qemu_laiocb *laiocb) + { +- int ret; ++ ssize_t ret; + + ret = laiocb->ret; + if (ret != -ECANCELED) { +- if (ret == laiocb->nbytes) { ++ if (ret == laiocb->nbytes - laiocb->total_done) { + ret = 0; ++ } else if (ret > 0 && (laiocb->type == QEMU_AIO_READ || ++ laiocb->type == QEMU_AIO_WRITE)) { ++ ret = laio_resubmit_short_io(laiocb, ret); ++ if (!ret) { ++ return; ++ } + } else if (ret >= 0) { +- /* Short reads mean EOF, pad with zeros. */ ++ /* ++ * For normal reads and writes, we only get here if ret == 0, which ++ * means EOF for reads and ENOSPC for writes. ++ * For zone-append, we get here with any ret >= 0, which we just ++ * treat as ENOSPC, too (safer than resubmitting, probably, but not ++ * 100 % clear). ++ */ + if (laiocb->type == QEMU_AIO_READ) { +- qemu_iovec_memset(laiocb->qiov, ret, 0, +- laiocb->qiov->size - ret); ++ qemu_iovec_memset(laiocb->qiov, laiocb->total_done, 0, ++ laiocb->qiov->size - laiocb->total_done); + } else { + ret = -ENOSPC; + } +@@ -103,6 +140,9 @@ static void qemu_laio_process_completion(struct qemu_laiocb *laiocb) + } + + laiocb->ret = ret; ++ if (laiocb->resubmit_qiov.iov) { ++ qemu_iovec_destroy(&laiocb->resubmit_qiov); ++ } + + /* + * If the coroutine is already entered it must be in ioq_submit() and +@@ -379,7 +419,11 @@ static int laio_do_submit(struct qemu_laiocb *laiocb) + struct iocb *iocbs = &laiocb->iocb; + QEMUIOVector *qiov = laiocb->qiov; + int fd = laiocb->fd; +- off_t offset = laiocb->offset; ++ off_t offset = laiocb->offset + laiocb->total_done; ++ ++ if (laiocb->resubmit_qiov.iov) { ++ qiov = &laiocb->resubmit_qiov; ++ } + + switch (laiocb->type) { + case QEMU_AIO_WRITE: +-- +2.47.3 + diff --git a/kvm-linux-aio-add-IO_CMD_FDSYNC-command-support.patch b/kvm-linux-aio-add-IO_CMD_FDSYNC-command-support.patch deleted file mode 100644 index 391ab43..0000000 --- a/kvm-linux-aio-add-IO_CMD_FDSYNC-command-support.patch +++ /dev/null @@ -1,126 +0,0 @@ -From 11faa773637f76f573f5320c063f7e55263c3a84 Mon Sep 17 00:00:00 2001 -From: Prasad Pandit -Date: Thu, 25 Apr 2024 12:34:12 +0530 -Subject: [PATCH 1/5] linux-aio: add IO_CMD_FDSYNC command support - -RH-Author: Prasad Pandit -RH-MergeRequest: 260: linux-aio: add IO_CMD_FDSYNC command support -RH-Jira: RHEL-51901 -RH-Acked-by: Miroslav Rezanina -RH-Commit: [1/1] 2830edc801f9fbbc373631cf5b12a396f4b2bced (pjp/cs-qemu-kvm) - -Libaio defines IO_CMD_FDSYNC command to sync all outstanding -asynchronous I/O operations, by flushing out file data to the -disk storage. Enable linux-aio to submit such aio request. - -When using aio=native without fdsync() support, QEMU creates -pthreads, and destroying these pthreads results in TLB flushes. -In a real-time guest environment, TLB flushes cause a latency -spike. This patch helps to avoid such spikes. - -Jira: https://issues.redhat.com/browse/RHEL-51901 -Reviewed-by: Stefan Hajnoczi -Signed-off-by: Prasad Pandit -Message-ID: <20240425070412.37248-1-ppandit@redhat.com> -Reviewed-by: Kevin Wolf -Signed-off-by: Kevin Wolf -(cherry picked from commit 24687abf237e3c15816d689a8e4b08d7c3190dcb) -Signed-off-by: Prasad Pandit ---- - block/file-posix.c | 9 +++++++++ - block/linux-aio.c | 21 ++++++++++++++++++++- - include/block/raw-aio.h | 1 + - 3 files changed, 30 insertions(+), 1 deletion(-) - -diff --git a/block/file-posix.c b/block/file-posix.c -index 35684f7e21..9831b08fb6 100644 ---- a/block/file-posix.c -+++ b/block/file-posix.c -@@ -159,6 +159,7 @@ typedef struct BDRVRawState { - bool has_discard:1; - bool has_write_zeroes:1; - bool use_linux_aio:1; -+ bool has_laio_fdsync:1; - bool use_linux_io_uring:1; - int page_cache_inconsistent; /* errno from fdatasync failure */ - bool has_fallocate; -@@ -718,6 +719,9 @@ static int raw_open_common(BlockDriverState *bs, QDict *options, - ret = -EINVAL; - goto fail; - } -+ if (s->use_linux_aio) { -+ s->has_laio_fdsync = laio_has_fdsync(s->fd); -+ } - #else - if (s->use_linux_aio) { - error_setg(errp, "aio=native was specified, but is not supported " -@@ -2599,6 +2603,11 @@ static int coroutine_fn raw_co_flush_to_disk(BlockDriverState *bs) - if (raw_check_linux_io_uring(s)) { - return luring_co_submit(bs, s->fd, 0, NULL, QEMU_AIO_FLUSH); - } -+#endif -+#ifdef CONFIG_LINUX_AIO -+ if (s->has_laio_fdsync && raw_check_linux_aio(s)) { -+ return laio_co_submit(s->fd, 0, NULL, QEMU_AIO_FLUSH, 0); -+ } - #endif - return raw_thread_pool_submit(handle_aiocb_flush, &acb); - } -diff --git a/block/linux-aio.c b/block/linux-aio.c -index ec05d946f3..e3b5ec9aba 100644 ---- a/block/linux-aio.c -+++ b/block/linux-aio.c -@@ -384,6 +384,9 @@ static int laio_do_submit(int fd, struct qemu_laiocb *laiocb, off_t offset, - case QEMU_AIO_READ: - io_prep_preadv(iocbs, fd, qiov->iov, qiov->niov, offset); - break; -+ case QEMU_AIO_FLUSH: -+ io_prep_fdsync(iocbs, fd); -+ break; - /* Currently Linux kernel does not support other operations */ - default: - fprintf(stderr, "%s: invalid AIO request type 0x%x.\n", -@@ -412,7 +415,7 @@ int coroutine_fn laio_co_submit(int fd, uint64_t offset, QEMUIOVector *qiov, - AioContext *ctx = qemu_get_current_aio_context(); - struct qemu_laiocb laiocb = { - .co = qemu_coroutine_self(), -- .nbytes = qiov->size, -+ .nbytes = qiov ? qiov->size : 0, - .ctx = aio_get_linux_aio(ctx), - .ret = -EINPROGRESS, - .is_read = (type == QEMU_AIO_READ), -@@ -486,3 +489,19 @@ void laio_cleanup(LinuxAioState *s) - } - g_free(s); - } -+ -+bool laio_has_fdsync(int fd) -+{ -+ struct iocb cb; -+ struct iocb *cbs[] = {&cb, NULL}; -+ -+ io_context_t ctx = 0; -+ io_setup(1, &ctx); -+ -+ /* check if host kernel supports IO_CMD_FDSYNC */ -+ io_prep_fdsync(&cb, fd); -+ int ret = io_submit(ctx, 1, cbs); -+ -+ io_destroy(ctx); -+ return (ret == -EINVAL) ? false : true; -+} -diff --git a/include/block/raw-aio.h b/include/block/raw-aio.h -index 20e000b8ef..626706827f 100644 ---- a/include/block/raw-aio.h -+++ b/include/block/raw-aio.h -@@ -60,6 +60,7 @@ void laio_cleanup(LinuxAioState *s); - int coroutine_fn laio_co_submit(int fd, uint64_t offset, QEMUIOVector *qiov, - int type, uint64_t dev_max_batch); - -+bool laio_has_fdsync(int); - void laio_detach_aio_context(LinuxAioState *s, AioContext *old_context); - void laio_attach_aio_context(LinuxAioState *s, AioContext *new_context); - #endif --- -2.39.3 - diff --git a/kvm-linux-headers-Update-to-Linux-v6.17-rc1.patch b/kvm-linux-headers-Update-to-Linux-v6.17-rc1.patch new file mode 100644 index 0000000..efc0d20 --- /dev/null +++ b/kvm-linux-headers-Update-to-Linux-v6.17-rc1.patch @@ -0,0 +1,941 @@ +From b5bbab573e01c1225e7368dba469086a78bc1570 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:17 +0200 +Subject: [PATCH 08/19] linux-headers: Update to Linux v6.17-rc1 + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [3/14] 5b5aab965427deef98614bcfc99c4b3b9818b46d (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Update headers to include the virtio GSO over UDP tunnel features + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Acked-by: Stefano Garzarella +Reviewed-by: Michael S. Tsirkin +Message-ID: <0b1f3c011f90583ab52aa4fef04df6db35cc4a69.1758549625.git.pabeni@redhat.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 8de6cd5452eb9c58c0d105dbc9718bd0e83cc70f) +Signed-off-by: Laurent Vivier +--- + include/standard-headers/drm/drm_fourcc.h | 56 ++++++- + include/standard-headers/linux/ethtool.h | 4 +- + .../linux/input-event-codes.h | 8 + + include/standard-headers/linux/input.h | 1 + + include/standard-headers/linux/pci_regs.h | 9 + + include/standard-headers/linux/vhost_types.h | 5 + + include/standard-headers/linux/virtio_net.h | 33 ++++ + linux-headers/LICENSES/preferred/GPL-2.0 | 10 +- + linux-headers/asm-arm64/unistd_64.h | 2 + + linux-headers/asm-generic/unistd.h | 8 +- + linux-headers/asm-loongarch/unistd_64.h | 2 + + linux-headers/asm-mips/unistd_n32.h | 2 + + linux-headers/asm-mips/unistd_n64.h | 2 + + linux-headers/asm-mips/unistd_o32.h | 2 + + linux-headers/asm-powerpc/kvm.h | 13 -- + linux-headers/asm-powerpc/unistd_32.h | 2 + + linux-headers/asm-powerpc/unistd_64.h | 2 + + linux-headers/asm-riscv/kvm.h | 1 + + linux-headers/asm-riscv/unistd_32.h | 2 + + linux-headers/asm-riscv/unistd_64.h | 2 + + linux-headers/asm-s390/unistd_32.h | 2 + + linux-headers/asm-s390/unistd_64.h | 2 + + linux-headers/asm-x86/unistd_32.h | 2 + + linux-headers/asm-x86/unistd_64.h | 2 + + linux-headers/asm-x86/unistd_x32.h | 2 + + linux-headers/linux/iommufd.h | 154 +++++++++++++++++- + linux-headers/linux/kvm.h | 2 + + linux-headers/linux/vfio.h | 12 +- + linux-headers/linux/vhost.h | 35 ++++ + 29 files changed, 352 insertions(+), 27 deletions(-) + +diff --git a/include/standard-headers/drm/drm_fourcc.h b/include/standard-headers/drm/drm_fourcc.h +index c8309d378b..cef077dfb3 100644 +--- a/include/standard-headers/drm/drm_fourcc.h ++++ b/include/standard-headers/drm/drm_fourcc.h +@@ -209,6 +209,10 @@ extern "C" { + #define DRM_FORMAT_RGBA1010102 fourcc_code('R', 'A', '3', '0') /* [31:0] R:G:B:A 10:10:10:2 little endian */ + #define DRM_FORMAT_BGRA1010102 fourcc_code('B', 'A', '3', '0') /* [31:0] B:G:R:A 10:10:10:2 little endian */ + ++/* 48 bpp RGB */ ++#define DRM_FORMAT_RGB161616 fourcc_code('R', 'G', '4', '8') /* [47:0] R:G:B 16:16:16 little endian */ ++#define DRM_FORMAT_BGR161616 fourcc_code('B', 'G', '4', '8') /* [47:0] B:G:R 16:16:16 little endian */ ++ + /* 64 bpp RGB */ + #define DRM_FORMAT_XRGB16161616 fourcc_code('X', 'R', '4', '8') /* [63:0] x:R:G:B 16:16:16:16 little endian */ + #define DRM_FORMAT_XBGR16161616 fourcc_code('X', 'B', '4', '8') /* [63:0] x:B:G:R 16:16:16:16 little endian */ +@@ -217,7 +221,7 @@ extern "C" { + #define DRM_FORMAT_ABGR16161616 fourcc_code('A', 'B', '4', '8') /* [63:0] A:B:G:R 16:16:16:16 little endian */ + + /* +- * Floating point 64bpp RGB ++ * Half-Floating point - 16b/component + * IEEE 754-2008 binary16 half-precision float + * [15:0] sign:exponent:mantissa 1:5:10 + */ +@@ -227,6 +231,20 @@ extern "C" { + #define DRM_FORMAT_ARGB16161616F fourcc_code('A', 'R', '4', 'H') /* [63:0] A:R:G:B 16:16:16:16 little endian */ + #define DRM_FORMAT_ABGR16161616F fourcc_code('A', 'B', '4', 'H') /* [63:0] A:B:G:R 16:16:16:16 little endian */ + ++#define DRM_FORMAT_R16F fourcc_code('R', ' ', ' ', 'H') /* [15:0] R 16 little endian */ ++#define DRM_FORMAT_GR1616F fourcc_code('G', 'R', ' ', 'H') /* [31:0] G:R 16:16 little endian */ ++#define DRM_FORMAT_BGR161616F fourcc_code('B', 'G', 'R', 'H') /* [47:0] B:G:R 16:16:16 little endian */ ++ ++/* ++ * Floating point - 32b/component ++ * IEEE 754-2008 binary32 float ++ * [31:0] sign:exponent:mantissa 1:8:23 ++ */ ++#define DRM_FORMAT_R32F fourcc_code('R', ' ', ' ', 'F') /* [31:0] R 32 little endian */ ++#define DRM_FORMAT_GR3232F fourcc_code('G', 'R', ' ', 'F') /* [63:0] R:G 32:32 little endian */ ++#define DRM_FORMAT_BGR323232F fourcc_code('B', 'G', 'R', 'F') /* [95:0] R:G:B 32:32:32 little endian */ ++#define DRM_FORMAT_ABGR32323232F fourcc_code('A', 'B', '8', 'F') /* [127:0] R:G:B:A 32:32:32:32 little endian */ ++ + /* + * RGBA format with 10-bit components packed in 64-bit per pixel, with 6 bits + * of unused padding per component: +@@ -376,6 +394,42 @@ extern "C" { + */ + #define DRM_FORMAT_Q401 fourcc_code('Q', '4', '0', '1') + ++/* ++ * 3 plane YCbCr LSB aligned ++ * In order to use these formats in a similar fashion to MSB aligned ones ++ * implementation can multiply the values by 2^6=64. For that reason the padding ++ * must only contain zeros. ++ * index 0 = Y plane, [15:0] z:Y [6:10] little endian ++ * index 1 = Cr plane, [15:0] z:Cr [6:10] little endian ++ * index 2 = Cb plane, [15:0] z:Cb [6:10] little endian ++ */ ++#define DRM_FORMAT_S010 fourcc_code('S', '0', '1', '0') /* 2x2 subsampled Cb (1) and Cr (2) planes 10 bits per channel */ ++#define DRM_FORMAT_S210 fourcc_code('S', '2', '1', '0') /* 2x1 subsampled Cb (1) and Cr (2) planes 10 bits per channel */ ++#define DRM_FORMAT_S410 fourcc_code('S', '4', '1', '0') /* non-subsampled Cb (1) and Cr (2) planes 10 bits per channel */ ++ ++/* ++ * 3 plane YCbCr LSB aligned ++ * In order to use these formats in a similar fashion to MSB aligned ones ++ * implementation can multiply the values by 2^4=16. For that reason the padding ++ * must only contain zeros. ++ * index 0 = Y plane, [15:0] z:Y [4:12] little endian ++ * index 1 = Cr plane, [15:0] z:Cr [4:12] little endian ++ * index 2 = Cb plane, [15:0] z:Cb [4:12] little endian ++ */ ++#define DRM_FORMAT_S012 fourcc_code('S', '0', '1', '2') /* 2x2 subsampled Cb (1) and Cr (2) planes 12 bits per channel */ ++#define DRM_FORMAT_S212 fourcc_code('S', '2', '1', '2') /* 2x1 subsampled Cb (1) and Cr (2) planes 12 bits per channel */ ++#define DRM_FORMAT_S412 fourcc_code('S', '4', '1', '2') /* non-subsampled Cb (1) and Cr (2) planes 12 bits per channel */ ++ ++/* ++ * 3 plane YCbCr ++ * index 0 = Y plane, [15:0] Y little endian ++ * index 1 = Cr plane, [15:0] Cr little endian ++ * index 2 = Cb plane, [15:0] Cb little endian ++ */ ++#define DRM_FORMAT_S016 fourcc_code('S', '0', '1', '6') /* 2x2 subsampled Cb (1) and Cr (2) planes 16 bits per channel */ ++#define DRM_FORMAT_S216 fourcc_code('S', '2', '1', '6') /* 2x1 subsampled Cb (1) and Cr (2) planes 16 bits per channel */ ++#define DRM_FORMAT_S416 fourcc_code('S', '4', '1', '6') /* non-subsampled Cb (1) and Cr (2) planes 16 bits per channel */ ++ + /* + * 3 plane YCbCr + * index 0: Y plane, [7:0] Y +diff --git a/include/standard-headers/linux/ethtool.h b/include/standard-headers/linux/ethtool.h +index cef0d207a6..eb80314028 100644 +--- a/include/standard-headers/linux/ethtool.h ++++ b/include/standard-headers/linux/ethtool.h +@@ -2314,7 +2314,7 @@ enum { + IPV6_USER_FLOW = 0x0e, /* spec only (usr_ip6_spec; nfc only) */ + IPV4_FLOW = 0x10, /* hash only */ + IPV6_FLOW = 0x11, /* hash only */ +- ETHER_FLOW = 0x12, /* spec only (ether_spec) */ ++ ETHER_FLOW = 0x12, /* hash or spec (ether_spec) */ + + /* Used for GTP-U IPv4 and IPv6. + * The format of GTP packets only includes +@@ -2371,7 +2371,7 @@ enum { + /* Flag to enable RSS spreading of traffic matching rule (nfc only) */ + #define FLOW_RSS 0x20000000 + +-/* L3-L4 network traffic flow hash options */ ++/* L2-L4 network traffic flow hash options */ + #define RXH_L2DA (1 << 1) + #define RXH_VLAN (1 << 2) + #define RXH_L3_PROTO (1 << 3) +diff --git a/include/standard-headers/linux/input-event-codes.h b/include/standard-headers/linux/input-event-codes.h +index a82ff795e0..00dc9caac9 100644 +--- a/include/standard-headers/linux/input-event-codes.h ++++ b/include/standard-headers/linux/input-event-codes.h +@@ -601,6 +601,11 @@ + #define BTN_DPAD_LEFT 0x222 + #define BTN_DPAD_RIGHT 0x223 + ++#define BTN_GRIPL 0x224 ++#define BTN_GRIPR 0x225 ++#define BTN_GRIPL2 0x226 ++#define BTN_GRIPR2 0x227 ++ + #define KEY_ALS_TOGGLE 0x230 /* Ambient light sensor */ + #define KEY_ROTATE_LOCK_TOGGLE 0x231 /* Display rotation lock */ + #define KEY_REFRESH_RATE_TOGGLE 0x232 /* Display refresh rate toggle */ +@@ -765,6 +770,9 @@ + #define KEY_KBD_LCD_MENU4 0x2bb + #define KEY_KBD_LCD_MENU5 0x2bc + ++/* Performance Boost key (Alienware)/G-Mode key (Dell) */ ++#define KEY_PERFORMANCE 0x2bd ++ + #define BTN_TRIGGER_HAPPY 0x2c0 + #define BTN_TRIGGER_HAPPY1 0x2c0 + #define BTN_TRIGGER_HAPPY2 0x2c1 +diff --git a/include/standard-headers/linux/input.h b/include/standard-headers/linux/input.h +index 942ea6aaa9..d4512c20b5 100644 +--- a/include/standard-headers/linux/input.h ++++ b/include/standard-headers/linux/input.h +@@ -272,6 +272,7 @@ struct input_mask { + #define BUS_CEC 0x1E + #define BUS_INTEL_ISHTP 0x1F + #define BUS_AMD_SFH 0x20 ++#define BUS_SDW 0x21 + + /* + * MT_TOOL types +diff --git a/include/standard-headers/linux/pci_regs.h b/include/standard-headers/linux/pci_regs.h +index a3a3e942de..f5b17745de 100644 +--- a/include/standard-headers/linux/pci_regs.h ++++ b/include/standard-headers/linux/pci_regs.h +@@ -745,6 +745,7 @@ + #define PCI_EXT_CAP_ID_L1SS 0x1E /* L1 PM Substates */ + #define PCI_EXT_CAP_ID_PTM 0x1F /* Precision Time Measurement */ + #define PCI_EXT_CAP_ID_DVSEC 0x23 /* Designated Vendor-Specific */ ++#define PCI_EXT_CAP_ID_VF_REBAR 0x24 /* VF Resizable BAR */ + #define PCI_EXT_CAP_ID_DLF 0x25 /* Data Link Feature */ + #define PCI_EXT_CAP_ID_PL_16GT 0x26 /* Physical Layer 16.0 GT/s */ + #define PCI_EXT_CAP_ID_NPEM 0x29 /* Native PCIe Enclosure Management */ +@@ -1141,6 +1142,14 @@ + #define PCI_DVSEC_HEADER2 0x8 /* Designated Vendor-Specific Header2 */ + #define PCI_DVSEC_HEADER2_ID(x) ((x) & 0xffff) + ++/* VF Resizable BARs, same layout as PCI_REBAR */ ++#define PCI_VF_REBAR_CAP PCI_REBAR_CAP ++#define PCI_VF_REBAR_CAP_SIZES PCI_REBAR_CAP_SIZES ++#define PCI_VF_REBAR_CTRL PCI_REBAR_CTRL ++#define PCI_VF_REBAR_CTRL_BAR_IDX PCI_REBAR_CTRL_BAR_IDX ++#define PCI_VF_REBAR_CTRL_NBAR_MASK PCI_REBAR_CTRL_NBAR_MASK ++#define PCI_VF_REBAR_CTRL_BAR_SIZE PCI_REBAR_CTRL_BAR_SIZE ++ + /* Data Link Feature */ + #define PCI_DLF_CAP 0x04 /* Capabilities Register */ + #define PCI_DLF_EXCHANGE_ENABLE 0x80000000 /* Data Link Feature Exchange Enable */ +diff --git a/include/standard-headers/linux/vhost_types.h b/include/standard-headers/linux/vhost_types.h +index fd54044936..79b53a931a 100644 +--- a/include/standard-headers/linux/vhost_types.h ++++ b/include/standard-headers/linux/vhost_types.h +@@ -110,6 +110,11 @@ struct vhost_msg_v2 { + }; + }; + ++struct vhost_features_array { ++ uint64_t count; /* number of entries present in features array */ ++ uint64_t features[] ; ++}; ++ + struct vhost_memory_region { + uint64_t guest_phys_addr; + uint64_t memory_size; /* bytes */ +diff --git a/include/standard-headers/linux/virtio_net.h b/include/standard-headers/linux/virtio_net.h +index 982e854f14..93abaae0b9 100644 +--- a/include/standard-headers/linux/virtio_net.h ++++ b/include/standard-headers/linux/virtio_net.h +@@ -70,6 +70,28 @@ + * with the same MAC. + */ + #define VIRTIO_NET_F_SPEED_DUPLEX 63 /* Device set linkspeed and duplex */ ++#define VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO 65 /* Driver can receive ++ * GSO-over-UDP-tunnel packets ++ */ ++#define VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM 66 /* Driver handles ++ * GSO-over-UDP-tunnel ++ * packets with partial csum ++ * for the outer header ++ */ ++#define VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO 67 /* Device can receive ++ * GSO-over-UDP-tunnel packets ++ */ ++#define VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO_CSUM 68 /* Device handles ++ * GSO-over-UDP-tunnel ++ * packets with partial csum ++ * for the outer header ++ */ ++ ++/* Offloads bits corresponding to VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO{,_CSUM} ++ * features ++ */ ++#define VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_MAPPED 46 ++#define VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM_MAPPED 47 + + #ifndef VIRTIO_NET_NO_LEGACY + #define VIRTIO_NET_F_GSO 6 /* Host handles pkts w/ any GSO type */ +@@ -131,12 +153,17 @@ struct virtio_net_hdr_v1 { + #define VIRTIO_NET_HDR_F_NEEDS_CSUM 1 /* Use csum_start, csum_offset */ + #define VIRTIO_NET_HDR_F_DATA_VALID 2 /* Csum is valid */ + #define VIRTIO_NET_HDR_F_RSC_INFO 4 /* rsc info in csum_ fields */ ++#define VIRTIO_NET_HDR_F_UDP_TUNNEL_CSUM 8 /* UDP tunnel csum offload */ + uint8_t flags; + #define VIRTIO_NET_HDR_GSO_NONE 0 /* Not a GSO frame */ + #define VIRTIO_NET_HDR_GSO_TCPV4 1 /* GSO frame, IPv4 TCP (TSO) */ + #define VIRTIO_NET_HDR_GSO_UDP 3 /* GSO frame, IPv4 UDP (UFO) */ + #define VIRTIO_NET_HDR_GSO_TCPV6 4 /* GSO frame, IPv6 TCP */ + #define VIRTIO_NET_HDR_GSO_UDP_L4 5 /* GSO frame, IPv4& IPv6 UDP (USO) */ ++#define VIRTIO_NET_HDR_GSO_UDP_TUNNEL_IPV4 0x20 /* UDPv4 tunnel present */ ++#define VIRTIO_NET_HDR_GSO_UDP_TUNNEL_IPV6 0x40 /* UDPv6 tunnel present */ ++#define VIRTIO_NET_HDR_GSO_UDP_TUNNEL (VIRTIO_NET_HDR_GSO_UDP_TUNNEL_IPV4 | \ ++ VIRTIO_NET_HDR_GSO_UDP_TUNNEL_IPV6) + #define VIRTIO_NET_HDR_GSO_ECN 0x80 /* TCP has ECN set */ + uint8_t gso_type; + __virtio16 hdr_len; /* Ethernet + IP + tcp/udp hdrs */ +@@ -181,6 +208,12 @@ struct virtio_net_hdr_v1_hash { + uint16_t padding; + }; + ++struct virtio_net_hdr_v1_hash_tunnel { ++ struct virtio_net_hdr_v1_hash hash_hdr; ++ uint16_t outer_th_offset; ++ uint16_t inner_nh_offset; ++}; ++ + #ifndef VIRTIO_NET_NO_LEGACY + /* This header comes first in the scatter-gather list. + * For legacy virtio, if VIRTIO_F_ANY_LAYOUT is not negotiated, it must +diff --git a/linux-headers/LICENSES/preferred/GPL-2.0 b/linux-headers/LICENSES/preferred/GPL-2.0 +index ff0812fd89..ea8e93dc44 100644 +--- a/linux-headers/LICENSES/preferred/GPL-2.0 ++++ b/linux-headers/LICENSES/preferred/GPL-2.0 +@@ -20,8 +20,8 @@ License-Text: + GNU GENERAL PUBLIC LICENSE + Version 2, June 1991 + +- Copyright (C) 1989, 1991 Free Software Foundation, Inc. +- 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA ++ Copyright (C) 1989, 1991 Free Software Foundation, Inc., ++ + Everyone is permitted to copy and distribute verbatim copies + of this license document, but changing it is not allowed. + +@@ -322,10 +322,8 @@ the "copyright" line and a pointer to where the full notice is found. + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + +- You should have received a copy of the GNU General Public License +- along with this program; if not, write to the Free Software +- Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA +- ++ You should have received a copy of the GNU General Public License along ++ with this program; if not, see . + + Also add information on how to contact you by electronic and paper mail. + +diff --git a/linux-headers/asm-arm64/unistd_64.h b/linux-headers/asm-arm64/unistd_64.h +index ee9aaebdf3..4ae25c2b91 100644 +--- a/linux-headers/asm-arm64/unistd_64.h ++++ b/linux-headers/asm-arm64/unistd_64.h +@@ -324,6 +324,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-generic/unistd.h b/linux-headers/asm-generic/unistd.h +index 2892a45023..04e0077fb4 100644 +--- a/linux-headers/asm-generic/unistd.h ++++ b/linux-headers/asm-generic/unistd.h +@@ -852,8 +852,14 @@ __SYSCALL(__NR_removexattrat, sys_removexattrat) + #define __NR_open_tree_attr 467 + __SYSCALL(__NR_open_tree_attr, sys_open_tree_attr) + ++/* fs/inode.c */ ++#define __NR_file_getattr 468 ++__SYSCALL(__NR_file_getattr, sys_file_getattr) ++#define __NR_file_setattr 469 ++__SYSCALL(__NR_file_setattr, sys_file_setattr) ++ + #undef __NR_syscalls +-#define __NR_syscalls 468 ++#define __NR_syscalls 470 + + /* + * 32 bit systems traditionally used different +diff --git a/linux-headers/asm-loongarch/unistd_64.h b/linux-headers/asm-loongarch/unistd_64.h +index 50d22df8f7..5033fc8f2f 100644 +--- a/linux-headers/asm-loongarch/unistd_64.h ++++ b/linux-headers/asm-loongarch/unistd_64.h +@@ -320,6 +320,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-mips/unistd_n32.h b/linux-headers/asm-mips/unistd_n32.h +index bdcc2f460b..c99c10e5bf 100644 +--- a/linux-headers/asm-mips/unistd_n32.h ++++ b/linux-headers/asm-mips/unistd_n32.h +@@ -396,5 +396,7 @@ + #define __NR_listxattrat (__NR_Linux + 465) + #define __NR_removexattrat (__NR_Linux + 466) + #define __NR_open_tree_attr (__NR_Linux + 467) ++#define __NR_file_getattr (__NR_Linux + 468) ++#define __NR_file_setattr (__NR_Linux + 469) + + #endif /* _ASM_UNISTD_N32_H */ +diff --git a/linux-headers/asm-mips/unistd_n64.h b/linux-headers/asm-mips/unistd_n64.h +index 3b6b0193b6..0d975bb185 100644 +--- a/linux-headers/asm-mips/unistd_n64.h ++++ b/linux-headers/asm-mips/unistd_n64.h +@@ -372,5 +372,7 @@ + #define __NR_listxattrat (__NR_Linux + 465) + #define __NR_removexattrat (__NR_Linux + 466) + #define __NR_open_tree_attr (__NR_Linux + 467) ++#define __NR_file_getattr (__NR_Linux + 468) ++#define __NR_file_setattr (__NR_Linux + 469) + + #endif /* _ASM_UNISTD_N64_H */ +diff --git a/linux-headers/asm-mips/unistd_o32.h b/linux-headers/asm-mips/unistd_o32.h +index 4609a4b4d3..86ac0ac84b 100644 +--- a/linux-headers/asm-mips/unistd_o32.h ++++ b/linux-headers/asm-mips/unistd_o32.h +@@ -442,5 +442,7 @@ + #define __NR_listxattrat (__NR_Linux + 465) + #define __NR_removexattrat (__NR_Linux + 466) + #define __NR_open_tree_attr (__NR_Linux + 467) ++#define __NR_file_getattr (__NR_Linux + 468) ++#define __NR_file_setattr (__NR_Linux + 469) + + #endif /* _ASM_UNISTD_O32_H */ +diff --git a/linux-headers/asm-powerpc/kvm.h b/linux-headers/asm-powerpc/kvm.h +index eaeda00178..077c5437f5 100644 +--- a/linux-headers/asm-powerpc/kvm.h ++++ b/linux-headers/asm-powerpc/kvm.h +@@ -1,18 +1,5 @@ + /* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */ + /* +- * This program is free software; you can redistribute it and/or modify +- * it under the terms of the GNU General Public License, version 2, as +- * published by the Free Software Foundation. +- * +- * This program is distributed in the hope that it will be useful, +- * but WITHOUT ANY WARRANTY; without even the implied warranty of +- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +- * GNU General Public License for more details. +- * +- * You should have received a copy of the GNU General Public License +- * along with this program; if not, write to the Free Software +- * Foundation, 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301, USA. +- * + * Copyright IBM Corp. 2007 + * + * Authors: Hollis Blanchard +diff --git a/linux-headers/asm-powerpc/unistd_32.h b/linux-headers/asm-powerpc/unistd_32.h +index 5d38a427e0..d7a32c5e06 100644 +--- a/linux-headers/asm-powerpc/unistd_32.h ++++ b/linux-headers/asm-powerpc/unistd_32.h +@@ -449,6 +449,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_32_H */ +diff --git a/linux-headers/asm-powerpc/unistd_64.h b/linux-headers/asm-powerpc/unistd_64.h +index 860a488e4d..ff35c51fc6 100644 +--- a/linux-headers/asm-powerpc/unistd_64.h ++++ b/linux-headers/asm-powerpc/unistd_64.h +@@ -421,6 +421,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-riscv/kvm.h b/linux-headers/asm-riscv/kvm.h +index 5f59fd226c..ef27d4289d 100644 +--- a/linux-headers/asm-riscv/kvm.h ++++ b/linux-headers/asm-riscv/kvm.h +@@ -18,6 +18,7 @@ + #define __KVM_HAVE_IRQ_LINE + + #define KVM_COALESCED_MMIO_PAGE_OFFSET 1 ++#define KVM_DIRTY_LOG_PAGE_OFFSET 64 + + #define KVM_INTERRUPT_SET -1U + #define KVM_INTERRUPT_UNSET -2U +diff --git a/linux-headers/asm-riscv/unistd_32.h b/linux-headers/asm-riscv/unistd_32.h +index a5e769f1d9..6083373e88 100644 +--- a/linux-headers/asm-riscv/unistd_32.h ++++ b/linux-headers/asm-riscv/unistd_32.h +@@ -315,6 +315,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_32_H */ +diff --git a/linux-headers/asm-riscv/unistd_64.h b/linux-headers/asm-riscv/unistd_64.h +index 8df4d64841..f0c7585c60 100644 +--- a/linux-headers/asm-riscv/unistd_64.h ++++ b/linux-headers/asm-riscv/unistd_64.h +@@ -325,6 +325,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-s390/unistd_32.h b/linux-headers/asm-s390/unistd_32.h +index 85eedbd18e..37b8f6f358 100644 +--- a/linux-headers/asm-s390/unistd_32.h ++++ b/linux-headers/asm-s390/unistd_32.h +@@ -440,5 +440,7 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + #endif /* _ASM_S390_UNISTD_32_H */ +diff --git a/linux-headers/asm-s390/unistd_64.h b/linux-headers/asm-s390/unistd_64.h +index c03b1b9701..0652ba6331 100644 +--- a/linux-headers/asm-s390/unistd_64.h ++++ b/linux-headers/asm-s390/unistd_64.h +@@ -388,5 +388,7 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + #endif /* _ASM_S390_UNISTD_64_H */ +diff --git a/linux-headers/asm-x86/unistd_32.h b/linux-headers/asm-x86/unistd_32.h +index 491d6b4eb6..8f784a5634 100644 +--- a/linux-headers/asm-x86/unistd_32.h ++++ b/linux-headers/asm-x86/unistd_32.h +@@ -458,6 +458,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_32_H */ +diff --git a/linux-headers/asm-x86/unistd_64.h b/linux-headers/asm-x86/unistd_64.h +index 7cf88bf9bd..2f55bebb81 100644 +--- a/linux-headers/asm-x86/unistd_64.h ++++ b/linux-headers/asm-x86/unistd_64.h +@@ -381,6 +381,8 @@ + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 + #define __NR_open_tree_attr 467 ++#define __NR_file_getattr 468 ++#define __NR_file_setattr 469 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-x86/unistd_x32.h b/linux-headers/asm-x86/unistd_x32.h +index 82959111e6..8cc8673f15 100644 +--- a/linux-headers/asm-x86/unistd_x32.h ++++ b/linux-headers/asm-x86/unistd_x32.h +@@ -334,6 +334,8 @@ + #define __NR_listxattrat (__X32_SYSCALL_BIT + 465) + #define __NR_removexattrat (__X32_SYSCALL_BIT + 466) + #define __NR_open_tree_attr (__X32_SYSCALL_BIT + 467) ++#define __NR_file_getattr (__X32_SYSCALL_BIT + 468) ++#define __NR_file_setattr (__X32_SYSCALL_BIT + 469) + #define __NR_rt_sigaction (__X32_SYSCALL_BIT + 512) + #define __NR_rt_sigreturn (__X32_SYSCALL_BIT + 513) + #define __NR_ioctl (__X32_SYSCALL_BIT + 514) +diff --git a/linux-headers/linux/iommufd.h b/linux-headers/linux/iommufd.h +index cb0f7d6b4d..2105a03955 100644 +--- a/linux-headers/linux/iommufd.h ++++ b/linux-headers/linux/iommufd.h +@@ -56,6 +56,7 @@ enum { + IOMMUFD_CMD_VDEVICE_ALLOC = 0x91, + IOMMUFD_CMD_IOAS_CHANGE_PROCESS = 0x92, + IOMMUFD_CMD_VEVENTQ_ALLOC = 0x93, ++ IOMMUFD_CMD_HW_QUEUE_ALLOC = 0x94, + }; + + /** +@@ -590,17 +591,44 @@ struct iommu_hw_info_arm_smmuv3 { + __u32 aidr; + }; + ++/** ++ * struct iommu_hw_info_tegra241_cmdqv - NVIDIA Tegra241 CMDQV Hardware ++ * Information (IOMMU_HW_INFO_TYPE_TEGRA241_CMDQV) ++ * ++ * @flags: Must be 0 ++ * @version: Version number for the CMDQ-V HW for PARAM bits[03:00] ++ * @log2vcmdqs: Log2 of the total number of VCMDQs for PARAM bits[07:04] ++ * @log2vsids: Log2 of the total number of SID replacements for PARAM bits[15:12] ++ * @__reserved: Must be 0 ++ * ++ * VMM can use these fields directly in its emulated global PARAM register. Note ++ * that only one Virtual Interface (VINTF) should be exposed to a VM, i.e. PARAM ++ * bits[11:08] should be set to 0 for log2 of the total number of VINTFs. ++ */ ++struct iommu_hw_info_tegra241_cmdqv { ++ __u32 flags; ++ __u8 version; ++ __u8 log2vcmdqs; ++ __u8 log2vsids; ++ __u8 __reserved; ++}; ++ + /** + * enum iommu_hw_info_type - IOMMU Hardware Info Types +- * @IOMMU_HW_INFO_TYPE_NONE: Used by the drivers that do not report hardware ++ * @IOMMU_HW_INFO_TYPE_NONE: Output by the drivers that do not report hardware + * info ++ * @IOMMU_HW_INFO_TYPE_DEFAULT: Input to request for a default type + * @IOMMU_HW_INFO_TYPE_INTEL_VTD: Intel VT-d iommu info type + * @IOMMU_HW_INFO_TYPE_ARM_SMMUV3: ARM SMMUv3 iommu info type ++ * @IOMMU_HW_INFO_TYPE_TEGRA241_CMDQV: NVIDIA Tegra241 CMDQV (extension for ARM ++ * SMMUv3) info type + */ + enum iommu_hw_info_type { + IOMMU_HW_INFO_TYPE_NONE = 0, ++ IOMMU_HW_INFO_TYPE_DEFAULT = 0, + IOMMU_HW_INFO_TYPE_INTEL_VTD = 1, + IOMMU_HW_INFO_TYPE_ARM_SMMUV3 = 2, ++ IOMMU_HW_INFO_TYPE_TEGRA241_CMDQV = 3, + }; + + /** +@@ -625,6 +653,15 @@ enum iommufd_hw_capabilities { + IOMMU_HW_CAP_PCI_PASID_PRIV = 1 << 2, + }; + ++/** ++ * enum iommufd_hw_info_flags - Flags for iommu_hw_info ++ * @IOMMU_HW_INFO_FLAG_INPUT_TYPE: If set, @in_data_type carries an input type ++ * for user space to request for a specific info ++ */ ++enum iommufd_hw_info_flags { ++ IOMMU_HW_INFO_FLAG_INPUT_TYPE = 1 << 0, ++}; ++ + /** + * struct iommu_hw_info - ioctl(IOMMU_GET_HW_INFO) + * @size: sizeof(struct iommu_hw_info) +@@ -634,6 +671,12 @@ enum iommufd_hw_capabilities { + * data that kernel supports + * @data_uptr: User pointer to a user-space buffer used by the kernel to fill + * the iommu type specific hardware information data ++ * @in_data_type: This shares the same field with @out_data_type, making it be ++ * a bidirectional field. When IOMMU_HW_INFO_FLAG_INPUT_TYPE is ++ * set, an input type carried via this @in_data_type field will ++ * be valid, requesting for the info data to the given type. If ++ * IOMMU_HW_INFO_FLAG_INPUT_TYPE is unset, any input value will ++ * be seen as IOMMU_HW_INFO_TYPE_DEFAULT + * @out_data_type: Output the iommu hardware info type as defined in the enum + * iommu_hw_info_type. + * @out_capabilities: Output the generic iommu capability info type as defined +@@ -663,7 +706,10 @@ struct iommu_hw_info { + __u32 dev_id; + __u32 data_len; + __aligned_u64 data_uptr; +- __u32 out_data_type; ++ union { ++ __u32 in_data_type; ++ __u32 out_data_type; ++ }; + __u8 out_max_pasid_log2; + __u8 __reserved[3]; + __aligned_u64 out_capabilities; +@@ -951,10 +997,29 @@ struct iommu_fault_alloc { + * enum iommu_viommu_type - Virtual IOMMU Type + * @IOMMU_VIOMMU_TYPE_DEFAULT: Reserved for future use + * @IOMMU_VIOMMU_TYPE_ARM_SMMUV3: ARM SMMUv3 driver specific type ++ * @IOMMU_VIOMMU_TYPE_TEGRA241_CMDQV: NVIDIA Tegra241 CMDQV (extension for ARM ++ * SMMUv3) enabled ARM SMMUv3 type + */ + enum iommu_viommu_type { + IOMMU_VIOMMU_TYPE_DEFAULT = 0, + IOMMU_VIOMMU_TYPE_ARM_SMMUV3 = 1, ++ IOMMU_VIOMMU_TYPE_TEGRA241_CMDQV = 2, ++}; ++ ++/** ++ * struct iommu_viommu_tegra241_cmdqv - NVIDIA Tegra241 CMDQV Virtual Interface ++ * (IOMMU_VIOMMU_TYPE_TEGRA241_CMDQV) ++ * @out_vintf_mmap_offset: mmap offset argument for VINTF's page0 ++ * @out_vintf_mmap_length: mmap length argument for VINTF's page0 ++ * ++ * Both @out_vintf_mmap_offset and @out_vintf_mmap_length are reported by kernel ++ * for user space to mmap the VINTF page0 from the host physical address space ++ * to the guest physical address space so that a guest kernel can directly R/W ++ * access to the VINTF page0 in order to control its virtual command queues. ++ */ ++struct iommu_viommu_tegra241_cmdqv { ++ __aligned_u64 out_vintf_mmap_offset; ++ __aligned_u64 out_vintf_mmap_length; + }; + + /** +@@ -965,6 +1030,9 @@ enum iommu_viommu_type { + * @dev_id: The device's physical IOMMU will be used to back the virtual IOMMU + * @hwpt_id: ID of a nesting parent HWPT to associate to + * @out_viommu_id: Output virtual IOMMU ID for the allocated object ++ * @data_len: Length of the type specific data ++ * @__reserved: Must be 0 ++ * @data_uptr: User pointer to a driver-specific virtual IOMMU data + * + * Allocate a virtual IOMMU object, representing the underlying physical IOMMU's + * virtualization support that is a security-isolated slice of the real IOMMU HW +@@ -985,6 +1053,9 @@ struct iommu_viommu_alloc { + __u32 dev_id; + __u32 hwpt_id; + __u32 out_viommu_id; ++ __u32 data_len; ++ __u32 __reserved; ++ __aligned_u64 data_uptr; + }; + #define IOMMU_VIOMMU_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_VIOMMU_ALLOC) + +@@ -995,10 +1066,15 @@ struct iommu_viommu_alloc { + * @dev_id: The physical device to allocate a virtual instance on the vIOMMU + * @out_vdevice_id: Object handle for the vDevice. Pass to IOMMU_DESTORY + * @virt_id: Virtual device ID per vIOMMU, e.g. vSID of ARM SMMUv3, vDeviceID +- * of AMD IOMMU, and vRID of a nested Intel VT-d to a Context Table ++ * of AMD IOMMU, and vRID of Intel VT-d + * + * Allocate a virtual device instance (for a physical device) against a vIOMMU. + * This instance holds the device's information (related to its vIOMMU) in a VM. ++ * User should use IOMMU_DESTROY to destroy the virtual device before ++ * destroying the physical device (by closing vfio_cdev fd). Otherwise the ++ * virtual device would be forcibly destroyed on physical device destruction, ++ * its vdevice_id would be permanently leaked (unremovable & unreusable) until ++ * iommu fd closed. + */ + struct iommu_vdevice_alloc { + __u32 size; +@@ -1075,10 +1151,12 @@ struct iommufd_vevent_header { + * enum iommu_veventq_type - Virtual Event Queue Type + * @IOMMU_VEVENTQ_TYPE_DEFAULT: Reserved for future use + * @IOMMU_VEVENTQ_TYPE_ARM_SMMUV3: ARM SMMUv3 Virtual Event Queue ++ * @IOMMU_VEVENTQ_TYPE_TEGRA241_CMDQV: NVIDIA Tegra241 CMDQV Extension IRQ + */ + enum iommu_veventq_type { + IOMMU_VEVENTQ_TYPE_DEFAULT = 0, + IOMMU_VEVENTQ_TYPE_ARM_SMMUV3 = 1, ++ IOMMU_VEVENTQ_TYPE_TEGRA241_CMDQV = 2, + }; + + /** +@@ -1102,6 +1180,19 @@ struct iommu_vevent_arm_smmuv3 { + __aligned_le64 evt[4]; + }; + ++/** ++ * struct iommu_vevent_tegra241_cmdqv - Tegra241 CMDQV IRQ ++ * (IOMMU_VEVENTQ_TYPE_TEGRA241_CMDQV) ++ * @lvcmdq_err_map: 128-bit logical vcmdq error map, little-endian. ++ * (Refer to register LVCMDQ_ERR_MAPs per VINTF ) ++ * ++ * The 128-bit register value from HW exclusively reflect the error bits for a ++ * Virtual Interface represented by a vIOMMU object. Read and report directly. ++ */ ++struct iommu_vevent_tegra241_cmdqv { ++ __aligned_le64 lvcmdq_err_map[2]; ++}; ++ + /** + * struct iommu_veventq_alloc - ioctl(IOMMU_VEVENTQ_ALLOC) + * @size: sizeof(struct iommu_veventq_alloc) +@@ -1141,4 +1232,61 @@ struct iommu_veventq_alloc { + __u32 __reserved; + }; + #define IOMMU_VEVENTQ_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_VEVENTQ_ALLOC) ++ ++/** ++ * enum iommu_hw_queue_type - HW Queue Type ++ * @IOMMU_HW_QUEUE_TYPE_DEFAULT: Reserved for future use ++ * @IOMMU_HW_QUEUE_TYPE_TEGRA241_CMDQV: NVIDIA Tegra241 CMDQV (extension for ARM ++ * SMMUv3) Virtual Command Queue (VCMDQ) ++ */ ++enum iommu_hw_queue_type { ++ IOMMU_HW_QUEUE_TYPE_DEFAULT = 0, ++ /* ++ * TEGRA241_CMDQV requirements (otherwise, allocation will fail) ++ * - alloc starts from the lowest @index=0 in ascending order ++ * - destroy starts from the last allocated @index in descending order ++ * - @base_addr must be aligned to @length in bytes and mapped in IOAS ++ * - @length must be a power of 2, with a minimum 32 bytes and a maximum ++ * 2 ^ idr[1].CMDQS * 16 bytes (use GET_HW_INFO call to read idr[1] ++ * from struct iommu_hw_info_arm_smmuv3) ++ * - suggest to back the queue memory with contiguous physical pages or ++ * a single huge page with alignment of the queue size, and limit the ++ * emulated vSMMU's IDR1.CMDQS to log2(huge page size / 16 bytes) ++ */ ++ IOMMU_HW_QUEUE_TYPE_TEGRA241_CMDQV = 1, ++}; ++ ++/** ++ * struct iommu_hw_queue_alloc - ioctl(IOMMU_HW_QUEUE_ALLOC) ++ * @size: sizeof(struct iommu_hw_queue_alloc) ++ * @flags: Must be 0 ++ * @viommu_id: Virtual IOMMU ID to associate the HW queue with ++ * @type: One of enum iommu_hw_queue_type ++ * @index: The logical index to the HW queue per virtual IOMMU for a multi-queue ++ * model ++ * @out_hw_queue_id: The ID of the new HW queue ++ * @nesting_parent_iova: Base address of the queue memory in the guest physical ++ * address space ++ * @length: Length of the queue memory ++ * ++ * Allocate a HW queue object for a vIOMMU-specific HW-accelerated queue, which ++ * allows HW to access a guest queue memory described using @nesting_parent_iova ++ * and @length. ++ * ++ * A vIOMMU can allocate multiple queues, but it must use a different @index per ++ * type to separate each allocation, e.g:: ++ * ++ * Type1 HW queue0, Type1 HW queue1, Type2 HW queue0, ... ++ */ ++struct iommu_hw_queue_alloc { ++ __u32 size; ++ __u32 flags; ++ __u32 viommu_id; ++ __u32 type; ++ __u32 index; ++ __u32 out_hw_queue_id; ++ __aligned_u64 nesting_parent_iova; ++ __aligned_u64 length; ++}; ++#define IOMMU_HW_QUEUE_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_HW_QUEUE_ALLOC) + #endif +diff --git a/linux-headers/linux/kvm.h b/linux-headers/linux/kvm.h +index 32c5885a3c..be704965d8 100644 +--- a/linux-headers/linux/kvm.h ++++ b/linux-headers/linux/kvm.h +@@ -636,6 +636,7 @@ struct kvm_ioeventfd { + #define KVM_X86_DISABLE_EXITS_HLT (1 << 1) + #define KVM_X86_DISABLE_EXITS_PAUSE (1 << 2) + #define KVM_X86_DISABLE_EXITS_CSTATE (1 << 3) ++#define KVM_X86_DISABLE_EXITS_APERFMPERF (1 << 4) + + /* for KVM_ENABLE_CAP */ + struct kvm_enable_cap { +@@ -952,6 +953,7 @@ struct kvm_enable_cap { + #define KVM_CAP_ARM_EL2 240 + #define KVM_CAP_ARM_EL2_E2H0 241 + #define KVM_CAP_RISCV_MP_STATE_RESET 242 ++#define KVM_CAP_ARM_CACHEABLE_PFNMAP_SUPPORTED 243 + + struct kvm_irq_routing_irqchip { + __u32 irqchip; +diff --git a/linux-headers/linux/vfio.h b/linux-headers/linux/vfio.h +index 79bf8c0cc5..4d96d1fc12 100644 +--- a/linux-headers/linux/vfio.h ++++ b/linux-headers/linux/vfio.h +@@ -905,10 +905,12 @@ struct vfio_device_feature { + * VFIO_DEVICE_BIND_IOMMUFD - _IOR(VFIO_TYPE, VFIO_BASE + 18, + * struct vfio_device_bind_iommufd) + * @argsz: User filled size of this data. +- * @flags: Must be 0. ++ * @flags: Must be 0 or a bit flags of VFIO_DEVICE_BIND_* + * @iommufd: iommufd to bind. + * @out_devid: The device id generated by this bind. devid is a handle for + * this device/iommufd bond and can be used in IOMMUFD commands. ++ * @token_uuid_ptr: Valid if VFIO_DEVICE_BIND_FLAG_TOKEN. Points to a 16 byte ++ * UUID in the same format as VFIO_DEVICE_FEATURE_PCI_VF_TOKEN. + * + * Bind a vfio_device to the specified iommufd. + * +@@ -917,13 +919,21 @@ struct vfio_device_feature { + * + * Unbind is automatically conducted when device fd is closed. + * ++ * A token is sometimes required to open the device, unless this is known to be ++ * needed VFIO_DEVICE_BIND_FLAG_TOKEN should not be set and token_uuid_ptr is ++ * ignored. The only case today is a PF/VF relationship where the VF bind must ++ * be provided the same token as VFIO_DEVICE_FEATURE_PCI_VF_TOKEN provided to ++ * the PF. ++ * + * Return: 0 on success, -errno on failure. + */ + struct vfio_device_bind_iommufd { + __u32 argsz; + __u32 flags; ++#define VFIO_DEVICE_BIND_FLAG_TOKEN (1 << 0) + __s32 iommufd; + __u32 out_devid; ++ __aligned_u64 token_uuid_ptr; + }; + + #define VFIO_DEVICE_BIND_IOMMUFD _IO(VFIO_TYPE, VFIO_BASE + 18) +diff --git a/linux-headers/linux/vhost.h b/linux-headers/linux/vhost.h +index d4b3e2ae13..283348b64a 100644 +--- a/linux-headers/linux/vhost.h ++++ b/linux-headers/linux/vhost.h +@@ -235,4 +235,39 @@ + */ + #define VHOST_VDPA_GET_VRING_SIZE _IOWR(VHOST_VIRTIO, 0x82, \ + struct vhost_vring_state) ++ ++/* Extended features manipulation */ ++#define VHOST_GET_FEATURES_ARRAY _IOR(VHOST_VIRTIO, 0x83, \ ++ struct vhost_features_array) ++#define VHOST_SET_FEATURES_ARRAY _IOW(VHOST_VIRTIO, 0x83, \ ++ struct vhost_features_array) ++ ++/* fork_owner values for vhost */ ++#define VHOST_FORK_OWNER_KTHREAD 0 ++#define VHOST_FORK_OWNER_TASK 1 ++ ++/** ++ * VHOST_SET_FORK_FROM_OWNER - Set the fork_owner flag for the vhost device, ++ * This ioctl must called before VHOST_SET_OWNER. ++ * Only available when CONFIG_VHOST_ENABLE_FORK_OWNER_CONTROL=y ++ * ++ * @param fork_owner: An 8-bit value that determines the vhost thread mode ++ * ++ * When fork_owner is set to VHOST_FORK_OWNER_TASK(default value): ++ * - Vhost will create vhost worker as tasks forked from the owner, ++ * inheriting all of the owner's attributes. ++ * ++ * When fork_owner is set to VHOST_FORK_OWNER_KTHREAD: ++ * - Vhost will create vhost workers as kernel threads. ++ */ ++#define VHOST_SET_FORK_FROM_OWNER _IOW(VHOST_VIRTIO, 0x83, __u8) ++ ++/** ++ * VHOST_GET_FORK_OWNER - Get the current fork_owner flag for the vhost device. ++ * Only available when CONFIG_VHOST_ENABLE_FORK_OWNER_CONTROL=y ++ * ++ * @return: An 8-bit value indicating the current thread mode. ++ */ ++#define VHOST_GET_FORK_FROM_OWNER _IOR(VHOST_VIRTIO, 0x84, __u8) ++ + #endif +-- +2.47.3 + diff --git a/kvm-linux-headers-deal-with-counted_by-annotation.patch b/kvm-linux-headers-deal-with-counted_by-annotation.patch new file mode 100644 index 0000000..eb0be84 --- /dev/null +++ b/kvm-linux-headers-deal-with-counted_by-annotation.patch @@ -0,0 +1,47 @@ +From aa56db1a8eb6ede62e58ab4827fc8a7ba8a2ec4a Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:16 +0200 +Subject: [PATCH 07/19] linux-headers: deal with counted_by annotation + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [2/14] dacb8059a0de06a0b9665152e0725da7da08467e (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Such annotation is present into the kernel uAPI headers since +v6.7, and will be used soon by the vhost_type.h. Deal with it +just stripping it. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit c3d9dcd87f0d228ea3ac5a42076da829cff401f0) +Signed-off-by: Laurent Vivier +--- + scripts/update-linux-headers.sh | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/scripts/update-linux-headers.sh b/scripts/update-linux-headers.sh +index 828a7809f7..844d9cb9f5 100755 +--- a/scripts/update-linux-headers.sh ++++ b/scripts/update-linux-headers.sh +@@ -90,6 +90,7 @@ cp_portable() { + -e 's/]*\)>/"standard-headers\/linux\/\1"/' \ + -e "$arch_cmd" \ + -e 's/__bitwise//' \ ++ -e 's/__counted_by(\w*)//' \ + -e 's/__attribute__((packed))/QEMU_PACKED/' \ + -e 's/__inline__/inline/' \ + -e 's/__BITS_PER_LONG/HOST_LONG_BITS/' \ +-- +2.47.3 + diff --git a/kvm-linux-headers-linux-Add-mshv.h-headers.patch b/kvm-linux-headers-linux-Add-mshv.h-headers.patch new file mode 100644 index 0000000..1c389db --- /dev/null +++ b/kvm-linux-headers-linux-Add-mshv.h-headers.patch @@ -0,0 +1,329 @@ +From 38642c3d90746f63ecc1f961ac7ef52b019c1132 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:26 +0200 +Subject: [PATCH 08/32] linux-headers/linux: Add mshv.h headers +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [6/30] ef60621970d29f626815bdeb68bc8bdfe0061d0e + +This file has been added to the tree by running `update-linux-header.sh` +on linux v6.16. + +Signed-off-by: Magnus Kulke +Reviewed-by: Daniel P. Berrangé +Link: https://lore.kernel.org/r/20250916164847.77883-7-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit a6d6878650a0f4982918e8cd9d7ea6c3c3c681f7) +Signed-off-by: Igor Mammedov +--- + linux-headers/linux/mshv.h | 291 +++++++++++++++++++++++++++++++++++++ + 1 file changed, 291 insertions(+) + create mode 100644 linux-headers/linux/mshv.h + +diff --git a/linux-headers/linux/mshv.h b/linux-headers/linux/mshv.h +new file mode 100644 +index 0000000000..5bc83db6a3 +--- /dev/null ++++ b/linux-headers/linux/mshv.h +@@ -0,0 +1,291 @@ ++/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */ ++/* ++ * Userspace interfaces for /dev/mshv* devices and derived fds ++ * ++ * This file is divided into sections containing data structures and IOCTLs for ++ * a particular set of related devices or derived file descriptors. ++ * ++ * The IOCTL definitions are at the end of each section. They are grouped by ++ * device/fd, so that new IOCTLs can easily be added with a monotonically ++ * increasing number. ++ */ ++#ifndef _LINUX_MSHV_H ++#define _LINUX_MSHV_H ++ ++#include ++ ++#define MSHV_IOCTL 0xB8 ++ ++/* ++ ******************************************* ++ * Entry point to main VMM APIs: /dev/mshv * ++ ******************************************* ++ */ ++ ++enum { ++ MSHV_PT_BIT_LAPIC, ++ MSHV_PT_BIT_X2APIC, ++ MSHV_PT_BIT_GPA_SUPER_PAGES, ++ MSHV_PT_BIT_COUNT, ++}; ++ ++#define MSHV_PT_FLAGS_MASK ((1 << MSHV_PT_BIT_COUNT) - 1) ++ ++enum { ++ MSHV_PT_ISOLATION_NONE, ++ MSHV_PT_ISOLATION_COUNT, ++}; ++ ++/** ++ * struct mshv_create_partition - arguments for MSHV_CREATE_PARTITION ++ * @pt_flags: Bitmask of 1 << MSHV_PT_BIT_* ++ * @pt_isolation: MSHV_PT_ISOLATION_* ++ * ++ * Returns a file descriptor to act as a handle to a guest partition. ++ * At this point the partition is not yet initialized in the hypervisor. ++ * Some operations must be done with the partition in this state, e.g. setting ++ * so-called "early" partition properties. The partition can then be ++ * initialized with MSHV_INITIALIZE_PARTITION. ++ */ ++struct mshv_create_partition { ++ __u64 pt_flags; ++ __u64 pt_isolation; ++}; ++ ++/* /dev/mshv */ ++#define MSHV_CREATE_PARTITION _IOW(MSHV_IOCTL, 0x00, struct mshv_create_partition) ++ ++/* ++ ************************ ++ * Child partition APIs * ++ ************************ ++ */ ++ ++struct mshv_create_vp { ++ __u32 vp_index; ++}; ++ ++enum { ++ MSHV_SET_MEM_BIT_WRITABLE, ++ MSHV_SET_MEM_BIT_EXECUTABLE, ++ MSHV_SET_MEM_BIT_UNMAP, ++ MSHV_SET_MEM_BIT_COUNT ++}; ++ ++#define MSHV_SET_MEM_FLAGS_MASK ((1 << MSHV_SET_MEM_BIT_COUNT) - 1) ++ ++/* The hypervisor's "native" page size */ ++#define MSHV_HV_PAGE_SIZE 0x1000 ++ ++/** ++ * struct mshv_user_mem_region - arguments for MSHV_SET_GUEST_MEMORY ++ * @size: Size of the memory region (bytes). Must be aligned to ++ * MSHV_HV_PAGE_SIZE ++ * @guest_pfn: Base guest page number to map ++ * @userspace_addr: Base address of userspace memory. Must be aligned to ++ * MSHV_HV_PAGE_SIZE ++ * @flags: Bitmask of 1 << MSHV_SET_MEM_BIT_*. If (1 << MSHV_SET_MEM_BIT_UNMAP) ++ * is set, ignore other bits. ++ * @rsvd: MBZ ++ * ++ * Map or unmap a region of userspace memory to Guest Physical Addresses (GPA). ++ * Mappings can't overlap in GPA space or userspace. ++ * To unmap, these fields must match an existing mapping. ++ */ ++struct mshv_user_mem_region { ++ __u64 size; ++ __u64 guest_pfn; ++ __u64 userspace_addr; ++ __u8 flags; ++ __u8 rsvd[7]; ++}; ++ ++enum { ++ MSHV_IRQFD_BIT_DEASSIGN, ++ MSHV_IRQFD_BIT_RESAMPLE, ++ MSHV_IRQFD_BIT_COUNT, ++}; ++ ++#define MSHV_IRQFD_FLAGS_MASK ((1 << MSHV_IRQFD_BIT_COUNT) - 1) ++ ++struct mshv_user_irqfd { ++ __s32 fd; ++ __s32 resamplefd; ++ __u32 gsi; ++ __u32 flags; ++}; ++ ++enum { ++ MSHV_IOEVENTFD_BIT_DATAMATCH, ++ MSHV_IOEVENTFD_BIT_PIO, ++ MSHV_IOEVENTFD_BIT_DEASSIGN, ++ MSHV_IOEVENTFD_BIT_COUNT, ++}; ++ ++#define MSHV_IOEVENTFD_FLAGS_MASK ((1 << MSHV_IOEVENTFD_BIT_COUNT) - 1) ++ ++struct mshv_user_ioeventfd { ++ __u64 datamatch; ++ __u64 addr; /* legal pio/mmio address */ ++ __u32 len; /* 1, 2, 4, or 8 bytes */ ++ __s32 fd; ++ __u32 flags; ++ __u8 rsvd[4]; ++}; ++ ++struct mshv_user_irq_entry { ++ __u32 gsi; ++ __u32 address_lo; ++ __u32 address_hi; ++ __u32 data; ++}; ++ ++struct mshv_user_irq_table { ++ __u32 nr; ++ __u32 rsvd; /* MBZ */ ++ struct mshv_user_irq_entry entries[]; ++}; ++ ++enum { ++ MSHV_GPAP_ACCESS_TYPE_ACCESSED, ++ MSHV_GPAP_ACCESS_TYPE_DIRTY, ++ MSHV_GPAP_ACCESS_TYPE_COUNT /* Count of enum members */ ++}; ++ ++enum { ++ MSHV_GPAP_ACCESS_OP_NOOP, ++ MSHV_GPAP_ACCESS_OP_CLEAR, ++ MSHV_GPAP_ACCESS_OP_SET, ++ MSHV_GPAP_ACCESS_OP_COUNT /* Count of enum members */ ++}; ++ ++/** ++ * struct mshv_gpap_access_bitmap - arguments for MSHV_GET_GPAP_ACCESS_BITMAP ++ * @access_type: MSHV_GPAP_ACCESS_TYPE_* - The type of access to record in the ++ * bitmap ++ * @access_op: MSHV_GPAP_ACCESS_OP_* - Allows an optional clear or set of all ++ * the access states in the range, after retrieving the current ++ * states. ++ * @rsvd: MBZ ++ * @page_count: Number of pages ++ * @gpap_base: Base gpa page number ++ * @bitmap_ptr: Output buffer for bitmap, at least (page_count + 7) / 8 bytes ++ * ++ * Retrieve a bitmap of either ACCESSED or DIRTY bits for a given range of guest ++ * memory, and optionally clear or set the bits. ++ */ ++struct mshv_gpap_access_bitmap { ++ __u8 access_type; ++ __u8 access_op; ++ __u8 rsvd[6]; ++ __u64 page_count; ++ __u64 gpap_base; ++ __u64 bitmap_ptr; ++}; ++ ++/** ++ * struct mshv_root_hvcall - arguments for MSHV_ROOT_HVCALL ++ * @code: Hypercall code (HVCALL_*) ++ * @reps: in: Rep count ('repcount') ++ * out: Reps completed ('repcomp'). MBZ unless rep hvcall ++ * @in_sz: Size of input incl rep data. <= MSHV_HV_PAGE_SIZE ++ * @out_sz: Size of output buffer. <= MSHV_HV_PAGE_SIZE. MBZ if out_ptr is 0 ++ * @status: in: MBZ ++ * out: HV_STATUS_* from hypercall ++ * @rsvd: MBZ ++ * @in_ptr: Input data buffer (struct hv_input_*). If used with partition or ++ * vp fd, partition id field is populated by kernel. ++ * @out_ptr: Output data buffer (optional) ++ */ ++struct mshv_root_hvcall { ++ __u16 code; ++ __u16 reps; ++ __u16 in_sz; ++ __u16 out_sz; ++ __u16 status; ++ __u8 rsvd[6]; ++ __u64 in_ptr; ++ __u64 out_ptr; ++}; ++ ++/* Partition fds created with MSHV_CREATE_PARTITION */ ++#define MSHV_INITIALIZE_PARTITION _IO(MSHV_IOCTL, 0x00) ++#define MSHV_CREATE_VP _IOW(MSHV_IOCTL, 0x01, struct mshv_create_vp) ++#define MSHV_SET_GUEST_MEMORY _IOW(MSHV_IOCTL, 0x02, struct mshv_user_mem_region) ++#define MSHV_IRQFD _IOW(MSHV_IOCTL, 0x03, struct mshv_user_irqfd) ++#define MSHV_IOEVENTFD _IOW(MSHV_IOCTL, 0x04, struct mshv_user_ioeventfd) ++#define MSHV_SET_MSI_ROUTING _IOW(MSHV_IOCTL, 0x05, struct mshv_user_irq_table) ++#define MSHV_GET_GPAP_ACCESS_BITMAP _IOWR(MSHV_IOCTL, 0x06, struct mshv_gpap_access_bitmap) ++/* Generic hypercall */ ++#define MSHV_ROOT_HVCALL _IOWR(MSHV_IOCTL, 0x07, struct mshv_root_hvcall) ++ ++/* ++ ******************************** ++ * VP APIs for child partitions * ++ ******************************** ++ */ ++ ++#define MSHV_RUN_VP_BUF_SZ 256 ++ ++/* ++ * VP state pages may be mapped to userspace via mmap(). ++ * To specify which state page, use MSHV_VP_MMAP_OFFSET_ values multiplied by ++ * the system page size. ++ * e.g. ++ * long page_size = sysconf(_SC_PAGE_SIZE); ++ * void *reg_page = mmap(NULL, MSHV_HV_PAGE_SIZE, PROT_READ|PROT_WRITE, ++ * MAP_SHARED, vp_fd, ++ * MSHV_VP_MMAP_OFFSET_REGISTERS * page_size); ++ */ ++enum { ++ MSHV_VP_MMAP_OFFSET_REGISTERS, ++ MSHV_VP_MMAP_OFFSET_INTERCEPT_MESSAGE, ++ MSHV_VP_MMAP_OFFSET_GHCB, ++ MSHV_VP_MMAP_OFFSET_COUNT ++}; ++ ++/** ++ * struct mshv_run_vp - argument for MSHV_RUN_VP ++ * @msg_buf: On success, the intercept message is copied here. It can be ++ * interpreted using the relevant hypervisor definitions. ++ */ ++struct mshv_run_vp { ++ __u8 msg_buf[MSHV_RUN_VP_BUF_SZ]; ++}; ++ ++enum { ++ MSHV_VP_STATE_LAPIC, /* Local interrupt controller state (either arch) */ ++ MSHV_VP_STATE_XSAVE, /* XSAVE data in compacted form (x86_64) */ ++ MSHV_VP_STATE_SIMP, ++ MSHV_VP_STATE_SIEFP, ++ MSHV_VP_STATE_SYNTHETIC_TIMERS, ++ MSHV_VP_STATE_COUNT, ++}; ++ ++/** ++ * struct mshv_get_set_vp_state - arguments for MSHV_[GET,SET]_VP_STATE ++ * @type: MSHV_VP_STATE_* ++ * @rsvd: MBZ ++ * @buf_sz: in: 4k page-aligned size of buffer ++ * out: Actual size of data (on EINVAL, check this to see if buffer ++ * was too small) ++ * @buf_ptr: 4k page-aligned data buffer ++ */ ++struct mshv_get_set_vp_state { ++ __u8 type; ++ __u8 rsvd[3]; ++ __u32 buf_sz; ++ __u64 buf_ptr; ++}; ++ ++/* VP fds created with MSHV_CREATE_VP */ ++#define MSHV_RUN_VP _IOR(MSHV_IOCTL, 0x00, struct mshv_run_vp) ++#define MSHV_GET_VP_STATE _IOWR(MSHV_IOCTL, 0x01, struct mshv_get_set_vp_state) ++#define MSHV_SET_VP_STATE _IOWR(MSHV_IOCTL, 0x02, struct mshv_get_set_vp_state) ++/* ++ * Generic hypercall ++ * Defined above in partition IOCTLs, avoid redefining it here ++ * #define MSHV_ROOT_HVCALL _IOWR(MSHV_IOCTL, 0x07, struct mshv_root_hvcall) ++ */ ++ ++#endif +-- +2.47.3 + diff --git a/kvm-mirror-Fix-missed-dirty-bitmap-writes-during-startup.patch b/kvm-mirror-Fix-missed-dirty-bitmap-writes-during-startup.patch new file mode 100644 index 0000000..9adf132 --- /dev/null +++ b/kvm-mirror-Fix-missed-dirty-bitmap-writes-during-startup.patch @@ -0,0 +1,162 @@ +From 19a86e8f7e88b47ac40014cd716146f0754939c2 Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Thu, 19 Feb 2026 21:24:46 +0100 +Subject: [PATCH] mirror: Fix missed dirty bitmap writes during startup + +RH-Author: Kevin Wolf +RH-MergeRequest: 474: mirror: Fix missed dirty bitmap writes during startup +RH-Jira: RHEL-155601 +RH-Acked-by: Hanna Czenczek +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [1/1] a888c4b2085d5fc3639bf6721f1f4e974db65dba (kmwolf/centos-qemu-kvm) + +Currently, mirror disables the block layer's dirty bitmap before its own +replacement is working. This means that during startup, there is a +window in which the allocation status of blocks in the source has +already been checked, but new writes coming in aren't tracked yet, +resulting in a corrupted copy: + +1. Dirty bitmap is disabled in mirror_start_job() +2. Some request are started in mirror_top_bs while s->job == NULL +3. mirror_dirty_init() -> bdrv_co_is_allocated_above() runs and because + the request hasn't completed yet, the block isn't allocated +4. The request completes, still sees s->job == NULL and skips the + bitmap, and nothing else will mark it dirty either + +One ingredient is that mirror_top_opaque->job is only set after the +job is fully initialized. For the rationale, see commit 32125b1460 +("mirror: Fix access of uninitialised fields during start"). + +Fix this by giving mirror_top_bs access to dirty_bitmap and enabling it +to track writes from the beginning. Disabling the block layer's tracking +and enabling the mirror_top_bs one happens in a drained section, so +there is no danger of races with in-flight requests any more. All of +this happens well before the block allocation status is checked, so we +can be sure that no writes will be missed. + +Cc: qemu-stable@nongnu.org +Closes: https://gitlab.com/qemu-project/qemu/-/issues/3273 +Fixes: 32125b14606a ('mirror: Fix access of uninitialised fields during start') +Signed-off-by: Kevin Wolf +Message-ID: <20260219202446.312493-1-kwolf@redhat.com> +Reviewed-by: Fiona Ebner +Tested-by: Jean-Louis Dupond +Signed-off-by: Kevin Wolf +(cherry picked from commit 0f51f9c3420b31bb383e456dd7bf24d3056eeb73) +Signed-off-by: Kevin Wolf +--- + block/mirror.c | 52 +++++++++++++++++++++++++++++++------------------- + 1 file changed, 32 insertions(+), 20 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index b344182c74..f01be99b55 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -99,6 +99,7 @@ typedef struct MirrorBlockJob { + + typedef struct MirrorBDSOpaque { + MirrorBlockJob *job; ++ BdrvDirtyBitmap *dirty_bitmap; + bool stop; + bool is_commit; + } MirrorBDSOpaque; +@@ -1672,9 +1673,11 @@ bdrv_mirror_top_do_write(BlockDriverState *bs, MirrorMethod method, + abort(); + } + +- if (!copy_to_target && s->job && s->job->dirty_bitmap) { +- qatomic_set(&s->job->actively_synced, false); +- bdrv_set_dirty_bitmap(s->job->dirty_bitmap, offset, bytes); ++ if (!copy_to_target) { ++ if (s->job) { ++ qatomic_set(&s->job->actively_synced, false); ++ } ++ bdrv_set_dirty_bitmap(s->dirty_bitmap, offset, bytes); + } + + if (ret < 0) { +@@ -1901,13 +1904,35 @@ static BlockJob *mirror_start_job( + + bdrv_drained_begin(bs); + ret = bdrv_append(mirror_top_bs, bs, errp); +- bdrv_drained_end(bs); +- + if (ret < 0) { ++ bdrv_drained_end(bs); ++ bdrv_unref(mirror_top_bs); ++ return NULL; ++ } ++ ++ bs_opaque->dirty_bitmap = bdrv_create_dirty_bitmap(mirror_top_bs, ++ granularity, ++ NULL, errp); ++ if (!bs_opaque->dirty_bitmap) { ++ bdrv_drained_end(bs); + bdrv_unref(mirror_top_bs); + return NULL; + } + ++ /* ++ * The mirror job doesn't use the block layer's dirty tracking because it ++ * needs to be able to switch seemlessly between background copy mode (which ++ * does need dirty tracking) and write blocking mode (which doesn't) and ++ * doing that would require draining the node. Instead, mirror_top_bs takes ++ * care of updating the dirty bitmap as appropriate. ++ * ++ * Note that write blocking mode only becomes effective after mirror_run() ++ * sets mirror_top_opaque->job (see should_copy_to_target()). Until then, ++ * we're still in background copy mode irrespective of @copy_mode. ++ */ ++ bdrv_disable_dirty_bitmap(bs_opaque->dirty_bitmap); ++ bdrv_drained_end(bs); ++ + /* Make sure that the source is not resized while the job is running */ + s = block_job_create(job_id, driver, NULL, mirror_top_bs, + BLK_PERM_CONSISTENT_READ, +@@ -2002,24 +2027,13 @@ static BlockJob *mirror_start_job( + s->base_overlay = bdrv_find_overlay(bs, base); + s->granularity = granularity; + s->buf_size = ROUND_UP(buf_size, granularity); ++ s->dirty_bitmap = bs_opaque->dirty_bitmap; + s->unmap = unmap; + if (auto_complete) { + s->should_complete = true; + } + bdrv_graph_rdunlock_main_loop(); + +- s->dirty_bitmap = bdrv_create_dirty_bitmap(s->mirror_top_bs, granularity, +- NULL, errp); +- if (!s->dirty_bitmap) { +- goto fail; +- } +- +- /* +- * The dirty bitmap is set by bdrv_mirror_top_do_write() when not in active +- * mode. +- */ +- bdrv_disable_dirty_bitmap(s->dirty_bitmap); +- + bdrv_graph_wrlock_drained(); + ret = block_job_add_bdrv(&s->common, "source", bs, 0, + BLK_PERM_WRITE_UNCHANGED | BLK_PERM_WRITE | +@@ -2099,9 +2113,6 @@ fail: + g_free(s->replaces); + blk_unref(s->target); + bs_opaque->job = NULL; +- if (s->dirty_bitmap) { +- bdrv_release_dirty_bitmap(s->dirty_bitmap); +- } + job_early_fail(&s->common.job); + } + +@@ -2115,6 +2126,7 @@ fail: + bdrv_graph_wrunlock(); + bdrv_drained_end(bs); + ++ bdrv_release_dirty_bitmap(bs_opaque->dirty_bitmap); + bdrv_unref(mirror_top_bs); + + return NULL; +-- +2.47.3 + diff --git a/kvm-monitor-generalize-query-mshv-info-mshv-to-query-acc.patch b/kvm-monitor-generalize-query-mshv-info-mshv-to-query-acc.patch new file mode 100644 index 0000000..3f8cbf7 --- /dev/null +++ b/kvm-monitor-generalize-query-mshv-info-mshv-to-query-acc.patch @@ -0,0 +1,235 @@ +From afb70cb5eea1e9dad7b5414713d2394ef2c88329 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Mon, 13 Oct 2025 12:49:04 +0200 +Subject: [PATCH 1/6] monitor: generalize query-mshv/"info mshv" to + query-accelerators/"info accelerators" +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Igor Mammedov +RH-MergeRequest: 443: [REHL10.2] L1HV: monitor: generalize query-mshv/"info mshv" to query-accelerators/"info accelerators" +RH-Jira: RHEL-134212 +RH-Acked-by: MST +RH-Acked-by: Paolo Bonzini +RH-Commit: [1/1] af946c572c3dab34cee6399bf4438e2722dd7e44 (imammedo/qemu-kvm-cs) + +The recently-introduced query-mshv command is a duplicate of query-kvm, +and neither provides a full view of which accelerators are supported +by a particular binary of QEMU and which is in use. + +KVM was the first accelerator added to QEMU, predating QOM and TYPE_ACCEL, +so it got a pass. But now, instead of adding a badly designed copy, solve +the problem completely for all accelerators with a command that provides +the whole picture: + + >> {"execute": "query-accelerators"} + << {"return": {"enabled": "tcg", "present": ["kvm", "mshv", "qtest", "tcg", "xen"]}} + +Cc: Praveen K Paladugu +Cc: Magnus Kulke +Suggested-by: Markus Armbruster +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Paolo Bonzini +(cherry picked from commit 71d5babbd6fffc7def1ecbf29f9753e3a2807761) +Signed-off-by: Igor Mammedov +--- + hmp-commands-info.hx | 15 ++++++++---- + hw/core/machine-hmp-cmds.c | 23 +++++++++++-------- + hw/core/machine-qmp-cmds.c | 24 +++++++++++++------ + include/monitor/hmp.h | 2 +- + qapi/accelerator.json | 47 +++++++++++++++++++++++++++++--------- + 5 files changed, 77 insertions(+), 34 deletions(-) + +diff --git a/hmp-commands-info.hx b/hmp-commands-info.hx +index eaaa880c1b..3ed636e4e8 100644 +--- a/hmp-commands-info.hx ++++ b/hmp-commands-info.hx +@@ -308,16 +308,21 @@ SRST + ERST + + { +- .name = "mshv", ++ .name = "accelerators", + .args_type = "", + .params = "", +- .help = "show MSHV information", +- .cmd = hmp_info_mshv, ++ .help = "show present and enabled information", ++ .cmd = hmp_info_accelerators, + }, + + SRST +- ``info mshv`` +- Show MSHV information. ++ ``info accelerators`` ++ Show which accelerators are compiled into a QEMU binary, and what accelerator ++ is in use. For example:: ++ ++ kvm qtest [tcg] ++ ++ indicates that TCG in use, and that KVM and qtest are also available. + ERST + + { +diff --git a/hw/core/machine-hmp-cmds.c b/hw/core/machine-hmp-cmds.c +index 682ed9f49b..74a56600be 100644 +--- a/hw/core/machine-hmp-cmds.c ++++ b/hw/core/machine-hmp-cmds.c +@@ -163,19 +163,22 @@ void hmp_info_kvm(Monitor *mon, const QDict *qdict) + qapi_free_KvmInfo(info); + } + +-void hmp_info_mshv(Monitor *mon, const QDict *qdict) ++void hmp_info_accelerators(Monitor *mon, const QDict *qdict) + { +- MshvInfo *info; +- +- info = qmp_query_mshv(NULL); +- monitor_printf(mon, "mshv support: "); +- if (info->present) { +- monitor_printf(mon, "%s\n", info->enabled ? "enabled" : "disabled"); +- } else { +- monitor_printf(mon, "not compiled\n"); ++ AcceleratorInfo *info; ++ AcceleratorList *accel; ++ ++ info = qmp_query_accelerators(NULL); ++ for (accel = info->present; accel; accel = accel->next) { ++ char trail = accel->next ? ' ' : '\n'; ++ if (info->enabled == accel->value) { ++ monitor_printf(mon, "[%s]%c", Accelerator_str(accel->value), trail); ++ } else { ++ monitor_printf(mon, "%s%c", Accelerator_str(accel->value), trail); ++ } + } + +- qapi_free_MshvInfo(info); ++ qapi_free_AcceleratorInfo(info); + } + + void hmp_info_uuid(Monitor *mon, const QDict *qdict) +diff --git a/hw/core/machine-qmp-cmds.c b/hw/core/machine-qmp-cmds.c +index e24bf0d97b..51d5c230f7 100644 +--- a/hw/core/machine-qmp-cmds.c ++++ b/hw/core/machine-qmp-cmds.c +@@ -31,15 +31,25 @@ + #include + + /* +- * QMP query for MSHV ++ * QMP query for enabled and present accelerators + */ +-MshvInfo *qmp_query_mshv(Error **errp) ++AcceleratorInfo *qmp_query_accelerators(Error **errp) + { +- MshvInfo *info = g_malloc0(sizeof(*info)); +- +- info->enabled = mshv_enabled(); +- info->present = accel_find("mshv"); +- ++ AcceleratorInfo *info = g_malloc0(sizeof(*info)); ++ AccelClass *current_class = ACCEL_GET_CLASS(current_accel()); ++ int i; ++ ++ for (i = ACCELERATOR__MAX; i-- > 0; ) { ++ const char *s = Accelerator_str(i); ++ AccelClass *this_class = accel_find(s); ++ ++ if (this_class) { ++ QAPI_LIST_PREPEND(info->present, i); ++ if (this_class == current_class) { ++ info->enabled = i; ++ } ++ } ++ } + return info; + } + +diff --git a/include/monitor/hmp.h b/include/monitor/hmp.h +index 31bd812e5f..897dfaa2b6 100644 +--- a/include/monitor/hmp.h ++++ b/include/monitor/hmp.h +@@ -24,7 +24,7 @@ strList *hmp_split_at_comma(const char *str); + void hmp_info_name(Monitor *mon, const QDict *qdict); + void hmp_info_version(Monitor *mon, const QDict *qdict); + void hmp_info_kvm(Monitor *mon, const QDict *qdict); +-void hmp_info_mshv(Monitor *mon, const QDict *qdict); ++void hmp_info_accelerators(Monitor *mon, const QDict *qdict); + void hmp_info_status(Monitor *mon, const QDict *qdict); + void hmp_info_uuid(Monitor *mon, const QDict *qdict); + void hmp_info_chardev(Monitor *mon, const QDict *qdict); +diff --git a/qapi/accelerator.json b/qapi/accelerator.json +index 664e027246..2b92060884 100644 +--- a/qapi/accelerator.json ++++ b/qapi/accelerator.json +@@ -56,30 +56,55 @@ + 'features': [ 'unstable' ] } + + ## +-# @MshvInfo: ++# @Accelerator: + # + # Information about support for MSHV acceleration + # +-# @enabled: true if MSHV acceleration is active ++# @hvf: Apple Hypervisor.framework + # +-# @present: true if MSHV acceleration is built into this executable ++# @kvm: KVM ++# ++# @mshv: Hyper-V ++# ++# @nvmm: NetBSD NVMM ++# ++# @qtest: QTest (dummy accelerator) ++# ++# @tcg: TCG (dynamic translation) ++# ++# @whpx: Windows Hypervisor Platform ++# ++# @xen: Xen ++# ++# Since: 10.2.0 ++## ++{ 'enum': 'Accelerator', 'data': ['hvf', 'kvm', 'mshv', 'nvmm', 'qtest', 'tcg', 'whpx', 'xen'] } ++ ++## ++# @AcceleratorInfo: ++# ++# Information about support for various accelerators ++# ++# @enabled: the accelerator that is in use ++# ++# @present: the list of accelerators that are built into this executable + # + # Since: 10.2.0 + ## +-{ 'struct': 'MshvInfo', 'data': {'enabled': 'bool', 'present': 'bool'} } ++{ 'struct': 'AcceleratorInfo', 'data': {'enabled': 'Accelerator', 'present': ['Accelerator']} } + + ## +-# @query-mshv: ++# @query-accelerators: + # +-# Return information about MSHV acceleration ++# Return information about accelerators + # +-# Returns: @MshvInfo ++# Returns: @AcceleratorInfo + # +-# Since: 10.0.92 ++# Since: 10.2.0 + # + # .. qmp-example:: + # +-# -> { "execute": "query-mshv" } +-# <- { "return": { "enabled": true, "present": true } } ++# -> { "execute": "query-accelerators" } ++# <- { "return": { "enabled": "mshv", "present": ["kvm", "mshv", "qtest", "tcg"] } } + ## +-{ 'command': 'query-mshv', 'returns': 'MshvInfo' } ++{ 'command': 'query-accelerators', 'returns': 'AcceleratorInfo' } +-- +2.47.3 + diff --git a/kvm-nbd-server-CVE-2024-7409-Avoid-use-after-free-when-c.patch b/kvm-nbd-server-CVE-2024-7409-Avoid-use-after-free-when-c.patch deleted file mode 100644 index 5b5b2e2..0000000 --- a/kvm-nbd-server-CVE-2024-7409-Avoid-use-after-free-when-c.patch +++ /dev/null @@ -1,101 +0,0 @@ -From 3732f1491d8981e85f699fcd125d903aba77fa32 Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Thu, 22 Aug 2024 09:35:29 -0500 -Subject: [PATCH] nbd/server: CVE-2024-7409: Avoid use-after-free when closing - server - -RH-Author: Eric Blake -RH-MergeRequest: 267: nbd/server: CVE-2024-7409: Avoid use-after-free when closing server -RH-Jira: RHEL-52599 -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Hanna Czenczek -RH-Commit: e7d52e5d1372eaec00325d4854772ee78fe650b7 (ebblake/centos-qemu-kvm) - -Commit 3e7ef738 plugged the use-after-free of the global nbd_server -object, but overlooked a use-after-free of nbd_server->listener. -Although this race is harder to hit, notice that our shutdown path -first drops the reference count of nbd_server->listener, then triggers -actions that can result in a pending client reaching the -nbd_blockdev_client_closed() callback, which in turn calls -qio_net_listener_set_client_func on a potentially stale object. - -If we know we don't want any more clients to connect, and have already -told the listener socket to shut down, then we should not be trying to -update the listener socket's associated function. - -Reproducer: - -> #!/usr/bin/python3 -> -> import os -> from threading import Thread -> -> def start_stop(): -> while 1: -> os.system('virsh qemu-monitor-command VM \'{"execute": "nbd-server-start", -+"arguments":{"addr":{"type":"unix","data":{"path":"/tmp/nbd-sock"}}}}\'') -> os.system('virsh qemu-monitor-command VM \'{"execute": "nbd-server-stop"}\'') -> -> def nbd_list(): -> while 1: -> os.system('/path/to/build/qemu-nbd -L -k /tmp/nbd-sock') -> -> def test(): -> sst = Thread(target=start_stop) -> sst.start() -> nlt = Thread(target=nbd_list) -> nlt.start() -> -> sst.join() -> nlt.join() -> -> test() - -Fixes: CVE-2024-7409 -Fixes: 3e7ef738c8 ("nbd/server: CVE-2024-7409: Close stray clients at server-stop") -CC: qemu-stable@nongnu.org -Reported-by: Andrey Drobyshev -Signed-off-by: Eric Blake -Message-ID: <20240822143617.800419-2-eblake@redhat.com> -Reviewed-by: Stefan Hajnoczi - -(cherry picked from commit 3874f5f73c441c52f1c699c848d463b0eda01e4c) -Jira: https://issues.redhat.com/browse/RHEL-52599 -Signed-off-by: Eric Blake ---- - blockdev-nbd.c | 12 ++++++++---- - 1 file changed, 8 insertions(+), 4 deletions(-) - -diff --git a/blockdev-nbd.c b/blockdev-nbd.c -index f73409ae49..b36f41b7c5 100644 ---- a/blockdev-nbd.c -+++ b/blockdev-nbd.c -@@ -92,10 +92,13 @@ static void nbd_accept(QIONetListener *listener, QIOChannelSocket *cioc, - - static void nbd_update_server_watch(NBDServerData *s) - { -- if (!s->max_connections || s->connections < s->max_connections) { -- qio_net_listener_set_client_func(s->listener, nbd_accept, NULL, NULL); -- } else { -- qio_net_listener_set_client_func(s->listener, NULL, NULL, NULL); -+ if (s->listener) { -+ if (!s->max_connections || s->connections < s->max_connections) { -+ qio_net_listener_set_client_func(s->listener, nbd_accept, NULL, -+ NULL); -+ } else { -+ qio_net_listener_set_client_func(s->listener, NULL, NULL, NULL); -+ } - } - } - -@@ -113,6 +116,7 @@ static void nbd_server_free(NBDServerData *server) - */ - qio_net_listener_disconnect(server->listener); - object_unref(OBJECT(server->listener)); -+ server->listener = NULL; - QLIST_FOREACH_SAFE(conn, &server->conns, next, tmp) { - qio_channel_shutdown(QIO_CHANNEL(conn->cioc), QIO_CHANNEL_SHUTDOWN_BOTH, - NULL); --- -2.39.3 - diff --git a/kvm-nbd-server-CVE-2024-7409-Cap-default-max-connections.patch b/kvm-nbd-server-CVE-2024-7409-Cap-default-max-connections.patch deleted file mode 100644 index f432ce3..0000000 --- a/kvm-nbd-server-CVE-2024-7409-Cap-default-max-connections.patch +++ /dev/null @@ -1,184 +0,0 @@ -From 20b179691fcd3a58aaf76269e66bd102dfbd0d2e Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Tue, 6 Aug 2024 13:53:00 -0500 -Subject: [PATCH 3/5] nbd/server: CVE-2024-7409: Cap default max-connections to - 100 -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Eric Blake -RH-MergeRequest: 263: nbd/server: fix CVE-2024-7409 (qemu crash on nbd-server-stop) [RHEL 10.0] -RH-Jira: RHEL-52599 -RH-Acked-by: Miroslav Rezanina -RH-Commit: [2/4] ad547c43ee9bae4cf6476408176aa7a7892427ff (redhat/centos-stream/src/qemu-kvm) - -Allowing an unlimited number of clients to any web service is a recipe -for a rudimentary denial of service attack: the client merely needs to -open lots of sockets without closing them, until qemu no longer has -any more fds available to allocate. - -For qemu-nbd, we default to allowing only 1 connection unless more are -explicitly asked for (-e or --shared); this was historically picked as -a nice default (without an explicit -t, a non-persistent qemu-nbd goes -away after a client disconnects, without needing any additional -follow-up commands), and we are not going to change that interface now -(besides, someday we want to point people towards qemu-storage-daemon -instead of qemu-nbd). - -But for qemu proper, and the newer qemu-storage-daemon, the QMP -nbd-server-start command has historically had a default of unlimited -number of connections, in part because unlike qemu-nbd it is -inherently persistent until nbd-server-stop. Allowing multiple client -sockets is particularly useful for clients that can take advantage of -MULTI_CONN (creating parallel sockets to increase throughput), -although known clients that do so (such as libnbd's nbdcopy) typically -use only 8 or 16 connections (the benefits of scaling diminish once -more sockets are competing for kernel attention). Picking a number -large enough for typical use cases, but not unlimited, makes it -slightly harder for a malicious client to perform a denial of service -merely by opening lots of connections withot progressing through the -handshake. - -This change does not eliminate CVE-2024-7409 on its own, but reduces -the chance for fd exhaustion or unlimited memory usage as an attack -surface. On the other hand, by itself, it makes it more obvious that -with a finite limit, we have the problem of an unauthenticated client -holding 100 fds opened as a way to block out a legitimate client from -being able to connect; thus, later patches will further add timeouts -to reject clients that are not making progress. - -This is an INTENTIONAL change in behavior, and will break any client -of nbd-server-start that was not passing an explicit max-connections -parameter, yet expects more than 100 simultaneous connections. We are -not aware of any such client (as stated above, most clients aware of -MULTI_CONN get by just fine on 8 or 16 connections, and probably cope -with later connections failing by relying on the earlier connections; -libvirt has not yet been passing max-connections, but generally -creates NBD servers with the intent for a single client for the sake -of live storage migration; meanwhile, the KubeSAN project anticipates -a large cluster sharing multiple clients [up to 8 per node, and up to -100 nodes in a cluster], but it currently uses qemu-nbd with an -explicit --shared=0 rather than qemu-storage-daemon with -nbd-server-start). - -We considered using a deprecation period (declare that omitting -max-parameters is deprecated, and make it mandatory in 3 releases - -then we don't need to pick an arbitrary default); that has zero risk -of breaking any apps that accidentally depended on more than 100 -connections, and where such breakage might not be noticed under unit -testing but only under the larger loads of production usage. But it -does not close the denial-of-service hole until far into the future, -and requires all apps to change to add the parameter even if 100 was -good enough. It also has a drawback that any app (like libvirt) that -is accidentally relying on an unlimited default should seriously -consider their own CVE now, at which point they are going to change to -pass explicit max-connections sooner than waiting for 3 qemu releases. -Finally, if our changed default breaks an app, that app can always -pass in an explicit max-parameters with a larger value. - -It is also intentional that the HMP interface to nbd-server-start is -not changed to expose max-connections (any client needing to fine-tune -things should be using QMP). - -Suggested-by: Daniel P. Berrangé -Signed-off-by: Eric Blake -Message-ID: <20240807174943.771624-12-eblake@redhat.com> -Reviewed-by: Daniel P. Berrangé -[ericb: Expand commit message to summarize Dan's argument for why we -break corner-case back-compat behavior without a deprecation period] -Signed-off-by: Eric Blake - -(cherry picked from commit c8a76dbd90c2f48df89b75bef74917f90a59b623) -Jira: https://issues.redhat.com/browse/RHEL-52599 -Signed-off-by: Eric Blake ---- - block/monitor/block-hmp-cmds.c | 3 ++- - blockdev-nbd.c | 8 ++++++++ - include/block/nbd.h | 7 +++++++ - qapi/block-export.json | 4 ++-- - 4 files changed, 19 insertions(+), 3 deletions(-) - -diff --git a/block/monitor/block-hmp-cmds.c b/block/monitor/block-hmp-cmds.c -index d954bec6f1..bdf2eb50b6 100644 ---- a/block/monitor/block-hmp-cmds.c -+++ b/block/monitor/block-hmp-cmds.c -@@ -402,7 +402,8 @@ void hmp_nbd_server_start(Monitor *mon, const QDict *qdict) - goto exit; - } - -- nbd_server_start(addr, NULL, NULL, 0, &local_err); -+ nbd_server_start(addr, NULL, NULL, NBD_DEFAULT_MAX_CONNECTIONS, -+ &local_err); - qapi_free_SocketAddress(addr); - if (local_err != NULL) { - goto exit; -diff --git a/blockdev-nbd.c b/blockdev-nbd.c -index 267a1de903..24ba5382db 100644 ---- a/blockdev-nbd.c -+++ b/blockdev-nbd.c -@@ -170,6 +170,10 @@ void nbd_server_start(SocketAddress *addr, const char *tls_creds, - - void nbd_server_start_options(NbdServerOptions *arg, Error **errp) - { -+ if (!arg->has_max_connections) { -+ arg->max_connections = NBD_DEFAULT_MAX_CONNECTIONS; -+ } -+ - nbd_server_start(arg->addr, arg->tls_creds, arg->tls_authz, - arg->max_connections, errp); - } -@@ -182,6 +186,10 @@ void qmp_nbd_server_start(SocketAddressLegacy *addr, - { - SocketAddress *addr_flat = socket_address_flatten(addr); - -+ if (!has_max_connections) { -+ max_connections = NBD_DEFAULT_MAX_CONNECTIONS; -+ } -+ - nbd_server_start(addr_flat, tls_creds, tls_authz, max_connections, errp); - qapi_free_SocketAddress(addr_flat); - } -diff --git a/include/block/nbd.h b/include/block/nbd.h -index 1d4d65922d..d4f8b21aec 100644 ---- a/include/block/nbd.h -+++ b/include/block/nbd.h -@@ -39,6 +39,13 @@ extern const BlockExportDriver blk_exp_nbd; - */ - #define NBD_DEFAULT_HANDSHAKE_MAX_SECS 10 - -+/* -+ * NBD_DEFAULT_MAX_CONNECTIONS: Number of client sockets to allow at -+ * once; must be large enough to allow a MULTI_CONN-aware client like -+ * nbdcopy to create its typical number of 8-16 sockets. -+ */ -+#define NBD_DEFAULT_MAX_CONNECTIONS 100 -+ - /* Handshake phase structs - this struct is passed on the wire */ - - typedef struct NBDOption { -diff --git a/qapi/block-export.json b/qapi/block-export.json -index 3919a2d5b9..f45e4fd481 100644 ---- a/qapi/block-export.json -+++ b/qapi/block-export.json -@@ -28,7 +28,7 @@ - # @max-connections: The maximum number of connections to allow at the - # same time, 0 for unlimited. Setting this to 1 also stops the - # server from advertising multiple client support (since 5.2; --# default: 0) -+# default: 100) - # - # Since: 4.2 - ## -@@ -63,7 +63,7 @@ - # @max-connections: The maximum number of connections to allow at the - # same time, 0 for unlimited. Setting this to 1 also stops the - # server from advertising multiple client support (since 5.2; --# default: 0). -+# default: 100). - # - # Errors: - # - if the server is already running --- -2.39.3 - diff --git a/kvm-nbd-server-CVE-2024-7409-Close-stray-clients-at-serv.patch b/kvm-nbd-server-CVE-2024-7409-Close-stray-clients-at-serv.patch deleted file mode 100644 index 1053fc6..0000000 --- a/kvm-nbd-server-CVE-2024-7409-Close-stray-clients-at-serv.patch +++ /dev/null @@ -1,173 +0,0 @@ -From 1b4bf69b064815a41ac18ef7276ceab0b9e0eb5b Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Wed, 7 Aug 2024 12:23:13 -0500 -Subject: [PATCH 5/5] nbd/server: CVE-2024-7409: Close stray clients at - server-stop -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Eric Blake -RH-MergeRequest: 263: nbd/server: fix CVE-2024-7409 (qemu crash on nbd-server-stop) [RHEL 10.0] -RH-Jira: RHEL-52599 -RH-Acked-by: Miroslav Rezanina -RH-Commit: [4/4] 6c5c7b5daa2b450122e98eb08ade1e1db56d20ae (redhat/centos-stream/src/qemu-kvm) - -A malicious client can attempt to connect to an NBD server, and then -intentionally delay progress in the handshake, including if it does -not know the TLS secrets. Although the previous two patches reduce -this behavior by capping the default max-connections parameter and -killing slow clients, they did not eliminate the possibility of a -client waiting to close the socket until after the QMP nbd-server-stop -command is executed, at which point qemu would SEGV when trying to -dereference the NULL nbd_server global which is no longer present. -This amounts to a denial of service attack. Worse, if another NBD -server is started before the malicious client disconnects, I cannot -rule out additional adverse effects when the old client interferes -with the connection count of the new server (although the most likely -is a crash due to an assertion failure when checking -nbd_server->connections > 0). - -For environments without this patch, the CVE can be mitigated by -ensuring (such as via a firewall) that only trusted clients can -connect to an NBD server. Note that using frameworks like libvirt -that ensure that TLS is used and that nbd-server-stop is not executed -while any trusted clients are still connected will only help if there -is also no possibility for an untrusted client to open a connection -but then stall on the NBD handshake. - -Given the previous patches, it would be possible to guarantee that no -clients remain connected by having nbd-server-stop sleep for longer -than the default handshake deadline before finally freeing the global -nbd_server object, but that could make QMP non-responsive for a long -time. So intead, this patch fixes the problem by tracking all client -sockets opened while the server is running, and forcefully closing any -such sockets remaining without a completed handshake at the time of -nbd-server-stop, then waiting until the coroutines servicing those -sockets notice the state change. nbd-server-stop now has a second -AIO_WAIT_WHILE_UNLOCKED (the first is indirectly through the -blk_exp_close_all_type() that disconnects all clients that completed -handshakes), but forced socket shutdown is enough to progress the -coroutines and quickly tear down all clients before the server is -freed, thus finally fixing the CVE. - -This patch relies heavily on the fact that nbd/server.c guarantees -that it only calls nbd_blockdev_client_closed() from the main loop -(see the assertion in nbd_client_put() and the hoops used in -nbd_client_put_nonzero() to achieve that); if we did not have that -guarantee, we would also need a mutex protecting our accesses of the -list of connections to survive re-entrancy from independent iothreads. - -Although I did not actually try to test old builds, it looks like this -problem has existed since at least commit 862172f45c (v2.12.0, 2017) - -even back when that patch started using a QIONetListener to handle -listening on multiple sockets, nbd_server_free() was already unaware -that the nbd_blockdev_client_closed callback can be reached later by a -client thread that has not completed handshakes (and therefore the -client's socket never got added to the list closed in -nbd_export_close_all), despite that patch intentionally tearing down -the QIONetListener to prevent new clients. - -Reported-by: Alexander Ivanov -Fixes: CVE-2024-7409 -CC: qemu-stable@nongnu.org -Signed-off-by: Eric Blake -Message-ID: <20240807174943.771624-14-eblake@redhat.com> -Reviewed-by: Daniel P. Berrangé - -(cherry picked from commit 3e7ef738c8462c45043a1d39f702a0990406a3b3) -Jira: https://issues.redhat.com/browse/RHEL-52599 -Signed-off-by: Eric Blake ---- - blockdev-nbd.c | 35 ++++++++++++++++++++++++++++++++++- - 1 file changed, 34 insertions(+), 1 deletion(-) - -diff --git a/blockdev-nbd.c b/blockdev-nbd.c -index 24ba5382db..f73409ae49 100644 ---- a/blockdev-nbd.c -+++ b/blockdev-nbd.c -@@ -21,12 +21,18 @@ - #include "io/channel-socket.h" - #include "io/net-listener.h" - -+typedef struct NBDConn { -+ QIOChannelSocket *cioc; -+ QLIST_ENTRY(NBDConn) next; -+} NBDConn; -+ - typedef struct NBDServerData { - QIONetListener *listener; - QCryptoTLSCreds *tlscreds; - char *tlsauthz; - uint32_t max_connections; - uint32_t connections; -+ QLIST_HEAD(, NBDConn) conns; - } NBDServerData; - - static NBDServerData *nbd_server; -@@ -51,6 +57,14 @@ int nbd_server_max_connections(void) - - static void nbd_blockdev_client_closed(NBDClient *client, bool ignored) - { -+ NBDConn *conn = nbd_client_owner(client); -+ -+ assert(qemu_in_main_thread() && nbd_server); -+ -+ object_unref(OBJECT(conn->cioc)); -+ QLIST_REMOVE(conn, next); -+ g_free(conn); -+ - nbd_client_put(client); - assert(nbd_server->connections > 0); - nbd_server->connections--; -@@ -60,14 +74,20 @@ static void nbd_blockdev_client_closed(NBDClient *client, bool ignored) - static void nbd_accept(QIONetListener *listener, QIOChannelSocket *cioc, - gpointer opaque) - { -+ NBDConn *conn = g_new0(NBDConn, 1); -+ -+ assert(qemu_in_main_thread() && nbd_server); - nbd_server->connections++; -+ object_ref(OBJECT(cioc)); -+ conn->cioc = cioc; -+ QLIST_INSERT_HEAD(&nbd_server->conns, conn, next); - nbd_update_server_watch(nbd_server); - - qio_channel_set_name(QIO_CHANNEL(cioc), "nbd-server"); - /* TODO - expose handshake timeout as QMP option */ - nbd_client_new(cioc, NBD_DEFAULT_HANDSHAKE_MAX_SECS, - nbd_server->tlscreds, nbd_server->tlsauthz, -- nbd_blockdev_client_closed, NULL); -+ nbd_blockdev_client_closed, conn); - } - - static void nbd_update_server_watch(NBDServerData *s) -@@ -81,12 +101,25 @@ static void nbd_update_server_watch(NBDServerData *s) - - static void nbd_server_free(NBDServerData *server) - { -+ NBDConn *conn, *tmp; -+ - if (!server) { - return; - } - -+ /* -+ * Forcefully close the listener socket, and any clients that have -+ * not yet disconnected on their own. -+ */ - qio_net_listener_disconnect(server->listener); - object_unref(OBJECT(server->listener)); -+ QLIST_FOREACH_SAFE(conn, &server->conns, next, tmp) { -+ qio_channel_shutdown(QIO_CHANNEL(conn->cioc), QIO_CHANNEL_SHUTDOWN_BOTH, -+ NULL); -+ } -+ -+ AIO_WAIT_WHILE_UNLOCKED(NULL, server->connections > 0); -+ - if (server->tlscreds) { - object_unref(OBJECT(server->tlscreds)); - } --- -2.39.3 - diff --git a/kvm-nbd-server-CVE-2024-7409-Drop-non-negotiating-client.patch b/kvm-nbd-server-CVE-2024-7409-Drop-non-negotiating-client.patch deleted file mode 100644 index 5f162e4..0000000 --- a/kvm-nbd-server-CVE-2024-7409-Drop-non-negotiating-client.patch +++ /dev/null @@ -1,134 +0,0 @@ -From 97012ea86a4a0a28fef68e43b989d858c8392e2a Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Thu, 8 Aug 2024 16:05:08 -0500 -Subject: [PATCH 4/5] nbd/server: CVE-2024-7409: Drop non-negotiating clients -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Eric Blake -RH-MergeRequest: 263: nbd/server: fix CVE-2024-7409 (qemu crash on nbd-server-stop) [RHEL 10.0] -RH-Jira: RHEL-52599 -RH-Acked-by: Miroslav Rezanina -RH-Commit: [3/4] c3dad94d423d2f431d1e605c412099b9fe0bd76e (redhat/centos-stream/src/qemu-kvm) - -A client that opens a socket but does not negotiate is merely hogging -qemu's resources (an open fd and a small amount of memory); and a -malicious client that can access the port where NBD is listening can -attempt a denial of service attack by intentionally opening and -abandoning lots of unfinished connections. The previous patch put a -default bound on the number of such ongoing connections, but once that -limit is hit, no more clients can connect (including legitimate ones). -The solution is to insist that clients complete handshake within a -reasonable time limit, defaulting to 10 seconds. A client that has -not successfully completed NBD_OPT_GO by then (including the case of -where the client didn't know TLS credentials to even reach the point -of NBD_OPT_GO) is wasting our time and does not deserve to stay -connected. Later patches will allow fine-tuning the limit away from -the default value (including disabling it for doing integration -testing of the handshake process itself). - -Note that this patch in isolation actually makes it more likely to see -qemu SEGV after nbd-server-stop, as any client socket still connected -when the server shuts down will now be closed after 10 seconds rather -than at the client's whims. That will be addressed in the next patch. - -For a demo of this patch in action: -$ qemu-nbd -f raw -r -t -e 10 file & -$ nbdsh --opt-mode -c ' -H = list() -for i in range(20): - print(i) - H.insert(i, nbd.NBD()) - H[i].set_opt_mode(True) - H[i].connect_uri("nbd://localhost") -' -$ kill $! - -where later connections get to start progressing once earlier ones are -forcefully dropped for taking too long, rather than hanging. - -Suggested-by: Daniel P. Berrangé -Signed-off-by: Eric Blake -Message-ID: <20240807174943.771624-13-eblake@redhat.com> -Reviewed-by: Daniel P. Berrangé -[eblake: rebase to changes earlier in series, reduce scope of timer] -Signed-off-by: Eric Blake - -(cherry picked from commit b9b72cb3ce15b693148bd09cef7e50110566d8a0) -Jira: https://issues.redhat.com/browse/RHEL-52599 -Signed-off-by: Eric Blake ---- - nbd/server.c | 28 +++++++++++++++++++++++++++- - nbd/trace-events | 1 + - 2 files changed, 28 insertions(+), 1 deletion(-) - -diff --git a/nbd/server.c b/nbd/server.c -index e50012499f..39285cc971 100644 ---- a/nbd/server.c -+++ b/nbd/server.c -@@ -3186,22 +3186,48 @@ static void nbd_client_receive_next_request(NBDClient *client) - } - } - -+static void nbd_handshake_timer_cb(void *opaque) -+{ -+ QIOChannel *ioc = opaque; -+ -+ trace_nbd_handshake_timer_cb(); -+ qio_channel_shutdown(ioc, QIO_CHANNEL_SHUTDOWN_BOTH, NULL); -+} -+ - static coroutine_fn void nbd_co_client_start(void *opaque) - { - NBDClient *client = opaque; - Error *local_err = NULL; -+ QEMUTimer *handshake_timer = NULL; - - qemu_co_mutex_init(&client->send_lock); - -- /* TODO - utilize client->handshake_max_secs */ -+ /* -+ * Create a timer to bound the time spent in negotiation. If the -+ * timer expires, it is likely nbd_negotiate will fail because the -+ * socket was shutdown. -+ */ -+ if (client->handshake_max_secs > 0) { -+ handshake_timer = aio_timer_new(qemu_get_aio_context(), -+ QEMU_CLOCK_REALTIME, -+ SCALE_NS, -+ nbd_handshake_timer_cb, -+ client->sioc); -+ timer_mod(handshake_timer, -+ qemu_clock_get_ns(QEMU_CLOCK_REALTIME) + -+ client->handshake_max_secs * NANOSECONDS_PER_SECOND); -+ } -+ - if (nbd_negotiate(client, &local_err)) { - if (local_err) { - error_report_err(local_err); - } -+ timer_free(handshake_timer); - client_close(client, false); - return; - } - -+ timer_free(handshake_timer); - WITH_QEMU_LOCK_GUARD(&client->lock) { - nbd_client_receive_next_request(client); - } -diff --git a/nbd/trace-events b/nbd/trace-events -index 00ae3216a1..cbd0a4ab7e 100644 ---- a/nbd/trace-events -+++ b/nbd/trace-events -@@ -76,6 +76,7 @@ nbd_co_receive_request_payload_received(uint64_t cookie, uint64_t len) "Payload - nbd_co_receive_ext_payload_compliance(uint64_t from, uint64_t len) "client sent non-compliant write without payload flag: from=0x%" PRIx64 ", len=0x%" PRIx64 - nbd_co_receive_align_compliance(const char *op, uint64_t from, uint64_t len, uint32_t align) "client sent non-compliant unaligned %s request: from=0x%" PRIx64 ", len=0x%" PRIx64 ", align=0x%" PRIx32 - nbd_trip(void) "Reading request" -+nbd_handshake_timer_cb(void) "client took too long to negotiate" - - # client-connection.c - nbd_connect_thread_sleep(uint64_t timeout) "timeout %" PRIu64 --- -2.39.3 - diff --git a/kvm-nbd-server-Mark-negotiation-functions-as-coroutine_f.patch b/kvm-nbd-server-Mark-negotiation-functions-as-coroutine_f.patch deleted file mode 100644 index c3cc85f..0000000 --- a/kvm-nbd-server-Mark-negotiation-functions-as-coroutine_f.patch +++ /dev/null @@ -1,330 +0,0 @@ -From 55e78a14c6a6956a3ac65f36b9b8b8c49eff959b Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Mon, 8 Apr 2024 11:00:44 -0500 -Subject: [PATCH 2/4] nbd/server: Mark negotiation functions as coroutine_fn - -RH-Author: Eric Blake -RH-MergeRequest: 257: nbd/server: fix TLS negotiation across coroutine context -RH-Jira: RHEL-40959 -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Miroslav Rezanina -RH-Commit: [2/4] f364e5cd2a9eac2d4f6af2841479f1dfb2f8df58 (ebblake/centos-qemu-kvm) - -nbd_negotiate() is already marked coroutine_fn. And given the fix in -the previous patch to have nbd_negotiate_handle_starttls not create -and wait on a g_main_loop (as that would violate coroutine -constraints), it is worth marking the rest of the related static -functions reachable only during option negotiation as also being -coroutine_fn. - -Suggested-by: Vladimir Sementsov-Ogievskiy -Signed-off-by: Eric Blake -Message-ID: <20240408160214.1200629-6-eblake@redhat.com> -Reviewed-by: Vladimir Sementsov-Ogievskiy -[eblake: drop one spurious coroutine_fn marking] -Signed-off-by: Eric Blake - -Jira: https://issues.redhat.com/browse/RHEL-40959 -(cherry picked from commit 4fa333e08dd96395a99ea8dd9e4c73a29dd23344) -Signed-off-by: Eric Blake ---- - nbd/server.c | 102 +++++++++++++++++++++++++++++---------------------- - 1 file changed, 59 insertions(+), 43 deletions(-) - -diff --git a/nbd/server.c b/nbd/server.c -index 98ae0e1632..892797bb11 100644 ---- a/nbd/server.c -+++ b/nbd/server.c -@@ -195,8 +195,9 @@ static inline void set_be_option_rep(NBDOptionReply *rep, uint32_t option, - - /* Send a reply header, including length, but no payload. - * Return -errno on error, 0 on success. */ --static int nbd_negotiate_send_rep_len(NBDClient *client, uint32_t type, -- uint32_t len, Error **errp) -+static coroutine_fn int -+nbd_negotiate_send_rep_len(NBDClient *client, uint32_t type, -+ uint32_t len, Error **errp) - { - NBDOptionReply rep; - -@@ -211,15 +212,15 @@ static int nbd_negotiate_send_rep_len(NBDClient *client, uint32_t type, - - /* Send a reply header with default 0 length. - * Return -errno on error, 0 on success. */ --static int nbd_negotiate_send_rep(NBDClient *client, uint32_t type, -- Error **errp) -+static coroutine_fn int -+nbd_negotiate_send_rep(NBDClient *client, uint32_t type, Error **errp) - { - return nbd_negotiate_send_rep_len(client, type, 0, errp); - } - - /* Send an error reply. - * Return -errno on error, 0 on success. */ --static int G_GNUC_PRINTF(4, 0) -+static coroutine_fn int G_GNUC_PRINTF(4, 0) - nbd_negotiate_send_rep_verr(NBDClient *client, uint32_t type, - Error **errp, const char *fmt, va_list va) - { -@@ -259,7 +260,7 @@ nbd_sanitize_name(const char *name) - - /* Send an error reply. - * Return -errno on error, 0 on success. */ --static int G_GNUC_PRINTF(4, 5) -+static coroutine_fn int G_GNUC_PRINTF(4, 5) - nbd_negotiate_send_rep_err(NBDClient *client, uint32_t type, - Error **errp, const char *fmt, ...) - { -@@ -275,7 +276,7 @@ nbd_negotiate_send_rep_err(NBDClient *client, uint32_t type, - /* Drop remainder of the current option, and send a reply with the - * given error type and message. Return -errno on read or write - * failure; or 0 if connection is still live. */ --static int G_GNUC_PRINTF(4, 0) -+static coroutine_fn int G_GNUC_PRINTF(4, 0) - nbd_opt_vdrop(NBDClient *client, uint32_t type, Error **errp, - const char *fmt, va_list va) - { -@@ -288,7 +289,7 @@ nbd_opt_vdrop(NBDClient *client, uint32_t type, Error **errp, - return ret; - } - --static int G_GNUC_PRINTF(4, 5) -+static coroutine_fn int G_GNUC_PRINTF(4, 5) - nbd_opt_drop(NBDClient *client, uint32_t type, Error **errp, - const char *fmt, ...) - { -@@ -302,7 +303,7 @@ nbd_opt_drop(NBDClient *client, uint32_t type, Error **errp, - return ret; - } - --static int G_GNUC_PRINTF(3, 4) -+static coroutine_fn int G_GNUC_PRINTF(3, 4) - nbd_opt_invalid(NBDClient *client, Error **errp, const char *fmt, ...) - { - int ret; -@@ -319,8 +320,9 @@ nbd_opt_invalid(NBDClient *client, Error **errp, const char *fmt, ...) - * If @check_nul, require that no NUL bytes appear in buffer. - * Return -errno on I/O error, 0 if option was completely handled by - * sending a reply about inconsistent lengths, or 1 on success. */ --static int nbd_opt_read(NBDClient *client, void *buffer, size_t size, -- bool check_nul, Error **errp) -+static coroutine_fn int -+nbd_opt_read(NBDClient *client, void *buffer, size_t size, -+ bool check_nul, Error **errp) - { - if (size > client->optlen) { - return nbd_opt_invalid(client, errp, -@@ -343,7 +345,8 @@ static int nbd_opt_read(NBDClient *client, void *buffer, size_t size, - /* Drop size bytes from the unparsed payload of the current option. - * Return -errno on I/O error, 0 if option was completely handled by - * sending a reply about inconsistent lengths, or 1 on success. */ --static int nbd_opt_skip(NBDClient *client, size_t size, Error **errp) -+static coroutine_fn int -+nbd_opt_skip(NBDClient *client, size_t size, Error **errp) - { - if (size > client->optlen) { - return nbd_opt_invalid(client, errp, -@@ -366,8 +369,9 @@ static int nbd_opt_skip(NBDClient *client, size_t size, Error **errp) - * Return -errno on I/O error, 0 if option was completely handled by - * sending a reply about inconsistent lengths, or 1 on success. - */ --static int nbd_opt_read_name(NBDClient *client, char **name, uint32_t *length, -- Error **errp) -+static coroutine_fn int -+nbd_opt_read_name(NBDClient *client, char **name, uint32_t *length, -+ Error **errp) - { - int ret; - uint32_t len; -@@ -402,8 +406,8 @@ static int nbd_opt_read_name(NBDClient *client, char **name, uint32_t *length, - - /* Send a single NBD_REP_SERVER reply to NBD_OPT_LIST, including payload. - * Return -errno on error, 0 on success. */ --static int nbd_negotiate_send_rep_list(NBDClient *client, NBDExport *exp, -- Error **errp) -+static coroutine_fn int -+nbd_negotiate_send_rep_list(NBDClient *client, NBDExport *exp, Error **errp) - { - ERRP_GUARD(); - size_t name_len, desc_len; -@@ -444,7 +448,8 @@ static int nbd_negotiate_send_rep_list(NBDClient *client, NBDExport *exp, - - /* Process the NBD_OPT_LIST command, with a potential series of replies. - * Return -errno on error, 0 on success. */ --static int nbd_negotiate_handle_list(NBDClient *client, Error **errp) -+static coroutine_fn int -+nbd_negotiate_handle_list(NBDClient *client, Error **errp) - { - NBDExport *exp; - assert(client->opt == NBD_OPT_LIST); -@@ -459,7 +464,8 @@ static int nbd_negotiate_handle_list(NBDClient *client, Error **errp) - return nbd_negotiate_send_rep(client, NBD_REP_ACK, errp); - } - --static void nbd_check_meta_export(NBDClient *client, NBDExport *exp) -+static coroutine_fn void -+nbd_check_meta_export(NBDClient *client, NBDExport *exp) - { - if (exp != client->contexts.exp) { - client->contexts.count = 0; -@@ -468,8 +474,9 @@ static void nbd_check_meta_export(NBDClient *client, NBDExport *exp) - - /* Send a reply to NBD_OPT_EXPORT_NAME. - * Return -errno on error, 0 on success. */ --static int nbd_negotiate_handle_export_name(NBDClient *client, bool no_zeroes, -- Error **errp) -+static coroutine_fn int -+nbd_negotiate_handle_export_name(NBDClient *client, bool no_zeroes, -+ Error **errp) - { - ERRP_GUARD(); - g_autofree char *name = NULL; -@@ -536,9 +543,9 @@ static int nbd_negotiate_handle_export_name(NBDClient *client, bool no_zeroes, - /* Send a single NBD_REP_INFO, with a buffer @buf of @length bytes. - * The buffer does NOT include the info type prefix. - * Return -errno on error, 0 if ready to send more. */ --static int nbd_negotiate_send_info(NBDClient *client, -- uint16_t info, uint32_t length, void *buf, -- Error **errp) -+static coroutine_fn int -+nbd_negotiate_send_info(NBDClient *client, uint16_t info, uint32_t length, -+ void *buf, Error **errp) - { - int rc; - -@@ -565,7 +572,8 @@ static int nbd_negotiate_send_info(NBDClient *client, - * -errno transmission error occurred or @fatal was requested, errp is set - * 0 error message successfully sent to client, errp is not set - */ --static int nbd_reject_length(NBDClient *client, bool fatal, Error **errp) -+static coroutine_fn int -+nbd_reject_length(NBDClient *client, bool fatal, Error **errp) - { - int ret; - -@@ -583,7 +591,8 @@ static int nbd_reject_length(NBDClient *client, bool fatal, Error **errp) - /* Handle NBD_OPT_INFO and NBD_OPT_GO. - * Return -errno on error, 0 if ready for next option, and 1 to move - * into transmission phase. */ --static int nbd_negotiate_handle_info(NBDClient *client, Error **errp) -+static coroutine_fn int -+nbd_negotiate_handle_info(NBDClient *client, Error **errp) - { - int rc; - g_autofree char *name = NULL; -@@ -755,7 +764,8 @@ struct NBDTLSServerHandshakeData { - Coroutine *co; - }; - --static void nbd_server_tls_handshake(QIOTask *task, void *opaque) -+static void -+nbd_server_tls_handshake(QIOTask *task, void *opaque) - { - struct NBDTLSServerHandshakeData *data = opaque; - -@@ -768,8 +778,8 @@ static void nbd_server_tls_handshake(QIOTask *task, void *opaque) - - /* Handle NBD_OPT_STARTTLS. Return NULL to drop connection, or else the - * new channel for all further (now-encrypted) communication. */ --static QIOChannel *nbd_negotiate_handle_starttls(NBDClient *client, -- Error **errp) -+static coroutine_fn QIOChannel * -+nbd_negotiate_handle_starttls(NBDClient *client, Error **errp) - { - QIOChannel *ioc; - QIOChannelTLS *tioc; -@@ -821,10 +831,9 @@ static QIOChannel *nbd_negotiate_handle_starttls(NBDClient *client, - * - * For NBD_OPT_LIST_META_CONTEXT @context_id is ignored, 0 is used instead. - */ --static int nbd_negotiate_send_meta_context(NBDClient *client, -- const char *context, -- uint32_t context_id, -- Error **errp) -+static coroutine_fn int -+nbd_negotiate_send_meta_context(NBDClient *client, const char *context, -+ uint32_t context_id, Error **errp) - { - NBDOptionReplyMetaContext opt; - struct iovec iov[] = { -@@ -849,8 +858,9 @@ static int nbd_negotiate_send_meta_context(NBDClient *client, - * Return true if @query matches @pattern, or if @query is empty when - * the @client is performing _LIST_. - */ --static bool nbd_meta_empty_or_pattern(NBDClient *client, const char *pattern, -- const char *query) -+static coroutine_fn bool -+nbd_meta_empty_or_pattern(NBDClient *client, const char *pattern, -+ const char *query) - { - if (!*query) { - trace_nbd_negotiate_meta_query_parse("empty"); -@@ -867,7 +877,8 @@ static bool nbd_meta_empty_or_pattern(NBDClient *client, const char *pattern, - /* - * Return true and adjust @str in place if it begins with @prefix. - */ --static bool nbd_strshift(const char **str, const char *prefix) -+static coroutine_fn bool -+nbd_strshift(const char **str, const char *prefix) - { - size_t len = strlen(prefix); - -@@ -883,8 +894,9 @@ static bool nbd_strshift(const char **str, const char *prefix) - * Handle queries to 'base' namespace. For now, only the base:allocation - * context is available. Return true if @query has been handled. - */ --static bool nbd_meta_base_query(NBDClient *client, NBDMetaContexts *meta, -- const char *query) -+static coroutine_fn bool -+nbd_meta_base_query(NBDClient *client, NBDMetaContexts *meta, -+ const char *query) - { - if (!nbd_strshift(&query, "base:")) { - return false; -@@ -903,8 +915,9 @@ static bool nbd_meta_base_query(NBDClient *client, NBDMetaContexts *meta, - * and qemu:allocation-depth contexts are available. Return true if @query - * has been handled. - */ --static bool nbd_meta_qemu_query(NBDClient *client, NBDMetaContexts *meta, -- const char *query) -+static coroutine_fn bool -+nbd_meta_qemu_query(NBDClient *client, NBDMetaContexts *meta, -+ const char *query) - { - size_t i; - -@@ -968,8 +981,9 @@ static bool nbd_meta_qemu_query(NBDClient *client, NBDMetaContexts *meta, - * - * Return -errno on I/O error, 0 if option was completely handled by - * sending a reply about inconsistent lengths, or 1 on success. */ --static int nbd_negotiate_meta_query(NBDClient *client, -- NBDMetaContexts *meta, Error **errp) -+static coroutine_fn int -+nbd_negotiate_meta_query(NBDClient *client, -+ NBDMetaContexts *meta, Error **errp) - { - int ret; - g_autofree char *query = NULL; -@@ -1008,7 +1022,8 @@ static int nbd_negotiate_meta_query(NBDClient *client, - * Handle NBD_OPT_LIST_META_CONTEXT and NBD_OPT_SET_META_CONTEXT - * - * Return -errno on I/O error, or 0 if option was completely handled. */ --static int nbd_negotiate_meta_queries(NBDClient *client, Error **errp) -+static coroutine_fn int -+nbd_negotiate_meta_queries(NBDClient *client, Error **errp) - { - int ret; - g_autofree char *export_name = NULL; -@@ -1136,7 +1151,8 @@ static int nbd_negotiate_meta_queries(NBDClient *client, Error **errp) - * 1 if client sent NBD_OPT_ABORT, i.e. on valid disconnect, - * errp is not set - */ --static int nbd_negotiate_options(NBDClient *client, Error **errp) -+static coroutine_fn int -+nbd_negotiate_options(NBDClient *client, Error **errp) - { - uint32_t flags; - bool fixedNewstyle = false; --- -2.39.3 - diff --git a/kvm-nbd-server-Plumb-in-new-args-to-nbd_client_add.patch b/kvm-nbd-server-Plumb-in-new-args-to-nbd_client_add.patch deleted file mode 100644 index e7d624e..0000000 --- a/kvm-nbd-server-Plumb-in-new-args-to-nbd_client_add.patch +++ /dev/null @@ -1,175 +0,0 @@ -From 785893c171d994bbcffe0585953ca0d290f3c27e Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Wed, 7 Aug 2024 08:50:01 -0500 -Subject: [PATCH 2/5] nbd/server: Plumb in new args to nbd_client_add() -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Eric Blake -RH-MergeRequest: 263: nbd/server: fix CVE-2024-7409 (qemu crash on nbd-server-stop) [RHEL 10.0] -RH-Jira: RHEL-52599 -RH-Acked-by: Miroslav Rezanina -RH-Commit: [1/4] 68e9f83d467704d3dfbf0a879b2bb4d9a568f81c (redhat/centos-stream/src/qemu-kvm) - -Upcoming patches to fix a CVE need to track an opaque pointer passed -in by the owner of a client object, as well as request for a time -limit on how fast negotiation must complete. Prepare for that by -changing the signature of nbd_client_new() and adding an accessor to -get at the opaque pointer, although for now the two servers -(qemu-nbd.c and blockdev-nbd.c) do not change behavior even though -they pass in a new default timeout value. - -Suggested-by: Vladimir Sementsov-Ogievskiy -Signed-off-by: Eric Blake -Message-ID: <20240807174943.771624-11-eblake@redhat.com> -Reviewed-by: Daniel P. Berrangé -[eblake: s/LIMIT/MAX_SECS/ as suggested by Dan] -Signed-off-by: Eric Blake - -(cherry picked from commit fb1c2aaa981e0a2fa6362c9985f1296b74f055ac) -Jira: https://issues.redhat.com/browse/RHEL-52599 -Signed-off-by: Eric Blake ---- - blockdev-nbd.c | 6 ++++-- - include/block/nbd.h | 11 ++++++++++- - nbd/server.c | 20 +++++++++++++++++--- - qemu-nbd.c | 4 +++- - 4 files changed, 34 insertions(+), 7 deletions(-) - -diff --git a/blockdev-nbd.c b/blockdev-nbd.c -index 213012435f..267a1de903 100644 ---- a/blockdev-nbd.c -+++ b/blockdev-nbd.c -@@ -64,8 +64,10 @@ static void nbd_accept(QIONetListener *listener, QIOChannelSocket *cioc, - nbd_update_server_watch(nbd_server); - - qio_channel_set_name(QIO_CHANNEL(cioc), "nbd-server"); -- nbd_client_new(cioc, nbd_server->tlscreds, nbd_server->tlsauthz, -- nbd_blockdev_client_closed); -+ /* TODO - expose handshake timeout as QMP option */ -+ nbd_client_new(cioc, NBD_DEFAULT_HANDSHAKE_MAX_SECS, -+ nbd_server->tlscreds, nbd_server->tlsauthz, -+ nbd_blockdev_client_closed, NULL); - } - - static void nbd_update_server_watch(NBDServerData *s) -diff --git a/include/block/nbd.h b/include/block/nbd.h -index 4e7bd6342f..1d4d65922d 100644 ---- a/include/block/nbd.h -+++ b/include/block/nbd.h -@@ -33,6 +33,12 @@ typedef struct NBDMetaContexts NBDMetaContexts; - - extern const BlockExportDriver blk_exp_nbd; - -+/* -+ * NBD_DEFAULT_HANDSHAKE_MAX_SECS: Number of seconds in which client must -+ * succeed at NBD_OPT_GO before being forcefully dropped as too slow. -+ */ -+#define NBD_DEFAULT_HANDSHAKE_MAX_SECS 10 -+ - /* Handshake phase structs - this struct is passed on the wire */ - - typedef struct NBDOption { -@@ -403,9 +409,12 @@ AioContext *nbd_export_aio_context(NBDExport *exp); - NBDExport *nbd_export_find(const char *name); - - void nbd_client_new(QIOChannelSocket *sioc, -+ uint32_t handshake_max_secs, - QCryptoTLSCreds *tlscreds, - const char *tlsauthz, -- void (*close_fn)(NBDClient *, bool)); -+ void (*close_fn)(NBDClient *, bool), -+ void *owner); -+void *nbd_client_owner(NBDClient *client); - void nbd_client_get(NBDClient *client); - void nbd_client_put(NBDClient *client); - -diff --git a/nbd/server.c b/nbd/server.c -index 892797bb11..e50012499f 100644 ---- a/nbd/server.c -+++ b/nbd/server.c -@@ -124,12 +124,14 @@ struct NBDMetaContexts { - struct NBDClient { - int refcount; /* atomic */ - void (*close_fn)(NBDClient *client, bool negotiated); -+ void *owner; - - QemuMutex lock; - - NBDExport *exp; - QCryptoTLSCreds *tlscreds; - char *tlsauthz; -+ uint32_t handshake_max_secs; - QIOChannelSocket *sioc; /* The underlying data channel */ - QIOChannel *ioc; /* The current I/O channel which may differ (eg TLS) */ - -@@ -3191,6 +3193,7 @@ static coroutine_fn void nbd_co_client_start(void *opaque) - - qemu_co_mutex_init(&client->send_lock); - -+ /* TODO - utilize client->handshake_max_secs */ - if (nbd_negotiate(client, &local_err)) { - if (local_err) { - error_report_err(local_err); -@@ -3205,14 +3208,17 @@ static coroutine_fn void nbd_co_client_start(void *opaque) - } - - /* -- * Create a new client listener using the given channel @sioc. -+ * Create a new client listener using the given channel @sioc and @owner. - * Begin servicing it in a coroutine. When the connection closes, call -- * @close_fn with an indication of whether the client completed negotiation. -+ * @close_fn with an indication of whether the client completed negotiation -+ * within @handshake_max_secs seconds (0 for unbounded). - */ - void nbd_client_new(QIOChannelSocket *sioc, -+ uint32_t handshake_max_secs, - QCryptoTLSCreds *tlscreds, - const char *tlsauthz, -- void (*close_fn)(NBDClient *, bool)) -+ void (*close_fn)(NBDClient *, bool), -+ void *owner) - { - NBDClient *client; - Coroutine *co; -@@ -3225,13 +3231,21 @@ void nbd_client_new(QIOChannelSocket *sioc, - object_ref(OBJECT(client->tlscreds)); - } - client->tlsauthz = g_strdup(tlsauthz); -+ client->handshake_max_secs = handshake_max_secs; - client->sioc = sioc; - qio_channel_set_delay(QIO_CHANNEL(sioc), false); - object_ref(OBJECT(client->sioc)); - client->ioc = QIO_CHANNEL(sioc); - object_ref(OBJECT(client->ioc)); - client->close_fn = close_fn; -+ client->owner = owner; - - co = qemu_coroutine_create(nbd_co_client_start, client); - qemu_coroutine_enter(co); - } -+ -+void * -+nbd_client_owner(NBDClient *client) -+{ -+ return client->owner; -+} -diff --git a/qemu-nbd.c b/qemu-nbd.c -index d7b3ccab21..48e2fa5858 100644 ---- a/qemu-nbd.c -+++ b/qemu-nbd.c -@@ -390,7 +390,9 @@ static void nbd_accept(QIONetListener *listener, QIOChannelSocket *cioc, - - nb_fds++; - nbd_update_server_watch(); -- nbd_client_new(cioc, tlscreds, tlsauthz, nbd_client_closed); -+ /* TODO - expose handshake timeout as command line option */ -+ nbd_client_new(cioc, NBD_DEFAULT_HANDSHAKE_MAX_SECS, -+ tlscreds, tlsauthz, nbd_client_closed, NULL); - } - - static void nbd_update_server_watch(void) --- -2.39.3 - diff --git a/kvm-nbd-server-do-not-poll-within-a-coroutine-context.patch b/kvm-nbd-server-do-not-poll-within-a-coroutine-context.patch deleted file mode 100644 index 55ca4a7..0000000 --- a/kvm-nbd-server-do-not-poll-within-a-coroutine-context.patch +++ /dev/null @@ -1,208 +0,0 @@ -From 484fe3af54a3e421be9e370d47eabe0d8cc5c50d Mon Sep 17 00:00:00 2001 -From: Zhu Yangyang -Date: Mon, 8 Apr 2024 11:00:43 -0500 -Subject: [PATCH 1/4] nbd/server: do not poll within a coroutine context - -RH-Author: Eric Blake -RH-MergeRequest: 257: nbd/server: fix TLS negotiation across coroutine context -RH-Jira: RHEL-40959 -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Miroslav Rezanina -RH-Commit: [1/4] 379f38d46d204890e47a5eb744292d728badc7db (ebblake/centos-qemu-kvm) - -Coroutines are not supposed to block. Instead, they should yield. - -The client performs TLS upgrade outside of an AIOContext, during -synchronous handshake; this still requires g_main_loop. But the -server responds to TLS upgrade inside a coroutine, so a nested -g_main_loop is wrong. Since the two callbacks no longer share more -than the setting of data.complete and data.error, it's just as easy to -use static helpers instead of trying to share a common code path. It -is also possible to add assertions that no other code is interfering -with the eventual path to qio reaching the callback, whether or not it -required a yield or main loop. - -Fixes: f95910f ("nbd: implement TLS support in the protocol negotiation") -Signed-off-by: Zhu Yangyang -[eblake: move callbacks to their use point, add assertions] -Signed-off-by: Eric Blake -Message-ID: <20240408160214.1200629-5-eblake@redhat.com> -Reviewed-by: Vladimir Sementsov-Ogievskiy - -Jira: https://issues.redhat.com/browse/RHEL-40959 -(cherry picked from commit ae6d91a7e9b77abb029ed3fa9fad461422286942) -Signed-off-by: Eric Blake ---- - nbd/client.c | 28 ++++++++++++++++++++++++---- - nbd/common.c | 11 ----------- - nbd/nbd-internal.h | 10 ---------- - nbd/server.c | 28 +++++++++++++++++++++++----- - 4 files changed, 47 insertions(+), 30 deletions(-) - -diff --git a/nbd/client.c b/nbd/client.c -index 29ffc609a4..c89c750467 100644 ---- a/nbd/client.c -+++ b/nbd/client.c -@@ -596,13 +596,31 @@ static int nbd_request_simple_option(QIOChannel *ioc, int opt, bool strict, - return 1; - } - -+/* Callback to learn when QIO TLS upgrade is complete */ -+struct NBDTLSClientHandshakeData { -+ bool complete; -+ Error *error; -+ GMainLoop *loop; -+}; -+ -+static void nbd_client_tls_handshake(QIOTask *task, void *opaque) -+{ -+ struct NBDTLSClientHandshakeData *data = opaque; -+ -+ qio_task_propagate_error(task, &data->error); -+ data->complete = true; -+ if (data->loop) { -+ g_main_loop_quit(data->loop); -+ } -+} -+ - static QIOChannel *nbd_receive_starttls(QIOChannel *ioc, - QCryptoTLSCreds *tlscreds, - const char *hostname, Error **errp) - { - int ret; - QIOChannelTLS *tioc; -- struct NBDTLSHandshakeData data = { 0 }; -+ struct NBDTLSClientHandshakeData data = { 0 }; - - ret = nbd_request_simple_option(ioc, NBD_OPT_STARTTLS, true, errp); - if (ret <= 0) { -@@ -619,18 +637,20 @@ static QIOChannel *nbd_receive_starttls(QIOChannel *ioc, - return NULL; - } - qio_channel_set_name(QIO_CHANNEL(tioc), "nbd-client-tls"); -- data.loop = g_main_loop_new(g_main_context_default(), FALSE); - trace_nbd_receive_starttls_tls_handshake(); - qio_channel_tls_handshake(tioc, -- nbd_tls_handshake, -+ nbd_client_tls_handshake, - &data, - NULL, - NULL); - - if (!data.complete) { -+ data.loop = g_main_loop_new(g_main_context_default(), FALSE); - g_main_loop_run(data.loop); -+ assert(data.complete); -+ g_main_loop_unref(data.loop); - } -- g_main_loop_unref(data.loop); -+ - if (data.error) { - error_propagate(errp, data.error); - object_unref(OBJECT(tioc)); -diff --git a/nbd/common.c b/nbd/common.c -index 3247c1d618..589a748cfe 100644 ---- a/nbd/common.c -+++ b/nbd/common.c -@@ -47,17 +47,6 @@ int nbd_drop(QIOChannel *ioc, size_t size, Error **errp) - } - - --void nbd_tls_handshake(QIOTask *task, -- void *opaque) --{ -- struct NBDTLSHandshakeData *data = opaque; -- -- qio_task_propagate_error(task, &data->error); -- data->complete = true; -- g_main_loop_quit(data->loop); --} -- -- - const char *nbd_opt_lookup(uint32_t opt) - { - switch (opt) { -diff --git a/nbd/nbd-internal.h b/nbd/nbd-internal.h -index dfa02f77ee..91895106a9 100644 ---- a/nbd/nbd-internal.h -+++ b/nbd/nbd-internal.h -@@ -72,16 +72,6 @@ static inline int nbd_write(QIOChannel *ioc, const void *buffer, size_t size, - return qio_channel_write_all(ioc, buffer, size, errp) < 0 ? -EIO : 0; - } - --struct NBDTLSHandshakeData { -- GMainLoop *loop; -- bool complete; -- Error *error; --}; -- -- --void nbd_tls_handshake(QIOTask *task, -- void *opaque); -- - int nbd_drop(QIOChannel *ioc, size_t size, Error **errp); - - #endif -diff --git a/nbd/server.c b/nbd/server.c -index c3484cc1eb..98ae0e1632 100644 ---- a/nbd/server.c -+++ b/nbd/server.c -@@ -748,6 +748,23 @@ static int nbd_negotiate_handle_info(NBDClient *client, Error **errp) - return rc; - } - -+/* Callback to learn when QIO TLS upgrade is complete */ -+struct NBDTLSServerHandshakeData { -+ bool complete; -+ Error *error; -+ Coroutine *co; -+}; -+ -+static void nbd_server_tls_handshake(QIOTask *task, void *opaque) -+{ -+ struct NBDTLSServerHandshakeData *data = opaque; -+ -+ qio_task_propagate_error(task, &data->error); -+ data->complete = true; -+ if (!qemu_coroutine_entered(data->co)) { -+ aio_co_wake(data->co); -+ } -+} - - /* Handle NBD_OPT_STARTTLS. Return NULL to drop connection, or else the - * new channel for all further (now-encrypted) communication. */ -@@ -756,7 +773,7 @@ static QIOChannel *nbd_negotiate_handle_starttls(NBDClient *client, - { - QIOChannel *ioc; - QIOChannelTLS *tioc; -- struct NBDTLSHandshakeData data = { 0 }; -+ struct NBDTLSServerHandshakeData data = { 0 }; - - assert(client->opt == NBD_OPT_STARTTLS); - -@@ -777,17 +794,18 @@ static QIOChannel *nbd_negotiate_handle_starttls(NBDClient *client, - - qio_channel_set_name(QIO_CHANNEL(tioc), "nbd-server-tls"); - trace_nbd_negotiate_handle_starttls_handshake(); -- data.loop = g_main_loop_new(g_main_context_default(), FALSE); -+ data.co = qemu_coroutine_self(); - qio_channel_tls_handshake(tioc, -- nbd_tls_handshake, -+ nbd_server_tls_handshake, - &data, - NULL, - NULL); - - if (!data.complete) { -- g_main_loop_run(data.loop); -+ qemu_coroutine_yield(); -+ assert(data.complete); - } -- g_main_loop_unref(data.loop); -+ - if (data.error) { - object_unref(OBJECT(tioc)); - error_propagate(errp, data.error); --- -2.39.3 - diff --git a/kvm-net-bundle-all-offloads-in-a-single-struct.patch b/kvm-net-bundle-all-offloads-in-a-single-struct.patch new file mode 100644 index 0000000..7d62dd7 --- /dev/null +++ b/kvm-net-bundle-all-offloads-in-a-single-struct.patch @@ -0,0 +1,366 @@ +From e98a5779344e632fab5cce16508fb6fbdb847332 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:15 +0200 +Subject: [PATCH 06/19] net: bundle all offloads in a single struct + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [1/14] 6895db45d4c3490df93d5156a72fa11f706fa5d2 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +The set_offload() argument list is already pretty long and +we are going to introduce soon a bunch of additional offloads. + +Replace the offload arguments with a single struct and update +all the relevant call-sites. + +No functional changes intended. + +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit e5fd02d8253abdc25c0eb145765734890c256b71) +Signed-off-by: Laurent Vivier +--- + hw/net/e1000e_core.c | 5 +++-- + hw/net/igb_core.c | 5 +++-- + hw/net/virtio-net.c | 19 +++++++++++-------- + hw/net/vmxnet3.c | 13 +++++-------- + include/net/net.h | 15 ++++++++++++--- + net/net.c | 5 ++--- + net/netmap.c | 3 +-- + net/tap-bsd.c | 3 +-- + net/tap-linux.c | 21 ++++++++++++--------- + net/tap-solaris.c | 4 ++-- + net/tap-stub.c | 3 +-- + net/tap.c | 8 ++++---- + net/tap_int.h | 4 ++-- + 13 files changed, 59 insertions(+), 49 deletions(-) + +diff --git a/hw/net/e1000e_core.c b/hw/net/e1000e_core.c +index 06657bb3ac..8fef598b49 100644 +--- a/hw/net/e1000e_core.c ++++ b/hw/net/e1000e_core.c +@@ -2822,8 +2822,9 @@ e1000e_update_rx_offloads(E1000ECore *core) + trace_e1000e_rx_set_cso(cso_state); + + if (core->has_vnet) { +- qemu_set_offload(qemu_get_queue(core->owner_nic)->peer, +- cso_state, 0, 0, 0, 0, 0, 0); ++ NetOffloads ol = { .csum = cso_state }; ++ ++ qemu_set_offload(qemu_get_queue(core->owner_nic)->peer, &ol); + } + } + +diff --git a/hw/net/igb_core.c b/hw/net/igb_core.c +index 39e3ce1c8f..45d8fd795b 100644 +--- a/hw/net/igb_core.c ++++ b/hw/net/igb_core.c +@@ -3058,8 +3058,9 @@ igb_update_rx_offloads(IGBCore *core) + trace_e1000e_rx_set_cso(cso_state); + + if (core->has_vnet) { +- qemu_set_offload(qemu_get_queue(core->owner_nic)->peer, +- cso_state, 0, 0, 0, 0, 0, 0); ++ NetOffloads ol = {.csum = cso_state }; ++ ++ qemu_set_offload(qemu_get_queue(core->owner_nic)->peer, &ol); + } + } + +diff --git a/hw/net/virtio-net.c b/hw/net/virtio-net.c +index 6b5b5dace3..b86ba1fd27 100644 +--- a/hw/net/virtio-net.c ++++ b/hw/net/virtio-net.c +@@ -773,14 +773,17 @@ static uint64_t virtio_net_bad_features(VirtIODevice *vdev) + + static void virtio_net_apply_guest_offloads(VirtIONet *n) + { +- qemu_set_offload(qemu_get_queue(n->nic)->peer, +- !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_CSUM)), +- !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_TSO4)), +- !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_TSO6)), +- !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_ECN)), +- !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_UFO)), +- !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_USO4)), +- !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_USO6))); ++ NetOffloads ol = { ++ .csum = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_CSUM)), ++ .tso4 = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_TSO4)), ++ .tso6 = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_TSO6)), ++ .ecn = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_ECN)), ++ .ufo = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_UFO)), ++ .uso4 = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_USO4)), ++ .uso6 = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_USO6)), ++ }; ++ ++ qemu_set_offload(qemu_get_queue(n->nic)->peer, &ol); + } + + static uint64_t virtio_net_guest_offloads_by_features(uint64_t features) +diff --git a/hw/net/vmxnet3.c b/hw/net/vmxnet3.c +index af73aa8ef2..03732375a7 100644 +--- a/hw/net/vmxnet3.c ++++ b/hw/net/vmxnet3.c +@@ -1322,14 +1322,11 @@ static void vmxnet3_update_features(VMXNET3State *s) + s->lro_supported, rxcso_supported, + s->rx_vlan_stripping); + if (s->peer_has_vhdr) { +- qemu_set_offload(qemu_get_queue(s->nic)->peer, +- rxcso_supported, +- s->lro_supported, +- s->lro_supported, +- 0, +- 0, +- 0, +- 0); ++ NetOffloads ol = { .csum = rxcso_supported, ++ .tso4 = s->lro_supported, ++ .tso6 = s->lro_supported }; ++ ++ qemu_set_offload(qemu_get_queue(s->nic)->peer, &ol); + } + } + +diff --git a/include/net/net.h b/include/net/net.h +index 84ee18e0f9..48ba333d02 100644 +--- a/include/net/net.h ++++ b/include/net/net.h +@@ -35,6 +35,16 @@ typedef struct NICConf { + int32_t bootindex; + } NICConf; + ++typedef struct NetOffloads { ++ bool csum; ++ bool tso4; ++ bool tso6; ++ bool ecn; ++ bool ufo; ++ bool uso4; ++ bool uso6; ++} NetOffloads; ++ + #define DEFINE_NIC_PROPERTIES(_state, _conf) \ + DEFINE_PROP_MACADDR("mac", _state, _conf.macaddr), \ + DEFINE_PROP_NETDEV("netdev", _state, _conf.peers) +@@ -57,7 +67,7 @@ typedef bool (HasUfo)(NetClientState *); + typedef bool (HasUso)(NetClientState *); + typedef bool (HasVnetHdr)(NetClientState *); + typedef bool (HasVnetHdrLen)(NetClientState *, int); +-typedef void (SetOffload)(NetClientState *, int, int, int, int, int, int, int); ++typedef void (SetOffload)(NetClientState *, const NetOffloads *); + typedef int (GetVnetHdrLen)(NetClientState *); + typedef void (SetVnetHdrLen)(NetClientState *, int); + typedef bool (GetVnetHashSupportedTypes)(NetClientState *, uint32_t *); +@@ -189,8 +199,7 @@ bool qemu_has_ufo(NetClientState *nc); + bool qemu_has_uso(NetClientState *nc); + bool qemu_has_vnet_hdr(NetClientState *nc); + bool qemu_has_vnet_hdr_len(NetClientState *nc, int len); +-void qemu_set_offload(NetClientState *nc, int csum, int tso4, int tso6, +- int ecn, int ufo, int uso4, int uso6); ++void qemu_set_offload(NetClientState *nc, const NetOffloads *ol); + int qemu_get_vnet_hdr_len(NetClientState *nc); + void qemu_set_vnet_hdr_len(NetClientState *nc, int len); + bool qemu_get_vnet_hash_supported_types(NetClientState *nc, uint32_t *types); +diff --git a/net/net.c b/net/net.c +index da275db86e..63872b6855 100644 +--- a/net/net.c ++++ b/net/net.c +@@ -540,14 +540,13 @@ bool qemu_has_vnet_hdr_len(NetClientState *nc, int len) + return nc->info->has_vnet_hdr_len(nc, len); + } + +-void qemu_set_offload(NetClientState *nc, int csum, int tso4, int tso6, +- int ecn, int ufo, int uso4, int uso6) ++void qemu_set_offload(NetClientState *nc, const NetOffloads *ol) + { + if (!nc || !nc->info->set_offload) { + return; + } + +- nc->info->set_offload(nc, csum, tso4, tso6, ecn, ufo, uso4, uso6); ++ nc->info->set_offload(nc, ol); + } + + int qemu_get_vnet_hdr_len(NetClientState *nc) +diff --git a/net/netmap.c b/net/netmap.c +index 297510e190..6cd8f2bdc5 100644 +--- a/net/netmap.c ++++ b/net/netmap.c +@@ -366,8 +366,7 @@ static void netmap_set_vnet_hdr_len(NetClientState *nc, int len) + } + } + +-static void netmap_set_offload(NetClientState *nc, int csum, int tso4, int tso6, +- int ecn, int ufo, int uso4, int uso6) ++static void netmap_set_offload(NetClientState *nc, const NetOffloads *ol) + { + NetmapState *s = DO_UPCAST(NetmapState, nc, nc); + +diff --git a/net/tap-bsd.c b/net/tap-bsd.c +index b4c84441ba..86b6edee94 100644 +--- a/net/tap-bsd.c ++++ b/net/tap-bsd.c +@@ -231,8 +231,7 @@ int tap_fd_set_vnet_be(int fd, int is_be) + return -EINVAL; + } + +-void tap_fd_set_offload(int fd, int csum, int tso4, +- int tso6, int ecn, int ufo, int uso4, int uso6) ++void tap_fd_set_offload(int fd, const NetOffloads *ol) + { + } + +diff --git a/net/tap-linux.c b/net/tap-linux.c +index 22ec2f45d2..a1c58f74f5 100644 +--- a/net/tap-linux.c ++++ b/net/tap-linux.c +@@ -239,8 +239,7 @@ int tap_fd_set_vnet_be(int fd, int is_be) + abort(); + } + +-void tap_fd_set_offload(int fd, int csum, int tso4, +- int tso6, int ecn, int ufo, int uso4, int uso6) ++void tap_fd_set_offload(int fd, const NetOffloads *ol) + { + unsigned int offload = 0; + +@@ -249,20 +248,24 @@ void tap_fd_set_offload(int fd, int csum, int tso4, + return; + } + +- if (csum) { ++ if (ol->csum) { + offload |= TUN_F_CSUM; +- if (tso4) ++ if (ol->tso4) { + offload |= TUN_F_TSO4; +- if (tso6) ++ } ++ if (ol->tso6) { + offload |= TUN_F_TSO6; +- if ((tso4 || tso6) && ecn) ++ } ++ if ((ol->tso4 || ol->tso6) && ol->ecn) { + offload |= TUN_F_TSO_ECN; +- if (ufo) ++ } ++ if (ol->ufo) { + offload |= TUN_F_UFO; +- if (uso4) { ++ } ++ if (ol->uso4) { + offload |= TUN_F_USO4; + } +- if (uso6) { ++ if (ol->uso6) { + offload |= TUN_F_USO6; + } + } +diff --git a/net/tap-solaris.c b/net/tap-solaris.c +index 51b7830bef..833c066bee 100644 +--- a/net/tap-solaris.c ++++ b/net/tap-solaris.c +@@ -27,6 +27,7 @@ + #include "tap_int.h" + #include "qemu/ctype.h" + #include "qemu/cutils.h" ++#include "net/net.h" + + #include + #include +@@ -235,8 +236,7 @@ int tap_fd_set_vnet_be(int fd, int is_be) + return -EINVAL; + } + +-void tap_fd_set_offload(int fd, int csum, int tso4, +- int tso6, int ecn, int ufo, int uso4, int uso6) ++void tap_fd_set_offload(int fd, const NetOffloads *ol) + { + } + +diff --git a/net/tap-stub.c b/net/tap-stub.c +index 38673434cb..67d14ad4d5 100644 +--- a/net/tap-stub.c ++++ b/net/tap-stub.c +@@ -66,8 +66,7 @@ int tap_fd_set_vnet_be(int fd, int is_be) + return -EINVAL; + } + +-void tap_fd_set_offload(int fd, int csum, int tso4, +- int tso6, int ecn, int ufo, int uso4, int uso6) ++void tap_fd_set_offload(int fd, const NetOffloads *ol) + { + } + +diff --git a/net/tap.c b/net/tap.c +index f7df702f97..72046a43aa 100644 +--- a/net/tap.c ++++ b/net/tap.c +@@ -285,15 +285,14 @@ static int tap_set_vnet_be(NetClientState *nc, bool is_be) + return tap_fd_set_vnet_be(s->fd, is_be); + } + +-static void tap_set_offload(NetClientState *nc, int csum, int tso4, +- int tso6, int ecn, int ufo, int uso4, int uso6) ++static void tap_set_offload(NetClientState *nc, const NetOffloads *ol) + { + TAPState *s = DO_UPCAST(TAPState, nc, nc); + if (s->fd < 0) { + return; + } + +- tap_fd_set_offload(s->fd, csum, tso4, tso6, ecn, ufo, uso4, uso6); ++ tap_fd_set_offload(s->fd, ol); + } + + static void tap_exit_notify(Notifier *notifier, void *data) +@@ -391,6 +390,7 @@ static TAPState *net_tap_fd_init(NetClientState *peer, + int fd, + int vnet_hdr) + { ++ NetOffloads ol = {}; + NetClientState *nc; + TAPState *s; + +@@ -404,7 +404,7 @@ static TAPState *net_tap_fd_init(NetClientState *peer, + s->has_ufo = tap_probe_has_ufo(s->fd); + s->has_uso = tap_probe_has_uso(s->fd); + s->enabled = true; +- tap_set_offload(&s->nc, 0, 0, 0, 0, 0, 0, 0); ++ tap_set_offload(&s->nc, &ol); + /* + * Make sure host header length is set correctly in tap: + * it might have been modified by another instance of qemu. +diff --git a/net/tap_int.h b/net/tap_int.h +index 8857ff299d..f8bbe1cb0c 100644 +--- a/net/tap_int.h ++++ b/net/tap_int.h +@@ -27,6 +27,7 @@ + #define NET_TAP_INT_H + + #include "qapi/qapi-types-net.h" ++#include "net/net.h" + + int tap_open(char *ifname, int ifname_size, int *vnet_hdr, + int vnet_hdr_required, int mq_required, Error **errp); +@@ -37,8 +38,7 @@ void tap_set_sndbuf(int fd, const NetdevTapOptions *tap, Error **errp); + int tap_probe_vnet_hdr(int fd, Error **errp); + int tap_probe_has_ufo(int fd); + int tap_probe_has_uso(int fd); +-void tap_fd_set_offload(int fd, int csum, int tso4, int tso6, int ecn, int ufo, +- int uso4, int uso6); ++void tap_fd_set_offload(int fd, const NetOffloads *ol); + void tap_fd_set_vnet_hdr_len(int fd, int len); + int tap_fd_set_vnet_le(int fd, int vnet_is_le); + int tap_fd_set_vnet_be(int fd, int vnet_is_be); +-- +2.47.3 + diff --git a/kvm-net-implement-UDP-tunnel-features-offloading.patch b/kvm-net-implement-UDP-tunnel-features-offloading.patch new file mode 100644 index 0000000..f7ecee1 --- /dev/null +++ b/kvm-net-implement-UDP-tunnel-features-offloading.patch @@ -0,0 +1,209 @@ +From 97f8e0bbaddf37dcf147f6405a8a96921469d60d Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:28 +0200 +Subject: [PATCH 19/19] net: implement UDP tunnel features offloading + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [14/14] 5ff3488d210020730a71b944519a9938f1628dcc (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +When any host or guest GSO over UDP tunnel offload is enabled the +virtio net header includes the additional tunnel-related fields, +update the size accordingly. + +Push the GSO over UDP tunnel offloads all the way down to the tap +device extending the newly introduced NetFeatures struct, and +eventually enable the associated features. + +As per virtio specification, to convert features bit to offload bit, +map the extended features into the reserved range. + +Finally, make the vhost backend aware of the exact header layout, to +copy it correctly. The tunnel-related field are present if either +the guest or the host negotiated any UDP tunnel related feature: +add them to the kernel supported features list, to allow qemu +transfer to the backend the needed information. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Acked-by: Stefano Garzarella +Reviewed-by: Michael S. Tsirkin +Message-ID: <093b4bc68368046bffbcab2202227632d6e4e83b.1758549625.git.pabeni@redhat.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit a5289563ad74a2a37e8d2101d82935454c71fef4) +Signed-off-by: Laurent Vivier +--- + hw/net/virtio-net.c | 34 ++++++++++++++++++++++++++-------- + include/net/net.h | 2 ++ + net/net.c | 3 ++- + net/tap-linux.c | 6 ++++++ + net/tap.c | 2 ++ + 5 files changed, 38 insertions(+), 9 deletions(-) + +diff --git a/hw/net/virtio-net.c b/hw/net/virtio-net.c +index 0abb8c8a62..f021663f92 100644 +--- a/hw/net/virtio-net.c ++++ b/hw/net/virtio-net.c +@@ -103,6 +103,12 @@ + #define VIRTIO_NET_F2O_SHIFT (VIRTIO_NET_OFFLOAD_MAP_MIN - \ + VIRTIO_NET_FEATURES_MAP_MIN + 64) + ++static bool virtio_has_tunnel_hdr(const uint64_t *features) ++{ ++ return virtio_has_feature_ex(features, VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO) || ++ virtio_has_feature_ex(features, VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO); ++} ++ + static const VirtIOFeature feature_sizes[] = { + {.flags = 1ULL << VIRTIO_NET_F_MAC, + .end = endof(struct virtio_net_config, mac)}, +@@ -659,7 +665,8 @@ static bool peer_has_tunnel(VirtIONet *n) + } + + static void virtio_net_set_mrg_rx_bufs(VirtIONet *n, int mergeable_rx_bufs, +- int version_1, int hash_report) ++ int version_1, int hash_report, ++ int tunnel) + { + int i; + NetClientState *nc; +@@ -667,9 +674,11 @@ static void virtio_net_set_mrg_rx_bufs(VirtIONet *n, int mergeable_rx_bufs, + n->mergeable_rx_bufs = mergeable_rx_bufs; + + if (version_1) { +- n->guest_hdr_len = hash_report ? +- sizeof(struct virtio_net_hdr_v1_hash) : +- sizeof(struct virtio_net_hdr_mrg_rxbuf); ++ n->guest_hdr_len = tunnel ? ++ sizeof(struct virtio_net_hdr_v1_hash_tunnel) : ++ (hash_report ? ++ sizeof(struct virtio_net_hdr_v1_hash) : ++ sizeof(struct virtio_net_hdr_mrg_rxbuf)); + n->rss_data.populate_hash = !!hash_report; + } else { + n->guest_hdr_len = n->mergeable_rx_bufs ? +@@ -803,6 +812,10 @@ static void virtio_net_apply_guest_offloads(VirtIONet *n) + .ufo = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_UFO)), + .uso4 = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_USO4)), + .uso6 = !!(n->curr_guest_offloads & (1ULL << VIRTIO_NET_F_GUEST_USO6)), ++ .tnl = !!(n->curr_guest_offloads & ++ (1ULL << VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_MAPPED)), ++ .tnl_csum = !!(n->curr_guest_offloads & ++ (1ULL << VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM_MAPPED)), + }; + + qemu_set_offload(qemu_get_queue(n->nic)->peer, &ol); +@@ -824,7 +837,9 @@ virtio_net_guest_offloads_by_features(const uint64_t *features) + (1ULL << VIRTIO_NET_F_GUEST_ECN) | + (1ULL << VIRTIO_NET_F_GUEST_UFO) | + (1ULL << VIRTIO_NET_F_GUEST_USO4) | +- (1ULL << VIRTIO_NET_F_GUEST_USO6); ++ (1ULL << VIRTIO_NET_F_GUEST_USO6) | ++ (1ULL << VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_MAPPED) | ++ (1ULL << VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM_MAPPED); + + return guest_offloads_mask & virtio_net_features_to_offload(features); + } +@@ -937,7 +952,8 @@ static void virtio_net_set_features(VirtIODevice *vdev, + virtio_has_feature_ex(features, + VIRTIO_F_VERSION_1), + virtio_has_feature_ex(features, +- VIRTIO_NET_F_HASH_REPORT)); ++ VIRTIO_NET_F_HASH_REPORT), ++ virtio_has_tunnel_hdr(features)); + + n->rsc4_enabled = virtio_has_feature_ex(features, VIRTIO_NET_F_RSC_EXT) && + virtio_has_feature_ex(features, VIRTIO_NET_F_GUEST_TSO4); +@@ -3163,13 +3179,15 @@ static int virtio_net_post_load_device(void *opaque, int version_id) + VirtIONet *n = opaque; + VirtIODevice *vdev = VIRTIO_DEVICE(n); + int i, link_down; ++ bool has_tunnel_hdr = virtio_has_tunnel_hdr(vdev->guest_features_ex); + + trace_virtio_net_post_load_device(); + virtio_net_set_mrg_rx_bufs(n, n->mergeable_rx_bufs, + virtio_vdev_has_feature(vdev, + VIRTIO_F_VERSION_1), + virtio_vdev_has_feature(vdev, +- VIRTIO_NET_F_HASH_REPORT)); ++ VIRTIO_NET_F_HASH_REPORT), ++ has_tunnel_hdr); + + /* MAC_TABLE_ENTRIES may be different from the saved image */ + if (n->mac_table.in_use > MAC_TABLE_ENTRIES) { +@@ -3989,7 +4007,7 @@ static void virtio_net_device_realize(DeviceState *dev, Error **errp) + + n->vqs[0].tx_waiting = 0; + n->tx_burst = n->net_conf.txburst; +- virtio_net_set_mrg_rx_bufs(n, 0, 0, 0); ++ virtio_net_set_mrg_rx_bufs(n, 0, 0, 0, 0); + n->promisc = 1; /* for compatibility */ + + n->mac_table.macs = g_malloc0(MAC_TABLE_ENTRIES * ETH_ALEN); +diff --git a/include/net/net.h b/include/net/net.h +index 9a9084690d..72b476ee1d 100644 +--- a/include/net/net.h ++++ b/include/net/net.h +@@ -43,6 +43,8 @@ typedef struct NetOffloads { + bool ufo; + bool uso4; + bool uso6; ++ bool tnl; ++ bool tnl_csum; + } NetOffloads; + + #define DEFINE_NIC_PROPERTIES(_state, _conf) \ +diff --git a/net/net.c b/net/net.c +index 9536184a0c..27e0d27807 100644 +--- a/net/net.c ++++ b/net/net.c +@@ -575,7 +575,8 @@ void qemu_set_vnet_hdr_len(NetClientState *nc, int len) + + assert(len == sizeof(struct virtio_net_hdr_mrg_rxbuf) || + len == sizeof(struct virtio_net_hdr) || +- len == sizeof(struct virtio_net_hdr_v1_hash)); ++ len == sizeof(struct virtio_net_hdr_v1_hash) || ++ len == sizeof(struct virtio_net_hdr_v1_hash_tunnel)); + + nc->vnet_hdr_len = len; + nc->info->set_vnet_hdr_len(nc, len); +diff --git a/net/tap-linux.c b/net/tap-linux.c +index e2628be798..8e275d2ea4 100644 +--- a/net/tap-linux.c ++++ b/net/tap-linux.c +@@ -279,6 +279,12 @@ void tap_fd_set_offload(int fd, const NetOffloads *ol) + if (ol->uso6) { + offload |= TUN_F_USO6; + } ++ if (ol->tnl) { ++ offload |= TUN_F_UDP_TUNNEL_GSO; ++ } ++ if (ol->tnl_csum) { ++ offload |= TUN_F_UDP_TUNNEL_GSO_CSUM; ++ } + } + + if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) { +diff --git a/net/tap.c b/net/tap.c +index 9f65e3fb3d..dc2a2859ec 100644 +--- a/net/tap.c ++++ b/net/tap.c +@@ -62,6 +62,8 @@ static const int kernel_feature_bits[] = { + VIRTIO_F_NOTIFICATION_DATA, + VIRTIO_NET_F_RSC_EXT, + VIRTIO_NET_F_HASH_REPORT, ++ VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO, ++ VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO, + VHOST_INVALID_FEATURE_BIT + }; + +-- +2.47.3 + diff --git a/kvm-net-implement-tunnel-probing.patch b/kvm-net-implement-tunnel-probing.patch new file mode 100644 index 0000000..24305cb --- /dev/null +++ b/kvm-net-implement-tunnel-probing.patch @@ -0,0 +1,317 @@ +From 0cad25c7acac95a4af0d46f4d0207de4ac973370 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:27 +0200 +Subject: [PATCH 18/19] net: implement tunnel probing + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [13/14] 3cd000a61b90ffc176f950686fb9d36d93bbee3f (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Tap devices support GSO over UDP tunnel offload. Probe for such +feature in a similar manner to other offloads. + +GSO over UDP tunnel needs to be enabled in addition to a "plain" +offload (TSO or USO). + +No need to check separately for the outer header checksum offload: +the kernel is going to support both of them or none. + +The new features are disabled by default to avoid compat issues, +and could be enabled, after that hw_compat_10_1 will be added, +together with the related compat entries. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Acked-by: Stefano Garzarella +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit fffac046282c99801b62fa7fa1032cdc261bca6d) +Signed-off-by: Laurent Vivier +--- + hw/net/virtio-net.c | 41 +++++++++++++++++++++++++++++++++++++++++ + include/net/net.h | 3 +++ + net/net.c | 9 +++++++++ + net/tap-bsd.c | 5 +++++ + net/tap-linux.c | 11 +++++++++++ + net/tap-linux.h | 9 +++++++++ + net/tap-solaris.c | 5 +++++ + net/tap-stub.c | 5 +++++ + net/tap.c | 11 +++++++++++ + net/tap_int.h | 1 + + 10 files changed, 100 insertions(+) + +diff --git a/hw/net/virtio-net.c b/hw/net/virtio-net.c +index 89cf008401..0abb8c8a62 100644 +--- a/hw/net/virtio-net.c ++++ b/hw/net/virtio-net.c +@@ -649,6 +649,15 @@ static int peer_has_uso(VirtIONet *n) + return qemu_has_uso(qemu_get_queue(n->nic)->peer); + } + ++static bool peer_has_tunnel(VirtIONet *n) ++{ ++ if (!peer_has_vnet_hdr(n)) { ++ return false; ++ } ++ ++ return qemu_has_tunnel(qemu_get_queue(n->nic)->peer); ++} ++ + static void virtio_net_set_mrg_rx_bufs(VirtIONet *n, int mergeable_rx_bufs, + int version_1, int hash_report) + { +@@ -3073,6 +3082,13 @@ static void virtio_net_get_features(VirtIODevice *vdev, uint64_t *features, + virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_USO4); + virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_USO6); + ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO); ++ virtio_clear_feature_ex(features, ++ VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM); ++ virtio_clear_feature_ex(features, ++ VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO_CSUM); ++ + virtio_clear_feature_ex(features, VIRTIO_NET_F_HASH_REPORT); + } + +@@ -3086,6 +3102,15 @@ static void virtio_net_get_features(VirtIODevice *vdev, uint64_t *features, + virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_USO6); + } + ++ if (!peer_has_tunnel(n)) { ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO); ++ virtio_clear_feature_ex(features, ++ VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM); ++ virtio_clear_feature_ex(features, ++ VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO_CSUM); ++ } ++ + if (!get_vhost_net(nc->peer)) { + if (!use_own_hash) { + virtio_clear_feature_ex(features, VIRTIO_NET_F_HASH_REPORT); +@@ -4248,6 +4273,22 @@ static const Property virtio_net_properties[] = { + rss_data.specified_hash_types, + VIRTIO_NET_HASH_REPORT_UDPv6_EX - 1, + ON_OFF_AUTO_AUTO), ++ VIRTIO_DEFINE_PROP_FEATURE("host_tunnel", VirtIONet, ++ host_features_ex, ++ VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO, ++ false), ++ VIRTIO_DEFINE_PROP_FEATURE("host_tunnel_csum", VirtIONet, ++ host_features_ex, ++ VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO_CSUM, ++ false), ++ VIRTIO_DEFINE_PROP_FEATURE("guest_tunnel", VirtIONet, ++ host_features_ex, ++ VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO, ++ false), ++ VIRTIO_DEFINE_PROP_FEATURE("guest_tunnel_csum", VirtIONet, ++ host_features_ex, ++ VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM, ++ false), + }; + + static void virtio_net_class_init(ObjectClass *klass, const void *data) +diff --git a/include/net/net.h b/include/net/net.h +index 48ba333d02..9a9084690d 100644 +--- a/include/net/net.h ++++ b/include/net/net.h +@@ -65,6 +65,7 @@ typedef void (NetClientDestructor)(NetClientState *); + typedef RxFilterInfo *(QueryRxFilter)(NetClientState *); + typedef bool (HasUfo)(NetClientState *); + typedef bool (HasUso)(NetClientState *); ++typedef bool (HasTunnel)(NetClientState *); + typedef bool (HasVnetHdr)(NetClientState *); + typedef bool (HasVnetHdrLen)(NetClientState *, int); + typedef void (SetOffload)(NetClientState *, const NetOffloads *); +@@ -95,6 +96,7 @@ typedef struct NetClientInfo { + NetPoll *poll; + HasUfo *has_ufo; + HasUso *has_uso; ++ HasTunnel *has_tunnel; + HasVnetHdr *has_vnet_hdr; + HasVnetHdrLen *has_vnet_hdr_len; + SetOffload *set_offload; +@@ -197,6 +199,7 @@ void qemu_set_info_str(NetClientState *nc, + void qemu_format_nic_info_str(NetClientState *nc, uint8_t macaddr[6]); + bool qemu_has_ufo(NetClientState *nc); + bool qemu_has_uso(NetClientState *nc); ++bool qemu_has_tunnel(NetClientState *nc); + bool qemu_has_vnet_hdr(NetClientState *nc); + bool qemu_has_vnet_hdr_len(NetClientState *nc, int len); + void qemu_set_offload(NetClientState *nc, const NetOffloads *ol); +diff --git a/net/net.c b/net/net.c +index 63872b6855..9536184a0c 100644 +--- a/net/net.c ++++ b/net/net.c +@@ -522,6 +522,15 @@ bool qemu_has_uso(NetClientState *nc) + return nc->info->has_uso(nc); + } + ++bool qemu_has_tunnel(NetClientState *nc) ++{ ++ if (!nc || !nc->info->has_tunnel) { ++ return false; ++ } ++ ++ return nc->info->has_tunnel(nc); ++} ++ + bool qemu_has_vnet_hdr(NetClientState *nc) + { + if (!nc || !nc->info->has_vnet_hdr) { +diff --git a/net/tap-bsd.c b/net/tap-bsd.c +index 86b6edee94..751d4c819c 100644 +--- a/net/tap-bsd.c ++++ b/net/tap-bsd.c +@@ -217,6 +217,11 @@ int tap_probe_has_uso(int fd) + return 0; + } + ++bool tap_probe_has_tunnel(int fd) ++{ ++ return false; ++} ++ + void tap_fd_set_vnet_hdr_len(int fd, int len) + { + } +diff --git a/net/tap-linux.c b/net/tap-linux.c +index a1c58f74f5..e2628be798 100644 +--- a/net/tap-linux.c ++++ b/net/tap-linux.c +@@ -196,6 +196,17 @@ int tap_probe_has_uso(int fd) + return 1; + } + ++bool tap_probe_has_tunnel(int fd) ++{ ++ unsigned offload; ++ ++ offload = TUN_F_CSUM | TUN_F_TSO4 | TUN_F_UDP_TUNNEL_GSO; ++ if (ioctl(fd, TUNSETOFFLOAD, offload) < 0) { ++ return false; ++ } ++ return true; ++} ++ + void tap_fd_set_vnet_hdr_len(int fd, int len) + { + if (ioctl(fd, TUNSETVNETHDRSZ, &len) == -1) { +diff --git a/net/tap-linux.h b/net/tap-linux.h +index 9a58cecb7f..8cd6b5874b 100644 +--- a/net/tap-linux.h ++++ b/net/tap-linux.h +@@ -53,4 +53,13 @@ + #define TUN_F_USO4 0x20 /* I can handle USO for IPv4 packets */ + #define TUN_F_USO6 0x40 /* I can handle USO for IPv6 packets */ + ++/* I can handle TSO/USO for UDP tunneled packets */ ++#define TUN_F_UDP_TUNNEL_GSO 0x080 ++ ++/* ++ * I can handle TSO/USO for UDP tunneled packets requiring csum offload for ++ * the outer header ++ */ ++#define TUN_F_UDP_TUNNEL_GSO_CSUM 0x100 ++ + #endif /* QEMU_TAP_LINUX_H */ +diff --git a/net/tap-solaris.c b/net/tap-solaris.c +index 833c066bee..ac1ae25761 100644 +--- a/net/tap-solaris.c ++++ b/net/tap-solaris.c +@@ -222,6 +222,11 @@ int tap_probe_has_uso(int fd) + return 0; + } + ++bool tap_probe_has_tunnel(int fd) ++{ ++ return false; ++} ++ + void tap_fd_set_vnet_hdr_len(int fd, int len) + { + } +diff --git a/net/tap-stub.c b/net/tap-stub.c +index 67d14ad4d5..f7a5e0c163 100644 +--- a/net/tap-stub.c ++++ b/net/tap-stub.c +@@ -52,6 +52,11 @@ int tap_probe_has_uso(int fd) + return 0; + } + ++bool tap_probe_has_tunnel(int fd) ++{ ++ return false; ++} ++ + void tap_fd_set_vnet_hdr_len(int fd, int len) + { + } +diff --git a/net/tap.c b/net/tap.c +index 72046a43aa..9f65e3fb3d 100644 +--- a/net/tap.c ++++ b/net/tap.c +@@ -76,6 +76,7 @@ typedef struct TAPState { + bool using_vnet_hdr; + bool has_ufo; + bool has_uso; ++ bool has_tunnel; + bool enabled; + VHostNetState *vhost_net; + unsigned host_vnet_hdr_len; +@@ -246,6 +247,14 @@ static bool tap_has_uso(NetClientState *nc) + return s->has_uso; + } + ++static bool tap_has_tunnel(NetClientState *nc) ++{ ++ TAPState *s = DO_UPCAST(TAPState, nc, nc); ++ ++ assert(nc->info->type == NET_CLIENT_DRIVER_TAP); ++ return s->has_tunnel; ++} ++ + static bool tap_has_vnet_hdr(NetClientState *nc) + { + TAPState *s = DO_UPCAST(TAPState, nc, nc); +@@ -374,6 +383,7 @@ static NetClientInfo net_tap_info = { + .cleanup = tap_cleanup, + .has_ufo = tap_has_ufo, + .has_uso = tap_has_uso, ++ .has_tunnel = tap_has_tunnel, + .has_vnet_hdr = tap_has_vnet_hdr, + .has_vnet_hdr_len = tap_has_vnet_hdr_len, + .set_offload = tap_set_offload, +@@ -403,6 +413,7 @@ static TAPState *net_tap_fd_init(NetClientState *peer, + s->using_vnet_hdr = false; + s->has_ufo = tap_probe_has_ufo(s->fd); + s->has_uso = tap_probe_has_uso(s->fd); ++ s->has_tunnel = tap_probe_has_tunnel(s->fd); + s->enabled = true; + tap_set_offload(&s->nc, &ol); + /* +diff --git a/net/tap_int.h b/net/tap_int.h +index f8bbe1cb0c..b76a05044b 100644 +--- a/net/tap_int.h ++++ b/net/tap_int.h +@@ -38,6 +38,7 @@ void tap_set_sndbuf(int fd, const NetdevTapOptions *tap, Error **errp); + int tap_probe_vnet_hdr(int fd, Error **errp); + int tap_probe_has_ufo(int fd); + int tap_probe_has_uso(int fd); ++bool tap_probe_has_tunnel(int fd); + void tap_fd_set_offload(int fd, const NetOffloads *ol); + void tap_fd_set_vnet_hdr_len(int fd, int len); + int tap_fd_set_vnet_le(int fd, int vnet_is_le); +-- +2.47.3 + diff --git a/kvm-pcie_sriov-Fix-broken-MMIO-accesses-from-SR-IOV-VFs.patch b/kvm-pcie_sriov-Fix-broken-MMIO-accesses-from-SR-IOV-VFs.patch new file mode 100644 index 0000000..4e87061 --- /dev/null +++ b/kvm-pcie_sriov-Fix-broken-MMIO-accesses-from-SR-IOV-VFs.patch @@ -0,0 +1,177 @@ +From 36d73722360db85d210bacef0c8386e5af031447 Mon Sep 17 00:00:00 2001 +From: Damien Bergamini +Date: Mon, 1 Sep 2025 15:14:23 +0000 +Subject: [PATCH 1/2] pcie_sriov: Fix broken MMIO accesses from SR-IOV VFs +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Laurent Vivier +RH-MergeRequest: 431: pcie_sriov: Fix broken MMIO accesses from SR-IOV VFs +RH-Jira: RHEL-120115 +RH-Acked-by: Eric Auger +RH-Acked-by: Cédric Le Goater +RH-Commit: [1/1] b6bb40d37376db589852cb07a915d6f0eae29f67 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-120115 + +Starting with commit cab1398a60eb, SR-IOV VFs are realized as soon as +pcie_sriov_pf_init() is called. Because pcie_sriov_pf_init() must be +called before pcie_sriov_pf_init_vf_bar(), the VF BARs types won't be +known when the VF realize function calls pcie_sriov_vf_register_bar(). + +This breaks the memory regions of the VFs (for instance with igbvf): + +$ lspci +... + Region 0: Memory at 281a00000 (64-bit, prefetchable) [virtual] [size=16K] + Region 3: Memory at 281a20000 (64-bit, prefetchable) [virtual] [size=16K] + +$ info mtree +... +address-space: pci_bridge_pci_mem + 0000000000000000-ffffffffffffffff (prio 0, i/o): pci_bridge_pci + 0000000081a00000-0000000081a03fff (prio 1, i/o): igbvf-mmio + 0000000081a20000-0000000081a23fff (prio 1, i/o): igbvf-msix + +and causes MMIO accesses to fail: + + Invalid write at addr 0x281A01520, size 4, region '(null)', reason: rejected + Invalid read at addr 0x281A00C40, size 4, region '(null)', reason: rejected + +To fix this, VF BARs are now registered with pci_register_bar() which +has a type parameter and pcie_sriov_vf_register_bar() is removed. + +Fixes: cab1398a60eb ("pcie_sriov: Reuse SR-IOV VF device instances") +Signed-off-by: Damien Bergamini +Signed-off-by: Clement Mathieu--Drif +Reviewed-by: Akihiko Odaki +Reviewed-by: Michael S. Tsirkin +Message-ID: <20250901151314.1038020-1-clement.mathieu--drif@eviden.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 2e54e5fda779a7ba45578884276dca62462f7a06) +Signed-off-by: Laurent Vivier +--- + docs/pcie_sriov.txt | 5 ++--- + hw/net/igbvf.c | 6 ++++-- + hw/nvme/ctrl.c | 8 ++------ + hw/pci/pci.c | 3 --- + hw/pci/pcie_sriov.c | 11 ----------- + include/hw/pci/pcie_sriov.h | 4 ---- + 6 files changed, 8 insertions(+), 29 deletions(-) + +diff --git a/docs/pcie_sriov.txt b/docs/pcie_sriov.txt +index ab2142807f..00d7bd93fd 100644 +--- a/docs/pcie_sriov.txt ++++ b/docs/pcie_sriov.txt +@@ -72,8 +72,7 @@ setting up a BAR for a VF. + 2) Similarly in the implementation of the virtual function, you need to + make it a PCI Express device and add a similar set of capabilities + except for the SR/IOV capability. Then you need to set up the VF BARs as +- subregions of the PFs SR/IOV VF BARs by calling +- pcie_sriov_vf_register_bar() instead of the normal pci_register_bar() call: ++ subregions of the PFs SR/IOV VF BARs by calling pci_register_bar(): + + pci_your_vf_dev_realize( ... ) + { +@@ -83,7 +82,7 @@ setting up a BAR for a VF. + pcie_ari_init(d, 0x100); + ... + memory_region_init(mr, ... ) +- pcie_sriov_vf_register_bar(d, bar_nr, mr); ++ pci_register_bar(d, bar_nr, bar_type, mr); + ... + } + +diff --git a/hw/net/igbvf.c b/hw/net/igbvf.c +index 31d72c4977..9b0db8f841 100644 +--- a/hw/net/igbvf.c ++++ b/hw/net/igbvf.c +@@ -251,10 +251,12 @@ static void igbvf_pci_realize(PCIDevice *dev, Error **errp) + + memory_region_init_io(&s->mmio, OBJECT(dev), &mmio_ops, s, "igbvf-mmio", + IGBVF_MMIO_SIZE); +- pcie_sriov_vf_register_bar(dev, IGBVF_MMIO_BAR_IDX, &s->mmio); ++ pci_register_bar(dev, IGBVF_MMIO_BAR_IDX, PCI_BASE_ADDRESS_MEM_TYPE_64 | ++ PCI_BASE_ADDRESS_MEM_PREFETCH, &s->mmio); + + memory_region_init(&s->msix, OBJECT(dev), "igbvf-msix", IGBVF_MSIX_SIZE); +- pcie_sriov_vf_register_bar(dev, IGBVF_MSIX_BAR_IDX, &s->msix); ++ pci_register_bar(dev, IGBVF_MSIX_BAR_IDX, PCI_BASE_ADDRESS_MEM_TYPE_64 | ++ PCI_BASE_ADDRESS_MEM_PREFETCH, &s->msix); + + ret = msix_init(dev, IGBVF_MSIX_VEC_NUM, &s->msix, IGBVF_MSIX_BAR_IDX, 0, + &s->msix, IGBVF_MSIX_BAR_IDX, 0x2000, 0x70, errp); +diff --git a/hw/nvme/ctrl.c b/hw/nvme/ctrl.c +index f5ee6bf260..cd81f73997 100644 +--- a/hw/nvme/ctrl.c ++++ b/hw/nvme/ctrl.c +@@ -8708,12 +8708,8 @@ static bool nvme_init_pci(NvmeCtrl *n, PCIDevice *pci_dev, Error **errp) + msix_table_offset); + memory_region_add_subregion(&n->bar0, 0, &n->iomem); + +- if (pci_is_vf(pci_dev)) { +- pcie_sriov_vf_register_bar(pci_dev, 0, &n->bar0); +- } else { +- pci_register_bar(pci_dev, 0, PCI_BASE_ADDRESS_SPACE_MEMORY | +- PCI_BASE_ADDRESS_MEM_TYPE_64, &n->bar0); +- } ++ pci_register_bar(pci_dev, 0, PCI_BASE_ADDRESS_SPACE_MEMORY | ++ PCI_BASE_ADDRESS_MEM_TYPE_64, &n->bar0); + + ret = msix_init(pci_dev, nr_vectors, + &n->bar0, 0, msix_table_offset, +diff --git a/hw/pci/pci.c b/hw/pci/pci.c +index 0012cc12e7..d2ebb066e1 100644 +--- a/hw/pci/pci.c ++++ b/hw/pci/pci.c +@@ -1490,9 +1490,6 @@ void pci_register_bar(PCIDevice *pci_dev, int region_num, + : pci_get_bus(pci_dev)->address_space_mem; + + if (pci_is_vf(pci_dev)) { +- PCIDevice *pf = pci_dev->exp.sriov_vf.pf; +- assert(!pf || type == pf->exp.sriov_pf.vf_bar_type[region_num]); +- + r->addr = pci_bar_address(pci_dev, region_num, r->type, r->size); + if (r->addr != PCI_BAR_UNMAPPED) { + memory_region_add_subregion_overlap(r->address_space, +diff --git a/hw/pci/pcie_sriov.c b/hw/pci/pcie_sriov.c +index cf1b5b5c05..c4f88f0975 100644 +--- a/hw/pci/pcie_sriov.c ++++ b/hw/pci/pcie_sriov.c +@@ -246,17 +246,6 @@ void pcie_sriov_pf_init_vf_bar(PCIDevice *dev, int region_num, + dev->exp.sriov_pf.vf_bar_type[region_num] = type; + } + +-void pcie_sriov_vf_register_bar(PCIDevice *dev, int region_num, +- MemoryRegion *memory) +-{ +- uint8_t type; +- +- assert(dev->exp.sriov_vf.pf); +- type = dev->exp.sriov_vf.pf->exp.sriov_pf.vf_bar_type[region_num]; +- +- return pci_register_bar(dev, region_num, type, memory); +-} +- + static gint compare_vf_devfns(gconstpointer a, gconstpointer b) + { + return (*(PCIDevice **)a)->devfn - (*(PCIDevice **)b)->devfn; +diff --git a/include/hw/pci/pcie_sriov.h b/include/hw/pci/pcie_sriov.h +index aeaa38cf34..b0ea6a62c7 100644 +--- a/include/hw/pci/pcie_sriov.h ++++ b/include/hw/pci/pcie_sriov.h +@@ -37,10 +37,6 @@ void pcie_sriov_pf_exit(PCIDevice *dev); + void pcie_sriov_pf_init_vf_bar(PCIDevice *dev, int region_num, + uint8_t type, dma_addr_t size); + +-/* Instantiate a bar for a VF */ +-void pcie_sriov_vf_register_bar(PCIDevice *dev, int region_num, +- MemoryRegion *memory); +- + /** + * pcie_sriov_pf_init_from_user_created_vfs() - Initialize PF with user-created + * VFs, adding ARI to PF +-- +2.47.3 + diff --git a/kvm-pcie_sriov-make-pcie_sriov_pf_exit-safe-on-non-SR-IO.patch b/kvm-pcie_sriov-make-pcie_sriov_pf_exit-safe-on-non-SR-IO.patch new file mode 100644 index 0000000..8790d13 --- /dev/null +++ b/kvm-pcie_sriov-make-pcie_sriov_pf_exit-safe-on-non-SR-IO.patch @@ -0,0 +1,73 @@ +From db20fe92c8cc22e3317787d6c83056126e912f3a Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Wed, 24 Sep 2025 11:51:53 -0400 +Subject: [PATCH 2/4] pcie_sriov: make pcie_sriov_pf_exit() safe on non-SR-IOV + devices + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 408: pcie_sriov: make pcie_sriov_pf_exit() safe on non-SR-IOV devices +RH-Jira: RHEL-116443 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/1] a4fea6ad073c9fa5fdc7f20b3dfeab4e4acef73f (stefanha/centos-stream-qemu-kvm) + +Commit 3f9cfaa92c96 ("virtio-pci: Implement SR-IOV PF") added an +unconditional call from virtio_pci_exit() to pcie_sriov_pf_exit(). + +pcie_sriov_pf_exit() reads from the SR-IOV Capability in Configuration +Space: + + uint8_t *cfg = dev->config + dev->exp.sriov_cap; + ... + unparent_vfs(dev, pci_get_word(cfg + PCI_SRIOV_TOTAL_VF)); + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +This results in undefined behavior when dev->exp.sriov_cap is 0 because +this is not an SR-IOV device. For example, unparent_vfs() segfaults when +total_vfs happens to be non-zero. + +Fix this by returning early from pcie_sriov_pf_exit() when +dev->exp.sriov_cap is 0 because this is not an SR-IOV device. + +Cc: Akihiko Odaki +Cc: Michael S. Tsirkin +Reported-by: Qing Wang +Buglink: https://issues.redhat.com/browse/RHEL-116443 +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Akihiko Odaki +Fixes: cab1398a60eb ("pcie_sriov: Reuse SR-IOV VF device instances") +Reviewed-by: Michael S. Tsirkin +Message-ID: <20250924155153.579495-1-stefanha@redhat.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit bab681f752048c3bc22d561b1d314c7ec16419c9) +Signed-off-by: Stefan Hajnoczi +--- + hw/pci/pcie_sriov.c | 6 +++++- + 1 file changed, 5 insertions(+), 1 deletion(-) + +diff --git a/hw/pci/pcie_sriov.c b/hw/pci/pcie_sriov.c +index 8a4bf0d6f7..cf1b5b5c05 100644 +--- a/hw/pci/pcie_sriov.c ++++ b/hw/pci/pcie_sriov.c +@@ -195,7 +195,9 @@ bool pcie_sriov_pf_init(PCIDevice *dev, uint16_t offset, + + void pcie_sriov_pf_exit(PCIDevice *dev) + { +- uint8_t *cfg = dev->config + dev->exp.sriov_cap; ++ if (dev->exp.sriov_cap == 0) { ++ return; ++ } + + if (dev->exp.sriov_pf.vf_user_created) { + uint16_t ven_id = pci_get_word(dev->config + PCI_VENDOR_ID); +@@ -211,6 +213,8 @@ void pcie_sriov_pf_exit(PCIDevice *dev) + pci_config_set_device_id(dev->exp.sriov_pf.vf[i]->config, vf_dev_id); + } + } else { ++ uint8_t *cfg = dev->config + dev->exp.sriov_cap; ++ + unparent_vfs(dev, pci_get_word(cfg + PCI_SRIOV_TOTAL_VF)); + } + } +-- +2.47.3 + diff --git a/kvm-q35-increase-default-tseg-size.patch b/kvm-q35-increase-default-tseg-size.patch new file mode 100644 index 0000000..9ac5d51 --- /dev/null +++ b/kvm-q35-increase-default-tseg-size.patch @@ -0,0 +1,63 @@ +From 50db9d559ad56dfa54de13adbb494edca24d9d43 Mon Sep 17 00:00:00 2001 +From: Gerd Hoffmann +Date: Thu, 6 Nov 2025 11:56:40 +0100 +Subject: [PATCH 2/4] q35: increase default tseg size + +RH-Author: Gerd Hoffmann +RH-MergeRequest: 447: q35: increase default tseg size +RH-Jira: RHEL-126707 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/2] 8cbbb7b55e2e83bb6129b2da92179ff7b969c4ed (kraxel.rh/centos-src-qemu-kvm) + +With virtual machines becoming larger (more CPUs, more memory) the +memory needed by the SMM code in OVMF to manage page tables and vcpu +state grows too. + +Default SMM memory (aka TSEG) size is 16 MB, and this often is not +enough. Bump it to 64 MB for new machine types. + +Signed-off-by: Gerd Hoffmann +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +Message-Id: <20251106105640.1642109-1-kraxel@redhat.com> +(cherry picked from commit fa4136387928749d5a1b4fa3606ae0ce6dce75aa) + +[ RHEL: drop compat property update for upstream machine types ] +[ RHEL: add compat property for rhel 10.0 machine type ] +[ NOTE: the 10.0 compat properties are used for + rhel-9.6 + older machine types too ] + +Resolves: RHEL-126707 +--- + hw/i386/pc.c | 1 + + hw/pci-host/q35.c | 2 +- + 2 files changed, 2 insertions(+), 1 deletion(-) + +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index 394a84eb8a..27900e071b 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -299,6 +299,7 @@ const size_t pc_rhel_compat_len = G_N_ELEMENTS(pc_rhel_compat); + + GlobalProperty pc_rhel_10_2_compat[] = { + /* pc_rhel_10_2_compat from pc_compat_10_0 */ ++ { "mch", "extended-tseg-mbytes", "16" }, + { TYPE_X86_CPU, "x-consistent-cache", "false" }, + { TYPE_X86_CPU, "x-vendor-cpuid-only-v2", "false" }, + { TYPE_X86_CPU, "x-arch-cap-always-on", "true" }, +diff --git a/hw/pci-host/q35.c b/hw/pci-host/q35.c +index 1951ae440c..a708758d36 100644 +--- a/hw/pci-host/q35.c ++++ b/hw/pci-host/q35.c +@@ -663,7 +663,7 @@ static void mch_realize(PCIDevice *d, Error **errp) + + static const Property mch_props[] = { + DEFINE_PROP_UINT16("extended-tseg-mbytes", MCHPCIState, ext_tseg_mbytes, +- 16), ++ 64), + DEFINE_PROP_BOOL("smbase-smram", MCHPCIState, has_smram_at_smbase, true), + }; + +-- +2.47.3 + diff --git a/kvm-qapi-accel-Allow-to-query-mshv-capabilities.patch b/kvm-qapi-accel-Allow-to-query-mshv-capabilities.patch new file mode 100644 index 0000000..5cac472 --- /dev/null +++ b/kvm-qapi-accel-Allow-to-query-mshv-capabilities.patch @@ -0,0 +1,169 @@ +From 4c29e56da6378dcd55336748d65a777ac85a7fc8 Mon Sep 17 00:00:00 2001 +From: Praveen K Paladugu +Date: Tue, 16 Sep 2025 18:48:44 +0200 +Subject: [PATCH 27/32] qapi/accel: Allow to query mshv capabilities + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [25/30] 50d73451b735ee139ca065c15a84e5f7e77e3904 + +Allow to query mshv capabilities via query-mshv QMP and info mshv HMP commands. + +Signed-off-by: Magnus Kulke +Acked-by: Dr. David Alan Gilbert +Link: https://lore.kernel.org/r/20250916164847.77883-25-magnuskulke@linux.microsoft.com +[Fix "since" version. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit e7b08dfb902430b4f8226d23d7cf9b2762b6fc83) +Signed-off-by: Igor Mammedov +--- + hmp-commands-info.hx | 13 +++++++++++++ + hw/core/machine-hmp-cmds.c | 15 +++++++++++++++ + hw/core/machine-qmp-cmds.c | 14 ++++++++++++++ + include/monitor/hmp.h | 1 + + include/system/hw_accel.h | 1 + + qapi/accelerator.json | 29 +++++++++++++++++++++++++++++ + 6 files changed, 73 insertions(+) + +diff --git a/hmp-commands-info.hx b/hmp-commands-info.hx +index 6142f60e7b..eaaa880c1b 100644 +--- a/hmp-commands-info.hx ++++ b/hmp-commands-info.hx +@@ -307,6 +307,19 @@ SRST + Show KVM information. + ERST + ++ { ++ .name = "mshv", ++ .args_type = "", ++ .params = "", ++ .help = "show MSHV information", ++ .cmd = hmp_info_mshv, ++ }, ++ ++SRST ++ ``info mshv`` ++ Show MSHV information. ++ERST ++ + { + .name = "numa", + .args_type = "", +diff --git a/hw/core/machine-hmp-cmds.c b/hw/core/machine-hmp-cmds.c +index 3a612e2232..682ed9f49b 100644 +--- a/hw/core/machine-hmp-cmds.c ++++ b/hw/core/machine-hmp-cmds.c +@@ -163,6 +163,21 @@ void hmp_info_kvm(Monitor *mon, const QDict *qdict) + qapi_free_KvmInfo(info); + } + ++void hmp_info_mshv(Monitor *mon, const QDict *qdict) ++{ ++ MshvInfo *info; ++ ++ info = qmp_query_mshv(NULL); ++ monitor_printf(mon, "mshv support: "); ++ if (info->present) { ++ monitor_printf(mon, "%s\n", info->enabled ? "enabled" : "disabled"); ++ } else { ++ monitor_printf(mon, "not compiled\n"); ++ } ++ ++ qapi_free_MshvInfo(info); ++} ++ + void hmp_info_uuid(Monitor *mon, const QDict *qdict) + { + UuidInfo *info; +diff --git a/hw/core/machine-qmp-cmds.c b/hw/core/machine-qmp-cmds.c +index 6aca1a626e..e24bf0d97b 100644 +--- a/hw/core/machine-qmp-cmds.c ++++ b/hw/core/machine-qmp-cmds.c +@@ -28,6 +28,20 @@ + #include "system/runstate.h" + #include "system/system.h" + #include "hw/s390x/storage-keys.h" ++#include ++ ++/* ++ * QMP query for MSHV ++ */ ++MshvInfo *qmp_query_mshv(Error **errp) ++{ ++ MshvInfo *info = g_malloc0(sizeof(*info)); ++ ++ info->enabled = mshv_enabled(); ++ info->present = accel_find("mshv"); ++ ++ return info; ++} + + /* + * fast means: we NEVER interrupt vCPU threads to retrieve +diff --git a/include/monitor/hmp.h b/include/monitor/hmp.h +index ae116d9804..31bd812e5f 100644 +--- a/include/monitor/hmp.h ++++ b/include/monitor/hmp.h +@@ -24,6 +24,7 @@ strList *hmp_split_at_comma(const char *str); + void hmp_info_name(Monitor *mon, const QDict *qdict); + void hmp_info_version(Monitor *mon, const QDict *qdict); + void hmp_info_kvm(Monitor *mon, const QDict *qdict); ++void hmp_info_mshv(Monitor *mon, const QDict *qdict); + void hmp_info_status(Monitor *mon, const QDict *qdict); + void hmp_info_uuid(Monitor *mon, const QDict *qdict); + void hmp_info_chardev(Monitor *mon, const QDict *qdict); +diff --git a/include/system/hw_accel.h b/include/system/hw_accel.h +index fa9228d5d2..55497edc29 100644 +--- a/include/system/hw_accel.h ++++ b/include/system/hw_accel.h +@@ -14,6 +14,7 @@ + #include "hw/core/cpu.h" + #include "system/kvm.h" + #include "system/hvf.h" ++#include "system/mshv.h" + #include "system/whpx.h" + #include "system/nvmm.h" + +diff --git a/qapi/accelerator.json b/qapi/accelerator.json +index fb28c8d920..664e027246 100644 +--- a/qapi/accelerator.json ++++ b/qapi/accelerator.json +@@ -54,3 +54,32 @@ + { 'command': 'x-accel-stats', + 'returns': 'HumanReadableText', + 'features': [ 'unstable' ] } ++ ++## ++# @MshvInfo: ++# ++# Information about support for MSHV acceleration ++# ++# @enabled: true if MSHV acceleration is active ++# ++# @present: true if MSHV acceleration is built into this executable ++# ++# Since: 10.2.0 ++## ++{ 'struct': 'MshvInfo', 'data': {'enabled': 'bool', 'present': 'bool'} } ++ ++## ++# @query-mshv: ++# ++# Return information about MSHV acceleration ++# ++# Returns: @MshvInfo ++# ++# Since: 10.0.92 ++# ++# .. qmp-example:: ++# ++# -> { "execute": "query-mshv" } ++# <- { "return": { "enabled": true, "present": true } } ++## ++{ 'command': 'query-mshv', 'returns': 'MshvInfo' } +-- +2.47.3 + diff --git a/kvm-qapi-machine-s390x-add-QAPI-event-SCLP_CPI_INFO_AVAI.patch b/kvm-qapi-machine-s390x-add-QAPI-event-SCLP_CPI_INFO_AVAI.patch new file mode 100644 index 0000000..796cb88 --- /dev/null +++ b/kvm-qapi-machine-s390x-add-QAPI-event-SCLP_CPI_INFO_AVAI.patch @@ -0,0 +1,87 @@ +From c616ee68c53e1ebc10f95e3401e505c034aeb361 Mon Sep 17 00:00:00 2001 +From: Shalini Chellathurai Saroja +Date: Thu, 16 Oct 2025 14:17:07 +0200 +Subject: [PATCH 01/10] qapi/machine-s390x: add QAPI event + SCLP_CPI_INFO_AVAILABLE +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 412: Add -rhel9.8.0 and -rhel10.2.0 s390x machine types and enable the CPI feature +RH-Jira: RHEL-104009 RHEL-105823 RHEL-73008 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/3] 24789cc7501d22168819bf459f7d880deba8b403 (thuth/qemu-kvm-cs) + +JIRA: https://issues.redhat.com/browse/RHEL-73008 + +Add QAPI event SCLP_CPI_INFO_AVAILABLE to notify the availability +of Control-Program Identification data in QOM. + +Signed-off-by: Shalini Chellathurai Saroja +Suggested-by: Thomas Huth +Reviewed-by: Hendrik Brueckner +Reviewed-by: Thomas Huth +Message-ID: <20251016121708.334133-1-shalini@linux.ibm.com> +Signed-off-by: Thomas Huth +(cherry picked from commit 0ce63280dc6fe9fce1d89922f2d17dcae77827e6) +--- + hw/s390x/sclpcpi.c | 4 ++++ + qapi/machine-s390x.json | 21 +++++++++++++++++++++ + 2 files changed, 25 insertions(+) + +diff --git a/hw/s390x/sclpcpi.c b/hw/s390x/sclpcpi.c +index 7aa039d510..68fc1b809b 100644 +--- a/hw/s390x/sclpcpi.c ++++ b/hw/s390x/sclpcpi.c +@@ -54,6 +54,7 @@ + #include "hw/s390x/event-facility.h" + #include "hw/s390x/ebcdic.h" + #include "qapi/qapi-visit-machine.h" ++#include "qapi/qapi-events-machine-s390x.h" + #include "migration/vmstate.h" + + typedef struct Data { +@@ -106,6 +107,9 @@ static int write_event_data(SCLPEvent *event, EventBufferHeader *evt_buf_hdr) + e->timestamp = qemu_clock_get_ns(QEMU_CLOCK_HOST); + + cpim->ebh.flags = SCLP_EVENT_BUFFER_ACCEPTED; ++ ++ qapi_event_send_sclp_cpi_info_available(); ++ + return SCLP_RC_NORMAL_COMPLETION; + } + +diff --git a/qapi/machine-s390x.json b/qapi/machine-s390x.json +index 966dbd61d2..8412668b67 100644 +--- a/qapi/machine-s390x.json ++++ b/qapi/machine-s390x.json +@@ -119,3 +119,24 @@ + { 'command': 'query-s390x-cpu-polarization', 'returns': 'CpuPolarizationInfo', + 'features': [ 'unstable' ] + } ++ ++## ++# @SCLP_CPI_INFO_AVAILABLE: ++# ++# Emitted when the Control-Program Identification data is available ++# in the QOM tree. ++# ++# Features: ++# ++# @unstable: This event is experimental. ++# ++# Since: 10.2 ++# ++# .. qmp-example:: ++# ++# <- { "event": "SCLP_CPI_INFO_AVAILABLE", ++# "timestamp": { "seconds": 1401385907, "microseconds": 422329 } } ++## ++{ 'event': 'SCLP_CPI_INFO_AVAILABLE', ++ 'features': [ 'unstable' ] ++} +-- +2.47.3 + diff --git a/kvm-qcow2-Don-t-open-data_file-with-BDRV_O_NO_IO.patch b/kvm-qcow2-Don-t-open-data_file-with-BDRV_O_NO_IO.patch deleted file mode 100644 index 71d0bfe..0000000 --- a/kvm-qcow2-Don-t-open-data_file-with-BDRV_O_NO_IO.patch +++ /dev/null @@ -1,117 +0,0 @@ -From 57ec055ce7615d4838ae19c4980c2a1799c6cb3d Mon Sep 17 00:00:00 2001 -From: Kevin Wolf -Date: Thu, 11 Apr 2024 15:06:01 +0200 -Subject: [PATCH 1/4] qcow2: Don't open data_file with BDRV_O_NO_IO - -RH-Author: Hana Czenczek -RH-MergeRequest: 1: CVE 2024-4467 (PRDSC) -RH-Jira: RHEL-46239 -RH-CVE: CVE-2024-4467 -RH-Acked-by: Kevin Wolf -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Eric Blake -RH-Commit: [1/4] f9843ce5c519901654a7d8ba43ee95ce25ca13c2 - -One use case for 'qemu-img info' is verifying that untrusted images -don't reference an unwanted external file, be it as a backing file or an -external data file. To make sure that calling 'qemu-img info' can't -already have undesired side effects with a malicious image, just don't -open the data file at all with BDRV_O_NO_IO. If nothing ever tries to do -I/O, we don't need to have it open. - -This changes the output of iotests case 061, which used 'qemu-img info' -to show that opening an image with an invalid data file fails. After -this patch, it succeeds. Replace this part of the test with a qemu-io -call, but keep the final 'qemu-img info' to show that the invalid data -file is correctly displayed in the output. - -Signed-off-by: Kevin Wolf -Reviewed-by: Eric Blake -Reviewed-by: Stefan Hajnoczi -Reviewed-by: Hanna Czenczek -Upstream: N/A, embargoed -Signed-off-by: Hanna Czenczek ---- - block/qcow2.c | 17 ++++++++++++++++- - tests/qemu-iotests/061 | 6 ++++-- - tests/qemu-iotests/061.out | 8 ++++++-- - 3 files changed, 26 insertions(+), 5 deletions(-) - -diff --git a/block/qcow2.c b/block/qcow2.c -index 0e8b2f7518..3b8d2db9f9 100644 ---- a/block/qcow2.c -+++ b/block/qcow2.c -@@ -1642,7 +1642,22 @@ qcow2_do_open(BlockDriverState *bs, QDict *options, int flags, - goto fail; - } - -- if (open_data_file) { -+ if (open_data_file && (flags & BDRV_O_NO_IO)) { -+ /* -+ * Don't open the data file for 'qemu-img info' so that it can be used -+ * to verify that an untrusted qcow2 image doesn't refer to external -+ * files. -+ * -+ * Note: This still makes has_data_file() return true. -+ */ -+ if (s->incompatible_features & QCOW2_INCOMPAT_DATA_FILE) { -+ s->data_file = NULL; -+ } else { -+ s->data_file = bs->file; -+ } -+ qdict_extract_subqdict(options, NULL, "data-file."); -+ qdict_del(options, "data-file"); -+ } else if (open_data_file) { - /* Open external data file */ - bdrv_graph_co_rdunlock(); - s->data_file = bdrv_co_open_child(NULL, options, "data-file", bs, -diff --git a/tests/qemu-iotests/061 b/tests/qemu-iotests/061 -index 53c7d428e3..b71ac097d1 100755 ---- a/tests/qemu-iotests/061 -+++ b/tests/qemu-iotests/061 -@@ -326,12 +326,14 @@ $QEMU_IMG amend -o "data_file=foo" "$TEST_IMG" - echo - _make_test_img -o "compat=1.1,data_file=$TEST_IMG.data" 64M - $QEMU_IMG amend -o "data_file=foo" "$TEST_IMG" --_img_info --format-specific -+$QEMU_IO -c "read 0 4k" "$TEST_IMG" 2>&1 | _filter_testdir | _filter_imgfmt -+$QEMU_IO -c "open -o data-file.filename=$TEST_IMG.data,file.filename=$TEST_IMG" -c "read 0 4k" | _filter_qemu_io - TEST_IMG="data-file.filename=$TEST_IMG.data,file.filename=$TEST_IMG" _img_info --format-specific --image-opts - - echo - $QEMU_IMG amend -o "data_file=" --image-opts "data-file.filename=$TEST_IMG.data,file.filename=$TEST_IMG" --_img_info --format-specific -+$QEMU_IO -c "read 0 4k" "$TEST_IMG" 2>&1 | _filter_testdir | _filter_imgfmt -+$QEMU_IO -c "open -o data-file.filename=$TEST_IMG.data,file.filename=$TEST_IMG" -c "read 0 4k" | _filter_qemu_io - TEST_IMG="data-file.filename=$TEST_IMG.data,file.filename=$TEST_IMG" _img_info --format-specific --image-opts - - echo -diff --git a/tests/qemu-iotests/061.out b/tests/qemu-iotests/061.out -index 139fc68177..24c33add7c 100644 ---- a/tests/qemu-iotests/061.out -+++ b/tests/qemu-iotests/061.out -@@ -545,7 +545,9 @@ Formatting 'TEST_DIR/t.IMGFMT', fmt=IMGFMT size=67108864 - qemu-img: data-file can only be set for images that use an external data file - - Formatting 'TEST_DIR/t.IMGFMT', fmt=IMGFMT size=67108864 data_file=TEST_DIR/t.IMGFMT.data --qemu-img: Could not open 'TEST_DIR/t.IMGFMT': Could not open 'foo': No such file or directory -+qemu-io: can't open device TEST_DIR/t.IMGFMT: Could not open 'foo': No such file or directory -+read 4096/4096 bytes at offset 0 -+4 KiB, X ops; XX:XX:XX.X (XXX YYY/sec and XXX ops/sec) - image: TEST_DIR/t.IMGFMT - file format: IMGFMT - virtual size: 64 MiB (67108864 bytes) -@@ -560,7 +562,9 @@ Format specific information: - corrupt: false - extended l2: false - --qemu-img: Could not open 'TEST_DIR/t.IMGFMT': 'data-file' is required for this image -+qemu-io: can't open device TEST_DIR/t.IMGFMT: 'data-file' is required for this image -+read 4096/4096 bytes at offset 0 -+4 KiB, X ops; XX:XX:XX.X (XXX YYY/sec and XXX ops/sec) - image: TEST_DIR/t.IMGFMT - file format: IMGFMT - virtual size: 64 MiB (67108864 bytes) --- -2.39.3 - diff --git a/kvm-qcow2-Fix-cache_clean_timer.patch b/kvm-qcow2-Fix-cache_clean_timer.patch new file mode 100644 index 0000000..00befca --- /dev/null +++ b/kvm-qcow2-Fix-cache_clean_timer.patch @@ -0,0 +1,338 @@ +From ebcb7dbc060ca295d2ec1daba67ba84fe2ede3bc Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Mon, 10 Nov 2025 16:48:47 +0100 +Subject: [PATCH 05/19] qcow2: Fix cache_clean_timer +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Hanna Czenczek +RH-MergeRequest: 454: Multithreading fixes for rbd, curl, qcow2 +RH-Jira: RHEL-79118 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Miroslav Rezanina +RH-Commit: [5/5] ff4bbc4f7c7cebeb8ba34a146854aca1f8aed86f (hreitz/qemu-kvm-c-9-s) + +The cache-cleaner runs as a timer CB in the BDS AioContext. With +multiqueue, it can run concurrently to I/O requests, and because it does +not take any lock, this can break concurrent cache accesses, corrupting +the image. While the chances of this happening are low, it can be +reproduced e.g. by modifying the code to schedule the timer CB every +5 ms (instead of at most once per second) and modifying the last (inner) +while loop of qcow2_cache_clean_unused() like so: + + while (i < c->size && can_clean_entry(c, i)) { + for (int j = 0; j < 1000 && can_clean_entry(c, i); j++) { + usleep(100); + } + c->entries[i].offset = 0; + c->entries[i].lru_counter = 0; + i++; + to_clean++; + } + +i.e. making it wait on purpose for the point in time where the cache is +in use by something else. + +The solution chosen for this in this patch is not the best solution, I +hope, but I admittedly can’t come up with anything strictly better. + +We can protect from concurrent cache accesses either by taking the +existing s->lock, or we introduce a new (non-coroutine) mutex +specifically for cache accesses. I would prefer to avoid the latter so +as not to introduce additional (very slight) overhead. + +Using s->lock, which is a coroutine mutex, however means that we need to +take it in a coroutine, so the timer must run in a coroutine. We can +transform it from the current timer CB style into a coroutine that +sleeps for the set interval. As a result, however, we can no longer +just deschedule the timer to instantly guarantee it won’t run anymore, +but have to await the coroutine’s exit. + +(Note even before this patch there were places that may not have been so +guaranteed after all: Anything calling cache_clean_timer_del() from the +QEMU main AioContext could have been running concurrently to an existing +timer CB invocation.) + +Polling to await the timer to actually settle seems very complicated for +something that’s rather a minor problem, but I can’t come up with any +better solution that doesn’t again just overlook potential problems. + +(Not Cc-ing qemu-stable, as the issue is quite unlikely to be hit, and +I’m not too fond of this solution.) + +Signed-off-by: Hanna Czenczek +Message-ID: <20251110154854.151484-13-hreitz@redhat.com> +Reviewed-by: Kevin Wolf +Signed-off-by: Kevin Wolf +(cherry picked from commit f86dde9a1524f0d7fca0f4037f560e6131d31f7f) +Signed-off-by: Hanna Czenczek +--- + block/qcow2.c | 141 ++++++++++++++++++++++++++++++++++++++++---------- + block/qcow2.h | 5 +- + 2 files changed, 117 insertions(+), 29 deletions(-) + +diff --git a/block/qcow2.c b/block/qcow2.c +index 374ca53829..a6adfa9f84 100644 +--- a/block/qcow2.c ++++ b/block/qcow2.c +@@ -835,41 +835,113 @@ static const char *overlap_bool_option_names[QCOW2_OL_MAX_BITNR] = { + [QCOW2_OL_BITMAP_DIRECTORY_BITNR] = QCOW2_OPT_OVERLAP_BITMAP_DIRECTORY, + }; + +-static void cache_clean_timer_cb(void *opaque) ++static void coroutine_fn cache_clean_timer(void *opaque) + { +- BlockDriverState *bs = opaque; +- BDRVQcow2State *s = bs->opaque; +- qcow2_cache_clean_unused(s->l2_table_cache); +- qcow2_cache_clean_unused(s->refcount_block_cache); +- timer_mod(s->cache_clean_timer, qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL) + +- (int64_t) s->cache_clean_interval * 1000); ++ BDRVQcow2State *s = opaque; ++ uint64_t wait_ns; ++ ++ WITH_QEMU_LOCK_GUARD(&s->lock) { ++ wait_ns = s->cache_clean_interval * NANOSECONDS_PER_SECOND; ++ } ++ ++ while (wait_ns > 0) { ++ qemu_co_sleep_ns_wakeable(&s->cache_clean_timer_wake, ++ QEMU_CLOCK_VIRTUAL, wait_ns); ++ ++ WITH_QEMU_LOCK_GUARD(&s->lock) { ++ if (s->cache_clean_interval > 0) { ++ qcow2_cache_clean_unused(s->l2_table_cache); ++ qcow2_cache_clean_unused(s->refcount_block_cache); ++ } ++ ++ wait_ns = s->cache_clean_interval * NANOSECONDS_PER_SECOND; ++ } ++ } ++ ++ WITH_QEMU_LOCK_GUARD(&s->lock) { ++ s->cache_clean_timer_co = NULL; ++ qemu_co_queue_restart_all(&s->cache_clean_timer_exit); ++ } + } + + static void cache_clean_timer_init(BlockDriverState *bs, AioContext *context) + { + BDRVQcow2State *s = bs->opaque; + if (s->cache_clean_interval > 0) { +- s->cache_clean_timer = +- aio_timer_new_with_attrs(context, QEMU_CLOCK_VIRTUAL, +- SCALE_MS, QEMU_TIMER_ATTR_EXTERNAL, +- cache_clean_timer_cb, bs); +- timer_mod(s->cache_clean_timer, qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL) + +- (int64_t) s->cache_clean_interval * 1000); ++ assert(!s->cache_clean_timer_co); ++ s->cache_clean_timer_co = qemu_coroutine_create(cache_clean_timer, s); ++ aio_co_enter(context, s->cache_clean_timer_co); + } + } + +-static void cache_clean_timer_del(BlockDriverState *bs) ++/** ++ * Delete the cache clean timer and await any yet running instance. ++ * Called holding s->lock. ++ */ ++static void coroutine_fn ++cache_clean_timer_co_locked_del_and_wait(BlockDriverState *bs) ++{ ++ BDRVQcow2State *s = bs->opaque; ++ ++ if (s->cache_clean_timer_co) { ++ s->cache_clean_interval = 0; ++ qemu_co_sleep_wake(&s->cache_clean_timer_wake); ++ qemu_co_queue_wait(&s->cache_clean_timer_exit, &s->lock); ++ } ++} ++ ++/** ++ * Same as cache_clean_timer_co_locked_del_and_wait(), but takes s->lock. ++ */ ++static void coroutine_fn ++cache_clean_timer_co_del_and_wait(BlockDriverState *bs) + { + BDRVQcow2State *s = bs->opaque; +- if (s->cache_clean_timer) { +- timer_free(s->cache_clean_timer); +- s->cache_clean_timer = NULL; ++ ++ WITH_QEMU_LOCK_GUARD(&s->lock) { ++ cache_clean_timer_co_locked_del_and_wait(bs); ++ } ++} ++ ++struct CacheCleanTimerDelAndWaitCoParams { ++ BlockDriverState *bs; ++ bool done; ++}; ++ ++static void coroutine_fn cache_clean_timer_del_and_wait_co_entry(void *opaque) ++{ ++ struct CacheCleanTimerDelAndWaitCoParams *p = opaque; ++ ++ cache_clean_timer_co_del_and_wait(p->bs); ++ p->done = true; ++ aio_wait_kick(); ++} ++ ++/** ++ * Delete the cache clean timer and await any yet running instance. ++ * Must be called from the main or BDS AioContext without s->lock held. ++ */ ++static void coroutine_mixed_fn ++cache_clean_timer_del_and_wait(BlockDriverState *bs) ++{ ++ IO_OR_GS_CODE(); ++ ++ if (qemu_in_coroutine()) { ++ cache_clean_timer_co_del_and_wait(bs); ++ } else { ++ struct CacheCleanTimerDelAndWaitCoParams p = { .bs = bs }; ++ Coroutine *co; ++ ++ co = qemu_coroutine_create(cache_clean_timer_del_and_wait_co_entry, &p); ++ qemu_coroutine_enter(co); ++ ++ BDRV_POLL_WHILE(bs, !p.done); + } + } + + static void qcow2_detach_aio_context(BlockDriverState *bs) + { +- cache_clean_timer_del(bs); ++ cache_clean_timer_del_and_wait(bs); + } + + static void qcow2_attach_aio_context(BlockDriverState *bs, +@@ -1214,12 +1286,24 @@ fail: + return ret; + } + ++/* s_locked specifies whether s->lock is held or not */ + static void qcow2_update_options_commit(BlockDriverState *bs, +- Qcow2ReopenState *r) ++ Qcow2ReopenState *r, ++ bool s_locked) + { + BDRVQcow2State *s = bs->opaque; + int i; + ++ /* ++ * We need to stop the cache-clean-timer before destroying the metadata ++ * table caches ++ */ ++ if (s_locked) { ++ cache_clean_timer_co_locked_del_and_wait(bs); ++ } else { ++ cache_clean_timer_del_and_wait(bs); ++ } ++ + if (s->l2_table_cache) { + qcow2_cache_destroy(s->l2_table_cache); + } +@@ -1228,6 +1312,7 @@ static void qcow2_update_options_commit(BlockDriverState *bs, + } + s->l2_table_cache = r->l2_table_cache; + s->refcount_block_cache = r->refcount_block_cache; ++ + s->l2_slice_size = r->l2_slice_size; + + s->overlap_check = r->overlap_check; +@@ -1239,11 +1324,8 @@ static void qcow2_update_options_commit(BlockDriverState *bs, + + s->discard_no_unref = r->discard_no_unref; + +- if (s->cache_clean_interval != r->cache_clean_interval) { +- cache_clean_timer_del(bs); +- s->cache_clean_interval = r->cache_clean_interval; +- cache_clean_timer_init(bs, bdrv_get_aio_context(bs)); +- } ++ s->cache_clean_interval = r->cache_clean_interval; ++ cache_clean_timer_init(bs, bdrv_get_aio_context(bs)); + + qapi_free_QCryptoBlockOpenOptions(s->crypto_opts); + s->crypto_opts = r->crypto_opts; +@@ -1261,6 +1343,7 @@ static void qcow2_update_options_abort(BlockDriverState *bs, + qapi_free_QCryptoBlockOpenOptions(r->crypto_opts); + } + ++/* Called with s->lock held */ + static int coroutine_fn GRAPH_RDLOCK + qcow2_update_options(BlockDriverState *bs, QDict *options, int flags, + Error **errp) +@@ -1270,7 +1353,7 @@ qcow2_update_options(BlockDriverState *bs, QDict *options, int flags, + + ret = qcow2_update_options_prepare(bs, &r, options, flags, errp); + if (ret >= 0) { +- qcow2_update_options_commit(bs, &r); ++ qcow2_update_options_commit(bs, &r, true); + } else { + qcow2_update_options_abort(bs, &r); + } +@@ -1914,7 +1997,7 @@ qcow2_do_open(BlockDriverState *bs, QDict *options, int flags, + qemu_vfree(s->l1_table); + /* else pre-write overlap checks in cache_destroy may crash */ + s->l1_table = NULL; +- cache_clean_timer_del(bs); ++ cache_clean_timer_co_locked_del_and_wait(bs); + if (s->l2_table_cache) { + qcow2_cache_destroy(s->l2_table_cache); + } +@@ -1969,6 +2052,7 @@ static int qcow2_open(BlockDriverState *bs, QDict *options, int flags, + + /* Initialise locks */ + qemu_co_mutex_init(&s->lock); ++ qemu_co_queue_init(&s->cache_clean_timer_exit); + + assert(!qemu_in_coroutine()); + assert(qemu_get_current_aio_context() == qemu_get_aio_context()); +@@ -2054,7 +2138,7 @@ static void qcow2_reopen_commit(BDRVReopenState *state) + + GRAPH_RDLOCK_GUARD_MAINLOOP(); + +- qcow2_update_options_commit(state->bs, state->opaque); ++ qcow2_update_options_commit(state->bs, state->opaque, false); + if (!s->data_file) { + /* + * If we don't have an external data file, s->data_file was cleared by +@@ -2811,7 +2895,7 @@ qcow2_do_close(BlockDriverState *bs, bool close_data_file) + qcow2_inactivate(bs); + } + +- cache_clean_timer_del(bs); ++ cache_clean_timer_del_and_wait(bs); + qcow2_cache_destroy(s->l2_table_cache); + qcow2_cache_destroy(s->refcount_block_cache); + +@@ -2881,6 +2965,7 @@ qcow2_co_invalidate_cache(BlockDriverState *bs, Error **errp) + s->data_file = data_file; + /* Re-initialize objects initialized in qcow2_open() */ + qemu_co_mutex_init(&s->lock); ++ qemu_co_queue_init(&s->cache_clean_timer_exit); + + options = qdict_clone_shallow(bs->options); + +diff --git a/block/qcow2.h b/block/qcow2.h +index a9e3481c6e..3e38bccd87 100644 +--- a/block/qcow2.h ++++ b/block/qcow2.h +@@ -345,8 +345,11 @@ typedef struct BDRVQcow2State { + + Qcow2Cache *l2_table_cache; + Qcow2Cache *refcount_block_cache; +- QEMUTimer *cache_clean_timer; ++ /* Non-NULL while the timer is running */ ++ Coroutine *cache_clean_timer_co; + unsigned cache_clean_interval; ++ QemuCoSleep cache_clean_timer_wake; ++ CoQueue cache_clean_timer_exit; + + QLIST_HEAD(, QCowL2Meta) cluster_allocs; + +-- +2.47.3 + diff --git a/kvm-qcow2-Re-initialize-lock-in-invalidate_cache.patch b/kvm-qcow2-Re-initialize-lock-in-invalidate_cache.patch new file mode 100644 index 0000000..738ae4a --- /dev/null +++ b/kvm-qcow2-Re-initialize-lock-in-invalidate_cache.patch @@ -0,0 +1,45 @@ +From cfe4b6407a08b56490c3cadd4e01f810fb22070c Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Mon, 10 Nov 2025 16:48:46 +0100 +Subject: [PATCH 04/19] qcow2: Re-initialize lock in invalidate_cache + +RH-Author: Hanna Czenczek +RH-MergeRequest: 454: Multithreading fixes for rbd, curl, qcow2 +RH-Jira: RHEL-79118 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Miroslav Rezanina +RH-Commit: [4/5] 48ead2f35d8558fda48a2098240c0203f4c547b0 (hreitz/qemu-kvm-c-9-s) + +After clearing our state (memset()-ing it to 0), we should +re-initialize objects that need it. Specifically, that applies to +s->lock, which is originally initialized in qcow2_open(). + +Given qemu_co_mutex_init() is just a memset() to 0, this is functionally +a no-op, but still seems like the right thing to do. + +Signed-off-by: Hanna Czenczek +Message-ID: <20251110154854.151484-12-hreitz@redhat.com> +Reviewed-by: Kevin Wolf +Signed-off-by: Kevin Wolf +(cherry picked from commit 90db3a1721b30daf839901813616c0854f383cc5) +Signed-off-by: Hanna Czenczek +--- + block/qcow2.c | 2 ++ + 1 file changed, 2 insertions(+) + +diff --git a/block/qcow2.c b/block/qcow2.c +index 6df65aab93..374ca53829 100644 +--- a/block/qcow2.c ++++ b/block/qcow2.c +@@ -2879,6 +2879,8 @@ qcow2_co_invalidate_cache(BlockDriverState *bs, Error **errp) + data_file = s->data_file; + memset(s, 0, sizeof(BDRVQcow2State)); + s->data_file = data_file; ++ /* Re-initialize objects initialized in qcow2_open() */ ++ qemu_co_mutex_init(&s->lock); + + options = qdict_clone_shallow(bs->options); + +-- +2.47.3 + diff --git a/kvm-qemu-img-info-Add-cache-mode-option.patch b/kvm-qemu-img-info-Add-cache-mode-option.patch new file mode 100644 index 0000000..c685595 --- /dev/null +++ b/kvm-qemu-img-info-Add-cache-mode-option.patch @@ -0,0 +1,153 @@ +From 67461afaa7e25c6a41cb542ecc1a2b8696d5489d Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Fri, 24 Oct 2025 14:30:40 +0200 +Subject: [PATCH 5/6] qemu-img info: Add cache mode option + +RH-Author: Kevin Wolf +RH-MergeRequest: 439: block: Expose block limits in monitor and qemu-img info +RH-Jira: RHEL-110003 +RH-Acked-by: Hanna Czenczek +RH-Acked-by: Miroslav Rezanina +RH-Commit: [4/4] 1d7c278d70f37528be71b31fe121cc531a13dde5 (kmwolf/centos-qemu-kvm) + +When querying block limits, different cache modes (in particular +O_DIRECT or not) can result in different limits. Add an option to +'qemu-img info' that allows the user to specify a cache mode, so that +they can get the block limits for the cache mode they intend to use with +their VM. + +Signed-off-by: Kevin Wolf +Message-ID: <20251024123041.51254-5-kwolf@redhat.com> +Reviewed-by: Eric Blake +Signed-off-by: Kevin Wolf +(cherry picked from commit 911992fd6ec7a84c7cc82831b4bcd8a2ca5ccc76) +Signed-off-by: Kevin Wolf +--- + docs/tools/qemu-img.rst | 2 +- + qemu-img-cmds.hx | 4 ++-- + qemu-img.c | 25 +++++++++++++++++++++---- + 3 files changed, 24 insertions(+), 7 deletions(-) + +diff --git a/docs/tools/qemu-img.rst b/docs/tools/qemu-img.rst +index fdc9ea9cf2..558b0eb84d 100644 +--- a/docs/tools/qemu-img.rst ++++ b/docs/tools/qemu-img.rst +@@ -503,7 +503,7 @@ Command description: + + The size syntax is similar to :manpage:`dd(1)`'s size syntax. + +-.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [--limits] [-U] FILENAME ++.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [--limits] [-t CACHE] [-U] FILENAME + + Give information about the disk image *FILENAME*. Use it in + particular to know the size reserved on disk which can be different +diff --git a/qemu-img-cmds.hx b/qemu-img-cmds.hx +index 74b66f9d42..6bc8265cfb 100644 +--- a/qemu-img-cmds.hx ++++ b/qemu-img-cmds.hx +@@ -66,9 +66,9 @@ SRST + ERST + + DEF("info", img_info, +- "info [--object objectdef] [--image-opts] [-f fmt] [--output=ofmt] [--backing-chain] [--limits] [-U] filename") ++ "info [--object objectdef] [--image-opts] [-f fmt] [--output=ofmt] [--backing-chain] [--limits] [-t CACHE] [-U] filename") + SRST +-.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [--limits] [-U] FILENAME ++.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [--limits] [-t CACHE] [-U] FILENAME + ERST + + DEF("map", img_map, +diff --git a/qemu-img.c b/qemu-img.c +index 5cdbeda969..a7791896c1 100644 +--- a/qemu-img.c ++++ b/qemu-img.c +@@ -3003,6 +3003,7 @@ static gboolean str_equal_func(gconstpointer a, gconstpointer b) + static BlockGraphInfoList *collect_image_info_list(bool image_opts, + const char *filename, + const char *fmt, ++ const char *cache, + bool chain, bool limits, + bool force_share) + { +@@ -3010,6 +3011,15 @@ static BlockGraphInfoList *collect_image_info_list(bool image_opts, + BlockGraphInfoList **tail = &head; + GHashTable *filenames; + Error *err = NULL; ++ int cache_flags = 0; ++ bool writethrough = false; ++ int ret; ++ ++ ret = bdrv_parse_cache_mode(cache, &cache_flags, &writethrough); ++ if (ret < 0) { ++ error_report("Invalid cache option: %s", cache); ++ return NULL; ++ } + + filenames = g_hash_table_new_full(g_str_hash, str_equal_func, NULL, NULL); + +@@ -3026,8 +3036,8 @@ static BlockGraphInfoList *collect_image_info_list(bool image_opts, + g_hash_table_insert(filenames, (gpointer)filename, NULL); + + blk = img_open(image_opts, filename, fmt, +- BDRV_O_NO_BACKING | BDRV_O_NO_IO, false, false, +- force_share); ++ BDRV_O_NO_BACKING | BDRV_O_NO_IO | cache_flags, ++ writethrough, false, force_share); + if (!blk) { + goto err; + } +@@ -3087,6 +3097,7 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + OutputFormat output_format = OFORMAT_HUMAN; + bool chain = false; + const char *filename, *fmt; ++ const char *cache = BDRV_DEFAULT_CACHE; + BlockGraphInfoList *list; + bool image_opts = false; + bool force_share = false; +@@ -3099,13 +3110,14 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + {"format", required_argument, 0, 'f'}, + {"image-opts", no_argument, 0, OPTION_IMAGE_OPTS}, + {"backing-chain", no_argument, 0, OPTION_BACKING_CHAIN}, ++ {"cache", required_argument, 0, 't'}, + {"force-share", no_argument, 0, 'U'}, + {"limits", no_argument, 0, OPTION_LIMITS}, + {"output", required_argument, 0, OPTION_OUTPUT}, + {"object", required_argument, 0, OPTION_OBJECT}, + {0, 0, 0, 0} + }; +- c = getopt_long(argc, argv, "hf:U", long_options, NULL); ++ c = getopt_long(argc, argv, "hf:t:U", long_options, NULL); + if (c == -1) { + break; + } +@@ -3121,6 +3133,8 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + " (incompatible with -f|--format)\n" + " --backing-chain\n" + " display information about the backing chain for copy-on-write overlays\n" ++" -t, --cache CACHE\n" ++" cache mode for FILE (default: " BDRV_DEFAULT_CACHE ")\n" + " -U, --force-share\n" + " open image in shared mode for concurrent access\n" + " --limits\n" +@@ -3143,6 +3157,9 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + case OPTION_BACKING_CHAIN: + chain = true; + break; ++ case 't': ++ cache = optarg; ++ break; + case 'U': + force_share = true; + break; +@@ -3164,7 +3181,7 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + } + filename = argv[optind++]; + +- list = collect_image_info_list(image_opts, filename, fmt, chain, ++ list = collect_image_info_list(image_opts, filename, fmt, cache, chain, + limits, force_share); + if (!list) { + return 1; +-- +2.47.3 + diff --git a/kvm-qemu-img-info-Optionally-show-block-limits.patch b/kvm-qemu-img-info-Optionally-show-block-limits.patch new file mode 100644 index 0000000..c528e64 --- /dev/null +++ b/kvm-qemu-img-info-Optionally-show-block-limits.patch @@ -0,0 +1,240 @@ +From 264e2c0bff1ae09ef89757eac655401dbdf10d90 Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Fri, 24 Oct 2025 14:30:39 +0200 +Subject: [PATCH 4/6] qemu-img info: Optionally show block limits + +RH-Author: Kevin Wolf +RH-MergeRequest: 439: block: Expose block limits in monitor and qemu-img info +RH-Jira: RHEL-110003 +RH-Acked-by: Hanna Czenczek +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/4] 688d6d1617c4d5916b299f08fad0e928341c73f7 (kmwolf/centos-qemu-kvm) + +Add a new --limits option to 'qemu-img info' that displays the block +limits for the image and all of its children, making the information +more accessible for human users than in QMP. This option is not enabled +by default because it can be a lot of output that isn't usually relevant +if you're not specifically trying to diagnose some I/O problem. + +This makes the same information automatically also available in HMP +'info block -v'. + +Signed-off-by: Kevin Wolf +Reviewed-by: Eric Blake +Reviewed-by: Hanna Czenczek +Message-ID: <20251024123041.51254-4-kwolf@redhat.com> +Signed-off-by: Kevin Wolf +(cherry picked from commit 5b4b3bfdfc28d2398f34194d260d6eef9a9048b4) +Signed-off-by: Kevin Wolf +--- + block/qapi.c | 34 ++++++++++++++++++++++++++++++++-- + docs/tools/qemu-img.rst | 6 +++++- + include/block/qapi.h | 2 +- + qemu-img-cmds.hx | 4 ++-- + qemu-img.c | 15 ++++++++++++--- + 5 files changed, 52 insertions(+), 9 deletions(-) + +diff --git a/block/qapi.c b/block/qapi.c +index 54521d0a68..9f5771e019 100644 +--- a/block/qapi.c ++++ b/block/qapi.c +@@ -417,6 +417,7 @@ fail: + */ + void bdrv_query_block_graph_info(BlockDriverState *bs, + BlockGraphInfo **p_info, ++ bool limits, + Error **errp) + { + ERRP_GUARD(); +@@ -425,7 +426,7 @@ void bdrv_query_block_graph_info(BlockDriverState *bs, + BdrvChild *c; + + info = g_new0(BlockGraphInfo, 1); +- bdrv_do_query_node_info(bs, qapi_BlockGraphInfo_base(info), false, errp); ++ bdrv_do_query_node_info(bs, qapi_BlockGraphInfo_base(info), limits, errp); + if (*errp) { + goto fail; + } +@@ -439,7 +440,7 @@ void bdrv_query_block_graph_info(BlockDriverState *bs, + QAPI_LIST_APPEND(children_list_tail, c_info); + + c_info->name = g_strdup(c->name); +- bdrv_query_block_graph_info(c->bs, &c_info->info, errp); ++ bdrv_query_block_graph_info(c->bs, &c_info->info, limits, errp); + if (*errp) { + goto fail; + } +@@ -936,6 +937,29 @@ void bdrv_image_info_specific_dump(ImageInfoSpecific *info_spec, + visit_free(v); + } + ++/** ++ * Dumps the given BlockLimitsInfo object in a human-readable form, ++ * prepending an optional prefix if the dump is not empty. ++ */ ++static void bdrv_image_info_limits_dump(BlockLimitsInfo *limits, ++ const char *prefix, ++ int indentation) ++{ ++ QObject *obj; ++ Visitor *v = qobject_output_visitor_new(&obj); ++ ++ visit_type_BlockLimitsInfo(v, NULL, &limits, &error_abort); ++ visit_complete(v, &obj); ++ if (!qobject_is_empty_dump(obj)) { ++ if (prefix) { ++ qemu_printf("%*s%s", indentation * 4, "", prefix); ++ } ++ dump_qobject(indentation + 1, obj); ++ } ++ qobject_unref(obj); ++ visit_free(v); ++} ++ + /** + * Print the given @info object in human-readable form. Every field is indented + * using the given @indentation (four spaces per indentation level). +@@ -1011,6 +1035,12 @@ void bdrv_node_info_dump(BlockNodeInfo *info, int indentation, bool protocol) + } + } + ++ if (info->limits) { ++ bdrv_image_info_limits_dump(info->limits, ++ "Block limits:\n", ++ indentation); ++ } ++ + if (info->has_snapshots) { + SnapshotInfoList *elem; + +diff --git a/docs/tools/qemu-img.rst b/docs/tools/qemu-img.rst +index 5e7b85079d..fdc9ea9cf2 100644 +--- a/docs/tools/qemu-img.rst ++++ b/docs/tools/qemu-img.rst +@@ -503,7 +503,7 @@ Command description: + + The size syntax is similar to :manpage:`dd(1)`'s size syntax. + +-.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [-U] FILENAME ++.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [--limits] [-U] FILENAME + + Give information about the disk image *FILENAME*. Use it in + particular to know the size reserved on disk which can be different +@@ -571,6 +571,10 @@ Command description: + ``ImageInfoSpecific*`` QAPI object (e.g. ``ImageInfoSpecificQCow2`` + for qcow2 images). + ++ *Block limits* ++ The block limits for I/O that QEMU detected for the image. ++ This information is only shown if the ``--limits`` option was specified. ++ + .. option:: map [--object OBJECTDEF] [--image-opts] [-f FMT] [--start-offset=OFFSET] [--max-length=LEN] [--output=OFMT] [-U] FILENAME + + Dump the metadata of image *FILENAME* and its backing file chain. +diff --git a/include/block/qapi.h b/include/block/qapi.h +index 54c48de26a..be554e53dc 100644 +--- a/include/block/qapi.h ++++ b/include/block/qapi.h +@@ -42,7 +42,7 @@ bdrv_query_image_info(BlockDriverState *bs, ImageInfo **p_info, bool flat, + bool skip_implicit_filters, Error **errp); + void GRAPH_RDLOCK + bdrv_query_block_graph_info(BlockDriverState *bs, BlockGraphInfo **p_info, +- Error **errp); ++ bool limits, Error **errp); + + void bdrv_snapshot_dump(QEMUSnapshotInfo *sn); + void bdrv_image_info_specific_dump(ImageInfoSpecific *info_spec, +diff --git a/qemu-img-cmds.hx b/qemu-img-cmds.hx +index 2c5a8a28f9..74b66f9d42 100644 +--- a/qemu-img-cmds.hx ++++ b/qemu-img-cmds.hx +@@ -66,9 +66,9 @@ SRST + ERST + + DEF("info", img_info, +- "info [--object objectdef] [--image-opts] [-f fmt] [--output=ofmt] [--backing-chain] [-U] filename") ++ "info [--object objectdef] [--image-opts] [-f fmt] [--output=ofmt] [--backing-chain] [--limits] [-U] filename") + SRST +-.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [-U] FILENAME ++.. option:: info [--object OBJECTDEF] [--image-opts] [-f FMT] [--output=OFMT] [--backing-chain] [--limits] [-U] FILENAME + ERST + + DEF("map", img_map, +diff --git a/qemu-img.c b/qemu-img.c +index 7a162fdc08..5cdbeda969 100644 +--- a/qemu-img.c ++++ b/qemu-img.c +@@ -86,6 +86,7 @@ enum { + OPTION_BITMAPS = 275, + OPTION_FORCE = 276, + OPTION_SKIP_BROKEN = 277, ++ OPTION_LIMITS = 278, + }; + + typedef enum OutputFormat { +@@ -3002,7 +3003,8 @@ static gboolean str_equal_func(gconstpointer a, gconstpointer b) + static BlockGraphInfoList *collect_image_info_list(bool image_opts, + const char *filename, + const char *fmt, +- bool chain, bool force_share) ++ bool chain, bool limits, ++ bool force_share) + { + BlockGraphInfoList *head = NULL; + BlockGraphInfoList **tail = &head; +@@ -3039,7 +3041,7 @@ static BlockGraphInfoList *collect_image_info_list(bool image_opts, + * the chain manually here. + */ + bdrv_graph_rdlock_main_loop(); +- bdrv_query_block_graph_info(bs, &info, &err); ++ bdrv_query_block_graph_info(bs, &info, limits, &err); + bdrv_graph_rdunlock_main_loop(); + + if (err) { +@@ -3088,6 +3090,7 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + BlockGraphInfoList *list; + bool image_opts = false; + bool force_share = false; ++ bool limits = false; + + fmt = NULL; + for(;;) { +@@ -3097,6 +3100,7 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + {"image-opts", no_argument, 0, OPTION_IMAGE_OPTS}, + {"backing-chain", no_argument, 0, OPTION_BACKING_CHAIN}, + {"force-share", no_argument, 0, 'U'}, ++ {"limits", no_argument, 0, OPTION_LIMITS}, + {"output", required_argument, 0, OPTION_OUTPUT}, + {"object", required_argument, 0, OPTION_OBJECT}, + {0, 0, 0, 0} +@@ -3119,6 +3123,8 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + " display information about the backing chain for copy-on-write overlays\n" + " -U, --force-share\n" + " open image in shared mode for concurrent access\n" ++" --limits\n" ++" show detected block limits (may depend on options, e.g. cache mode)\n" + " --output human|json\n" + " specify output format (default: human)\n" + " --object OBJDEF\n" +@@ -3140,6 +3146,9 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + case 'U': + force_share = true; + break; ++ case OPTION_LIMITS: ++ limits = true; ++ break; + case OPTION_OUTPUT: + output_format = parse_output_format(argv[0], optarg); + break; +@@ -3156,7 +3165,7 @@ static int img_info(const img_cmd_t *ccmd, int argc, char **argv) + filename = argv[optind++]; + + list = collect_image_info_list(image_opts, filename, fmt, chain, +- force_share); ++ limits, force_share); + if (!list) { + return 1; + } +-- +2.47.3 + diff --git a/kvm-qemu-options.hx-Document-the-arm-smmuv3-device.patch b/kvm-qemu-options.hx-Document-the-arm-smmuv3-device.patch new file mode 100644 index 0000000..a6fee5b --- /dev/null +++ b/kvm-qemu-options.hx-Document-the-arm-smmuv3-device.patch @@ -0,0 +1,53 @@ +From bf0ecadea242c05671bf057fc45d8c58862032d3 Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:30 +0100 +Subject: [PATCH 12/16] qemu-options.hx: Document the arm-smmuv3 device + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [8/11] 775310b784fd7631a58e8bab1e8fcc36973fceca (eauger1/centos-qemu-kvm) + +Now that arm,virt can have user-creatable smmuv3 devices, document it. + +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Reviewed-by: Nicolin Chen +Message-id: 20250829082543.7680-9-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 73d3d0187bc6b482d8b15116edce1475c7975b89) +Signed-off-by: Eric Auger +--- + qemu-options.hx | 7 +++++++ + 1 file changed, 7 insertions(+) + +diff --git a/qemu-options.hx b/qemu-options.hx +index 3837456a61..5f146c1860 100644 +--- a/qemu-options.hx ++++ b/qemu-options.hx +@@ -1231,6 +1231,13 @@ SRST + ``aw-bits=val`` (val between 32 and 64, default depends on machine) + This decides the address width of the IOVA address space. + ++``-device arm-smmuv3,primary-bus=id`` ++ This is only supported by ``-machine virt`` (ARM). ++ ++ ``primary-bus=id`` ++ Accepts either the default root complex (pcie.0) or a ++ pxb-pcie based root complex. ++ + ERST + + DEF("name", HAS_ARG, QEMU_OPTION_name, +-- +2.47.3 + diff --git a/kvm-qio-Inherit-follow_coroutine_ctx-across-TLS.patch b/kvm-qio-Inherit-follow_coroutine_ctx-across-TLS.patch deleted file mode 100644 index fcbfbfb..0000000 --- a/kvm-qio-Inherit-follow_coroutine_ctx-across-TLS.patch +++ /dev/null @@ -1,130 +0,0 @@ -From 120a2c8a7d936e24948f8f4ada6b781b6cbc9931 Mon Sep 17 00:00:00 2001 -From: Eric Blake -Date: Fri, 17 May 2024 21:50:14 -0500 -Subject: [PATCH 3/4] qio: Inherit follow_coroutine_ctx across TLS -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Eric Blake -RH-MergeRequest: 257: nbd/server: fix TLS negotiation across coroutine context -RH-Jira: RHEL-40959 -RH-Acked-by: Stefan Hajnoczi -RH-Acked-by: Miroslav Rezanina -RH-Commit: [3/4] b7fd03af5985bbc5504b1a8e2f5cd165f6e438e5 (ebblake/centos-qemu-kvm) - -Since qemu 8.2, the combination of NBD + TLS + iothread crashes on an -assertion failure: - -qemu-kvm: ../io/channel.c:534: void qio_channel_restart_read(void *): Assertion `qemu_get_current_aio_context() == qemu_coroutine_get_aio_context(co)' failed. - -It turns out that when we removed AioContext locking, we did so by -having NBD tell its qio channels that it wanted to opt in to -qio_channel_set_follow_coroutine_ctx(); but while we opted in on the -main channel, we did not opt in on the TLS wrapper channel. -qemu-iotests has coverage of NBD+iothread and NBD+TLS, but apparently -no coverage of NBD+TLS+iothread, or we would have noticed this -regression sooner. (I'll add that in the next patch) - -But while we could manually opt in to the TLS channel in nbd/server.c -(a one-line change), it is more generic if all qio channels that wrap -other channels inherit the follow status, in the same way that they -inherit feature bits. - -CC: Stefan Hajnoczi -CC: Daniel P. Berrangé -CC: qemu-stable@nongnu.org -Fixes: https://issues.redhat.com/browse/RHEL-34786 -Fixes: 06e0f098 ("io: follow coroutine AioContext in qio_channel_yield()", v8.2.0) -Signed-off-by: Eric Blake -Reviewed-by: Stefan Hajnoczi -Reviewed-by: Daniel P. Berrangé -Message-ID: <20240518025246.791593-5-eblake@redhat.com> - -(cherry picked from commit 199e84de1c903ba5aa1f7256310bbc4a20dd930b) -Jira: https://issues.redhat.com/browse/RHEL-40959 -Signed-off-by: Eric Blake ---- - io/channel-tls.c | 26 +++++++++++++++----------- - io/channel-websock.c | 1 + - 2 files changed, 16 insertions(+), 11 deletions(-) - -diff --git a/io/channel-tls.c b/io/channel-tls.c -index 1d9c9c72bf..67b9700006 100644 ---- a/io/channel-tls.c -+++ b/io/channel-tls.c -@@ -69,37 +69,40 @@ qio_channel_tls_new_server(QIOChannel *master, - const char *aclname, - Error **errp) - { -- QIOChannelTLS *ioc; -+ QIOChannelTLS *tioc; -+ QIOChannel *ioc; - -- ioc = QIO_CHANNEL_TLS(object_new(TYPE_QIO_CHANNEL_TLS)); -+ tioc = QIO_CHANNEL_TLS(object_new(TYPE_QIO_CHANNEL_TLS)); -+ ioc = QIO_CHANNEL(tioc); - -- ioc->master = master; -+ tioc->master = master; -+ ioc->follow_coroutine_ctx = master->follow_coroutine_ctx; - if (qio_channel_has_feature(master, QIO_CHANNEL_FEATURE_SHUTDOWN)) { -- qio_channel_set_feature(QIO_CHANNEL(ioc), QIO_CHANNEL_FEATURE_SHUTDOWN); -+ qio_channel_set_feature(ioc, QIO_CHANNEL_FEATURE_SHUTDOWN); - } - object_ref(OBJECT(master)); - -- ioc->session = qcrypto_tls_session_new( -+ tioc->session = qcrypto_tls_session_new( - creds, - NULL, - aclname, - QCRYPTO_TLS_CREDS_ENDPOINT_SERVER, - errp); -- if (!ioc->session) { -+ if (!tioc->session) { - goto error; - } - - qcrypto_tls_session_set_callbacks( -- ioc->session, -+ tioc->session, - qio_channel_tls_write_handler, - qio_channel_tls_read_handler, -- ioc); -+ tioc); - -- trace_qio_channel_tls_new_server(ioc, master, creds, aclname); -- return ioc; -+ trace_qio_channel_tls_new_server(tioc, master, creds, aclname); -+ return tioc; - - error: -- object_unref(OBJECT(ioc)); -+ object_unref(OBJECT(tioc)); - return NULL; - } - -@@ -116,6 +119,7 @@ qio_channel_tls_new_client(QIOChannel *master, - ioc = QIO_CHANNEL(tioc); - - tioc->master = master; -+ ioc->follow_coroutine_ctx = master->follow_coroutine_ctx; - if (qio_channel_has_feature(master, QIO_CHANNEL_FEATURE_SHUTDOWN)) { - qio_channel_set_feature(ioc, QIO_CHANNEL_FEATURE_SHUTDOWN); - } -diff --git a/io/channel-websock.c b/io/channel-websock.c -index a12acc27cf..de39f0d182 100644 ---- a/io/channel-websock.c -+++ b/io/channel-websock.c -@@ -883,6 +883,7 @@ qio_channel_websock_new_server(QIOChannel *master) - ioc = QIO_CHANNEL(wioc); - - wioc->master = master; -+ ioc->follow_coroutine_ctx = master->follow_coroutine_ctx; - if (qio_channel_has_feature(master, QIO_CHANNEL_FEATURE_SHUTDOWN)) { - qio_channel_set_feature(ioc, QIO_CHANNEL_FEATURE_SHUTDOWN); - } --- -2.39.3 - diff --git a/kvm-qmp-update-virtio-features-map-to-support-extended-f.patch b/kvm-qmp-update-virtio-features-map-to-support-extended-f.patch new file mode 100644 index 0000000..279bf95 --- /dev/null +++ b/kvm-qmp-update-virtio-features-map-to-support-extended-f.patch @@ -0,0 +1,329 @@ +From af8775763edabeeb175f72d3db0b23a20fc177e0 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:23 +0200 +Subject: [PATCH 14/19] qmp: update virtio features map to support extended + features + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [9/14] c4ebb3187b6b7aa0c0e631363110a05f538c827b (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Extend the VirtioDeviceFeatures struct with an additional u64 +to track unknown features in the 64-127 bit range and decode +the full virtio features spaces for vhost and virtio devices. + +Also add entries for the soon-to-be-supported virtio net GSO over +UDP features. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Markus Armbruster +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit a76f5b795cab8ea0654e4813caa694470d7250a9) +Signed-off-by: Laurent Vivier +--- + hw/virtio/virtio-hmp-cmds.c | 3 +- + hw/virtio/virtio-qmp.c | 91 +++++++++++++++++++++++++------------ + hw/virtio/virtio-qmp.h | 3 +- + qapi/virtio.json | 9 +++- + 4 files changed, 74 insertions(+), 32 deletions(-) + +diff --git a/hw/virtio/virtio-hmp-cmds.c b/hw/virtio/virtio-hmp-cmds.c +index 7d8677bcf0..1daae482d3 100644 +--- a/hw/virtio/virtio-hmp-cmds.c ++++ b/hw/virtio/virtio-hmp-cmds.c +@@ -74,7 +74,8 @@ static void hmp_virtio_dump_features(Monitor *mon, + } + + if (features->has_unknown_dev_features) { +- monitor_printf(mon, " unknown-features(0x%016"PRIx64")\n", ++ monitor_printf(mon, " unknown-features(0x%016"PRIx64"%016"PRIx64")\n", ++ features->unknown_dev_features2, + features->unknown_dev_features); + } + } +diff --git a/hw/virtio/virtio-qmp.c b/hw/virtio/virtio-qmp.c +index 3b6377cf0d..b338344c6c 100644 +--- a/hw/virtio/virtio-qmp.c ++++ b/hw/virtio/virtio-qmp.c +@@ -325,6 +325,20 @@ static const qmp_virtio_feature_map_t virtio_net_feature_map[] = { + FEATURE_ENTRY(VHOST_USER_F_PROTOCOL_FEATURES, \ + "VHOST_USER_F_PROTOCOL_FEATURES: Vhost-user protocol features " + "negotiation supported"), ++ FEATURE_ENTRY(VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO, \ ++ "VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO: Driver can receive GSO over " ++ "UDP tunnel packets"), ++ FEATURE_ENTRY(VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO_CSUM, \ ++ "VIRTIO_NET_F_GUEST_UDP_TUNNEL_GSO: Driver can receive GSO over " ++ "UDP tunnel packets requiring checksum offload for the outer " ++ "header"), ++ FEATURE_ENTRY(VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO, \ ++ "VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO: Device can receive GSO over " ++ "UDP tunnel packets"), ++ FEATURE_ENTRY(VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO_CSUM, \ ++ "VIRTIO_NET_F_HOST_UDP_TUNNEL_GSO_CSUM: Device can receive GSO over " ++ "UDP tunnel packets requiring checksum offload for the outer " ++ "header"), + { -1, "" } + }; + #endif +@@ -510,6 +524,24 @@ static const qmp_virtio_feature_map_t virtio_gpio_feature_map[] = { + list; \ + }) + ++#define CONVERT_FEATURES_EX(type, map, bitmap) \ ++ ({ \ ++ type *list = NULL; \ ++ type *node; \ ++ for (i = 0; map[i].virtio_bit != -1; i++) { \ ++ bit = map[i].virtio_bit; \ ++ if (!virtio_has_feature_ex(bitmap, bit)) { \ ++ continue; \ ++ } \ ++ node = g_new0(type, 1); \ ++ node->value = g_strdup(map[i].feature_desc); \ ++ node->next = list; \ ++ list = node; \ ++ virtio_clear_feature_ex(bitmap, bit); \ ++ } \ ++ list; \ ++ }) ++ + VirtioDeviceStatus *qmp_decode_status(uint8_t bitmap) + { + VirtioDeviceStatus *status; +@@ -545,109 +577,112 @@ VhostDeviceProtocols *qmp_decode_protocols(uint64_t bitmap) + return vhu_protocols; + } + +-VirtioDeviceFeatures *qmp_decode_features(uint16_t device_id, uint64_t bitmap) ++VirtioDeviceFeatures *qmp_decode_features(uint16_t device_id, ++ const uint64_t *bmap) + { ++ uint64_t bitmap[VIRTIO_FEATURES_NU64S]; + VirtioDeviceFeatures *features; + uint64_t bit; + int i; + ++ virtio_features_copy(bitmap, bmap); + features = g_new0(VirtioDeviceFeatures, 1); + features->has_dev_features = true; + + /* transport features */ +- features->transports = CONVERT_FEATURES(strList, virtio_transport_map, 0, +- bitmap); ++ features->transports = CONVERT_FEATURES_EX(strList, virtio_transport_map, ++ bitmap); + + /* device features */ + switch (device_id) { + #ifdef CONFIG_VIRTIO_SERIAL + case VIRTIO_ID_CONSOLE: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_serial_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_serial_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_BLK + case VIRTIO_ID_BLOCK: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_blk_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_blk_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_GPU + case VIRTIO_ID_GPU: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_gpu_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_gpu_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_NET + case VIRTIO_ID_NET: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_net_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_net_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_SCSI + case VIRTIO_ID_SCSI: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_scsi_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_scsi_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_BALLOON + case VIRTIO_ID_BALLOON: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_balloon_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_balloon_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_IOMMU + case VIRTIO_ID_IOMMU: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_iommu_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_iommu_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_INPUT + case VIRTIO_ID_INPUT: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_input_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_input_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VHOST_USER_FS + case VIRTIO_ID_FS: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_fs_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_fs_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VHOST_VSOCK + case VIRTIO_ID_VSOCK: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_vsock_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_vsock_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_CRYPTO + case VIRTIO_ID_CRYPTO: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_crypto_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_crypto_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_MEM + case VIRTIO_ID_MEM: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_mem_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_mem_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_I2C_ADAPTER + case VIRTIO_ID_I2C_ADAPTER: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_i2c_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_i2c_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VIRTIO_RNG + case VIRTIO_ID_RNG: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_rng_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_rng_feature_map, bitmap); + break; + #endif + #ifdef CONFIG_VHOST_USER_GPIO + case VIRTIO_ID_GPIO: + features->dev_features = +- CONVERT_FEATURES(strList, virtio_gpio_feature_map, 0, bitmap); ++ CONVERT_FEATURES_EX(strList, virtio_gpio_feature_map, bitmap); + break; + #endif + /* No features */ +@@ -680,10 +715,9 @@ VirtioDeviceFeatures *qmp_decode_features(uint16_t device_id, uint64_t bitmap) + g_assert_not_reached(); + } + +- features->has_unknown_dev_features = bitmap != 0; +- if (features->has_unknown_dev_features) { +- features->unknown_dev_features = bitmap; +- } ++ features->has_unknown_dev_features = !virtio_features_empty(bitmap); ++ features->unknown_dev_features = bitmap[0]; ++ features->unknown_dev_features2 = bitmap[1]; + + return features; + } +@@ -743,11 +777,11 @@ VirtioStatus *qmp_x_query_virtio_status(const char *path, Error **errp) + status->device_id = vdev->device_id; + status->vhost_started = vdev->vhost_started; + status->guest_features = qmp_decode_features(vdev->device_id, +- vdev->guest_features); ++ vdev->guest_features_ex); + status->host_features = qmp_decode_features(vdev->device_id, +- vdev->host_features); ++ vdev->host_features_ex); + status->backend_features = qmp_decode_features(vdev->device_id, +- vdev->backend_features); ++ vdev->backend_features_ex); + + switch (vdev->device_endian) { + case VIRTIO_DEVICE_ENDIAN_LITTLE: +@@ -785,11 +819,12 @@ VirtioStatus *qmp_x_query_virtio_status(const char *path, Error **errp) + status->vhost_dev->nvqs = hdev->nvqs; + status->vhost_dev->vq_index = hdev->vq_index; + status->vhost_dev->features = +- qmp_decode_features(vdev->device_id, hdev->features); ++ qmp_decode_features(vdev->device_id, hdev->features_ex); + status->vhost_dev->acked_features = +- qmp_decode_features(vdev->device_id, hdev->acked_features); ++ qmp_decode_features(vdev->device_id, hdev->acked_features_ex); + status->vhost_dev->backend_features = +- qmp_decode_features(vdev->device_id, hdev->backend_features); ++ qmp_decode_features(vdev->device_id, hdev->backend_features_ex); ++ + status->vhost_dev->protocol_features = + qmp_decode_protocols(hdev->protocol_features); + status->vhost_dev->max_queues = hdev->max_queues; +diff --git a/hw/virtio/virtio-qmp.h b/hw/virtio/virtio-qmp.h +index 245a446a56..e0a1e49035 100644 +--- a/hw/virtio/virtio-qmp.h ++++ b/hw/virtio/virtio-qmp.h +@@ -18,6 +18,7 @@ + VirtIODevice *qmp_find_virtio_device(const char *path); + VirtioDeviceStatus *qmp_decode_status(uint8_t bitmap); + VhostDeviceProtocols *qmp_decode_protocols(uint64_t bitmap); +-VirtioDeviceFeatures *qmp_decode_features(uint16_t device_id, uint64_t bitmap); ++VirtioDeviceFeatures *qmp_decode_features(uint16_t device_id, ++ const uint64_t *bitmap); + + #endif +diff --git a/qapi/virtio.json b/qapi/virtio.json +index 9d652fe4a8..05295ab665 100644 +--- a/qapi/virtio.json ++++ b/qapi/virtio.json +@@ -247,6 +247,7 @@ + # }, + # "host-features": { + # "unknown-dev-features": 1073741824, ++# "unknown-dev-features2": 0, + # "dev-features": [], + # "transports": [ + # "VIRTIO_RING_F_EVENT_IDX: Used & avail. event fields enabled", +@@ -490,14 +491,18 @@ + # unique features) + # + # @unknown-dev-features: Virtio device features bitmap that have not +-# been decoded ++# been decoded (bits 0-63) ++# ++# @unknown-dev-features2: Virtio device features bitmap that have not ++# been decoded (bits 64-127) (since 10.2) + # + # Since: 7.2 + ## + { 'struct': 'VirtioDeviceFeatures', + 'data': { 'transports': [ 'str' ], + '*dev-features': [ 'str' ], +- '*unknown-dev-features': 'uint64' } } ++ '*unknown-dev-features': 'uint64', ++ '*unknown-dev-features2': 'uint64' } } + + ## + # @VirtQueueStatus: +-- +2.47.3 + diff --git a/kvm-qtest-Do-not-run-bios-tables-test-on-aarch64.patch b/kvm-qtest-Do-not-run-bios-tables-test-on-aarch64.patch new file mode 100644 index 0000000..d048675 --- /dev/null +++ b/kvm-qtest-Do-not-run-bios-tables-test-on-aarch64.patch @@ -0,0 +1,30 @@ +From 3b21c60b771087e7d566bf738e04e01a7a1bdf09 Mon Sep 17 00:00:00 2001 +From: Miroslav Rezanina +Date: Fri, 14 Nov 2025 06:46:07 +0100 +Subject: [PATCH 16/16] qtest: Do not run bios-tables-test on aarch64 + +We do several disruptive downstream only changes that make +bios-tables-test to fail. Disabling it for now. + +This is done to enable fixing RHEL-126573 and RHEL-67323. + +Signed-off-by: Miroslav Rezanina +--- + tests/qtest/meson.build | 1 - + 1 file changed, 1 deletion(-) + +diff --git a/tests/qtest/meson.build b/tests/qtest/meson.build +index ef44ffaf78..13b52ea41a 100644 +--- a/tests/qtest/meson.build ++++ b/tests/qtest/meson.build +@@ -252,7 +252,6 @@ qtests_arm = \ + + # TODO: once aarch64 TCG is fixed on ARM 32 bit host, make bios-tables-test unconditional + qtests_aarch64 = \ +- (cpu != 'arm' and unpack_edk2_blobs ? ['bios-tables-test'] : []) + \ + (config_all_accel.has_key('CONFIG_TCG') and config_all_devices.has_key('CONFIG_TPM_TIS_SYSBUS') ? \ + ['tpm-tis-device-test', 'tpm-tis-device-swtpm-test'] : []) + \ + (config_all_devices.has_key('CONFIG_XLNX_ZYNQMP_ARM') ? ['xlnx-can-test', 'fuzz-xlnx-dp-test'] : []) + \ +-- +2.47.3 + diff --git a/kvm-qtest-bios-tables-test-Add-tests-for-legacy-smmuv3-a.patch b/kvm-qtest-bios-tables-test-Add-tests-for-legacy-smmuv3-a.patch new file mode 100644 index 0000000..4ab008d --- /dev/null +++ b/kvm-qtest-bios-tables-test-Add-tests-for-legacy-smmuv3-a.patch @@ -0,0 +1,160 @@ +From 51ec91309c99a5d81b53c2762d18c073f672e45a Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:32 +0100 +Subject: [PATCH 14/16] qtest/bios-tables-test: Add tests for legacy smmuv3 and + smmuv3 device + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [10/11] 101ed34313636fdc11f7fbdedbe2f91671b84c4b (eauger1/centos-qemu-kvm) + +For the legacy SMMUv3 test, the setup includes three PCIe Root Complexes, +one of which has bypass_iommu enabled. The generated IORT table contains +a single SMMUv3 node, a Root Complex(RC) node and 1 ITS node. +RC node features 4 ID mappings, of which 2 points to SMMU node and the +remaining ones points to ITS. + + pcie.0 -> {SMMU0} -> {ITS} +{RC} pcie.1 -> {SMMU0} -> {ITS} + pcie.2 -> {ITS} + [all other ids] -> {ITS} + +For the -device arm-smmuv3,... test, the configuration also includes three +Root Complexes, with two connected to separate SMMUv3 devices. +The resulting IORT table contains 1 RC node, 2 SMMU nodes and 1 ITS node. +RC node features 4 ID mappings. 2 of them target the 2 SMMU nodes while +the others targets the ITS. + + pcie.0 -> {SMMU0} -> {ITS} +{RC} pcie.1 -> {SMMU1} -> {ITS} + pcie.2 -> {ITS} + [all other ids] -> {ITS} + +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Reviewed-by: Nicolin Chen +Message-id: 20250829082543.7680-11-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit 3f8cd046c151c471d9a34181320f4a7d3f72b32a) +Signed-off-by: Eric Auger +--- + tests/qtest/bios-tables-test.c | 86 ++++++++++++++++++++++++++++++++++ + 1 file changed, 86 insertions(+) + +diff --git a/tests/qtest/bios-tables-test.c b/tests/qtest/bios-tables-test.c +index 386196edc8..a384aac1be 100644 +--- a/tests/qtest/bios-tables-test.c ++++ b/tests/qtest/bios-tables-test.c +@@ -2343,6 +2343,86 @@ static void test_acpi_aarch64_virt_viot(void) + free_test_data(&data); + } + ++static void test_acpi_aarch64_virt_smmuv3_legacy(void) ++{ ++ test_data data = { ++ .machine = "virt", ++ .arch = "aarch64", ++ .tcg_only = true, ++ .uefi_fl1 = "pc-bios/edk2-aarch64-code.fd", ++ .uefi_fl2 = "pc-bios/edk2-arm-vars.fd", ++ .ram_start = 0x40000000ULL, ++ .scan_len = 128ULL * MiB, ++ }; ++ ++ /* ++ * cdrom is plugged into scsi controller to avoid conflict ++ * with pxb-pcie. See comments in test_acpi_aarch64_virt_tcg_pxb() for ++ * details. ++ * ++ * The setup includes three PCIe root complexes, one of which has ++ * bypass_iommu enabled. The generated IORT table contains a single ++ * SMMUv3 node and a Root Complex node with three ID mappings. Two ++ * of the ID mappings have output references pointing to the SMMUv3 ++ * node and the remaining one points to ITS. ++ */ ++ data.variant = ".smmuv3-legacy"; ++ test_acpi_one(" -device pcie-root-port,chassis=1,id=pci.1" ++ " -device virtio-scsi-pci,id=scsi0,bus=pci.1" ++ " -drive file=" ++ "tests/data/uefi-boot-images/bios-tables-test.aarch64.iso.qcow2," ++ "if=none,media=cdrom,id=drive-scsi0-0-0-1,readonly=on" ++ " -device scsi-cd,bus=scsi0.0,scsi-id=0," ++ "drive=drive-scsi0-0-0-1,id=scsi0-0-0-1,bootindex=1" ++ " -cpu cortex-a57" ++ " -M iommu=smmuv3" ++ " -device pxb-pcie,id=pcie.1,bus=pcie.0,bus_nr=0x10" ++ " -device pxb-pcie,id=pcie.2,bus=pcie.0,bus_nr=0x20,bypass_iommu=on", ++ &data); ++ free_test_data(&data); ++} ++ ++static void test_acpi_aarch64_virt_smmuv3_dev(void) ++{ ++ test_data data = { ++ .machine = "virt", ++ .arch = "aarch64", ++ .tcg_only = true, ++ .uefi_fl1 = "pc-bios/edk2-aarch64-code.fd", ++ .uefi_fl2 = "pc-bios/edk2-arm-vars.fd", ++ .ram_start = 0x40000000ULL, ++ .scan_len = 128ULL * MiB, ++ }; ++ ++ /* ++ * cdrom is plugged into scsi controller to avoid conflict ++ * with pxb-pcie. See comments in test_acpi_aarch64_virt_tcg_pxb() ++ * for details. ++ * ++ * The setup includes three PCie root complexes, two of which are ++ * connected to separate SMMUv3 devices. The resulting IORT table ++ * contains two SMMUv3 nodes and a Root Complex node with ID mappings ++ * of which two of the ID mappings have output references pointing ++ * to two different SMMUv3 nodes and the remaining ones pointing to ++ * ITS. ++ */ ++ data.variant = ".smmuv3-dev"; ++ test_acpi_one(" -device pcie-root-port,chassis=1,id=pci.1" ++ " -device virtio-scsi-pci,id=scsi0,bus=pci.1" ++ " -drive file=" ++ "tests/data/uefi-boot-images/bios-tables-test.aarch64.iso.qcow2," ++ "if=none,media=cdrom,id=drive-scsi0-0-0-1,readonly=on" ++ " -device scsi-cd,bus=scsi0.0,scsi-id=0," ++ "drive=drive-scsi0-0-0-1,id=scsi0-0-0-1,bootindex=1" ++ " -cpu cortex-a57" ++ " -device arm-smmuv3,primary-bus=pcie.0,id=smmuv3.0" ++ " -device pxb-pcie,id=pcie.1,bus=pcie.0,bus_nr=0x10" ++ " -device arm-smmuv3,primary-bus=pcie.1,id=smmuv3.1" ++ " -device pxb-pcie,id=pcie.2,bus=pcie.0,bus_nr=0x20", ++ &data); ++ free_test_data(&data); ++} ++ + #ifndef _WIN32 + # define DEV_NULL "/dev/null" + #else +@@ -2776,6 +2856,12 @@ int main(int argc, char *argv[]) + if (qtest_has_device("virtio-iommu-pci")) { + qtest_add_func("acpi/virt/viot", test_acpi_aarch64_virt_viot); + } ++ qtest_add_func("acpi/virt/smmuv3-legacy", ++ test_acpi_aarch64_virt_smmuv3_legacy); ++ if (qtest_has_device("arm-smmuv3")) { ++ qtest_add_func("acpi/virt/smmuv3-dev", ++ test_acpi_aarch64_virt_smmuv3_dev); ++ } + } + #if 0 /* Disabled for Red Hat Enterprise Linux */ + } else if (strcmp(arch, "riscv64") == 0) { +-- +2.47.3 + diff --git a/kvm-qtest-bios-tables-test-Update-tables-for-smmuv3-test.patch b/kvm-qtest-bios-tables-test-Update-tables-for-smmuv3-test.patch new file mode 100644 index 0000000..9273081 --- /dev/null +++ b/kvm-qtest-bios-tables-test-Update-tables-for-smmuv3-test.patch @@ -0,0 +1,282 @@ +From d6f27731c3d469f4ba68807a4c1f8ee534cc9d57 Mon Sep 17 00:00:00 2001 +From: Shameer Kolothum +Date: Fri, 29 Aug 2025 09:25:33 +0100 +Subject: [PATCH 15/16] qtest/bios-tables-test: Update tables for smmuv3 tests + +RH-Author: Eric Auger +RH-MergeRequest: 423: hw/arm/virt: Add support for user creatable SMMUv3 device +RH-Jira: RHEL-73800 +RH-Acked-by: Gavin Shan +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Sebastian Ott +RH-Acked-by: Donald Dutile +RH-Commit: [11/11] 88cb1daec4e92b759f12a96daacd46fc4656eacd (eauger1/centos-qemu-kvm) + +For the legacy smmuv3 test case, generated IORT has a single SMMUv3 node, +a Root Complex(RC) node and 1 ITS node. +RC node features 4 ID mappings, of which 2 points to SMMU node and the +remaining ones points to ITS. + + pcie.0 -> {SMMU0} -> {ITS} +{RC} pcie.1 -> {SMMU0} -> {ITS} + pcie.2 -> {ITS} + [all other ids] -> {ITS} + +... +[030h 0048 1] Type : 00 +[031h 0049 2] Length : 0018 +[033h 0051 1] Revision : 01 +[034h 0052 4] Identifier : 00000000 +[038h 0056 4] Mapping Count : 00000000 +[03Ch 0060 4] Mapping Offset : 00000000 + +[040h 0064 4] ItsCount : 00000001 +[044h 0068 4] Identifiers : 00000000 + +[048h 0072 1] Type : 04 +[049h 0073 2] Length : 0058 +[04Bh 0075 1] Revision : 04 +[04Ch 0076 4] Identifier : 00000001 +[050h 0080 4] Mapping Count : 00000001 +[054h 0084 4] Mapping Offset : 00000044 + +[058h 0088 8] Base Address : 0000000009050000 +[060h 0096 4] Flags (decoded below) : 00000001 + COHACC Override : 1 + HTTU Override : 0 + Proximity Domain Valid : 0 +[064h 0100 4] Reserved : 00000000 +[068h 0104 8] VATOS Address : 0000000000000000 +[070h 0112 4] Model : 00000000 +[074h 0116 4] Event GSIV : 0000006A +[078h 0120 4] PRI GSIV : 0000006B +[07Ch 0124 4] GERR GSIV : 0000006D +[080h 0128 4] Sync GSIV : 0000006C +[084h 0132 4] Proximity Domain : 00000000 +[088h 0136 4] Device ID Mapping Index : 00000000 + +[08Ch 0140 4] Input base : 00000000 +[090h 0144 4] ID Count : 0000FFFF +[094h 0148 4] Output Base : 00000000 +[098h 0152 4] Output Reference : 00000030 +[09Ch 0156 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[0A0h 0160 1] Type : 02 +[0A1h 0161 2] Length : 0074 +[0A3h 0163 1] Revision : 03 +[0A4h 0164 4] Identifier : 00000002 +[0A8h 0168 4] Mapping Count : 00000004 +[0ACh 0172 4] Mapping Offset : 00000024 + +[0B0h 0176 8] Memory Properties : [IORT Memory Access Properties] +[0B0h 0176 4] Cache Coherency : 00000001 +[0B4h 0180 1] Hints (decoded below) : 00 + Transient : 0 + Write Allocate : 0 + Read Allocate : 0 + Override : 0 +[0B5h 0181 2] Reserved : 0000 +[0B7h 0183 1] Memory Flags (decoded below) : 03 + Coherency : 1 + Device Attribute : 1 +[0B8h 0184 4] ATS Attribute : 00000000 +[0BCh 0188 4] PCI Segment Number : 00000000 +[0C0h 0192 1] Memory Size Limit : 40 +[0C1h 0193 2] PASID Capabilities : 0000 +[0C3h 0195 1] Reserved : 00 + +[0C4h 0196 4] Input base : 00000000 +[0C8h 0200 4] ID Count : 000001FF +[0CCh 0204 4] Output Base : 00000000 +[0D0h 0208 4] Output Reference : 00000048 +[0D4h 0212 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[0D8h 0216 4] Input base : 00001000 +[0DCh 0220 4] ID Count : 000000FF +[0E0h 0224 4] Output Base : 00001000 +[0E4h 0228 4] Output Reference : 00000048 +[0E8h 0232 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[0ECh 0236 4] Input base : 00000200 +[0F0h 0240 4] ID Count : 00000DFF +[0F4h 0244 4] Output Base : 00000200 +[0F8h 0248 4] Output Reference : 00000030 +[0FCh 0252 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[100h 0256 4] Input base : 00001100 +[104h 0260 4] ID Count : 0000EEFF +[108h 0264 4] Output Base : 00001100 +[10Ch 0268 4] Output Reference : 00000030 +[110h 0272 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +For the smmuv3-dev test case, IORT has 2 SMMUV3 nodes, +1 RC node and 1 ITS node. +RC node features 4 ID mappings. 2 of them target the 2 +SMMU nodes while the others targets the ITS. + + pcie.0 -> {SMMU0} -> {ITS} +{RC} pcie.1 -> {SMMU1} -> {ITS} + pcie.2 -> {ITS} + [all other ids] -> {ITS} +... +[030h 0048 1] Type : 00 +[031h 0049 2] Length : 0018 +[033h 0051 1] Revision : 01 +[034h 0052 4] Identifier : 00000000 +[038h 0056 4] Mapping Count : 00000000 +[03Ch 0060 4] Mapping Offset : 00000000 + +[040h 0064 4] ItsCount : 00000001 +[044h 0068 4] Identifiers : 00000000 + +[048h 0072 1] Type : 04 +[049h 0073 2] Length : 0058 +[04Bh 0075 1] Revision : 04 +[04Ch 0076 4] Identifier : 00000001 +[050h 0080 4] Mapping Count : 00000001 +[054h 0084 4] Mapping Offset : 00000044 + +[058h 0088 8] Base Address : 000000000C000000 +[060h 0096 4] Flags (decoded below) : 00000001 + COHACC Override : 1 + HTTU Override : 0 + Proximity Domain Valid : 0 +[064h 0100 4] Reserved : 00000000 +[068h 0104 8] VATOS Address : 0000000000000000 +[070h 0112 4] Model : 00000000 +[074h 0116 4] Event GSIV : 00000090 +[078h 0120 4] PRI GSIV : 00000091 +[07Ch 0124 4] GERR GSIV : 00000093 +[080h 0128 4] Sync GSIV : 00000092 +[084h 0132 4] Proximity Domain : 00000000 +[088h 0136 4] Device ID Mapping Index : 00000000 + +[08Ch 0140 4] Input base : 00000000 +[090h 0144 4] ID Count : 0000FFFF +[094h 0148 4] Output Base : 00000000 +[098h 0152 4] Output Reference : 00000030 +[09Ch 0156 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[0A0h 0160 1] Type : 04 +[0A1h 0161 2] Length : 0058 +[0A3h 0163 1] Revision : 04 +[0A4h 0164 4] Identifier : 00000002 +[0A8h 0168 4] Mapping Count : 00000001 +[0ACh 0172 4] Mapping Offset : 00000044 + +[0B0h 0176 8] Base Address : 000000000C020000 +[0B8h 0184 4] Flags (decoded below) : 00000001 + COHACC Override : 1 + HTTU Override : 0 + Proximity Domain Valid : 0 +[0BCh 0188 4] Reserved : 00000000 +[0C0h 0192 8] VATOS Address : 0000000000000000 +[0C8h 0200 4] Model : 00000000 +[0CCh 0204 4] Event GSIV : 00000094 +[0D0h 0208 4] PRI GSIV : 00000095 +[0D4h 0212 4] GERR GSIV : 00000097 +[0D8h 0216 4] Sync GSIV : 00000096 +[0DCh 0220 4] Proximity Domain : 00000000 +[0E0h 0224 4] Device ID Mapping Index : 00000000 + +[0E4h 0228 4] Input base : 00000000 +[0E8h 0232 4] ID Count : 0000FFFF +[0ECh 0236 4] Output Base : 00000000 +[0F0h 0240 4] Output Reference : 00000030 +[0F4h 0244 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[0F8h 0248 1] Type : 02 +[0F9h 0249 2] Length : 0074 +[0FBh 0251 1] Revision : 03 +[0FCh 0252 4] Identifier : 00000003 +[100h 0256 4] Mapping Count : 00000004 +[104h 0260 4] Mapping Offset : 00000024 + +[108h 0264 8] Memory Properties : [IORT Memory Access Properties] +[108h 0264 4] Cache Coherency : 00000001 +[10Ch 0268 1] Hints (decoded below) : 00 + Transient : 0 + Write Allocate : 0 + Read Allocate : 0 + Override : 0 +[10Dh 0269 2] Reserved : 0000 +[10Fh 0271 1] Memory Flags (decoded below) : 03 + Coherency : 1 + Device Attribute : 1 +[110h 0272 4] ATS Attribute : 00000000 +[114h 0276 4] PCI Segment Number : 00000000 +[118h 0280 1] Memory Size Limit : 40 +[119h 0281 2] PASID Capabilities : 0000 +[11Bh 0283 1] Reserved : 00 + +[11Ch 0284 4] Input base : 00000000 +[120h 0288 4] ID Count : 000001FF +[124h 0292 4] Output Base : 00000000 +[128h 0296 4] Output Reference : 00000048 +[12Ch 0300 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[130h 0304 4] Input base : 00001000 +[134h 0308 4] ID Count : 000000FF +[138h 0312 4] Output Base : 00001000 +[13Ch 0316 4] Output Reference : 000000A0 +[140h 0320 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[144h 0324 4] Input base : 00000200 +[148h 0328 4] ID Count : 00000DFF +[14Ch 0332 4] Output Base : 00000200 +[150h 0336 4] Output Reference : 00000030 +[154h 0340 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +[158h 0344 4] Input base : 00001100 +[15Ch 0348 4] ID Count : 0000EEFF +[160h 0352 4] Output Base : 00001100 +[164h 0356 4] Output Reference : 00000030 +[168h 0360 4] Flags (decoded below) : 00000000 + Single Mapping : 0 + +Note: DSDT changes are not described here as it is not impacted by the +way the SMMUv3 is instantiated. + +Reviewed-by: Jonathan Cameron +Reviewed-by: Eric Auger +Tested-by: Eric Auger +Tested-by: Nicolin Chen +Signed-off-by: Shameer Kolothum +Signed-off-by: Shameer Kolothum +Reviewed-by: Donald Dutile +Reviewed-by: Nicolin Chen +Message-id: 20250829082543.7680-12-skolothumtho@nvidia.com +Signed-off-by: Peter Maydell +(cherry picked from commit d35146a6606cf6ebb4e24bb97dfc0330f074f6e3) +Signed-off-by: Eric Auger +--- + tests/data/acpi/aarch64/virt/DSDT.smmuv3-dev | Bin 0 -> 10230 bytes + tests/data/acpi/aarch64/virt/DSDT.smmuv3-legacy | Bin 0 -> 10230 bytes + tests/data/acpi/aarch64/virt/IORT.smmuv3-dev | Bin 0 -> 364 bytes + tests/data/acpi/aarch64/virt/IORT.smmuv3-legacy | Bin 0 -> 276 bytes + tests/qtest/bios-tables-test-allowed-diff.h | 4 ---- + 5 files changed, 4 deletions(-) + +diff --git a/tests/qtest/bios-tables-test-allowed-diff.h b/tests/qtest/bios-tables-test-allowed-diff.h +index 2e3e3ccdce..dfb8523c8b 100644 +--- a/tests/qtest/bios-tables-test-allowed-diff.h ++++ b/tests/qtest/bios-tables-test-allowed-diff.h +@@ -1,5 +1 @@ + /* List of comma-separated changed AML files to ignore */ +-"tests/data/acpi/aarch64/virt/DSDT.smmuv3-legacy", +-"tests/data/acpi/aarch64/virt/DSDT.smmuv3-dev", +-"tests/data/acpi/aarch64/virt/IORT.smmuv3-legacy", +-"tests/data/acpi/aarch64/virt/IORT.smmuv3-dev", +-- +2.47.3 + diff --git a/kvm-qtest-x86-numa-test-do-not-use-the-obsolete-pentium-.patch b/kvm-qtest-x86-numa-test-do-not-use-the-obsolete-pentium-.patch deleted file mode 100644 index d5e1077..0000000 --- a/kvm-qtest-x86-numa-test-do-not-use-the-obsolete-pentium-.patch +++ /dev/null @@ -1,46 +0,0 @@ -From 2c7512b27b8d8862e26c6e07169752078513f40c Mon Sep 17 00:00:00 2001 -From: Ani Sinha -Date: Mon, 10 Jun 2024 21:22:58 +0530 -Subject: [PATCH 01/14] qtest/x86/numa-test: do not use the obsolete 'pentium' - cpu - -RH-Author: Ani Sinha -RH-MergeRequest: 243: target/cpu-models/x86: Remove the existing deprecated CPU models on c10s -RH-Jira: RHEL-28972 -RH-Acked-by: Thomas Huth -RH-Acked-by: Igor Mammedov -RH-Acked-by: MST -RH-Commit: [1/4] a9b38ebd4e772a0a1fe40301a6f1abab6b961cd7 (anisinha/centos-qemu-kvm) - -'pentium' cpu is old and obsolete and should be avoided for running tests if -its not strictly needed. Use 'max' cpu instead for generic non-cpu specific -numa test. - -Reviewed-by: Thomas Huth -Reviewed-by: Igor Mammedov -Tested-by: Mario Casquero -Signed-off-by: Ani Sinha -Message-ID: <20240610155303.7933-2-anisinha@redhat.com> -Signed-off-by: Thomas Huth -(cherry picked from commit 07c8d9ac0fa30712fdf78046a7998ee8d2231d6f) ---- - tests/qtest/numa-test.c | 3 ++- - 1 file changed, 2 insertions(+), 1 deletion(-) - -diff --git a/tests/qtest/numa-test.c b/tests/qtest/numa-test.c -index 4f4404a4b1..a512f743c4 100644 ---- a/tests/qtest/numa-test.c -+++ b/tests/qtest/numa-test.c -@@ -125,7 +125,8 @@ static void pc_numa_cpu(const void *data) - QTestState *qts; - g_autofree char *cli = NULL; - -- cli = make_cli(data, "-cpu pentium -machine smp.cpus=8,smp.sockets=2,smp.cores=2,smp.threads=2 " -+ cli = make_cli(data, -+ "-cpu max -machine smp.cpus=8,smp.sockets=2,smp.cores=2,smp.threads=2 " - "-numa node,nodeid=0,memdev=ram -numa node,nodeid=1 " - "-numa cpu,node-id=1,socket-id=0 " - "-numa cpu,node-id=0,socket-id=1,core-id=0 " --- -2.39.3 - diff --git a/kvm-ram-block-attributes-Unify-the-retrieval-of-the-bloc.patch b/kvm-ram-block-attributes-Unify-the-retrieval-of-the-bloc.patch new file mode 100644 index 0000000..87c0288 --- /dev/null +++ b/kvm-ram-block-attributes-Unify-the-retrieval-of-the-bloc.patch @@ -0,0 +1,47 @@ +From 0e0b1ba11faad72338036b9dfa97a3a8a8bc56dc Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Wed, 19 Nov 2025 12:51:32 +0100 +Subject: [PATCH 2/4] ram-block-attributes: Unify the retrieval of the block + size + +RH-Author: Paolo Bonzini +RH-MergeRequest: 426: Fix hugetlb memory backends in confidential VMs +RH-Jira: RHEL-126708 +RH-Acked-by: Bandan Das +RH-Acked-by: Eric Blake +RH-Acked-by: Peter Xu +RH-Commit: [2/2] 0d6f2045a97f04348c4d07ef331e42f47eaba160 (bonzini/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-126708 + +There's an existing helper function designed to obtain the block size. +Modify ram_block_attribute_create() to use this function for +consistency. + +Tested-by: Farrah Chen +Signed-off-by: Chenyi Qiang +Link: https://lore.kernel.org/r/20251023095526.48365-3-chenyi.qiang@intel.com +[peterx: fix double spaces, per david] +Signed-off-by: Peter Xu +(cherry picked from commit b2ceb87b1a210d91a29d525590eb164d1121b8a1) +Signed-off-by: Paolo Bonzini +--- + system/ram-block-attributes.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/system/ram-block-attributes.c b/system/ram-block-attributes.c +index a7579de5b4..fb7c5c2746 100644 +--- a/system/ram-block-attributes.c ++++ b/system/ram-block-attributes.c +@@ -390,7 +390,7 @@ int ram_block_attributes_state_change(RamBlockAttributes *attr, + + RamBlockAttributes *ram_block_attributes_create(RAMBlock *ram_block) + { +- const int block_size = qemu_real_host_page_size(); ++ const int block_size = ram_block_attributes_get_block_size(); + RamBlockAttributes *attr; + MemoryRegion *mr = ram_block->mr; + +-- +2.47.3 + diff --git a/kvm-ram-block-attributes-fix-interaction-with-hugetlb-me.patch b/kvm-ram-block-attributes-fix-interaction-with-hugetlb-me.patch new file mode 100644 index 0000000..5e5f04d --- /dev/null +++ b/kvm-ram-block-attributes-fix-interaction-with-hugetlb-me.patch @@ -0,0 +1,125 @@ +From 6b03dd00ce169886ec7aa866c14daa99209a3137 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Wed, 19 Nov 2025 12:51:32 +0100 +Subject: [PATCH 1/4] ram-block-attributes: fix interaction with hugetlb memory + backends + +RH-Author: Paolo Bonzini +RH-MergeRequest: 426: Fix hugetlb memory backends in confidential VMs +RH-Jira: RHEL-126708 +RH-Acked-by: Bandan Das +RH-Acked-by: Eric Blake +RH-Acked-by: Peter Xu +RH-Commit: [1/2] 050a6c78989db4ba3111fb903647270c7ab12e53 (bonzini/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-126708 + +Currently, CoCo VMs can perform conversion at the base page granularity, +which is the granularity that has to be tracked. In relevant setups, the +target page size is assumed to be equal to the host page size, thus +fixing the block size to the host page size. + +However, since private memory and shared memory have different backend +at present, users can specify shared memory with a hugetlbfs backend +while private memory with guest_memfd backend only supports 4K page +size. In this scenario, ram_block->page_size is different from the host +page size which will trigger an assertion when retrieving the block +size. + +To address this, return the host page size directly to relax the +restriction. This changes fixes a regression of using hugetlbfs backend +for shared memory within CoCo VMs, with or without VFIO devices' presence. + +Acked-by: David Hildenbrand +Tested-by: Farrah Chen +Signed-off-by: Chenyi Qiang +Link: https://lore.kernel.org/r/20251023095526.48365-2-chenyi.qiang@intel.com +[peterx: fix subject, per david] +Cc: qemu-stable +Signed-off-by: Peter Xu +(cherry picked from commit 8922a758b29251d9009ec509e7f580b76509ab3d) +Signed-off-by: Paolo Bonzini +--- + system/ram-block-attributes.c | 18 ++++++++---------- + 1 file changed, 8 insertions(+), 10 deletions(-) + +diff --git a/system/ram-block-attributes.c b/system/ram-block-attributes.c +index 68e8a02703..a7579de5b4 100644 +--- a/system/ram-block-attributes.c ++++ b/system/ram-block-attributes.c +@@ -22,16 +22,14 @@ OBJECT_DEFINE_SIMPLE_TYPE_WITH_INTERFACES(RamBlockAttributes, + { }) + + static size_t +-ram_block_attributes_get_block_size(const RamBlockAttributes *attr) ++ram_block_attributes_get_block_size(void) + { + /* + * Because page conversion could be manipulated in the size of at least 4K + * or 4K aligned, Use the host page size as the granularity to track the + * memory attribute. + */ +- g_assert(attr && attr->ram_block); +- g_assert(attr->ram_block->page_size == qemu_real_host_page_size()); +- return attr->ram_block->page_size; ++ return qemu_real_host_page_size(); + } + + +@@ -40,7 +38,7 @@ ram_block_attributes_rdm_is_populated(const RamDiscardManager *rdm, + const MemoryRegionSection *section) + { + const RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); +- const size_t block_size = ram_block_attributes_get_block_size(attr); ++ const size_t block_size = ram_block_attributes_get_block_size(); + const uint64_t first_bit = section->offset_within_region / block_size; + const uint64_t last_bit = + first_bit + int128_get64(section->size) / block_size - 1; +@@ -81,7 +79,7 @@ ram_block_attributes_for_each_populated_section(const RamBlockAttributes *attr, + { + unsigned long first_bit, last_bit; + uint64_t offset, size; +- const size_t block_size = ram_block_attributes_get_block_size(attr); ++ const size_t block_size = ram_block_attributes_get_block_size(); + int ret = 0; + + first_bit = section->offset_within_region / block_size; +@@ -122,7 +120,7 @@ ram_block_attributes_for_each_discarded_section(const RamBlockAttributes *attr, + { + unsigned long first_bit, last_bit; + uint64_t offset, size; +- const size_t block_size = ram_block_attributes_get_block_size(attr); ++ const size_t block_size = ram_block_attributes_get_block_size(); + int ret = 0; + + first_bit = section->offset_within_region / block_size; +@@ -163,7 +161,7 @@ ram_block_attributes_rdm_get_min_granularity(const RamDiscardManager *rdm, + const RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); + + g_assert(mr == attr->ram_block->mr); +- return ram_block_attributes_get_block_size(attr); ++ return ram_block_attributes_get_block_size(); + } + + static void +@@ -265,7 +263,7 @@ ram_block_attributes_is_valid_range(RamBlockAttributes *attr, uint64_t offset, + g_assert(mr); + + uint64_t region_size = memory_region_size(mr); +- const size_t block_size = ram_block_attributes_get_block_size(attr); ++ const size_t block_size = ram_block_attributes_get_block_size(); + + if (!QEMU_IS_ALIGNED(offset, block_size) || + !QEMU_IS_ALIGNED(size, block_size)) { +@@ -322,7 +320,7 @@ int ram_block_attributes_state_change(RamBlockAttributes *attr, + uint64_t offset, uint64_t size, + bool to_discard) + { +- const size_t block_size = ram_block_attributes_get_block_size(attr); ++ const size_t block_size = ram_block_attributes_get_block_size(); + const unsigned long first_bit = offset / block_size; + const unsigned long nbits = size / block_size; + const unsigned long last_bit = first_bit + nbits - 1; +-- +2.47.3 + diff --git a/kvm-rbd-Run-co-BH-CB-in-the-coroutine-s-AioContext.patch b/kvm-rbd-Run-co-BH-CB-in-the-coroutine-s-AioContext.patch new file mode 100644 index 0000000..789185b --- /dev/null +++ b/kvm-rbd-Run-co-BH-CB-in-the-coroutine-s-AioContext.patch @@ -0,0 +1,127 @@ +From 339f5506d877c9e32852aff3cbe165a0aa2964c5 Mon Sep 17 00:00:00 2001 +From: Hanna Czenczek +Date: Mon, 10 Nov 2025 16:48:37 +0100 +Subject: [PATCH 01/19] =?UTF-8?q?rbd:=20Run=20co=20BH=20CB=20in=20the=20co?= + =?UTF-8?q?routine=E2=80=99s=20AioContext?= +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Hanna Czenczek +RH-MergeRequest: 454: Multithreading fixes for rbd, curl, qcow2 +RH-Jira: RHEL-79118 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/5] 6d858f7d65d6358bf7b8b58dc6e5a2ea9e8599fe (hreitz/qemu-kvm-c-9-s) + +qemu_rbd_completion_cb() schedules the request completion code +(qemu_rbd_finish_bh()) to run in the BDS’s AioContext, assuming that +this is the same thread in which qemu_rbd_start_co() runs. + +To explain, this is how both latter functions interact: + +In qemu_rbd_start_co(): + + while (!task.complete) + qemu_coroutine_yield(); + +In qemu_rbd_finish_bh(): + + task->complete = true; + aio_co_wake(task->co); // task->co is qemu_rbd_start_co() + +For this interaction to work reliably, both must run in the same thread +so that qemu_rbd_finish_bh() can only run once the coroutine yields. +Otherwise, finish_bh() may run before start_co() checks task.complete, +which will result in the latter seeing .complete as true immediately and +skipping the yield altogether, even though finish_bh() still wakes it. + +With multiqueue, the BDS’s AioContext is not necessarily the thread +start_co() runs in, and so finish_bh() may be scheduled to run in a +different thread than start_co(). With the right timing, this will +cause the problems described above; waking a non-yielding coroutine is +not good, as can be reproduced by putting e.g. a usleep(100000) above +the while loop in start_co() (and using multiqueue), giving finish_bh() +a much better chance at exiting before start_co() can yield. + +So instead of scheduling finish_bh() in the BDS’s AioContext, schedule +finish_bh() in task->co’s AioContext. + +In addition, we can get rid of task.complete altogether because we will +get woken exactly once, when the task is indeed complete, no need to +check. + +(We could go further and drop the BH, running aio_co_wake() directly in +qemu_rbd_completion_cb() because we are allowed to do that even if the +coroutine isn’t yet yielding and we’re in a different thread – but the +doc comment on qemu_rbd_completion_cb() says to be careful, so I decided +not to go so far here.) + +Buglink: https://issues.redhat.com/browse/RHEL-67115 +Reported-by: Junyao Zhao +Cc: qemu-stable@nongnu.org +Signed-off-by: Hanna Czenczek +Message-ID: <20251110154854.151484-3-hreitz@redhat.com> +Reviewed-by: Kevin Wolf +Signed-off-by: Kevin Wolf +(cherry picked from commit 89d22536d1a1715083ef8118fe7e6e9239f900c1) +Signed-off-by: Hanna Czenczek +--- + block/rbd.c | 12 ++++-------- + 1 file changed, 4 insertions(+), 8 deletions(-) + +diff --git a/block/rbd.c b/block/rbd.c +index 3611dc81cf..2a70b5a983 100644 +--- a/block/rbd.c ++++ b/block/rbd.c +@@ -110,9 +110,7 @@ typedef struct BDRVRBDState { + } BDRVRBDState; + + typedef struct RBDTask { +- BlockDriverState *bs; + Coroutine *co; +- bool complete; + int64_t ret; + } RBDTask; + +@@ -1309,7 +1307,6 @@ static int qemu_rbd_resize(BlockDriverState *bs, uint64_t size) + static void qemu_rbd_finish_bh(void *opaque) + { + RBDTask *task = opaque; +- task->complete = true; + aio_co_wake(task->co); + } + +@@ -1326,7 +1323,7 @@ static void qemu_rbd_completion_cb(rbd_completion_t c, RBDTask *task) + { + task->ret = rbd_aio_get_return_value(c); + rbd_aio_release(c); +- aio_bh_schedule_oneshot(bdrv_get_aio_context(task->bs), ++ aio_bh_schedule_oneshot(qemu_coroutine_get_aio_context(task->co), + qemu_rbd_finish_bh, task); + } + +@@ -1338,7 +1335,7 @@ static int coroutine_fn qemu_rbd_start_co(BlockDriverState *bs, + RBDAIOCmd cmd) + { + BDRVRBDState *s = bs->opaque; +- RBDTask task = { .bs = bs, .co = qemu_coroutine_self() }; ++ RBDTask task = { .co = qemu_coroutine_self() }; + rbd_completion_t c; + int r; + +@@ -1401,9 +1398,8 @@ static int coroutine_fn qemu_rbd_start_co(BlockDriverState *bs, + return r; + } + +- while (!task.complete) { +- qemu_coroutine_yield(); +- } ++ /* Expect exactly a single wake from qemu_rbd_finish_bh() */ ++ qemu_coroutine_yield(); + + if (task.ret < 0) { + error_report("rbd request failed: cmd %d offset %" PRIu64 " bytes %" +-- +2.47.3 + diff --git a/kvm-redhat-Add-new-rhel9.8.0-and-rhel10.2.0-machine-type.patch b/kvm-redhat-Add-new-rhel9.8.0-and-rhel10.2.0-machine-type.patch new file mode 100644 index 0000000..5a719ff --- /dev/null +++ b/kvm-redhat-Add-new-rhel9.8.0-and-rhel10.2.0-machine-type.patch @@ -0,0 +1,99 @@ +From 5c61c4d31ea23e30d79fb7d25d078d47701b2378 Mon Sep 17 00:00:00 2001 +From: Thomas Huth +Date: Thu, 18 Sep 2025 17:41:25 +0200 +Subject: [PATCH 03/10] redhat: Add new -rhel9.8.0 and -rhel10.2.0 machine + types on s390x +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 412: Add -rhel9.8.0 and -rhel10.2.0 s390x machine types and enable the CPI feature +RH-Jira: RHEL-104009 RHEL-105823 RHEL-73008 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/3] e9ad89069cb4b9badb1b5f57d78a043c64411b65 (thuth/qemu-kvm-cs) + +Upstream Status: RHEL only +JIRA: https://issues.redhat.com/browse/RHEL-105823 +JIRA: https://issues.redhat.com/browse/RHEL-104009 + +Add new -rhel9.8.0 and -rhel10.2.0 machine types that enable the +CPI feature and have the "relaxed-translation" for PCI passthrough +devices enabled by default. + +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-virtio-ccw.c | 34 +++++++++++++++++++++++++++++++--- + 1 file changed, 31 insertions(+), 3 deletions(-) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 9be423858d..4937a0c3b8 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -1172,8 +1172,18 @@ DEFINE_CCW_MACHINE(4, 2); + + #endif /* disabled for RHEL */ + ++static void ccw_rhel_machine_10_2_0_instance_options(MachineState *machine) ++{ ++} ++ ++static void ccw_rhel_machine_10_2_0_class_options(MachineClass *mc) ++{ ++} ++DEFINE_CCW_MACHINE_AS_LATEST(10, 2, 0); ++ + static void ccw_rhel_machine_10_0_0_instance_options(MachineState *machine) + { ++ ccw_rhel_machine_10_2_0_instance_options(machine); + } + + static void ccw_rhel_machine_10_0_0_class_options(MachineClass *mc) +@@ -1183,12 +1193,28 @@ static void ccw_rhel_machine_10_0_0_class_options(MachineClass *mc) + { TYPE_S390_PCI_DEVICE, "relaxed-translation", "off", }, + }; + ++ ccw_rhel_machine_10_2_0_class_options(mc); ++ + compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); + compat_props_add(mc->compat_props, hw_compat_rhel_10_2, hw_compat_rhel_10_2_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_1, hw_compat_rhel_10_1_len); + s390mc->use_cpi = false; + } +-DEFINE_CCW_MACHINE_AS_LATEST(10, 0, 0); ++DEFINE_CCW_MACHINE(10, 0, 0); ++ ++static void ccw_rhel_machine_9_8_0_instance_options(MachineState *machine) ++{ ++ ccw_rhel_machine_10_2_0_instance_options(machine); ++} ++ ++static void ccw_rhel_machine_9_8_0_class_options(MachineClass *mc) ++{ ++ ccw_rhel_machine_10_2_0_class_options(mc); ++ ++ /* NB: remember to copy this line to the *latest* RHEL 9 machine */ ++ compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); ++} ++DEFINE_CCW_MACHINE(9, 8, 0); + + static void ccw_rhel_machine_9_6_0_instance_options(MachineState *machine) + { +@@ -1197,9 +1223,11 @@ static void ccw_rhel_machine_9_6_0_instance_options(MachineState *machine) + + static void ccw_rhel_machine_9_6_0_class_options(MachineClass *mc) + { ++ /* ++ * NB: -rhel9.6 was on a par with -rhel10.0, so we derive from that ++ * instead of deriving from the 9.8 machine type ++ */ + ccw_rhel_machine_10_0_0_class_options(mc); +- +- /* NB: remember to move this line to the *latest* RHEL 9 machine */ + compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } + DEFINE_CCW_MACHINE(9, 6, 0); +-- +2.47.3 + diff --git a/kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch b/kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch new file mode 100644 index 0000000..533cfcf --- /dev/null +++ b/kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch @@ -0,0 +1,43 @@ +From 966c9312152322e6c2bb434861388a34754911c8 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 4/4] redhat: allow 5-level paging for TDX VMs +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 450: redhat: allow 5-level paging for TDX VMs +RH-Jira: RHEL-111853 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Daniel P. Berrangé +RH-Commit: [1/1] d3ffac324cfbf79d99763b9d3d29797d883f4969 (bonzini/qemu-kvm-centos) + +Without this patch, when booting a TDX guest, qemu outputs, + +qemu-kvm: TDX requires guest CPU physical bits (48) to match host CPU physical bits (52) + +This is due to different machine types in RHEL vs. upstream, and therefore +must be kept forever as a delta from upstream. + +Resolves: https://issues.redhat.com/browse/RHEL-111853 +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index dbf0fa2c91..91fe19b7d4 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -754,6 +754,7 @@ static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) + } + + object_property_set_bool(OBJECT(cpu), "pmu", false, &error_abort); ++ object_property_set_int(OBJECT(cpu), "host-phys-bits-limit", 0, &error_abort); + + /* invtsc is fixed1 for TD guest */ + object_property_set_bool(OBJECT(cpu), "invtsc", true, &error_abort); +-- +2.47.3 + diff --git a/kvm-rh-configs-enable-CONFIG_TDX-for-x86_64.patch b/kvm-rh-configs-enable-CONFIG_TDX-for-x86_64.patch new file mode 100644 index 0000000..68bc045 --- /dev/null +++ b/kvm-rh-configs-enable-CONFIG_TDX-for-x86_64.patch @@ -0,0 +1,37 @@ +From 94e93fc7673af3236dc13cd8cfe90953c48bc813 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 12 Dec 2025 12:30:30 +0100 +Subject: [PATCH 6/6] rh: configs: enable CONFIG_TDX for x86_64 + +RH-Author: Paolo Bonzini +RH-MergeRequest: 445: rh: configs: enable CONFIG_TDX for x86_64 +RH-Jira: RHEL-111853 +RH-Acked-by: Igor Mammedov +RH-Acked-by: Stefano Garzarella +RH-Commit: [1/1] 282972aa9e4f94097c68b046f6bb09c4b03705fc (bonzini/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-111853 + +TDX support is included in 10.1 but we still need to enable it, +because RHEL is built --without-default-devices. + +Signed-off-by: Paolo Bonzini +--- + configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +index 1f244f766d..9063ae1b53 100644 +--- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak ++++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +@@ -74,6 +74,7 @@ CONFIG_SERIAL_PCI=y + CONFIG_SEV=y + CONFIG_SMBIOS=y + CONFIG_SMBUS_EEPROM=y ++CONFIG_TDX=y + CONFIG_TEST_DEVICES=y + CONFIG_USB=y + CONFIG_USB_EHCI=y +-- +2.47.3 + diff --git a/kvm-rh-enable-CONFIG_USB_STORAGE_BOT.patch b/kvm-rh-enable-CONFIG_USB_STORAGE_BOT.patch new file mode 100644 index 0000000..8a0984d --- /dev/null +++ b/kvm-rh-enable-CONFIG_USB_STORAGE_BOT.patch @@ -0,0 +1,62 @@ +From c70f4d1ef65ee432a5f2cedb94447d705f6d8686 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Marc-Andr=C3=A9=20Lureau?= +Date: Mon, 3 Nov 2025 14:35:38 +0400 +Subject: [PATCH 10/10] rh: enable CONFIG_USB_STORAGE_BOT +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Marc-André Lureau +RH-MergeRequest: 417: rh: enable CONFIG_USB_STORAGE_BOT +RH-Jira: RHEL-101929 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/1] 612acd1d19c799c7441f63548648eb70b9e1dfd6 (marcandre.lureau-rh/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-101929 + +Signed-off-by: Marc-André Lureau +--- + configs/devices/aarch64-softmmu/aarch64-rh-devices.mak | 1 + + configs/devices/riscv64-softmmu/riscv64-rh-devices.mak | 1 + + configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 1 + + 3 files changed, 3 insertions(+) + +diff --git a/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak +index 855278f70e..92d0b322d0 100644 +--- a/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak ++++ b/configs/devices/aarch64-softmmu/aarch64-rh-devices.mak +@@ -19,6 +19,7 @@ CONFIG_SEMIHOSTING=y + CONFIG_USB=y + CONFIG_USB_XHCI=y + CONFIG_USB_XHCI_PCI=y ++CONFIG_USB_STORAGE_BOT=y + CONFIG_USB_STORAGE_CORE=y + CONFIG_USB_STORAGE_CLASSIC=y + CONFIG_USB_HUB=y +diff --git a/configs/devices/riscv64-softmmu/riscv64-rh-devices.mak b/configs/devices/riscv64-softmmu/riscv64-rh-devices.mak +index b5e55de916..5ac051f90b 100644 +--- a/configs/devices/riscv64-softmmu/riscv64-rh-devices.mak ++++ b/configs/devices/riscv64-softmmu/riscv64-rh-devices.mak +@@ -15,6 +15,7 @@ CONFIG_SEMIHOSTING=y + CONFIG_USB=y + CONFIG_USB_XHCI=y + CONFIG_USB_XHCI_PCI=y ++CONFIG_USB_STORAGE_BOT=y + CONFIG_USB_STORAGE_CORE=y + CONFIG_USB_STORAGE_CLASSIC=y + CONFIG_USB_HUB=y +diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +index 828cb8aa6f..1f244f766d 100644 +--- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak ++++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +@@ -79,6 +79,7 @@ CONFIG_USB=y + CONFIG_USB_EHCI=y + CONFIG_USB_EHCI_PCI=y + CONFIG_USB_SMARTCARD=y ++CONFIG_USB_STORAGE_BOT=y + CONFIG_USB_STORAGE_CORE=y + CONFIG_USB_STORAGE_CLASSIC=y + CONFIG_USB_UHCI=y +-- +2.47.3 + diff --git a/kvm-rhel-9.4.0-machine-type-compat-for-virtio-gpu-migrat.patch b/kvm-rhel-9.4.0-machine-type-compat-for-virtio-gpu-migrat.patch deleted file mode 100644 index e61f0a7..0000000 --- a/kvm-rhel-9.4.0-machine-type-compat-for-virtio-gpu-migrat.patch +++ /dev/null @@ -1,36 +0,0 @@ -From 44ee061e1904c20cae9cab5e8a62f1b506395383 Mon Sep 17 00:00:00 2001 -From: =?UTF-8?q?Marc-Andr=C3=A9=20Lureau?= -Date: Wed, 5 Jun 2024 10:28:20 +0400 -Subject: [PATCH 07/14] rhel 9.4.0 machine type compat for virtio-gpu migration -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Marc-André Lureau -RH-MergeRequest: 250: virtio-gpu: fix v2 migration -RH-Jira: RHEL-36329 -RH-Acked-by: Peter Xu -RH-Acked-by: Miroslav Rezanina -RH-Commit: [2/2] 66c98702c691e3454377f5a98230fd1f619a9a87 (marcandre.lureau-rh/qemu-kvm-centos) - -Signed-off-by: Marc-André Lureau ---- - hw/core/machine.c | 2 ++ - 1 file changed, 2 insertions(+) - -diff --git a/hw/core/machine.c b/hw/core/machine.c -index cf1d7faaaf..92609aae27 100644 ---- a/hw/core/machine.c -+++ b/hw/core/machine.c -@@ -310,6 +310,8 @@ GlobalProperty hw_compat_rhel_9_5[] = { - { TYPE_VIRTIO_IOMMU_PCI, "granule", "4k" }, - /* hw_compat_rhel_9_5 from hw_compat_8_2 */ - { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "64" }, -+ /* hw_compat_rhel_9_5 from hw_compat_8_2 */ -+ { "virtio-gpu-device", "x-scanout-vmstate-version", "1" }, - }; - const size_t hw_compat_rhel_9_5_len = G_N_ELEMENTS(hw_compat_rhel_9_5); - --- -2.39.3 - diff --git a/kvm-s390x-remove-deprecated-rhel-machine-types.patch b/kvm-s390x-remove-deprecated-rhel-machine-types.patch deleted file mode 100644 index f2615bd..0000000 --- a/kvm-s390x-remove-deprecated-rhel-machine-types.patch +++ /dev/null @@ -1,164 +0,0 @@ -From eb773f38d127117597a1640cd623f1fcd000c067 Mon Sep 17 00:00:00 2001 -From: Sebastian Ott -Date: Fri, 19 Apr 2024 16:37:57 +0200 -Subject: [PATCH 08/14] s390x: remove deprecated rhel machine types -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Thomas Huth -RH-MergeRequest: 252: s390x: remove legacy CPU types -RH-Jira: RHEL-39898 -RH-Acked-by: Cédric Le Goater -RH-Acked-by: Miroslav Rezanina -RH-Commit: [1/5] 5ed0651c38584980b1fe51592a788032526c0f2f (thuth/qemu-kvm-cs9) - -Upstream-status: N/A - -Remove the following deprecated s390x rhel specific machine types: -s390-ccw-virtio-rhel8.6.0 -s390-ccw-virtio-rhel8.5.0 -s390-ccw-virtio-rhel8.4.0 -s390-ccw-virtio-rhel8.2.0 -s390-ccw-virtio-rhel7.6.0 - -Signed-off-by: Sebastian Ott -Signed-off-by: Thomas Huth ---- - hw/s390x/s390-virtio-ccw.c | 106 +------------------------------------ - 1 file changed, 2 insertions(+), 104 deletions(-) - -diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c -index 9ad54682c6..b0b903b78c 100644 ---- a/hw/s390x/s390-virtio-ccw.c -+++ b/hw/s390x/s390-virtio-ccw.c -@@ -610,6 +610,7 @@ static void s390_nmi(NMIState *n, int cpu_index, Error **errp) - s390_cpu_restart(S390_CPU(cs)); - } - -+#if 0 /* Disabled for Red Hat Enterprise Linux */ - static ram_addr_t s390_fixup_ram_size(ram_addr_t sz) - { - /* same logic as in sclp.c */ -@@ -629,6 +630,7 @@ static ram_addr_t s390_fixup_ram_size(ram_addr_t sz) - } - return newsz; - } -+#endif /* disabled for RHEL */ - - static inline bool machine_get_aes_key_wrap(Object *obj, Error **errp) - { -@@ -1329,110 +1331,6 @@ static void ccw_machine_rhel900_class_options(MachineClass *mc) - } - DEFINE_CCW_MACHINE(rhel900, "rhel9.0.0", false); - --static void ccw_machine_rhel860_instance_options(MachineState *machine) --{ -- /* Note: The -rhel8.6.0 and -rhel9.0.0 machines are technically identical */ -- ccw_machine_rhel900_instance_options(machine); --} -- --static void ccw_machine_rhel860_class_options(MachineClass *mc) --{ -- static GlobalProperty compat[] = { -- { TYPE_S390_PCI_DEVICE, "interpret", "on", }, -- { TYPE_S390_PCI_DEVICE, "forwarding-assist", "on", }, -- }; -- -- ccw_machine_rhel900_class_options(mc); -- compat_props_add(mc->compat_props, hw_compat_rhel_8_6, hw_compat_rhel_8_6_len); -- compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); -- -- /* All RHEL machines for prior major releases are deprecated */ -- mc->deprecation_reason = rhel_old_machine_deprecation; --} --DEFINE_CCW_MACHINE(rhel860, "rhel8.6.0", false); -- --static void ccw_machine_rhel850_instance_options(MachineState *machine) --{ -- static const S390FeatInit qemu_cpu_feat = { S390_FEAT_LIST_QEMU_V6_0 }; -- -- ccw_machine_rhel860_instance_options(machine); -- -- s390_set_qemu_cpu_model(0x2964, 13, 2, qemu_cpu_feat); -- -- s390_cpudef_featoff_greater(16, 1, S390_FEAT_NNPA); -- s390_cpudef_featoff_greater(16, 1, S390_FEAT_VECTOR_PACKED_DECIMAL_ENH2); -- s390_cpudef_featoff_greater(16, 1, S390_FEAT_BEAR_ENH); -- s390_cpudef_featoff_greater(16, 1, S390_FEAT_RDP); -- s390_cpudef_featoff_greater(16, 1, S390_FEAT_PAI); --} -- --static void ccw_machine_rhel850_class_options(MachineClass *mc) --{ -- static GlobalProperty compat[] = { -- { TYPE_S390_PCI_DEVICE, "interpret", "off", }, -- { TYPE_S390_PCI_DEVICE, "forwarding-assist", "off", }, -- }; -- -- ccw_machine_rhel860_class_options(mc); -- compat_props_add(mc->compat_props, hw_compat_rhel_8_5, hw_compat_rhel_8_5_len); -- compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); -- mc->smp_props.prefer_sockets = true; --} --DEFINE_CCW_MACHINE(rhel850, "rhel8.5.0", false); -- --static void ccw_machine_rhel840_instance_options(MachineState *machine) --{ -- ccw_machine_rhel850_instance_options(machine); --} -- --static void ccw_machine_rhel840_class_options(MachineClass *mc) --{ -- ccw_machine_rhel850_class_options(mc); -- compat_props_add(mc->compat_props, hw_compat_rhel_8_4, hw_compat_rhel_8_4_len); --} --DEFINE_CCW_MACHINE(rhel840, "rhel8.4.0", false); -- --static void ccw_machine_rhel820_instance_options(MachineState *machine) --{ -- ccw_machine_rhel840_instance_options(machine); --} -- --static void ccw_machine_rhel820_class_options(MachineClass *mc) --{ -- ccw_machine_rhel840_class_options(mc); -- mc->fixup_ram_size = s390_fixup_ram_size; -- /* we did not publish a rhel8.3.0 machine */ -- compat_props_add(mc->compat_props, hw_compat_rhel_8_3, hw_compat_rhel_8_3_len); -- compat_props_add(mc->compat_props, hw_compat_rhel_8_2, hw_compat_rhel_8_2_len); --} --DEFINE_CCW_MACHINE(rhel820, "rhel8.2.0", false); -- --static void ccw_machine_rhel760_instance_options(MachineState *machine) --{ -- static const S390FeatInit qemu_cpu_feat = { S390_FEAT_LIST_QEMU_V3_1 }; -- -- ccw_machine_rhel820_instance_options(machine); -- -- s390_set_qemu_cpu_model(0x2827, 12, 2, qemu_cpu_feat); -- -- /* The multiple-epoch facility was not available with rhel7.6.0 on z14GA1 */ -- s390_cpudef_featoff(14, 1, S390_FEAT_MULTIPLE_EPOCH); -- s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_QSIE); -- s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_QTOUE); -- s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_STOE); -- s390_cpudef_featoff(14, 1, S390_FEAT_PTFF_STOUE); --} -- --static void ccw_machine_rhel760_class_options(MachineClass *mc) --{ -- ccw_machine_rhel820_class_options(mc); -- /* We never published the s390x version of RHEL-AV 8.0 and 8.1, so add this here */ -- compat_props_add(mc->compat_props, hw_compat_rhel_8_1, hw_compat_rhel_8_1_len); -- compat_props_add(mc->compat_props, hw_compat_rhel_8_0, hw_compat_rhel_8_0_len); -- compat_props_add(mc->compat_props, hw_compat_rhel_7_6, hw_compat_rhel_7_6_len); --} --DEFINE_CCW_MACHINE(rhel760, "rhel7.6.0", false); -- - static void ccw_machine_register_types(void) - { - type_register_static(&ccw_machine_info); --- -2.39.3 - diff --git a/kvm-s390x-select-correct-components-for-no-board-build.patch b/kvm-s390x-select-correct-components-for-no-board-build.patch deleted file mode 100644 index f2fc71b..0000000 --- a/kvm-s390x-select-correct-components-for-no-board-build.patch +++ /dev/null @@ -1,41 +0,0 @@ -From 874c2ad98804caf0db862c2a45db66a9bceb4fc4 Mon Sep 17 00:00:00 2001 -From: Paolo Bonzini -Date: Thu, 9 May 2024 19:00:35 +0200 -Subject: [PATCH 09/14] s390x: select correct components for no-board build -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Thomas Huth -RH-MergeRequest: 252: s390x: remove legacy CPU types -RH-Jira: RHEL-39898 -RH-Acked-by: Cédric Le Goater -RH-Acked-by: Miroslav Rezanina -RH-Commit: [2/5] 441dfae234f21f801ac9f9e417e96e2edff48bd4 (thuth/qemu-kvm-cs9) - -Signed-off-by: Paolo Bonzini -Reviewed-by: Thomas Huth -Message-ID: <20240509170044.190795-5-pbonzini@redhat.com> -Signed-off-by: Paolo Bonzini -(cherry picked from commit e799b65faef129f2905bd9bf66c30aaaa7115dac) -Conflicts: - .gitlab-ci.d/buildtest.yml - (skipped the changes to the CI files, they don't apply and - are not needed in downstream) -Signed-off-by: Thomas Huth ---- - target/s390x/Kconfig | 2 ++ - 1 file changed, 2 insertions(+) - -diff --git a/target/s390x/Kconfig b/target/s390x/Kconfig -index 72da48136c..d886be48b4 100644 ---- a/target/s390x/Kconfig -+++ b/target/s390x/Kconfig -@@ -1,2 +1,4 @@ - config S390X - bool -+ select PCI -+ select S390_FLIC --- -2.39.3 - diff --git a/kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch b/kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch new file mode 100644 index 0000000..3f4a9a4 --- /dev/null +++ b/kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch @@ -0,0 +1,131 @@ +From d6784a187e586d6e8541ab8b1ab41541c768e614 Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:32 -0500 +Subject: [PATCH 3/7] scsi: add error reporting to scsi_SG_IO() + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [2/5] 7858df3edd4f42eb1b9d3bf515b3470dd812327e (stefanha/centos-stream-qemu-kvm) + +Report the details of the SG_IO ioctl failure if an Error pointer is +provided. This information aids troubleshooting and will be used by the +SCSI Persistent Reservations migration code. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-3-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit 6302598fe538206fd02494007ab5d218524dc7a7) +Signed-off-by: Stefan Hajnoczi +--- + hw/scsi/scsi-disk.c | 2 +- + hw/scsi/scsi-generic.c | 33 ++++++++++++++++++++++++++++----- + include/hw/scsi/scsi.h | 2 +- + 3 files changed, 30 insertions(+), 7 deletions(-) + +diff --git a/hw/scsi/scsi-disk.c b/hw/scsi/scsi-disk.c +index ca0214c404..c736988ef6 100644 +--- a/hw/scsi/scsi-disk.c ++++ b/hw/scsi/scsi-disk.c +@@ -2749,7 +2749,7 @@ static int get_device_type(SCSIDiskState *s) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(s->qdev.conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), s->qdev.io_timeout); ++ buf, sizeof(buf), s->qdev.io_timeout, NULL); + if (ret < 0) { + return -1; + } +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index b27ad48f18..4f851186f6 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -526,10 +526,10 @@ static int read_naa_id(const uint8_t *p, uint64_t *p_wwn) + + int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, + uint8_t cmd_size, uint8_t *buf, uint8_t buf_size, +- uint32_t timeout) ++ uint32_t timeout, Error **errp) + { + sg_io_hdr_t io_header; +- uint8_t sensebuf[8]; ++ uint8_t sensebuf[8] = {}; + int ret; + + memset(&io_header, 0, sizeof(io_header)); +@@ -549,6 +549,29 @@ int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, + io_header.driver_status || io_header.host_status) { + trace_scsi_generic_ioctl_sgio_done(cmd[0], ret, io_header.status, + io_header.host_status); ++ if (ret < 0) { ++ error_setg_errno(errp, -ret, "SG_IO ioctl failed"); ++ } else { ++ g_autofree char *sensebuf_hex = ++ g_strdup_printf("%02x%02x%02x%02x%02x%02x%02x%02x", ++ sensebuf[0], ++ sensebuf[1], ++ sensebuf[2], ++ sensebuf[3], ++ sensebuf[4], ++ sensebuf[5], ++ sensebuf[6], ++ sensebuf[7]); ++ ++ error_setg(errp, "SG_IO SCSI command failed with status=0x%x " ++ "driver_status=0x%x host_status=0x%x sensebuf=%s " ++ "sb_len_wr=%u", ++ io_header.status, ++ io_header.driver_status, ++ io_header.host_status, ++ sensebuf_hex, ++ io_header.sb_len_wr); ++ } + return -1; + } + return 0; +@@ -575,7 +598,7 @@ static void scsi_generic_set_vpd_bl_emulation(SCSIDevice *s) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ buf, sizeof(buf), s->io_timeout, NULL); + if (ret < 0) { + /* + * Do not assume anything if we can't retrieve the +@@ -611,7 +634,7 @@ static void scsi_generic_read_device_identification(SCSIDevice *s) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ buf, sizeof(buf), s->io_timeout, NULL); + if (ret < 0) { + return; + } +@@ -663,7 +686,7 @@ static int get_stream_blocksize(BlockBackend *blk) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), 6); ++ buf, sizeof(buf), 6, NULL); + if (ret < 0) { + return -1; + } +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index a79c3aadfc..977b7e4d55 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -237,7 +237,7 @@ void scsi_device_unit_attention_reported(SCSIDevice *dev); + void scsi_generic_read_device_inquiry(SCSIDevice *dev); + int scsi_device_get_sense(SCSIDevice *dev, uint8_t *buf, int len, bool fixed); + int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, uint8_t cmd_size, +- uint8_t *buf, uint8_t buf_size, uint32_t timeout); ++ uint8_t *buf, uint8_t buf_size, uint32_t timeout, Error **errp); + SCSIDevice *scsi_device_find(SCSIBus *bus, int channel, int target, int lun); + SCSIDevice *scsi_device_get(SCSIBus *bus, int channel, int target, int lun); + +-- +2.47.3 + diff --git a/kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch b/kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch new file mode 100644 index 0000000..3224362 --- /dev/null +++ b/kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch @@ -0,0 +1,117 @@ +From af75da80d87f9b2490368edb3818bf0692c66882 Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:31 -0500 +Subject: [PATCH 2/7] scsi: generalize scsi_SG_IO_FROM_DEV() to scsi_SG_IO() + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [1/5] 45295e0aa8da535c0fca388e6b8e3a0665a4cf7c (stefanha/centos-stream-qemu-kvm) + +Add a direction argument so that scsi_SG_IO() can be used for +SG_DXFER_FROM_DEV and SG_DXFER_TO_DEV transfers. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-2-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit 03396b9afcf93964bb4dbb9d0cd7387ba0f63aa3) +Signed-off-by: Stefan Hajnoczi +--- + hw/scsi/scsi-disk.c | 4 ++-- + hw/scsi/scsi-generic.c | 18 ++++++++++-------- + include/hw/scsi/scsi.h | 4 ++-- + 3 files changed, 14 insertions(+), 12 deletions(-) + +diff --git a/hw/scsi/scsi-disk.c b/hw/scsi/scsi-disk.c +index b4782c6248..ca0214c404 100644 +--- a/hw/scsi/scsi-disk.c ++++ b/hw/scsi/scsi-disk.c +@@ -2748,8 +2748,8 @@ static int get_device_type(SCSIDiskState *s) + cmd[0] = INQUIRY; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(s->qdev.conf.blk, cmd, sizeof(cmd), +- buf, sizeof(buf), s->qdev.io_timeout); ++ ret = scsi_SG_IO(s->qdev.conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->qdev.io_timeout); + if (ret < 0) { + return -1; + } +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index 9e380a2109..b27ad48f18 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -524,8 +524,9 @@ static int read_naa_id(const uint8_t *p, uint64_t *p_wwn) + return -EINVAL; + } + +-int scsi_SG_IO_FROM_DEV(BlockBackend *blk, uint8_t *cmd, uint8_t cmd_size, +- uint8_t *buf, uint8_t buf_size, uint32_t timeout) ++int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, ++ uint8_t cmd_size, uint8_t *buf, uint8_t buf_size, ++ uint32_t timeout) + { + sg_io_hdr_t io_header; + uint8_t sensebuf[8]; +@@ -533,7 +534,7 @@ int scsi_SG_IO_FROM_DEV(BlockBackend *blk, uint8_t *cmd, uint8_t cmd_size, + + memset(&io_header, 0, sizeof(io_header)); + io_header.interface_id = 'S'; +- io_header.dxfer_direction = SG_DXFER_FROM_DEV; ++ io_header.dxfer_direction = direction; + io_header.dxfer_len = buf_size; + io_header.dxferp = buf; + io_header.cmdp = cmd; +@@ -573,8 +574,8 @@ static void scsi_generic_set_vpd_bl_emulation(SCSIDevice *s) + cmd[2] = 0x00; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(s->conf.blk, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout); + if (ret < 0) { + /* + * Do not assume anything if we can't retrieve the +@@ -609,8 +610,8 @@ static void scsi_generic_read_device_identification(SCSIDevice *s) + cmd[2] = 0x83; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(s->conf.blk, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout); + if (ret < 0) { + return; + } +@@ -661,7 +662,8 @@ static int get_stream_blocksize(BlockBackend *blk) + cmd[0] = MODE_SENSE; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(blk, cmd, sizeof(cmd), buf, sizeof(buf), 6); ++ ret = scsi_SG_IO(blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), 6); + if (ret < 0) { + return -1; + } +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index 90ee192b4d..a79c3aadfc 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -236,8 +236,8 @@ void scsi_device_report_change(SCSIDevice *dev, SCSISense sense); + void scsi_device_unit_attention_reported(SCSIDevice *dev); + void scsi_generic_read_device_inquiry(SCSIDevice *dev); + int scsi_device_get_sense(SCSIDevice *dev, uint8_t *buf, int len, bool fixed); +-int scsi_SG_IO_FROM_DEV(BlockBackend *blk, uint8_t *cmd, uint8_t cmd_size, +- uint8_t *buf, uint8_t buf_size, uint32_t timeout); ++int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, uint8_t cmd_size, ++ uint8_t *buf, uint8_t buf_size, uint32_t timeout); + SCSIDevice *scsi_device_find(SCSIBus *bus, int channel, int target, int lun); + SCSIDevice *scsi_device_get(SCSIBus *bus, int channel, int target, int lun); + +-- +2.47.3 + diff --git a/kvm-scsi-save-load-SCSI-reservation-state.patch b/kvm-scsi-save-load-SCSI-reservation-state.patch new file mode 100644 index 0000000..7660bff --- /dev/null +++ b/kvm-scsi-save-load-SCSI-reservation-state.patch @@ -0,0 +1,348 @@ +From 91e9a0fbb3d6ee8726ea0ef17ff17404f823577a Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:34 -0500 +Subject: [PATCH 5/7] scsi: save/load SCSI reservation state + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [4/5] 2ebcfedf1a9e54433ed1ba3593d4fbcbadcd644b (stefanha/centos-stream-qemu-kvm) + +Add a vmstate subsection to SCSIDiskState so that scsi-block devices can +transfer their reservation state during live migration. Upon loading the +subsection, the destination QEMU invokes the PERSISTENT RESERVE OUT +command's PREEMPT service action to atomically move the reservation from +the source I_T nexus to the destination I_T nexus. This results in +transparent live migration of SCSI reservations. + +This approach is incomplete since SCSI reservations are cooperative and +other hosts could interfere. Neither the source QEMU nor the destination +QEMU are aware of changes made by other hosts. The assumption is that +reservation is not taken over by a third host without cooperation from +the source host. + +I considered adding the vmstate subsection to SCSIDevice instead of +SCSIDiskState, since reservations are part of the SCSI Primary Commands +that other devices apart from disks could support. However, due to +fragility of migrating reservations, we will probably limit support to +scsi-block and maybe scsi-disk in the future. In the end, I think it +makes sense to place this within scsi-disk.c. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-5-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit ab57b51f1375b6a6f098a74c6f79207a9630948d) +Signed-off-by: Stefan Hajnoczi + +Conflicts: +- hw/core/machine.c + + Downstream does not have hw_compat_10_2. Make sure that machine types + prior to RHEL 10.2 default to migrate-pr=off. + +- hw/scsi/scsi-disk.c + + Downstream is missing commit 40de712a89d8f ("migration: Add + error-parameterized function variants in VMSD struct"), so add a + .post_load() wrapper function that reports the Error and returns + -EINVAL. +--- + hw/core/machine.c | 1 + + hw/scsi/scsi-disk.c | 99 +++++++++++++++++++++++++++++++++++++++++- + hw/scsi/scsi-generic.c | 83 +++++++++++++++++++++++++++++++++++ + hw/scsi/trace-events | 1 + + include/hw/scsi/scsi.h | 1 + + 5 files changed, 184 insertions(+), 1 deletion(-) + +diff --git a/hw/core/machine.c b/hw/core/machine.c +index 2a1a42cebc..2b339f6a13 100644 +--- a/hw/core/machine.c ++++ b/hw/core/machine.c +@@ -295,6 +295,7 @@ const char *rhel_old_machine_deprecation = + "machine types for previous major releases are deprecated"; + + GlobalProperty hw_compat_rhel_10_2[] = { ++ { "scsi-block", "migrate-pr", "off" }, + /* hw_compat_rhel_10_2 from hw_compat_10_0 */ + { "scsi-hd", "dpofua", "off" }, + /* hw_compat_rhel_10_2 from hw_compat_10_0 */ +diff --git a/hw/scsi/scsi-disk.c b/hw/scsi/scsi-disk.c +index c736988ef6..e7b821ad19 100644 +--- a/hw/scsi/scsi-disk.c ++++ b/hw/scsi/scsi-disk.c +@@ -28,6 +28,7 @@ + #include "qemu/hw-version.h" + #include "qemu/memalign.h" + #include "hw/scsi/scsi.h" ++#include "migration/misc.h" + #include "migration/qemu-file-types.h" + #include "migration/vmstate.h" + #include "hw/scsi/emulation.h" +@@ -122,6 +123,7 @@ struct SCSIDiskState { + */ + uint16_t rotation_rate; + bool migrate_emulated_scsi_request; ++ NotifierWithReturn migration_notifier; + }; + + static void scsi_free_request(SCSIRequest *req) +@@ -2737,6 +2739,29 @@ static SCSIRequest *scsi_new_request(SCSIDevice *d, uint32_t tag, uint32_t lun, + } + + #ifdef __linux__ ++/* ++ * Preempt on the SCSI Persistent Reservation on the source when migration ++ * fails because the destination may have already preempted and we need to get ++ * the reservation back. ++ */ ++static int scsi_block_migration_notifier(NotifierWithReturn *notifier, ++ MigrationEvent *e, Error **errp) ++{ ++ if (e->type == MIG_EVENT_PRECOPY_FAILED) { ++ SCSIDiskState *s = ++ container_of(notifier, SCSIDiskState, migration_notifier); ++ SCSIDevice *d = &s->qdev; ++ Error *local_err = NULL; ++ ++ if (!scsi_generic_pr_state_preempt(d, &local_err)) { ++ /* MIG_EVENT_PRECOPY_FAILED cannot fail, so just warn */ ++ error_prepend(&local_err, "scsi-block migration rollback: "); ++ warn_report_err(local_err); ++ } ++ } ++ return 0; ++} ++ + static int get_device_type(SCSIDiskState *s) + { + uint8_t cmd[16]; +@@ -2815,6 +2840,16 @@ static void scsi_block_realize(SCSIDevice *dev, Error **errp) + + scsi_realize(&s->qdev, errp); + scsi_generic_read_device_inquiry(&s->qdev); ++ ++ migration_add_notifier(&s->migration_notifier, ++ scsi_block_migration_notifier); ++} ++ ++static void scsi_block_unrealize(SCSIDevice *dev) ++{ ++ SCSIDiskState *s = DO_UPCAST(SCSIDiskState, qdev, dev); ++ ++ migration_remove_notifier(&s->migration_notifier); + } + + typedef struct SCSIBlockReq { +@@ -3209,6 +3244,60 @@ static const Property scsi_hd_properties[] = { + DEFINE_BLOCK_CHS_PROPERTIES(SCSIDiskState, qdev.conf), + }; + ++#ifdef __linux__ ++static bool scsi_disk_pr_state_post_load_errp(void *opaque, int version_id, ++ Error **errp) ++{ ++ SCSIDiskState *s = opaque; ++ SCSIDevice *dev = &s->qdev; ++ ++ return scsi_generic_pr_state_preempt(dev, errp); ++} ++ ++static int scsi_disk_pr_state_post_load(void *opaque, int version_id) ++{ ++ SCSIDiskState *s = opaque; ++ Error *errp = NULL; ++ ++ if (scsi_disk_pr_state_post_load_errp(s, version_id, &errp)) { ++ return 0; ++ } else { ++ error_report_err(errp); ++ return -EINVAL; ++ } ++} ++ ++static bool scsi_disk_pr_state_needed(void *opaque) ++{ ++ SCSIDiskState *s = opaque; ++ SCSIPRState *pr_state = &s->qdev.pr_state; ++ bool ret; ++ ++ if (!s->qdev.migrate_pr) { ++ return false; ++ } ++ ++ /* A reservation requires a key, so checking this field is enough */ ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ ret = pr_state->key; ++ } ++ return ret; ++} ++ ++static const VMStateDescription vmstate_scsi_disk_pr_state = { ++ .name = "scsi-disk/pr", ++ .version_id = 1, ++ .minimum_version_id = 1, ++ .post_load = scsi_disk_pr_state_post_load, ++ .needed = scsi_disk_pr_state_needed, ++ .fields = (const VMStateField[]) { ++ VMSTATE_UINT64(qdev.pr_state.key, SCSIDiskState), ++ VMSTATE_UINT8(qdev.pr_state.resv_type, SCSIDiskState), ++ VMSTATE_END_OF_LIST() ++ } ++}; ++#endif /* __linux__ */ ++ + static const VMStateDescription vmstate_scsi_disk_state = { + .name = "scsi-disk", + .version_id = 1, +@@ -3221,7 +3310,13 @@ static const VMStateDescription vmstate_scsi_disk_state = { + VMSTATE_BOOL(tray_open, SCSIDiskState), + VMSTATE_BOOL(tray_locked, SCSIDiskState), + VMSTATE_END_OF_LIST() +- } ++ }, ++ .subsections = (const VMStateDescription * const []) { ++#ifdef __linux__ ++ &vmstate_scsi_disk_pr_state, ++#endif ++ NULL ++ }, + }; + + static void scsi_hd_class_initfn(ObjectClass *klass, const void *data) +@@ -3301,6 +3396,7 @@ static const Property scsi_block_properties[] = { + -1), + DEFINE_PROP_UINT32("io_timeout", SCSIDiskState, qdev.io_timeout, + DEFAULT_IO_TIMEOUT), ++ DEFINE_PROP_BOOL("migrate-pr", SCSIDiskState, qdev.migrate_pr, true), + }; + + static void scsi_block_class_initfn(ObjectClass *klass, const void *data) +@@ -3310,6 +3406,7 @@ static void scsi_block_class_initfn(ObjectClass *klass, const void *data) + SCSIDiskClass *sdc = SCSI_DISK_BASE_CLASS(klass); + + sc->realize = scsi_block_realize; ++ sc->unrealize = scsi_block_unrealize; + sc->alloc_req = scsi_block_new_request; + sc->parse_cdb = scsi_block_parse_cdb; + sdc->dma_readv = scsi_block_dma_readv; +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index a6280eaa87..b8b3f399f0 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -423,6 +423,89 @@ static void scsi_handle_persistent_reserve_out_reply( + } + } + ++static bool scsi_generic_pr_register(SCSIDevice *s, uint64_t key, Error **errp) ++{ ++ uint8_t cmd[10] = {}; ++ uint8_t buf[24] = {}; ++ uint64_t key_be = cpu_to_be64(key); ++ int ret; ++ ++ cmd[0] = PERSISTENT_RESERVE_OUT; ++ cmd[1] = PRO_REGISTER; ++ cmd[8] = sizeof(buf); ++ memcpy(&buf[8], &key_be, sizeof(key_be)); ++ ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_TO_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout, errp); ++ if (ret < 0) { ++ error_prepend(errp, "PERSISTENT RESERVE OUT with REGISTER"); ++ return false; ++ } ++ return true; ++} ++ ++static bool scsi_generic_pr_preempt(SCSIDevice *s, uint64_t key, ++ uint8_t resv_type, Error **errp) ++{ ++ uint8_t cmd[10] = {}; ++ uint8_t buf[24] = {}; ++ uint64_t key_be = cpu_to_be64(key); ++ int ret; ++ ++ cmd[0] = PERSISTENT_RESERVE_OUT; ++ cmd[1] = PRO_PREEMPT; ++ cmd[2] = resv_type & 0xf; ++ cmd[8] = sizeof(buf); ++ memcpy(&buf[0], &key_be, sizeof(key_be)); ++ memcpy(&buf[8], &key_be, sizeof(key_be)); ++ ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_TO_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout, errp); ++ if (ret < 0) { ++ error_prepend(errp, "PERSISTENT RESERVE OUT with PREEMPT"); ++ return false; ++ } ++ return true; ++} ++ ++/* Register keys and preempt reservations after live migration */ ++bool scsi_generic_pr_state_preempt(SCSIDevice *s, Error **errp) ++{ ++ SCSIPRState *pr_state = &s->pr_state; ++ uint64_t key; ++ uint8_t resv_type; ++ ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ key = pr_state->key; ++ resv_type = pr_state->resv_type; ++ } ++ ++ trace_scsi_generic_pr_state_preempt(key, resv_type); ++ ++ if (key) { ++ if (!scsi_generic_pr_register(s, key, errp)) { ++ return false; ++ } ++ ++ /* ++ * Two cases: ++ * ++ * 1. There is no reservation (resv_type is 0) and the other I_T nexus ++ * will be unregistered. This is important so the source host does ++ * not leak registered keys across live migration. ++ * ++ * 2. There is a reservation (resv_type is not 0) and the other I_T ++ * nexus will be unregistered and its reservation is atomically ++ * taken over by us. This is the scenario where a reservation is ++ * migrated along with the guest. ++ */ ++ if (!scsi_generic_pr_preempt(s, key, resv_type, errp)) { ++ return false; ++ } ++ } ++ return true; ++} ++ + static void scsi_read_complete(void * opaque, int ret) + { + SCSIGenericReq *r = (SCSIGenericReq *)opaque; +diff --git a/hw/scsi/trace-events b/hw/scsi/trace-events +index fdb87a237f..ab7a6f4cea 100644 +--- a/hw/scsi/trace-events ++++ b/hw/scsi/trace-events +@@ -362,3 +362,4 @@ scsi_generic_aio_sgio_command(uint32_t tag, uint8_t cmd, uint32_t timeout) "gene + scsi_generic_ioctl_sgio_command(uint8_t cmd, uint32_t timeout) "generic ioctl sgio: cmd=0x%x timeout=%u" + scsi_generic_ioctl_sgio_done(uint8_t cmd, int ret, uint8_t status, uint8_t host_status) "generic ioctl sgio: cmd=0x%x ret=%d status=0x%x host_status=0x%x" + scsi_generic_persistent_reserve_out_reply(uint8_t service_action, uint8_t resv_type, uint64_t old_key, uint64_t new_key) "persistent reserve out reply service_action=%u resv_type=%u old_key=0x%" PRIx64 " new_key=0x%" PRIx64 ++scsi_generic_pr_state_preempt(uint64_t key, uint8_t resv_type) "key=0x%" PRIx64 " resv_type=%u" +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index 7e120fdd6a..f61c63c5ea 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -253,6 +253,7 @@ SCSIDevice *scsi_device_get(SCSIBus *bus, int channel, int target, int lun); + + /* scsi-generic.c. */ + extern const SCSIReqOps scsi_generic_req_ops; ++bool scsi_generic_pr_state_preempt(SCSIDevice *s, Error **errp); + + /* scsi-disk.c */ + #define SCSI_DISK_QUIRK_MODE_PAGE_APPLE_VENDOR 0 +-- +2.47.3 + diff --git a/kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch b/kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch new file mode 100644 index 0000000..82a3403 --- /dev/null +++ b/kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch @@ -0,0 +1,331 @@ +From 220f405f15c997b636048aae3d4df44c4518e71d Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:33 -0500 +Subject: [PATCH 4/7] scsi: track SCSI reservation state for live migration + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [3/5] 563e4c4917709c65ca503898488131e3babf16e7 (stefanha/centos-stream-qemu-kvm) + +SCSI Persistent Reservations are stateful and external to the guest. In +order to transparently move reservations to the destination host during +live migration, it is necessary to track the state built up on the +source host before migration. Only then can the destination host ensure +an equivalent state is restored upon migration. + +Snoop on successful PERSISTENT RESERVE OUT commands and save the +reservation key and reservation type. This will allow registered keys +and reservations to be migrated. + +Also patch PERSISTENT RESERVE IN replies with the REPORT CAPABILITIES +service action since features that involve the physical SCSI bus target +ports must not be exposed to the guest (it sees a virtual SCSI bus). + +Usually this plays out as follows: +1. The guest invokes the REGISTER service action to register a + reservation key on its I_T nexus. +2. The guest invokes the RESERVE service action to create a reservation + using the previously-registered key. + +This commit implements the snooping and stores the reservation key and +type (if any) for each LUN. The snooped PR state and the migrate_pr flag +to enable PR migration will be used in later commits. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-4-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit 70f0e0cedb2e0d7511cdebbce9b21a01bba55b74) +Signed-off-by: Stefan Hajnoczi +--- + hw/scsi/scsi-bus.c | 3 + + hw/scsi/scsi-generic.c | 165 +++++++++++++++++++++++++++++++++++++++ + hw/scsi/trace-events | 1 + + include/hw/scsi/scsi.h | 10 +++ + include/scsi/constants.h | 21 +++++ + 5 files changed, 200 insertions(+) + +diff --git a/hw/scsi/scsi-bus.c b/hw/scsi/scsi-bus.c +index 9b12ee7f1c..878ccf62c9 100644 +--- a/hw/scsi/scsi-bus.c ++++ b/hw/scsi/scsi-bus.c +@@ -393,6 +393,7 @@ static void scsi_qdev_realize(DeviceState *qdev, Error **errp) + } + + qemu_mutex_init(&dev->requests_lock); ++ qemu_mutex_init(&dev->pr_state.mutex); + QTAILQ_INIT(&dev->requests); + scsi_device_realize(dev, &local_err); + if (local_err) { +@@ -417,6 +418,8 @@ static void scsi_qdev_unrealize(DeviceState *qdev) + + scsi_device_unrealize(dev); + ++ qemu_mutex_destroy(&dev->pr_state.mutex); ++ + blockdev_mark_auto_del(dev->conf.blk); + } + +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index 4f851186f6..a6280eaa87 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -264,6 +264,165 @@ static int scsi_generic_emulate_block_limits(SCSIGenericReq *r, SCSIDevice *s) + return r->buflen; + } + ++/* ++ * Patch persistent reservation capabilities that are not emulated. ++ */ ++static void scsi_handle_persistent_reserve_in_reply(SCSIGenericReq *r, ++ SCSIDevice *s) ++{ ++ uint8_t service_action = r->req.cmd.buf[1] & 0x1f; ++ ++ if (!s->migrate_pr) { ++ return; /* when migration is disabled there is no need for patching */ ++ } ++ ++ if (service_action == PRI_REPORT_CAPABILITIES) { ++ assert(r->buflen >= 3); ++ ++ /* ++ * Clear specify initiator ports capable (SIP_C) and all target ports ++ * capable (ATC_C). ++ * ++ * SPEC_I_PT is not supported because the guest sees an emulated SCSI ++ * bus and does not have the underlying transport IDs needed to use ++ * SPEC_I_PT. ++ * ++ * ALL_TG_PT is not supported because we only track the state of this ++ * emulated I_T nexus, not the underlying device's target ports. ++ */ ++ r->buf[2] &= ~0xc; ++ } ++} ++ ++static int scsi_generic_read_reservation(SCSIDevice *s, uint64_t *key, ++ uint8_t *resv_type, Error **errp) ++{ ++ uint8_t cmd[10] = {}; ++ uint8_t buf[24] = {}; ++ uint32_t additional_length; ++ int ret; ++ ++ *key = 0; ++ *resv_type = 0; ++ ++ cmd[0] = PERSISTENT_RESERVE_IN; ++ cmd[1] = PRI_READ_RESERVATION; ++ cmd[8] = sizeof(buf); ++ ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout, errp); ++ if (ret < 0) { ++ return ret; ++ } ++ ++ memcpy(&additional_length, &buf[4], sizeof(additional_length)); ++ be32_to_cpus(&additional_length); ++ ++ if (additional_length >= 0x10) { ++ memcpy(key, &buf[8], sizeof(*key)); ++ be64_to_cpus(key); ++ ++ *resv_type = buf[21] & 0xf; ++ } ++ return 0; ++} ++ ++/* ++ * Snoop changes to registered keys and reservations so that this information ++ * can be transferred during live migration. ++ */ ++static void scsi_handle_persistent_reserve_out_reply( ++ SCSIGenericReq *r, ++ SCSIDevice *s) ++{ ++ SCSIPRState *pr_state = &s->pr_state; ++ uint8_t service_action = r->req.cmd.buf[1] & 0x1f; ++ uint8_t resv_type = r->req.cmd.buf[2] & 0xf; ++ uint64_t old_key; ++ uint64_t new_key; ++ ++ assert(r->buflen >= 16); ++ memcpy(&old_key, &r->buf[0], sizeof(old_key)); ++ memcpy(&new_key, &r->buf[8], sizeof(new_key)); ++ be64_to_cpus(&old_key); ++ be64_to_cpus(&new_key); ++ ++ trace_scsi_generic_persistent_reserve_out_reply(service_action, resv_type, ++ old_key, new_key); ++ ++ switch (service_action) { ++ case PRO_REGISTER: /* fallthrough */ ++ case PRO_REGISTER_AND_IGNORE_EXISTING_KEY: ++ if (service_action == PRO_REGISTER && old_key == 0 && new_key == 0) { ++ /* Do nothing */ ++ } else { ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->key = new_key; ++ if (new_key == 0) { ++ pr_state->resv_type = 0; /* release reservation */ ++ } ++ } ++ } ++ break; ++ ++ case PRO_RESERVE: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->resv_type = resv_type; ++ } ++ break; ++ ++ case PRO_RELEASE: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->resv_type = 0; ++ } ++ break; ++ ++ case PRO_CLEAR: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->key = 0; ++ pr_state->resv_type = 0; ++ } ++ break; ++ ++ case PRO_REPLACE_LOST_RESERVATION: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->key = new_key; ++ pr_state->resv_type = resv_type; ++ } ++ break; ++ ++ case PRO_PREEMPT: /* fallthrough */ ++ case PRO_PREEMPT_AND_ABORT: { ++ uint64_t dev_key; ++ uint8_t dev_resv_type; ++ Error *local_err = NULL; ++ ++ /* Not enough information to know actual state, ask the device */ ++ if (!scsi_generic_read_reservation(s, &dev_key, &dev_resv_type, ++ &local_err)) { ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ if (pr_state->key == dev_key) { ++ pr_state->resv_type = dev_resv_type; ++ } else { ++ pr_state->resv_type = 0; ++ } ++ } ++ } ++ if (local_err) { ++ warn_report_err(local_err); ++ } ++ break; ++ } ++ ++ /* ++ * PRO_REGISTER_AND_MOVE cannot be implemented since it involves the ++ * physical SCSI bus target ports. ++ */ ++ default: ++ break; /* do nothing */ ++ } ++} ++ + static void scsi_read_complete(void * opaque, int ret) + { + SCSIGenericReq *r = (SCSIGenericReq *)opaque; +@@ -346,6 +505,9 @@ static void scsi_read_complete(void * opaque, int ret) + if (r->req.cmd.buf[0] == INQUIRY) { + len = scsi_handle_inquiry_reply(r, s, len); + } ++ if (r->req.cmd.buf[0] == PERSISTENT_RESERVE_IN) { ++ scsi_handle_persistent_reserve_in_reply(r, s); ++ } + + req_complete: + scsi_req_data(&r->req, len); +@@ -395,6 +557,9 @@ static void scsi_write_complete(void * opaque, int ret) + s->blocksize = (r->buf[9] << 16) | (r->buf[10] << 8) | r->buf[11]; + trace_scsi_generic_write_complete_blocksize(s->blocksize); + } ++ if (r->req.cmd.buf[0] == PERSISTENT_RESERVE_OUT) { ++ scsi_handle_persistent_reserve_out_reply(r, s); ++ } + + scsi_command_complete_noio(r, ret); + } +diff --git a/hw/scsi/trace-events b/hw/scsi/trace-events +index 6c2788e202..fdb87a237f 100644 +--- a/hw/scsi/trace-events ++++ b/hw/scsi/trace-events +@@ -361,3 +361,4 @@ scsi_generic_realize_blocksize(int blocksize) "block size %d" + scsi_generic_aio_sgio_command(uint32_t tag, uint8_t cmd, uint32_t timeout) "generic aio sgio: tag=0x%x cmd=0x%x timeout=%u" + scsi_generic_ioctl_sgio_command(uint8_t cmd, uint32_t timeout) "generic ioctl sgio: cmd=0x%x timeout=%u" + scsi_generic_ioctl_sgio_done(uint8_t cmd, int ret, uint8_t status, uint8_t host_status) "generic ioctl sgio: cmd=0x%x ret=%d status=0x%x host_status=0x%x" ++scsi_generic_persistent_reserve_out_reply(uint8_t service_action, uint8_t resv_type, uint64_t old_key, uint64_t new_key) "persistent reserve out reply service_action=%u resv_type=%u old_key=0x%" PRIx64 " new_key=0x%" PRIx64 +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index 977b7e4d55..7e120fdd6a 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -54,6 +54,13 @@ struct SCSIRequest { + QTAILQ_ENTRY(SCSIRequest) next; + }; + ++/* Per-SCSIDevice Persistent Reservation state */ ++typedef struct { ++ QemuMutex mutex; /* protects all fields (e.g. from multiple IOThreads) */ ++ uint64_t key; /* 0 if no registered key */ ++ uint8_t resv_type; /* 0 if no reservation */ ++} SCSIPRState; ++ + #define TYPE_SCSI_DEVICE "scsi-device" + OBJECT_DECLARE_TYPE(SCSIDevice, SCSIDeviceClass, SCSI_DEVICE) + +@@ -94,6 +101,9 @@ struct SCSIDevice + uint32_t io_timeout; + bool needs_vpd_bl_emulation; + bool hba_supports_iothread; ++ ++ bool migrate_pr; ++ SCSIPRState pr_state; + }; + + extern const VMStateDescription vmstate_scsi_device; +diff --git a/include/scsi/constants.h b/include/scsi/constants.h +index 9b98451912..cb97bdb636 100644 +--- a/include/scsi/constants.h ++++ b/include/scsi/constants.h +@@ -319,4 +319,25 @@ + #define IDENT_DESCR_TGT_DESCR_SIZE 32 + #define XCOPY_BLK2BLK_SEG_DESC_SIZE 28 + ++/* ++ * PERSISTENT RESERVATION IN service action codes ++ */ ++#define PRI_READ_KEYS 0x00 ++#define PRI_READ_RESERVATION 0x01 ++#define PRI_REPORT_CAPABILITIES 0x02 ++#define PRI_READ_FULL_STATUS 0x03 ++ ++/* ++ * PERSISTENT RESERVATION OUT service action codes ++ */ ++#define PRO_REGISTER 0x00 ++#define PRO_RESERVE 0x01 ++#define PRO_RELEASE 0x02 ++#define PRO_CLEAR 0x03 ++#define PRO_PREEMPT 0x04 ++#define PRO_PREEMPT_AND_ABORT 0x05 ++#define PRO_REGISTER_AND_IGNORE_EXISTING_KEY 0x06 ++#define PRO_REGISTER_AND_MOVE 0x07 ++#define PRO_REPLACE_LOST_RESERVATION 0x08 ++ + #endif +-- +2.47.3 + diff --git a/kvm-target-cpu-models-x86-Remove-the-existing-deprecated.patch b/kvm-target-cpu-models-x86-Remove-the-existing-deprecated.patch deleted file mode 100644 index e7e94d8..0000000 --- a/kvm-target-cpu-models-x86-Remove-the-existing-deprecated.patch +++ /dev/null @@ -1,62 +0,0 @@ -From 0d3444e4ba998bbebce282fe1367ef16b635e3ae Mon Sep 17 00:00:00 2001 -From: Ani Sinha -Date: Fri, 14 Jun 2024 13:34:47 +0530 -Subject: [PATCH 04/14] target/cpu-models/x86: Remove the existing deprecated - CPU models on c10s - -RH-Author: Ani Sinha -RH-MergeRequest: 243: target/cpu-models/x86: Remove the existing deprecated CPU models on c10s -RH-Jira: RHEL-28972 -RH-Acked-by: Thomas Huth -RH-Acked-by: Igor Mammedov -RH-Acked-by: MST -RH-Commit: [4/4] ca6905d2f6cae5f120d3acef973cadb1164e0864 (anisinha/centos-qemu-kvm) - -The cpu models that were deprecated in c9s can be removed in c10s. This change -compiled out these cpu models. For x86, 'qemu64' cpu model is still kept as is -as its the default cpu model. - -Signed-off-by: Ani Sinha ---- - target/i386/cpu.c | 4 ++++ - 1 file changed, 4 insertions(+) - -diff --git a/target/i386/cpu.c b/target/i386/cpu.c -index be7b0663cd..c83d585c9b 100644 ---- a/target/i386/cpu.c -+++ b/target/i386/cpu.c -@@ -2215,6 +2215,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - .xlevel = 0x8000000A, - .model_id = "QEMU Virtual CPU version " QEMU_HW_VERSION, - }, -+#if 0 // Deprecated CPU models are removed in RHEL-10 - { - .name = "phenom", - .deprecation_note = RHEL_CPU_DEPRECATION, -@@ -2593,6 +2594,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - .xlevel = 0x80000008, - .model_id = "Intel Core 2 Duo P9xxx (Penryn Class Core 2)", - }, -+#endif // Removal of deprecated CPU models in RHEL-10 - { - .name = "Nehalem", - .level = 11, -@@ -4410,6 +4412,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - .xlevel = 0x80000008, - .model_id = "Intel Xeon Phi Processor (Knights Mill)", - }, -+#if 0 // Deprecated CPU models are removed in RHEL-10 - { - .name = "Opteron_G1", - .deprecation_note = RHEL_CPU_DEPRECATION, -@@ -4480,6 +4483,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - .xlevel = 0x80000008, - .model_id = "AMD Opteron 23xx (Gen 3 Class Opteron)", - }, -+#endif - { - .name = "Opteron_G4", - .level = 0xd, --- -2.39.3 - diff --git a/kvm-target-i386-add-compatibility-property-for-arch_capa.patch b/kvm-target-i386-add-compatibility-property-for-arch_capa.patch new file mode 100644 index 0000000..78aa945 --- /dev/null +++ b/kvm-target-i386-add-compatibility-property-for-arch_capa.patch @@ -0,0 +1,135 @@ +From ae1b11511be8c6ea7ac3b6dc46e106bb19829b0f Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Thu, 9 Oct 2025 16:49:59 +0200 +Subject: [PATCH 3/4] target/i386: add compatibility property for + arch_capabilities + +RH-Author: Paolo Bonzini +RH-MergeRequest: 411: fix x86-64 migration regression in QEMU 10.1 +RH-Jira: RHEL-120253 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/2] 6bfa7c38398623ac7c3392c0ad17312e6f641f4d (bonzini/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-120253 + +Prior to v10.1, if requested by user, arch-capabilities is always on +despite the fact that CPUID advertises it to be off/unvailable. +This causes a migration issue for VMs that are run on a machine +without arch-capabilities and expect this feature to be present +on the destination host with QEMU 10.1. + +Add a compatibility property to restore the legacy behavior for all +machines with version prior to 10.1. + +To preserve the functionality (added by 10.1) of turning off +ARCH_CAPABILITIES where Windows does not like it, use directly +the guest CPU vendor: x86_cpu_get_supported_feature_word is not +KVM-specific and therefore should not necessarily use the host +CPUID. + +Co-authored-by: Hector Cao +Signed-off-by: Hector Cao +Fixes: d3a24134e37 ("target/i386: do not expose ARCH_CAPABILITIES on AMD CPU", 2025-07-17) +Signed-off-by: Paolo Bonzini +(cherry picked from commit e9efa4a77168ac2816bf9471f878252ce6224710) +Signed-off-by: Paolo Bonzini +--- + hw/i386/pc.c | 2 ++ + target/i386/cpu.c | 17 +++++++++++++++++ + target/i386/cpu.h | 6 ++++++ + target/i386/kvm/kvm.c | 6 +----- + 4 files changed, 26 insertions(+), 5 deletions(-) + +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index 439abe8f46..625a89d097 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -84,6 +84,7 @@ + GlobalProperty pc_compat_10_0[] = { + { TYPE_X86_CPU, "x-consistent-cache", "false" }, + { TYPE_X86_CPU, "x-vendor-cpuid-only-v2", "false" }, ++ { TYPE_X86_CPU, "x-arch-cap-always-on", "true" }, + }; + const size_t pc_compat_10_0_len = G_N_ELEMENTS(pc_compat_10_0); + +@@ -299,6 +300,7 @@ GlobalProperty pc_rhel_10_2_compat[] = { + /* pc_rhel_10_2_compat from pc_compat_10_0 */ + { TYPE_X86_CPU, "x-consistent-cache", "false" }, + { TYPE_X86_CPU, "x-vendor-cpuid-only-v2", "false" }, ++ { TYPE_X86_CPU, "x-arch-cap-always-on", "true" }, + }; + const size_t pc_rhel_10_2_compat_len = G_N_ELEMENTS(pc_compat_10_0); + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 9c756a05f2..de288dc5ac 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -7559,6 +7559,20 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w) + #endif + break; + ++ case FEAT_7_0_EDX: ++ /* ++ * Windows does not like ARCH_CAPABILITIES on AMD machines at all. ++ * Do not show the fake ARCH_CAPABILITIES MSR that KVM sets up, ++ * except if needed for migration. ++ * ++ * When arch_cap_always_on is removed, this tweak can move to ++ * kvm_arch_get_supported_cpuid. ++ */ ++ if (cpu && IS_AMD_CPU(&cpu->env) && !cpu->arch_cap_always_on) { ++ unavail = CPUID_7_0_EDX_ARCH_CAPABILITIES; ++ } ++ break; ++ + default: + break; + } +@@ -10024,6 +10038,9 @@ static const Property x86_cpu_properties[] = { + true), + DEFINE_PROP_BOOL("x-l1-cache-per-thread", X86CPU, l1_cache_per_core, true), + DEFINE_PROP_BOOL("x-force-cpuid-0x1f", X86CPU, force_cpuid_0x1f, false), ++ ++ DEFINE_PROP_BOOL("x-arch-cap-always-on", X86CPU, ++ arch_cap_always_on, false), + }; + + #ifndef CONFIG_USER_ONLY +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index f977fc49a7..b966bc997c 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2314,6 +2314,12 @@ struct ArchCPU { + /* Forcefully disable KVM PV features not exposed in guest CPUIDs */ + bool kvm_pv_enforce_cpuid; + ++ /* ++ * Expose arch-capabilities unconditionally even on AMD models, for backwards ++ * compatibility with QEMU <10.1. ++ */ ++ bool arch_cap_always_on; ++ + /* Number of physical address bits supported */ + uint32_t phys_bits; + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 0eb39d22d6..baa2f80beb 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -503,12 +503,8 @@ uint32_t kvm_arch_get_supported_cpuid(KVMState *s, uint32_t function, + * Linux v4.17-v4.20 incorrectly return ARCH_CAPABILITIES on SVM hosts. + * We can detect the bug by checking if MSR_IA32_ARCH_CAPABILITIES is + * returned by KVM_GET_MSR_INDEX_LIST. +- * +- * But also, because Windows does not like ARCH_CAPABILITIES on AMD +- * mcahines at all, do not show the fake ARCH_CAPABILITIES MSR that +- * KVM sets up. + */ +- if (!has_msr_arch_capabs || !(edx & CPUID_7_0_EDX_ARCH_CAPABILITIES)) { ++ if (!has_msr_arch_capabs) { + ret &= ~CPUID_7_0_EDX_ARCH_CAPABILITIES; + } + } else if (function == 7 && index == 1 && reg == R_EAX) { +-- +2.47.3 + diff --git a/kvm-target-i386-add-compatibility-property-for-pdcm-feat.patch b/kvm-target-i386-add-compatibility-property-for-pdcm-feat.patch new file mode 100644 index 0000000..dfd644c --- /dev/null +++ b/kvm-target-i386-add-compatibility-property-for-pdcm-feat.patch @@ -0,0 +1,115 @@ +From 0d2ec98960c89003d3040818b5c5493cd636b98d Mon Sep 17 00:00:00 2001 +From: Hector Cao +Date: Tue, 23 Sep 2025 12:16:41 +0200 +Subject: [PATCH 4/4] target/i386: add compatibility property for pdcm feature + +RH-Author: Paolo Bonzini +RH-MergeRequest: 411: fix x86-64 migration regression in QEMU 10.1 +RH-Jira: RHEL-120253 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/2] 9d76b45215d7ae2bc5ea2e61fe01780059d6d44c (bonzini/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-120253 + +The pdcm feature is supposed to be disabled when PMU is not +available. Up until v10.1, pdcm feature is enabled even when PMU +is off. This behavior has been fixed but this change breaks the +migration of VMs that are run with QEMU < 10.0 and expect the pdcm +feature to be enabled on the destination host. + +This commit restores the legacy behavior for machines with version +prior to 10.1 to allow the migration from older QEMU to QEMU 10.1. + +Signed-off-by: Hector Cao +Link: https://lore.kernel.org/r/20250910115733.21149-3-hector.cao@canonical.com +Fixes: e68ec298090 ("i386/cpu: Move adjustment of CPUID_EXT_PDCM before feature_dependencies[] check", 2025-06-20) +[Move property from migration object to CPU. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 6529f31e0dccadb532c80b36e3efe7aef83f9cad) +Signed-off-by: Paolo Bonzini +--- + hw/i386/pc.c | 2 ++ + target/i386/cpu.c | 15 ++++++++++++--- + target/i386/cpu.h | 6 ++++++ + 3 files changed, 20 insertions(+), 3 deletions(-) + +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index 625a89d097..446d4a7c93 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -85,6 +85,7 @@ GlobalProperty pc_compat_10_0[] = { + { TYPE_X86_CPU, "x-consistent-cache", "false" }, + { TYPE_X86_CPU, "x-vendor-cpuid-only-v2", "false" }, + { TYPE_X86_CPU, "x-arch-cap-always-on", "true" }, ++ { TYPE_X86_CPU, "x-pdcm-on-even-without-pmu", "true" }, + }; + const size_t pc_compat_10_0_len = G_N_ELEMENTS(pc_compat_10_0); + +@@ -301,6 +302,7 @@ GlobalProperty pc_rhel_10_2_compat[] = { + { TYPE_X86_CPU, "x-consistent-cache", "false" }, + { TYPE_X86_CPU, "x-vendor-cpuid-only-v2", "false" }, + { TYPE_X86_CPU, "x-arch-cap-always-on", "true" }, ++ { TYPE_X86_CPU, "x-pdcm-on-even-without-pmu", "true" }, + }; + const size_t pc_rhel_10_2_compat_len = G_N_ELEMENTS(pc_compat_10_0); + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index de288dc5ac..dfcdfd3da6 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -7928,6 +7928,11 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + /* Fixup overflow: max value for bits 23-16 is 255. */ + *ebx |= MIN(num, 255) << 16; + } ++ if (cpu->pdcm_on_even_without_pmu) { ++ if (!cpu->enable_pmu) { ++ *ecx &= ~CPUID_EXT_PDCM; ++ } ++ } + break; + case 2: { /* cache info: needed for Pentium Pro compatibility */ + const CPUCaches *caches; +@@ -8978,9 +8983,11 @@ void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + } + } + +- /* PDCM is fixed1 bit for TDX */ +- if (!cpu->enable_pmu && !is_tdx_vm()) { +- env->features[FEAT_1_ECX] &= ~CPUID_EXT_PDCM; ++ if (!cpu->pdcm_on_even_without_pmu) { ++ /* PDCM is fixed1 bit for TDX */ ++ if (!cpu->enable_pmu && !is_tdx_vm()) { ++ env->features[FEAT_1_ECX] &= ~CPUID_EXT_PDCM; ++ } + } + + for (i = 0; i < ARRAY_SIZE(feature_dependencies); i++) { +@@ -10041,6 +10048,8 @@ static const Property x86_cpu_properties[] = { + + DEFINE_PROP_BOOL("x-arch-cap-always-on", X86CPU, + arch_cap_always_on, false), ++ DEFINE_PROP_BOOL("x-pdcm-on-even-without-pmu", X86CPU, ++ pdcm_on_even_without_pmu, false), + }; + + #ifndef CONFIG_USER_ONLY +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index b966bc997c..2187e61654 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2320,6 +2320,12 @@ struct ArchCPU { + */ + bool arch_cap_always_on; + ++ /* ++ * Backwards compatibility with QEMU <10.1. The PDCM feature is now disabled when ++ * PMU is not available, but prior to 10.1 it was enabled even if PMU is off. ++ */ ++ bool pdcm_on_even_without_pmu; ++ + /* Number of physical address bits supported */ + uint32_t phys_bits; + +-- +2.47.3 + diff --git a/kvm-target-i386-emulate-Allow-instruction-decoding-from-.patch b/kvm-target-i386-emulate-Allow-instruction-decoding-from-.patch new file mode 100644 index 0000000..068fd00 --- /dev/null +++ b/kvm-target-i386-emulate-Allow-instruction-decoding-from-.patch @@ -0,0 +1,150 @@ +From 35aea0fcc61b1796268f3eae3549721ecd0165fc Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:22 +0200 +Subject: [PATCH 04/32] target/i386/emulate: Allow instruction decoding from + stream + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/30] 36aa0cc1832e8162cb7bba8001b2362245cdb80b + +Introduce a new helper function to decode x86 instructions from a +raw instruction byte stream. MSHV delivers an instruction stream in a +buffer of the vm_exit message. It can be used to speed up MMIO +emulation, since instructions do not have to be fetched and translated. + +Added "fetch_instruction()" op to x86_emul_ops() to improve +traceability. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-3-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 1e25327b244a217078e3fa5df18322c70932f478) +Signed-off-by: Igor Mammedov +--- + target/i386/emulate/x86_decode.c | 27 +++++++++++++++++++++++---- + target/i386/emulate/x86_decode.h | 9 +++++++++ + target/i386/emulate/x86_emu.c | 3 ++- + target/i386/emulate/x86_emu.h | 2 ++ + 4 files changed, 36 insertions(+), 5 deletions(-) + +diff --git a/target/i386/emulate/x86_decode.c b/target/i386/emulate/x86_decode.c +index 2eca39802e..97bd6f1a3b 100644 +--- a/target/i386/emulate/x86_decode.c ++++ b/target/i386/emulate/x86_decode.c +@@ -71,10 +71,16 @@ static inline uint64_t decode_bytes(CPUX86State *env, struct x86_decode *decode, + VM_PANIC_EX("%s invalid size %d\n", __func__, size); + break; + } +- target_ulong va = linear_rip(env_cpu(env), env->eip) + decode->len; +- emul_ops->read_mem(env_cpu(env), &val, va, size); ++ ++ /* copy the bytes from the instruction stream, if available */ ++ if (decode->stream && decode->len + size <= decode->stream->len) { ++ memcpy(&val, decode->stream->bytes + decode->len, size); ++ } else { ++ target_ulong va = linear_rip(env_cpu(env), env->eip) + decode->len; ++ emul_ops->fetch_instruction(env_cpu(env), &val, va, size); ++ } + decode->len += size; +- ++ + return val; + } + +@@ -2076,9 +2082,10 @@ static void decode_opcodes(CPUX86State *env, struct x86_decode *decode) + } + } + +-uint32_t decode_instruction(CPUX86State *env, struct x86_decode *decode) ++static uint32_t decode_opcode(CPUX86State *env, struct x86_decode *decode) + { + memset(decode, 0, sizeof(*decode)); ++ + decode_prefix(env, decode); + set_addressing_size(env, decode); + set_operand_size(env, decode); +@@ -2088,6 +2095,18 @@ uint32_t decode_instruction(CPUX86State *env, struct x86_decode *decode) + return decode->len; + } + ++uint32_t decode_instruction(CPUX86State *env, struct x86_decode *decode) ++{ ++ return decode_opcode(env, decode); ++} ++ ++uint32_t decode_instruction_stream(CPUX86State *env, struct x86_decode *decode, ++ struct x86_insn_stream *stream) ++{ ++ decode->stream = stream; ++ return decode_opcode(env, decode); ++} ++ + void init_decoder(void) + { + int i; +diff --git a/target/i386/emulate/x86_decode.h b/target/i386/emulate/x86_decode.h +index 927645af1a..1cadf3694f 100644 +--- a/target/i386/emulate/x86_decode.h ++++ b/target/i386/emulate/x86_decode.h +@@ -272,6 +272,11 @@ typedef struct x86_decode_op { + }; + } x86_decode_op; + ++typedef struct x86_insn_stream { ++ const uint8_t *bytes; ++ size_t len; ++} x86_insn_stream; ++ + typedef struct x86_decode { + int len; + uint8_t opcode[4]; +@@ -298,11 +303,15 @@ typedef struct x86_decode { + struct x86_modrm modrm; + struct x86_decode_op op[4]; + bool is_fpu; ++ ++ x86_insn_stream *stream; + } x86_decode; + + uint64_t sign(uint64_t val, int size); + + uint32_t decode_instruction(CPUX86State *env, struct x86_decode *decode); ++uint32_t decode_instruction_stream(CPUX86State *env, struct x86_decode *decode, ++ struct x86_insn_stream *stream); + + void *get_reg_ref(CPUX86State *env, int reg, int rex_present, + int is_extended, int size); +diff --git a/target/i386/emulate/x86_emu.c b/target/i386/emulate/x86_emu.c +index db7a7f7437..4409f7bc13 100644 +--- a/target/i386/emulate/x86_emu.c ++++ b/target/i386/emulate/x86_emu.c +@@ -1246,7 +1246,8 @@ static void init_cmd_handler(void) + bool exec_instruction(CPUX86State *env, struct x86_decode *ins) + { + if (!_cmd_handler[ins->cmd].handler) { +- printf("Unimplemented handler (" TARGET_FMT_lx ") for %d (%x %x) \n", env->eip, ++ printf("Unimplemented handler (" TARGET_FMT_lx ") for %d (%x %x)\n", ++ env->eip, + ins->cmd, ins->opcode[0], + ins->opcode_len > 1 ? ins->opcode[1] : 0); + env->eip += ins->len; +diff --git a/target/i386/emulate/x86_emu.h b/target/i386/emulate/x86_emu.h +index a1a961284b..05686b162f 100644 +--- a/target/i386/emulate/x86_emu.h ++++ b/target/i386/emulate/x86_emu.h +@@ -24,6 +24,8 @@ + #include "cpu.h" + + struct x86_emul_ops { ++ void (*fetch_instruction)(CPUState *cpu, void *data, target_ulong addr, ++ int bytes); + void (*read_mem)(CPUState *cpu, void *data, target_ulong addr, int bytes); + void (*write_mem)(CPUState *cpu, void *data, target_ulong addr, int bytes); + void (*read_segment_descriptor)(CPUState *cpu, struct x86_segment_descriptor *desc, +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Add-CPU-create-and-remove-logic.patch b/kvm-target-i386-mshv-Add-CPU-create-and-remove-logic.patch new file mode 100644 index 0000000..a5f72f5 --- /dev/null +++ b/kvm-target-i386-mshv-Add-CPU-create-and-remove-logic.patch @@ -0,0 +1,76 @@ +From 890cb9e42509e42471bc48dc6ddf0283a0b5e8f5 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:32 +0200 +Subject: [PATCH 15/32] target/i386/mshv: Add CPU create and remove logic + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [13/30] cdb6e3cec7a38a6328be8facbbdf19fd0fd9ec78 + +Implement MSHV-specific hooks for vCPU creation and teardown in the +i386 target. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-13-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 4ed605c06e743b2ec0bfe35954d88b8c64dd3767) +Signed-off-by: Igor Mammedov +--- + target/i386/mshv/mshv-cpu.c | 23 +++++++++++++++++------ + 1 file changed, 17 insertions(+), 6 deletions(-) + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 02d71ebc14..5069ab7a22 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -30,6 +30,8 @@ + #include "trace-accel_mshv.h" + #include "trace.h" + ++#include ++ + int mshv_store_regs(CPUState *cpu) + { + error_report("unimplemented"); +@@ -62,20 +64,29 @@ int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit) + + void mshv_remove_vcpu(int vm_fd, int cpu_fd) + { +- error_report("unimplemented"); +- abort(); ++ close(cpu_fd); + } + ++ + int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd) + { +- error_report("unimplemented"); +- abort(); ++ int ret; ++ struct mshv_create_vp vp_arg = { ++ .vp_index = vp_index, ++ }; ++ ret = ioctl(vm_fd, MSHV_CREATE_VP, &vp_arg); ++ if (ret < 0) { ++ error_report("failed to create mshv vcpu: %s", strerror(errno)); ++ return -1; ++ } ++ ++ *cpu_fd = ret; ++ ++ return 0; + } + + void mshv_init_mmio_emu(void) + { +- error_report("unimplemented"); +- abort(); + } + + void mshv_arch_init_vcpu(CPUState *cpu) +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Add-x86-decoder-emu-implementation.patch b/kvm-target-i386-mshv-Add-x86-decoder-emu-implementation.patch new file mode 100644 index 0000000..d64687d --- /dev/null +++ b/kvm-target-i386-mshv-Add-x86-decoder-emu-implementation.patch @@ -0,0 +1,431 @@ +From 84edc1855313a6a06517794e1b2e65e75691725e Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:23 +0200 +Subject: [PATCH 05/32] target/i386/mshv: Add x86 decoder/emu implementation + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/30] 8fa2e2f50c18532d98ef3917f624bc3e65ad167d + +The MSHV accelerator requires a x86 decoder/emulator in userland to +emulate MMIO instructions. This change contains the implementations for +the generalized i386 instruction decoder/emulator. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-4-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 0daf817c80b57e58168309420abf0a8a3d2a60f6) +Signed-off-by: Igor Mammedov +--- + include/system/mshv.h | 25 +++ + target/i386/cpu.h | 2 +- + target/i386/emulate/meson.build | 7 +- + target/i386/meson.build | 2 + + target/i386/mshv/meson.build | 7 + + target/i386/mshv/x86.c | 297 ++++++++++++++++++++++++++++++++ + 6 files changed, 337 insertions(+), 3 deletions(-) + create mode 100644 include/system/mshv.h + create mode 100644 target/i386/mshv/meson.build + create mode 100644 target/i386/mshv/x86.c + +diff --git a/include/system/mshv.h b/include/system/mshv.h +new file mode 100644 +index 0000000000..342f1ef6a9 +--- /dev/null ++++ b/include/system/mshv.h +@@ -0,0 +1,25 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Ziqiao Zhou ++ * Magnus Kulke ++ * Jinank Jain ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ * ++ */ ++ ++#ifndef QEMU_MSHV_H ++#define QEMU_MSHV_H ++ ++#ifdef COMPILING_PER_TARGET ++#ifdef CONFIG_MSHV ++#define CONFIG_MSHV_IS_POSSIBLE ++#endif ++#else ++#define CONFIG_MSHV_IS_POSSIBLE ++#endif ++ ++#endif +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 2187e61654..4b7eae43e1 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2126,7 +2126,7 @@ typedef struct CPUArchState { + QEMUTimer *xen_periodic_timer; + QemuMutex xen_timers_lock; + #endif +-#if defined(CONFIG_HVF) ++#if defined(CONFIG_HVF) || defined(CONFIG_MSHV) + void *emu_mmio_buf; + #endif + +diff --git a/target/i386/emulate/meson.build b/target/i386/emulate/meson.build +index 4edd4f462f..b6dafb6a5b 100644 +--- a/target/i386/emulate/meson.build ++++ b/target/i386/emulate/meson.build +@@ -1,5 +1,8 @@ +-i386_system_ss.add(when: [hvf, 'CONFIG_HVF'], if_true: files( ++emulator_files = files( + 'x86_decode.c', + 'x86_emu.c', + 'x86_flags.c', +-)) ++) ++ ++i386_system_ss.add(when: [hvf, 'CONFIG_HVF'], if_true: emulator_files) ++i386_system_ss.add(when: 'CONFIG_MSHV', if_true: emulator_files) +diff --git a/target/i386/meson.build b/target/i386/meson.build +index 092af34e2d..89ba4912aa 100644 +--- a/target/i386/meson.build ++++ b/target/i386/meson.build +@@ -13,6 +13,7 @@ i386_ss.add(when: 'CONFIG_KVM', if_true: files('host-cpu.c')) + i386_ss.add(when: 'CONFIG_HVF', if_true: files('host-cpu.c')) + i386_ss.add(when: 'CONFIG_WHPX', if_true: files('host-cpu.c')) + i386_ss.add(when: 'CONFIG_NVMM', if_true: files('host-cpu.c')) ++i386_ss.add(when: 'CONFIG_MSHV', if_true: files('host-cpu.c')) + + i386_system_ss = ss.source_set() + i386_system_ss.add(files( +@@ -34,6 +35,7 @@ subdir('nvmm') + subdir('hvf') + subdir('tcg') + subdir('emulate') ++subdir('mshv') + + target_arch += {'i386': i386_ss} + target_system_arch += {'i386': i386_system_ss} +diff --git a/target/i386/mshv/meson.build b/target/i386/mshv/meson.build +new file mode 100644 +index 0000000000..8ddaa7c11d +--- /dev/null ++++ b/target/i386/mshv/meson.build +@@ -0,0 +1,7 @@ ++i386_mshv_ss = ss.source_set() ++ ++i386_mshv_ss.add(files( ++ 'x86.c', ++)) ++ ++i386_system_ss.add_all(when: 'CONFIG_MSHV', if_true: i386_mshv_ss) +diff --git a/target/i386/mshv/x86.c b/target/i386/mshv/x86.c +new file mode 100644 +index 0000000000..d574b3bc52 +--- /dev/null ++++ b/target/i386/mshv/x86.c +@@ -0,0 +1,297 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Magnus Kulke ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++ ++#include "cpu.h" ++#include "emulate/x86_decode.h" ++#include "emulate/x86_emu.h" ++#include "qemu/typedefs.h" ++#include "qemu/error-report.h" ++#include "system/mshv.h" ++ ++/* RW or Exec segment */ ++static const uint8_t RWRX_SEGMENT_TYPE = 0x2; ++static const uint8_t CODE_SEGMENT_TYPE = 0x8; ++static const uint8_t EXPAND_DOWN_SEGMENT_TYPE = 0x4; ++ ++typedef enum CpuMode { ++ REAL_MODE, ++ PROTECTED_MODE, ++ LONG_MODE, ++} CpuMode; ++ ++static CpuMode cpu_mode(CPUState *cpu) ++{ ++ enum CpuMode m = REAL_MODE; ++ ++ if (x86_is_protected(cpu)) { ++ m = PROTECTED_MODE; ++ ++ if (x86_is_long_mode(cpu)) { ++ m = LONG_MODE; ++ } ++ } ++ ++ return m; ++} ++ ++static bool segment_type_ro(const SegmentCache *seg) ++{ ++ uint32_t type_ = (seg->flags >> DESC_TYPE_SHIFT) & 15; ++ return (type_ & (~RWRX_SEGMENT_TYPE)) == 0; ++} ++ ++static bool segment_type_code(const SegmentCache *seg) ++{ ++ uint32_t type_ = (seg->flags >> DESC_TYPE_SHIFT) & 15; ++ return (type_ & CODE_SEGMENT_TYPE) != 0; ++} ++ ++static bool segment_expands_down(const SegmentCache *seg) ++{ ++ uint32_t type_ = (seg->flags >> DESC_TYPE_SHIFT) & 15; ++ ++ if (segment_type_code(seg)) { ++ return false; ++ } ++ ++ return (type_ & EXPAND_DOWN_SEGMENT_TYPE) != 0; ++} ++ ++static uint32_t segment_limit(const SegmentCache *seg) ++{ ++ uint32_t limit = seg->limit; ++ uint32_t granularity = (seg->flags & DESC_G_MASK) != 0; ++ ++ if (granularity != 0) { ++ limit = (limit << 12) | 0xFFF; ++ } ++ ++ return limit; ++} ++ ++static uint8_t segment_db(const SegmentCache *seg) ++{ ++ return (seg->flags >> DESC_B_SHIFT) & 1; ++} ++ ++static uint32_t segment_max_limit(const SegmentCache *seg) ++{ ++ if (segment_db(seg) != 0) { ++ return 0xFFFFFFFF; ++ } ++ return 0xFFFF; ++} ++ ++static int linearize(CPUState *cpu, ++ target_ulong logical_addr, target_ulong *linear_addr, ++ X86Seg seg_idx) ++{ ++ enum CpuMode mode; ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ SegmentCache *seg = &env->segs[seg_idx]; ++ target_ulong base = seg->base; ++ target_ulong logical_addr_32b; ++ uint32_t limit; ++ /* TODO: the emulator will not pass us "write" indicator yet */ ++ bool write = false; ++ ++ mode = cpu_mode(cpu); ++ ++ switch (mode) { ++ case LONG_MODE: ++ if (__builtin_add_overflow(logical_addr, base, linear_addr)) { ++ error_report("Address overflow"); ++ return -1; ++ } ++ break; ++ case PROTECTED_MODE: ++ case REAL_MODE: ++ if (segment_type_ro(seg) && write) { ++ error_report("Cannot write to read-only segment"); ++ return -1; ++ } ++ ++ logical_addr_32b = logical_addr & 0xFFFFFFFF; ++ limit = segment_limit(seg); ++ ++ if (segment_expands_down(seg)) { ++ if (logical_addr_32b >= limit) { ++ error_report("Address exceeds limit (expands down)"); ++ return -1; ++ } ++ ++ limit = segment_max_limit(seg); ++ } ++ ++ if (logical_addr_32b > limit) { ++ error_report("Address exceeds limit %u", limit); ++ return -1; ++ } ++ *linear_addr = logical_addr_32b + base; ++ break; ++ default: ++ error_report("Unknown cpu mode: %d", mode); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++bool x86_read_segment_descriptor(CPUState *cpu, ++ struct x86_segment_descriptor *desc, ++ x86_segment_selector sel) ++{ ++ target_ulong base; ++ uint32_t limit; ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ target_ulong gva; ++ ++ memset(desc, 0, sizeof(*desc)); ++ ++ /* valid gdt descriptors start from index 1 */ ++ if (!sel.index && GDT_SEL == sel.ti) { ++ return false; ++ } ++ ++ if (GDT_SEL == sel.ti) { ++ base = env->gdt.base; ++ limit = env->gdt.limit; ++ } else { ++ base = env->ldt.base; ++ limit = env->ldt.limit; ++ } ++ ++ if (sel.index * 8 >= limit) { ++ return false; ++ } ++ ++ gva = base + sel.index * 8; ++ emul_ops->read_mem(cpu, desc, gva, sizeof(*desc)); ++ ++ return true; ++} ++ ++bool x86_read_call_gate(CPUState *cpu, struct x86_call_gate *idt_desc, ++ int gate) ++{ ++ target_ulong base; ++ uint32_t limit; ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ target_ulong gva; ++ ++ base = env->idt.base; ++ limit = env->idt.limit; ++ ++ memset(idt_desc, 0, sizeof(*idt_desc)); ++ if (gate * 8 >= limit) { ++ perror("call gate exceeds idt limit"); ++ return false; ++ } ++ ++ gva = base + gate * 8; ++ emul_ops->read_mem(cpu, idt_desc, gva, sizeof(*idt_desc)); ++ ++ return true; ++} ++ ++bool x86_is_protected(CPUState *cpu) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ uint64_t cr0 = env->cr[0]; ++ ++ return cr0 & CR0_PE_MASK; ++} ++ ++bool x86_is_real(CPUState *cpu) ++{ ++ return !x86_is_protected(cpu); ++} ++ ++bool x86_is_v8086(CPUState *cpu) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ return x86_is_protected(cpu) && (env->eflags & VM_MASK); ++} ++ ++bool x86_is_long_mode(CPUState *cpu) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ uint64_t efer = env->efer; ++ uint64_t lme_lma = (MSR_EFER_LME | MSR_EFER_LMA); ++ ++ return ((efer & lme_lma) == lme_lma); ++} ++ ++bool x86_is_long64_mode(CPUState *cpu) ++{ ++ error_report("unimplemented: is_long64_mode()"); ++ abort(); ++} ++ ++bool x86_is_paging_mode(CPUState *cpu) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ uint64_t cr0 = env->cr[0]; ++ ++ return cr0 & CR0_PG_MASK; ++} ++ ++bool x86_is_pae_enabled(CPUState *cpu) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ uint64_t cr4 = env->cr[4]; ++ ++ return cr4 & CR4_PAE_MASK; ++} ++ ++target_ulong linear_addr(CPUState *cpu, target_ulong addr, X86Seg seg) ++{ ++ int ret; ++ target_ulong linear_addr; ++ ++ ret = linearize(cpu, addr, &linear_addr, seg); ++ if (ret < 0) { ++ error_report("failed to linearize address"); ++ abort(); ++ } ++ ++ return linear_addr; ++} ++ ++target_ulong linear_addr_size(CPUState *cpu, target_ulong addr, int size, ++ X86Seg seg) ++{ ++ switch (size) { ++ case 2: ++ addr = (uint16_t)addr; ++ break; ++ case 4: ++ addr = (uint32_t)addr; ++ break; ++ default: ++ break; ++ } ++ return linear_addr(cpu, addr, seg); ++} ++ ++target_ulong linear_rip(CPUState *cpu, target_ulong rip) ++{ ++ return linear_addr(cpu, rip, R_CS); ++} +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Implement-mshv_arch_put_registers.patch b/kvm-target-i386-mshv-Implement-mshv_arch_put_registers.patch new file mode 100644 index 0000000..b73fa87 --- /dev/null +++ b/kvm-target-i386-mshv-Implement-mshv_arch_put_registers.patch @@ -0,0 +1,319 @@ +From 78ed99e6783511368371defaebda4cc450a6e266 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:36 +0200 +Subject: [PATCH 19/32] target/i386/mshv: Implement mshv_arch_put_registers() + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [17/30] e2648880fed0b7eb9097768c3e4cfc914baa104a + +Write CPU register state to MSHV vCPUs. Various mapping functions to +prepare the payload for the HV call have been implemented. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-17-magnuskulke@linux.microsoft.com +[mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 25a1d871e0f19bbf8fee471af5fefec52465bb09) +Signed-off-by: Igor Mammedov +--- + include/system/mshv_int.h | 15 +++ + target/i386/mshv/mshv-cpu.c | 237 ++++++++++++++++++++++++++++++++++++ + 2 files changed, 252 insertions(+) + +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index c6e6e8af30..0ea8d504fa 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -49,6 +49,20 @@ typedef struct MshvMsiControl { + #define mshv_vcpufd(cpu) (cpu->accel->cpufd) + + /* cpu */ ++typedef struct MshvFPU { ++ uint8_t fpr[8][16]; ++ uint16_t fcw; ++ uint16_t fsw; ++ uint8_t ftwx; ++ uint8_t pad1; ++ uint16_t last_opcode; ++ uint64_t last_ip; ++ uint64_t last_dp; ++ uint8_t xmm[16][16]; ++ uint32_t mxcsr; ++ uint32_t pad2; ++} MshvFPU; ++ + typedef enum MshvVmExit { + MshvVmExitIgnore = 0, + MshvVmExitShutdown = 1, +@@ -58,6 +72,7 @@ typedef enum MshvVmExit { + void mshv_init_mmio_emu(void); + int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd); + void mshv_remove_vcpu(int vm_fd, int cpu_fd); ++int mshv_configure_vcpu(const CPUState *cpu, const MshvFPU *fpu, uint64_t xcr0); + int mshv_get_standard_regs(CPUState *cpu); + int mshv_get_special_regs(CPUState *cpu); + int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit); +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index bc75686f82..8b10c79e54 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -73,6 +73,35 @@ static enum hv_register_name SPECIAL_REGISTER_NAMES[17] = { + HV_X64_REGISTER_APIC_BASE, + }; + ++static enum hv_register_name FPU_REGISTER_NAMES[26] = { ++ HV_X64_REGISTER_XMM0, ++ HV_X64_REGISTER_XMM1, ++ HV_X64_REGISTER_XMM2, ++ HV_X64_REGISTER_XMM3, ++ HV_X64_REGISTER_XMM4, ++ HV_X64_REGISTER_XMM5, ++ HV_X64_REGISTER_XMM6, ++ HV_X64_REGISTER_XMM7, ++ HV_X64_REGISTER_XMM8, ++ HV_X64_REGISTER_XMM9, ++ HV_X64_REGISTER_XMM10, ++ HV_X64_REGISTER_XMM11, ++ HV_X64_REGISTER_XMM12, ++ HV_X64_REGISTER_XMM13, ++ HV_X64_REGISTER_XMM14, ++ HV_X64_REGISTER_XMM15, ++ HV_X64_REGISTER_FP_MMX0, ++ HV_X64_REGISTER_FP_MMX1, ++ HV_X64_REGISTER_FP_MMX2, ++ HV_X64_REGISTER_FP_MMX3, ++ HV_X64_REGISTER_FP_MMX4, ++ HV_X64_REGISTER_FP_MMX5, ++ HV_X64_REGISTER_FP_MMX6, ++ HV_X64_REGISTER_FP_MMX7, ++ HV_X64_REGISTER_FP_CONTROL_STATUS, ++ HV_X64_REGISTER_XMM_CONTROL_STATUS, ++}; ++ + int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, + size_t n_regs) + { +@@ -372,8 +401,216 @@ int mshv_load_regs(CPUState *cpu) + return 0; + } + ++static inline void populate_hv_segment_reg(SegmentCache *seg, ++ hv_x64_segment_register *hv_reg) ++{ ++ uint32_t flags = seg->flags; ++ ++ hv_reg->base = seg->base; ++ hv_reg->limit = seg->limit; ++ hv_reg->selector = seg->selector; ++ hv_reg->segment_type = (flags >> DESC_TYPE_SHIFT) & 0xF; ++ hv_reg->non_system_segment = (flags & DESC_S_MASK) != 0; ++ hv_reg->descriptor_privilege_level = (flags >> DESC_DPL_SHIFT) & 0x3; ++ hv_reg->present = (flags & DESC_P_MASK) != 0; ++ hv_reg->reserved = 0; ++ hv_reg->available = (flags & DESC_AVL_MASK) != 0; ++ hv_reg->_long = (flags >> DESC_L_SHIFT) & 0x1; ++ hv_reg->_default = (flags >> DESC_B_SHIFT) & 0x1; ++ hv_reg->granularity = (flags & DESC_G_MASK) != 0; ++} ++ ++static inline void populate_hv_table_reg(const struct SegmentCache *seg, ++ hv_x64_table_register *hv_reg) ++{ ++ memset(hv_reg, 0, sizeof(*hv_reg)); ++ ++ hv_reg->base = seg->base; ++ hv_reg->limit = seg->limit; ++} ++ ++static int set_special_regs(const CPUState *cpu) ++{ ++ X86CPU *x86cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86cpu->env; ++ struct hv_register_assoc assocs[ARRAY_SIZE(SPECIAL_REGISTER_NAMES)]; ++ size_t n_regs = ARRAY_SIZE(SPECIAL_REGISTER_NAMES); ++ int ret; ++ ++ /* set names */ ++ for (size_t i = 0; i < n_regs; i++) { ++ assocs[i].name = SPECIAL_REGISTER_NAMES[i]; ++ } ++ populate_hv_segment_reg(&env->segs[R_CS], &assocs[0].value.segment); ++ populate_hv_segment_reg(&env->segs[R_DS], &assocs[1].value.segment); ++ populate_hv_segment_reg(&env->segs[R_ES], &assocs[2].value.segment); ++ populate_hv_segment_reg(&env->segs[R_FS], &assocs[3].value.segment); ++ populate_hv_segment_reg(&env->segs[R_GS], &assocs[4].value.segment); ++ populate_hv_segment_reg(&env->segs[R_SS], &assocs[5].value.segment); ++ populate_hv_segment_reg(&env->tr, &assocs[6].value.segment); ++ populate_hv_segment_reg(&env->ldt, &assocs[7].value.segment); ++ ++ populate_hv_table_reg(&env->gdt, &assocs[8].value.table); ++ populate_hv_table_reg(&env->idt, &assocs[9].value.table); ++ ++ assocs[10].value.reg64 = env->cr[0]; ++ assocs[11].value.reg64 = env->cr[2]; ++ assocs[12].value.reg64 = env->cr[3]; ++ assocs[13].value.reg64 = env->cr[4]; ++ assocs[14].value.reg64 = cpu_get_apic_tpr(x86cpu->apic_state); ++ assocs[15].value.reg64 = env->efer; ++ assocs[16].value.reg64 = cpu_get_apic_base(x86cpu->apic_state); ++ ++ ret = mshv_set_generic_regs(cpu, assocs, n_regs); ++ if (ret < 0) { ++ error_report("failed to set special registers"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int set_fpu(const CPUState *cpu, const struct MshvFPU *regs) ++{ ++ struct hv_register_assoc assocs[ARRAY_SIZE(FPU_REGISTER_NAMES)]; ++ union hv_register_value *value; ++ size_t fp_i; ++ union hv_x64_fp_control_status_register *ctrl_status; ++ union hv_x64_xmm_control_status_register *xmm_ctrl_status; ++ int ret; ++ size_t n_regs = ARRAY_SIZE(FPU_REGISTER_NAMES); ++ ++ /* first 16 registers are xmm0-xmm15 */ ++ for (size_t i = 0; i < 16; i++) { ++ assocs[i].name = FPU_REGISTER_NAMES[i]; ++ value = &assocs[i].value; ++ memcpy(&value->reg128, ®s->xmm[i], 16); ++ } ++ ++ /* next 8 registers are fp_mmx0-fp_mmx7 */ ++ for (size_t i = 16; i < 24; i++) { ++ assocs[i].name = FPU_REGISTER_NAMES[i]; ++ fp_i = (i - 16); ++ value = &assocs[i].value; ++ memcpy(&value->reg128, ®s->fpr[fp_i], 16); ++ } ++ ++ /* last two registers are fp_control_status and xmm_control_status */ ++ assocs[24].name = FPU_REGISTER_NAMES[24]; ++ value = &assocs[24].value; ++ ctrl_status = &value->fp_control_status; ++ ctrl_status->fp_control = regs->fcw; ++ ctrl_status->fp_status = regs->fsw; ++ ctrl_status->fp_tag = regs->ftwx; ++ ctrl_status->reserved = 0; ++ ctrl_status->last_fp_op = regs->last_opcode; ++ ctrl_status->last_fp_rip = regs->last_ip; ++ ++ assocs[25].name = FPU_REGISTER_NAMES[25]; ++ value = &assocs[25].value; ++ xmm_ctrl_status = &value->xmm_control_status; ++ xmm_ctrl_status->xmm_status_control = regs->mxcsr; ++ xmm_ctrl_status->xmm_status_control_mask = 0; ++ xmm_ctrl_status->last_fp_rdp = regs->last_dp; ++ ++ ret = mshv_set_generic_regs(cpu, assocs, n_regs); ++ if (ret < 0) { ++ error_report("failed to set fpu registers"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int set_xc_reg(const CPUState *cpu, uint64_t xcr0) ++{ ++ int ret; ++ struct hv_register_assoc assoc = { ++ .name = HV_X64_REGISTER_XFEM, ++ .value.reg64 = xcr0, ++ }; ++ ++ ret = mshv_set_generic_regs(cpu, &assoc, 1); ++ if (ret < 0) { ++ error_report("failed to set xcr0"); ++ return -errno; ++ } ++ return 0; ++} ++ ++static int set_cpu_state(const CPUState *cpu, const MshvFPU *fpu_regs, ++ uint64_t xcr0) ++{ ++ int ret; ++ ++ ret = set_standard_regs(cpu); ++ if (ret < 0) { ++ return ret; ++ } ++ ret = set_special_regs(cpu); ++ if (ret < 0) { ++ return ret; ++ } ++ ret = set_fpu(cpu, fpu_regs); ++ if (ret < 0) { ++ return ret; ++ } ++ ret = set_xc_reg(cpu, xcr0); ++ if (ret < 0) { ++ return ret; ++ } ++ return 0; ++} ++ ++/* ++ * TODO: populate topology info: ++ * ++ * X86CPU *x86cpu = X86_CPU(cpu); ++ * CPUX86State *env = &x86cpu->env; ++ * X86CPUTopoInfo *topo_info = &env->topo_info; ++ */ ++int mshv_configure_vcpu(const CPUState *cpu, const struct MshvFPU *fpu, ++ uint64_t xcr0) ++{ ++ int ret; ++ ++ ret = set_cpu_state(cpu, fpu, xcr0); ++ if (ret < 0) { ++ error_report("failed to set cpu state"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int put_regs(const CPUState *cpu) ++{ ++ X86CPU *x86cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86cpu->env; ++ MshvFPU fpu = {0}; ++ int ret; ++ ++ memset(&fpu, 0, sizeof(fpu)); ++ ++ ret = mshv_configure_vcpu(cpu, &fpu, env->xcr0); ++ if (ret < 0) { ++ error_report("failed to configure vcpu"); ++ return ret; ++ } ++ ++ return 0; ++} ++ + int mshv_arch_put_registers(const CPUState *cpu) + { ++ int ret; ++ ++ ret = put_regs(cpu); ++ if (ret < 0) { ++ error_report("Failed to put registers"); ++ return -1; ++ } ++ + error_report("unimplemented"); + abort(); + } +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Implement-mshv_get_special_regs.patch b/kvm-target-i386-mshv-Implement-mshv_get_special_regs.patch new file mode 100644 index 0000000..9cede6e --- /dev/null +++ b/kvm-target-i386-mshv-Implement-mshv_get_special_regs.patch @@ -0,0 +1,173 @@ +From 30862bd55b3d5c95a051638ec49013e64596f46d Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:35 +0200 +Subject: [PATCH 18/32] target/i386/mshv: Implement mshv_get_special_regs() + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [16/30] 429625fae745bc6c3c9e9412741c875331a92180 + +Retrieve special registers (e.g. segment, control, and descriptor +table registers) from MSHV vCPUs. + +Various helper functions to map register state representations between +Qemu and MSHV are introduced. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-16-magnuskulke@linux.microsoft.com +[mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 0382c2c8544da66619d16988caaab8e3d9e881d4) +Signed-off-by: Igor Mammedov +--- + include/system/mshv_int.h | 1 + + target/i386/mshv/mshv-cpu.c | 104 ++++++++++++++++++++++++++++++++++++ + 2 files changed, 105 insertions(+) + +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index b0a79296ad..c6e6e8af30 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -59,6 +59,7 @@ void mshv_init_mmio_emu(void); + int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd); + void mshv_remove_vcpu(int vm_fd, int cpu_fd); + int mshv_get_standard_regs(CPUState *cpu); ++int mshv_get_special_regs(CPUState *cpu); + int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit); + int mshv_load_regs(CPUState *cpu); + int mshv_store_regs(CPUState *cpu); +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 4e3eee113b..bc75686f82 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -53,6 +53,26 @@ static enum hv_register_name STANDARD_REGISTER_NAMES[18] = { + HV_X64_REGISTER_RFLAGS, + }; + ++static enum hv_register_name SPECIAL_REGISTER_NAMES[17] = { ++ HV_X64_REGISTER_CS, ++ HV_X64_REGISTER_DS, ++ HV_X64_REGISTER_ES, ++ HV_X64_REGISTER_FS, ++ HV_X64_REGISTER_GS, ++ HV_X64_REGISTER_SS, ++ HV_X64_REGISTER_TR, ++ HV_X64_REGISTER_LDTR, ++ HV_X64_REGISTER_GDTR, ++ HV_X64_REGISTER_IDTR, ++ HV_X64_REGISTER_CR0, ++ HV_X64_REGISTER_CR2, ++ HV_X64_REGISTER_CR3, ++ HV_X64_REGISTER_CR4, ++ HV_X64_REGISTER_CR8, ++ HV_X64_REGISTER_EFER, ++ HV_X64_REGISTER_APIC_BASE, ++}; ++ + int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, + size_t n_regs) + { +@@ -255,6 +275,84 @@ int mshv_get_standard_regs(CPUState *cpu) + return 0; + } + ++static inline void populate_segment_reg(const hv_x64_segment_register *hv_seg, ++ SegmentCache *seg) ++{ ++ memset(seg, 0, sizeof(SegmentCache)); ++ ++ seg->base = hv_seg->base; ++ seg->limit = hv_seg->limit; ++ seg->selector = hv_seg->selector; ++ ++ seg->flags = (hv_seg->segment_type << DESC_TYPE_SHIFT) ++ | (hv_seg->present * DESC_P_MASK) ++ | (hv_seg->descriptor_privilege_level << DESC_DPL_SHIFT) ++ | (hv_seg->_default << DESC_B_SHIFT) ++ | (hv_seg->non_system_segment * DESC_S_MASK) ++ | (hv_seg->_long << DESC_L_SHIFT) ++ | (hv_seg->granularity * DESC_G_MASK) ++ | (hv_seg->available * DESC_AVL_MASK); ++ ++} ++ ++static inline void populate_table_reg(const hv_x64_table_register *hv_seg, ++ SegmentCache *tbl) ++{ ++ memset(tbl, 0, sizeof(SegmentCache)); ++ ++ tbl->base = hv_seg->base; ++ tbl->limit = hv_seg->limit; ++} ++ ++static void populate_special_regs(const hv_register_assoc *assocs, ++ X86CPU *x86cpu) ++{ ++ CPUX86State *env = &x86cpu->env; ++ ++ populate_segment_reg(&assocs[0].value.segment, &env->segs[R_CS]); ++ populate_segment_reg(&assocs[1].value.segment, &env->segs[R_DS]); ++ populate_segment_reg(&assocs[2].value.segment, &env->segs[R_ES]); ++ populate_segment_reg(&assocs[3].value.segment, &env->segs[R_FS]); ++ populate_segment_reg(&assocs[4].value.segment, &env->segs[R_GS]); ++ populate_segment_reg(&assocs[5].value.segment, &env->segs[R_SS]); ++ ++ populate_segment_reg(&assocs[6].value.segment, &env->tr); ++ populate_segment_reg(&assocs[7].value.segment, &env->ldt); ++ ++ populate_table_reg(&assocs[8].value.table, &env->gdt); ++ populate_table_reg(&assocs[9].value.table, &env->idt); ++ ++ env->cr[0] = assocs[10].value.reg64; ++ env->cr[2] = assocs[11].value.reg64; ++ env->cr[3] = assocs[12].value.reg64; ++ env->cr[4] = assocs[13].value.reg64; ++ ++ cpu_set_apic_tpr(x86cpu->apic_state, assocs[14].value.reg64); ++ env->efer = assocs[15].value.reg64; ++ cpu_set_apic_base(x86cpu->apic_state, assocs[16].value.reg64); ++} ++ ++ ++int mshv_get_special_regs(CPUState *cpu) ++{ ++ struct hv_register_assoc assocs[ARRAY_SIZE(SPECIAL_REGISTER_NAMES)]; ++ int ret; ++ X86CPU *x86cpu = X86_CPU(cpu); ++ size_t n_regs = ARRAY_SIZE(SPECIAL_REGISTER_NAMES); ++ ++ for (size_t i = 0; i < n_regs; i++) { ++ assocs[i].name = SPECIAL_REGISTER_NAMES[i]; ++ } ++ ret = get_generic_regs(cpu, assocs, n_regs); ++ if (ret < 0) { ++ error_report("failed to get special registers"); ++ return -errno; ++ } ++ ++ populate_special_regs(assocs, x86cpu); ++ return 0; ++} ++ + int mshv_load_regs(CPUState *cpu) + { + int ret; +@@ -265,6 +363,12 @@ int mshv_load_regs(CPUState *cpu) + return -1; + } + ++ ret = mshv_get_special_regs(cpu); ++ if (ret < 0) { ++ error_report("Failed to load special registers"); ++ return -1; ++ } ++ + return 0; + } + +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Implement-mshv_get_standard_regs.patch b/kvm-target-i386-mshv-Implement-mshv_get_standard_regs.patch new file mode 100644 index 0000000..e68b281 --- /dev/null +++ b/kvm-target-i386-mshv-Implement-mshv_get_standard_regs.patch @@ -0,0 +1,182 @@ +From 11393f9a7ff5fa2ec3182025fe5ba052d5fecee0 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:34 +0200 +Subject: [PATCH 17/32] target/i386/mshv: Implement mshv_get_standard_regs() + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [15/30] ccb135407183081206513dd2de2410a9c1768f15 + +Fetch standard register state from MSHV vCPUs to support debugging, +migration, and other introspection features in QEMU. + +Fetch standard register state from a MHSV vCPU's. A generic get_regs() +function and a mapper to map the different register representations are +introduced. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-15-magnuskulke@linux.microsoft.com +[mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 66480a048a01fc833ba153410282609f26166141) +Signed-off-by: Igor Mammedov +--- + include/system/mshv_int.h | 1 + + target/i386/mshv/mshv-cpu.c | 116 +++++++++++++++++++++++++++++++++++- + 2 files changed, 115 insertions(+), 2 deletions(-) + +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index 731841af92..b0a79296ad 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -58,6 +58,7 @@ typedef enum MshvVmExit { + void mshv_init_mmio_emu(void); + int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd); + void mshv_remove_vcpu(int vm_fd, int cpu_fd); ++int mshv_get_standard_regs(CPUState *cpu); + int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit); + int mshv_load_regs(CPUState *cpu); + int mshv_store_regs(CPUState *cpu); +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 9ead03ca2d..4e3eee113b 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -96,6 +96,66 @@ int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, + return 0; + } + ++static int get_generic_regs(CPUState *cpu, hv_register_assoc *assocs, ++ size_t n_regs) ++{ ++ int cpu_fd = mshv_vcpufd(cpu); ++ int vp_index = cpu->cpu_index; ++ hv_input_get_vp_registers *in; ++ hv_register_value *values; ++ size_t in_sz, names_sz, values_sz; ++ int i, ret; ++ struct mshv_root_hvcall args = {0}; ++ ++ /* find out the size of the struct w/ a flexible array at the tail */ ++ names_sz = n_regs * sizeof(hv_register_name); ++ in_sz = sizeof(hv_input_get_vp_registers) + names_sz; ++ ++ /* fill the input struct */ ++ in = g_malloc0(in_sz); ++ in->vp_index = vp_index; ++ for (i = 0; i < n_regs; i++) { ++ in->names[i] = assocs[i].name; ++ } ++ ++ /* allocate value output buffer */ ++ values_sz = n_regs * sizeof(union hv_register_value); ++ values = g_malloc0(values_sz); ++ ++ /* create the hvcall envelope */ ++ args.code = HVCALL_GET_VP_REGISTERS; ++ args.in_sz = in_sz; ++ args.in_ptr = (uint64_t) in; ++ args.out_sz = values_sz; ++ args.out_ptr = (uint64_t) values; ++ args.reps = (uint16_t) n_regs; ++ ++ /* perform the call */ ++ ret = mshv_hvcall(cpu_fd, &args); ++ g_free(in); ++ if (ret < 0) { ++ g_free(values); ++ error_report("Failed to retrieve registers"); ++ return -1; ++ } ++ ++ /* assert we got all registers */ ++ if (args.reps != n_regs) { ++ g_free(values); ++ error_report("Failed to retrieve registers: expected %zu elements" ++ ", got %u", n_regs, args.reps); ++ return -1; ++ } ++ ++ /* copy values into assoc */ ++ for (i = 0; i < n_regs; i++) { ++ assocs[i].value = values[i]; ++ } ++ g_free(values); ++ ++ return 0; ++} ++ + static int set_standard_regs(const CPUState *cpu) + { + X86CPU *x86cpu = X86_CPU(cpu); +@@ -149,11 +209,63 @@ int mshv_store_regs(CPUState *cpu) + return 0; + } + ++static void populate_standard_regs(const hv_register_assoc *assocs, ++ CPUX86State *env) ++{ ++ env->regs[R_EAX] = assocs[0].value.reg64; ++ env->regs[R_EBX] = assocs[1].value.reg64; ++ env->regs[R_ECX] = assocs[2].value.reg64; ++ env->regs[R_EDX] = assocs[3].value.reg64; ++ env->regs[R_ESI] = assocs[4].value.reg64; ++ env->regs[R_EDI] = assocs[5].value.reg64; ++ env->regs[R_ESP] = assocs[6].value.reg64; ++ env->regs[R_EBP] = assocs[7].value.reg64; ++ env->regs[R_R8] = assocs[8].value.reg64; ++ env->regs[R_R9] = assocs[9].value.reg64; ++ env->regs[R_R10] = assocs[10].value.reg64; ++ env->regs[R_R11] = assocs[11].value.reg64; ++ env->regs[R_R12] = assocs[12].value.reg64; ++ env->regs[R_R13] = assocs[13].value.reg64; ++ env->regs[R_R14] = assocs[14].value.reg64; ++ env->regs[R_R15] = assocs[15].value.reg64; ++ ++ env->eip = assocs[16].value.reg64; ++ env->eflags = assocs[17].value.reg64; ++ rflags_to_lflags(env); ++} ++ ++int mshv_get_standard_regs(CPUState *cpu) ++{ ++ struct hv_register_assoc assocs[ARRAY_SIZE(STANDARD_REGISTER_NAMES)]; ++ int ret; ++ X86CPU *x86cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86cpu->env; ++ size_t n_regs = ARRAY_SIZE(STANDARD_REGISTER_NAMES); ++ ++ for (size_t i = 0; i < n_regs; i++) { ++ assocs[i].name = STANDARD_REGISTER_NAMES[i]; ++ } ++ ret = get_generic_regs(cpu, assocs, n_regs); ++ if (ret < 0) { ++ error_report("failed to get standard registers"); ++ return -1; ++ } ++ ++ populate_standard_regs(assocs, env); ++ return 0; ++} + + int mshv_load_regs(CPUState *cpu) + { +- error_report("unimplemented"); +- abort(); ++ int ret; ++ ++ ret = mshv_get_standard_regs(cpu); ++ if (ret < 0) { ++ error_report("Failed to load standard registers"); ++ return -1; ++ } ++ ++ return 0; + } + + int mshv_arch_put_registers(const CPUState *cpu) +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Implement-mshv_store_regs.patch b/kvm-target-i386-mshv-Implement-mshv_store_regs.patch new file mode 100644 index 0000000..aa3e8d0 --- /dev/null +++ b/kvm-target-i386-mshv-Implement-mshv_store_regs.patch @@ -0,0 +1,188 @@ +From 00d2e933423d093c34265d5e8056461308a738f3 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Thu, 2 Oct 2025 18:13:31 +0200 +Subject: [PATCH 16/32] target/i386/mshv: Implement mshv_store_regs() + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [14/30] 2f4fe2c0b042ec06d351929a7b090e554a459be1 + +Add support for writing general-purpose registers to MSHV vCPUs +during initialization or migration using the MSHV register interface. A +generic set_register call is introduced to abstract the HV call over +the various register types. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-14-magnuskulke@linux.microsoft.com +[mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 2bd7a6aa4743ee85c77c0520e8b422b3bfeab386) +Signed-off-by: Igor Mammedov +--- + include/system/mshv.h | 1 + + include/system/mshv_int.h | 2 + + target/i386/mshv/mshv-cpu.c | 116 +++++++++++++++++++++++++++++++++++- + 3 files changed, 117 insertions(+), 2 deletions(-) + +diff --git a/include/system/mshv.h b/include/system/mshv.h +index bbc42f4dc3..8b1fc20c80 100644 +--- a/include/system/mshv.h ++++ b/include/system/mshv.h +@@ -18,6 +18,7 @@ + #include "qemu/accel.h" + #include "hw/hyperv/hyperv-proto.h" + #include "hw/hyperv/hvhdk.h" ++#include "hw/hyperv/hvgdk_mini.h" + #include "qapi/qapi-types-common.h" + #include "system/memory.h" + #include "accel/accel-ops.h" +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index fb80f69772..731841af92 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -61,6 +61,8 @@ void mshv_remove_vcpu(int vm_fd, int cpu_fd); + int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit); + int mshv_load_regs(CPUState *cpu); + int mshv_store_regs(CPUState *cpu); ++int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, ++ size_t n_regs); + int mshv_arch_put_registers(const CPUState *cpu); + void mshv_arch_init_vcpu(CPUState *cpu); + void mshv_arch_destroy_vcpu(CPUState *cpu); +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 5069ab7a22..9ead03ca2d 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -32,12 +32,124 @@ + + #include + ++static enum hv_register_name STANDARD_REGISTER_NAMES[18] = { ++ HV_X64_REGISTER_RAX, ++ HV_X64_REGISTER_RBX, ++ HV_X64_REGISTER_RCX, ++ HV_X64_REGISTER_RDX, ++ HV_X64_REGISTER_RSI, ++ HV_X64_REGISTER_RDI, ++ HV_X64_REGISTER_RSP, ++ HV_X64_REGISTER_RBP, ++ HV_X64_REGISTER_R8, ++ HV_X64_REGISTER_R9, ++ HV_X64_REGISTER_R10, ++ HV_X64_REGISTER_R11, ++ HV_X64_REGISTER_R12, ++ HV_X64_REGISTER_R13, ++ HV_X64_REGISTER_R14, ++ HV_X64_REGISTER_R15, ++ HV_X64_REGISTER_RIP, ++ HV_X64_REGISTER_RFLAGS, ++}; ++ ++int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, ++ size_t n_regs) ++{ ++ int cpu_fd = mshv_vcpufd(cpu); ++ int vp_index = cpu->cpu_index; ++ size_t in_sz, assocs_sz; ++ hv_input_set_vp_registers *in; ++ struct mshv_root_hvcall args = {0}; ++ int ret; ++ ++ /* find out the size of the struct w/ a flexible array at the tail */ ++ assocs_sz = n_regs * sizeof(hv_register_assoc); ++ in_sz = sizeof(hv_input_set_vp_registers) + assocs_sz; ++ ++ /* fill the input struct */ ++ in = g_malloc0(in_sz); ++ in->vp_index = vp_index; ++ memcpy(in->elements, assocs, assocs_sz); ++ ++ /* create the hvcall envelope */ ++ args.code = HVCALL_SET_VP_REGISTERS; ++ args.in_sz = in_sz; ++ args.in_ptr = (uint64_t) in; ++ args.reps = (uint16_t) n_regs; ++ ++ /* perform the call */ ++ ret = mshv_hvcall(cpu_fd, &args); ++ g_free(in); ++ if (ret < 0) { ++ error_report("Failed to set registers"); ++ return -1; ++ } ++ ++ /* assert we set all registers */ ++ if (args.reps != n_regs) { ++ error_report("Failed to set registers: expected %zu elements" ++ ", got %u", n_regs, args.reps); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int set_standard_regs(const CPUState *cpu) ++{ ++ X86CPU *x86cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86cpu->env; ++ hv_register_assoc assocs[ARRAY_SIZE(STANDARD_REGISTER_NAMES)]; ++ int ret; ++ size_t n_regs = ARRAY_SIZE(STANDARD_REGISTER_NAMES); ++ ++ /* set names */ ++ for (size_t i = 0; i < ARRAY_SIZE(STANDARD_REGISTER_NAMES); i++) { ++ assocs[i].name = STANDARD_REGISTER_NAMES[i]; ++ } ++ assocs[0].value.reg64 = env->regs[R_EAX]; ++ assocs[1].value.reg64 = env->regs[R_EBX]; ++ assocs[2].value.reg64 = env->regs[R_ECX]; ++ assocs[3].value.reg64 = env->regs[R_EDX]; ++ assocs[4].value.reg64 = env->regs[R_ESI]; ++ assocs[5].value.reg64 = env->regs[R_EDI]; ++ assocs[6].value.reg64 = env->regs[R_ESP]; ++ assocs[7].value.reg64 = env->regs[R_EBP]; ++ assocs[8].value.reg64 = env->regs[R_R8]; ++ assocs[9].value.reg64 = env->regs[R_R9]; ++ assocs[10].value.reg64 = env->regs[R_R10]; ++ assocs[11].value.reg64 = env->regs[R_R11]; ++ assocs[12].value.reg64 = env->regs[R_R12]; ++ assocs[13].value.reg64 = env->regs[R_R13]; ++ assocs[14].value.reg64 = env->regs[R_R14]; ++ assocs[15].value.reg64 = env->regs[R_R15]; ++ assocs[16].value.reg64 = env->eip; ++ lflags_to_rflags(env); ++ assocs[17].value.reg64 = env->eflags; ++ ++ ret = mshv_set_generic_regs(cpu, assocs, n_regs); ++ if (ret < 0) { ++ error_report("failed to set standard registers"); ++ return -errno; ++ } ++ return 0; ++} ++ + int mshv_store_regs(CPUState *cpu) + { +- error_report("unimplemented"); +- abort(); ++ int ret; ++ ++ ret = set_standard_regs(cpu); ++ if (ret < 0) { ++ error_report("Failed to store standard registers"); ++ return -1; ++ } ++ ++ return 0; + } + ++ + int mshv_load_regs(CPUState *cpu) + { + error_report("unimplemented"); +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Implement-mshv_vcpu_run.patch b/kvm-target-i386-mshv-Implement-mshv_vcpu_run.patch new file mode 100644 index 0000000..b76928c --- /dev/null +++ b/kvm-target-i386-mshv-Implement-mshv_vcpu_run.patch @@ -0,0 +1,489 @@ +From 334e33f443ce36355a29909ee41f49789ddae6d4 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:42 +0200 +Subject: [PATCH 25/32] target/i386/mshv: Implement mshv_vcpu_run() + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [23/30] f21e5df491be51f4256237f9497fbc73f44cc386 + +Add the main vCPU execution loop for MSHV using the MSHV_RUN_VP ioctl. + +The execution loop handles guest entry and VM exits. There are handlers for +memory r/w, PIO and MMIO to which the exit events are dispatched. + +In case of MMIO the i386 instruction decoder/emulator is invoked to +perform the operation in user space. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-23-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 6dec60528c419781f1f79cd9837d5f26415a3fd7) +Signed-off-by: Igor Mammedov +--- + target/i386/mshv/mshv-cpu.c | 444 +++++++++++++++++++++++++++++++++++- + 1 file changed, 442 insertions(+), 2 deletions(-) + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 33a3ce8b11..7edc032cea 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -1082,10 +1082,450 @@ void mshv_arch_amend_proc_features( + features->access_guest_idle_reg = 1; + } + ++static int set_memory_info(const struct hyperv_message *msg, ++ struct hv_x64_memory_intercept_message *info) ++{ ++ if (msg->header.message_type != HVMSG_GPA_INTERCEPT ++ && msg->header.message_type != HVMSG_UNMAPPED_GPA ++ && msg->header.message_type != HVMSG_UNACCEPTED_GPA) { ++ error_report("invalid message type"); ++ return -1; ++ } ++ memcpy(info, msg->payload, sizeof(*info)); ++ ++ return 0; ++} ++ ++static int emulate_instruction(CPUState *cpu, ++ const uint8_t *insn_bytes, size_t insn_len, ++ uint64_t gva, uint64_t gpa) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ struct x86_decode decode = { 0 }; ++ int ret; ++ x86_insn_stream stream = { .bytes = insn_bytes, .len = insn_len }; ++ ++ ret = mshv_load_regs(cpu); ++ if (ret < 0) { ++ error_report("failed to load registers"); ++ return -1; ++ } ++ ++ decode_instruction_stream(env, &decode, &stream); ++ exec_instruction(env, &decode); ++ ++ ret = mshv_store_regs(cpu); ++ if (ret < 0) { ++ error_report("failed to store registers"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int handle_mmio(CPUState *cpu, const struct hyperv_message *msg, ++ MshvVmExit *exit_reason) ++{ ++ struct hv_x64_memory_intercept_message info = { 0 }; ++ size_t insn_len; ++ uint8_t access_type; ++ uint8_t *instruction_bytes; ++ int ret; ++ ++ ret = set_memory_info(msg, &info); ++ if (ret < 0) { ++ error_report("failed to convert message to memory info"); ++ return -1; ++ } ++ insn_len = info.instruction_byte_count; ++ access_type = info.header.intercept_access_type; ++ ++ if (access_type == HV_X64_INTERCEPT_ACCESS_TYPE_EXECUTE) { ++ error_report("invalid intercept access type: execute"); ++ return -1; ++ } ++ ++ if (insn_len > 16) { ++ error_report("invalid mmio instruction length: %zu", insn_len); ++ return -1; ++ } ++ ++ trace_mshv_handle_mmio(info.guest_virtual_address, ++ info.guest_physical_address, ++ info.instruction_byte_count, access_type); ++ ++ instruction_bytes = info.instruction_bytes; ++ ++ ret = emulate_instruction(cpu, instruction_bytes, insn_len, ++ info.guest_virtual_address, ++ info.guest_physical_address); ++ if (ret < 0) { ++ error_report("failed to emulate mmio"); ++ return -1; ++ } ++ ++ *exit_reason = MshvVmExitIgnore; ++ ++ return 0; ++} ++ ++static int set_ioport_info(const struct hyperv_message *msg, ++ hv_x64_io_port_intercept_message *info) ++{ ++ if (msg->header.message_type != HVMSG_X64_IO_PORT_INTERCEPT) { ++ error_report("Invalid message type"); ++ return -1; ++ } ++ memcpy(info, msg->payload, sizeof(*info)); ++ ++ return 0; ++} ++ ++static int set_x64_registers(const CPUState *cpu, const uint32_t *names, ++ const uint64_t *values) ++{ ++ ++ hv_register_assoc assocs[2]; ++ int ret; ++ ++ for (size_t i = 0; i < ARRAY_SIZE(assocs); i++) { ++ assocs[i].name = names[i]; ++ assocs[i].value.reg64 = values[i]; ++ } ++ ++ ret = mshv_set_generic_regs(cpu, assocs, ARRAY_SIZE(assocs)); ++ if (ret < 0) { ++ error_report("failed to set x64 registers"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static inline MemTxAttrs get_mem_attrs(bool is_secure_mode) ++{ ++ MemTxAttrs memattr = {0}; ++ memattr.secure = is_secure_mode; ++ return memattr; ++} ++ ++static void pio_read(uint64_t port, uint8_t *data, uintptr_t size, ++ bool is_secure_mode) ++{ ++ int ret = 0; ++ MemTxAttrs memattr = get_mem_attrs(is_secure_mode); ++ ret = address_space_rw(&address_space_io, port, memattr, (void *)data, size, ++ false); ++ if (ret != MEMTX_OK) { ++ error_report("Failed to read from port %lx: %d", port, ret); ++ abort(); ++ } ++} ++ ++static int pio_write(uint64_t port, const uint8_t *data, uintptr_t size, ++ bool is_secure_mode) ++{ ++ int ret = 0; ++ MemTxAttrs memattr = get_mem_attrs(is_secure_mode); ++ ret = address_space_rw(&address_space_io, port, memattr, (void *)data, size, ++ true); ++ return ret; ++} ++ ++static int handle_pio_non_str(const CPUState *cpu, ++ hv_x64_io_port_intercept_message *info) ++{ ++ size_t len = info->access_info.access_size; ++ uint8_t access_type = info->header.intercept_access_type; ++ int ret; ++ uint32_t val, eax; ++ const uint32_t eax_mask = 0xffffffffu >> (32 - len * 8); ++ size_t insn_len; ++ uint64_t rip, rax; ++ uint32_t reg_names[2]; ++ uint64_t reg_values[2]; ++ uint16_t port = info->port_number; ++ ++ if (access_type == HV_X64_INTERCEPT_ACCESS_TYPE_WRITE) { ++ union { ++ uint32_t u32; ++ uint8_t bytes[4]; ++ } conv; ++ ++ /* convert the first 4 bytes of rax to bytes */ ++ conv.u32 = (uint32_t)info->rax; ++ /* secure mode is set to false */ ++ ret = pio_write(port, conv.bytes, len, false); ++ if (ret < 0) { ++ error_report("Failed to write to io port"); ++ return -1; ++ } ++ } else { ++ uint8_t data[4] = { 0 }; ++ /* secure mode is set to false */ ++ pio_read(info->port_number, data, len, false); ++ ++ /* Preserve high bits in EAX, but clear out high bits in RAX */ ++ val = *(uint32_t *)data; ++ eax = (((uint32_t)info->rax) & ~eax_mask) | (val & eax_mask); ++ info->rax = (uint64_t)eax; ++ } ++ ++ insn_len = info->header.instruction_length; ++ ++ /* Advance RIP and update RAX */ ++ rip = info->header.rip + insn_len; ++ rax = info->rax; ++ ++ reg_names[0] = HV_X64_REGISTER_RIP; ++ reg_values[0] = rip; ++ reg_names[1] = HV_X64_REGISTER_RAX; ++ reg_values[1] = rax; ++ ++ ret = set_x64_registers(cpu, reg_names, reg_values); ++ if (ret < 0) { ++ error_report("Failed to set x64 registers"); ++ return -1; ++ } ++ ++ cpu->accel->dirty = false; ++ ++ return 0; ++} ++ ++static int fetch_guest_state(CPUState *cpu) ++{ ++ int ret; ++ ++ ret = mshv_get_standard_regs(cpu); ++ if (ret < 0) { ++ error_report("Failed to get standard registers"); ++ return -1; ++ } ++ ++ ret = mshv_get_special_regs(cpu); ++ if (ret < 0) { ++ error_report("Failed to get special registers"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int read_memory(const CPUState *cpu, uint64_t initial_gva, ++ uint64_t initial_gpa, uint64_t gva, uint8_t *data, ++ size_t len) ++{ ++ int ret; ++ uint64_t gpa, flags; ++ ++ if (gva == initial_gva) { ++ gpa = initial_gpa; ++ } else { ++ flags = HV_TRANSLATE_GVA_VALIDATE_READ; ++ ret = translate_gva(cpu, gva, &gpa, flags); ++ if (ret < 0) { ++ return -1; ++ } ++ ++ ret = mshv_guest_mem_read(gpa, data, len, false, false); ++ if (ret < 0) { ++ error_report("failed to read guest mem"); ++ return -1; ++ } ++ } ++ ++ return 0; ++} ++ ++static int write_memory(const CPUState *cpu, uint64_t initial_gva, ++ uint64_t initial_gpa, uint64_t gva, const uint8_t *data, ++ size_t len) ++{ ++ int ret; ++ uint64_t gpa, flags; ++ ++ if (gva == initial_gva) { ++ gpa = initial_gpa; ++ } else { ++ flags = HV_TRANSLATE_GVA_VALIDATE_WRITE; ++ ret = translate_gva(cpu, gva, &gpa, flags); ++ if (ret < 0) { ++ error_report("failed to translate gva to gpa"); ++ return -1; ++ } ++ } ++ ret = mshv_guest_mem_write(gpa, data, len, false); ++ if (ret != MEMTX_OK) { ++ error_report("failed to write to mmio"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int handle_pio_str_write(CPUState *cpu, ++ hv_x64_io_port_intercept_message *info, ++ size_t repeat, uint16_t port, ++ bool direction_flag) ++{ ++ int ret; ++ uint64_t src; ++ uint8_t data[4] = { 0 }; ++ size_t len = info->access_info.access_size; ++ ++ src = linear_addr(cpu, info->rsi, R_DS); ++ ++ for (size_t i = 0; i < repeat; i++) { ++ ret = read_memory(cpu, 0, 0, src, data, len); ++ if (ret < 0) { ++ error_report("Failed to read memory"); ++ return -1; ++ } ++ ret = pio_write(port, data, len, false); ++ if (ret < 0) { ++ error_report("Failed to write to io port"); ++ return -1; ++ } ++ src += direction_flag ? -len : len; ++ info->rsi += direction_flag ? -len : len; ++ } ++ ++ return 0; ++} ++ ++static int handle_pio_str_read(CPUState *cpu, ++ hv_x64_io_port_intercept_message *info, ++ size_t repeat, uint16_t port, ++ bool direction_flag) ++{ ++ int ret; ++ uint64_t dst; ++ size_t len = info->access_info.access_size; ++ uint8_t data[4] = { 0 }; ++ ++ dst = linear_addr(cpu, info->rdi, R_ES); ++ ++ for (size_t i = 0; i < repeat; i++) { ++ pio_read(port, data, len, false); ++ ++ ret = write_memory(cpu, 0, 0, dst, data, len); ++ if (ret < 0) { ++ error_report("Failed to write memory"); ++ return -1; ++ } ++ dst += direction_flag ? -len : len; ++ info->rdi += direction_flag ? -len : len; ++ } ++ ++ return 0; ++} ++ ++static int handle_pio_str(CPUState *cpu, hv_x64_io_port_intercept_message *info) ++{ ++ uint8_t access_type = info->header.intercept_access_type; ++ uint16_t port = info->port_number; ++ bool repop = info->access_info.rep_prefix == 1; ++ size_t repeat = repop ? info->rcx : 1; ++ size_t insn_len = info->header.instruction_length; ++ bool direction_flag; ++ uint32_t reg_names[3]; ++ uint64_t reg_values[3]; ++ int ret; ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ ++ ret = fetch_guest_state(cpu); ++ if (ret < 0) { ++ error_report("Failed to fetch guest state"); ++ return -1; ++ } ++ ++ direction_flag = (env->eflags & DESC_E_MASK) != 0; ++ ++ if (access_type == HV_X64_INTERCEPT_ACCESS_TYPE_WRITE) { ++ ret = handle_pio_str_write(cpu, info, repeat, port, direction_flag); ++ if (ret < 0) { ++ error_report("Failed to handle pio str write"); ++ return -1; ++ } ++ reg_names[0] = HV_X64_REGISTER_RSI; ++ reg_values[0] = info->rsi; ++ } else { ++ ret = handle_pio_str_read(cpu, info, repeat, port, direction_flag); ++ reg_names[0] = HV_X64_REGISTER_RDI; ++ reg_values[0] = info->rdi; ++ } ++ ++ reg_names[1] = HV_X64_REGISTER_RIP; ++ reg_values[1] = info->header.rip + insn_len; ++ reg_names[2] = HV_X64_REGISTER_RAX; ++ reg_values[2] = info->rax; ++ ++ ret = set_x64_registers(cpu, reg_names, reg_values); ++ if (ret < 0) { ++ error_report("Failed to set x64 registers"); ++ return -1; ++ } ++ ++ cpu->accel->dirty = false; ++ ++ return 0; ++} ++ ++static int handle_pio(CPUState *cpu, const struct hyperv_message *msg) ++{ ++ struct hv_x64_io_port_intercept_message info = { 0 }; ++ int ret; ++ ++ ret = set_ioport_info(msg, &info); ++ if (ret < 0) { ++ error_report("Failed to convert message to ioport info"); ++ return -1; ++ } ++ ++ if (info.access_info.string_op) { ++ return handle_pio_str(cpu, &info); ++ } ++ ++ return handle_pio_non_str(cpu, &info); ++} ++ + int mshv_run_vcpu(int vm_fd, CPUState *cpu, hv_message *msg, MshvVmExit *exit) + { +- error_report("unimplemented"); +- abort(); ++ int ret; ++ enum MshvVmExit exit_reason; ++ int cpu_fd = mshv_vcpufd(cpu); ++ ++ ret = ioctl(cpu_fd, MSHV_RUN_VP, msg); ++ if (ret < 0) { ++ return MshvVmExitShutdown; ++ } ++ ++ switch (msg->header.message_type) { ++ case HVMSG_UNRECOVERABLE_EXCEPTION: ++ return MshvVmExitShutdown; ++ case HVMSG_UNMAPPED_GPA: ++ case HVMSG_GPA_INTERCEPT: ++ ret = handle_mmio(cpu, msg, &exit_reason); ++ if (ret < 0) { ++ error_report("failed to handle mmio"); ++ return -1; ++ } ++ return exit_reason; ++ case HVMSG_X64_IO_PORT_INTERCEPT: ++ ret = handle_pio(cpu, msg); ++ if (ret < 0) { ++ return MshvVmExitSpecial; ++ } ++ return MshvVmExitIgnore; ++ default: ++ break; ++ } ++ ++ *exit = MshvVmExitIgnore; ++ return 0; + } + + void mshv_remove_vcpu(int vm_fd, int cpu_fd) +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Integrate-x86-instruction-decoder-e.patch b/kvm-target-i386-mshv-Integrate-x86-instruction-decoder-e.patch new file mode 100644 index 0000000..3dc0f45 --- /dev/null +++ b/kvm-target-i386-mshv-Integrate-x86-instruction-decoder-e.patch @@ -0,0 +1,283 @@ +From 3d2a5157c0ad300f954ba9b3fddadd4da5120eac Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:40 +0200 +Subject: [PATCH 23/32] target/i386/mshv: Integrate x86 instruction + decoder/emulator + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [21/30] c9dba7c3e0a4c896fa581b97f121a293606e074f + +Connect the x86 instruction decoder and emulator to the MSHV backend +to handle intercepted instructions. This enables software emulation +of MMIO operations in MSHV guests. MSHV has a translate_gva hypercall +that is used to accessing the physical guest memory. + +A guest might read from unmapped memory regions (e.g. OVMF will probe +0xfed40000 for a vTPM). In those cases 0xFF bytes is returned instead of +aborting the execution. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-21-magnuskulke@linux.microsoft.com +[mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit 9bc6a1d29605a13c541629a651e41787af65a963) +Signed-off-by: Igor Mammedov +--- + accel/mshv/mem.c | 65 +++++++++++++++++ + include/system/mshv_int.h | 4 ++ + target/i386/mshv/mshv-cpu.c | 135 ++++++++++++++++++++++++++++++++++++ + 3 files changed, 204 insertions(+) + +diff --git a/accel/mshv/mem.c b/accel/mshv/mem.c +index a0a40eb333..e55c38d4db 100644 +--- a/accel/mshv/mem.c ++++ b/accel/mshv/mem.c +@@ -59,6 +59,71 @@ static int map_or_unmap(int vm_fd, const MshvMemoryRegion *mr, bool map) + return set_guest_memory(vm_fd, ®ion); + } + ++static int handle_unmapped_mmio_region_read(uint64_t gpa, uint64_t size, ++ uint8_t *data) ++{ ++ warn_report("read from unmapped mmio region gpa=0x%lx size=%lu", gpa, size); ++ ++ if (size == 0 || size > 8) { ++ error_report("invalid size %lu for reading from unmapped mmio region", ++ size); ++ return -1; ++ } ++ ++ memset(data, 0xFF, size); ++ ++ return 0; ++} ++ ++int mshv_guest_mem_read(uint64_t gpa, uint8_t *data, uintptr_t size, ++ bool is_secure_mode, bool instruction_fetch) ++{ ++ int ret; ++ MemTxAttrs memattr = { .secure = is_secure_mode }; ++ ++ if (instruction_fetch) { ++ trace_mshv_insn_fetch(gpa, size); ++ } else { ++ trace_mshv_mem_read(gpa, size); ++ } ++ ++ ret = address_space_rw(&address_space_memory, gpa, memattr, (void *)data, ++ size, false); ++ if (ret == MEMTX_OK) { ++ return 0; ++ } ++ ++ if (ret == MEMTX_DECODE_ERROR) { ++ return handle_unmapped_mmio_region_read(gpa, size, data); ++ } ++ ++ error_report("failed to read guest memory at 0x%lx", gpa); ++ return -1; ++} ++ ++int mshv_guest_mem_write(uint64_t gpa, const uint8_t *data, uintptr_t size, ++ bool is_secure_mode) ++{ ++ int ret; ++ MemTxAttrs memattr = { .secure = is_secure_mode }; ++ ++ trace_mshv_mem_write(gpa, size); ++ ret = address_space_rw(&address_space_memory, gpa, memattr, (void *)data, ++ size, true); ++ if (ret == MEMTX_OK) { ++ return 0; ++ } ++ ++ if (ret == MEMTX_DECODE_ERROR) { ++ warn_report("write to unmapped mmio region gpa=0x%lx size=%lu", gpa, ++ size); ++ return 0; ++ } ++ ++ error_report("Failed to write guest memory"); ++ return -1; ++} ++ + static int set_memory(const MshvMemoryRegion *mshv_mr, bool add) + { + int ret = 0; +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index 6649438313..b29d39911d 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -101,6 +101,10 @@ typedef struct MshvMemoryRegion { + bool readonly; + } MshvMemoryRegion; + ++int mshv_guest_mem_read(uint64_t gpa, uint8_t *data, uintptr_t size, ++ bool is_secure_mode, bool instruction_fetch); ++int mshv_guest_mem_write(uint64_t gpa, const uint8_t *data, uintptr_t size, ++ bool is_secure_mode); + void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, + bool add); + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 1f43dfc58a..424ebdb122 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -104,6 +104,47 @@ static enum hv_register_name FPU_REGISTER_NAMES[26] = { + HV_X64_REGISTER_XMM_CONTROL_STATUS, + }; + ++static int translate_gva(const CPUState *cpu, uint64_t gva, uint64_t *gpa, ++ uint64_t flags) ++{ ++ int ret; ++ int cpu_fd = mshv_vcpufd(cpu); ++ int vp_index = cpu->cpu_index; ++ ++ hv_input_translate_virtual_address in = { 0 }; ++ hv_output_translate_virtual_address out = { 0 }; ++ struct mshv_root_hvcall args = {0}; ++ uint64_t gva_page = gva >> HV_HYP_PAGE_SHIFT; ++ ++ in.vp_index = vp_index; ++ in.control_flags = flags; ++ in.gva_page = gva_page; ++ ++ /* create the hvcall envelope */ ++ args.code = HVCALL_TRANSLATE_VIRTUAL_ADDRESS; ++ args.in_sz = sizeof(in); ++ args.in_ptr = (uint64_t) ∈ ++ args.out_sz = sizeof(out); ++ args.out_ptr = (uint64_t) &out; ++ ++ /* perform the call */ ++ ret = mshv_hvcall(cpu_fd, &args); ++ if (ret < 0) { ++ error_report("Failed to invoke gva->gpa translation"); ++ return -errno; ++ } ++ ++ if (out.translation_result.result_code != HV_TRANSLATE_GVA_SUCCESS) { ++ error_report("Failed to translate gva (" TARGET_FMT_lx ") to gpa", gva); ++ return -1; ++ } ++ ++ *gpa = ((out.gpa_page << HV_HYP_PAGE_SHIFT) ++ | (gva & ~(uint64_t)HV_HYP_PAGE_MASK)); ++ ++ return 0; ++} ++ + int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, + size_t n_regs) + { +@@ -1006,8 +1047,102 @@ int mshv_create_vcpu(int vm_fd, uint8_t vp_index, int *cpu_fd) + return 0; + } + ++static int guest_mem_read_with_gva(const CPUState *cpu, uint64_t gva, ++ uint8_t *data, uintptr_t size, ++ bool fetch_instruction) ++{ ++ int ret; ++ uint64_t gpa, flags; ++ ++ flags = HV_TRANSLATE_GVA_VALIDATE_READ; ++ ret = translate_gva(cpu, gva, &gpa, flags); ++ if (ret < 0) { ++ error_report("failed to translate gva to gpa"); ++ return -1; ++ } ++ ++ ret = mshv_guest_mem_read(gpa, data, size, false, fetch_instruction); ++ if (ret < 0) { ++ error_report("failed to read from guest memory"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int guest_mem_write_with_gva(const CPUState *cpu, uint64_t gva, ++ const uint8_t *data, uintptr_t size) ++{ ++ int ret; ++ uint64_t gpa, flags; ++ ++ flags = HV_TRANSLATE_GVA_VALIDATE_WRITE; ++ ret = translate_gva(cpu, gva, &gpa, flags); ++ if (ret < 0) { ++ error_report("failed to translate gva to gpa"); ++ return -1; ++ } ++ ret = mshv_guest_mem_write(gpa, data, size, false); ++ if (ret < 0) { ++ error_report("failed to write to guest memory"); ++ return -1; ++ } ++ return 0; ++} ++ ++static void write_mem(CPUState *cpu, void *data, target_ulong addr, int bytes) ++{ ++ if (guest_mem_write_with_gva(cpu, addr, data, bytes) < 0) { ++ error_report("failed to write memory"); ++ abort(); ++ } ++} ++ ++static void fetch_instruction(CPUState *cpu, void *data, ++ target_ulong addr, int bytes) ++{ ++ if (guest_mem_read_with_gva(cpu, addr, data, bytes, true) < 0) { ++ error_report("failed to fetch instruction"); ++ abort(); ++ } ++} ++ ++static void read_mem(CPUState *cpu, void *data, target_ulong addr, int bytes) ++{ ++ if (guest_mem_read_with_gva(cpu, addr, data, bytes, false) < 0) { ++ error_report("failed to read memory"); ++ abort(); ++ } ++} ++ ++static void read_segment_descriptor(CPUState *cpu, ++ struct x86_segment_descriptor *desc, ++ enum X86Seg seg_idx) ++{ ++ bool ret; ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ SegmentCache *seg = &env->segs[seg_idx]; ++ x86_segment_selector sel = { .sel = seg->selector & 0xFFFF }; ++ ++ ret = x86_read_segment_descriptor(cpu, desc, sel); ++ if (ret == false) { ++ error_report("failed to read segment descriptor"); ++ abort(); ++ } ++} ++ ++static const struct x86_emul_ops mshv_x86_emul_ops = { ++ .fetch_instruction = fetch_instruction, ++ .read_mem = read_mem, ++ .write_mem = write_mem, ++ .read_segment_descriptor = read_segment_descriptor, ++}; ++ + void mshv_init_mmio_emu(void) + { ++ init_decoder(); ++ init_emu(&mshv_x86_emul_ops); + } + + void mshv_arch_init_vcpu(CPUState *cpu) +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Register-CPUID-entries-with-MSHV.patch b/kvm-target-i386-mshv-Register-CPUID-entries-with-MSHV.patch new file mode 100644 index 0000000..678f176 --- /dev/null +++ b/kvm-target-i386-mshv-Register-CPUID-entries-with-MSHV.patch @@ -0,0 +1,252 @@ +From d122c4cbb920e5d20ee70da4496b2b686319532f Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:38 +0200 +Subject: [PATCH 21/32] target/i386/mshv: Register CPUID entries with MSHV + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [19/30] 4a1e70942d21acee2c90095afd82e9f376b87b56 + +Convert the guest CPU's CPUID model into MSHV's format and register it +with the hypervisor. This ensures that the guest observes the correct +CPU feature set during CPUID instructions. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-19-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 4fa04dd16216e266564606b7da582e5fce07bced) +Signed-off-by: Igor Mammedov +--- + target/i386/mshv/mshv-cpu.c | 206 ++++++++++++++++++++++++++++++++++++ + 1 file changed, 206 insertions(+) + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 0fe3cbb48d..2b7a81274b 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -403,6 +403,206 @@ int mshv_load_regs(CPUState *cpu) + return 0; + } + ++static void add_cpuid_entry(GList *cpuid_entries, ++ uint32_t function, uint32_t index, ++ uint32_t eax, uint32_t ebx, ++ uint32_t ecx, uint32_t edx) ++{ ++ struct hv_cpuid_entry *entry; ++ ++ entry = g_malloc0(sizeof(struct hv_cpuid_entry)); ++ entry->function = function; ++ entry->index = index; ++ entry->eax = eax; ++ entry->ebx = ebx; ++ entry->ecx = ecx; ++ entry->edx = edx; ++ ++ cpuid_entries = g_list_append(cpuid_entries, entry); ++} ++ ++static void collect_cpuid_entries(const CPUState *cpu, GList *cpuid_entries) ++{ ++ X86CPU *x86_cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86_cpu->env; ++ uint32_t eax, ebx, ecx, edx; ++ uint32_t leaf, subleaf; ++ size_t max_leaf = 0x1F; ++ size_t max_subleaf = 0x20; ++ ++ uint32_t leaves_with_subleaves[] = {0x4, 0x7, 0xD, 0xF, 0x10}; ++ int n_subleaf_leaves = ARRAY_SIZE(leaves_with_subleaves); ++ ++ /* Regular leaves without subleaves */ ++ for (leaf = 0; leaf <= max_leaf; leaf++) { ++ bool has_subleaves = false; ++ for (int i = 0; i < n_subleaf_leaves; i++) { ++ if (leaf == leaves_with_subleaves[i]) { ++ has_subleaves = true; ++ break; ++ } ++ } ++ ++ if (!has_subleaves) { ++ cpu_x86_cpuid(env, leaf, 0, &eax, &ebx, &ecx, &edx); ++ if (eax == 0 && ebx == 0 && ecx == 0 && edx == 0) { ++ /* all zeroes indicates no more leaves */ ++ continue; ++ } ++ ++ add_cpuid_entry(cpuid_entries, leaf, 0, eax, ebx, ecx, edx); ++ continue; ++ } ++ ++ subleaf = 0; ++ while (subleaf < max_subleaf) { ++ cpu_x86_cpuid(env, leaf, subleaf, &eax, &ebx, &ecx, &edx); ++ ++ if (eax == 0 && ebx == 0 && ecx == 0 && edx == 0) { ++ /* all zeroes indicates no more leaves */ ++ break; ++ } ++ add_cpuid_entry(cpuid_entries, leaf, 0, eax, ebx, ecx, edx); ++ subleaf++; ++ } ++ } ++} ++ ++static int register_intercept_result_cpuid_entry(const CPUState *cpu, ++ uint8_t subleaf_specific, ++ uint8_t always_override, ++ struct hv_cpuid_entry *entry) ++{ ++ int ret; ++ int vp_index = cpu->cpu_index; ++ int cpu_fd = mshv_vcpufd(cpu); ++ ++ struct hv_register_x64_cpuid_result_parameters cpuid_params = { ++ .input.eax = entry->function, ++ .input.ecx = entry->index, ++ .input.subleaf_specific = subleaf_specific, ++ .input.always_override = always_override, ++ .input.padding = 0, ++ /* ++ * With regard to masks - these are to specify bits to be overwritten ++ * The current CpuidEntry structure wouldn't allow to carry the masks ++ * in addition to the actual register values. For this reason, the ++ * masks are set to the exact values of the corresponding register bits ++ * to be registered for an overwrite. To view resulting values the ++ * hypervisor would return, HvCallGetVpCpuidValues hypercall can be ++ * used. ++ */ ++ .result.eax = entry->eax, ++ .result.eax_mask = entry->eax, ++ .result.ebx = entry->ebx, ++ .result.ebx_mask = entry->ebx, ++ .result.ecx = entry->ecx, ++ .result.ecx_mask = entry->ecx, ++ .result.edx = entry->edx, ++ .result.edx_mask = entry->edx, ++ }; ++ union hv_register_intercept_result_parameters parameters = { ++ .cpuid = cpuid_params, ++ }; ++ ++ hv_input_register_intercept_result in = {0}; ++ in.vp_index = vp_index; ++ in.intercept_type = HV_INTERCEPT_TYPE_X64_CPUID; ++ in.parameters = parameters; ++ ++ struct mshv_root_hvcall args = {0}; ++ args.code = HVCALL_REGISTER_INTERCEPT_RESULT; ++ args.in_sz = sizeof(in); ++ args.in_ptr = (uint64_t)∈ ++ ++ ret = mshv_hvcall(cpu_fd, &args); ++ if (ret < 0) { ++ error_report("failed to register intercept result for cpuid"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int register_intercept_result_cpuid(const CPUState *cpu, ++ struct hv_cpuid *cpuid) ++{ ++ int ret = 0, entry_ret; ++ struct hv_cpuid_entry *entry; ++ uint8_t subleaf_specific, always_override; ++ ++ for (size_t i = 0; i < cpuid->nent; i++) { ++ entry = &cpuid->entries[i]; ++ ++ /* set defaults */ ++ subleaf_specific = 0; ++ always_override = 1; ++ ++ /* Intel */ ++ /* 0xb - Extended Topology Enumeration Leaf */ ++ /* 0x1f - V2 Extended Topology Enumeration Leaf */ ++ /* AMD */ ++ /* 0x8000_001e - Processor Topology Information */ ++ /* 0x8000_0026 - Extended CPU Topology */ ++ if (entry->function == 0xb ++ || entry->function == 0x1f ++ || entry->function == 0x8000001e ++ || entry->function == 0x80000026) { ++ subleaf_specific = 1; ++ always_override = 1; ++ } else if (entry->function == 0x00000001 ++ || entry->function == 0x80000000 ++ || entry->function == 0x80000001 ++ || entry->function == 0x80000008) { ++ subleaf_specific = 0; ++ always_override = 1; ++ } ++ ++ entry_ret = register_intercept_result_cpuid_entry(cpu, subleaf_specific, ++ always_override, ++ entry); ++ if ((entry_ret < 0) && (ret == 0)) { ++ ret = entry_ret; ++ } ++ } ++ ++ return ret; ++} ++ ++static int set_cpuid2(const CPUState *cpu) ++{ ++ int ret; ++ size_t n_entries, cpuid_size; ++ struct hv_cpuid *cpuid; ++ struct hv_cpuid_entry *entry; ++ GList *entries = NULL; ++ ++ collect_cpuid_entries(cpu, entries); ++ n_entries = g_list_length(entries); ++ ++ cpuid_size = sizeof(struct hv_cpuid) ++ + n_entries * sizeof(struct hv_cpuid_entry); ++ ++ cpuid = g_malloc0(cpuid_size); ++ cpuid->nent = n_entries; ++ cpuid->padding = 0; ++ ++ for (size_t i = 0; i < n_entries; i++) { ++ entry = g_list_nth_data(entries, i); ++ cpuid->entries[i] = *entry; ++ g_free(entry); ++ } ++ g_list_free(entries); ++ ++ ret = register_intercept_result_cpuid(cpu, cpuid); ++ g_free(cpuid); ++ if (ret < 0) { ++ return ret; ++ } ++ ++ return 0; ++} ++ + static inline void populate_hv_segment_reg(SegmentCache *seg, + hv_x64_segment_register *hv_reg) + { +@@ -685,6 +885,12 @@ int mshv_configure_vcpu(const CPUState *cpu, const struct MshvFPU *fpu, + int ret; + int cpu_fd = mshv_vcpufd(cpu); + ++ ret = set_cpuid2(cpu); ++ if (ret < 0) { ++ error_report("failed to set cpuid"); ++ return -1; ++ } ++ + ret = set_cpu_state(cpu, fpu, xcr0); + if (ret < 0) { + error_report("failed to set cpu state"); +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Register-MSRs-with-MSHV.patch b/kvm-target-i386-mshv-Register-MSRs-with-MSHV.patch new file mode 100644 index 0000000..abda761 --- /dev/null +++ b/kvm-target-i386-mshv-Register-MSRs-with-MSHV.patch @@ -0,0 +1,528 @@ +From 5bbfd219b62bdea50ee0d16396f7eed138f7771f Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Thu, 2 Oct 2025 18:19:22 +0200 +Subject: [PATCH 22/32] target/i386/mshv: Register MSRs with MSHV + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [20/30] 43ca27cba532fd1d562fa10ade6261df58af82fd + +Build and register the guest vCPU's model-specific registers using +the MSHV interface. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-20-magnuskulke@linux.microsoft.com +[mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit f38e2a63e541730e114f6ed09e5f8719e388c8db) +Signed-off-by: Igor Mammedov +--- + accel/mshv/meson.build | 1 + + accel/mshv/msr.c | 375 ++++++++++++++++++++++++++++++++++++ + include/system/mshv_int.h | 17 ++ + target/i386/cpu.h | 2 + + target/i386/mshv/mshv-cpu.c | 33 ++++ + 5 files changed, 428 insertions(+) + create mode 100644 accel/mshv/msr.c + +diff --git a/accel/mshv/meson.build b/accel/mshv/meson.build +index f88fc8678c..d3a2b32581 100644 +--- a/accel/mshv/meson.build ++++ b/accel/mshv/meson.build +@@ -2,6 +2,7 @@ mshv_ss = ss.source_set() + mshv_ss.add(if_true: files( + 'irq.c', + 'mem.c', ++ 'msr.c', + 'mshv-all.c' + )) + +diff --git a/accel/mshv/msr.c b/accel/mshv/msr.c +new file mode 100644 +index 0000000000..e6e5baef50 +--- /dev/null ++++ b/accel/mshv/msr.c +@@ -0,0 +1,375 @@ ++/* ++ * QEMU MSHV support ++ * ++ * Copyright Microsoft, Corp. 2025 ++ * ++ * Authors: Magnus Kulke ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "system/mshv.h" ++#include "system/mshv_int.h" ++#include "hw/hyperv/hvgdk_mini.h" ++#include "linux/mshv.h" ++#include "qemu/error-report.h" ++ ++static uint32_t supported_msrs[64] = { ++ IA32_MSR_TSC, ++ IA32_MSR_EFER, ++ IA32_MSR_KERNEL_GS_BASE, ++ IA32_MSR_APIC_BASE, ++ IA32_MSR_PAT, ++ IA32_MSR_SYSENTER_CS, ++ IA32_MSR_SYSENTER_ESP, ++ IA32_MSR_SYSENTER_EIP, ++ IA32_MSR_STAR, ++ IA32_MSR_LSTAR, ++ IA32_MSR_CSTAR, ++ IA32_MSR_SFMASK, ++ IA32_MSR_MTRR_DEF_TYPE, ++ IA32_MSR_MTRR_PHYSBASE0, ++ IA32_MSR_MTRR_PHYSMASK0, ++ IA32_MSR_MTRR_PHYSBASE1, ++ IA32_MSR_MTRR_PHYSMASK1, ++ IA32_MSR_MTRR_PHYSBASE2, ++ IA32_MSR_MTRR_PHYSMASK2, ++ IA32_MSR_MTRR_PHYSBASE3, ++ IA32_MSR_MTRR_PHYSMASK3, ++ IA32_MSR_MTRR_PHYSBASE4, ++ IA32_MSR_MTRR_PHYSMASK4, ++ IA32_MSR_MTRR_PHYSBASE5, ++ IA32_MSR_MTRR_PHYSMASK5, ++ IA32_MSR_MTRR_PHYSBASE6, ++ IA32_MSR_MTRR_PHYSMASK6, ++ IA32_MSR_MTRR_PHYSBASE7, ++ IA32_MSR_MTRR_PHYSMASK7, ++ IA32_MSR_MTRR_FIX64K_00000, ++ IA32_MSR_MTRR_FIX16K_80000, ++ IA32_MSR_MTRR_FIX16K_A0000, ++ IA32_MSR_MTRR_FIX4K_C0000, ++ IA32_MSR_MTRR_FIX4K_C8000, ++ IA32_MSR_MTRR_FIX4K_D0000, ++ IA32_MSR_MTRR_FIX4K_D8000, ++ IA32_MSR_MTRR_FIX4K_E0000, ++ IA32_MSR_MTRR_FIX4K_E8000, ++ IA32_MSR_MTRR_FIX4K_F0000, ++ IA32_MSR_MTRR_FIX4K_F8000, ++ IA32_MSR_TSC_AUX, ++ IA32_MSR_DEBUG_CTL, ++ HV_X64_MSR_GUEST_OS_ID, ++ HV_X64_MSR_SINT0, ++ HV_X64_MSR_SINT1, ++ HV_X64_MSR_SINT2, ++ HV_X64_MSR_SINT3, ++ HV_X64_MSR_SINT4, ++ HV_X64_MSR_SINT5, ++ HV_X64_MSR_SINT6, ++ HV_X64_MSR_SINT7, ++ HV_X64_MSR_SINT8, ++ HV_X64_MSR_SINT9, ++ HV_X64_MSR_SINT10, ++ HV_X64_MSR_SINT11, ++ HV_X64_MSR_SINT12, ++ HV_X64_MSR_SINT13, ++ HV_X64_MSR_SINT14, ++ HV_X64_MSR_SINT15, ++ HV_X64_MSR_SCONTROL, ++ HV_X64_MSR_SIEFP, ++ HV_X64_MSR_SIMP, ++ HV_X64_MSR_REFERENCE_TSC, ++ HV_X64_MSR_EOM, ++}; ++static const size_t msr_count = ARRAY_SIZE(supported_msrs); ++ ++static int compare_msr_index(const void *a, const void *b) ++{ ++ return *(uint32_t *)a - *(uint32_t *)b; ++} ++ ++__attribute__((constructor)) ++static void init_sorted_msr_map(void) ++{ ++ qsort(supported_msrs, msr_count, sizeof(uint32_t), compare_msr_index); ++} ++ ++static int mshv_is_supported_msr(uint32_t msr) ++{ ++ return bsearch(&msr, supported_msrs, msr_count, sizeof(uint32_t), ++ compare_msr_index) != NULL; ++} ++ ++static int mshv_msr_to_hv_reg_name(uint32_t msr, uint32_t *hv_reg) ++{ ++ switch (msr) { ++ case IA32_MSR_TSC: ++ *hv_reg = HV_X64_REGISTER_TSC; ++ return 0; ++ case IA32_MSR_EFER: ++ *hv_reg = HV_X64_REGISTER_EFER; ++ return 0; ++ case IA32_MSR_KERNEL_GS_BASE: ++ *hv_reg = HV_X64_REGISTER_KERNEL_GS_BASE; ++ return 0; ++ case IA32_MSR_APIC_BASE: ++ *hv_reg = HV_X64_REGISTER_APIC_BASE; ++ return 0; ++ case IA32_MSR_PAT: ++ *hv_reg = HV_X64_REGISTER_PAT; ++ return 0; ++ case IA32_MSR_SYSENTER_CS: ++ *hv_reg = HV_X64_REGISTER_SYSENTER_CS; ++ return 0; ++ case IA32_MSR_SYSENTER_ESP: ++ *hv_reg = HV_X64_REGISTER_SYSENTER_ESP; ++ return 0; ++ case IA32_MSR_SYSENTER_EIP: ++ *hv_reg = HV_X64_REGISTER_SYSENTER_EIP; ++ return 0; ++ case IA32_MSR_STAR: ++ *hv_reg = HV_X64_REGISTER_STAR; ++ return 0; ++ case IA32_MSR_LSTAR: ++ *hv_reg = HV_X64_REGISTER_LSTAR; ++ return 0; ++ case IA32_MSR_CSTAR: ++ *hv_reg = HV_X64_REGISTER_CSTAR; ++ return 0; ++ case IA32_MSR_SFMASK: ++ *hv_reg = HV_X64_REGISTER_SFMASK; ++ return 0; ++ case IA32_MSR_MTRR_CAP: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_CAP; ++ return 0; ++ case IA32_MSR_MTRR_DEF_TYPE: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_DEF_TYPE; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE0: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE0; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK0: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK0; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE1: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE1; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK1: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK1; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE2: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE2; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK2: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK2; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE3: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE3; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK3: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK3; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE4: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE4; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK4: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK4; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE5: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE5; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK5: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK5; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE6: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE6; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK6: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK6; ++ return 0; ++ case IA32_MSR_MTRR_PHYSBASE7: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_BASE7; ++ return 0; ++ case IA32_MSR_MTRR_PHYSMASK7: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_PHYS_MASK7; ++ return 0; ++ case IA32_MSR_MTRR_FIX64K_00000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX64K00000; ++ return 0; ++ case IA32_MSR_MTRR_FIX16K_80000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX16K80000; ++ return 0; ++ case IA32_MSR_MTRR_FIX16K_A0000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX16KA0000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_C0000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KC0000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_C8000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KC8000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_D0000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KD0000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_D8000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KD8000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_E0000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KE0000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_E8000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KE8000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_F0000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KF0000; ++ return 0; ++ case IA32_MSR_MTRR_FIX4K_F8000: ++ *hv_reg = HV_X64_REGISTER_MSR_MTRR_FIX4KF8000; ++ return 0; ++ case IA32_MSR_TSC_AUX: ++ *hv_reg = HV_X64_REGISTER_TSC_AUX; ++ return 0; ++ case IA32_MSR_BNDCFGS: ++ *hv_reg = HV_X64_REGISTER_BNDCFGS; ++ return 0; ++ case IA32_MSR_DEBUG_CTL: ++ *hv_reg = HV_X64_REGISTER_DEBUG_CTL; ++ return 0; ++ case IA32_MSR_TSC_ADJUST: ++ *hv_reg = HV_X64_REGISTER_TSC_ADJUST; ++ return 0; ++ case IA32_MSR_SPEC_CTRL: ++ *hv_reg = HV_X64_REGISTER_SPEC_CTRL; ++ return 0; ++ case HV_X64_MSR_GUEST_OS_ID: ++ *hv_reg = HV_REGISTER_GUEST_OS_ID; ++ return 0; ++ case HV_X64_MSR_SINT0: ++ *hv_reg = HV_REGISTER_SINT0; ++ return 0; ++ case HV_X64_MSR_SINT1: ++ *hv_reg = HV_REGISTER_SINT1; ++ return 0; ++ case HV_X64_MSR_SINT2: ++ *hv_reg = HV_REGISTER_SINT2; ++ return 0; ++ case HV_X64_MSR_SINT3: ++ *hv_reg = HV_REGISTER_SINT3; ++ return 0; ++ case HV_X64_MSR_SINT4: ++ *hv_reg = HV_REGISTER_SINT4; ++ return 0; ++ case HV_X64_MSR_SINT5: ++ *hv_reg = HV_REGISTER_SINT5; ++ return 0; ++ case HV_X64_MSR_SINT6: ++ *hv_reg = HV_REGISTER_SINT6; ++ return 0; ++ case HV_X64_MSR_SINT7: ++ *hv_reg = HV_REGISTER_SINT7; ++ return 0; ++ case HV_X64_MSR_SINT8: ++ *hv_reg = HV_REGISTER_SINT8; ++ return 0; ++ case HV_X64_MSR_SINT9: ++ *hv_reg = HV_REGISTER_SINT9; ++ return 0; ++ case HV_X64_MSR_SINT10: ++ *hv_reg = HV_REGISTER_SINT10; ++ return 0; ++ case HV_X64_MSR_SINT11: ++ *hv_reg = HV_REGISTER_SINT11; ++ return 0; ++ case HV_X64_MSR_SINT12: ++ *hv_reg = HV_REGISTER_SINT12; ++ return 0; ++ case HV_X64_MSR_SINT13: ++ *hv_reg = HV_REGISTER_SINT13; ++ return 0; ++ case HV_X64_MSR_SINT14: ++ *hv_reg = HV_REGISTER_SINT14; ++ return 0; ++ case HV_X64_MSR_SINT15: ++ *hv_reg = HV_REGISTER_SINT15; ++ return 0; ++ case IA32_MSR_MISC_ENABLE: ++ *hv_reg = HV_X64_REGISTER_MSR_IA32_MISC_ENABLE; ++ return 0; ++ case HV_X64_MSR_SCONTROL: ++ *hv_reg = HV_REGISTER_SCONTROL; ++ return 0; ++ case HV_X64_MSR_SIEFP: ++ *hv_reg = HV_REGISTER_SIEFP; ++ return 0; ++ case HV_X64_MSR_SIMP: ++ *hv_reg = HV_REGISTER_SIMP; ++ return 0; ++ case HV_X64_MSR_REFERENCE_TSC: ++ *hv_reg = HV_REGISTER_REFERENCE_TSC; ++ return 0; ++ case HV_X64_MSR_EOM: ++ *hv_reg = HV_REGISTER_EOM; ++ return 0; ++ default: ++ error_report("failed to map MSR %u to HV register name", msr); ++ return -1; ++ } ++} ++ ++static int set_msrs(const CPUState *cpu, GList *msrs) ++{ ++ size_t n_msrs; ++ GList *entries; ++ MshvMsrEntry *entry; ++ enum hv_register_name name; ++ struct hv_register_assoc *assoc; ++ int ret; ++ size_t i = 0; ++ ++ n_msrs = g_list_length(msrs); ++ hv_register_assoc *assocs = g_new0(hv_register_assoc, n_msrs); ++ ++ entries = msrs; ++ for (const GList *elem = entries; elem != NULL; elem = elem->next) { ++ entry = elem->data; ++ ret = mshv_msr_to_hv_reg_name(entry->index, &name); ++ if (ret < 0) { ++ g_free(assocs); ++ return ret; ++ } ++ assoc = &assocs[i]; ++ assoc->name = name; ++ /* the union has been initialized to 0 */ ++ assoc->value.reg64 = entry->data; ++ i++; ++ } ++ ret = mshv_set_generic_regs(cpu, assocs, n_msrs); ++ g_free(assocs); ++ if (ret < 0) { ++ error_report("failed to set msrs"); ++ return -1; ++ } ++ return 0; ++} ++ ++ ++int mshv_configure_msr(const CPUState *cpu, const MshvMsrEntry *msrs, ++ size_t n_msrs) ++{ ++ GList *valid_msrs = NULL; ++ uint32_t msr_index; ++ int ret; ++ ++ for (size_t i = 0; i < n_msrs; i++) { ++ msr_index = msrs[i].index; ++ /* check whether index of msrs is in SUPPORTED_MSRS */ ++ if (mshv_is_supported_msr(msr_index)) { ++ valid_msrs = g_list_append(valid_msrs, (void *) &msrs[i]); ++ } ++ } ++ ++ ret = set_msrs(cpu, valid_msrs); ++ g_list_free(valid_msrs); ++ ++ return ret; ++} +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index 0ea8d504fa..6649438313 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -14,6 +14,8 @@ + #ifndef QEMU_MSHV_INT_H + #define QEMU_MSHV_INT_H + ++#define MSHV_MSR_ENTRIES_COUNT 64 ++ + typedef struct hyperv_message hv_message; + + struct AccelCPUState { +@@ -102,6 +104,21 @@ typedef struct MshvMemoryRegion { + void mshv_set_phys_mem(MshvMemoryListener *mml, MemoryRegionSection *section, + bool add); + ++/* msr */ ++typedef struct MshvMsrEntry { ++ uint32_t index; ++ uint32_t reserved; ++ uint64_t data; ++} MshvMsrEntry; ++ ++typedef struct MshvMsrEntries { ++ MshvMsrEntry entries[MSHV_MSR_ENTRIES_COUNT]; ++ uint32_t nmsrs; ++} MshvMsrEntries; ++ ++int mshv_configure_msr(const CPUState *cpu, const MshvMsrEntry *msrs, ++ size_t n_msrs); ++ + /* interrupt */ + void mshv_init_msicontrol(void); + int mshv_reserve_ioapic_msi_routes(int vm_fd); +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 4b7eae43e1..8f780a3abe 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -435,9 +435,11 @@ typedef enum X86Seg { + #define MSR_SMI_COUNT 0x34 + #define MSR_CORE_THREAD_COUNT 0x35 + #define MSR_MTRRcap 0xfe ++#define MSR_MTRR_MEM_TYPE_WB 0x06 + #define MSR_MTRRcap_VCNT 8 + #define MSR_MTRRcap_FIXRANGE_SUPPORT (1 << 8) + #define MSR_MTRRcap_WC_SUPPORTED (1 << 10) ++#define MSR_MTRR_ENABLE (1 << 11) + + #define MSR_IA32_SYSENTER_CS 0x174 + #define MSR_IA32_SYSENTER_ESP 0x175 +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 2b7a81274b..1f43dfc58a 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -872,6 +872,33 @@ static int set_lint(int cpu_fd) + return set_lapic(cpu_fd, &lapic_state); + } + ++static int setup_msrs(const CPUState *cpu) ++{ ++ int ret; ++ uint64_t default_type = MSR_MTRR_ENABLE | MSR_MTRR_MEM_TYPE_WB; ++ ++ /* boot msr entries */ ++ MshvMsrEntry msrs[9] = { ++ { .index = IA32_MSR_SYSENTER_CS, .data = 0x0, }, ++ { .index = IA32_MSR_SYSENTER_ESP, .data = 0x0, }, ++ { .index = IA32_MSR_SYSENTER_EIP, .data = 0x0, }, ++ { .index = IA32_MSR_STAR, .data = 0x0, }, ++ { .index = IA32_MSR_CSTAR, .data = 0x0, }, ++ { .index = IA32_MSR_LSTAR, .data = 0x0, }, ++ { .index = IA32_MSR_KERNEL_GS_BASE, .data = 0x0, }, ++ { .index = IA32_MSR_SFMASK, .data = 0x0, }, ++ { .index = IA32_MSR_MTRR_DEF_TYPE, .data = default_type, }, ++ }; ++ ++ ret = mshv_configure_msr(cpu, msrs, 9); ++ if (ret < 0) { ++ error_report("failed to setup msrs"); ++ return -1; ++ } ++ ++ return 0; ++} ++ + /* + * TODO: populate topology info: + * +@@ -891,6 +918,12 @@ int mshv_configure_vcpu(const CPUState *cpu, const struct MshvFPU *fpu, + return -1; + } + ++ ret = setup_msrs(cpu); ++ if (ret < 0) { ++ error_report("failed to setup msrs"); ++ return -1; ++ } ++ + ret = set_cpu_state(cpu, fpu, xcr0); + if (ret < 0) { + error_report("failed to set cpu state"); +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Set-local-interrupt-controller-stat.patch b/kvm-target-i386-mshv-Set-local-interrupt-controller-stat.patch new file mode 100644 index 0000000..0a0ef11 --- /dev/null +++ b/kvm-target-i386-mshv-Set-local-interrupt-controller-stat.patch @@ -0,0 +1,183 @@ +From 140f8cbaa8665f281fe3e47c81cfd0c306f02d8b Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:37 +0200 +Subject: [PATCH 20/32] target/i386/mshv: Set local interrupt controller state + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [18/30] ded044d2fd9c55a149d7e71d952c10ed6ea29505 + +To set the local interrupt controller state, perform hv calls retrieving +partition state from the hypervisor. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-18-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit ca20d46fa94887a6e42061646a71ee44cbc9adb0) +Signed-off-by: Igor Mammedov +--- + target/i386/mshv/mshv-cpu.c | 117 ++++++++++++++++++++++++++++++++++++ + 1 file changed, 117 insertions(+) + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 8b10c79e54..0fe3cbb48d 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -12,6 +12,7 @@ + + #include "qemu/osdep.h" + #include "qemu/error-report.h" ++#include "qemu/memalign.h" + #include "qemu/typedefs.h" + + #include "system/mshv.h" +@@ -21,6 +22,7 @@ + #include "hw/hyperv/hvgdk.h" + #include "hw/hyperv/hvgdk_mini.h" + #include "hw/hyperv/hvhdk_mini.h" ++#include "hw/i386/apic_internal.h" + + #include "cpu.h" + #include "emulate/x86_decode.h" +@@ -562,6 +564,114 @@ static int set_cpu_state(const CPUState *cpu, const MshvFPU *fpu_regs, + return 0; + } + ++static int get_vp_state(int cpu_fd, struct mshv_get_set_vp_state *state) ++{ ++ int ret; ++ ++ ret = ioctl(cpu_fd, MSHV_GET_VP_STATE, state); ++ if (ret < 0) { ++ error_report("failed to get partition state: %s", strerror(errno)); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int get_lapic(int cpu_fd, ++ struct hv_local_interrupt_controller_state *state) ++{ ++ int ret; ++ size_t size = 4096; ++ /* buffer aligned to 4k, as *state requires that */ ++ void *buffer = qemu_memalign(size, size); ++ struct mshv_get_set_vp_state mshv_state = { 0 }; ++ ++ mshv_state.buf_ptr = (uint64_t) buffer; ++ mshv_state.buf_sz = size; ++ mshv_state.type = MSHV_VP_STATE_LAPIC; ++ ++ ret = get_vp_state(cpu_fd, &mshv_state); ++ if (ret == 0) { ++ memcpy(state, buffer, sizeof(*state)); ++ } ++ qemu_vfree(buffer); ++ if (ret < 0) { ++ error_report("failed to get lapic"); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static uint32_t set_apic_delivery_mode(uint32_t reg, uint32_t mode) ++{ ++ return ((reg) & ~0x700) | ((mode) << 8); ++} ++ ++static int set_vp_state(int cpu_fd, const struct mshv_get_set_vp_state *state) ++{ ++ int ret; ++ ++ ret = ioctl(cpu_fd, MSHV_SET_VP_STATE, state); ++ if (ret < 0) { ++ error_report("failed to set partition state: %s", strerror(errno)); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int set_lapic(int cpu_fd, ++ const struct hv_local_interrupt_controller_state *state) ++{ ++ int ret; ++ size_t size = 4096; ++ /* buffer aligned to 4k, as *state requires that */ ++ void *buffer = qemu_memalign(size, size); ++ struct mshv_get_set_vp_state mshv_state = { 0 }; ++ ++ if (!state) { ++ error_report("lapic state is NULL"); ++ return -1; ++ } ++ memcpy(buffer, state, sizeof(*state)); ++ ++ mshv_state.buf_ptr = (uint64_t) buffer; ++ mshv_state.buf_sz = size; ++ mshv_state.type = MSHV_VP_STATE_LAPIC; ++ ++ ret = set_vp_state(cpu_fd, &mshv_state); ++ qemu_vfree(buffer); ++ if (ret < 0) { ++ error_report("failed to set lapic: %s", strerror(errno)); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int set_lint(int cpu_fd) ++{ ++ int ret; ++ uint32_t *lvt_lint0, *lvt_lint1; ++ ++ struct hv_local_interrupt_controller_state lapic_state = { 0 }; ++ ret = get_lapic(cpu_fd, &lapic_state); ++ if (ret < 0) { ++ return ret; ++ } ++ ++ lvt_lint0 = &lapic_state.apic_lvt_lint0; ++ *lvt_lint0 = set_apic_delivery_mode(*lvt_lint0, APIC_DM_EXTINT); ++ ++ lvt_lint1 = &lapic_state.apic_lvt_lint1; ++ *lvt_lint1 = set_apic_delivery_mode(*lvt_lint1, APIC_DM_NMI); ++ ++ /* TODO: should we skip setting lapic if the values are the same? */ ++ ++ return set_lapic(cpu_fd, &lapic_state); ++} ++ + /* + * TODO: populate topology info: + * +@@ -573,6 +683,7 @@ int mshv_configure_vcpu(const CPUState *cpu, const struct MshvFPU *fpu, + uint64_t xcr0) + { + int ret; ++ int cpu_fd = mshv_vcpufd(cpu); + + ret = set_cpu_state(cpu, fpu, xcr0); + if (ret < 0) { +@@ -580,6 +691,12 @@ int mshv_configure_vcpu(const CPUState *cpu, const struct MshvFPU *fpu, + return -1; + } + ++ ret = set_lint(cpu_fd); ++ if (ret < 0) { ++ error_report("failed to set lpic int"); ++ return -1; ++ } ++ + return 0; + } + +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Use-preallocated-page-for-hvcall.patch b/kvm-target-i386-mshv-Use-preallocated-page-for-hvcall.patch new file mode 100644 index 0000000..259c3b4 --- /dev/null +++ b/kvm-target-i386-mshv-Use-preallocated-page-for-hvcall.patch @@ -0,0 +1,193 @@ +From 553c58c0e1560ee83e90b1751ec79874a0a7addf Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Thu, 2 Oct 2025 09:50:12 +0200 +Subject: [PATCH 28/32] target/i386/mshv: Use preallocated page for hvcall + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [26/30] 338fa20beaaa44e30512d49026d2a9e9dd016b53 + +There are hvcalls that are invoked during MMIO exits, the payload is of +dynamic size. To avoid heap allocations we can use preallocated pages as +in/out buffer for those calls. A page is reserved per vCPU and used for +set/get register hv calls. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-26-magnuskulke@linux.microsoft.com +[Use standard MAX_CONST macro; mshv.h/mshv_int.h split. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit e4a20afce59937073298d716cc829bdc026542dc) +Signed-off-by: Igor Mammedov +--- + accel/mshv/mshv-all.c | 2 +- + include/system/mshv_int.h | 7 +++++++ + target/i386/mshv/mshv-cpu.c | 38 +++++++++++++++++++++++++------------ + 3 files changed, 34 insertions(+), 13 deletions(-) + +diff --git a/accel/mshv/mshv-all.c b/accel/mshv/mshv-all.c +index 5edfcbad9d..45174f7c4e 100644 +--- a/accel/mshv/mshv-all.c ++++ b/accel/mshv/mshv-all.c +@@ -399,8 +399,8 @@ static int mshv_init_vcpu(CPUState *cpu) + uint8_t vp_index = cpu->cpu_index; + int ret; + +- mshv_arch_init_vcpu(cpu); + cpu->accel = g_new0(AccelCPUState, 1); ++ mshv_arch_init_vcpu(cpu); + + ret = mshv_create_vcpu(vm_fd, vp_index, &cpu->accel->cpufd); + if (ret < 0) { +diff --git a/include/system/mshv_int.h b/include/system/mshv_int.h +index 6350c69e9d..490563c1ab 100644 +--- a/include/system/mshv_int.h ++++ b/include/system/mshv_int.h +@@ -20,9 +20,16 @@ + + typedef struct hyperv_message hv_message; + ++typedef struct MshvHvCallArgs { ++ void *base; ++ void *input_page; ++ void *output_page; ++} MshvHvCallArgs; ++ + struct AccelCPUState { + int cpufd; + bool dirty; ++ MshvHvCallArgs hvcall_args; + }; + + typedef struct MshvMemoryListener { +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index de87142bff..1f7b9cb37e 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -34,6 +34,10 @@ + + #include + ++#define MAX_REGISTER_COUNT (MAX_CONST(ARRAY_SIZE(STANDARD_REGISTER_NAMES), \ ++ MAX_CONST(ARRAY_SIZE(SPECIAL_REGISTER_NAMES), \ ++ ARRAY_SIZE(FPU_REGISTER_NAMES)))) ++ + static enum hv_register_name STANDARD_REGISTER_NAMES[18] = { + HV_X64_REGISTER_RAX, + HV_X64_REGISTER_RBX, +@@ -151,7 +155,7 @@ int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, + int cpu_fd = mshv_vcpufd(cpu); + int vp_index = cpu->cpu_index; + size_t in_sz, assocs_sz; +- hv_input_set_vp_registers *in; ++ hv_input_set_vp_registers *in = cpu->accel->hvcall_args.input_page; + struct mshv_root_hvcall args = {0}; + int ret; + +@@ -160,7 +164,7 @@ int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, + in_sz = sizeof(hv_input_set_vp_registers) + assocs_sz; + + /* fill the input struct */ +- in = g_malloc0(in_sz); ++ memset(in, 0, sizeof(hv_input_set_vp_registers)); + in->vp_index = vp_index; + memcpy(in->elements, assocs, assocs_sz); + +@@ -172,7 +176,6 @@ int mshv_set_generic_regs(const CPUState *cpu, const hv_register_assoc *assocs, + + /* perform the call */ + ret = mshv_hvcall(cpu_fd, &args); +- g_free(in); + if (ret < 0) { + error_report("Failed to set registers"); + return -1; +@@ -193,8 +196,8 @@ static int get_generic_regs(CPUState *cpu, hv_register_assoc *assocs, + { + int cpu_fd = mshv_vcpufd(cpu); + int vp_index = cpu->cpu_index; +- hv_input_get_vp_registers *in; +- hv_register_value *values; ++ hv_input_get_vp_registers *in = cpu->accel->hvcall_args.input_page; ++ hv_register_value *values = cpu->accel->hvcall_args.output_page; + size_t in_sz, names_sz, values_sz; + int i, ret; + struct mshv_root_hvcall args = {0}; +@@ -204,15 +207,14 @@ static int get_generic_regs(CPUState *cpu, hv_register_assoc *assocs, + in_sz = sizeof(hv_input_get_vp_registers) + names_sz; + + /* fill the input struct */ +- in = g_malloc0(in_sz); ++ memset(in, 0, sizeof(hv_input_get_vp_registers)); + in->vp_index = vp_index; + for (i = 0; i < n_regs; i++) { + in->names[i] = assocs[i].name; + } + +- /* allocate value output buffer */ ++ /* determine size of value output buffer */ + values_sz = n_regs * sizeof(union hv_register_value); +- values = g_malloc0(values_sz); + + /* create the hvcall envelope */ + args.code = HVCALL_GET_VP_REGISTERS; +@@ -224,16 +226,13 @@ static int get_generic_regs(CPUState *cpu, hv_register_assoc *assocs, + + /* perform the call */ + ret = mshv_hvcall(cpu_fd, &args); +- g_free(in); + if (ret < 0) { +- g_free(values); + error_report("Failed to retrieve registers"); + return -1; + } + + /* assert we got all registers */ + if (args.reps != n_regs) { +- g_free(values); + error_report("Failed to retrieve registers: expected %zu elements" + ", got %u", n_regs, args.reps); + return -1; +@@ -243,7 +242,6 @@ static int get_generic_regs(CPUState *cpu, hv_register_assoc *assocs, + for (i = 0; i < n_regs; i++) { + assocs[i].value = values[i]; + } +- g_free(values); + + return 0; + } +@@ -1696,6 +1694,19 @@ void mshv_arch_init_vcpu(CPUState *cpu) + { + X86CPU *x86_cpu = X86_CPU(cpu); + CPUX86State *env = &x86_cpu->env; ++ AccelCPUState *state = cpu->accel; ++ size_t page = HV_HYP_PAGE_SIZE; ++ void *mem = qemu_memalign(page, 2 * page); ++ ++ /* sanity check, to make sure we don't overflow the page */ ++ QEMU_BUILD_BUG_ON((MAX_REGISTER_COUNT ++ * sizeof(hv_register_assoc) ++ + sizeof(hv_input_get_vp_registers) ++ > HV_HYP_PAGE_SIZE)); ++ ++ state->hvcall_args.base = mem; ++ state->hvcall_args.input_page = mem; ++ state->hvcall_args.output_page = (uint8_t *)mem + page; + + env->emu_mmio_buf = g_new(char, 4096); + } +@@ -1704,7 +1715,10 @@ void mshv_arch_destroy_vcpu(CPUState *cpu) + { + X86CPU *x86_cpu = X86_CPU(cpu); + CPUX86State *env = &x86_cpu->env; ++ AccelCPUState *state = cpu->accel; + ++ g_free(state->hvcall_args.base); ++ state->hvcall_args = (MshvHvCallArgs){0}; + g_clear_pointer(&env->emu_mmio_buf, g_free); + } + +-- +2.47.3 + diff --git a/kvm-target-i386-mshv-Write-MSRs-to-the-hypervisor.patch b/kvm-target-i386-mshv-Write-MSRs-to-the-hypervisor.patch new file mode 100644 index 0000000..6ec7aa4 --- /dev/null +++ b/kvm-target-i386-mshv-Write-MSRs-to-the-hypervisor.patch @@ -0,0 +1,113 @@ +From 07b03c6440308cb6294ee2d2e4518df052b2a363 Mon Sep 17 00:00:00 2001 +From: Magnus Kulke +Date: Tue, 16 Sep 2025 18:48:41 +0200 +Subject: [PATCH 24/32] target/i386/mshv: Write MSRs to the hypervisor + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [22/30] 8b9720dbc31e48c43e618a0650a21eb5a43b7954 + +Push current model-specific register (MSR) values to MSHV's vCPUs as +part of setting state to the hypervisor. + +Signed-off-by: Magnus Kulke +Link: https://lore.kernel.org/r/20250916164847.77883-22-magnuskulke@linux.microsoft.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 64118f452cbd97cd9fa790b1c15b65b435f136d2) +Signed-off-by: Igor Mammedov +--- + target/i386/mshv/mshv-cpu.c | 68 +++++++++++++++++++++++++++++++++++-- + 1 file changed, 66 insertions(+), 2 deletions(-) + +diff --git a/target/i386/mshv/mshv-cpu.c b/target/i386/mshv/mshv-cpu.c +index 424ebdb122..33a3ce8b11 100644 +--- a/target/i386/mshv/mshv-cpu.c ++++ b/target/i386/mshv/mshv-cpu.c +@@ -998,6 +998,65 @@ static int put_regs(const CPUState *cpu) + return 0; + } + ++struct MsrPair { ++ uint32_t index; ++ uint64_t value; ++}; ++ ++static int put_msrs(const CPUState *cpu) ++{ ++ int ret = 0; ++ X86CPU *x86cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86cpu->env; ++ MshvMsrEntries *msrs = g_malloc0(sizeof(MshvMsrEntries)); ++ ++ struct MsrPair pairs[] = { ++ { MSR_IA32_SYSENTER_CS, env->sysenter_cs }, ++ { MSR_IA32_SYSENTER_ESP, env->sysenter_esp }, ++ { MSR_IA32_SYSENTER_EIP, env->sysenter_eip }, ++ { MSR_EFER, env->efer }, ++ { MSR_PAT, env->pat }, ++ { MSR_STAR, env->star }, ++ { MSR_CSTAR, env->cstar }, ++ { MSR_LSTAR, env->lstar }, ++ { MSR_KERNELGSBASE, env->kernelgsbase }, ++ { MSR_FMASK, env->fmask }, ++ { MSR_MTRRdefType, env->mtrr_deftype }, ++ { MSR_VM_HSAVE_PA, env->vm_hsave }, ++ { MSR_SMI_COUNT, env->msr_smi_count }, ++ { MSR_IA32_PKRS, env->pkrs }, ++ { MSR_IA32_BNDCFGS, env->msr_bndcfgs }, ++ { MSR_IA32_XSS, env->xss }, ++ { MSR_IA32_UMWAIT_CONTROL, env->umwait }, ++ { MSR_IA32_TSX_CTRL, env->tsx_ctrl }, ++ { MSR_AMD64_TSC_RATIO, env->amd_tsc_scale_msr }, ++ { MSR_TSC_AUX, env->tsc_aux }, ++ { MSR_TSC_ADJUST, env->tsc_adjust }, ++ { MSR_IA32_SMBASE, env->smbase }, ++ { MSR_IA32_SPEC_CTRL, env->spec_ctrl }, ++ { MSR_VIRT_SSBD, env->virt_ssbd }, ++ }; ++ ++ if (ARRAY_SIZE(pairs) > MSHV_MSR_ENTRIES_COUNT) { ++ error_report("MSR entries exceed maximum size"); ++ g_free(msrs); ++ return -1; ++ } ++ ++ for (size_t i = 0; i < ARRAY_SIZE(pairs); i++) { ++ MshvMsrEntry *entry = &msrs->entries[i]; ++ entry->index = pairs[i].index; ++ entry->reserved = 0; ++ entry->data = pairs[i].value; ++ msrs->nmsrs++; ++ } ++ ++ ret = mshv_configure_msr(cpu, &msrs->entries[0], msrs->nmsrs); ++ g_free(msrs); ++ return ret; ++} ++ ++ + int mshv_arch_put_registers(const CPUState *cpu) + { + int ret; +@@ -1008,8 +1067,13 @@ int mshv_arch_put_registers(const CPUState *cpu) + return -1; + } + +- error_report("unimplemented"); +- abort(); ++ ret = put_msrs(cpu); ++ if (ret < 0) { ++ error_report("Failed to put msrs"); ++ return -1; ++ } ++ ++ return 0; + } + + void mshv_arch_amend_proc_features( +-- +2.47.3 + diff --git a/kvm-target-s390x-Add-a-CONFIG-switch-to-disable-legacy-C.patch b/kvm-target-s390x-Add-a-CONFIG-switch-to-disable-legacy-C.patch deleted file mode 100644 index 29bf5d7..0000000 --- a/kvm-target-s390x-Add-a-CONFIG-switch-to-disable-legacy-C.patch +++ /dev/null @@ -1,116 +0,0 @@ -From d0f88c7a0c95b4d9ab03221400736cb17cb4b995 Mon Sep 17 00:00:00 2001 -From: Thomas Huth -Date: Thu, 13 Jun 2024 16:14:22 +0200 -Subject: [PATCH 10/14] target/s390x: Add a CONFIG switch to disable legacy - CPUs -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Thomas Huth -RH-MergeRequest: 252: s390x: remove legacy CPU types -RH-Jira: RHEL-39898 -RH-Acked-by: Cédric Le Goater -RH-Acked-by: Miroslav Rezanina -RH-Commit: [3/5] f8e78c8e0349c8645e7df7b0bebed1635865b454 (thuth/qemu-kvm-cs9) - -The oldest model that IBM still supports is the z13. Considering -that each generation can "emulate" the previous two generations -in hardware (via the "IBC" feature of the CPUs), this means that -everything that is older than z114/196 is not an officially supported -CPU model anymore. The Linux kernel still support the z10, so if -we also take this into account, everything older than that can -definitely be considered as a legacy CPU model. - -For downstream builds of QEMU, we would like to be able to disable -these legacy CPUs in the build. Thus add a CONFIG switch that can be -used to disable them (and old machine types that use them by default). - -Message-Id: <20240614125019.588928-1-thuth@redhat.com> -Signed-off-by: Thomas Huth -(cherry picked from commit d6a7c3f44cf3f60c066dbf087ef79d4b12acc642) ---- - hw/s390x/s390-virtio-ccw.c | 4 ++++ - target/s390x/Kconfig | 5 +++++ - target/s390x/cpu_models.c | 9 +++++++++ - 3 files changed, 18 insertions(+) - -diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c -index b0b903b78c..527b05d1d6 100644 ---- a/hw/s390x/s390-virtio-ccw.c -+++ b/hw/s390x/s390-virtio-ccw.c -@@ -46,6 +46,7 @@ - #include "migration/blocker.h" - #include "qapi/visitor.h" - #include "hw/s390x/cpu-topology.h" -+#include CONFIG_DEVICES - - static Error *pv_mig_blocker; - -@@ -1130,6 +1131,8 @@ static void ccw_machine_2_12_class_options(MachineClass *mc) - } - DEFINE_CCW_MACHINE(2_12, "2.12", false); - -+#ifdef CONFIG_S390X_LEGACY_CPUS -+ - static void ccw_machine_2_11_instance_options(MachineState *machine) - { - static const S390FeatInit qemu_cpu_feat = { S390_FEAT_LIST_QEMU_V2_11 }; -@@ -1277,6 +1280,7 @@ static void ccw_machine_2_4_class_options(MachineClass *mc) - DEFINE_CCW_MACHINE(2_4, "2.4", false); - #endif - -+#endif - - static void ccw_machine_rhel940_instance_options(MachineState *machine) - { -diff --git a/target/s390x/Kconfig b/target/s390x/Kconfig -index d886be48b4..8a95f2bc3f 100644 ---- a/target/s390x/Kconfig -+++ b/target/s390x/Kconfig -@@ -2,3 +2,8 @@ config S390X - bool - select PCI - select S390_FLIC -+ -+config S390X_LEGACY_CPUS -+ bool -+ default y -+ depends on S390X -diff --git a/target/s390x/cpu_models.c b/target/s390x/cpu_models.c -index 370b3b3065..f4dbcc67bb 100644 ---- a/target/s390x/cpu_models.c -+++ b/target/s390x/cpu_models.c -@@ -25,6 +25,7 @@ - #ifndef CONFIG_USER_ONLY - #include "sysemu/sysemu.h" - #include "target/s390x/kvm/pv.h" -+#include CONFIG_DEVICES - #endif - - #define CPUDEF_INIT(_type, _gen, _ec_ga, _mha_pow, _hmfai, _name, _desc) \ -@@ -50,6 +51,13 @@ - #define RHEL_CPU_DEPRECATION "use at least 'z14', or 'host' / 'qemu' / 'max'" - - static S390CPUDef s390_cpu_defs[] = { -+ /* -+ * Linux requires at least z10 nowadays, and IBM only supports recent CPUs -+ * (see https://www.ibm.com/support/pages/ibm-mainframe-life-cycle-history), -+ * so we consider older CPUs as legacy that can optionally be disabled via -+ * the CONFIG_S390X_LEGACY_CPUS config switch. -+ */ -+#if defined(CONFIG_S390X_LEGACY_CPUS) || defined(CONFIG_USER_ONLY) - CPUDEF_INIT(0x2064, 7, 1, 38, 0x00000000U, "z900", "IBM zSeries 900 GA1"), - CPUDEF_INIT(0x2064, 7, 2, 38, 0x00000000U, "z900.2", "IBM zSeries 900 GA2"), - CPUDEF_INIT(0x2064, 7, 3, 38, 0x00000000U, "z900.3", "IBM zSeries 900 GA3"), -@@ -67,6 +75,7 @@ static S390CPUDef s390_cpu_defs[] = { - CPUDEF_INIT(0x2096, 9, 2, 40, 0x00000000U, "z9BC", "IBM System z9 BC GA1"), - CPUDEF_INIT(0x2094, 9, 3, 40, 0x00000000U, "z9EC.3", "IBM System z9 EC GA3"), - CPUDEF_INIT(0x2096, 9, 3, 40, 0x00000000U, "z9BC.2", "IBM System z9 BC GA2"), -+#endif - CPUDEF_INIT(0x2097, 10, 1, 43, 0x00000000U, "z10EC", "IBM System z10 EC GA1"), - CPUDEF_INIT(0x2097, 10, 2, 43, 0x00000000U, "z10EC.2", "IBM System z10 EC GA2"), - CPUDEF_INIT(0x2098, 10, 2, 43, 0x00000000U, "z10BC", "IBM System z10 BC GA1"), --- -2.39.3 - diff --git a/kvm-target-s390x-Revert-the-old-s390x-CPU-model-disablem.patch b/kvm-target-s390x-Revert-the-old-s390x-CPU-model-disablem.patch deleted file mode 100644 index cf89fcb..0000000 --- a/kvm-target-s390x-Revert-the-old-s390x-CPU-model-disablem.patch +++ /dev/null @@ -1,66 +0,0 @@ -From 64eecc611dfdb9252b5e9d20b96cba715ecc1d07 Mon Sep 17 00:00:00 2001 -From: Thomas Huth -Date: Mon, 24 Jun 2024 14:26:14 +0200 -Subject: [PATCH 12/14] target/s390x: Revert the old s390x CPU model - disablement code -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Thomas Huth -RH-MergeRequest: 252: s390x: remove legacy CPU types -RH-Jira: RHEL-39898 -RH-Acked-by: Cédric Le Goater -RH-Acked-by: Miroslav Rezanina -RH-Commit: [5/5] da022e5acaeb1c86fba6245aa2c20491ac83046f (thuth/qemu-kvm-cs9) - -Upstream-Status: N/A - -We now completely disable the old CPU models up to the z12 in -target/s390x/cpu_models.c, so we don't need these old checks -anymore. - -This patch should get squashed into the downstream patch -"Enable/disable devices for RHEL" during the next rebase. - -Signed-off-by: Thomas Huth ---- - target/s390x/cpu_models_sysemu.c | 3 --- - target/s390x/kvm/kvm.c | 7 ------- - 2 files changed, 10 deletions(-) - -diff --git a/target/s390x/cpu_models_sysemu.c b/target/s390x/cpu_models_sysemu.c -index ca2e5d91e2..906d5d42b7 100644 ---- a/target/s390x/cpu_models_sysemu.c -+++ b/target/s390x/cpu_models_sysemu.c -@@ -34,9 +34,6 @@ static void check_unavailable_features(const S390CPUModel *max_model, - (max_model->def->gen == model->def->gen && - max_model->def->ec_ga < model->def->ec_ga)) { - list_add_feat("type", unavailable); -- } else if (model->def->gen < 11 && kvm_enabled()) { -- /* Older CPU models are not supported on Red Hat Enterprise Linux */ -- list_add_feat("type", unavailable); - } - - /* detect missing features if any to properly report them */ -diff --git a/target/s390x/kvm/kvm.c b/target/s390x/kvm/kvm.c -index 55fb4855b1..6dcb8dba2d 100644 ---- a/target/s390x/kvm/kvm.c -+++ b/target/s390x/kvm/kvm.c -@@ -2566,13 +2566,6 @@ void kvm_s390_apply_cpu_model(const S390CPUModel *model, Error **errp) - return; - } - -- /* Older CPU models are not supported on Red Hat Enterprise Linux */ -- if (model->def->gen < 11) { -- error_setg(errp, "KVM: Unsupported CPU type specified: %s", -- MACHINE(qdev_get_machine())->cpu_type); -- return; -- } -- - prop.cpuid = s390_cpuid_from_cpu_model(model); - prop.ibc = s390_ibc_from_cpu_model(model); - /* configure cpu features indicated via STFL(e) */ --- -2.39.3 - diff --git a/kvm-target-s390x-cpu_models-Disable-everything-up-to-the.patch b/kvm-target-s390x-cpu_models-Disable-everything-up-to-the.patch deleted file mode 100644 index 34e08fe..0000000 --- a/kvm-target-s390x-cpu_models-Disable-everything-up-to-the.patch +++ /dev/null @@ -1,56 +0,0 @@ -From 947ee045103e9148c80a1df0dc300fc840df2680 Mon Sep 17 00:00:00 2001 -From: Thomas Huth -Date: Mon, 24 Jun 2024 14:15:08 +0200 -Subject: [PATCH 11/14] target/s390x/cpu_models: Disable everything up to the - z12 CPU model -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Thomas Huth -RH-MergeRequest: 252: s390x: remove legacy CPU types -RH-Jira: RHEL-39898 -RH-Acked-by: Cédric Le Goater -RH-Acked-by: Miroslav Rezanina -RH-Commit: [4/5] f5236c8041bfcb63df4046f7bb0a12c1fa90062d (thuth/qemu-kvm-cs9) - -Upstream-Status: N/A -JIRA: https://issues.redhat.com/browse/RHEL-39898 - -When RHEL 10.0 gets released, the z14 will be the oldest mainframe -that is still officially supported by IBM, see: -https://www.ibm.com/support/pages/ibm-mainframe-life-cycle-history - -Now each IBM Z machine can "emulate" the previous two CPU types in -hardware for virtual guests, so we should still allow the z12 and -z13 in our qemu-kvm builds, too. But everything that is older than -the z12 can be disabled now. - -Signed-off-by: Thomas Huth ---- - target/s390x/cpu_models.c | 2 +- - 1 file changed, 1 insertion(+), 1 deletion(-) - -diff --git a/target/s390x/cpu_models.c b/target/s390x/cpu_models.c -index f4dbcc67bb..ad65149844 100644 ---- a/target/s390x/cpu_models.c -+++ b/target/s390x/cpu_models.c -@@ -75,7 +75,6 @@ static S390CPUDef s390_cpu_defs[] = { - CPUDEF_INIT(0x2096, 9, 2, 40, 0x00000000U, "z9BC", "IBM System z9 BC GA1"), - CPUDEF_INIT(0x2094, 9, 3, 40, 0x00000000U, "z9EC.3", "IBM System z9 EC GA3"), - CPUDEF_INIT(0x2096, 9, 3, 40, 0x00000000U, "z9BC.2", "IBM System z9 BC GA2"), --#endif - CPUDEF_INIT(0x2097, 10, 1, 43, 0x00000000U, "z10EC", "IBM System z10 EC GA1"), - CPUDEF_INIT(0x2097, 10, 2, 43, 0x00000000U, "z10EC.2", "IBM System z10 EC GA2"), - CPUDEF_INIT(0x2098, 10, 2, 43, 0x00000000U, "z10BC", "IBM System z10 BC GA1"), -@@ -84,6 +83,7 @@ static S390CPUDef s390_cpu_defs[] = { - CPUDEF_INIT(0x2817, 11, 1, 44, 0x08000000U, "z196", "IBM zEnterprise 196 GA1"), - CPUDEF_INIT(0x2817, 11, 2, 44, 0x08000000U, "z196.2", "IBM zEnterprise 196 GA2"), - CPUDEF_INIT(0x2818, 11, 2, 44, 0x08000000U, "z114", "IBM zEnterprise 114 GA1"), -+#endif - CPUDEF_INIT(0x2827, 12, 1, 44, 0x08000000U, "zEC12", "IBM zEnterprise EC12 GA1"), - CPUDEF_INIT(0x2827, 12, 2, 44, 0x08000000U, "zEC12.2", "IBM zEnterprise EC12 GA2"), - CPUDEF_INIT(0x2828, 12, 2, 44, 0x08000000U, "zBC12", "IBM zEnterprise BC12 GA1"), --- -2.39.3 - diff --git a/kvm-tests-functional-add-tests-for-SCLP-event-CPI.patch b/kvm-tests-functional-add-tests-for-SCLP-event-CPI.patch new file mode 100644 index 0000000..502a79a --- /dev/null +++ b/kvm-tests-functional-add-tests-for-SCLP-event-CPI.patch @@ -0,0 +1,76 @@ +From c8082b70cdd198a4a5236ca0543e2653663fb1d4 Mon Sep 17 00:00:00 2001 +From: Shalini Chellathurai Saroja +Date: Thu, 16 Oct 2025 14:17:08 +0200 +Subject: [PATCH 02/10] tests/functional: add tests for SCLP event CPI +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 412: Add -rhel9.8.0 and -rhel10.2.0 s390x machine types and enable the CPI feature +RH-Jira: RHEL-104009 RHEL-105823 RHEL-73008 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/3] 7ff3e50bf038921d4514db88ef147ba92f960780 (thuth/qemu-kvm-cs) + +JIRA: https://issues.redhat.com/browse/RHEL-73008 + +Add tests for SCLP event type Control-Program Identification. + +Signed-off-by: Shalini Chellathurai Saroja +Suggested-by: Thomas Huth +Reviewed-by: Hendrik Brueckner +Reviewed-by Thomas Huth +Message-ID: <20251016121708.334133-2-shalini@linux.ibm.com> +Signed-off-by: Thomas Huth +(cherry picked from commit bc436b739c3c8683ef5e8c22391952dcaa95242e) +--- + tests/functional/test_s390x_ccw_virtio.py | 26 +++++++++++++++++++++++ + 1 file changed, 26 insertions(+) + +diff --git a/tests/functional/test_s390x_ccw_virtio.py b/tests/functional/test_s390x_ccw_virtio.py +index 453711aa0f..0455337856 100755 +--- a/tests/functional/test_s390x_ccw_virtio.py ++++ b/tests/functional/test_s390x_ccw_virtio.py +@@ -15,6 +15,7 @@ + import tempfile + + from qemu_test import QemuSystemTest, Asset ++from qemu_test import exec_command + from qemu_test import exec_command_and_wait_for_pattern + from qemu_test import wait_for_console_pattern + +@@ -270,5 +271,30 @@ def test_s390x_fedora(self): + 'while ! (dmesg -c | grep Start.virtcrypto_remove) ; do' + ' sleep 1 ; done', 'Start virtcrypto_remove.') + ++ # Test SCLP event Control-Program Identification (CPI) ++ cpi = '/sys/firmware/cpi/' ++ sclpcpi = '/machine/sclp/s390-sclp-event-facility/sclpcpi' ++ self.log.info("Test SCLP event CPI") ++ exec_command(self, 'echo TESTVM > ' + cpi + 'system_name') ++ exec_command(self, 'echo LINUX > ' + cpi + 'system_type') ++ exec_command(self, 'echo TESTPLEX > ' + cpi + 'sysplex_name') ++ exec_command(self, 'echo 0x001a000000060b00 > ' + cpi + 'system_level') ++ exec_command_and_wait_for_pattern(self, ++ 'echo 1 > ' + cpi + 'set', ':/#') ++ try: ++ event = self.vm.event_wait('SCLP_CPI_INFO_AVAILABLE') ++ except TimeoutError: ++ self.fail('Timed out waiting for the SCLP_CPI_INFO_AVAILABLE event') ++ ts = self.vm.cmd('qom-get', path=sclpcpi, property='timestamp') ++ self.assertNotEqual(ts, 0) ++ name = self.vm.cmd('qom-get', path=sclpcpi, property='system_name') ++ self.assertEqual(name.strip(), 'TESTVM') ++ typ = self.vm.cmd('qom-get', path=sclpcpi, property='system_type') ++ self.assertEqual(typ.strip(), 'LINUX') ++ sysplex = self.vm.cmd('qom-get', path=sclpcpi, property='sysplex_name') ++ self.assertEqual(sysplex.strip(), 'TESTPLEX') ++ level = self.vm.cmd('qom-get', path=sclpcpi, property='system_level') ++ self.assertEqual(level, 0x001a000000060b00) ++ + if __name__ == '__main__': + QemuSystemTest.main() +-- +2.47.3 + diff --git a/kvm-tests-qtest-libqtest-add-qtest_has_cpu_model-api.patch b/kvm-tests-qtest-libqtest-add-qtest_has_cpu_model-api.patch deleted file mode 100644 index 106e9e2..0000000 --- a/kvm-tests-qtest-libqtest-add-qtest_has_cpu_model-api.patch +++ /dev/null @@ -1,162 +0,0 @@ -From 83bed1458ca3c0137658b53f0a1115d232091703 Mon Sep 17 00:00:00 2001 -From: Ani Sinha -Date: Mon, 10 Jun 2024 21:22:59 +0530 -Subject: [PATCH 02/14] tests/qtest/libqtest: add qtest_has_cpu_model() api -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Ani Sinha -RH-MergeRequest: 243: target/cpu-models/x86: Remove the existing deprecated CPU models on c10s -RH-Jira: RHEL-28972 -RH-Acked-by: Thomas Huth -RH-Acked-by: Igor Mammedov -RH-Acked-by: MST -RH-Commit: [2/4] af128c3ae0a563ca5e2b50bdbdf44f6ce1404aad (anisinha/centos-qemu-kvm) - -Added a new test api qtest_has_cpu_model() in order to check availability of -some cpu models in the current QEMU binary. The specific architecture of the -QEMU binary is selected using the QTEST_QEMU_BINARY environment variable. -This api would be useful to run tests against some older cpu models after -checking if QEMU actually supported these models. - -Signed-off-by: Ani Sinha -Reviewed-by: Reviewed-by: Daniel P. Berrangé -Message-ID: <20240610155303.7933-3-anisinha@redhat.com> -Signed-off-by: Thomas Huth -(cherry picked from commit f43f8abe457a4aa32441bd190638e1118d291c42) ---- - tests/qtest/libqtest.c | 83 ++++++++++++++++++++++++++++++++++++++++++ - tests/qtest/libqtest.h | 8 ++++ - 2 files changed, 91 insertions(+) - -diff --git a/tests/qtest/libqtest.c b/tests/qtest/libqtest.c -index d8f80d335e..18e2f7f282 100644 ---- a/tests/qtest/libqtest.c -+++ b/tests/qtest/libqtest.c -@@ -37,6 +37,7 @@ - #include "qapi/qmp/qjson.h" - #include "qapi/qmp/qlist.h" - #include "qapi/qmp/qstring.h" -+#include "qapi/qmp/qbool.h" - - #define MAX_IRQ 256 - -@@ -1471,6 +1472,12 @@ struct MachInfo { - char *alias; - }; - -+struct CpuModel { -+ char *name; -+ char *alias_of; -+ bool deprecated; -+}; -+ - static void qtest_free_machine_list(struct MachInfo *machines) - { - if (machines) { -@@ -1550,6 +1557,82 @@ static struct MachInfo *qtest_get_machines(const char *var) - return machines; - } - -+static struct CpuModel *qtest_get_cpu_models(void) -+{ -+ static struct CpuModel *cpus; -+ QDict *response, *minfo; -+ QList *list; -+ const QListEntry *p; -+ QObject *qobj; -+ QString *qstr; -+ QBool *qbool; -+ QTestState *qts; -+ int idx; -+ -+ if (cpus) { -+ return cpus; -+ } -+ -+ silence_spawn_log = !g_test_verbose(); -+ -+ qts = qtest_init_with_env(NULL, "-machine none"); -+ response = qtest_qmp(qts, "{ 'execute': 'query-cpu-definitions' }"); -+ g_assert(response); -+ list = qdict_get_qlist(response, "return"); -+ g_assert(list); -+ -+ cpus = g_new0(struct CpuModel, qlist_size(list) + 1); -+ -+ for (p = qlist_first(list), idx = 0; p; p = qlist_next(p), idx++) { -+ minfo = qobject_to(QDict, qlist_entry_obj(p)); -+ g_assert(minfo); -+ -+ qobj = qdict_get(minfo, "name"); -+ g_assert(qobj); -+ qstr = qobject_to(QString, qobj); -+ g_assert(qstr); -+ cpus[idx].name = g_strdup(qstring_get_str(qstr)); -+ -+ qobj = qdict_get(minfo, "alias_of"); -+ if (qobj) { /* old machines do not report aliases */ -+ qstr = qobject_to(QString, qobj); -+ g_assert(qstr); -+ cpus[idx].alias_of = g_strdup(qstring_get_str(qstr)); -+ } else { -+ cpus[idx].alias_of = NULL; -+ } -+ -+ qobj = qdict_get(minfo, "deprecated"); -+ qbool = qobject_to(QBool, qobj); -+ g_assert(qbool); -+ cpus[idx].deprecated = qbool_get_bool(qbool); -+ } -+ -+ qtest_quit(qts); -+ qobject_unref(response); -+ -+ silence_spawn_log = false; -+ -+ return cpus; -+} -+ -+bool qtest_has_cpu_model(const char *cpu) -+{ -+ struct CpuModel *cpus; -+ int i; -+ -+ cpus = qtest_get_cpu_models(); -+ -+ for (i = 0; cpus[i].name != NULL; i++) { -+ if (g_str_equal(cpu, cpus[i].name) || -+ (cpus[i].alias_of && g_str_equal(cpu, cpus[i].alias_of))) { -+ return true; -+ } -+ } -+ -+ return false; -+} -+ - void qtest_cb_for_every_machine(void (*cb)(const char *machine), - bool skip_old_versioned) - { -diff --git a/tests/qtest/libqtest.h b/tests/qtest/libqtest.h -index 6e3d3525bf..beb96b18eb 100644 ---- a/tests/qtest/libqtest.h -+++ b/tests/qtest/libqtest.h -@@ -949,6 +949,14 @@ bool qtest_has_machine(const char *machine); - */ - bool qtest_has_machine_with_env(const char *var, const char *machine); - -+/** -+ * qtest_has_cpu_model: -+ * @cpu: The cpu to look for -+ * -+ * Returns: true if the cpu is available in the target binary. -+ */ -+bool qtest_has_cpu_model(const char *cpu); -+ - /** - * qtest_has_device: - * @device: The device to look for --- -2.39.3 - diff --git a/kvm-tests-qtest-x86-check-for-availability-of-older-cpu-.patch b/kvm-tests-qtest-x86-check-for-availability-of-older-cpu-.patch deleted file mode 100644 index 40b1c45..0000000 --- a/kvm-tests-qtest-x86-check-for-availability-of-older-cpu-.patch +++ /dev/null @@ -1,359 +0,0 @@ -From 31bce7b3e6776e60e0994a45691bded22cc68476 Mon Sep 17 00:00:00 2001 -From: Ani Sinha -Date: Mon, 10 Jun 2024 21:23:00 +0530 -Subject: [PATCH 03/14] tests/qtest/x86: check for availability of older cpu - models before running tests -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Ani Sinha -RH-MergeRequest: 243: target/cpu-models/x86: Remove the existing deprecated CPU models on c10s -RH-Jira: RHEL-28972 -RH-Acked-by: Thomas Huth -RH-Acked-by: Igor Mammedov -RH-Acked-by: MST -RH-Commit: [3/4] 5a049fbd48fda9c1b2d74dc8b389c43547029df2 (anisinha/centos-qemu-kvm) - -It is better to check if some older cpu models like 486, athlon, pentium, -penryn, phenom, core2duo etc are available before running their corresponding -tests. Some downstream distributions may no longer support these older cpu -models. - -Signature of add_feature_test() has been modified to return void as -FeatureTestArgs* was not used by the caller. - -One minor correction. Replaced 'phenom' with '486' in the test -'x86/cpuid/auto-level/phenom/arat' matching the cpu used. - -Signed-off-by: Ani Sinha -Reviewed-by: Daniel P. Berrangé -Message-ID: <20240610155303.7933-4-anisinha@redhat.com> -Signed-off-by: Thomas Huth -(cherry picked from commit e08f6e0b9fcf708f641bbb8839b7e30d857989d9) ---- - tests/qtest/test-x86-cpuid-compat.c | 170 ++++++++++++++++++---------- - 1 file changed, 108 insertions(+), 62 deletions(-) - -diff --git a/tests/qtest/test-x86-cpuid-compat.c b/tests/qtest/test-x86-cpuid-compat.c -index 6a39454fce..b9e7e5ef7b 100644 ---- a/tests/qtest/test-x86-cpuid-compat.c -+++ b/tests/qtest/test-x86-cpuid-compat.c -@@ -67,10 +67,29 @@ static void test_cpuid_prop(const void *data) - g_free(path); - } - --static void add_cpuid_test(const char *name, const char *cmdline, -+static void add_cpuid_test(const char *name, const char *cpu, -+ const char *cpufeat, const char *machine, - const char *property, int64_t expected_value) - { - CpuidTestArgs *args = g_new0(CpuidTestArgs, 1); -+ char *cmdline; -+ char *save; -+ -+ if (!qtest_has_cpu_model(cpu)) { -+ return; -+ } -+ cmdline = g_strdup_printf("-cpu %s", cpu); -+ -+ if (cpufeat) { -+ save = cmdline; -+ cmdline = g_strdup_printf("%s,%s", cmdline, cpufeat); -+ g_free(save); -+ } -+ if (machine) { -+ save = cmdline; -+ cmdline = g_strdup_printf("-machine %s %s", machine, cmdline); -+ g_free(save); -+ } - args->cmdline = cmdline; - args->property = property; - args->expected_value = expected_value; -@@ -149,12 +168,24 @@ static void test_feature_flag(const void *data) - * either "feature-words" or "filtered-features", when running QEMU - * using cmdline - */ --static FeatureTestArgs *add_feature_test(const char *name, const char *cmdline, -- uint32_t eax, uint32_t ecx, -- const char *reg, int bitnr, -- bool expected_value) -+static void add_feature_test(const char *name, const char *cpu, -+ const char *cpufeat, uint32_t eax, -+ uint32_t ecx, const char *reg, -+ int bitnr, bool expected_value) - { - FeatureTestArgs *args = g_new0(FeatureTestArgs, 1); -+ char *cmdline; -+ -+ if (!qtest_has_cpu_model(cpu)) { -+ return; -+ } -+ -+ if (cpufeat) { -+ cmdline = g_strdup_printf("-cpu %s,%s", cpu, cpufeat); -+ } else { -+ cmdline = g_strdup_printf("-cpu %s", cpu); -+ } -+ - args->cmdline = cmdline; - args->in_eax = eax; - args->in_ecx = ecx; -@@ -162,13 +193,17 @@ static FeatureTestArgs *add_feature_test(const char *name, const char *cmdline, - args->bitnr = bitnr; - args->expected_value = expected_value; - qtest_add_data_func(name, args, test_feature_flag); -- return args; -+ return; - } - - static void test_plus_minus_subprocess(void) - { - char *path; - -+ if (!qtest_has_cpu_model("pentium")) { -+ return; -+ } -+ - /* Rules: - * 1)"-foo" overrides "+foo" - * 2) "[+-]foo" overrides "foo=..." -@@ -198,6 +233,10 @@ static void test_plus_minus_subprocess(void) - - static void test_plus_minus(void) - { -+ if (!qtest_has_cpu_model("pentium")) { -+ return; -+ } -+ - g_test_trap_subprocess("/x86/cpuid/parsing-plus-minus/subprocess", 0, 0); - g_test_trap_assert_passed(); - g_test_trap_assert_stderr("*Ambiguous CPU model string. " -@@ -217,99 +256,105 @@ int main(int argc, char **argv) - - /* Original level values for CPU models: */ - add_cpuid_test("x86/cpuid/phenom/level", -- "-cpu phenom", "level", 5); -+ "phenom", NULL, NULL, "level", 5); - add_cpuid_test("x86/cpuid/Conroe/level", -- "-cpu Conroe", "level", 10); -+ "Conroe", NULL, NULL, "level", 10); - add_cpuid_test("x86/cpuid/SandyBridge/level", -- "-cpu SandyBridge", "level", 0xd); -+ "SandyBridge", NULL, NULL, "level", 0xd); - add_cpuid_test("x86/cpuid/486/xlevel", -- "-cpu 486", "xlevel", 0); -+ "486", NULL, NULL, "xlevel", 0); - add_cpuid_test("x86/cpuid/core2duo/xlevel", -- "-cpu core2duo", "xlevel", 0x80000008); -+ "core2duo", NULL, NULL, "xlevel", 0x80000008); - add_cpuid_test("x86/cpuid/phenom/xlevel", -- "-cpu phenom", "xlevel", 0x8000001A); -+ "phenom", NULL, NULL, "xlevel", 0x8000001A); - add_cpuid_test("x86/cpuid/athlon/xlevel", -- "-cpu athlon", "xlevel", 0x80000008); -+ "athlon", NULL, NULL, "xlevel", 0x80000008); - - /* If level is not large enough, it should increase automatically: */ - /* CPUID[6].EAX: */ -- add_cpuid_test("x86/cpuid/auto-level/phenom/arat", -- "-cpu 486,arat=on", "level", 6); -+ add_cpuid_test("x86/cpuid/auto-level/486/arat", -+ "486", "arat=on", NULL, "level", 6); - /* CPUID[EAX=7,ECX=0].EBX: */ - add_cpuid_test("x86/cpuid/auto-level/phenom/fsgsbase", -- "-cpu phenom,fsgsbase=on", "level", 7); -+ "phenom", "fsgsbase=on", NULL, "level", 7); - /* CPUID[EAX=7,ECX=0].ECX: */ - add_cpuid_test("x86/cpuid/auto-level/phenom/avx512vbmi", -- "-cpu phenom,avx512vbmi=on", "level", 7); -+ "phenom", "avx512vbmi=on", NULL, "level", 7); - /* CPUID[EAX=0xd,ECX=1].EAX: */ - add_cpuid_test("x86/cpuid/auto-level/phenom/xsaveopt", -- "-cpu phenom,xsaveopt=on", "level", 0xd); -+ "phenom", "xsaveopt=on", NULL, "level", 0xd); - /* CPUID[8000_0001].EDX: */ - add_cpuid_test("x86/cpuid/auto-xlevel/486/3dnow", -- "-cpu 486,3dnow=on", "xlevel", 0x80000001); -+ "486", "3dnow=on", NULL, "xlevel", 0x80000001); - /* CPUID[8000_0001].ECX: */ - add_cpuid_test("x86/cpuid/auto-xlevel/486/sse4a", -- "-cpu 486,sse4a=on", "xlevel", 0x80000001); -+ "486", "sse4a=on", NULL, "xlevel", 0x80000001); - /* CPUID[8000_0007].EDX: */ - add_cpuid_test("x86/cpuid/auto-xlevel/486/invtsc", -- "-cpu 486,invtsc=on", "xlevel", 0x80000007); -+ "486", "invtsc=on", NULL, "xlevel", 0x80000007); - /* CPUID[8000_000A].EDX: */ - add_cpuid_test("x86/cpuid/auto-xlevel/486/npt", -- "-cpu 486,svm=on,npt=on", "xlevel", 0x8000000A); -+ "486", "svm=on,npt=on", NULL, "xlevel", 0x8000000A); - /* CPUID[C000_0001].EDX: */ - add_cpuid_test("x86/cpuid/auto-xlevel2/phenom/xstore", -- "-cpu phenom,xstore=on", "xlevel2", 0xC0000001); -+ "phenom", "xstore=on", NULL, "xlevel2", 0xC0000001); - /* SVM needs CPUID[0x8000000A] */ - add_cpuid_test("x86/cpuid/auto-xlevel/athlon/svm", -- "-cpu athlon,svm=on", "xlevel", 0x8000000A); -+ "athlon", "svm=on", NULL, "xlevel", 0x8000000A); - - - /* If level is already large enough, it shouldn't change: */ - add_cpuid_test("x86/cpuid/auto-level/SandyBridge/multiple", -- "-cpu SandyBridge,arat=on,fsgsbase=on,avx512vbmi=on", -- "level", 0xd); -+ "SandyBridge", "arat=on,fsgsbase=on,avx512vbmi=on", -+ NULL, "level", 0xd); - /* If level is explicitly set, it shouldn't change: */ - add_cpuid_test("x86/cpuid/auto-level/486/fixed/0xF", -- "-cpu 486,level=0xF,arat=on,fsgsbase=on,avx512vbmi=on,xsaveopt=on", -- "level", 0xF); -+ "486", -+ "level=0xF,arat=on,fsgsbase=on,avx512vbmi=on,xsaveopt=on", -+ NULL, "level", 0xF); - add_cpuid_test("x86/cpuid/auto-level/486/fixed/2", -- "-cpu 486,level=2,arat=on,fsgsbase=on,avx512vbmi=on,xsaveopt=on", -- "level", 2); -+ "486", -+ "level=2,arat=on,fsgsbase=on,avx512vbmi=on,xsaveopt=on", -+ NULL, "level", 2); - add_cpuid_test("x86/cpuid/auto-level/486/fixed/0", -- "-cpu 486,level=0,arat=on,fsgsbase=on,avx512vbmi=on,xsaveopt=on", -- "level", 0); -+ "486", -+ "level=0,arat=on,fsgsbase=on,avx512vbmi=on,xsaveopt=on", -+ NULL, "level", 0); - - /* if xlevel is already large enough, it shouldn't change: */ - add_cpuid_test("x86/cpuid/auto-xlevel/phenom/3dnow", -- "-cpu phenom,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -- "xlevel", 0x8000001A); -+ "phenom", "3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -+ NULL, "xlevel", 0x8000001A); - /* If xlevel is explicitly set, it shouldn't change: */ - add_cpuid_test("x86/cpuid/auto-xlevel/486/fixed/80000002", -- "-cpu 486,xlevel=0x80000002,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -- "xlevel", 0x80000002); -+ "486", -+ "xlevel=0x80000002,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -+ NULL, "xlevel", 0x80000002); - add_cpuid_test("x86/cpuid/auto-xlevel/486/fixed/8000001A", -- "-cpu 486,xlevel=0x8000001A,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -- "xlevel", 0x8000001A); -+ "486", -+ "xlevel=0x8000001A,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -+ NULL, "xlevel", 0x8000001A); - add_cpuid_test("x86/cpuid/auto-xlevel/phenom/fixed/0", -- "-cpu 486,xlevel=0,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -- "xlevel", 0); -+ "486", -+ "xlevel=0,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -+ NULL, "xlevel", 0); - - /* if xlevel2 is already large enough, it shouldn't change: */ - add_cpuid_test("x86/cpuid/auto-xlevel2/486/fixed", -- "-cpu 486,xlevel2=0xC0000002,xstore=on", -- "xlevel2", 0xC0000002); -+ "486", "xlevel2=0xC0000002,xstore=on", -+ NULL, "xlevel2", 0xC0000002); - - /* Check compatibility of old machine-types that didn't - * auto-increase level/xlevel/xlevel2: */ - if (qtest_has_machine("pc-i440fx-2.7")) { - add_cpuid_test("x86/cpuid/auto-level/pc-2.7", -- "-machine pc-i440fx-2.7 -cpu 486,arat=on,avx512vbmi=on,xsaveopt=on", -- "level", 1); -+ "486", "arat=on,avx512vbmi=on,xsaveopt=on", -+ "pc-i440fx-2.7", "level", 1); - add_cpuid_test("x86/cpuid/auto-xlevel/pc-2.7", -- "-machine pc-i440fx-2.7 -cpu 486,3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -- "xlevel", 0); -+ "486", "3dnow=on,sse4a=on,invtsc=on,npt=on,svm=on", -+ "pc-i440fx-2.7", "xlevel", 0); - add_cpuid_test("x86/cpuid/auto-xlevel2/pc-2.7", -- "-machine pc-i440fx-2.7 -cpu 486,xstore=on", -+ "486", "xstore=on", "pc-i440fx-2.7", - "xlevel2", 0); - } - /* -@@ -319,18 +364,18 @@ int main(int argc, char **argv) - */ - if (qtest_has_machine("pc-i440fx-2.3")) { - add_cpuid_test("x86/cpuid/auto-level7/pc-i440fx-2.3/off", -- "-machine pc-i440fx-2.3 -cpu Penryn", -+ "Penryn", NULL, "pc-i440fx-2.3", - "level", 4); - add_cpuid_test("x86/cpuid/auto-level7/pc-i440fx-2.3/on", -- "-machine pc-i440fx-2.3 -cpu Penryn,erms=on", -+ "Penryn", "erms=on", "pc-i440fx-2.3", - "level", 7); - } - if (qtest_has_machine("pc-i440fx-2.9")) { - add_cpuid_test("x86/cpuid/auto-level7/pc-i440fx-2.9/off", -- "-machine pc-i440fx-2.9 -cpu Conroe", -+ "Conroe", NULL, "pc-i440fx-2.9", - "level", 10); - add_cpuid_test("x86/cpuid/auto-level7/pc-i440fx-2.9/on", -- "-machine pc-i440fx-2.9 -cpu Conroe,erms=on", -+ "Conroe", "erms=on", "pc-i440fx-2.9", - "level", 10); - } - -@@ -341,42 +386,43 @@ int main(int argc, char **argv) - */ - if (qtest_has_machine("pc-i440fx-2.3")) { - add_cpuid_test("x86/cpuid/xlevel-compat/pc-i440fx-2.3", -- "-machine pc-i440fx-2.3 -cpu SandyBridge", -+ "SandyBridge", NULL, "pc-i440fx-2.3", - "xlevel", 0x8000000a); - } - if (qtest_has_machine("pc-i440fx-2.4")) { - add_cpuid_test("x86/cpuid/xlevel-compat/pc-i440fx-2.4/npt-off", -- "-machine pc-i440fx-2.4 -cpu SandyBridge,", -+ "SandyBridge", NULL, "pc-i440fx-2.4", - "xlevel", 0x80000008); - add_cpuid_test("x86/cpuid/xlevel-compat/pc-i440fx-2.4/npt-on", -- "-machine pc-i440fx-2.4 -cpu SandyBridge,svm=on,npt=on", -+ "SandyBridge", "svm=on,npt=on", "pc-i440fx-2.4", - "xlevel", 0x80000008); - } - - /* Test feature parsing */ - add_feature_test("x86/cpuid/features/plus", -- "-cpu 486,+arat", -+ "486", "+arat", - 6, 0, "EAX", 2, true); - add_feature_test("x86/cpuid/features/minus", -- "-cpu pentium,-mmx", -+ "pentium", "-mmx", - 1, 0, "EDX", 23, false); - add_feature_test("x86/cpuid/features/on", -- "-cpu 486,arat=on", -+ "486", "arat=on", - 6, 0, "EAX", 2, true); - add_feature_test("x86/cpuid/features/off", -- "-cpu pentium,mmx=off", -+ "pentium", "mmx=off", - 1, 0, "EDX", 23, false); -+ - add_feature_test("x86/cpuid/features/max-plus-invtsc", -- "-cpu max,+invtsc", -+ "max" , "+invtsc", - 0x80000007, 0, "EDX", 8, true); - add_feature_test("x86/cpuid/features/max-invtsc-on", -- "-cpu max,invtsc=on", -+ "max", "invtsc=on", - 0x80000007, 0, "EDX", 8, true); - add_feature_test("x86/cpuid/features/max-minus-mmx", -- "-cpu max,-mmx", -+ "max", "-mmx", - 1, 0, "EDX", 23, false); - add_feature_test("x86/cpuid/features/max-invtsc-on,mmx=off", -- "-cpu max,mmx=off", -+ "max", "mmx=off", - 1, 0, "EDX", 23, false); - - return g_test_run(); --- -2.39.3 - diff --git a/kvm-treewide-rename-qemu_wait_io_event-qemu_wait_io_even.patch b/kvm-treewide-rename-qemu_wait_io_event-qemu_wait_io_even.patch new file mode 100644 index 0000000..c894679 --- /dev/null +++ b/kvm-treewide-rename-qemu_wait_io_event-qemu_wait_io_even.patch @@ -0,0 +1,219 @@ +From 8aa1e85d4a67cbe79da787ddfbe5d65395cd5daf Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Tue, 2 Sep 2025 07:17:09 +0200 +Subject: [PATCH 12/32] treewide: rename + qemu_wait_io_event/qemu_wait_io_event_common + +RH-Author: Igor Mammedov +RH-MergeRequest: 437: el10: x86: enablement for Azure L1VH OCP readiness +RH-Jira: RHEL-134212 +RH-Acked-by: Vitaly Kuznetsov +RH-Acked-by: Miroslav Rezanina +RH-Commit: [10/30] 501e99aab6fb49429b23c688cffbecdf283c89dd + +Do so before extending it to the user-mode emulators, where there is no +such thing as an "I/O thread". + +Reviewed-by: Richard Henderson +Signed-off-by: Paolo Bonzini +(cherry picked from commit 871de7078fcaf597605576b97b32fab14722ea43) +Signed-off-by: Igor Mammedov + + Conflicts: (due to missing cpu exit/interrupt_request refactoring) + cpu-common.c + include/hw/core/cpu.h +--- + accel/dummy-cpus.c | 2 +- + accel/hvf/hvf-accel-ops.c | 2 +- + accel/kvm/kvm-accel-ops.c | 2 +- + accel/tcg/tcg-accel-ops-mttcg.c | 2 +- + accel/tcg/tcg-accel-ops-rr.c | 4 ++-- + cpu-common.c | 1 + + include/hw/core/cpu.h | 9 +++++++++ + include/system/cpus.h | 4 ++-- + system/cpus.c | 6 +++--- + target/i386/nvmm/nvmm-accel-ops.c | 2 +- + target/i386/whpx/whpx-accel-ops.c | 2 +- + 11 files changed, 23 insertions(+), 13 deletions(-) + +diff --git a/accel/dummy-cpus.c b/accel/dummy-cpus.c +index 03cfc0fa01..225a47c31f 100644 +--- a/accel/dummy-cpus.c ++++ b/accel/dummy-cpus.c +@@ -57,7 +57,7 @@ static void *dummy_cpu_thread_fn(void *arg) + qemu_sem_wait(&cpu->sem); + #endif + bql_lock(); +- qemu_wait_io_event(cpu); ++ qemu_process_cpu_events(cpu); + } while (!cpu->unplug); + + bql_unlock(); +diff --git a/accel/hvf/hvf-accel-ops.c b/accel/hvf/hvf-accel-ops.c +index d488d6afba..7a27bdadb4 100644 +--- a/accel/hvf/hvf-accel-ops.c ++++ b/accel/hvf/hvf-accel-ops.c +@@ -198,7 +198,7 @@ static void *hvf_cpu_thread_fn(void *arg) + cpu_handle_guest_debug(cpu); + } + } +- qemu_wait_io_event(cpu); ++ qemu_process_cpu_events(cpu); + } while (!cpu->unplug || cpu_can_run(cpu)); + + hvf_vcpu_destroy(cpu); +diff --git a/accel/kvm/kvm-accel-ops.c b/accel/kvm/kvm-accel-ops.c +index b709187c7d..65a7f76a69 100644 +--- a/accel/kvm/kvm-accel-ops.c ++++ b/accel/kvm/kvm-accel-ops.c +@@ -53,7 +53,7 @@ static void *kvm_vcpu_thread_fn(void *arg) + cpu_handle_guest_debug(cpu); + } + } +- qemu_wait_io_event(cpu); ++ qemu_process_cpu_events(cpu); + } while (!cpu->unplug || cpu_can_run(cpu)); + + kvm_destroy_vcpu(cpu); +diff --git a/accel/tcg/tcg-accel-ops-mttcg.c b/accel/tcg/tcg-accel-ops-mttcg.c +index 337b993d3d..a3255000df 100644 +--- a/accel/tcg/tcg-accel-ops-mttcg.c ++++ b/accel/tcg/tcg-accel-ops-mttcg.c +@@ -113,7 +113,7 @@ static void *mttcg_cpu_thread_fn(void *arg) + } + } + +- qemu_wait_io_event(cpu); ++ qemu_process_cpu_events(cpu); + } while (!cpu->unplug || cpu_can_run(cpu)); + + tcg_cpu_destroy(cpu); +diff --git a/accel/tcg/tcg-accel-ops-rr.c b/accel/tcg/tcg-accel-ops-rr.c +index 6eec5c9eee..60ad1a39f0 100644 +--- a/accel/tcg/tcg-accel-ops-rr.c ++++ b/accel/tcg/tcg-accel-ops-rr.c +@@ -117,7 +117,7 @@ static void rr_wait_io_event(void) + rr_start_kick_timer(); + + CPU_FOREACH(cpu) { +- qemu_wait_io_event_common(cpu); ++ qemu_process_cpu_events_common(cpu); + } + } + +@@ -203,7 +203,7 @@ static void *rr_cpu_thread_fn(void *arg) + /* process any pending work */ + CPU_FOREACH(cpu) { + current_cpu = cpu; +- qemu_wait_io_event_common(cpu); ++ qemu_process_cpu_events_common(cpu); + } + } + +diff --git a/cpu-common.c b/cpu-common.c +index ef5757d23b..8e4cbac290 100644 +--- a/cpu-common.c ++++ b/cpu-common.c +@@ -137,6 +137,7 @@ static void queue_work_on_cpu(CPUState *cpu, struct qemu_work_item *wi) + wi->done = false; + qemu_mutex_unlock(&cpu->work_mutex); + ++ /* conflict fixup due missing exit and interrupt refactoring */ + qemu_cpu_kick(cpu); + } + +diff --git a/include/hw/core/cpu.h b/include/hw/core/cpu.h +index 5eaf41a566..fadd6c5ac0 100644 +--- a/include/hw/core/cpu.h ++++ b/include/hw/core/cpu.h +@@ -422,6 +422,15 @@ struct qemu_work_item; + * valid under cpu_list_lock. + * @created: Indicates whether the CPU thread has been successfully created. + * @halt_cond: condition variable sleeping threads can wait on. ++ * @exit_request: Another thread requests the CPU to call qemu_process_cpu_events(). ++ * Should be read only by CPU thread with load-acquire, to synchronize with ++ * other threads' store-release operation. ++ * ++ * In some cases, accelerator-specific code will write exit_request from ++ * within the same thread, to "bump" the effect of qemu_cpu_kick() to ++ * the one provided by cpu_exit(), especially when processing interrupt ++ * flags. In this case, the write and read happen in the same thread ++ * and the write therefore can use qemu_atomic_set(). + * @interrupt_request: Indicates a pending interrupt request. + * @halted: Nonzero if the CPU is in suspended state. + * @stop: Indicates a pending stop request. +diff --git a/include/system/cpus.h b/include/system/cpus.h +index 69be6a77a7..4aebec4870 100644 +--- a/include/system/cpus.h ++++ b/include/system/cpus.h +@@ -17,8 +17,8 @@ bool cpu_work_list_empty(CPUState *cpu); + bool cpu_thread_is_idle(CPUState *cpu); + bool all_cpu_threads_idle(void); + bool cpu_can_run(CPUState *cpu); +-void qemu_wait_io_event_common(CPUState *cpu); +-void qemu_wait_io_event(CPUState *cpu); ++void qemu_process_cpu_events_common(CPUState *cpu); ++void qemu_process_cpu_events(CPUState *cpu); + void cpu_thread_signal_created(CPUState *cpu); + void cpu_thread_signal_destroyed(CPUState *cpu); + void cpu_handle_guest_debug(CPUState *cpu); +diff --git a/system/cpus.c b/system/cpus.c +index 256723558d..efffbd8df6 100644 +--- a/system/cpus.c ++++ b/system/cpus.c +@@ -444,7 +444,7 @@ static void qemu_cpu_stop(CPUState *cpu, bool exit) + qemu_cond_broadcast(&qemu_pause_cond); + } + +-void qemu_wait_io_event_common(CPUState *cpu) ++void qemu_process_cpu_events_common(CPUState *cpu) + { + qatomic_set_mb(&cpu->thread_kicked, false); + if (cpu->stop) { +@@ -453,7 +453,7 @@ void qemu_wait_io_event_common(CPUState *cpu) + process_queued_cpu_work(cpu); + } + +-void qemu_wait_io_event(CPUState *cpu) ++void qemu_process_cpu_events(CPUState *cpu) + { + bool slept = false; + +@@ -468,7 +468,7 @@ void qemu_wait_io_event(CPUState *cpu) + qemu_plugin_vcpu_resume_cb(cpu); + } + +- qemu_wait_io_event_common(cpu); ++ qemu_process_cpu_events_common(cpu); + } + + void cpus_kick_thread(CPUState *cpu) +diff --git a/target/i386/nvmm/nvmm-accel-ops.c b/target/i386/nvmm/nvmm-accel-ops.c +index 3799260bbd..c7b1d0f1d3 100644 +--- a/target/i386/nvmm/nvmm-accel-ops.c ++++ b/target/i386/nvmm/nvmm-accel-ops.c +@@ -51,7 +51,7 @@ static void *qemu_nvmm_cpu_thread_fn(void *arg) + while (cpu_thread_is_idle(cpu)) { + qemu_cond_wait_bql(cpu->halt_cond); + } +- qemu_wait_io_event_common(cpu); ++ qemu_process_cpu_events_common(cpu); + } while (!cpu->unplug || cpu_can_run(cpu)); + + nvmm_destroy_vcpu(cpu); +diff --git a/target/i386/whpx/whpx-accel-ops.c b/target/i386/whpx/whpx-accel-ops.c +index da58805b1a..2ca4ee0263 100644 +--- a/target/i386/whpx/whpx-accel-ops.c ++++ b/target/i386/whpx/whpx-accel-ops.c +@@ -51,7 +51,7 @@ static void *whpx_cpu_thread_fn(void *arg) + while (cpu_thread_is_idle(cpu)) { + qemu_cond_wait_bql(cpu->halt_cond); + } +- qemu_wait_io_event_common(cpu); ++ qemu_process_cpu_events_common(cpu); + } while (!cpu->unplug || cpu_can_run(cpu)); + + whpx_destroy_vcpu(cpu); +-- +2.47.3 + diff --git a/kvm-vfio-Disable-VFIO-migration-with-MultiFD-support.patch b/kvm-vfio-Disable-VFIO-migration-with-MultiFD-support.patch new file mode 100644 index 0000000..5219676 --- /dev/null +++ b/kvm-vfio-Disable-VFIO-migration-with-MultiFD-support.patch @@ -0,0 +1,47 @@ +From 66bd3c1e7702962060d23fdc3084f0ace26b94e6 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?C=C3=A9dric=20Le=20Goater?= +Date: Thu, 6 Nov 2025 16:39:53 +0100 +Subject: [PATCH 03/16] vfio: Disable VFIO migration with MultiFD support +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Cédric Le Goater +RH-MergeRequest: 421: vfio: Disable VFIO migration with MultiFD support +RH-Jira: RHEL-126573 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Thomas Huth +RH-Commit: [1/1] b3ec6731c96e5650c66ece6e3b8728a7b94353f2 (clegoate/qemu-kvm-centos) + +QEMU 10.0 extends VFIO migration with MultiFD support, which can be +controlled through the 'vfio-pci' device property +'x-migration-multifd-transfer'. By default, this property is set to +'auto', meaning its activation depends on the availability of other +related features. However, it should be set to 'off' in RHEL until +more testing has been completed. + +Signed-off-by: Cédric Le Goater +--- + hw/vfio/pci.c | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) + +diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c +index 9486521a90..83ecffb535 100644 +--- a/hw/vfio/pci.c ++++ b/hw/vfio/pci.c +@@ -3686,10 +3686,11 @@ static const Property vfio_pci_dev_properties[] = { + igd_legacy_mode, ON_OFF_AUTO_AUTO), + DEFINE_PROP_ON_OFF_AUTO("enable-migration", VFIOPCIDevice, + vbasedev.enable_migration, ON_OFF_AUTO_AUTO), ++ /* RHEL only. Disable VFIO migration with MultiFD support */ + DEFINE_PROP("x-migration-multifd-transfer", VFIOPCIDevice, + vbasedev.migration_multifd_transfer, + vfio_pci_migration_multifd_transfer_prop, OnOffAuto, +- .set_default = true, .defval.i = ON_OFF_AUTO_AUTO), ++ .set_default = true, .defval.i = ON_OFF_AUTO_OFF), + DEFINE_PROP_ON_OFF_AUTO("x-migration-load-config-after-iter", VFIOPCIDevice, + vbasedev.migration_load_config_after_iter, + ON_OFF_AUTO_AUTO), +-- +2.47.3 + diff --git a/kvm-vfio-only-check-region-info-cache-for-initial-region.patch b/kvm-vfio-only-check-region-info-cache-for-initial-region.patch new file mode 100644 index 0000000..02b4fc2 --- /dev/null +++ b/kvm-vfio-only-check-region-info-cache-for-initial-region.patch @@ -0,0 +1,83 @@ +From 9659c700e0afab65e7993459764b2e802178873b Mon Sep 17 00:00:00 2001 +From: John Levon +Date: Tue, 14 Oct 2025 17:12:27 +0200 +Subject: [PATCH 05/10] vfio: only check region info cache for initial regions +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Cédric Le Goater +RH-MergeRequest: 414: Fixes for vfio region cache +RH-Jira: RHEL-118810 +RH-Acked-by: Eric Auger +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/2] 30b628642ee153e7da612e43aead3c6cd2de9769 (clegoate/qemu-kvm-centos) + +It is semantically valid for a VFIO device to increase the number of +regions after initialization. In this case, we'd attempt to check for +cached region info past the size of the ->reginfo array. Check for the +region index and skip the cache in these cases. + +This also works around some VGPU use cases which appear to be a bug, +where VFIO_DEVICE_QUERY_GFX_PLANE returns a region index beyond the +reported ->num_regions. + +Fixes: 95cdb024 ("vfio: add region info cache") +Signed-off-by: John Levon +Reviewed-by: Cédric Le Goater +Reviewed-by: Alex Williamson +Link: https://lore.kernel.org/qemu-devel/20251014151227.2298892-3-john.levon@nutanix.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit ecbe424a63c9f860a901d6a4a75724b046abd796) +--- + hw/vfio/device.c | 27 +++++++++++++++++++-------- + 1 file changed, 19 insertions(+), 8 deletions(-) + +diff --git a/hw/vfio/device.c b/hw/vfio/device.c +index 0b459c0f7c..7ebf41c95e 100644 +--- a/hw/vfio/device.c ++++ b/hw/vfio/device.c +@@ -205,10 +205,19 @@ int vfio_device_get_region_info(VFIODevice *vbasedev, int index, + int fd = -1; + int ret; + +- /* check cache */ +- if (vbasedev->reginfo[index] != NULL) { +- *info = vbasedev->reginfo[index]; +- return 0; ++ /* ++ * We only set up the region info cache for the initial number of regions. ++ * ++ * Since a VFIO device may later increase the number of regions then use ++ * such regions with an index past ->num_initial_regions, don't attempt to ++ * use the info cache in those cases. ++ */ ++ if (index < vbasedev->num_initial_regions) { ++ /* check cache */ ++ if (vbasedev->reginfo[index] != NULL) { ++ *info = vbasedev->reginfo[index]; ++ return 0; ++ } + } + + *info = g_malloc0(argsz); +@@ -236,10 +245,12 @@ retry: + goto retry; + } + +- /* fill cache */ +- vbasedev->reginfo[index] = *info; +- if (vbasedev->region_fds != NULL) { +- vbasedev->region_fds[index] = fd; ++ if (index < vbasedev->num_initial_regions) { ++ /* fill cache */ ++ vbasedev->reginfo[index] = *info; ++ if (vbasedev->region_fds != NULL) { ++ vbasedev->region_fds[index] = fd; ++ } + } + + return 0; +-- +2.47.3 + diff --git a/kvm-vfio-rename-field-to-num_initial_regions.patch b/kvm-vfio-rename-field-to-num_initial_regions.patch new file mode 100644 index 0000000..84b624d --- /dev/null +++ b/kvm-vfio-rename-field-to-num_initial_regions.patch @@ -0,0 +1,253 @@ +From de33fcdabae841dba89fae781d6391164b22740f Mon Sep 17 00:00:00 2001 +From: John Levon +Date: Tue, 14 Oct 2025 17:12:26 +0200 +Subject: [PATCH 04/10] vfio: rename field to "num_initial_regions" +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Cédric Le Goater +RH-MergeRequest: 414: Fixes for vfio region cache +RH-Jira: RHEL-118810 +RH-Acked-by: Eric Auger +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/2] 62524d77e79b6a57c4e61762cd61449280e4f2cd (clegoate/qemu-kvm-centos) + +We set VFIODevice::num_regions at initialization time, and do not +otherwise refresh it. As it is valid in theory for a VFIO device to +later increase the number of supported regions, rename the field to +"num_initial_regions" to better reflect its semantics. + +Signed-off-by: John Levon +Reviewed-by: Cédric Le Goater +Reviewed-by: Alex Williamson +Link: https://lore.kernel.org/qemu-devel/20251014151227.2298892-2-john.levon@nutanix.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit aaca725884b57c9245528a0afb3f32e078543faf) + +Conflicts: Modified hw/core/sysbus-fdt.c and hw/vfio/platform.c +--- + hw/core/sysbus-fdt.c | 14 +++++++------- + hw/vfio-user/device.c | 2 +- + hw/vfio/ccw.c | 4 ++-- + hw/vfio/device.c | 12 ++++++------ + hw/vfio/iommufd.c | 3 ++- + hw/vfio/pci.c | 4 ++-- + hw/vfio/platform.c | 10 +++++----- + include/hw/vfio/vfio-device.h | 2 +- + 8 files changed, 26 insertions(+), 25 deletions(-) + +diff --git a/hw/core/sysbus-fdt.c b/hw/core/sysbus-fdt.c +index c339a27875..1e1966813f 100644 +--- a/hw/core/sysbus-fdt.c ++++ b/hw/core/sysbus-fdt.c +@@ -236,15 +236,15 @@ static int add_calxeda_midway_xgmac_fdt_node(SysBusDevice *sbdev, void *opaque) + + qemu_fdt_setprop(fdt, nodename, "dma-coherent", "", 0); + +- reg_attr = g_new(uint32_t, vbasedev->num_regions * 2); +- for (i = 0; i < vbasedev->num_regions; i++) { ++ reg_attr = g_new(uint32_t, vbasedev->num_initial_regions * 2); ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + mmio_base = platform_bus_get_mmio_addr(pbus, sbdev, i); + reg_attr[2 * i] = cpu_to_be32(mmio_base); + reg_attr[2 * i + 1] = cpu_to_be32( + memory_region_size(vdev->regions[i]->mem)); + } + qemu_fdt_setprop(fdt, nodename, "reg", reg_attr, +- vbasedev->num_regions * 2 * sizeof(uint32_t)); ++ vbasedev->num_initial_regions * 2 * sizeof(uint32_t)); + + irq_attr = g_new(uint32_t, vbasedev->num_irqs * 3); + for (i = 0; i < vbasedev->num_irqs; i++) { +@@ -330,7 +330,7 @@ static int add_amd_xgbe_fdt_node(SysBusDevice *sbdev, void *opaque) + + g_free(dt_name); + +- if (vbasedev->num_regions != 5) { ++ if (vbasedev->num_initial_regions != 5) { + error_report("%s Does the host dt node combine XGBE/PHY?", __func__); + exit(1); + } +@@ -374,15 +374,15 @@ static int add_amd_xgbe_fdt_node(SysBusDevice *sbdev, void *opaque) + guest_clock_phandles[0], + guest_clock_phandles[1]); + +- reg_attr = g_new(uint32_t, vbasedev->num_regions * 2); +- for (i = 0; i < vbasedev->num_regions; i++) { ++ reg_attr = g_new(uint32_t, vbasedev->num_initial_regions * 2); ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + mmio_base = platform_bus_get_mmio_addr(pbus, sbdev, i); + reg_attr[2 * i] = cpu_to_be32(mmio_base); + reg_attr[2 * i + 1] = cpu_to_be32( + memory_region_size(vdev->regions[i]->mem)); + } + qemu_fdt_setprop(guest_fdt, nodename, "reg", reg_attr, +- vbasedev->num_regions * 2 * sizeof(uint32_t)); ++ vbasedev->num_initial_regions * 2 * sizeof(uint32_t)); + + irq_attr = g_new(uint32_t, vbasedev->num_irqs * 3); + for (i = 0; i < vbasedev->num_irqs; i++) { +diff --git a/hw/vfio-user/device.c b/hw/vfio-user/device.c +index 0609a7dc25..64ef35b320 100644 +--- a/hw/vfio-user/device.c ++++ b/hw/vfio-user/device.c +@@ -134,7 +134,7 @@ static int vfio_user_device_io_get_region_info(VFIODevice *vbasedev, + VFIOUserFDs fds = { 0, 1, fd}; + int ret; + +- if (info->index > vbasedev->num_regions) { ++ if (info->index > vbasedev->num_initial_regions) { + return -EINVAL; + } + +diff --git a/hw/vfio/ccw.c b/hw/vfio/ccw.c +index 9560b8d851..4d9588e7aa 100644 +--- a/hw/vfio/ccw.c ++++ b/hw/vfio/ccw.c +@@ -484,9 +484,9 @@ static bool vfio_ccw_get_region(VFIOCCWDevice *vcdev, Error **errp) + * We always expect at least the I/O region to be present. We also + * may have a variable number of regions governed by capabilities. + */ +- if (vdev->num_regions < VFIO_CCW_CONFIG_REGION_INDEX + 1) { ++ if (vdev->num_initial_regions < VFIO_CCW_CONFIG_REGION_INDEX + 1) { + error_setg(errp, "vfio: too few regions (%u), expected at least %u", +- vdev->num_regions, VFIO_CCW_CONFIG_REGION_INDEX + 1); ++ vdev->num_initial_regions, VFIO_CCW_CONFIG_REGION_INDEX + 1); + return false; + } + +diff --git a/hw/vfio/device.c b/hw/vfio/device.c +index 52a1996dc4..0b459c0f7c 100644 +--- a/hw/vfio/device.c ++++ b/hw/vfio/device.c +@@ -257,7 +257,7 @@ int vfio_device_get_region_info_type(VFIODevice *vbasedev, uint32_t type, + { + int i; + +- for (i = 0; i < vbasedev->num_regions; i++) { ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + struct vfio_info_cap_header *hdr; + struct vfio_region_info_cap_type *cap_type; + +@@ -466,7 +466,7 @@ void vfio_device_prepare(VFIODevice *vbasedev, VFIOContainerBase *bcontainer, + int i; + + vbasedev->num_irqs = info->num_irqs; +- vbasedev->num_regions = info->num_regions; ++ vbasedev->num_initial_regions = info->num_regions; + vbasedev->flags = info->flags; + vbasedev->reset_works = !!(info->flags & VFIO_DEVICE_FLAGS_RESET); + +@@ -476,10 +476,10 @@ void vfio_device_prepare(VFIODevice *vbasedev, VFIOContainerBase *bcontainer, + QLIST_INSERT_HEAD(&vfio_device_list, vbasedev, global_next); + + vbasedev->reginfo = g_new0(struct vfio_region_info *, +- vbasedev->num_regions); ++ vbasedev->num_initial_regions); + if (vbasedev->use_region_fds) { +- vbasedev->region_fds = g_new0(int, vbasedev->num_regions); +- for (i = 0; i < vbasedev->num_regions; i++) { ++ vbasedev->region_fds = g_new0(int, vbasedev->num_initial_regions); ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + vbasedev->region_fds[i] = -1; + } + } +@@ -489,7 +489,7 @@ void vfio_device_unprepare(VFIODevice *vbasedev) + { + int i; + +- for (i = 0; i < vbasedev->num_regions; i++) { ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + g_free(vbasedev->reginfo[i]); + if (vbasedev->region_fds != NULL && vbasedev->region_fds[i] != -1) { + close(vbasedev->region_fds[i]); +diff --git a/hw/vfio/iommufd.c b/hw/vfio/iommufd.c +index 48c590b6a9..dbcd861b27 100644 +--- a/hw/vfio/iommufd.c ++++ b/hw/vfio/iommufd.c +@@ -668,7 +668,8 @@ found_container: + vfio_iommufd_cpr_register_device(vbasedev); + + trace_iommufd_cdev_device_info(vbasedev->name, devfd, vbasedev->num_irqs, +- vbasedev->num_regions, vbasedev->flags); ++ vbasedev->num_initial_regions, ++ vbasedev->flags); + return true; + + err_listener_register: +diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c +index 48da233cb2..9486521a90 100644 +--- a/hw/vfio/pci.c ++++ b/hw/vfio/pci.c +@@ -2933,9 +2933,9 @@ bool vfio_pci_populate_device(VFIOPCIDevice *vdev, Error **errp) + return false; + } + +- if (vbasedev->num_regions < VFIO_PCI_CONFIG_REGION_INDEX + 1) { ++ if (vbasedev->num_initial_regions < VFIO_PCI_CONFIG_REGION_INDEX + 1) { + error_setg(errp, "unexpected number of io regions %u", +- vbasedev->num_regions); ++ vbasedev->num_initial_regions); + return false; + } + +diff --git a/hw/vfio/platform.c b/hw/vfio/platform.c +index 5c1795a26f..c9349ba7b7 100644 +--- a/hw/vfio/platform.c ++++ b/hw/vfio/platform.c +@@ -148,7 +148,7 @@ static void vfio_mmap_set_enabled(VFIOPlatformDevice *vdev, bool enabled) + { + int i; + +- for (i = 0; i < vdev->vbasedev.num_regions; i++) { ++ for (i = 0; i < vdev->vbasedev.num_initial_regions; i++) { + vfio_region_mmaps_set_enabled(vdev->regions[i], enabled); + } + } +@@ -453,9 +453,9 @@ static bool vfio_populate_device(VFIODevice *vbasedev, Error **errp) + return false; + } + +- vdev->regions = g_new0(VFIORegion *, vbasedev->num_regions); ++ vdev->regions = g_new0(VFIORegion *, vbasedev->num_initial_regions); + +- for (i = 0; i < vbasedev->num_regions; i++) { ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + char *name = g_strdup_printf("VFIO %s region %d\n", vbasedev->name, i); + + vdev->regions[i] = g_new0(VFIORegion, 1); +@@ -499,7 +499,7 @@ irq_err: + g_free(intp); + } + reg_error: +- for (i = 0; i < vbasedev->num_regions; i++) { ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + if (vdev->regions[i]) { + vfio_region_finalize(vdev->regions[i]); + } +@@ -608,7 +608,7 @@ static void vfio_platform_realize(DeviceState *dev, Error **errp) + } + } + +- for (i = 0; i < vbasedev->num_regions; i++) { ++ for (i = 0; i < vbasedev->num_initial_regions; i++) { + if (vfio_region_mmap(vdev->regions[i])) { + warn_report("%s mmap unsupported, performance may be slow", + memory_region_name(vdev->regions[i]->mem)); +diff --git a/include/hw/vfio/vfio-device.h b/include/hw/vfio/vfio-device.h +index 9290774299..df81d319b2 100644 +--- a/include/hw/vfio/vfio-device.h ++++ b/include/hw/vfio/vfio-device.h +@@ -74,7 +74,7 @@ typedef struct VFIODevice { + VFIODeviceOps *ops; + VFIODeviceIOOps *io_ops; + unsigned int num_irqs; +- unsigned int num_regions; ++ unsigned int num_initial_regions; + unsigned int flags; + VFIOMigration *migration; + Error *migration_blocker; +-- +2.47.3 + diff --git a/kvm-vhost-add-support-for-negotiating-extended-features.patch b/kvm-vhost-add-support-for-negotiating-extended-features.patch new file mode 100644 index 0000000..d3e50e5 --- /dev/null +++ b/kvm-vhost-add-support-for-negotiating-extended-features.patch @@ -0,0 +1,292 @@ +From 32cf8e1f3e289c67f49240e8301979d78801258b Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:22 +0200 +Subject: [PATCH 13/19] vhost: add support for negotiating extended features + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [8/14] 7546d4a0222e49da3a0536a703b89a015a80932c (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Similar to virtio infra, vhost core maintains the features status +in the full extended format and allows the devices to implement +extended version of the getter/setter. + +Note that 'protocol_features' are not extended: they are only +used by vhost-user, and the latter device is not going to implement +extended features soon. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 9f979ef0e01a2dd47d167c482a9e2d1dcdff2d3f) +Signed-off-by: Laurent Vivier +--- + hw/virtio/vhost.c | 68 ++++++++++++++++++++++--------- + include/hw/virtio/vhost-backend.h | 6 +++ + include/hw/virtio/vhost.h | 56 +++++++++++++++++++++---- + 3 files changed, 103 insertions(+), 27 deletions(-) + +diff --git a/hw/virtio/vhost.c b/hw/virtio/vhost.c +index 6557c58d12..5f485ad6cb 100644 +--- a/hw/virtio/vhost.c ++++ b/hw/virtio/vhost.c +@@ -972,20 +972,34 @@ static int vhost_virtqueue_set_addr(struct vhost_dev *dev, + static int vhost_dev_set_features(struct vhost_dev *dev, + bool enable_log) + { +- uint64_t features = dev->acked_features; ++ uint64_t features[VIRTIO_FEATURES_NU64S]; + int r; ++ ++ virtio_features_copy(features, dev->acked_features_ex); + if (enable_log) { +- features |= 0x1ULL << VHOST_F_LOG_ALL; ++ virtio_add_feature_ex(features, VHOST_F_LOG_ALL); + } + if (!vhost_dev_has_iommu(dev)) { +- features &= ~(0x1ULL << VIRTIO_F_IOMMU_PLATFORM); ++ virtio_clear_feature_ex(features, VIRTIO_F_IOMMU_PLATFORM); + } + if (dev->vhost_ops->vhost_force_iommu) { + if (dev->vhost_ops->vhost_force_iommu(dev) == true) { +- features |= 0x1ULL << VIRTIO_F_IOMMU_PLATFORM; ++ virtio_add_feature_ex(features, VIRTIO_F_IOMMU_PLATFORM); + } + } +- r = dev->vhost_ops->vhost_set_features(dev, features); ++ ++ if (virtio_features_use_ex(features) && ++ !dev->vhost_ops->vhost_set_features_ex) { ++ r = -EINVAL; ++ VHOST_OPS_DEBUG(r, "extended features without device support"); ++ goto out; ++ } ++ ++ if (dev->vhost_ops->vhost_set_features_ex) { ++ r = dev->vhost_ops->vhost_set_features_ex(dev, features); ++ } else { ++ r = dev->vhost_ops->vhost_set_features(dev, features[0]); ++ } + if (r < 0) { + VHOST_OPS_DEBUG(r, "vhost_set_features failed"); + goto out; +@@ -1508,12 +1522,27 @@ static void vhost_virtqueue_cleanup(struct vhost_virtqueue *vq) + } + } + ++static int vhost_dev_get_features(struct vhost_dev *hdev, ++ uint64_t *features) ++{ ++ uint64_t features64; ++ int r; ++ ++ if (hdev->vhost_ops->vhost_get_features_ex) { ++ return hdev->vhost_ops->vhost_get_features_ex(hdev, features); ++ } ++ ++ r = hdev->vhost_ops->vhost_get_features(hdev, &features64); ++ virtio_features_from_u64(features, features64); ++ return r; ++} ++ + int vhost_dev_init(struct vhost_dev *hdev, void *opaque, + VhostBackendType backend_type, uint32_t busyloop_timeout, + Error **errp) + { ++ uint64_t features[VIRTIO_FEATURES_NU64S]; + unsigned int used, reserved, limit; +- uint64_t features; + int i, r, n_initialized_vqs = 0; + + hdev->vdev = NULL; +@@ -1533,7 +1562,7 @@ int vhost_dev_init(struct vhost_dev *hdev, void *opaque, + goto fail; + } + +- r = hdev->vhost_ops->vhost_get_features(hdev, &features); ++ r = vhost_dev_get_features(hdev, features); + if (r < 0) { + error_setg_errno(errp, -r, "vhost_get_features failed"); + goto fail; +@@ -1571,7 +1600,7 @@ int vhost_dev_init(struct vhost_dev *hdev, void *opaque, + } + } + +- hdev->features = features; ++ virtio_features_copy(hdev->features_ex, features); + + hdev->memory_listener = (MemoryListener) { + .name = "vhost", +@@ -1594,7 +1623,7 @@ int vhost_dev_init(struct vhost_dev *hdev, void *opaque, + }; + + if (hdev->migration_blocker == NULL) { +- if (!(hdev->features & (0x1ULL << VHOST_F_LOG_ALL))) { ++ if (!virtio_has_feature_ex(hdev->features_ex, VHOST_F_LOG_ALL)) { + error_setg(&hdev->migration_blocker, + "Migration disabled: vhost lacks VHOST_F_LOG_ALL feature."); + } else if (vhost_dev_log_is_shared(hdev) && !qemu_memfd_alloc_check()) { +@@ -1859,28 +1888,27 @@ static void vhost_start_config_intr(struct vhost_dev *dev) + } + } + +-uint64_t vhost_get_features(struct vhost_dev *hdev, const int *feature_bits, +- uint64_t features) ++void vhost_get_features_ex(struct vhost_dev *hdev, ++ const int *feature_bits, ++ uint64_t *features) + { + const int *bit = feature_bits; ++ + while (*bit != VHOST_INVALID_FEATURE_BIT) { +- uint64_t bit_mask = (1ULL << *bit); +- if (!(hdev->features & bit_mask)) { +- features &= ~bit_mask; ++ if (!virtio_has_feature_ex(hdev->features_ex, *bit)) { ++ virtio_clear_feature_ex(features, *bit); + } + bit++; + } +- return features; + } + +-void vhost_ack_features(struct vhost_dev *hdev, const int *feature_bits, +- uint64_t features) ++void vhost_ack_features_ex(struct vhost_dev *hdev, const int *feature_bits, ++ const uint64_t *features) + { + const int *bit = feature_bits; + while (*bit != VHOST_INVALID_FEATURE_BIT) { +- uint64_t bit_mask = (1ULL << *bit); +- if (features & bit_mask) { +- hdev->acked_features |= bit_mask; ++ if (virtio_has_feature_ex(features, *bit)) { ++ virtio_add_feature_ex(hdev->acked_features_ex, *bit); + } + bit++; + } +diff --git a/include/hw/virtio/vhost-backend.h b/include/hw/virtio/vhost-backend.h +index d6df209a2f..ff94fa1734 100644 +--- a/include/hw/virtio/vhost-backend.h ++++ b/include/hw/virtio/vhost-backend.h +@@ -95,6 +95,10 @@ typedef int (*vhost_new_worker_op)(struct vhost_dev *dev, + struct vhost_worker_state *worker); + typedef int (*vhost_free_worker_op)(struct vhost_dev *dev, + struct vhost_worker_state *worker); ++typedef int (*vhost_set_features_ex_op)(struct vhost_dev *dev, ++ const uint64_t *features); ++typedef int (*vhost_get_features_ex_op)(struct vhost_dev *dev, ++ uint64_t *features); + typedef int (*vhost_set_features_op)(struct vhost_dev *dev, + uint64_t features); + typedef int (*vhost_get_features_op)(struct vhost_dev *dev, +@@ -186,6 +190,8 @@ typedef struct VhostOps { + vhost_free_worker_op vhost_free_worker; + vhost_get_vring_worker_op vhost_get_vring_worker; + vhost_attach_vring_worker_op vhost_attach_vring_worker; ++ vhost_set_features_ex_op vhost_set_features_ex; ++ vhost_get_features_ex_op vhost_get_features_ex; + vhost_set_features_op vhost_set_features; + vhost_get_features_op vhost_get_features; + vhost_set_backend_cap_op vhost_set_backend_cap; +diff --git a/include/hw/virtio/vhost.h b/include/hw/virtio/vhost.h +index 66be6afc88..08bbb4dfe9 100644 +--- a/include/hw/virtio/vhost.h ++++ b/include/hw/virtio/vhost.h +@@ -107,9 +107,9 @@ struct vhost_dev { + * future use should be discouraged and the variable retired as + * its easy to confuse with the VirtIO backend_features. + */ +- uint64_t features; +- uint64_t acked_features; +- uint64_t backend_features; ++ VIRTIO_DECLARE_FEATURES(features); ++ VIRTIO_DECLARE_FEATURES(acked_features); ++ VIRTIO_DECLARE_FEATURES(backend_features); + + /** + * @protocol_features: is the vhost-user only feature set by +@@ -320,6 +320,20 @@ bool vhost_virtqueue_pending(struct vhost_dev *hdev, int n); + void vhost_virtqueue_mask(struct vhost_dev *hdev, VirtIODevice *vdev, int n, + bool mask); + ++/** ++ * vhost_get_features_ex() - sanitize the extended features set ++ * @hdev: common vhost_dev structure ++ * @feature_bits: pointer to terminated table of feature bits ++ * @features: original features set, filtered out on return ++ * ++ * This is the extended variant of vhost_get_features(), supporting the ++ * the extended features set. Filter it with the intersection of what is ++ * supported by the vhost backend (hdev->features) and the supported ++ * feature_bits. ++ */ ++void vhost_get_features_ex(struct vhost_dev *hdev, ++ const int *feature_bits, ++ uint64_t *features); + /** + * vhost_get_features() - return a sanitised set of feature bits + * @hdev: common vhost_dev structure +@@ -330,8 +344,28 @@ void vhost_virtqueue_mask(struct vhost_dev *hdev, VirtIODevice *vdev, int n, + * is supported by the vhost backend (hdev->features), the supported + * feature_bits and the requested feature set. + */ +-uint64_t vhost_get_features(struct vhost_dev *hdev, const int *feature_bits, +- uint64_t features); ++static inline uint64_t vhost_get_features(struct vhost_dev *hdev, ++ const int *feature_bits, ++ uint64_t features) ++{ ++ uint64_t features_ex[VIRTIO_FEATURES_NU64S]; ++ ++ virtio_features_from_u64(features_ex, features); ++ vhost_get_features_ex(hdev, feature_bits, features_ex); ++ return features_ex[0]; ++} ++ ++/** ++ * vhost_ack_features_ex() - set vhost full set of acked_features ++ * @hdev: common vhost_dev structure ++ * @feature_bits: pointer to terminated table of feature bits ++ * @features: requested feature set ++ * ++ * This sets the internal hdev->acked_features to the intersection of ++ * the backends advertised features and the supported feature_bits. ++ */ ++void vhost_ack_features_ex(struct vhost_dev *hdev, const int *feature_bits, ++ const uint64_t *features); + + /** + * vhost_ack_features() - set vhost acked_features +@@ -342,8 +376,16 @@ uint64_t vhost_get_features(struct vhost_dev *hdev, const int *feature_bits, + * This sets the internal hdev->acked_features to the intersection of + * the backends advertised features and the supported feature_bits. + */ +-void vhost_ack_features(struct vhost_dev *hdev, const int *feature_bits, +- uint64_t features); ++static inline void vhost_ack_features(struct vhost_dev *hdev, ++ const int *feature_bits, ++ uint64_t features) ++{ ++ uint64_t features_ex[VIRTIO_FEATURES_NU64S]; ++ ++ virtio_features_from_u64(features_ex, features); ++ vhost_ack_features_ex(hdev, feature_bits, features_ex); ++} ++ + unsigned int vhost_get_max_memslots(void); + unsigned int vhost_get_free_memslots(void); + +-- +2.47.3 + diff --git a/kvm-vhost-backend-implement-extended-features-support.patch b/kvm-vhost-backend-implement-extended-features-support.patch new file mode 100644 index 0000000..fe43faa --- /dev/null +++ b/kvm-vhost-backend-implement-extended-features-support.patch @@ -0,0 +1,133 @@ +From d0e31af77cd51df8c17fa4d304571f8b1eb5ce29 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:24 +0200 +Subject: [PATCH 15/19] vhost-backend: implement extended features support + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [10/14] bd12c2f08dad502a7359d4344a2220b10ac96a48 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Leverage the kernel extended features manipulation ioctls(), if +available, and fallback to old ops otherwise. Error out when setting +extended features but kernel support is not available. + +Note that extended support for get/set backend features is not needed, +as the only feature that can be changed belongs to the 64 bit range. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: <150daade3d59e77629276920e014ee8e5fc12121.1758549625.git.pabeni@redhat.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit f412c1f57ab58fd595efee26264193759220ca6f) +Signed-off-by: Laurent Vivier +--- + hw/virtio/vhost-backend.c | 62 ++++++++++++++++++++++++++++++++------- + 1 file changed, 51 insertions(+), 11 deletions(-) + +diff --git a/hw/virtio/vhost-backend.c b/hw/virtio/vhost-backend.c +index 833804dd40..4367db0d95 100644 +--- a/hw/virtio/vhost-backend.c ++++ b/hw/virtio/vhost-backend.c +@@ -20,6 +20,11 @@ + #include + #include + ++struct vhost_features { ++ uint64_t count; ++ uint64_t features[VIRTIO_FEATURES_NU64S]; ++}; ++ + static int vhost_kernel_call(struct vhost_dev *dev, unsigned long int request, + void *arg) + { +@@ -182,12 +187,6 @@ static int vhost_kernel_get_vring_worker(struct vhost_dev *dev, + return vhost_kernel_call(dev, VHOST_GET_VRING_WORKER, worker); + } + +-static int vhost_kernel_set_features(struct vhost_dev *dev, +- uint64_t features) +-{ +- return vhost_kernel_call(dev, VHOST_SET_FEATURES, &features); +-} +- + static int vhost_kernel_set_backend_cap(struct vhost_dev *dev) + { + uint64_t features; +@@ -210,10 +209,51 @@ static int vhost_kernel_set_backend_cap(struct vhost_dev *dev) + return 0; + } + +-static int vhost_kernel_get_features(struct vhost_dev *dev, +- uint64_t *features) ++static int vhost_kernel_set_features(struct vhost_dev *dev, ++ const uint64_t *features) + { +- return vhost_kernel_call(dev, VHOST_GET_FEATURES, features); ++ struct vhost_features farray; ++ bool extended_in_use; ++ int r; ++ ++ farray.count = VIRTIO_FEATURES_NU64S; ++ virtio_features_copy(farray.features, features); ++ extended_in_use = virtio_features_use_ex(farray.features); ++ ++ /* ++ * Can't check for ENOTTY: for unknown ioctls the kernel interprets ++ * the argument as a virtio queue id and most likely errors out validating ++ * such id, instead of reporting an unknown operation. ++ */ ++ r = vhost_kernel_call(dev, VHOST_SET_FEATURES_ARRAY, &farray); ++ if (!r) { ++ return 0; ++ } ++ ++ if (extended_in_use) { ++ error_report("Trying to set extended features without kernel support"); ++ return -EINVAL; ++ } ++ return vhost_kernel_call(dev, VHOST_SET_FEATURES, &farray.features[0]); ++} ++ ++static int vhost_kernel_get_features(struct vhost_dev *dev, uint64_t *features) ++{ ++ struct vhost_features farray; ++ int r; ++ ++ farray.count = VIRTIO_FEATURES_NU64S; ++ r = vhost_kernel_call(dev, VHOST_GET_FEATURES_ARRAY, &farray); ++ if (r) { ++ memset(&farray, 0, sizeof(farray)); ++ r = vhost_kernel_call(dev, VHOST_GET_FEATURES, &farray.features[0]); ++ } ++ if (r) { ++ return r; ++ } ++ ++ virtio_features_copy(features, farray.features); ++ return 0; + } + + static int vhost_kernel_set_owner(struct vhost_dev *dev) +@@ -341,8 +381,8 @@ const VhostOps kernel_ops = { + .vhost_attach_vring_worker = vhost_kernel_attach_vring_worker, + .vhost_new_worker = vhost_kernel_new_worker, + .vhost_free_worker = vhost_kernel_free_worker, +- .vhost_set_features = vhost_kernel_set_features, +- .vhost_get_features = vhost_kernel_get_features, ++ .vhost_set_features_ex = vhost_kernel_set_features, ++ .vhost_get_features_ex = vhost_kernel_get_features, + .vhost_set_backend_cap = vhost_kernel_set_backend_cap, + .vhost_set_owner = vhost_kernel_set_owner, + .vhost_get_vq_index = vhost_kernel_get_vq_index, +-- +2.47.3 + diff --git a/kvm-vhost-net-implement-extended-features-support.patch b/kvm-vhost-net-implement-extended-features-support.patch new file mode 100644 index 0000000..bf04384 --- /dev/null +++ b/kvm-vhost-net-implement-extended-features-support.patch @@ -0,0 +1,237 @@ +From 7eb0a4d5d5a03fd631b87c59dc531080f93a0640 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:25 +0200 +Subject: [PATCH 16/19] vhost-net: implement extended features support + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [11/14] 48cf0458bd30eeca10dace74ca2a8871dc0c8bca (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Provide extended version of the features manipulation helpers, +and let the device initialization deal with the full features space, +adjusting the relevant format strings accordingly. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Acked-by: Stefano Garzarella +Reviewed-by: Michael S. Tsirkin +Message-ID: <69c78c432e28e146a8874b2a7d00e9cbd111b1ba.1758549625.git.pabeni@redhat.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit d55ad8c9a9a5edd8152f13fc97879d66972f103e) +Signed-off-by: Laurent Vivier +--- + hw/net/vhost_net-stub.c | 8 +++----- + hw/net/vhost_net.c | 45 +++++++++++++++++++++++------------------ + include/net/vhost_net.h | 33 +++++++++++++++++++++++++++--- + 3 files changed, 58 insertions(+), 28 deletions(-) + +diff --git a/hw/net/vhost_net-stub.c b/hw/net/vhost_net-stub.c +index 7d49f82906..0740d5a2eb 100644 +--- a/hw/net/vhost_net-stub.c ++++ b/hw/net/vhost_net-stub.c +@@ -46,9 +46,8 @@ void vhost_net_cleanup(struct vhost_net *net) + { + } + +-uint64_t vhost_net_get_features(struct vhost_net *net, uint64_t features) ++void vhost_net_get_features_ex(struct vhost_net *net, uint64_t *features) + { +- return features; + } + + int vhost_net_get_config(struct vhost_net *net, uint8_t *config, +@@ -62,13 +61,12 @@ int vhost_net_set_config(struct vhost_net *net, const uint8_t *data, + return 0; + } + +-void vhost_net_ack_features(struct vhost_net *net, uint64_t features) ++void vhost_net_ack_features_ex(struct vhost_net *net, const uint64_t *features) + { + } + +-uint64_t vhost_net_get_acked_features(VHostNetState *net) ++void vhost_net_get_acked_features_ex(VHostNetState *net, uint64_t *features) + { +- return 0; + } + + bool vhost_net_virtqueue_pending(VHostNetState *net, int idx) +diff --git a/hw/net/vhost_net.c b/hw/net/vhost_net.c +index 540492b37d..a8ee18a912 100644 +--- a/hw/net/vhost_net.c ++++ b/hw/net/vhost_net.c +@@ -35,10 +35,9 @@ + #include "hw/virtio/virtio-bus.h" + #include "linux-headers/linux/vhost.h" + +-uint64_t vhost_net_get_features(struct vhost_net *net, uint64_t features) ++void vhost_net_get_features_ex(struct vhost_net *net, uint64_t *features) + { +- return vhost_get_features(&net->dev, net->feature_bits, +- features); ++ vhost_get_features_ex(&net->dev, net->feature_bits, features); + } + int vhost_net_get_config(struct vhost_net *net, uint8_t *config, + uint32_t config_len) +@@ -51,10 +50,11 @@ int vhost_net_set_config(struct vhost_net *net, const uint8_t *data, + return vhost_dev_set_config(&net->dev, data, offset, size, flags); + } + +-void vhost_net_ack_features(struct vhost_net *net, uint64_t features) ++void vhost_net_ack_features_ex(struct vhost_net *net, const uint64_t *features) + { +- net->dev.acked_features = net->dev.backend_features; +- vhost_ack_features(&net->dev, net->feature_bits, features); ++ virtio_features_copy(net->dev.acked_features_ex, ++ net->dev.backend_features_ex); ++ vhost_ack_features_ex(&net->dev, net->feature_bits, features); + } + + uint64_t vhost_net_get_max_queues(VHostNetState *net) +@@ -62,9 +62,9 @@ uint64_t vhost_net_get_max_queues(VHostNetState *net) + return net->dev.max_queues; + } + +-uint64_t vhost_net_get_acked_features(VHostNetState *net) ++void vhost_net_get_acked_features_ex(VHostNetState *net, uint64_t *features) + { +- return net->dev.acked_features; ++ virtio_features_copy(features, net->dev.acked_features_ex); + } + + void vhost_net_save_acked_features(NetClientState *nc) +@@ -234,7 +234,8 @@ struct vhost_net *vhost_net_init(VhostNetOptions *options) + int r; + bool backend_kernel = options->backend_type == VHOST_BACKEND_TYPE_KERNEL; + struct vhost_net *net = g_new0(struct vhost_net, 1); +- uint64_t features = 0; ++ uint64_t missing_features[VIRTIO_FEATURES_NU64S]; ++ uint64_t features[VIRTIO_FEATURES_NU64S]; + Error *local_err = NULL; + + if (!options->net_backend) { +@@ -247,6 +248,7 @@ struct vhost_net *vhost_net_init(VhostNetOptions *options) + net->save_acked_features = options->save_acked_features; + net->max_tx_queue_size = options->max_tx_queue_size; + net->is_vhost_user = options->is_vhost_user; ++ virtio_features_clear(features); + + net->dev.max_queues = 1; + net->dev.vqs = net->vqs; +@@ -261,7 +263,7 @@ struct vhost_net *vhost_net_init(VhostNetOptions *options) + net->backend = r; + net->dev.protocol_features = 0; + } else { +- net->dev.backend_features = 0; ++ virtio_features_clear(net->dev.backend_features_ex); + net->dev.protocol_features = 0; + net->backend = -1; + +@@ -281,26 +283,29 @@ struct vhost_net *vhost_net_init(VhostNetOptions *options) + sizeof(struct virtio_net_hdr_mrg_rxbuf))) { + net->dev.features &= ~(1ULL << VIRTIO_NET_F_MRG_RXBUF); + } +- if (~net->dev.features & net->dev.backend_features) { +- fprintf(stderr, "vhost lacks feature mask 0x%" PRIx64 +- " for backend\n", +- (uint64_t)(~net->dev.features & net->dev.backend_features)); ++ ++ if (virtio_features_andnot(missing_features, ++ net->dev.backend_features_ex, ++ net->dev.features_ex)) { ++ fprintf(stderr, "vhost lacks feature mask 0x" VIRTIO_FEATURES_FMT ++ " for backend\n", VIRTIO_FEATURES_PR(missing_features)); + goto fail; + } + } + + /* Set sane init value. Override when guest acks. */ + if (options->get_acked_features) { +- features = options->get_acked_features(net->nc); +- if (~net->dev.features & features) { +- fprintf(stderr, "vhost lacks feature mask 0x%" PRIx64 +- " for backend\n", +- (uint64_t)(~net->dev.features & features)); ++ virtio_features_from_u64(features, ++ options->get_acked_features(net->nc)); ++ if (virtio_features_andnot(missing_features, features, ++ net->dev.features_ex)) { ++ fprintf(stderr, "vhost lacks feature mask 0x" VIRTIO_FEATURES_FMT ++ " for backend\n", VIRTIO_FEATURES_PR(missing_features)); + goto fail; + } + } + +- vhost_net_ack_features(net, features); ++ vhost_net_ack_features_ex(net, features); + + return net; + +diff --git a/include/net/vhost_net.h b/include/net/vhost_net.h +index 879781dad7..0225207491 100644 +--- a/include/net/vhost_net.h ++++ b/include/net/vhost_net.h +@@ -2,6 +2,7 @@ + #define VHOST_NET_H + + #include "net/net.h" ++#include "hw/virtio/virtio-features.h" + #include "hw/virtio/vhost-backend.h" + + struct vhost_net; +@@ -33,8 +34,26 @@ void vhost_net_stop(VirtIODevice *dev, NetClientState *ncs, + + void vhost_net_cleanup(VHostNetState *net); + +-uint64_t vhost_net_get_features(VHostNetState *net, uint64_t features); +-void vhost_net_ack_features(VHostNetState *net, uint64_t features); ++void vhost_net_get_features_ex(VHostNetState *net, uint64_t *features); ++static inline uint64_t vhost_net_get_features(VHostNetState *net, ++ uint64_t features) ++{ ++ uint64_t features_array[VIRTIO_FEATURES_NU64S]; ++ ++ virtio_features_from_u64(features_array, features); ++ vhost_net_get_features_ex(net, features_array); ++ return features_array[0]; ++} ++ ++void vhost_net_ack_features_ex(VHostNetState *net, const uint64_t *features); ++static inline void vhost_net_ack_features(VHostNetState *net, ++ uint64_t features) ++{ ++ uint64_t features_array[VIRTIO_FEATURES_NU64S]; ++ ++ virtio_features_from_u64(features_array, features); ++ vhost_net_ack_features_ex(net, features_array); ++} + + int vhost_net_get_config(struct vhost_net *net, uint8_t *config, + uint32_t config_len); +@@ -51,7 +70,15 @@ VHostNetState *get_vhost_net(NetClientState *nc); + + int vhost_net_set_vring_enable(NetClientState *nc, int enable); + +-uint64_t vhost_net_get_acked_features(VHostNetState *net); ++void vhost_net_get_acked_features_ex(VHostNetState *net, uint64_t *features); ++static inline uint64_t vhost_net_get_acked_features(VHostNetState *net) ++{ ++ uint64_t features[VIRTIO_FEATURES_NU64S]; ++ ++ vhost_net_get_acked_features_ex(net, features); ++ assert(!virtio_features_use_ex(features)); ++ return features[0]; ++} + + int vhost_net_set_mtu(struct vhost_net *net, uint16_t mtu); + +-- +2.47.3 + diff --git a/kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch b/kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch new file mode 100644 index 0000000..a6a8242 --- /dev/null +++ b/kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch @@ -0,0 +1,123 @@ +From 2643a61dd6de41945d714aac210173754a7b5a7f Mon Sep 17 00:00:00 2001 +From: German Maglione +Date: Wed, 22 Oct 2025 18:24:05 +0200 +Subject: [PATCH 1/7] vhost-user: make vhost_set_vring_file() synchronous +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Hanna Czenczek +RH-MergeRequest: 462: vhost-user: make vhost_set_vring_file() synchronous +RH-Jira: RHEL-147425 +RH-Acked-by: Stefano Garzarella +RH-Acked-by: Eugenio Pérez +RH-Commit: [1/1] 92476878c6fa5cb8ac1b308d2c8d4275767fd780 (hreitz/qemu-kvm-c-9-s) + +QEMU sends all of VHOST_USER_SET_VRING_KICK, _CALL, and _ERR without +setting the NEED_REPLY flag, i.e. by the time the respective +vhost_user_set_vring_*() function returns, it is completely up to chance +whether the back-end has already processed the request and switched over +to the new FD for interrupts. + +At least for vhost_user_set_vring_call(), that is a problem: It is +called through vhost_virtqueue_mask(), which is generally used in the +VirtioDeviceClass.guest_notifier_mask() implementation, which is in turn +called by virtio_pci_one_vector_unmask(). The fact that we do not wait +for the back-end to install the FD leads to a race there: + +Masking interrupts is implemented by redirecting interrupts to an +internal event FD that is not connected to the guest. Unmasking then +re-installs the guest-connected IRQ FD, then checks if there are pending +interrupts left on the masked event FD, and if so, issues an interrupt +to the guest. + +Because guest_notifier_mask() (through vhost_user_set_vring_call()) +doesn't wait for the back-end to switch over to the actual IRQ FD, it's +possible we check for pending interrupts while the back-end is still +using the masked event FD, and then we will lose interrupts that occur +before the back-end finally does switch over. + +Fix this by setting NEED_REPLY on those VHOST_USER_SET_VRING_* messages, +so when we get that reply, we know that the back-end is now using the +new FD. + +We have a few reports of a virtiofs mount hanging: +- https://gitlab.com/virtio-fs/virtiofsd/-/issues/101 +- https://gitlab.com/virtio-fs/virtiofsd/-/issues/133 +- https://gitlab.com/virtio-fs/virtiofsd/-/issues/213 + +This is quite difficult bug to reproduce, even for the reporters. +It only happens on production, every few weeks, and/or on 1 in 300 VMs. +So, we are not 100% sure this fixes that issue. However, we think this +is still a bug, and at least we have one report that claims this fixed +the issue: + +https://gitlab.com/virtio-fs/virtiofsd/-/issues/133#note_2743209419 + +Fixes: 5f6f6664bf24 ("Add vhost-user as a vhost backend.") +Signed-off-by: German Maglione +Signed-off-by: Hanna Czenczek +Reviewed-by: Eugenio Pérez +Reviewed-by: Stefano Garzarella +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +Message-Id: <20251022162405.318672-1-gmaglione@redhat.com> +(cherry picked from commit 1ba9a5220325dd5260a0c37b6299ce38364a5120) +Signed-off-by: Hanna Czenczek +--- + hw/virtio/vhost-user.c | 24 +++++++++++++++++++++++- + 1 file changed, 23 insertions(+), 1 deletion(-) + +diff --git a/hw/virtio/vhost-user.c b/hw/virtio/vhost-user.c +index 1e1d6b0d6e..3e1e1d8d7e 100644 +--- a/hw/virtio/vhost-user.c ++++ b/hw/virtio/vhost-user.c +@@ -1327,8 +1327,11 @@ static int vhost_set_vring_file(struct vhost_dev *dev, + VhostUserRequest request, + struct vhost_vring_file *file) + { ++ int ret; + int fds[VHOST_USER_MAX_RAM_SLOTS]; + size_t fd_num = 0; ++ bool reply_supported = virtio_has_feature(dev->protocol_features, ++ VHOST_USER_PROTOCOL_F_REPLY_ACK); + VhostUserMsg msg = { + .hdr.request = request, + .hdr.flags = VHOST_USER_VERSION, +@@ -1336,13 +1339,32 @@ static int vhost_set_vring_file(struct vhost_dev *dev, + .hdr.size = sizeof(msg.payload.u64), + }; + ++ if (reply_supported) { ++ msg.hdr.flags |= VHOST_USER_NEED_REPLY_MASK; ++ } ++ + if (file->fd > 0) { + fds[fd_num++] = file->fd; + } else { + msg.payload.u64 |= VHOST_USER_VRING_NOFD_MASK; + } + +- return vhost_user_write(dev, &msg, fds, fd_num); ++ ret = vhost_user_write(dev, &msg, fds, fd_num); ++ if (ret < 0) { ++ return ret; ++ } ++ ++ if (reply_supported) { ++ /* ++ * wait for the back-end's confirmation that the new FD is active, ++ * otherwise guest_notifier_mask() could check for pending interrupts ++ * while the back-end is still using the masked event FD, losing ++ * interrupts that occur before the back-end installs the FD ++ */ ++ return process_message_reply(dev, &msg); ++ } ++ ++ return 0; + } + + static int vhost_user_set_vring_kick(struct vhost_dev *dev, +-- +2.47.3 + diff --git a/kvm-virtio-add-support-for-negotiating-extended-features.patch b/kvm-virtio-add-support-for-negotiating-extended-features.patch new file mode 100644 index 0000000..53a5d72 --- /dev/null +++ b/kvm-virtio-add-support-for-negotiating-extended-features.patch @@ -0,0 +1,133 @@ +From d7e0acde126d1f5385096d6b7247de82ee6b0e8d Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:20 +0200 +Subject: [PATCH 11/19] virtio: add support for negotiating extended features + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [6/14] 56553123a4941b457a34af8be01b5757a97706e6 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +The virtio specifications allows for a device features space up +to 128 bits and more. Soon we are going to use some of the 'extended' +bits features for the virtio net driver. + +Add support to allow extended features negotiation on a per +devices basis. Devices willing to negotiated extended features +need to implemented a new pair of features getter/setter, the +core will conditionally use them instead of the basic one. + +Note that 'bad_features' don't need to be extended, as they are +bound to the 64 bits limit. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: <9bb29d70adc3f2b8c7756d4e3cd076cffee87826.1758549625.git.pabeni@redhat.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 64a6a336f42bc6305ab7589fd874cb4a3d403bd0) +Signed-off-by: Laurent Vivier +--- + hw/virtio/virtio-bus.c | 11 ++++++++--- + hw/virtio/virtio.c | 14 +++++++++++--- + include/hw/virtio/virtio.h | 4 ++++ + 3 files changed, 23 insertions(+), 6 deletions(-) + +diff --git a/hw/virtio/virtio-bus.c b/hw/virtio/virtio-bus.c +index 11adfbf3ab..cef944e015 100644 +--- a/hw/virtio/virtio-bus.c ++++ b/hw/virtio/virtio-bus.c +@@ -62,9 +62,14 @@ void virtio_bus_device_plugged(VirtIODevice *vdev, Error **errp) + } + + /* Get the features of the plugged device. */ +- assert(vdc->get_features != NULL); +- vdev->host_features = vdc->get_features(vdev, vdev->host_features, +- &local_err); ++ if (vdc->get_features_ex) { ++ vdc->get_features_ex(vdev, vdev->host_features_ex, &local_err); ++ } else { ++ assert(vdc->get_features != NULL); ++ virtio_features_from_u64(vdev->host_features_ex, ++ vdc->get_features(vdev, vdev->host_features, ++ &local_err)); ++ } + if (local_err) { + error_propagate(errp, local_err); + return; +diff --git a/hw/virtio/virtio.c b/hw/virtio/virtio.c +index bf53c211e5..34f977a3c9 100644 +--- a/hw/virtio/virtio.c ++++ b/hw/virtio/virtio.c +@@ -3103,7 +3103,9 @@ static int virtio_set_features_nocheck(VirtIODevice *vdev, const uint64_t *val) + bad = virtio_features_andnot(tmp, val, vdev->host_features_ex); + virtio_features_and(tmp, val, vdev->host_features_ex); + +- if (k->set_features) { ++ if (k->set_features_ex) { ++ k->set_features_ex(vdev, val); ++ } else if (k->set_features) { + bad = bad || virtio_features_use_ex(tmp); + k->set_features(vdev, tmp[0]); + } +@@ -3149,6 +3151,13 @@ virtio_set_features_nocheck_maybe_co(VirtIODevice *vdev, + int virtio_set_features(VirtIODevice *vdev, uint64_t val) + { + uint64_t features[VIRTIO_FEATURES_NU64S]; ++ ++ virtio_features_from_u64(features, val); ++ return virtio_set_features_ex(vdev, features); ++} ++ ++int virtio_set_features_ex(VirtIODevice *vdev, const uint64_t *features) ++{ + int ret; + /* + * The driver must not attempt to set features after feature negotiation +@@ -3158,13 +3167,12 @@ int virtio_set_features(VirtIODevice *vdev, uint64_t val) + return -EINVAL; + } + +- if (val & (1ull << VIRTIO_F_BAD_FEATURE)) { ++ if (features[0] & (1ull << VIRTIO_F_BAD_FEATURE)) { + qemu_log_mask(LOG_GUEST_ERROR, + "%s: guest driver for %s has enabled UNUSED(30) feature bit!\n", + __func__, vdev->name); + } + +- virtio_features_from_u64(features, val); + ret = virtio_set_features_nocheck(vdev, features); + if (virtio_vdev_has_feature(vdev, VIRTIO_RING_F_EVENT_IDX)) { + /* VIRTIO_RING_F_EVENT_IDX changes the size of the caches. */ +diff --git a/include/hw/virtio/virtio.h b/include/hw/virtio/virtio.h +index 39e4059a66..2aeb021fb3 100644 +--- a/include/hw/virtio/virtio.h ++++ b/include/hw/virtio/virtio.h +@@ -178,6 +178,9 @@ struct VirtioDeviceClass { + /* This is what a VirtioDevice must implement */ + DeviceRealize realize; + DeviceUnrealize unrealize; ++ void (*get_features_ex)(VirtIODevice *vdev, uint64_t *requested_features, ++ Error **errp); ++ void (*set_features_ex)(VirtIODevice *vdev, const uint64_t *val); + uint64_t (*get_features)(VirtIODevice *vdev, + uint64_t requested_features, + Error **errp); +@@ -373,6 +376,7 @@ void virtio_queue_reset(VirtIODevice *vdev, uint32_t queue_index); + void virtio_queue_enable(VirtIODevice *vdev, uint32_t queue_index); + void virtio_update_irq(VirtIODevice *vdev); + int virtio_set_features(VirtIODevice *vdev, uint64_t val); ++int virtio_set_features_ex(VirtIODevice *vdev, const uint64_t *val); + + /* Base devices. */ + typedef struct VirtIOBlkConf VirtIOBlkConf; +-- +2.47.3 + diff --git a/kvm-virtio-gpu-fix-v2-migration.patch b/kvm-virtio-gpu-fix-v2-migration.patch deleted file mode 100644 index 4c8e609..0000000 --- a/kvm-virtio-gpu-fix-v2-migration.patch +++ /dev/null @@ -1,122 +0,0 @@ -From 77e24d71549454d7d7b9e83f882e2817a5da7fac Mon Sep 17 00:00:00 2001 -From: =?UTF-8?q?Marc-Andr=C3=A9=20Lureau?= -Date: Thu, 16 May 2024 12:40:22 +0400 -Subject: [PATCH 06/14] virtio-gpu: fix v2 migration -MIME-Version: 1.0 -Content-Type: text/plain; charset=UTF-8 -Content-Transfer-Encoding: 8bit - -RH-Author: Marc-André Lureau -RH-MergeRequest: 250: virtio-gpu: fix v2 migration -RH-Jira: RHEL-36329 -RH-Acked-by: Peter Xu -RH-Acked-by: Miroslav Rezanina -RH-Commit: [1/2] 55624c9074aaf1226ca3ae8a34744134cd8a4d9f (marcandre.lureau-rh/qemu-kvm-centos) - -Commit dfcf74fa ("virtio-gpu: fix scanout migration post-load") broke -forward/backward version migration. Versioning of nested VMSD structures -is not straightforward, as the wire format doesn't have nested -structures versions. Introduce x-scanout-vmstate-version and a field -test to save/load appropriately according to the machine version. - -Fixes: dfcf74fa ("virtio-gpu: fix scanout migration post-load") -Signed-off-by: Marc-André Lureau -Signed-off-by: Peter Xu -Reviewed-by: Fiona Ebner -Tested-by: Fiona Ebner -[fixed long lines] -Signed-off-by: Fabiano Rosas - -Jira: https://issues.redhat.com/browse/RHEL-36329 -Signed-off-by: Marc-André Lureau -(cherry picked from commit 40a23ef643664b5c1021a9789f9d680b6294fb50) ---- - hw/core/machine.c | 1 + - hw/display/virtio-gpu.c | 30 ++++++++++++++++++++++-------- - include/hw/virtio/virtio-gpu.h | 1 + - 3 files changed, 24 insertions(+), 8 deletions(-) - -diff --git a/hw/core/machine.c b/hw/core/machine.c -index 0f256d9633..cf1d7faaaf 100644 ---- a/hw/core/machine.c -+++ b/hw/core/machine.c -@@ -37,6 +37,7 @@ GlobalProperty hw_compat_8_2[] = { - { "migration", "zero-page-detection", "legacy"}, - { TYPE_VIRTIO_IOMMU_PCI, "granule", "4k" }, - { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "64" }, -+ { "virtio-gpu-device", "x-scanout-vmstate-version", "1" }, - }; - const size_t hw_compat_8_2_len = G_N_ELEMENTS(hw_compat_8_2); - -diff --git a/hw/display/virtio-gpu.c b/hw/display/virtio-gpu.c -index ae831b6b3e..d60b1b2973 100644 ---- a/hw/display/virtio-gpu.c -+++ b/hw/display/virtio-gpu.c -@@ -1166,10 +1166,17 @@ static void virtio_gpu_cursor_bh(void *opaque) - virtio_gpu_handle_cursor(&g->parent_obj.parent_obj, g->cursor_vq); - } - -+static bool scanout_vmstate_after_v2(void *opaque, int version) -+{ -+ struct VirtIOGPUBase *base = container_of(opaque, VirtIOGPUBase, scanout); -+ struct VirtIOGPU *gpu = container_of(base, VirtIOGPU, parent_obj); -+ -+ return gpu->scanout_vmstate_version >= 2; -+} -+ - static const VMStateDescription vmstate_virtio_gpu_scanout = { - .name = "virtio-gpu-one-scanout", -- .version_id = 2, -- .minimum_version_id = 1, -+ .version_id = 1, - .fields = (const VMStateField[]) { - VMSTATE_UINT32(resource_id, struct virtio_gpu_scanout), - VMSTATE_UINT32(width, struct virtio_gpu_scanout), -@@ -1181,12 +1188,18 @@ static const VMStateDescription vmstate_virtio_gpu_scanout = { - VMSTATE_UINT32(cursor.hot_y, struct virtio_gpu_scanout), - VMSTATE_UINT32(cursor.pos.x, struct virtio_gpu_scanout), - VMSTATE_UINT32(cursor.pos.y, struct virtio_gpu_scanout), -- VMSTATE_UINT32_V(fb.format, struct virtio_gpu_scanout, 2), -- VMSTATE_UINT32_V(fb.bytes_pp, struct virtio_gpu_scanout, 2), -- VMSTATE_UINT32_V(fb.width, struct virtio_gpu_scanout, 2), -- VMSTATE_UINT32_V(fb.height, struct virtio_gpu_scanout, 2), -- VMSTATE_UINT32_V(fb.stride, struct virtio_gpu_scanout, 2), -- VMSTATE_UINT32_V(fb.offset, struct virtio_gpu_scanout, 2), -+ VMSTATE_UINT32_TEST(fb.format, struct virtio_gpu_scanout, -+ scanout_vmstate_after_v2), -+ VMSTATE_UINT32_TEST(fb.bytes_pp, struct virtio_gpu_scanout, -+ scanout_vmstate_after_v2), -+ VMSTATE_UINT32_TEST(fb.width, struct virtio_gpu_scanout, -+ scanout_vmstate_after_v2), -+ VMSTATE_UINT32_TEST(fb.height, struct virtio_gpu_scanout, -+ scanout_vmstate_after_v2), -+ VMSTATE_UINT32_TEST(fb.stride, struct virtio_gpu_scanout, -+ scanout_vmstate_after_v2), -+ VMSTATE_UINT32_TEST(fb.offset, struct virtio_gpu_scanout, -+ scanout_vmstate_after_v2), - VMSTATE_END_OF_LIST() - }, - }; -@@ -1659,6 +1672,7 @@ static Property virtio_gpu_properties[] = { - DEFINE_PROP_BIT("blob", VirtIOGPU, parent_obj.conf.flags, - VIRTIO_GPU_FLAG_BLOB_ENABLED, false), - DEFINE_PROP_SIZE("hostmem", VirtIOGPU, parent_obj.conf.hostmem, 0), -+ DEFINE_PROP_UINT8("x-scanout-vmstate-version", VirtIOGPU, scanout_vmstate_version, 2), - DEFINE_PROP_END_OF_LIST(), - }; - -diff --git a/include/hw/virtio/virtio-gpu.h b/include/hw/virtio/virtio-gpu.h -index ed44cdad6b..842315d51d 100644 ---- a/include/hw/virtio/virtio-gpu.h -+++ b/include/hw/virtio/virtio-gpu.h -@@ -177,6 +177,7 @@ typedef struct VGPUDMABuf { - struct VirtIOGPU { - VirtIOGPUBase parent_obj; - -+ uint8_t scanout_vmstate_version; - uint64_t conf_max_hostmem; - - VirtQueue *ctrl_vq; --- -2.39.3 - diff --git a/kvm-virtio-introduce-extended-features-type.patch b/kvm-virtio-introduce-extended-features-type.patch new file mode 100644 index 0000000..caf0e14 --- /dev/null +++ b/kvm-virtio-introduce-extended-features-type.patch @@ -0,0 +1,201 @@ +From 4c9ae4b3a9e2c556dcf284924a154ed85534d77f Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:18 +0200 +Subject: [PATCH 09/19] virtio: introduce extended features type + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [4/14] fa4ef2d101589c700358f90be08943f1198503e4 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +The virtio specifications allows for up to 128 bits for the +device features. Soon we are going to use some of the 'extended' +bits features (bit 64 and above) for the virtio net driver. + +Represent the virtio features bitmask with a fixed size array, and +introduce a few helpers to help manipulate them. + +Most drivers will keep using only 64 bits features space: use union +to allow them access the lower part of the extended space without any +per driver change. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: <6a9bbb5eb33830f20afbcb7e64d300af4126dd98.1758549625.git.pabeni@redhat.com> +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit b15a61fdae976eb1ca8f2deee6a63dc3407d7ec6) +Signed-off-by: Laurent Vivier +--- + include/hw/virtio/virtio-features.h | 126 ++++++++++++++++++++++++++++ + include/hw/virtio/virtio.h | 7 +- + 2 files changed, 130 insertions(+), 3 deletions(-) + create mode 100644 include/hw/virtio/virtio-features.h + +diff --git a/include/hw/virtio/virtio-features.h b/include/hw/virtio/virtio-features.h +new file mode 100644 +index 0000000000..e29b7fe48f +--- /dev/null ++++ b/include/hw/virtio/virtio-features.h +@@ -0,0 +1,126 @@ ++/* ++ * Virtio features helpers ++ * ++ * Copyright 2025 Red Hat, Inc. ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef QEMU_VIRTIO_FEATURES_H ++#define QEMU_VIRTIO_FEATURES_H ++ ++#include "qemu/bitops.h" ++ ++#define VIRTIO_FEATURES_FMT "%016"PRIx64"%016"PRIx64 ++#define VIRTIO_FEATURES_PR(f) (f)[1], (f)[0] ++ ++#define VIRTIO_FEATURES_MAX 128 ++#define VIRTIO_FEATURES_BIT(b) BIT_ULL((b) % 64) ++#define VIRTIO_FEATURES_U64(b) ((b) / 64) ++#define VIRTIO_FEATURES_NU32S (VIRTIO_FEATURES_MAX / 32) ++#define VIRTIO_FEATURES_NU64S (VIRTIO_FEATURES_MAX / 64) ++ ++#define VIRTIO_DECLARE_FEATURES(name) \ ++ union { \ ++ uint64_t name; \ ++ uint64_t name##_ex[VIRTIO_FEATURES_NU64S]; \ ++ } ++ ++#define VIRTIO_DEFINE_PROP_FEATURE(_name, _state, _field, _bit, _defval) \ ++ DEFINE_PROP_BIT64(_name, _state, _field[VIRTIO_FEATURES_U64(_bit)], \ ++ (_bit) % 64, _defval) ++ ++static inline void virtio_features_clear(uint64_t *features) ++{ ++ memset(features, 0, sizeof(features[0]) * VIRTIO_FEATURES_NU64S); ++} ++ ++static inline void virtio_features_from_u64(uint64_t *features, uint64_t from) ++{ ++ virtio_features_clear(features); ++ features[0] = from; ++} ++ ++static inline bool virtio_has_feature_ex(const uint64_t *features, ++ unsigned int fbit) ++{ ++ assert(fbit < VIRTIO_FEATURES_MAX); ++ return features[VIRTIO_FEATURES_U64(fbit)] & VIRTIO_FEATURES_BIT(fbit); ++} ++ ++static inline void virtio_add_feature_ex(uint64_t *features, ++ unsigned int fbit) ++{ ++ assert(fbit < VIRTIO_FEATURES_MAX); ++ features[VIRTIO_FEATURES_U64(fbit)] |= VIRTIO_FEATURES_BIT(fbit); ++} ++ ++static inline void virtio_clear_feature_ex(uint64_t *features, ++ unsigned int fbit) ++{ ++ assert(fbit < VIRTIO_FEATURES_MAX); ++ features[VIRTIO_FEATURES_U64(fbit)] &= ~VIRTIO_FEATURES_BIT(fbit); ++} ++ ++static inline bool virtio_features_equal(const uint64_t *f1, ++ const uint64_t *f2) ++{ ++ return !memcmp(f1, f2, sizeof(uint64_t) * VIRTIO_FEATURES_NU64S); ++} ++ ++static inline bool virtio_features_use_ex(const uint64_t *features) ++{ ++ int i; ++ ++ for (i = 1; i < VIRTIO_FEATURES_NU64S; ++i) { ++ if (features[i]) { ++ return true; ++ } ++ } ++ return false; ++} ++ ++static inline bool virtio_features_empty(const uint64_t *features) ++{ ++ return !virtio_features_use_ex(features) && !features[0]; ++} ++ ++static inline void virtio_features_copy(uint64_t *to, const uint64_t *from) ++{ ++ memcpy(to, from, sizeof(to[0]) * VIRTIO_FEATURES_NU64S); ++} ++ ++static inline bool virtio_features_andnot(uint64_t *to, const uint64_t *f1, ++ const uint64_t *f2) ++{ ++ uint64_t diff = 0; ++ int i; ++ ++ for (i = 0; i < VIRTIO_FEATURES_NU64S; i++) { ++ to[i] = f1[i] & ~f2[i]; ++ diff |= to[i]; ++ } ++ return diff; ++} ++ ++static inline void virtio_features_and(uint64_t *to, const uint64_t *f1, ++ const uint64_t *f2) ++{ ++ int i; ++ ++ for (i = 0; i < VIRTIO_FEATURES_NU64S; i++) { ++ to[i] = f1[i] & f2[i]; ++ } ++} ++ ++static inline void virtio_features_or(uint64_t *to, const uint64_t *f1, ++ const uint64_t *f2) ++{ ++ int i; ++ ++ for (i = 0; i < VIRTIO_FEATURES_NU64S; i++) { ++ to[i] = f1[i] | f2[i]; ++ } ++} ++ ++#endif +diff --git a/include/hw/virtio/virtio.h b/include/hw/virtio/virtio.h +index c594764f23..39e4059a66 100644 +--- a/include/hw/virtio/virtio.h ++++ b/include/hw/virtio/virtio.h +@@ -16,6 +16,7 @@ + + #include "system/memory.h" + #include "hw/qdev-core.h" ++#include "hw/virtio/virtio-features.h" + #include "net/net.h" + #include "migration/vmstate.h" + #include "qemu/event_notifier.h" +@@ -121,9 +122,9 @@ struct VirtIODevice + * backend (e.g. vhost) and could potentially be a subset of the + * total feature set offered by QEMU. + */ +- uint64_t host_features; +- uint64_t guest_features; +- uint64_t backend_features; ++ VIRTIO_DECLARE_FEATURES(host_features); ++ VIRTIO_DECLARE_FEATURES(guest_features); ++ VIRTIO_DECLARE_FEATURES(backend_features); + + size_t config_len; + void *config; +-- +2.47.3 + diff --git a/kvm-virtio-net-implement-extended-features-support.patch b/kvm-virtio-net-implement-extended-features-support.patch new file mode 100644 index 0000000..1af6377 --- /dev/null +++ b/kvm-virtio-net-implement-extended-features-support.patch @@ -0,0 +1,331 @@ +From cd61fca5f53b5047a432db36f33ee85edf4f6ac3 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:26 +0200 +Subject: [PATCH 17/19] virtio-net: implement extended features support + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [12/14] 22829a4e87459998cbb2a658d24434f1e94c8597 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Use the extended types and helpers to manipulate the virtio_net +features. + +Note that offloads are still 64bits wide, as per specification, +and extended offloads will be mapped into such range. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Acked-by: Stefano Garzarella +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 3a7741c3bdc3537de4159418d712debbd22e4df6) +Signed-off-by: Laurent Vivier +--- + hw/net/virtio-net.c | 140 +++++++++++++++++++-------------- + include/hw/virtio/virtio-net.h | 2 +- + 2 files changed, 83 insertions(+), 59 deletions(-) + +diff --git a/hw/net/virtio-net.c b/hw/net/virtio-net.c +index b86ba1fd27..89cf008401 100644 +--- a/hw/net/virtio-net.c ++++ b/hw/net/virtio-net.c +@@ -90,6 +90,19 @@ + VIRTIO_NET_RSS_HASH_TYPE_TCP_EX | \ + VIRTIO_NET_RSS_HASH_TYPE_UDP_EX) + ++/* ++ * Features starting from VIRTIO_NET_FEATURES_MAP_MIN bit correspond ++ * to guest offloads in the VIRTIO_NET_OFFLOAD_MAP range ++ */ ++#define VIRTIO_NET_OFFLOAD_MAP_MIN 46 ++#define VIRTIO_NET_OFFLOAD_MAP_LENGTH 4 ++#define VIRTIO_NET_OFFLOAD_MAP MAKE_64BIT_MASK( \ ++ VIRTIO_NET_OFFLOAD_MAP_MIN, \ ++ VIRTIO_NET_OFFLOAD_MAP_LENGTH) ++#define VIRTIO_NET_FEATURES_MAP_MIN 65 ++#define VIRTIO_NET_F2O_SHIFT (VIRTIO_NET_OFFLOAD_MAP_MIN - \ ++ VIRTIO_NET_FEATURES_MAP_MIN + 64) ++ + static const VirtIOFeature feature_sizes[] = { + {.flags = 1ULL << VIRTIO_NET_F_MAC, + .end = endof(struct virtio_net_config, mac)}, +@@ -786,7 +799,14 @@ static void virtio_net_apply_guest_offloads(VirtIONet *n) + qemu_set_offload(qemu_get_queue(n->nic)->peer, &ol); + } + +-static uint64_t virtio_net_guest_offloads_by_features(uint64_t features) ++static uint64_t virtio_net_features_to_offload(const uint64_t *features) ++{ ++ return (features[0] & ~VIRTIO_NET_OFFLOAD_MAP) | ++ ((features[1] << VIRTIO_NET_F2O_SHIFT) & VIRTIO_NET_OFFLOAD_MAP); ++} ++ ++static uint64_t ++virtio_net_guest_offloads_by_features(const uint64_t *features) + { + static const uint64_t guest_offloads_mask = + (1ULL << VIRTIO_NET_F_GUEST_CSUM) | +@@ -797,13 +817,13 @@ static uint64_t virtio_net_guest_offloads_by_features(uint64_t features) + (1ULL << VIRTIO_NET_F_GUEST_USO4) | + (1ULL << VIRTIO_NET_F_GUEST_USO6); + +- return guest_offloads_mask & features; ++ return guest_offloads_mask & virtio_net_features_to_offload(features); + } + + uint64_t virtio_net_supported_guest_offloads(const VirtIONet *n) + { + VirtIODevice *vdev = VIRTIO_DEVICE(n); +- return virtio_net_guest_offloads_by_features(vdev->guest_features); ++ return virtio_net_guest_offloads_by_features(vdev->guest_features_ex); + } + + typedef struct { +@@ -882,34 +902,39 @@ static void failover_add_primary(VirtIONet *n, Error **errp) + error_propagate(errp, err); + } + +-static void virtio_net_set_features(VirtIODevice *vdev, uint64_t features) ++static void virtio_net_set_features(VirtIODevice *vdev, ++ const uint64_t *in_features) + { ++ uint64_t features[VIRTIO_FEATURES_NU64S]; + VirtIONet *n = VIRTIO_NET(vdev); + Error *err = NULL; + int i; + ++ virtio_features_copy(features, in_features); + if (n->mtu_bypass_backend && + !virtio_has_feature(vdev->backend_features, VIRTIO_NET_F_MTU)) { +- features &= ~(1ULL << VIRTIO_NET_F_MTU); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_MTU); + } + + virtio_net_set_multiqueue(n, +- virtio_has_feature(features, VIRTIO_NET_F_RSS) || +- virtio_has_feature(features, VIRTIO_NET_F_MQ)); ++ virtio_has_feature_ex(features, ++ VIRTIO_NET_F_RSS) || ++ virtio_has_feature_ex(features, ++ VIRTIO_NET_F_MQ)); + + virtio_net_set_mrg_rx_bufs(n, +- virtio_has_feature(features, ++ virtio_has_feature_ex(features, + VIRTIO_NET_F_MRG_RXBUF), +- virtio_has_feature(features, ++ virtio_has_feature_ex(features, + VIRTIO_F_VERSION_1), +- virtio_has_feature(features, ++ virtio_has_feature_ex(features, + VIRTIO_NET_F_HASH_REPORT)); + +- n->rsc4_enabled = virtio_has_feature(features, VIRTIO_NET_F_RSC_EXT) && +- virtio_has_feature(features, VIRTIO_NET_F_GUEST_TSO4); +- n->rsc6_enabled = virtio_has_feature(features, VIRTIO_NET_F_RSC_EXT) && +- virtio_has_feature(features, VIRTIO_NET_F_GUEST_TSO6); +- n->rss_data.redirect = virtio_has_feature(features, VIRTIO_NET_F_RSS); ++ n->rsc4_enabled = virtio_has_feature_ex(features, VIRTIO_NET_F_RSC_EXT) && ++ virtio_has_feature_ex(features, VIRTIO_NET_F_GUEST_TSO4); ++ n->rsc6_enabled = virtio_has_feature_ex(features, VIRTIO_NET_F_RSC_EXT) && ++ virtio_has_feature_ex(features, VIRTIO_NET_F_GUEST_TSO6); ++ n->rss_data.redirect = virtio_has_feature_ex(features, VIRTIO_NET_F_RSS); + + if (n->has_vnet_hdr) { + n->curr_guest_offloads = +@@ -923,7 +948,7 @@ static void virtio_net_set_features(VirtIODevice *vdev, uint64_t features) + if (!get_vhost_net(nc->peer)) { + continue; + } +- vhost_net_ack_features(get_vhost_net(nc->peer), features); ++ vhost_net_ack_features_ex(get_vhost_net(nc->peer), features); + + /* + * keep acked_features in NetVhostUserState up-to-date so it +@@ -932,12 +957,14 @@ static void virtio_net_set_features(VirtIODevice *vdev, uint64_t features) + vhost_net_save_acked_features(nc->peer); + } + +- if (virtio_has_feature(vdev->guest_features ^ features, VIRTIO_NET_F_CTRL_VLAN)) { +- bool vlan = virtio_has_feature(features, VIRTIO_NET_F_CTRL_VLAN); ++ if (virtio_has_feature_ex(features, VIRTIO_NET_F_CTRL_VLAN) != ++ virtio_has_feature_ex(vdev->guest_features_ex, ++ VIRTIO_NET_F_CTRL_VLAN)) { ++ bool vlan = virtio_has_feature_ex(features, VIRTIO_NET_F_CTRL_VLAN); + memset(n->vlans, vlan ? 0 : 0xff, MAX_VLAN >> 3); + } + +- if (virtio_has_feature(features, VIRTIO_NET_F_STANDBY)) { ++ if (virtio_has_feature_ex(features, VIRTIO_NET_F_STANDBY)) { + qapi_event_send_failover_negotiated(n->netclient_name); + qatomic_set(&n->failover_primary_hidden, false); + failover_add_primary(n, &err); +@@ -1902,10 +1929,10 @@ static ssize_t virtio_net_receive_rcu(NetClientState *nc, const uint8_t *buf, + virtio_error(vdev, "virtio-net unexpected empty queue: " + "i %zd mergeable %d offset %zd, size %zd, " + "guest hdr len %zd, host hdr len %zd " +- "guest features 0x%" PRIx64, ++ "guest features 0x" VIRTIO_FEATURES_FMT, + i, n->mergeable_rx_bufs, offset, size, + n->guest_hdr_len, n->host_hdr_len, +- vdev->guest_features); ++ VIRTIO_FEATURES_PR(vdev->guest_features_ex)); + } + err = -1; + goto err; +@@ -3012,8 +3039,8 @@ static int virtio_net_pre_load_queues(VirtIODevice *vdev, uint32_t n) + return 0; + } + +-static uint64_t virtio_net_get_features(VirtIODevice *vdev, uint64_t features, +- Error **errp) ++static void virtio_net_get_features(VirtIODevice *vdev, uint64_t *features, ++ Error **errp) + { + VirtIONet *n = VIRTIO_NET(vdev); + NetClientState *nc = qemu_get_queue(n->nic); +@@ -3027,68 +3054,67 @@ static uint64_t virtio_net_get_features(VirtIODevice *vdev, uint64_t features, + (supported_hash_types & peer_hash_types) == supported_hash_types; + + /* Firstly sync all virtio-net possible supported features */ +- features |= n->host_features; ++ virtio_features_or(features, features, n->host_features_ex); + +- virtio_add_feature(&features, VIRTIO_NET_F_MAC); ++ virtio_add_feature_ex(features, VIRTIO_NET_F_MAC); + + if (!peer_has_vnet_hdr(n)) { +- virtio_clear_feature(&features, VIRTIO_NET_F_CSUM); +- virtio_clear_feature(&features, VIRTIO_NET_F_HOST_TSO4); +- virtio_clear_feature(&features, VIRTIO_NET_F_HOST_TSO6); +- virtio_clear_feature(&features, VIRTIO_NET_F_HOST_ECN); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_CSUM); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_TSO4); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_TSO6); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_ECN); + +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_CSUM); +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_TSO4); +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_TSO6); +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_ECN); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_CSUM); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_TSO4); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_TSO6); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_ECN); + +- virtio_clear_feature(&features, VIRTIO_NET_F_HOST_USO); +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_USO4); +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_USO6); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_USO); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_USO4); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_USO6); + +- virtio_clear_feature(&features, VIRTIO_NET_F_HASH_REPORT); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HASH_REPORT); + } + + if (!peer_has_vnet_hdr(n) || !peer_has_ufo(n)) { +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_UFO); +- virtio_clear_feature(&features, VIRTIO_NET_F_HOST_UFO); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_UFO); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_UFO); + } +- + if (!peer_has_uso(n)) { +- virtio_clear_feature(&features, VIRTIO_NET_F_HOST_USO); +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_USO4); +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_USO6); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HOST_USO); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_USO4); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_USO6); + } + + if (!get_vhost_net(nc->peer)) { + if (!use_own_hash) { +- virtio_clear_feature(&features, VIRTIO_NET_F_HASH_REPORT); +- virtio_clear_feature(&features, VIRTIO_NET_F_RSS); +- } else if (virtio_has_feature(features, VIRTIO_NET_F_RSS)) { ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HASH_REPORT); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_RSS); ++ } else if (virtio_has_feature_ex(features, VIRTIO_NET_F_RSS)) { + virtio_net_load_ebpf(n, errp); + } + +- return features; ++ return; + } + + if (!use_peer_hash) { +- virtio_clear_feature(&features, VIRTIO_NET_F_HASH_REPORT); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_HASH_REPORT); + + if (!use_own_hash || !virtio_net_attach_ebpf_to_backend(n->nic, -1)) { + if (!virtio_net_load_ebpf(n, errp)) { +- return features; ++ return; + } + +- virtio_clear_feature(&features, VIRTIO_NET_F_RSS); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_RSS); + } + } + +- features = vhost_net_get_features(get_vhost_net(nc->peer), features); +- vdev->backend_features = features; ++ vhost_net_get_features_ex(get_vhost_net(nc->peer), features); ++ virtio_features_copy(vdev->backend_features_ex, features); + + if (n->mtu_bypass_backend && + (n->host_features & 1ULL << VIRTIO_NET_F_MTU)) { +- features |= (1ULL << VIRTIO_NET_F_MTU); ++ virtio_add_feature_ex(features, VIRTIO_NET_F_MTU); + } + + /* +@@ -3103,10 +3129,8 @@ static uint64_t virtio_net_get_features(VirtIODevice *vdev, uint64_t features, + * support it. + */ + if (!virtio_has_feature(vdev->backend_features, VIRTIO_NET_F_CTRL_VQ)) { +- virtio_clear_feature(&features, VIRTIO_NET_F_GUEST_ANNOUNCE); ++ virtio_clear_feature_ex(features, VIRTIO_NET_F_GUEST_ANNOUNCE); + } +- +- return features; + } + + static int virtio_net_post_load_device(void *opaque, int version_id) +@@ -4238,8 +4262,8 @@ static void virtio_net_class_init(ObjectClass *klass, const void *data) + vdc->unrealize = virtio_net_device_unrealize; + vdc->get_config = virtio_net_get_config; + vdc->set_config = virtio_net_set_config; +- vdc->get_features = virtio_net_get_features; +- vdc->set_features = virtio_net_set_features; ++ vdc->get_features_ex = virtio_net_get_features; ++ vdc->set_features_ex = virtio_net_set_features; + vdc->bad_features = virtio_net_bad_features; + vdc->reset = virtio_net_reset; + vdc->queue_reset = virtio_net_queue_reset; +diff --git a/include/hw/virtio/virtio-net.h b/include/hw/virtio/virtio-net.h +index 73fdefc0dc..5b8ab7bda7 100644 +--- a/include/hw/virtio/virtio-net.h ++++ b/include/hw/virtio/virtio-net.h +@@ -182,7 +182,7 @@ struct VirtIONet { + uint32_t has_vnet_hdr; + size_t host_hdr_len; + size_t guest_hdr_len; +- uint64_t host_features; ++ VIRTIO_DECLARE_FEATURES(host_features); + uint32_t rsc_timeout; + uint8_t rsc4_enabled; + uint8_t rsc6_enabled; +-- +2.47.3 + diff --git a/kvm-virtio-pci-implement-support-for-extended-features.patch b/kvm-virtio-pci-implement-support-for-extended-features.patch new file mode 100644 index 0000000..98821d9 --- /dev/null +++ b/kvm-virtio-pci-implement-support-for-extended-features.patch @@ -0,0 +1,196 @@ +From 71096da8a19183bbe11f28d5dfe6cf6cc38d9585 Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:21 +0200 +Subject: [PATCH 12/19] virtio-pci: implement support for extended features + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [7/14] 7d5683a98169722acb250f9e01344ec3d56faad4 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +Extend the features configuration space to 128 bits. If the virtio +device supports any extended features, allow the common read/write +operation to access all of it, otherwise keep exposing only the +lower 64 bits. + +On migration, save the 128 bit version of the features only if the +upper bits are non zero. Relay on reset to clear all the feature +space before load. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 712c79d6d374e7abe94599de5ba2d155d5a79955) +Signed-off-by: Laurent Vivier +--- + hw/virtio/virtio-pci.c | 76 ++++++++++++++++++++++++++++++---- + include/hw/virtio/virtio-pci.h | 2 +- + 2 files changed, 68 insertions(+), 10 deletions(-) + +diff --git a/hw/virtio/virtio-pci.c b/hw/virtio/virtio-pci.c +index 0cdc16217f..d38d0a5e9a 100644 +--- a/hw/virtio/virtio-pci.c ++++ b/hw/virtio/virtio-pci.c +@@ -110,6 +110,29 @@ static const VMStateDescription vmstate_virtio_pci_modern_queue_state = { + } + }; + ++static bool virtio_pci_modern_state_features128_needed(void *opaque) ++{ ++ VirtIOPCIProxy *proxy = opaque; ++ uint32_t features = 0; ++ int i; ++ ++ for (i = 2; i < ARRAY_SIZE(proxy->guest_features); ++i) { ++ features |= proxy->guest_features[i]; ++ } ++ return features; ++} ++ ++static const VMStateDescription vmstate_virtio_pci_modern_state_features128 = { ++ .name = "virtio_pci/modern_state/features128", ++ .version_id = 1, ++ .minimum_version_id = 1, ++ .needed = &virtio_pci_modern_state_features128_needed, ++ .fields = (const VMStateField[]) { ++ VMSTATE_UINT32_SUB_ARRAY(guest_features, VirtIOPCIProxy, 2, 2), ++ VMSTATE_END_OF_LIST() ++ } ++}; ++ + static bool virtio_pci_modern_state_needed(void *opaque) + { + VirtIOPCIProxy *proxy = opaque; +@@ -117,6 +140,12 @@ static bool virtio_pci_modern_state_needed(void *opaque) + return virtio_pci_modern(proxy); + } + ++/* ++ * Avoid silently breaking migration should the feature space increase ++ * even more in the (far away) future ++ */ ++QEMU_BUILD_BUG_ON(VIRTIO_FEATURES_NU32S != 4); ++ + static const VMStateDescription vmstate_virtio_pci_modern_state_sub = { + .name = "virtio_pci/modern_state", + .version_id = 1, +@@ -125,11 +154,15 @@ static const VMStateDescription vmstate_virtio_pci_modern_state_sub = { + .fields = (const VMStateField[]) { + VMSTATE_UINT32(dfselect, VirtIOPCIProxy), + VMSTATE_UINT32(gfselect, VirtIOPCIProxy), +- VMSTATE_UINT32_ARRAY(guest_features, VirtIOPCIProxy, 2), ++ VMSTATE_UINT32_SUB_ARRAY(guest_features, VirtIOPCIProxy, 0, 2), + VMSTATE_STRUCT_ARRAY(vqs, VirtIOPCIProxy, VIRTIO_QUEUE_MAX, 0, + vmstate_virtio_pci_modern_queue_state, + VirtIOPCIQueue), + VMSTATE_END_OF_LIST() ++ }, ++ .subsections = (const VMStateDescription * const []) { ++ &vmstate_virtio_pci_modern_state_features128, ++ NULL + } + }; + +@@ -1478,6 +1511,19 @@ int virtio_pci_add_shm_cap(VirtIOPCIProxy *proxy, + return virtio_pci_add_mem_cap(proxy, &cap.cap); + } + ++static int virtio_pci_select_max(const VirtIODevice *vdev) ++{ ++ int i; ++ ++ for (i = VIRTIO_FEATURES_NU64S - 1; i > 0; i--) { ++ if (vdev->host_features_ex[i]) { ++ return (i + 1) * 2; ++ } ++ } ++ ++ return 2; ++} ++ + static uint64_t virtio_pci_common_read(void *opaque, hwaddr addr, + unsigned size) + { +@@ -1495,18 +1541,21 @@ static uint64_t virtio_pci_common_read(void *opaque, hwaddr addr, + val = proxy->dfselect; + break; + case VIRTIO_PCI_COMMON_DF: +- if (proxy->dfselect <= 1) { ++ if (proxy->dfselect < virtio_pci_select_max(vdev)) { + VirtioDeviceClass *vdc = VIRTIO_DEVICE_GET_CLASS(vdev); + +- val = (vdev->host_features & ~vdc->legacy_features) >> +- (32 * proxy->dfselect); ++ val = vdev->host_features_ex[proxy->dfselect >> 1] >> ++ (32 * (proxy->dfselect & 1)); ++ if (proxy->dfselect <= 1) { ++ val &= (~vdc->legacy_features) >> (32 * proxy->dfselect); ++ } + } + break; + case VIRTIO_PCI_COMMON_GFSELECT: + val = proxy->gfselect; + break; + case VIRTIO_PCI_COMMON_GF: +- if (proxy->gfselect < ARRAY_SIZE(proxy->guest_features)) { ++ if (proxy->gfselect < virtio_pci_select_max(vdev)) { + val = proxy->guest_features[proxy->gfselect]; + } + break; +@@ -1589,11 +1638,18 @@ static void virtio_pci_common_write(void *opaque, hwaddr addr, + proxy->gfselect = val; + break; + case VIRTIO_PCI_COMMON_GF: +- if (proxy->gfselect < ARRAY_SIZE(proxy->guest_features)) { ++ if (proxy->gfselect < virtio_pci_select_max(vdev)) { ++ uint64_t features[VIRTIO_FEATURES_NU64S]; ++ int i; ++ + proxy->guest_features[proxy->gfselect] = val; +- virtio_set_features(vdev, +- (((uint64_t)proxy->guest_features[1]) << 32) | +- proxy->guest_features[0]); ++ virtio_features_clear(features); ++ for (i = 0; i < ARRAY_SIZE(proxy->guest_features); ++i) { ++ uint64_t cur = proxy->guest_features[i]; ++ ++ features[i >> 1] |= cur << ((i & 1) * 32); ++ } ++ virtio_set_features_ex(vdev, features); + } + break; + case VIRTIO_PCI_COMMON_MSIX: +@@ -2312,6 +2368,8 @@ static void virtio_pci_reset(DeviceState *qdev) + virtio_bus_reset(bus); + msix_unuse_all_vectors(&proxy->pci_dev); + ++ memset(proxy->guest_features, 0, sizeof(proxy->guest_features)); ++ + for (i = 0; i < VIRTIO_QUEUE_MAX; i++) { + proxy->vqs[i].enabled = 0; + proxy->vqs[i].reset = 0; +diff --git a/include/hw/virtio/virtio-pci.h b/include/hw/virtio/virtio-pci.h +index eab5394898..639752977e 100644 +--- a/include/hw/virtio/virtio-pci.h ++++ b/include/hw/virtio/virtio-pci.h +@@ -158,7 +158,7 @@ struct VirtIOPCIProxy { + uint32_t nvectors; + uint32_t dfselect; + uint32_t gfselect; +- uint32_t guest_features[2]; ++ uint32_t guest_features[VIRTIO_FEATURES_NU32S]; + VirtIOPCIQueue vqs[VIRTIO_QUEUE_MAX]; + + VirtIOIRQFD *vector_irqfd; +-- +2.47.3 + diff --git a/kvm-virtio-serialize-extended-features-state.patch b/kvm-virtio-serialize-extended-features-state.patch new file mode 100644 index 0000000..97a19d7 --- /dev/null +++ b/kvm-virtio-serialize-extended-features-state.patch @@ -0,0 +1,222 @@ +From 0f4e78c05a49b1ab51bd057df02b5c42f7914ebf Mon Sep 17 00:00:00 2001 +From: Paolo Abeni +Date: Mon, 22 Sep 2025 16:18:19 +0200 +Subject: [PATCH 10/19] virtio: serialize extended features state + +RH-Author: Laurent Vivier +RH-MergeRequest: 456: backport support for GSO over UDP tunnel offload +RH-Jira: RHEL-143785 +RH-Acked-by: Cindy Lu +RH-Acked-by: MST +RH-Commit: [5/14] 3448a1d2d4b7b6952819257c1bc2f5e5dcee8935 (lvivier/qemu-kvm-centos) + +JIRA: https://issues.redhat.com/browse/RHEL-143785 + +If the driver uses any of the extended features (i.e. 64 or above), +store the extended features range (64-127 bits). + +At load time, let legacy features initialize the full features range +and pass it to the set helper; sub-states loading will have filled-up +the extended part as needed. + +This is one of the few spots that need explicitly to know and set +in stone the extended features array size; add a build bug to prevent +breaking the migration should such size change again in the future: +more serialization plumbing will be needed. + +Reviewed-by: Akihiko Odaki +Acked-by: Jason Wang +Acked-by: Stefano Garzarella +Signed-off-by: Paolo Abeni +Tested-by: Lei Yang +Reviewed-by: Michael S. Tsirkin +Message-ID: +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 0a49a97433279512a03f3d9f36164a46caf498c6) +Signed-off-by: Laurent Vivier +--- + hw/virtio/virtio.c | 88 ++++++++++++++++++++++++++++++---------------- + 1 file changed, 57 insertions(+), 31 deletions(-) + +diff --git a/hw/virtio/virtio.c b/hw/virtio/virtio.c +index 9a81ad912e..bf53c211e5 100644 +--- a/hw/virtio/virtio.c ++++ b/hw/virtio/virtio.c +@@ -2964,6 +2964,30 @@ static const VMStateDescription vmstate_virtio_disabled = { + } + }; + ++static bool virtio_128bit_features_needed(void *opaque) ++{ ++ VirtIODevice *vdev = opaque; ++ ++ return virtio_features_use_ex(vdev->host_features_ex); ++} ++ ++static const VMStateDescription vmstate_virtio_128bit_features = { ++ .name = "virtio/128bit_features", ++ .version_id = 1, ++ .minimum_version_id = 1, ++ .needed = &virtio_128bit_features_needed, ++ .fields = (const VMStateField[]) { ++ VMSTATE_UINT64(guest_features_ex[1], VirtIODevice), ++ VMSTATE_END_OF_LIST() ++ } ++}; ++ ++/* ++ * Avoid silently breaking migration should the feature space increase ++ * even more in the (far away) future ++ */ ++QEMU_BUILD_BUG_ON(VIRTIO_FEATURES_NU64S != 2); ++ + static const VMStateDescription vmstate_virtio = { + .name = "virtio", + .version_id = 1, +@@ -2973,6 +2997,7 @@ static const VMStateDescription vmstate_virtio = { + }, + .subsections = (const VMStateDescription * const []) { + &vmstate_virtio_device_endian, ++ &vmstate_virtio_128bit_features, + &vmstate_virtio_64bit_features, + &vmstate_virtio_virtqueues, + &vmstate_virtio_ringsize, +@@ -3069,23 +3094,28 @@ const VMStateInfo virtio_vmstate_info = { + .put = virtio_device_put, + }; + +-static int virtio_set_features_nocheck(VirtIODevice *vdev, uint64_t val) ++static int virtio_set_features_nocheck(VirtIODevice *vdev, const uint64_t *val) + { + VirtioDeviceClass *k = VIRTIO_DEVICE_GET_CLASS(vdev); +- bool bad = (val & ~(vdev->host_features)) != 0; ++ uint64_t tmp[VIRTIO_FEATURES_NU64S]; ++ bool bad; ++ ++ bad = virtio_features_andnot(tmp, val, vdev->host_features_ex); ++ virtio_features_and(tmp, val, vdev->host_features_ex); + +- val &= vdev->host_features; + if (k->set_features) { +- k->set_features(vdev, val); ++ bad = bad || virtio_features_use_ex(tmp); ++ k->set_features(vdev, tmp[0]); + } +- vdev->guest_features = val; ++ ++ virtio_features_copy(vdev->guest_features_ex, tmp); + return bad ? -1 : 0; + } + + typedef struct VirtioSetFeaturesNocheckData { + Coroutine *co; + VirtIODevice *vdev; +- uint64_t val; ++ uint64_t val[VIRTIO_FEATURES_NU64S]; + int ret; + } VirtioSetFeaturesNocheckData; + +@@ -3098,14 +3128,15 @@ static void virtio_set_features_nocheck_bh(void *opaque) + } + + static int coroutine_mixed_fn +-virtio_set_features_nocheck_maybe_co(VirtIODevice *vdev, uint64_t val) ++virtio_set_features_nocheck_maybe_co(VirtIODevice *vdev, ++ const uint64_t *val) + { + if (qemu_in_coroutine()) { + VirtioSetFeaturesNocheckData data = { + .co = qemu_coroutine_self(), + .vdev = vdev, +- .val = val, + }; ++ virtio_features_copy(data.val, val); + aio_bh_schedule_oneshot(qemu_get_current_aio_context(), + virtio_set_features_nocheck_bh, &data); + qemu_coroutine_yield(); +@@ -3117,6 +3148,7 @@ virtio_set_features_nocheck_maybe_co(VirtIODevice *vdev, uint64_t val) + + int virtio_set_features(VirtIODevice *vdev, uint64_t val) + { ++ uint64_t features[VIRTIO_FEATURES_NU64S]; + int ret; + /* + * The driver must not attempt to set features after feature negotiation +@@ -3132,7 +3164,8 @@ int virtio_set_features(VirtIODevice *vdev, uint64_t val) + __func__, vdev->name); + } + +- ret = virtio_set_features_nocheck(vdev, val); ++ virtio_features_from_u64(features, val); ++ ret = virtio_set_features_nocheck(vdev, features); + if (virtio_vdev_has_feature(vdev, VIRTIO_RING_F_EVENT_IDX)) { + /* VIRTIO_RING_F_EVENT_IDX changes the size of the caches. */ + int i; +@@ -3155,6 +3188,7 @@ void virtio_reset(void *opaque) + { + VirtIODevice *vdev = opaque; + VirtioDeviceClass *k = VIRTIO_DEVICE_GET_CLASS(vdev); ++ uint64_t features[VIRTIO_FEATURES_NU64S]; + int i; + + virtio_set_status(vdev, 0); +@@ -3181,7 +3215,8 @@ void virtio_reset(void *opaque) + vdev->start_on_kick = false; + vdev->started = false; + vdev->broken = false; +- virtio_set_features_nocheck(vdev, 0); ++ virtio_features_clear(features); ++ virtio_set_features_nocheck(vdev, features); + vdev->queue_sel = 0; + vdev->status = 0; + vdev->disabled = false; +@@ -3264,7 +3299,7 @@ virtio_load(VirtIODevice *vdev, QEMUFile *f, int version_id) + * Note: devices should always test host features in future - don't create + * new dependencies like this. + */ +- vdev->guest_features = features; ++ virtio_features_from_u64(vdev->guest_features_ex, features); + + config_len = qemu_get_be32(f); + +@@ -3343,26 +3378,17 @@ virtio_load(VirtIODevice *vdev, QEMUFile *f, int version_id) + vdev->device_endian = virtio_default_endian(); + } + +- if (virtio_64bit_features_needed(vdev)) { +- /* +- * Subsection load filled vdev->guest_features. Run them +- * through virtio_set_features to sanity-check them against +- * host_features. +- */ +- uint64_t features64 = vdev->guest_features; +- if (virtio_set_features_nocheck_maybe_co(vdev, features64) < 0) { +- error_report("Features 0x%" PRIx64 " unsupported. " +- "Allowed features: 0x%" PRIx64, +- features64, vdev->host_features); +- return -1; +- } +- } else { +- if (virtio_set_features_nocheck_maybe_co(vdev, features) < 0) { +- error_report("Features 0x%x unsupported. " +- "Allowed features: 0x%" PRIx64, +- features, vdev->host_features); +- return -1; +- } ++ /* ++ * guest_features_ex is fully initialized with u32 features and upper ++ * bits have been filled as needed by the later load. ++ */ ++ if (virtio_set_features_nocheck_maybe_co(vdev, ++ vdev->guest_features_ex) < 0) { ++ error_report("Features 0x" VIRTIO_FEATURES_FMT " unsupported. " ++ "Allowed features: 0x" VIRTIO_FEATURES_FMT, ++ VIRTIO_FEATURES_PR(vdev->guest_features_ex), ++ VIRTIO_FEATURES_PR(vdev->host_features_ex)); ++ return -1; + } + + if (!virtio_device_started(vdev, vdev->status) && +-- +2.47.3 + diff --git a/kvm-x86-cpu-deprecate-cpu-models-that-do-not-support-x86.patch b/kvm-x86-cpu-deprecate-cpu-models-that-do-not-support-x86.patch deleted file mode 100644 index c5a06e4..0000000 --- a/kvm-x86-cpu-deprecate-cpu-models-that-do-not-support-x86.patch +++ /dev/null @@ -1,98 +0,0 @@ -From 8c735b34df1902f32eb68bb3e6c3e8f04b010bd4 Mon Sep 17 00:00:00 2001 -From: Ani Sinha -Date: Mon, 10 Jun 2024 15:34:22 +0530 -Subject: [PATCH 05/14] x86/cpu: deprecate cpu models that do not support - x86-64-v3 - -RH-Author: Ani Sinha -RH-MergeRequest: 247: x86/cpu: deprecate cpu models that do not support x86-64-v3 -RH-Jira: RHEL-28971 -RH-Acked-by: Igor Mammedov -RH-Acked-by: MST -RH-Commit: [1/1] 1afb03048c674b54da8cd4ad5174f767a7514b51 (anisinha/centos-qemu-kvm) - -RHEL-10 has switched to a new baseline microarchitecture called "x86-64-v3". -Deprecate the CPU models that do not support x86-64-v3. The following are the -CPU models that do not support v3: - -Intel: Denverton, IvyBridge, Nehalem, SandyBridge, Snowridge, Westmere. -AMD: Opteron_G4 and Opteron_G5. - -See also https://www.qemu.org/docs/master/system/i386/cpu.html#abi-compatibility-levels-for-cpu-models - -Signed-off-by: Ani Sinha ---- - target/i386/cpu.c | 8 ++++++++ - 1 file changed, 8 insertions(+) - -diff --git a/target/i386/cpu.c b/target/i386/cpu.c -index c83d585c9b..3eac3135a6 100644 ---- a/target/i386/cpu.c -+++ b/target/i386/cpu.c -@@ -2597,6 +2597,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - #endif // Removal of deprecated CPU models in RHEL-10 - { - .name = "Nehalem", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 11, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2674,6 +2675,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Westmere", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 11, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2755,6 +2757,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "SandyBridge", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 0xd, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -2841,6 +2844,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "IvyBridge", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 0xd, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -4121,6 +4125,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Denverton", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 21, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -4231,6 +4236,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Snowridge", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 27, - .vendor = CPUID_VENDOR_INTEL, - .family = 6, -@@ -4486,6 +4492,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - #endif - { - .name = "Opteron_G4", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 0xd, - .vendor = CPUID_VENDOR_AMD, - .family = 21, -@@ -4518,6 +4525,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { - }, - { - .name = "Opteron_G5", -+ .deprecation_note = RHEL_CPU_DEPRECATION, - .level = 0xd, - .vendor = CPUID_VENDOR_AMD, - .family = 21, --- -2.39.3 - diff --git a/kvm-x86-cpu-update-deprecation-string-to-match-lowest-un.patch b/kvm-x86-cpu-update-deprecation-string-to-match-lowest-un.patch deleted file mode 100644 index 853b6f7..0000000 --- a/kvm-x86-cpu-update-deprecation-string-to-match-lowest-un.patch +++ /dev/null @@ -1,38 +0,0 @@ -From 03615078bc2e2f238e3eb00b11f697a7e68477df Mon Sep 17 00:00:00 2001 -From: Ani Sinha -Date: Tue, 20 Aug 2024 13:32:49 +0530 -Subject: [PATCH] x86/cpu: update deprecation string to match lowest - undeprecated model - -RH-Author: Ani Sinha -RH-MergeRequest: 264: x86/cpu: update deprecation string to match lowest undeprecated model -RH-Jira: RHEL-54260 -RH-Commit: [1/1] 834ef2694b441431c3da48fefde307eea96d90e4 (anisinha/centos-qemu-kvm) - -Commit a581f2824dce64 ("x86/cpu: deprecate cpu models that do not support x86-64-v3") -deprecated a bunch of cpu models in RHEL-10 that do not support x86-64-v3. The -deprecation string was not updated to match what was the lowest model that was -still available and not deprecated. Update the string to reflect the new -reality. - -Signed-off-by: Ani Sinha ---- - target/i386/cpu.c | 2 +- - 1 file changed, 1 insertion(+), 1 deletion(-) - -diff --git a/target/i386/cpu.c b/target/i386/cpu.c -index 3eac3135a6..46f82974fb 100644 ---- a/target/i386/cpu.c -+++ b/target/i386/cpu.c -@@ -2191,7 +2191,7 @@ static const CPUCaches epyc_genoa_cache_info = { - */ - - #define RHEL_CPU_DEPRECATION \ -- "use at least 'Nehalem' / 'Opteron_G4', or 'host' / 'max'" -+ "use at least 'Haswell' / 'EPYC', or 'host' / 'max'" - - static const X86CPUDefinition builtin_x86_defs[] = { - { --- -2.39.3 - diff --git a/kvm-x86-create-new-rhel-10.2-specific-pc-q35-machine-typ.patch b/kvm-x86-create-new-rhel-10.2-specific-pc-q35-machine-typ.patch new file mode 100644 index 0000000..06321af --- /dev/null +++ b/kvm-x86-create-new-rhel-10.2-specific-pc-q35-machine-typ.patch @@ -0,0 +1,56 @@ +From 95622c4261ef6e0eb6688196d87d3bed838aa91d Mon Sep 17 00:00:00 2001 +From: Sebastian Ott +Date: Wed, 29 Oct 2025 17:59:07 +0100 +Subject: [PATCH 08/10] x86: create new rhel 10.2 specific pc-q35 machine type + +RH-Author: Sebastian Ott +RH-MergeRequest: 416: x86, arm: create new rhel 9.8, 10.2 specific machine types +RH-Jira: RHEL-105826 RHEL-105828 +RH-Acked-by: Eric Auger +RH-Acked-by: Gavin Shan +RH-Acked-by: Cornelia Huck +RH-Commit: [3/4] 7f5a5fa45696e579cd08fb3869884ce9543badd5 (seott1/cos-qemu-kvm) + +Signed-off-by: Sebastian Ott +--- + hw/i386/pc_q35.c | 14 ++++++++++++-- + 1 file changed, 12 insertions(+), 2 deletions(-) + +diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c +index e5d10c0335..aea4735dd3 100644 +--- a/hw/i386/pc_q35.c ++++ b/hw/i386/pc_q35.c +@@ -690,10 +690,20 @@ DEFINE_Q35_MACHINE(2, 6); + + /* Red Hat Enterprise Linux machine types */ + +-static void pc_q35_rhel_machine_10_0_0_options(MachineClass *m) ++static void pc_q35_rhel_machine_10_2_0_options(MachineClass *m) + { + PCMachineClass *pcmc = PC_MACHINE_CLASS(m); + pc_q35_machine_options(m); ++ m->desc = "RHEL-10.2.0 PC (Q35 + ICH9, 2009)"; ++ pcmc->smbios_stream_product = "RHEL"; ++ pcmc->smbios_stream_version = "10.2.0"; ++} ++DEFINE_Q35_MACHINE_AS_LATEST(10, 2, 0); ++ ++static void pc_q35_rhel_machine_10_0_0_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ pc_q35_rhel_machine_10_2_0_options(m); + m->desc = "RHEL-10.0.0 PC (Q35 + ICH9, 2009)"; + pcmc->smbios_stream_product = "RHEL"; + pcmc->smbios_stream_version = "10.0.0"; +@@ -707,7 +717,7 @@ static void pc_q35_rhel_machine_10_0_0_options(MachineClass *m) + compat_props_add(m->compat_props, pc_rhel_10_1_compat, + pc_rhel_10_1_compat_len); + } +-DEFINE_Q35_MACHINE_AS_LATEST(10, 0, 0); ++DEFINE_Q35_MACHINE(10, 0, 0); + + static void pc_q35_rhel_machine_9_6_0_options(MachineClass *m) + { +-- +2.47.3 + diff --git a/kvm-x86-create-new-rhel-9.8-specific-pc-q35-machine-type.patch b/kvm-x86-create-new-rhel-9.8-specific-pc-q35-machine-type.patch new file mode 100644 index 0000000..0354574 --- /dev/null +++ b/kvm-x86-create-new-rhel-9.8-specific-pc-q35-machine-type.patch @@ -0,0 +1,56 @@ +From ee136e68450beb87b55e3668538b3a52fdf98736 Mon Sep 17 00:00:00 2001 +From: Sebastian Ott +Date: Wed, 29 Oct 2025 18:05:21 +0100 +Subject: [PATCH 09/10] x86: create new rhel 9.8 specific pc-q35 machine type + +RH-Author: Sebastian Ott +RH-MergeRequest: 416: x86, arm: create new rhel 9.8, 10.2 specific machine types +RH-Jira: RHEL-105826 RHEL-105828 +RH-Acked-by: Eric Auger +RH-Acked-by: Gavin Shan +RH-Acked-by: Cornelia Huck +RH-Commit: [4/4] 2bfa55e063b4551e8e3299cf4f7aa2520943076a (seott1/cos-qemu-kvm) + +Signed-off-by: Sebastian Ott +--- + hw/i386/pc_q35.c | 17 ++++++++++++++--- + 1 file changed, 14 insertions(+), 3 deletions(-) + +diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c +index aea4735dd3..efbb5695c2 100644 +--- a/hw/i386/pc_q35.c ++++ b/hw/i386/pc_q35.c +@@ -719,16 +719,27 @@ static void pc_q35_rhel_machine_10_0_0_options(MachineClass *m) + } + DEFINE_Q35_MACHINE(10, 0, 0); + ++static void pc_q35_rhel_machine_9_8_0_options(MachineClass *m) ++{ ++ PCMachineClass *pcmc = PC_MACHINE_CLASS(m); ++ pc_q35_machine_options(m); ++ m->desc = "RHEL-9.8.0 PC (Q35 + ICH9, 2009)"; ++ pcmc->smbios_stream_product = "RHEL"; ++ pcmc->smbios_stream_version = "9.8.0"; ++ ++ /* NB: remember to move this line to the *latest* RHEL 9 machine */ ++ compat_props_add(m->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); ++} ++DEFINE_Q35_MACHINE(9, 8, 0); ++ + static void pc_q35_rhel_machine_9_6_0_options(MachineClass *m) + { + PCMachineClass *pcmc = PC_MACHINE_CLASS(m); + pc_q35_rhel_machine_10_0_0_options(m); ++ pc_q35_rhel_machine_9_8_0_options(m); + m->desc = "RHEL-9.6.0 PC (Q35 + ICH9, 2009)"; + pcmc->smbios_stream_product = "RHEL"; + pcmc->smbios_stream_version = "9.6.0"; +- +- /* NB: remember to move this line to the *latest* RHEL 9 machine */ +- compat_props_add(m->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } + + DEFINE_Q35_MACHINE(9, 6, 0); +-- +2.47.3 + diff --git a/qemu-ga.sysconfig b/qemu-ga.sysconfig index 736b471..b574514 100644 --- a/qemu-ga.sysconfig +++ b/qemu-ga.sysconfig @@ -13,7 +13,7 @@ # # You can get the list of RPC commands using "qemu-ga --allow-rpcs='?'". # There should be no spaces between commas and commands in the allow list. -FILTER_RPC_ARGS="--allow-rpcs=guest-sync-delimited,guest-sync,guest-ping,guest-get-time,guest-set-time,guest-info,guest-shutdown,guest-fsfreeze-status,guest-fsfreeze-freeze,guest-fsfreeze-freeze-list,guest-fsfreeze-thaw,guest-fstrim,guest-suspend-disk,guest-suspend-ram,guest-suspend-hybrid,guest-network-get-interfaces,guest-get-vcpus,guest-set-vcpus,guest-get-disks,guest-get-fsinfo,guest-set-user-password,guest-get-memory-blocks,guest-set-memory-blocks,guest-get-memory-block-info,guest-get-host-name,guest-get-users,guest-get-timezone,guest-get-osinfo,guest-get-devices,guest-ssh-get-authorized-keys,guest-ssh-add-authorized-keys,guest-ssh-remove-authorized-keys,guest-get-diskstats,guest-get-cpustats" +FILTER_RPC_ARGS="--allow-rpcs=guest-sync-delimited,guest-sync,guest-ping,guest-get-time,guest-set-time,guest-info,guest-shutdown,guest-fsfreeze-status,guest-fsfreeze-freeze,guest-fsfreeze-freeze-list,guest-fsfreeze-thaw,guest-fstrim,guest-suspend-disk,guest-suspend-ram,guest-suspend-hybrid,guest-network-get-interfaces,guest-get-vcpus,guest-set-vcpus,guest-get-disks,guest-get-fsinfo,guest-set-user-password,guest-get-memory-blocks,guest-set-memory-blocks,guest-get-memory-block-info,guest-get-host-name,guest-get-users,guest-get-timezone,guest-get-osinfo,guest-get-devices,guest-ssh-get-authorized-keys,guest-ssh-add-authorized-keys,guest-ssh-remove-authorized-keys,guest-get-diskstats,guest-get-cpustats,guest-network-get-route,guest-get-load" # Fsfreeze hook script specification. # diff --git a/qemu-kvm.spec b/qemu-kvm.spec index ba25a9d..7cd5d33 100644 --- a/qemu-kvm.spec +++ b/qemu-kvm.spec @@ -57,7 +57,7 @@ %global tools_only 1 %endif -%ifnarch %{ix86} x86_64 aarch64 +%ifnarch x86_64 aarch64 %global have_usbredir 0 %endif @@ -66,13 +66,10 @@ %ifarch s390x %global modprobe_kvm_conf %{_sourcedir}/kvm-s390x.conf %endif -%ifarch %{ix86} x86_64 +%ifarch x86_64 %global modprobe_kvm_conf %{_sourcedir}/kvm-x86.conf %endif -%ifarch %{ix86} - %global kvm_target i386 -%endif %ifarch x86_64 %global kvm_target x86_64 %else @@ -86,12 +83,13 @@ %global kvm_target s390x %global have_modules_load 1 %endif -%ifarch ppc - %global kvm_target ppc -%endif %ifarch aarch64 %global kvm_target aarch64 %endif +%ifarch riscv64 + %global kvm_target riscv64 +%endif + %global target_list %{kvm_target}-softmmu %global block_drivers_rw_list qcow2,raw,file,host_device,nbd,iscsi,rbd,blkdebug,luks,null-co,nvme,copy-on-read,throttle,compress,virtio-blk-vhost-vdpa,virtio-blk-vfio-pci,virtio-blk-vhost-user,io_uring,nvme-io_uring @@ -120,7 +118,9 @@ Requires: %{name}-device-usb-host = %{epoch}:%{version}-%{release} \ Requires: %{name}-device-usb-redirect = %{epoch}:%{version}-%{release} \ %endif \ Requires: %{name}-block-blkio = %{epoch}:%{version}-%{release} \ +%if %{have_block_rbd} \ Requires: %{name}-block-rbd = %{epoch}:%{version}-%{release} \ +%endif \ Requires: %{name}-audio-pa = %{epoch}:%{version}-%{release} # Since SPICE is removed from RHEL-9, the following Obsoletes: @@ -142,15 +142,15 @@ Obsoletes: %{name}-block-ssh <= %{epoch}:%{version} \ Summary: QEMU is a machine emulator and virtualizer Name: qemu-kvm -Version: 9.0.0 -Release: 9%{?rcrel}%{?dist}%{?cc_suffix} +Version: 10.1.0 +Release: 16%{?rcrel}%{?dist}%{?cc_suffix} # Epoch because we pushed a qemu-1.0 package. AIUI this can't ever be dropped # Epoch 15 used for RHEL 8 # Epoch 17 used for RHEL 9 (due to release versioning offset in RHEL 8.5) Epoch: 18 License: GPL-2.0-only AND GPL-2.0-or-later AND CC-BY-3.0 URL: http://www.qemu.org/ -ExclusiveArch: x86_64 %{power64} aarch64 s390x +ExclusiveArch: x86_64 %{power64} aarch64 s390x riscv64 Source0: http://wiki.qemu.org/download/qemu-%{version}%{?rcstr}.tar.xz @@ -171,81 +171,257 @@ Source36: README.tests Patch0004: 0004-Initial-redhat-build.patch Patch0005: 0005-Enable-disable-devices-for-RHEL.patch Patch0006: 0006-Machine-type-related-general-changes.patch -Patch0007: 0007-Add-aarch64-machine-types.patch -Patch0008: 0008-Add-s390x-machine-types.patch -Patch0009: 0009-Add-x86_64-machine-types.patch -Patch0010: 0010-Enable-make-check.patch -Patch0011: 0011-vfio-cap-number-of-devices-that-can-be-assigned.patch -Patch0012: 0012-Add-support-statement-to-help-output.patch -Patch0013: 0013-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch -Patch0014: 0014-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch -Patch0015: 0015-Add-upstream-compatibility-bits.patch -Patch0016: 0016-Disable-FDC-devices.patch -Patch0017: 0017-Disable-vga-cirrus-device.patch -# For RHEL-37563 - Enable 'vhost-user-snd-pci' in qemu-kvm for RHIVOS -Patch18: kvm-Enable-vhost-user-snd-pci-device.patch -# For RHEL-28972 - x86: Remove the existing deprecated CPU models on RHEL10 -Patch19: kvm-qtest-x86-numa-test-do-not-use-the-obsolete-pentium-.patch -# For RHEL-28972 - x86: Remove the existing deprecated CPU models on RHEL10 -Patch20: kvm-tests-qtest-libqtest-add-qtest_has_cpu_model-api.patch -# For RHEL-28972 - x86: Remove the existing deprecated CPU models on RHEL10 -Patch21: kvm-tests-qtest-x86-check-for-availability-of-older-cpu-.patch -# For RHEL-28972 - x86: Remove the existing deprecated CPU models on RHEL10 -Patch22: kvm-target-cpu-models-x86-Remove-the-existing-deprecated.patch -# For RHEL-28971 - Consider deprecating CPU models like "Nehalem" / "IvyBridge" on RHEL 10 -Patch23: kvm-x86-cpu-deprecate-cpu-models-that-do-not-support-x86.patch -# For RHEL-36329 - [RHEL10.0.beta][stable_guest_abi]Failed to migrate VM with (qemu) qemu-kvm: Missing section footer for 0000:00:01.0/virtio-gpu qemu-kvm: load of migration failed: Invalid argument -Patch24: kvm-virtio-gpu-fix-v2-migration.patch -# For RHEL-36329 - [RHEL10.0.beta][stable_guest_abi]Failed to migrate VM with (qemu) qemu-kvm: Missing section footer for 0000:00:01.0/virtio-gpu qemu-kvm: load of migration failed: Invalid argument -Patch25: kvm-rhel-9.4.0-machine-type-compat-for-virtio-gpu-migrat.patch -# For RHEL-39898 - s390: Remove the legacy CPU models on RHEL10 -Patch26: kvm-s390x-remove-deprecated-rhel-machine-types.patch -# For RHEL-39898 - s390: Remove the legacy CPU models on RHEL10 -Patch27: kvm-s390x-select-correct-components-for-no-board-build.patch -# For RHEL-39898 - s390: Remove the legacy CPU models on RHEL10 -Patch28: kvm-target-s390x-Add-a-CONFIG-switch-to-disable-legacy-C.patch -# For RHEL-39898 - s390: Remove the legacy CPU models on RHEL10 -Patch29: kvm-target-s390x-cpu_models-Disable-everything-up-to-the.patch -# For RHEL-39898 - s390: Remove the legacy CPU models on RHEL10 -Patch30: kvm-target-s390x-Revert-the-old-s390x-CPU-model-disablem.patch -# For RHEL-43409 - aio=io_uring: Assertion failure `luringcb->co->ctx == s->aio_context' with block_resize -# For RHEL-43410 - aio=native: Assertion failure `laiocb->co->ctx == laiocb->ctx->aio_context' with block_resize -Patch31: kvm-Revert-monitor-use-aio_co_reschedule_self.patch -# For RHEL-43409 - aio=io_uring: Assertion failure `luringcb->co->ctx == s->aio_context' with block_resize -# For RHEL-43410 - aio=native: Assertion failure `laiocb->co->ctx == laiocb->ctx->aio_context' with block_resize -Patch32: kvm-aio-warn-about-iohandler_ctx-special-casing.patch -# For RHEL-46239 - CVE-2024-4467 qemu-kvm: QEMU: 'qemu-img info' leads to host file read/write [rhel-10.0] -Patch33: kvm-qcow2-Don-t-open-data_file-with-BDRV_O_NO_IO.patch -# For RHEL-46239 - CVE-2024-4467 qemu-kvm: QEMU: 'qemu-img info' leads to host file read/write [rhel-10.0] -Patch34: kvm-iotests-244-Don-t-store-data-file-with-protocol-in-i.patch -# For RHEL-46239 - CVE-2024-4467 qemu-kvm: QEMU: 'qemu-img info' leads to host file read/write [rhel-10.0] -Patch35: kvm-iotests-270-Don-t-store-data-file-with-json-prefix-i.patch -# For RHEL-46239 - CVE-2024-4467 qemu-kvm: QEMU: 'qemu-img info' leads to host file read/write [rhel-10.0] -Patch36: kvm-block-Parse-filenames-only-when-explicitly-requested.patch -# For RHEL-40959 - Qemu hang when quit dst vm after storage migration(nbd+tls) -Patch37: kvm-nbd-server-do-not-poll-within-a-coroutine-context.patch -# For RHEL-40959 - Qemu hang when quit dst vm after storage migration(nbd+tls) -Patch38: kvm-nbd-server-Mark-negotiation-functions-as-coroutine_f.patch -# For RHEL-40959 - Qemu hang when quit dst vm after storage migration(nbd+tls) -Patch39: kvm-qio-Inherit-follow_coroutine_ctx-across-TLS.patch -# For RHEL-40959 - Qemu hang when quit dst vm after storage migration(nbd+tls) -Patch40: kvm-iotests-test-NBD-TLS-iothread.patch -# For RHEL-50165 - Enable 'vhost-user-scmi-pci' and 'vhost-user-scmi' in qemu-kvm for RHIVOS -Patch41: kvm-Enable-vhost-user-scmi-devices.patch -# For RHEL-51901 - qemu-kvm: linux-aio: add support for IO_CMD_FDSYNC command[RHEL-10] -Patch42: kvm-linux-aio-add-IO_CMD_FDSYNC-command-support.patch -# For RHEL-52599 - CVE-2024-7409 qemu-kvm: Denial of Service via Improper Synchronization in QEMU NBD Server During Socket Closure [rhel-10.0] -Patch43: kvm-nbd-server-Plumb-in-new-args-to-nbd_client_add.patch -# For RHEL-52599 - CVE-2024-7409 qemu-kvm: Denial of Service via Improper Synchronization in QEMU NBD Server During Socket Closure [rhel-10.0] -Patch44: kvm-nbd-server-CVE-2024-7409-Cap-default-max-connections.patch -# For RHEL-52599 - CVE-2024-7409 qemu-kvm: Denial of Service via Improper Synchronization in QEMU NBD Server During Socket Closure [rhel-10.0] -Patch45: kvm-nbd-server-CVE-2024-7409-Drop-non-negotiating-client.patch -# For RHEL-52599 - CVE-2024-7409 qemu-kvm: Denial of Service via Improper Synchronization in QEMU NBD Server During Socket Closure [rhel-10.0] -Patch46: kvm-nbd-server-CVE-2024-7409-Close-stray-clients-at-serv.patch -# For RHEL-54260 - [RHEL10] Need to update the deprecated CPU model warning message -Patch47: kvm-x86-cpu-update-deprecation-string-to-match-lowest-un.patch -# For RHEL-52599 - CVE-2024-7409 qemu-kvm: Denial of Service via Improper Synchronization in QEMU NBD Server During Socket Closure [rhel-10.0] -Patch48: kvm-nbd-server-CVE-2024-7409-Avoid-use-after-free-when-c.patch +Patch0007: 0007-meson-temporarily-disable-Wunused-function.patch +Patch0008: 0008-Remove-upstream-machine-types-for-aarch64-s390x-and-.patch +Patch0009: 0009-Adapt-versioned-machine-type-macros-for-RHEL.patch +Patch0010: 0010-Increase-deletion-schedule-to-4-releases.patch +Patch0011: 0011-Add-downstream-aarch64-versioned-virt-machine-types.patch +Patch0012: 0012-Add-downstream-s390x-versioned-s390-ccw-virtio-machi.patch +Patch0013: 0013-Add-downstream-x86_64-versioned-pc-q35-machine-types.patch +Patch0014: 0014-Disable-virtio-net-pci-romfile-loading-on-riscv64.patch +Patch0015: 0015-Revert-meson-temporarily-disable-Wunused-function.patch +Patch0016: 0016-Enable-make-check.patch +Patch0017: 0017-vfio-cap-number-of-devices-that-can-be-assigned.patch +Patch0018: 0018-Add-support-statement-to-help-output.patch +Patch0019: 0019-Use-qemu-kvm-in-documentation-instead-of-qemu-system.patch +Patch0020: 0020-qcow2-Deprecation-warning-when-opening-v2-images-rw.patch +Patch0021: 0021-file-posix-Define-DM_MPATH_PROBE_PATHS.patch +# For RHEL-112882 - [DEV Task]: Assertion `core->delayed_causes == 0' failed with e1000e NIC +Patch22: kvm-e1000e-Prevent-crash-from-legacy-interrupt-firing-af.patch +# For RHEL-119368 - [rhel10] Backport "arm/kvm: report registers we failed to set" +Patch23: kvm-arm-kvm-report-registers-we-failed-to-set.patch +# For RHEL-116443 - qemu crash after hot-unplug disk from the multifunction enabled bus,crash point PCIDevice *vf = dev->exp.sriov_pf.vf[i] +Patch24: kvm-pcie_sriov-make-pcie_sriov_pf_exit-safe-on-non-SR-IO.patch +# For RHEL-120253 - Backport fixes for PDCM and ARCH_CAPABILITIES migration incompatibility +Patch25: kvm-target-i386-add-compatibility-property-for-arch_capa.patch +# For RHEL-120253 - Backport fixes for PDCM and ARCH_CAPABILITIES migration incompatibility +Patch26: kvm-target-i386-add-compatibility-property-for-pdcm-feat.patch +# For RHEL-104009 - [IBM 10.2 FEAT] KVM: Enhance machine type definition to include CPI and PCI passthru capabilities (qemu) +# For RHEL-105823 - Add new -rhel10.2.0 machine type to qemu-kvm [s390x] +# For RHEL-73008 - [IBM 10.2 FEAT] KVM: Implement Control Program Identification (qemu) +Patch27: kvm-qapi-machine-s390x-add-QAPI-event-SCLP_CPI_INFO_AVAI.patch +# For RHEL-104009 - [IBM 10.2 FEAT] KVM: Enhance machine type definition to include CPI and PCI passthru capabilities (qemu) +# For RHEL-105823 - Add new -rhel10.2.0 machine type to qemu-kvm [s390x] +# For RHEL-73008 - [IBM 10.2 FEAT] KVM: Implement Control Program Identification (qemu) +Patch28: kvm-tests-functional-add-tests-for-SCLP-event-CPI.patch +# For RHEL-104009 - [IBM 10.2 FEAT] KVM: Enhance machine type definition to include CPI and PCI passthru capabilities (qemu) +# For RHEL-105823 - Add new -rhel10.2.0 machine type to qemu-kvm [s390x] +# For RHEL-73008 - [IBM 10.2 FEAT] KVM: Implement Control Program Identification (qemu) +Patch29: kvm-redhat-Add-new-rhel9.8.0-and-rhel10.2.0-machine-type.patch +# For RHEL-118810 - [RHEL 10.2] Windows 11 VM fails to boot up with ramfb='on' with QEMU 10.1 +Patch30: kvm-vfio-rename-field-to-num_initial_regions.patch +# For RHEL-118810 - [RHEL 10.2] Windows 11 VM fails to boot up with ramfb='on' with QEMU 10.1 +Patch31: kvm-vfio-only-check-region-info-cache-for-initial-region.patch +# For RHEL-105826 - Add new -rhel10.2.0 machine type to qemu-kvm [aarch64] +# For RHEL-105828 - Add new -rhel10.2.0 machine type to qemu-kvm [x86_64] +Patch32: kvm-arm-create-new-rhel-10.2-specific-virt-machine-type.patch +# For RHEL-105826 - Add new -rhel10.2.0 machine type to qemu-kvm [aarch64] +# For RHEL-105828 - Add new -rhel10.2.0 machine type to qemu-kvm [x86_64] +Patch33: kvm-arm-create-new-rhel-9.8-specific-virt-machine-type.patch +# For RHEL-105826 - Add new -rhel10.2.0 machine type to qemu-kvm [aarch64] +# For RHEL-105828 - Add new -rhel10.2.0 machine type to qemu-kvm [x86_64] +Patch34: kvm-x86-create-new-rhel-10.2-specific-pc-q35-machine-typ.patch +# For RHEL-105826 - Add new -rhel10.2.0 machine type to qemu-kvm [aarch64] +# For RHEL-105828 - Add new -rhel10.2.0 machine type to qemu-kvm [x86_64] +Patch35: kvm-x86-create-new-rhel-9.8-specific-pc-q35-machine-type.patch +# For RHEL-101929 - enable 'usb-bot' device for proper support of USB CD-ROM drives via libvirt +Patch36: kvm-rh-enable-CONFIG_USB_STORAGE_BOT.patch +# For RHEL-120116 - CVE-2025-11234 qemu-kvm: VNC WebSocket handshake use-after-free [rhel-10.2] +Patch37: kvm-io-move-websock-resource-release-to-close-method.patch +# For RHEL-120116 - CVE-2025-11234 qemu-kvm: VNC WebSocket handshake use-after-free [rhel-10.2] +Patch38: kvm-io-fix-use-after-free-in-websocket-handshake-code.patch +# For RHEL-126573 - VFIO migration using multifd should be disabled by default +Patch39: kvm-vfio-Disable-VFIO-migration-with-MultiFD-support.patch +# For RHEL-67323 - [aarch64] Support ACPI based PCI hotplug on ARM +Patch40: kvm-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-from-10..patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch41: kvm-hw-arm-smmu-common-Check-SMMU-has-PCIe-Root-Complex-.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch42: kvm-hw-arm-virt-acpi-build-Re-arrange-SMMUv3-IORT-build.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch43: kvm-hw-arm-virt-acpi-build-Update-IORT-for-multiple-smmu.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch44: kvm-hw-arm-virt-Factor-out-common-SMMUV3-dt-bindings-cod.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch45: kvm-hw-arm-virt-Add-an-SMMU_IO_LEN-macro.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch46: kvm-hw-pci-Introduce-pci_setup_iommu_per_bus-for-per-bus.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch47: kvm-hw-arm-virt-Allow-user-creatable-SMMUv3-dev-instanti.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch48: kvm-qemu-options.hx-Document-the-arm-smmuv3-device.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch49: kvm-bios-tables-test-Allow-for-smmuv3-test-data.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch50: kvm-qtest-bios-tables-test-Add-tests-for-legacy-smmuv3-a.patch +# For RHEL-73800 - NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1 +Patch51: kvm-qtest-bios-tables-test-Update-tables-for-smmuv3-test.patch +Patch52: kvm-qtest-Do-not-run-bios-tables-test-on-aarch64.patch +# For RHEL-126708 - [RHEL 10]snp guest fail to boot with hugepage +Patch53: kvm-ram-block-attributes-fix-interaction-with-hugetlb-me.patch +# For RHEL-126708 - [RHEL 10]snp guest fail to boot with hugepage +Patch54: kvm-ram-block-attributes-Unify-the-retrieval-of-the-bloc.patch +# For RHEL-128085 - VM crashes during boot when virtio device is attached through vfio_ccw +Patch55: kvm-hw-s390x-Fix-a-possible-crash-with-passed-through-vi.patch +# For RHEL-130704 - [rhel10] Fix the typo under vfio-pci device's enable-migration option +Patch56: kvm-Fix-the-typo-of-vfio-pci-device-s-enable-migration-o.patch +# For RHEL-120115 - The vf nic created using the IGB emulated nic can not obtain ip address +Patch57: kvm-pcie_sriov-Fix-broken-MMIO-accesses-from-SR-IOV-VFs.patch +# For RHEL-130478 - Migration from RHEL 10.2 to RHEL 10.1 with virt-rhel10.0.0 machine type fails on Grace +Patch58: kvm-arm-fix-oob-access-in-compat-handling.patch +# For RHEL-129540 - Assertion failure on drain with iothread and I/O load +Patch59: kvm-block-backend-Fix-race-when-resuming-queued-requests.patch +# For RHEL-121543 - The VM hit io error when do S3-PR integration on the pass-through failover multipath device +Patch60: kvm-file-posix-Handle-suspended-dm-multipath-better-for-.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch61: kvm-accel-Add-Meson-and-config-support-for-MSHV-accelera.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch62: kvm-target-i386-emulate-Allow-instruction-decoding-from-.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch63: kvm-target-i386-mshv-Add-x86-decoder-emu-implementation.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch64: kvm-hw-intc-Generalize-APIC-helper-names-from-kvm_-to-ac.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch65: kvm-include-hw-hyperv-Add-MSHV-ABI-header-definitions.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch66: kvm-linux-headers-linux-Add-mshv.h-headers.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch67: kvm-accel-mshv-Add-accelerator-skeleton.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch68: kvm-accel-mshv-Register-memory-region-listeners.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch69: kvm-accel-mshv-Initialize-VM-partition.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch70: kvm-treewide-rename-qemu_wait_io_event-qemu_wait_io_even.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch71: kvm-accel-mshv-Add-vCPU-creation-and-execution-loop.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch72: kvm-accel-mshv-Add-vCPU-signal-handling.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch73: kvm-target-i386-mshv-Add-CPU-create-and-remove-logic.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch74: kvm-target-i386-mshv-Implement-mshv_store_regs.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch75: kvm-target-i386-mshv-Implement-mshv_get_standard_regs.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch76: kvm-target-i386-mshv-Implement-mshv_get_special_regs.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch77: kvm-target-i386-mshv-Implement-mshv_arch_put_registers.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch78: kvm-target-i386-mshv-Set-local-interrupt-controller-stat.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch79: kvm-target-i386-mshv-Register-CPUID-entries-with-MSHV.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch80: kvm-target-i386-mshv-Register-MSRs-with-MSHV.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch81: kvm-target-i386-mshv-Integrate-x86-instruction-decoder-e.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch82: kvm-target-i386-mshv-Write-MSRs-to-the-hypervisor.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch83: kvm-target-i386-mshv-Implement-mshv_vcpu_run.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch84: kvm-accel-mshv-Handle-overlapping-mem-mappings.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch85: kvm-qapi-accel-Allow-to-query-mshv-capabilities.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch86: kvm-target-i386-mshv-Use-preallocated-page-for-hvcall.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch87: kvm-docs-Add-mshv-to-documentation.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch88: kvm-MAINTAINERS-Add-maintainers-for-mshv-accelerator.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch89: kvm-accel-mshv-initialize-thread-name.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch90: kvm-accel-mshv-use-return-value-of-handle_pio_str_read.patch +# For RHEL-134212 - [RHEL10.2] L1VH qemu downstream initial merge RHEL10.2 +Patch91: kvm-monitor-generalize-query-mshv-info-mshv-to-query-acc.patch +# For RHEL-110003 - Expose block limits of block nodes in QMP and qemu-img +Patch92: kvm-block-Improve-comments-in-BlockLimits.patch +# For RHEL-110003 - Expose block limits of block nodes in QMP and qemu-img +Patch93: kvm-block-Expose-block-limits-for-images-in-QMP.patch +# For RHEL-110003 - Expose block limits of block nodes in QMP and qemu-img +Patch94: kvm-qemu-img-info-Optionally-show-block-limits.patch +# For RHEL-110003 - Expose block limits of block nodes in QMP and qemu-img +Patch95: kvm-qemu-img-info-Add-cache-mode-option.patch +# For RHEL-111853 - [Intel 10.0 FEAT] [SPR] TDX: Virt-QEMU: QEMU Support [rhel-10] +Patch96: kvm-rh-configs-enable-CONFIG_TDX-for-x86_64.patch +# For RHEL-108142 - QEMU crashes when stopping source VM during live migration +Patch97: kvm-block-Fix-BDS-use-after-free-during-shutdown.patch +# For RHEL-126707 - [qemu, rhel-10] increase default TSEG size +Patch98: kvm-fix-pc_rhel_10_2_compat_len.patch +# For RHEL-126707 - [qemu, rhel-10] increase default TSEG size +Patch99: kvm-q35-increase-default-tseg-size.patch +# For RHEL-139028 - Intel IOMMU VM freezes: "call_irq_handler: 3.37 No irq handler for vector"[rhel-10.2] +Patch100: kvm-hw-intc-ioapic-Fix-ACCEL_KERNEL_GSI_IRQFD_POSSIBLE-t.patch +# For RHEL-111853 - [Intel 10.0 FEAT] [SPR] TDX: Virt-QEMU: QEMU Support [rhel-10] +Patch101: kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch +# For RHEL-79118 - [network-storage][rbd][core-dump]installation of guest failed sometimes with multiqueue enabled [rhel10] +Patch102: kvm-rbd-Run-co-BH-CB-in-the-coroutine-s-AioContext.patch +# For RHEL-79118 - [network-storage][rbd][core-dump]installation of guest failed sometimes with multiqueue enabled [rhel10] +Patch103: kvm-curl-Fix-coroutine-waking.patch +# For RHEL-79118 - [network-storage][rbd][core-dump]installation of guest failed sometimes with multiqueue enabled [rhel10] +Patch104: kvm-block-io-Take-reqs_lock-for-tracked_requests.patch +# For RHEL-79118 - [network-storage][rbd][core-dump]installation of guest failed sometimes with multiqueue enabled [rhel10] +Patch105: kvm-qcow2-Re-initialize-lock-in-invalidate_cache.patch +# For RHEL-79118 - [network-storage][rbd][core-dump]installation of guest failed sometimes with multiqueue enabled [rhel10] +Patch106: kvm-qcow2-Fix-cache_clean_timer.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch107: kvm-net-bundle-all-offloads-in-a-single-struct.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch108: kvm-linux-headers-deal-with-counted_by-annotation.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch109: kvm-linux-headers-Update-to-Linux-v6.17-rc1.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch110: kvm-virtio-introduce-extended-features-type.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch111: kvm-virtio-serialize-extended-features-state.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch112: kvm-virtio-add-support-for-negotiating-extended-features.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch113: kvm-virtio-pci-implement-support-for-extended-features.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch114: kvm-vhost-add-support-for-negotiating-extended-features.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch115: kvm-qmp-update-virtio-features-map-to-support-extended-f.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch116: kvm-vhost-backend-implement-extended-features-support.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch117: kvm-vhost-net-implement-extended-features-support.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch118: kvm-virtio-net-implement-extended-features-support.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch119: kvm-net-implement-tunnel-probing.patch +# For RHEL-143785 - backport support for GSO over UDP tunnel offload +Patch120: kvm-net-implement-UDP-tunnel-features-offloading.patch +# For RHEL-147425 - virtiofs: processes become stuck in request_wait_answer on virtiofs mounts +Patch121: kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch122: kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch123: kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch124: kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch125: kvm-scsi-save-load-SCSI-reservation-state.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch126: kvm-docs-add-SCSI-migrate-pr-documentation.patch +# For RHEL-134989 - Hotplugged interface device can not be shown in the guest +# For RHEL-146584 - [RHEL-10.2][ARM]: Unable to Check the mem prefetched size on Guest +Patch127: kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch +# For RHEL-153058 - Qemu crashes with "double free" during restore --reset-nvram with uefi-vars secure boot +Patch128: kvm-hw-uefi-add-variable-digest-to-vmstate.patch +# For RHEL-144004 - [rhel-10] Regression in BLOCK_IO_ERROR event delivery with (w|r)error setting of 'stop' or 'enospc' due to event rate limiting +Patch129: kvm-block-Never-drop-BLOCK_IO_ERROR-with-action-stop-for.patch +# For RHEL-155601 - Mirror job can miss writes during startup, corrupting the copy [rhel-10.2] +Patch130: kvm-mirror-Fix-missed-dirty-bitmap-writes-during-startup.patch +# For RHEL-158224 - qemu-kvm: disk writes of fewer bytes than requested is a retry condition, not necessarily an indication of ENOSPC [rhel-10.2] +Patch131: kvm-linux-aio-Put-all-parameters-into-qemu_laiocb.patch +# For RHEL-158224 - qemu-kvm: disk writes of fewer bytes than requested is a retry condition, not necessarily an indication of ENOSPC [rhel-10.2] +Patch132: kvm-linux-aio-Resubmit-tails-of-short-reads-writes.patch +# For RHEL-158224 - qemu-kvm: disk writes of fewer bytes than requested is a retry condition, not necessarily an indication of ENOSPC [rhel-10.2] +Patch133: kvm-block-io_uring-avoid-potentially-getting-stuck-after.patch +# For RHEL-158224 - qemu-kvm: disk writes of fewer bytes than requested is a retry condition, not necessarily an indication of ENOSPC [rhel-10.2] +Patch134: kvm-io-uring-Resubmit-tails-of-short-writes.patch %if %{have_clang} BuildRequires: clang @@ -283,6 +459,8 @@ BuildRequires: librbd-devel # We need both because the 'stap' binary is probed for by configure BuildRequires: systemtap BuildRequires: systemtap-sdt-devel +# Required as we use dtrace for trace backend +BuildRequires: /usr/bin/dtrace # For VNC PNG support BuildRequires: libpng-devel # For virtiofs @@ -320,6 +498,9 @@ BuildRequires: libslirp-devel BuildRequires: pulseaudio-libs-devel BuildRequires: spice-protocol BuildRequires: capstone-devel +%ifarch %{valgrind_arches} +BuildRequires: valgrind-devel +%endif # Requires for qemu-kvm package Requires: %{name}-core = %{epoch}:%{version}-%{release} @@ -341,12 +522,15 @@ Summary: %{name} core components %{obsoletes_some_modules} Requires: %{name}-common = %{epoch}:%{version}-%{release} Requires: qemu-img = %{epoch}:%{version}-%{release} -%ifarch %{ix86} x86_64 +%ifarch x86_64 Requires: edk2-ovmf %endif %ifarch aarch64 Requires: edk2-aarch64 %endif +%ifarch riscv64 +Requires: edk2-riscv64 +%endif Requires: libseccomp >= %{libseccomp_version} Requires: libusbx >= %{libusbx_version} @@ -378,10 +562,10 @@ Requires(post): /usr/sbin/useradd Requires(post): systemd-units Requires(preun): systemd-units Requires(postun): systemd-units -%ifarch %{ix86} x86_64 +%ifarch x86_64 Requires: seabios-bin >= 1.10.2-1 %endif -%ifnarch aarch64 s390x +%ifarch x86_64 %{power64} Requires: seavgabios-bin >= 1.12.0-3 Requires: ipxe-roms-qemu >= %{ipxe_version} %endif @@ -399,6 +583,8 @@ This package provides documentation and auxiliary programs used with %{name}. %package tools Summary: %{name} support tools +Recommends: systemtap-client +Recommends: systemtap-devel %description tools %{name}-tools provides various tools related to %{name} usage. @@ -447,8 +633,8 @@ Requires: %{name} = %{epoch}:%{version}-%{release} The %{name}-tests rpm contains tests that can be used to verify the functionality of the installed %{name} package -Install this package if you want access to the avocado_qemu -tests, or qemu-iotests. +Install this package if you want access to the qemu tests, +or qemu-iotests. %package block-blkio @@ -587,11 +773,9 @@ ulimit -n 10240 %define disable_everything \\\ --audio-drv-list= \\\ --disable-alsa \\\ + --disable-asan \\\ --disable-attr \\\ --disable-auth-pam \\\ - --disable-avx2 \\\ - --disable-avx512f \\\ - --disable-avx512bw \\\ --disable-blkio \\\ --disable-block-drv-whitelist-in-tools \\\ --disable-bochs \\\ @@ -646,7 +830,6 @@ ulimit -n 10240 --disable-linux-aio \\\ --disable-linux-io-uring \\\ --disable-linux-user \\\ - --disable-live-block-migration \\\ --disable-lto \\\ --disable-lzfse \\\ --disable-lzo \\\ @@ -666,7 +849,7 @@ ulimit -n 10240 --disable-parallels \\\ --disable-pie \\\ --disable-plugins \\\ - --disable-pvrdma \\\ + --disable-pvg \\\ --disable-qcow1 \\\ --disable-qed \\\ --disable-qga-vss \\\ @@ -676,7 +859,6 @@ ulimit -n 10240 --disable-replication \\\ --disable-rng-none \\\ --disable-safe-stack \\\ - --disable-sanitizers \\\ --disable-sdl \\\ --disable-sdl-image \\\ --disable-seccomp \\\ @@ -695,8 +877,10 @@ ulimit -n 10240 --disable-tools \\\ --disable-tpm \\\ --disable-u2f \\\ + --disable-ubsan \\\ --disable-usb-redir \\\ --disable-user \\\ + --disable-valgrind \\\ --disable-vde \\\ --disable-vdi \\\ --disable-vduse-blk-export \\\ @@ -744,7 +928,10 @@ run_configure() { --with-coroutine=ucontext \ --tls-priority=@QEMU,SYSTEM \ %{disable_everything} \ +%ifarch aarch64 s390x x86_64 riscv64 --with-devices-%{kvm_target}=%{kvm_target}-rh-devices \ +%endif + --rhel-version=10 \ "$@" echo "config-host.mak contents:" @@ -818,6 +1005,9 @@ run_configure \ --enable-tpm \ %if %{have_usbredir} --enable-usb-redir \ +%endif +%ifarch %{valgrind_arches} + --enable-valgrind \ %endif --enable-vdi \ --enable-vhost-kernel \ @@ -874,9 +1064,10 @@ cp -a qemu-system-%{kvm_target} qemu-kvm %ifarch s390x # Copy the built new images into place for "make check": - cp pc-bios/s390-ccw/s390-ccw.img pc-bios/s390-ccw/s390-netboot.img pc-bios/ + cp pc-bios/s390-ccw/s390-ccw.img pc-bios/ %endif + popd # endif !tools_only %endif @@ -927,7 +1118,6 @@ install -D -p -m 0644 %{modprobe_kvm_conf} $RPM_BUILD_ROOT%{_sysconfdir}/modprob # Create new directories and put them all under tests-src mkdir -p %{buildroot}%{testsdir}/python mkdir -p %{buildroot}%{testsdir}/tests -mkdir -p %{buildroot}%{testsdir}/tests/avocado mkdir -p %{buildroot}%{testsdir}/tests/qemu-iotests mkdir -p %{buildroot}%{testsdir}/scripts/qmp @@ -935,10 +1125,8 @@ mkdir -p %{buildroot}%{testsdir}/scripts/qmp install -m 0644 scripts/dump-guest-memory.py \ %{buildroot}%{_datadir}/%{name} -# Install avocado_qemu tests -cp -R %{qemu_kvm_build}/tests/avocado/* %{buildroot}%{testsdir}/tests/avocado/ -# Install qemu.py and qmp/ scripts required to run avocado_qemu tests +# Install qemu.py and qmp/ scripts required to run tests cp -R %{qemu_kvm_build}/python/qemu %{buildroot}%{testsdir}/python cp -R %{qemu_kvm_build}/scripts/qmp/* %{buildroot}%{testsdir}/scripts/qmp install -p -m 0644 tests/Makefile.include %{buildroot}%{testsdir}/tests/ @@ -1003,20 +1191,20 @@ rm -rf %{buildroot}%{_datadir}/%{name}/slof.bin # Remove unpackaged files. rm -rf %{buildroot}%{_datadir}/%{name}/palcode-clipper -rm -rf %{buildroot}%{_datadir}/%{name}/petalogix*.dtb -rm -f %{buildroot}%{_datadir}/%{name}/bamboo.dtb +rm -rf %{buildroot}%{_datadir}/%{name}/dtb/petalogix*.dtb +rm -f %{buildroot}%{_datadir}/%{name}/dtb/bamboo.dtb rm -f %{buildroot}%{_datadir}/%{name}/ppc_rom.bin rm -rf %{buildroot}%{_datadir}/%{name}/s390-zipl.rom rm -rf %{buildroot}%{_datadir}/%{name}/u-boot.e500 rm -rf %{buildroot}%{_datadir}/%{name}/qemu_vga.ndrv rm -rf %{buildroot}%{_datadir}/%{name}/skiboot.lid rm -rf %{buildroot}%{_datadir}/%{name}/qboot.rom +rm -rf %{buildroot}%{_datadir}/%{name}/pnv-pnor.bin rm -rf %{buildroot}%{_datadir}/%{name}/s390-ccw.img -rm -rf %{buildroot}%{_datadir}/%{name}/s390-netboot.img rm -rf %{buildroot}%{_datadir}/%{name}/hppa-firmware.img rm -rf %{buildroot}%{_datadir}/%{name}/hppa-firmware64.img -rm -rf %{buildroot}%{_datadir}/%{name}/canyonlands.dtb +rm -rf %{buildroot}%{_datadir}/%{name}/dtb/canyonlands.dtb rm -rf %{buildroot}%{_datadir}/%{name}/u-boot-sam460-20100605.bin rm -rf %{buildroot}%{_datadir}/%{name}/firmware @@ -1031,15 +1219,18 @@ rm -rf %{buildroot}%{_datadir}/%{name}/opensbi-riscv64-virt-fw_jump.bin rm -rf %{buildroot}%{_datadir}/%{name}/opensbi-riscv64-generic-fw_dynamic.* rm -rf %{buildroot}%{_datadir}/%{name}/qemu-nsis.bmp rm -rf %{buildroot}%{_datadir}/%{name}/npcm7xx_bootrom.bin +rm -rf %{buildroot}%{_datadir}/%{name}/npcm8xx_bootrom.bin +rm -rf %{buildroot}%{_datadir}/%{name}/ast27x0_bootrom.bin # Remove virtfs-proxy-helper files rm -rf %{buildroot}%{_libexecdir}/virtfs-proxy-helper rm -rf %{buildroot}%{_mandir}/man1/virtfs-proxy-helper* %ifarch s390x - # Use the s390-*.img that we've just built, not the pre-built ones + # Use the s390-ccw.img that we've just built, not the pre-built one install -m 0644 %{qemu_kvm_build}/pc-bios/s390-ccw/s390-ccw.img %{buildroot}%{_datadir}/%{name}/ - install -m 0644 %{qemu_kvm_build}/pc-bios/s390-ccw/s390-netboot.img %{buildroot}%{_datadir}/%{name}/ + # Remove uefi vars + rm -rf %{buildroot}%{_libdir}/%{name}/hw-uefi-vars.so %else rm -rf %{buildroot}%{_libdir}/%{name}/hw-s390x-virtio-gpu-ccw.so %endif @@ -1050,6 +1241,8 @@ rm -rf %{buildroot}%{_mandir}/man1/virtfs-proxy-helper* rm -rf %{buildroot}%{_datadir}/%{name}/multiboot.bin rm -rf %{buildroot}%{_datadir}/%{name}/multiboot_dma.bin rm -rf %{buildroot}%{_datadir}/%{name}/pvh.bin +%else + rm -rf %{buildroot}%{_bindir}/qemu-vmsr-helper %endif # Remove sparc files @@ -1227,7 +1420,8 @@ useradd -r -u 107 -g qemu -G kvm -d / -s /sbin/nologin \ %endif %ifarch s390x %{_datadir}/%{name}/s390-ccw.img - %{_datadir}/%{name}/s390-netboot.img +%else + %{_libdir}/%{name}/hw-uefi-vars.so %endif %{_datadir}/icons/* %{_datadir}/%{name}/linuxboot_dma.bin @@ -1250,10 +1444,6 @@ useradd -r -u 107 -g qemu -G kvm -d / -s /sbin/nologin \ %{_datadir}/systemtap/tapset/qemu-nbd*.stp %{_datadir}/systemtap/tapset/qemu-storage-daemon*.stp -%ifarch x86_64 - %{_libdir}/%{name}/accel-tcg-%{kvm_target}.so -%endif - %files device-display-virtio-gpu %{_libdir}/%{name}/hw-display-virtio-gpu.so @@ -1311,10 +1501,684 @@ useradd -r -u 107 -g qemu -G kvm -d / -s /sbin/nologin \ %endif %changelog -* Wed Sep 18 2024 Miroslav Rezanina - 9.0.0-9 -- kvm-nbd-server-CVE-2024-7409-Avoid-use-after-free-when-c.patch [RHEL-52599] -- Resolves: RHEL-52599 - (CVE-2024-7409 qemu-kvm: Denial of Service via Improper Synchronization in QEMU NBD Server During Socket Closure [rhel-10.0]) +* Mon Mar 30 2026 Miroslav Rezanina - 10.1.0-16 +- kvm-linux-aio-Put-all-parameters-into-qemu_laiocb.patch [RHEL-158224] +- kvm-linux-aio-Resubmit-tails-of-short-reads-writes.patch [RHEL-158224] +- kvm-block-io_uring-avoid-potentially-getting-stuck-after.patch [RHEL-158224] +- kvm-io-uring-Resubmit-tails-of-short-writes.patch [RHEL-158224] +- Resolves: RHEL-158224 + (qemu-kvm: disk writes of fewer bytes than requested is a retry condition, not necessarily an indication of ENOSPC [rhel-10.2]) + +* Thu Mar 26 2026 Miroslav Rezanina - 10.1.0-15 +- kvm-mirror-Fix-missed-dirty-bitmap-writes-during-startup.patch [RHEL-155601] +- Resolves: RHEL-155601 + (Mirror job can miss writes during startup, corrupting the copy [rhel-10.2]) + +* Wed Mar 18 2026 Miroslav Rezanina - 10.1.0-14 +- kvm-hw-uefi-add-variable-digest-to-vmstate.patch [RHEL-153058] +- kvm-block-Never-drop-BLOCK_IO_ERROR-with-action-stop-for.patch [RHEL-144004] +- Resolves: RHEL-153058 + (Qemu crashes with "double free" during restore --reset-nvram with uefi-vars secure boot) +- Resolves: RHEL-144004 + ([rhel-10] Regression in BLOCK_IO_ERROR event delivery with (w|r)error setting of 'stop' or 'enospc' due to event rate limiting) + +* Thu Feb 19 2026 Miroslav Rezanina - 10.1.0-13 +- kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch [RHEL-147425] +- kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch [RHEL-132749] +- kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch [RHEL-132749] +- kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch [RHEL-132749] +- kvm-scsi-save-load-SCSI-reservation-state.patch [RHEL-132749] +- kvm-docs-add-SCSI-migrate-pr-documentation.patch [RHEL-132749] +- kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch [RHEL-134989 RHEL-146584] +- Resolves: RHEL-147425 + (virtiofs: processes become stuck in request_wait_answer on virtiofs mounts) +- Resolves: RHEL-132749 + (Migrate SCSI PR state and preempt reservation upon live migration) +- Resolves: RHEL-134989 + (Hotplugged interface device can not be shown in the guest) +- Resolves: RHEL-146584 + ([RHEL-10.2][ARM]: Unable to Check the mem prefetched size on Guest) + +* Mon Feb 02 2026 Miroslav Rezanina - 10.1.0-12 +- kvm-rbd-Run-co-BH-CB-in-the-coroutine-s-AioContext.patch [RHEL-79118] +- kvm-curl-Fix-coroutine-waking.patch [RHEL-79118] +- kvm-block-io-Take-reqs_lock-for-tracked_requests.patch [RHEL-79118] +- kvm-qcow2-Re-initialize-lock-in-invalidate_cache.patch [RHEL-79118] +- kvm-qcow2-Fix-cache_clean_timer.patch [RHEL-79118] +- kvm-net-bundle-all-offloads-in-a-single-struct.patch [RHEL-143785] +- kvm-linux-headers-deal-with-counted_by-annotation.patch [RHEL-143785] +- kvm-linux-headers-Update-to-Linux-v6.17-rc1.patch [RHEL-143785] +- kvm-virtio-introduce-extended-features-type.patch [RHEL-143785] +- kvm-virtio-serialize-extended-features-state.patch [RHEL-143785] +- kvm-virtio-add-support-for-negotiating-extended-features.patch [RHEL-143785] +- kvm-virtio-pci-implement-support-for-extended-features.patch [RHEL-143785] +- kvm-vhost-add-support-for-negotiating-extended-features.patch [RHEL-143785] +- kvm-qmp-update-virtio-features-map-to-support-extended-f.patch [RHEL-143785] +- kvm-vhost-backend-implement-extended-features-support.patch [RHEL-143785] +- kvm-vhost-net-implement-extended-features-support.patch [RHEL-143785] +- kvm-virtio-net-implement-extended-features-support.patch [RHEL-143785] +- kvm-net-implement-tunnel-probing.patch [RHEL-143785] +- kvm-net-implement-UDP-tunnel-features-offloading.patch [RHEL-143785] +- Resolves: RHEL-79118 + ([network-storage][rbd][core-dump]installation of guest failed sometimes with multiqueue enabled [rhel10]) +- Resolves: RHEL-143785 + (backport support for GSO over UDP tunnel offload) + +* Tue Jan 13 2026 Miroslav Rezanina - 10.1.0-11 +- kvm-fix-pc_rhel_10_2_compat_len.patch [RHEL-126707] +- kvm-q35-increase-default-tseg-size.patch [RHEL-126707] +- kvm-hw-intc-ioapic-Fix-ACCEL_KERNEL_GSI_IRQFD_POSSIBLE-t.patch [RHEL-139028] +- kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch [RHEL-111853] +- Resolves: RHEL-126707 + ([qemu, rhel-10] increase default TSEG size) +- Resolves: RHEL-139028 + (Intel IOMMU VM freezes: "call_irq_handler: 3.37 No irq handler for vector"[rhel-10.2]) +- Resolves: RHEL-111853 + ([Intel 10.0 FEAT] [SPR] TDX: Virt-QEMU: QEMU Support [rhel-10]) + +* Mon Jan 05 2026 Miroslav Rezanina - 10.1.0-10 +- kvm-block-Fix-BDS-use-after-free-during-shutdown.patch [RHEL-108142] +- Resolves: RHEL-108142 + (QEMU crashes when stopping source VM during live migration) + +* Mon Dec 15 2025 Miroslav Rezanina - 10.1.0-9 +- kvm-monitor-generalize-query-mshv-info-mshv-to-query-acc.patch [RHEL-134212] +- kvm-block-Improve-comments-in-BlockLimits.patch [RHEL-110003] +- kvm-block-Expose-block-limits-for-images-in-QMP.patch [RHEL-110003] +- kvm-qemu-img-info-Optionally-show-block-limits.patch [RHEL-110003] +- kvm-qemu-img-info-Add-cache-mode-option.patch [RHEL-110003] +- kvm-rh-configs-enable-CONFIG_TDX-for-x86_64.patch [RHEL-111853] +- Resolves: RHEL-134212 + ([RHEL10.2] L1VH qemu downstream initial merge RHEL10.2) +- Resolves: RHEL-110003 + (Expose block limits of block nodes in QMP and qemu-img) +- Resolves: RHEL-111853 + ([Intel 10.0 FEAT] [SPR] TDX: Virt-QEMU: QEMU Support [rhel-10]) + +* Tue Dec 09 2025 Miroslav Rezanina - 10.1.0-8 +- kvm-block-backend-Fix-race-when-resuming-queued-requests.patch [RHEL-129540] +- kvm-file-posix-Handle-suspended-dm-multipath-better-for-.patch [RHEL-121543] +- kvm-accel-Add-Meson-and-config-support-for-MSHV-accelera.patch [RHEL-134212] +- kvm-target-i386-emulate-Allow-instruction-decoding-from-.patch [RHEL-134212] +- kvm-target-i386-mshv-Add-x86-decoder-emu-implementation.patch [RHEL-134212] +- kvm-hw-intc-Generalize-APIC-helper-names-from-kvm_-to-ac.patch [RHEL-134212] +- kvm-include-hw-hyperv-Add-MSHV-ABI-header-definitions.patch [RHEL-134212] +- kvm-linux-headers-linux-Add-mshv.h-headers.patch [RHEL-134212] +- kvm-accel-mshv-Add-accelerator-skeleton.patch [RHEL-134212] +- kvm-accel-mshv-Register-memory-region-listeners.patch [RHEL-134212] +- kvm-accel-mshv-Initialize-VM-partition.patch [RHEL-134212] +- kvm-treewide-rename-qemu_wait_io_event-qemu_wait_io_even.patch [RHEL-134212] +- kvm-accel-mshv-Add-vCPU-creation-and-execution-loop.patch [RHEL-134212] +- kvm-accel-mshv-Add-vCPU-signal-handling.patch [RHEL-134212] +- kvm-target-i386-mshv-Add-CPU-create-and-remove-logic.patch [RHEL-134212] +- kvm-target-i386-mshv-Implement-mshv_store_regs.patch [RHEL-134212] +- kvm-target-i386-mshv-Implement-mshv_get_standard_regs.patch [RHEL-134212] +- kvm-target-i386-mshv-Implement-mshv_get_special_regs.patch [RHEL-134212] +- kvm-target-i386-mshv-Implement-mshv_arch_put_registers.patch [RHEL-134212] +- kvm-target-i386-mshv-Set-local-interrupt-controller-stat.patch [RHEL-134212] +- kvm-target-i386-mshv-Register-CPUID-entries-with-MSHV.patch [RHEL-134212] +- kvm-target-i386-mshv-Register-MSRs-with-MSHV.patch [RHEL-134212] +- kvm-target-i386-mshv-Integrate-x86-instruction-decoder-e.patch [RHEL-134212] +- kvm-target-i386-mshv-Write-MSRs-to-the-hypervisor.patch [RHEL-134212] +- kvm-target-i386-mshv-Implement-mshv_vcpu_run.patch [RHEL-134212] +- kvm-accel-mshv-Handle-overlapping-mem-mappings.patch [RHEL-134212] +- kvm-qapi-accel-Allow-to-query-mshv-capabilities.patch [RHEL-134212] +- kvm-target-i386-mshv-Use-preallocated-page-for-hvcall.patch [RHEL-134212] +- kvm-docs-Add-mshv-to-documentation.patch [RHEL-134212] +- kvm-MAINTAINERS-Add-maintainers-for-mshv-accelerator.patch [RHEL-134212] +- kvm-accel-mshv-initialize-thread-name.patch [RHEL-134212] +- kvm-accel-mshv-use-return-value-of-handle_pio_str_read.patch [RHEL-134212] +- Resolves: RHEL-129540 + (Assertion failure on drain with iothread and I/O load) +- Resolves: RHEL-121543 + (The VM hit io error when do S3-PR integration on the pass-through failover multipath device) +- Resolves: RHEL-134212 + ([RHEL10.2] L1VH qemu downstream initial merge RHEL10.2) + +* Mon Dec 01 2025 Miroslav Rezanina - 10.1.0-7 +- kvm-pcie_sriov-Fix-broken-MMIO-accesses-from-SR-IOV-VFs.patch [RHEL-120115] +- kvm-arm-fix-oob-access-in-compat-handling.patch [RHEL-130478] +- Resolves: RHEL-120115 + (The vf nic created using the IGB emulated nic can not obtain ip address ) +- Resolves: RHEL-130478 + (Migration from RHEL 10.2 to RHEL 10.1 with virt-rhel10.0.0 machine type fails on Grace) + +* Tue Nov 25 2025 Miroslav Rezanina - 10.1.0-6 +- kvm-ram-block-attributes-fix-interaction-with-hugetlb-me.patch [RHEL-126708] +- kvm-ram-block-attributes-Unify-the-retrieval-of-the-bloc.patch [RHEL-126708] +- kvm-hw-s390x-Fix-a-possible-crash-with-passed-through-vi.patch [RHEL-128085] +- kvm-Fix-the-typo-of-vfio-pci-device-s-enable-migration-o.patch [RHEL-130704] +- Resolves: RHEL-126708 + ([RHEL 10]snp guest fail to boot with hugepage) +- Resolves: RHEL-128085 + (VM crashes during boot when virtio device is attached through vfio_ccw) +- Resolves: RHEL-130704 + ([rhel10] Fix the typo under vfio-pci device's enable-migration option ) + +* Fri Nov 14 2025 Miroslav Rezanina - 10.1.0-5 +- kvm-io-move-websock-resource-release-to-close-method.patch [RHEL-120116] +- kvm-io-fix-use-after-free-in-websocket-handshake-code.patch [RHEL-120116] +- kvm-vfio-Disable-VFIO-migration-with-MultiFD-support.patch [RHEL-126573] +- kvm-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-from-10..patch [RHEL-67323] +- kvm-hw-arm-smmu-common-Check-SMMU-has-PCIe-Root-Complex-.patch [RHEL-73800] +- kvm-hw-arm-virt-acpi-build-Re-arrange-SMMUv3-IORT-build.patch [RHEL-73800] +- kvm-hw-arm-virt-acpi-build-Update-IORT-for-multiple-smmu.patch [RHEL-73800] +- kvm-hw-arm-virt-Factor-out-common-SMMUV3-dt-bindings-cod.patch [RHEL-73800] +- kvm-hw-arm-virt-Add-an-SMMU_IO_LEN-macro.patch [RHEL-73800] +- kvm-hw-pci-Introduce-pci_setup_iommu_per_bus-for-per-bus.patch [RHEL-73800] +- kvm-hw-arm-virt-Allow-user-creatable-SMMUv3-dev-instanti.patch [RHEL-73800] +- kvm-qemu-options.hx-Document-the-arm-smmuv3-device.patch [RHEL-73800] +- kvm-bios-tables-test-Allow-for-smmuv3-test-data.patch [RHEL-73800] +- kvm-qtest-bios-tables-test-Add-tests-for-legacy-smmuv3-a.patch [RHEL-73800] +- kvm-qtest-bios-tables-test-Update-tables-for-smmuv3-test.patch [RHEL-73800] +- kvm-qtest-Do-not-run-bios-tables-test-on-aarch64.patch [] +- Resolves: RHEL-120116 + (CVE-2025-11234 qemu-kvm: VNC WebSocket handshake use-after-free [rhel-10.2]) +- Resolves: RHEL-126573 + (VFIO migration using multifd should be disabled by default) +- Resolves: RHEL-67323 + ([aarch64] Support ACPI based PCI hotplug on ARM) +- Resolves: RHEL-73800 + (NVIDIA:Grace-Hopper:Backport support for user-creatable nested SMMUv3 - RHEL 10.1) + +* Mon Nov 03 2025 Miroslav Rezanina - 10.1.0-4 +- kvm-qapi-machine-s390x-add-QAPI-event-SCLP_CPI_INFO_AVAI.patch [RHEL-104009 RHEL-105823 RHEL-73008] +- kvm-tests-functional-add-tests-for-SCLP-event-CPI.patch [RHEL-104009 RHEL-105823 RHEL-73008] +- kvm-redhat-Add-new-rhel9.8.0-and-rhel10.2.0-machine-type.patch [RHEL-104009 RHEL-105823 RHEL-73008] +- kvm-vfio-rename-field-to-num_initial_regions.patch [RHEL-118810] +- kvm-vfio-only-check-region-info-cache-for-initial-region.patch [RHEL-118810] +- kvm-arm-create-new-rhel-10.2-specific-virt-machine-type.patch [RHEL-105826 RHEL-105828] +- kvm-arm-create-new-rhel-9.8-specific-virt-machine-type.patch [RHEL-105826 RHEL-105828] +- kvm-x86-create-new-rhel-10.2-specific-pc-q35-machine-typ.patch [RHEL-105826 RHEL-105828] +- kvm-x86-create-new-rhel-9.8-specific-pc-q35-machine-type.patch [RHEL-105826 RHEL-105828] +- kvm-rh-enable-CONFIG_USB_STORAGE_BOT.patch [RHEL-101929] +- Resolves: RHEL-104009 + ([IBM 10.2 FEAT] KVM: Enhance machine type definition to include CPI and PCI passthru capabilities (qemu)) +- Resolves: RHEL-105823 + (Add new -rhel10.2.0 machine type to qemu-kvm [s390x]) +- Resolves: RHEL-73008 + ([IBM 10.2 FEAT] KVM: Implement Control Program Identification (qemu)) +- Resolves: RHEL-118810 + ([RHEL 10.2] Windows 11 VM fails to boot up with ramfb='on' with QEMU 10.1) +- Resolves: RHEL-105826 + (Add new -rhel10.2.0 machine type to qemu-kvm [aarch64]) +- Resolves: RHEL-105828 + (Add new -rhel10.2.0 machine type to qemu-kvm [x86_64]) +- Resolves: RHEL-101929 + (enable 'usb-bot' device for proper support of USB CD-ROM drives via libvirt ) + +* Mon Oct 20 2025 Miroslav Rezanina - 10.1.0-3 +- kvm-arm-kvm-report-registers-we-failed-to-set.patch [RHEL-119368] +- kvm-pcie_sriov-make-pcie_sriov_pf_exit-safe-on-non-SR-IO.patch [RHEL-116443] +- kvm-target-i386-add-compatibility-property-for-arch_capa.patch [RHEL-120253] +- kvm-target-i386-add-compatibility-property-for-pdcm-feat.patch [RHEL-120253] +- Resolves: RHEL-119368 + ([rhel10] Backport "arm/kvm: report registers we failed to set") +- Resolves: RHEL-116443 + (qemu crash after hot-unplug disk from the multifunction enabled bus,crash point PCIDevice *vf = dev->exp.sriov_pf.vf[i]) +- Resolves: RHEL-120253 + (Backport fixes for PDCM and ARCH_CAPABILITIES migration incompatibility) + +* Mon Sep 15 2025 Miroslav Rezanina - 10.1.0-2 +- kvm-e1000e-Prevent-crash-from-legacy-interrupt-firing-af.patch [RHEL-112882] +- Resolves: RHEL-112882 + ([DEV Task]: Assertion `core->delayed_causes == 0' failed with e1000e NIC) + +* Fri Aug 29 2025 Miroslav Rezanina - 10.1.0-1 +- Rebase to QEMU 10.1.0 [RHEL-105035] +- Resolves: RHEL-105035 + (Rebase qemu-kvm to QEMU 10.1.0) + +* Thu Aug 21 2025 Miroslav Rezanina - 10.0.0-12 +- kvm-RHEL-Pack-uefi-vars-module.patch [RHEL-102325] +- Resolves: RHEL-102325 + ([qemu] enable variable service for edk2) + +* Mon Aug 18 2025 Miroslav Rezanina - 10.0.0-11 +- kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch [RHEL-105440] +- Resolves: RHEL-105440 + (Openstack guest becomes inaccessible via network when storage network on the hypervisor is disabled/lost [rhel-10.1]) + +* Tue Aug 12 2025 Miroslav Rezanina - 10.0.0-10 +- kvm-Enable-uefi-variable-service-for-edk2.patch [RHEL-102325] +- Resolves: RHEL-102325 + ([qemu] enable variable service for edk2) + +* Mon Aug 04 2025 Miroslav Rezanina - 10.0.0-9 +- kvm-Declare-rtl8139-as-deprecated.patch [RHEL-45624] +- Resolves: RHEL-45624 + (Deprecate rtl8139 NIC in QEMU) + +* Mon Jul 28 2025 Miroslav Rezanina - 10.0.0-8 +- kvm-migration-multifd-move-macros-to-multifd-header.patch [RHEL-59697] +- kvm-migration-refactor-channel-discovery-mechanism.patch [RHEL-59697] +- kvm-migration-Add-save_postcopy_prepare-savevm-handler.patch [RHEL-59697] +- kvm-migration-ram-Implement-save_postcopy_prepare.patch [RHEL-59697] +- kvm-tests-qtest-migration-consolidate-set-capabilities.patch [RHEL-59697] +- kvm-migration-write-zero-pages-when-postcopy-enabled.patch [RHEL-59697] +- kvm-migration-enable-multifd-and-postcopy-together.patch [RHEL-59697] +- kvm-migration-Add-qtest-for-migration-over-RDMA.patch [RHEL-59697] +- kvm-qtest-migration-rdma-Enforce-RLIMIT_MEMLOCK-128MB-re.patch [RHEL-59697] +- kvm-qtest-migration-rdma-Add-test-for-rdma-migration-wit.patch [RHEL-59697] +- kvm-tests-qtest-migration-add-postcopy-tests-with-multif.patch [RHEL-59697] +- kvm-file-posix-Fix-aio-threads-performance-regression-af.patch [RHEL-96854] +- kvm-block-remove-outdated-comments-about-AioContext-lock.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-read-locked-bdrv_reopen_.patch [RHEL-88561] +- kvm-block-snapshot-move-drain-outside-of-read-locked-bdr.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-read-locked-bdrv_inactiv.patch [RHEL-88561] +- kvm-block-mark-bdrv_parent_change_aio_context-GRAPH_RDLO.patch [RHEL-88561] +- kvm-block-mark-change_aio_ctx-callback-and-instances-as-.patch [RHEL-88561] +- kvm-block-mark-bdrv_child_change_aio_context-GRAPH_RDLOC.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-bdrv_change_aio_context-.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-bdrv_try_change_aio_cont.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-bdrv_attach_child_common.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-bdrv_set_backing_hd_drai.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-bdrv_root_attach_child.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-bdrv_attach_child.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-quorum_add_child.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-bdrv_root_unref_child.patch [RHEL-88561] +- kvm-block-move-drain-outside-of-quorum_del_child.patch [RHEL-88561] +- kvm-blockdev-drain-while-unlocked-in-internal_snapshot_a.patch [RHEL-88561] +- kvm-blockdev-drain-while-unlocked-in-external_snapshot_a.patch [RHEL-88561] +- kvm-block-mark-bdrv_drained_begin-and-friends-as-GRAPH_U.patch [RHEL-88561] +- kvm-iotests-graph-changes-while-io-remove-image-file-aft.patch [RHEL-88561] +- kvm-iotests-graph-changes-while-io-add-test-case-with-re.patch [RHEL-88561] +- Resolves: RHEL-59697 + (Allow multifd+postcopy features being enabled together, but only use multifd during precopy ) +- Resolves: RHEL-96854 + (Performance Degradation(aio=threads) between Upstream Commit b75c5f9 and 984a32f) +- Resolves: RHEL-88561 + (qemu graph deadlock during job-dismiss) + +* Mon Jul 07 2025 Miroslav Rezanina - 10.0.0-7 +- kvm-s390x-Fix-leak-in-machine_set_loadparm.patch [RHEL-98555] +- kvm-hw-s390x-ccw-device-Fix-memory-leak-in-loadparm-sett.patch [RHEL-98555] +- kvm-target-i386-Update-EPYC-CPU-model-for-Cache-property.patch [RHEL-52650] +- kvm-target-i386-Update-EPYC-Rome-CPU-model-for-Cache-pro.patch [RHEL-52650] +- kvm-target-i386-Update-EPYC-Milan-CPU-model-for-Cache-pr.patch [RHEL-52650] +- kvm-target-i386-Add-couple-of-feature-bits-in-CPUID_Fn80.patch [RHEL-52650] +- kvm-target-i386-Update-EPYC-Genoa-for-Cache-property-per.patch [RHEL-52650] +- kvm-target-i386-Add-support-for-EPYC-Turin-model.patch [RHEL-52650] +- kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch [RHEL-95479] +- kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch [RHEL-95479] +- kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch [RHEL-95479] +- kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch [RHEL-95479] +- kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch [RHEL-95479] +- kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch [RHEL-95479] +- kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch [RHEL-95479] +- kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch [RHEL-95479] +- kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch [RHEL-95479] +- kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch [RHEL-95479] +- kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch [RHEL-95479] +- kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch [RHEL-95479] +- kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch [RHEL-95479] +- kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch [RHEL-95479] +- kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch [RHEL-95479] +- kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch [RHEL-95479] +- kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch [RHEL-95479] +- kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch [RHEL-95479] +- kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch [RHEL-95479] +- kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch [RHEL-95479] +- kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch [RHEL-95479] +- kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch [RHEL-95479] +- kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch [RHEL-95479] +- kvm-hw-ppc-pnv_occ-skip-automatic-zero-init-of-large-str.patch [RHEL-95479] +- kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch [RHEL-95479] +- kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch [RHEL-95479] +- kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch [RHEL-95479] +- kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch [RHEL-95479] +- kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch [RHEL-95479] +- kvm-net-socket-skip-automatic-zero-init-of-large-array.patch [RHEL-95479] +- kvm-net-stream-skip-automatic-zero-init-of-large-array.patch [RHEL-95479] +- kvm-hw-i386-amd_iommu-Isolate-AMDVI-PCI-from-amd-iommu-d.patch [RHEL-85649] +- kvm-hw-i386-amd_iommu-Allow-migration-when-explicitly-cr.patch [RHEL-85649] +- kvm-Enable-amd-iommu-device.patch [RHEL-85649] +- kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch [RHEL-83883] +- Resolves: RHEL-98555 + ([s390x][RHEL10.1][ccw-device] there would be memory leak with virtio_blk disks) +- Resolves: RHEL-52650 + ([AMDSERVER 10.1 Feature] Turin: Qemu EPYC-Turin Model) +- Resolves: RHEL-95479 + (-ftrivial-auto-var-init=zero reduced performance) +- Resolves: RHEL-85649 + ([RHEL 10]Qemu/amd-iommu: Add ability to manually specify the AMDVI-PCI device) +- Resolves: RHEL-83883 + (Video stuck after switchover phase when play one video during migration) + +* Fri Jun 20 2025 Miroslav Rezanina - 10.0.0-6 +- kvm-scsi-disk-Add-native-FUA-write-support.patch [RHEL-71962] +- kvm-Fix-handling-of-have_block_rbd.patch [RHEL-96057] +- kvm-Delete-obsolete-references-to-architectures.patch [RHEL-96057] +- kvm-Fix-arch-list-for-vgabios-and-ipxe-roms.patch [RHEL-96057] +- kvm-Disable-virtio-net-pci-romfile-loading-on-riscv64.patch [RHEL-96057] +- Resolves: RHEL-71962 + ([RFE] Implement FUA support in scsi-disk) +- Resolves: RHEL-96057 + (qemu-kvm: Various small issues in the spec file) + +* Mon Jun 09 2025 Miroslav Rezanina - 10.0.0-5 +- kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch [RHEL-65852] +- kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch [RHEL-65852] +- kvm-io-Fix-partial-struct-copy-in-qio_dns_resolver_looku.patch [RHEL-67706] +- kvm-util-qemu-sockets-Refactor-setting-client-sockopts-i.patch [RHEL-67706] +- kvm-util-qemu-sockets-Refactor-success-and-failure-paths.patch [RHEL-67706] +- kvm-util-qemu-sockets-Add-support-for-keep-alive-flag-to.patch [RHEL-67706] +- kvm-util-qemu-sockets-Refactor-inet_parse-to-use-QemuOpt.patch [RHEL-67706] +- kvm-util-qemu-sockets-Introduce-inet-socket-options-cont.patch [RHEL-67706] +- kvm-tests-unit-test-util-sockets-fix-mem-leak-on-error-o.patch [RHEL-67706] +- Resolves: RHEL-65852 + (Support multipath failover with scsi-block) +- Resolves: RHEL-67706 + (postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S') + +* Mon May 26 2025 Miroslav Rezanina - 10.0.0-4 +- kvm-block-Expand-block-status-mode-from-bool-to-flags.patch [RHEL-88435 RHEL-88437] +- kvm-file-posix-gluster-Handle-zero-block-status-hint-bet.patch [RHEL-88435 RHEL-88437] +- kvm-block-Let-bdrv_co_is_zero_fast-consolidate-adjacent-.patch [RHEL-88435 RHEL-88437] +- kvm-block-Add-new-bdrv_co_is_all_zeroes-function.patch [RHEL-88435 RHEL-88437] +- kvm-iotests-Improve-iotest-194-to-mirror-data.patch [RHEL-88435 RHEL-88437] +- kvm-mirror-Minor-refactoring.patch [RHEL-88435 RHEL-88437] +- kvm-mirror-Pass-full-sync-mode-rather-than-bool-to-inter.patch [RHEL-88435 RHEL-88437] +- kvm-mirror-Allow-QMP-override-to-declare-target-already-.patch [RHEL-88435 RHEL-88437] +- kvm-mirror-Drop-redundant-zero_target-parameter.patch [RHEL-88435 RHEL-88437] +- kvm-mirror-Skip-pre-zeroing-destination-if-it-is-already.patch [RHEL-88435 RHEL-88437] +- kvm-mirror-Skip-writing-zeroes-when-target-is-already-ze.patch [RHEL-88435 RHEL-88437] +- kvm-iotests-common.rc-add-disk_usage-function.patch [RHEL-88435 RHEL-88437] +- kvm-tests-Add-iotest-mirror-sparse-for-recent-patches.patch [RHEL-88435 RHEL-88437] +- kvm-mirror-Reduce-I-O-when-destination-is-detect-zeroes-.patch [RHEL-88435 RHEL-88437] +- Resolves: RHEL-88435 + (--migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-10.1]) +- Resolves: RHEL-88437 + (Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-10.1]) + +* Mon May 19 2025 Miroslav Rezanina - 10.0.0-3 +- kvm-migration-postcopy-Spatial-locality-page-hint-for-pr.patch [RHEL-85635] +- kvm-meson-configure-add-valgrind-option-en-dis-able-valg.patch [RHEL-88457] +- kvm-distro-add-an-explicit-valgrind-devel-build-dep.patch [RHEL-88457] +- kvm-Allow-guest-get-load-QGA-command.patch [RHEL-91219] +- Resolves: RHEL-85635 + (Video stuck about 1 min after switchover phase when play one video during postcopy-preempt migration) +- Resolves: RHEL-88457 + (qemu inadvertantly built with valgrind coroutine stack debugging on x86_64) +- Resolves: RHEL-91219 + ([qemu-guest-agent] Enable 'guest-get-load' by default [RHEL-10]) + +* Mon May 12 2025 Miroslav Rezanina - 10.0.0-2 +- kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch [RHEL-87642] +- kvm-block-io-skip-head-tail-requests-on-EINVAL.patch [RHEL-87642] +- kvm-file-posix-Fix-crash-on-discard_granularity-0.patch [RHEL-87642] +- kvm-Enable-vhost-user-gpu-pci-for-RHIVOS.patch [RHEL-86056] +- Resolves: RHEL-87642 + (QEMU sends unaligned discards on 4K devices[RHEL-10]) +- Resolves: RHEL-86056 + (Enable 'vhost-user-gpu-pci' in qemu-kvm for RHIVOS) + +* Wed Apr 23 2025 Miroslav Rezanina - 10.0.0-1 +- Rebase to QEMU 10.0.0 [RHEL-74473] +- Resolves: RHEL-74473 + (Rebase qemu-kvm to QEMU 10.0.0) + +* Mon Apr 07 2025 Miroslav Rezanina - 9.1.0-17 +- kvm-Also-recommend-systemtap-devel-from-qemu-tools.patch [RHEL-83535] +- Resolves: RHEL-83535 + ([Qemu RHEL-10] qemu-trace-stap should handle lack of stap more gracefully) + +* Tue Mar 25 2025 Miroslav Rezanina - 9.1.0-16 +- kvm-migration-Fix-UAF-for-incoming-migration-on-Migratio.patch [RHEL-69776] +- kvm-scripts-improve-error-from-qemu-trace-stap-on-missin.patch [RHEL-83535] +- kvm-Recommend-systemtap-client-from-qemu-tools.patch [RHEL-83535] +- Resolves: RHEL-69776 + ([rhel10]Guest crashed on the target host when the migration was canceled) +- Resolves: RHEL-83535 + ([Qemu RHEL-10] qemu-trace-stap should handle lack of stap more gracefully) + +* Mon Feb 17 2025 Miroslav Rezanina - 9.1.0-15 +- kvm-migration-Add-helper-to-get-target-runstate.patch [RHEL-54670] +- kvm-qmp-cont-Only-activate-disks-if-migration-completed.patch [RHEL-54670] +- kvm-migration-block-Make-late-block-active-the-default.patch [RHEL-54670] +- kvm-migration-block-Apply-late-block-active-behavior-to-.patch [RHEL-54670] +- kvm-migration-block-Fix-possible-race-with-block_inactiv.patch [RHEL-54670] +- kvm-migration-block-Rewrite-disk-activation.patch [RHEL-54670] +- kvm-block-Add-active-field-to-BlockDeviceInfo.patch [RHEL-54670] +- kvm-block-Allow-inactivating-already-inactive-nodes.patch [RHEL-54670] +- kvm-block-Inactivate-external-snapshot-overlays-when-nec.patch [RHEL-54670] +- kvm-migration-block-active-Remove-global-active-flag.patch [RHEL-54670] +- kvm-block-Don-t-attach-inactive-child-to-active-node.patch [RHEL-54670] +- kvm-block-Fix-crash-on-block_resize-on-inactive-node.patch [RHEL-54670] +- kvm-block-Add-option-to-create-inactive-nodes.patch [RHEL-54670] +- kvm-block-Add-blockdev-set-active-QMP-command.patch [RHEL-54670] +- kvm-block-Support-inactive-nodes-in-blk_insert_bs.patch [RHEL-54670] +- kvm-block-export-Don-t-ignore-image-activation-error-in-.patch [RHEL-54670] +- kvm-block-Drain-nodes-before-inactivating-them.patch [RHEL-54670] +- kvm-block-export-Add-option-to-allow-export-of-inactive-.patch [RHEL-54670] +- kvm-nbd-server-Support-inactive-nodes.patch [RHEL-54670] +- kvm-iotests-Add-filter_qtest.patch [RHEL-54670] +- kvm-iotests-Add-qsd-migrate-case.patch [RHEL-54670] +- kvm-iotests-Add-NBD-based-tests-for-inactive-nodes.patch [RHEL-54670] +- Resolves: RHEL-54670 + (Provide QMP command for block device reactivation after migration [rhel-10.0]) + +* Mon Feb 10 2025 Miroslav Rezanina - 9.1.0-14 +- kvm-net-Fix-announce_self.patch [RHEL-73894] +- kvm-vhost-Add-stubs-for-the-migration-state-transfer-int.patch [RHEL-78370] +- kvm-virtio-net-vhost-user-Implement-internal-migration.patch [RHEL-78370] +- Resolves: RHEL-73894 + (No RARP packets on the destination after migration [rhel-10]) +- Resolves: RHEL-78370 + (Add vhost-user internal migration for passt) + +* Mon Feb 03 2025 Miroslav Rezanina - 9.1.0-13 +- kvm-nbd-server-Silence-server-warnings-on-port-probes.patch [RHEL-76908] +- Resolves: RHEL-76908 + (Ensure qemu as NBD server does not flood logs [rhel-10]) + +* Mon Jan 27 2025 Miroslav Rezanina - 9.1.0-12 +- kvm-pci-ensure-valid-link-status-bits-for-downstream-por.patch [RHEL-65618] +- kvm-pc-bios-s390-ccw-Abort-IPL-on-invalid-loadparm.patch [RHEL-72717] +- kvm-pc-bios-s390-ccw-virtio-Add-a-function-to-reset-a-vi.patch [RHEL-72717] +- kvm-pc-bios-s390-ccw-Fix-boot-problem-with-virtio-net-de.patch [RHEL-72717] +- kvm-pc-bios-s390-ccw-netmain-Fix-error-messages-with-reg.patch [RHEL-72717] +- kvm-arm-disable-pauth-for-virt-rhel9-in-RHEL10.patch [RHEL-71761] +- Resolves: RHEL-65618 + ([RHEL10] Failed to hot add PCIe device behind xio3130 downstream) +- Resolves: RHEL-72717 + (Boot fall back to cdrom from network not always working) +- Resolves: RHEL-71761 + ([Nvidia "Grace"] Lack of "PAuth" CPU feature results in live migration failure from RHEL 9.6 to 10) + +* Mon Jan 20 2025 Miroslav Rezanina - 9.1.0-11 +- kvm-target-i386-Make-sure-SynIC-state-is-really-updated-.patch [RHEL-73002] +- kvm-hw-virtio-fix-crash-in-processing-balloon-stats.patch [RHEL-73835] +- kvm-qga-Add-log-to-guest-fsfreeze-thaw-command.patch [RHEL-74361] +- kvm-qemu-ga-Optimize-freeze-hook-script-logic-of-logging.patch [RHEL-74461] +- Resolves: RHEL-73002 + (kvm-unti kvm-hyperv_synic test is stuck on AMD with COS9 [rhel-10]) +- Resolves: RHEL-73835 + (VM crashes when requesting domstats [rhel-10]) +- Resolves: RHEL-74361 + (qemu-ga logs only "guest-fsfreeze called" (but not "guest-fsthaw called")) +- Resolves: RHEL-74461 + (fsfreeze hooks doesn't log error on system logs when running hook fails [rhel-10]) + +* Mon Jan 13 2025 Miroslav Rezanina - 9.1.0-10 +- kvm-qdev-Fix-set_pci_devfn-to-visit-option-only-once.patch [RHEL-43412] +- kvm-tests-avocado-hotplug_blk-Fix-addr-in-device_add-com.patch [RHEL-43412] +- kvm-qdev-monitor-avoid-QemuOpts-in-QMP-device_add.patch [RHEL-43412] +- kvm-vl-use-qmp_device_add-in-qemu_create_cli_devices.patch [RHEL-43412] +- kvm-pc-q35-Bump-max_cpus-to-4096-vcpus.patch [RHEL-57668] +- kvm-vhost-fail-device-start-if-iotlb-update-fails.patch [RHEL-73005] +- kvm-virtio-net-disable-USO-for-all-RHEL9.patch [RHEL-69500] +- Resolves: RHEL-43412 + (qom-get iothread-vq-mapping is empty on new hotplug disk [rhel-10.0-beta]) +- Resolves: RHEL-57668 + ([RFE] [HPEMC] [RHEL-10.0] qemu-kvm: support up to 4096 VCPUs) +- Resolves: RHEL-73005 + (qemu-kvm: vhost: reports error while updating IOTLB entries) +- Resolves: RHEL-69500 + ([Stable_Guest_ABI][USO][9.6.0-machine-type]From 10.0 to RHEL.9.6.0 the guest with 9.6 machine type only, the guest crashed with - qemu-kvm: Features 0x1c0010130afffa7 unsupported. Allowed features: 0x10179bfffe7) + +* Mon Jan 06 2025 Miroslav Rezanina - 9.1.0-9 +- kvm-linux-headers-Update-to-Linux-v6.12-rc5.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-msa10-subfunctions.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-msa11-subfunctions.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-msa12-changes.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-msa13-subfunctions.patch [RHEL-32665] +- kvm-s390x-cpumodel-Add-ptff-Query-Time-Stamp-Event-QTSE-.patch [RHEL-32665] +- kvm-linux-headers-Update-to-Linux-6.13-rc1.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-Concurrent-functions-facility-sup.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-Vector-Enhancements-facility-3.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-Miscellaneous-Instruction-Extensi.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-Vector-Packed-Decimal-Enhancement.patch [RHEL-32665] +- kvm-s390x-cpumodel-add-Ineffective-nonconstrained-transa.patch [RHEL-32665] +- kvm-s390x-cpumodel-Add-Sequential-Instruction-Fetching-f.patch [RHEL-32665] +- kvm-s390x-cpumodel-correct-PLO-feature-wording.patch [RHEL-32665] +- kvm-s390x-cpumodel-Add-PLO-extension-facility.patch [RHEL-32665] +- kvm-s390x-cpumodel-gen17-model.patch [RHEL-32665] +- kvm-qga-skip-bind-mounts-in-fs-list.patch [RHEL-71939] +- kvm-hw-char-pl011-Use-correct-masks-for-IBRD-and-FBRD.patch [RHEL-67108] +- Resolves: RHEL-32665 + ([IBM 10.0 FEAT] KVM: CPU model for new IBM Z HW - qemu-kvm part) +- Resolves: RHEL-71939 + (qemu-ga cannot freeze filesystems with sentinelone) +- Resolves: RHEL-67108 + ([aarch64] [rhel-10.0] Backport some important post 9.1 qemu fixes) + +* Fri Dec 13 2024 Miroslav Rezanina - 9.1.0-8 +- kvm-migration-Allow-pipes-to-keep-working-for-fd-migrati.patch [RHEL-69047] +- Resolves: RHEL-69047 + (warning: fd: migration to a file is deprecated when create or revert a snapshot) + +* Tue Dec 03 2024 Miroslav Rezanina - 9.1.0-7 +- kvm-virtio-net-Add-queues-before-loading-them.patch [RHEL-58316] +- kvm-docs-system-s390x-bootdevices-Update-loadparm-docume.patch [RHEL-68444] +- kvm-docs-system-bootindex-Make-it-clear-that-s390x-can-a.patch [RHEL-68444] +- kvm-hw-s390x-Restrict-loadparm-property-to-devices-that-.patch [RHEL-68444] +- kvm-hw-Add-loadparm-property-to-scsi-disk-devices-for-bo.patch [RHEL-68444] +- kvm-scsi-fix-allocation-for-s390x-loadparm.patch [RHEL-68444] +- kvm-pc-bios-s390x-Initialize-cdrom-type-to-false-for-eac.patch [RHEL-68444] +- kvm-pc-bios-s390x-Initialize-machine-loadparm-before-pro.patch [RHEL-68444] +- kvm-pc-bios-s390-ccw-Re-initialize-receive-queue-index-b.patch [RHEL-68444] +- Resolves: RHEL-58316 + (qemu crashed when migrate vm with multiqueue from rhel9.4 to rhel10.0) +- Resolves: RHEL-68444 + (The new "boot order" feature is sometimes not working as expected [RHEL 10]) + +* Mon Nov 25 2024 Miroslav Rezanina - 9.1.0-6 +- kvm-vfio-container-Fix-container-object-destruction.patch [RHEL-67936] +- kvm-virtio-net-disable-USO-for-RHEL9.patch [RHEL-40950] +- kvm-qemu-guest-agent-add-new-api-to-allow-rpc.patch [RHEL-60223] +- Resolves: RHEL-67936 + (QEMU should fail gracefully with passthrough devices in SEV-SNP guests) +- Resolves: RHEL-40950 + ([Stable_Guest_ABI][USO]From 10-beta to RHEL.9.5.0 the guest with 9.4 machine type only, the guest crashed with - qemu-kvm: Features 0x1c0010130afffa7 unsupported. Allowed features: 0x10179bfffe7 ) +- Resolves: RHEL-60223 + ([qemu-guest-agent] Add new api 'guest-network-get-route' to allow-rpc) + +* Tue Nov 19 2024 Miroslav Rezanina - 9.1.0-5 +- kvm-migration-Ensure-vmstate_save-sets-errp.patch [RHEL-63051] +- kvm-kvm-replace-fprintf-with-error_report-printf-in-kvm_.patch [RHEL-57685] +- kvm-kvm-refactor-core-virtual-machine-creation-into-its-.patch [RHEL-57685] +- kvm-accel-kvm-refactor-dirty-ring-setup.patch [RHEL-57685] +- kvm-KVM-Dynamic-sized-kvm-memslots-array.patch [RHEL-57685] +- kvm-KVM-Define-KVM_MEMSLOTS_NUM_MAX_DEFAULT.patch [RHEL-57685] +- kvm-KVM-Rename-KVMMemoryListener.nr_used_slots-to-nr_slo.patch [RHEL-57685] +- kvm-KVM-Rename-KVMState-nr_slots-to-nr_slots_max.patch [RHEL-57685] +- kvm-Require-new-dtrace-package.patch [RHEL-67899] +- Resolves: RHEL-63051 + (qemu crashed after killed virtiofsd during migration) +- Resolves: RHEL-57685 + (Bad migration performance when performing vGPU VM live migration ) +- Resolves: RHEL-67899 + (Failed to build qemu-kvm due to missing dtrace [rhel-10.0]) + +* Tue Nov 12 2024 Miroslav Rezanina - 9.1.0-4.el10 +- kvm-accel-kvm-check-for-KVM_CAP_READONLY_MEM-on-VM.patch [RHEL-58928] +- kvm-hw-s390x-ipl-Provide-more-memory-to-the-s390-ccw.img.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Use-the-libc-from-SLOF-and-remove-s.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Link-the-netboot-code-into-the-main.patch [RHEL-58153] +- kvm-redhat-Remove-the-s390-netboot.img-from-the-spec-fil.patch [RHEL-58153] +- kvm-hw-s390x-Remove-the-possibility-to-load-the-s390-net.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Merge-netboot.mak-into-the-main-Mak.patch [RHEL-58153] +- kvm-docs-system-s390x-bootdevices-Update-the-documentati.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Remove-panics-from-ISO-IPL-path.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Remove-panics-from-ECKD-IPL-path.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Remove-panics-from-SCSI-IPL-path.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Remove-panics-from-DASD-IPL-path.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Remove-panics-from-Netboot-IPL-path.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Enable-failed-IPL-to-return-after-e.patch [RHEL-58153] +- kvm-include-hw-s390x-Add-include-files-for-common-IPL-st.patch [RHEL-58153] +- kvm-s390x-Add-individual-loadparm-assignment-to-CCW-devi.patch [RHEL-58153] +- kvm-hw-s390x-Build-an-IPLB-for-each-boot-device.patch [RHEL-58153] +- kvm-s390x-Rebuild-IPLB-for-SCSI-device-directly-from-DIA.patch [RHEL-58153] +- kvm-pc-bios-s390x-Enable-multi-device-boot-loop.patch [RHEL-58153] +- kvm-docs-system-Update-documentation-for-s390x-IPL.patch [RHEL-58153] +- kvm-tests-qtest-Add-s390x-boot-order-tests-to-cdrom-test.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Clarify-alignment-is-in-bytes.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Don-t-generate-TEXTRELs.patch [RHEL-58153] +- kvm-pc-bios-s390-ccw-Introduce-EXTRA_LDFLAGS.patch [RHEL-58153] +- kvm-vnc-fix-crash-when-no-console-attached.patch [RHEL-50529] +- kvm-vfio-migration-Report-only-stop-copy-size-in-vfio_st.patch [RHEL-64308] +- kvm-vfio-migration-Change-trace-formats-from-hex-to-deci.patch [RHEL-64308] +- kvm-kvm-Allow-kvm_arch_get-put_registers-to-accept-Error.patch [RHEL-20574] +- kvm-target-i386-kvm-Report-which-action-failed-in-kvm_ar.patch [RHEL-20574] +- kvm-target-i386-cpu-set-correct-supported-XCR0-features-.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-do-not-rely-on-ExtSaveArea-for-accelerat.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-return-bool-from-x86_cpu_filter_features.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-add-AVX10-feature-and-AVX10-version-prop.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-add-CPUID.24-features-for-AVX10.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-Add-feature-dependencies-for-AVX10.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-Add-AVX512-state-when-AVX10-is-supported.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-Introduce-GraniteRapids-v2-model.patch [RHEL-30315 RHEL-45110] +- kvm-target-i386-add-sha512-sm3-sm4-feature-bits.patch [RHEL-30315 RHEL-45110] +- Resolves: RHEL-58928 + (Boot SNP guests failed with qemu-kvm: kvm_set_user_memory_region) +- Resolves: RHEL-58153 + ([IBM 10.0 FEAT] KVM: Full boot order support - qemu part) +- Resolves: RHEL-50529 + (Qemu-kvm crashed if no display device setting and switching display by remote-viewer) +- Resolves: RHEL-64308 + (High threshold value observed in vGPU live migration) +- Resolves: RHEL-20574 + (Fail migration properly when put cpu register fails) +- Resolves: RHEL-30315 + ([Intel 10.0 FEAT] [GNR] Virt-QEMU: Add AVX10.1 instruction support) +- Resolves: RHEL-45110 + ([Intel 10.0 FEAT] [CWF][DMR] Virt-QEMU: Advertise new instructions SHA2-512NI, SM3, and SM4) + +* Mon Oct 07 2024 Miroslav Rezanina - 9.1.0-3 +- kvm-hostmem-Apply-merge-property-after-the-memory-region.patch [RHEL-58936] +- Resolves: RHEL-58936 + ([RHEL-10.0] QEMU core dump on applying merge property to memory backend) + +* Mon Sep 30 2024 Miroslav Rezanina - 9.1.0-2 +- kvm-x86-create-new-pc-q35-machine-type-for-rhel-9.6.patch [RHEL-29002 RHEL-29003 RHEL-35587 RHEL-38411 RHEL-45141] +- kvm-arm-create-new-virt-machine-type-for-rhel-9.6.patch [RHEL-29002 RHEL-29003 RHEL-35587 RHEL-38411 RHEL-45141] +- kvm-x86-create-pc-i440fx-machine-type-for-rhel10.patch [RHEL-29002 RHEL-29003 RHEL-35587 RHEL-38411 RHEL-45141] +- kvm-x86-create-pc-q35-machine-type-for-rhel10.patch [RHEL-29002 RHEL-29003 RHEL-35587 RHEL-38411 RHEL-45141] +- kvm-arm-create-virt-machine-type-for-rhel10.patch [RHEL-29002 RHEL-29003 RHEL-35587 RHEL-38411 RHEL-45141] +- kvm-x86-remove-deprecated-rhel-machine-types.patch [RHEL-29002 RHEL-29003 RHEL-35587 RHEL-38411 RHEL-45141] +- kvm-remove-stale-compat-definitions.patch [RHEL-29002 RHEL-29003 RHEL-35587 RHEL-38411 RHEL-45141] +- kvm-RH-Author-Shaoqin-Huang-shahuang-redhat.com.patch [RHEL-38374] +- kvm-qemu-guest-agent-Update-the-logfile-path-of-qga-fsfr.patch [RHEL-57028] +- Resolves: RHEL-29002 + (Remove the existing deprecated machine types in RHEL-10) +- Resolves: RHEL-29003 + (Deprecate RHEL-9 machine types in RHEL-10) +- Resolves: RHEL-35587 + (Create a pc-i440fx-rhel10.0 machine type) +- Resolves: RHEL-38411 + ([Fujitsu 10.0 FEAT]: qemu-kvm: Continue to support i440fx for RHEL10) +- Resolves: RHEL-45141 + (Introduce virt-rhel10.0 arm-virt machine type [aarch64]) +- Resolves: RHEL-38374 + (aarch64 SMBIOS 'Manufacturer' and 'Product Name' differ from x86 ones [rhel-10]) +- Resolves: RHEL-57028 + (fsfreeze hooks break on the systems first restorecon [rhel-10]) + +* Tue Sep 10 2024 Miroslav Rezanina - 9.1.0-1 +- Rebase to QEMU 9.1.0 [RHEL-41246] +- Resolves: RHEL-41246 + (Rebase qemu-9.1 for RHEL 10.0) * Mon Aug 26 2024 Miroslav Rezanina - 9.0.0-8 - kvm-x86-cpu-update-deprecation-string-to-match-lowest-un.patch [RHEL-54260] @@ -1381,9 +2245,6 @@ useradd -r -u 107 -g qemu -G kvm -d / -s /sbin/nologin \ - Resolves: RHEL-43410 (aio=native: Assertion failure `laiocb->co->ctx == laiocb->ctx->aio_context' with block_resize) -* Mon Jun 24 2024 Troy Dawson - 18:9.0.0-2.1 -- Bump release for June 2024 mass rebuild - * Mon Jun 10 2024 Miroslav Rezanina - 9.0.0-2 - kvm-Enable-vhost-user-snd-pci-device.patch [RHEL-37563] - Resolves: RHEL-37563 diff --git a/sources b/sources index a4e2afa..cf81344 100644 --- a/sources +++ b/sources @@ -1 +1 @@ -SHA512 (qemu-9.0.0.tar.xz) = 1603517cd4c93632ba60ad7261eb67374f12a744bf58f10b0e8686e46d3a02d8b6bf58a0c617f23a1868084aaba6386c24341894f75539e0b816091718721427 +SHA512 (qemu-10.1.0.tar.xz) = 20552a524b6b298181df1af7084b470ded3fe8d1505f05011dda3c33cbc3d91f518ce026b44ba1a8b7f34c64ae81afddceda383066f4772a3a2a6333a2638caf