diff --git a/kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch b/kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch new file mode 100644 index 0000000..cd7c009 --- /dev/null +++ b/kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch @@ -0,0 +1,81 @@ +From eee1f8abab9cbcb64ab690737f1a8db293d87c05 Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Wed, 11 Feb 2026 09:57:37 -0500 +Subject: [PATCH 7/7] Revert "hw/arm/virt: Use ACPI PCI hotplug by default from + 10.2 onwards" + +RH-Author: Eric Auger +RH-MergeRequest: 465: Revert "hw/arm/virt: Use ACPI PCI hotplug by default from 10.2 onwards" +RH-Jira: RHEL-134989 RHEL-146584 +RH-Acked-by: Sebastian Ott +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Cornelia Huck +RH-Commit: [1/1] e22612dc81813762f8d7f4bc9f75df0b9f2135e0 (eauger1/centos-qemu-kvm) + +JIRA: https://issues.redhat.com/browse/RHEL-134989 +JIRA: https://issues.redhat.com/browse/RHEL-146584 +UPSTREAM: RHEL-only + +This reverts commit 58cba97a715fa3f506234e718191fcc34286f333. + +Conflicts: small contextual conflict when reverting changes in +hw/arm/virt.c due to subsequent fix by commit +fa0a758781fc ("arm: fix oob access in compat handling") + +Unfortunately the change of the default for the PCI hotplug method +introduced some regressions that cannot be fixed in 10.2 cycle. An +example is hotplugging a virtio-net-pci device with page-per-vq=true. +This induces an increase in the BAR size which is larger than the +default size the FW accomodates. At the moment we do not have any +workaround for those devices with large BARs, ie. we noticed +pcie-root-port pref64-reserve does not work as on x86 and we do not +have any way to opt-in for legacy PCIe hotplug at libvirt +level. So let's revert the change until we get all those stuff +properly fixed. + +Signed-off-by: Eric Auger +--- + hw/arm/virt.c | 10 ---------- + 1 file changed, 10 deletions(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index 1cfb386f64..752dc08720 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -95,16 +95,9 @@ + + static GlobalProperty arm_virt_compat[] = { + { TYPE_VIRTIO_IOMMU_PCI, "aw-bits", "48" }, +- { TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "on" }, + }; + static const size_t arm_virt_compat_len = G_N_ELEMENTS(arm_virt_compat); + +-GlobalProperty arm_acpi_pci_hp_disabled_compat[] = { +- { TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "off" }, +-}; +-static const size_t arm_acpi_pci_hp_disabled_compat_len = +- G_N_ELEMENTS(arm_acpi_pci_hp_disabled_compat); +- + /* + * RHEL9 kernels have pauth disabled while RHEL10 has it enabled, + * since qemu will setup the VM with pauth when KVM supports it we +@@ -112,7 +105,6 @@ static const size_t arm_acpi_pci_hp_disabled_compat_len = + */ + GlobalProperty arm_rhel9_compat[] = { + {TYPE_ARM_CPU, "pauth", "off", .optional = true}, +- {TYPE_ACPI_GED, "acpi-pci-hotplug-with-bridge-support", "off" }, + }; + const size_t arm_rhel9_compat_len = G_N_ELEMENTS(arm_rhel9_compat); + +@@ -3768,8 +3760,6 @@ static void virt_rhel_machine_10_0_0_options(MachineClass *mc) + + /* QEMU 9.1 and earlier have only a stage-1 SMMU, not a nested s1+2 one */ + vmc->no_nested_smmu = true; +- compat_props_add(mc->compat_props, arm_acpi_pci_hp_disabled_compat, +- arm_acpi_pci_hp_disabled_compat_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_2, hw_compat_rhel_10_2_len); + compat_props_add(mc->compat_props, hw_compat_rhel_10_1, hw_compat_rhel_10_1_len); + } +-- +2.47.3 + diff --git a/kvm-docs-add-SCSI-migrate-pr-documentation.patch b/kvm-docs-add-SCSI-migrate-pr-documentation.patch new file mode 100644 index 0000000..507524b --- /dev/null +++ b/kvm-docs-add-SCSI-migrate-pr-documentation.patch @@ -0,0 +1,118 @@ +From a47cd8b532de2235c3be76a79f42d40c32a4fa58 Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:35 -0500 +Subject: [PATCH 6/7] docs: add SCSI migrate-pr documentation + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [5/5] e081d8f5e12d1228cd3de395964c0cb4d483559e (stefanha/centos-stream-qemu-kvm) + +Suggested-by: Paolo Bonzini +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-6-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit a67819adb2212977360e9290bd005badb07dd2e4) +Signed-off-by: Stefan Hajnoczi +--- + docs/system/device-emulation.rst | 1 + + docs/system/devices/scsi/index.rst | 10 +++++ + docs/system/devices/scsi/migrate-pr.rst | 54 +++++++++++++++++++++++++ + 3 files changed, 65 insertions(+) + create mode 100644 docs/system/devices/scsi/index.rst + create mode 100644 docs/system/devices/scsi/migrate-pr.rst + +diff --git a/docs/system/device-emulation.rst b/docs/system/device-emulation.rst +index 911381643f..72a9dbd54d 100644 +--- a/docs/system/device-emulation.rst ++++ b/docs/system/device-emulation.rst +@@ -91,6 +91,7 @@ Emulated Devices + devices/keyboard.rst + devices/net.rst + devices/nvme.rst ++ devices/scsi/index.rst + devices/usb.rst + devices/vhost-user.rst + devices/virtio-gpu.rst +diff --git a/docs/system/devices/scsi/index.rst b/docs/system/devices/scsi/index.rst +new file mode 100644 +index 0000000000..4f0929b0ca +--- /dev/null ++++ b/docs/system/devices/scsi/index.rst +@@ -0,0 +1,10 @@ ++SCSI Devices ++============ ++ ++Several SCSI devices are available in QEMU. They are primarily used for block ++storage. ++ ++.. toctree:: ++ :maxdepth: 1 ++ ++ migrate-pr.rst +diff --git a/docs/system/devices/scsi/migrate-pr.rst b/docs/system/devices/scsi/migrate-pr.rst +new file mode 100644 +index 0000000000..a8f2790a86 +--- /dev/null ++++ b/docs/system/devices/scsi/migrate-pr.rst +@@ -0,0 +1,54 @@ ++.. ++ SPDX-License-Identifier: GPL-2.0-or-later ++ ++.. _scsi_migrate_pr: ++ ++SCSI Persistent Reservation Live Migration ++========================================== ++ ++This document explains how to live migrate SCSI Persistent Reservations. ++ ++The ``scsi-block`` device migrates SCSI Persistent Reservations when the ++``migrate-pr=on`` parameter is given. Migration is enabled by default in ++versioned machine types since QEMU 11.0. It is disabled by default on older ++machine types and needs to be explicitly enabled with ``--device ++scsi-block,migrate-pr=on,...``. ++ ++When migration is enabled, QEMU snoops PERSISTENT RESERVATION OUT commands and ++tracks the reservation key registered by the guest as well as reservations that ++the guest acquires. This information is migrated along with the guest and the ++destination QEMU submits a PERSISTENT RESERVATION OUT command with the PREEMPT ++service action to atomically transfer the reservation to the destination before ++the guest starts running on the destination. ++ ++The following persistent reservation capabilities reported by the PERSISTENT ++RESERVATION IN command with the REPORT CAPABILITIES service action are masked ++from the guest by QEMU when migration is enabled: ++ ++ * Specify Initiator Ports Capable (SIP_C) ++ * All Target Ports Capable (ATC_C) ++ ++When migration is disabled, the ``scsi-block`` device is live migrated but ++reservations remain in place on the source. Usually this is not the intended ++behavior unless there is another mechanism to update reservations during ++migration. The PERSISTENT RESERVATION IN command also does not mask ++capabilities reported to the guest when migration is disabled. ++ ++Limitations ++----------- ++ ++QEMU does not remember snooped reservation details across restart, so software ++inside the guest must acquire the reservation after boot in order for live ++migration to work. Similarly, if the reservation is acquired outside the guest ++then it will not live migrate along with the guest. ++ ++Snooping only considers the PERSISTENT RESERVATION OUT commands from the guest ++and does not track reservation changes made by other SCSI initiators. QEMU's ++snooped reservation details can become stale if another SCSI initiator ++makes changes to the reservation. ++ ++Guests running on the same host share a single SCSI initiator identity unless ++Fibre Channel N_Port ID Virtualization is configured. As a consequence, ++multiple guests on the same hosts may observe unexpected behavior if they use ++the same physical LUN. From the LUN's perspective all guests are the same ++initiator and there is no way to distinguish between guests. +-- +2.47.3 + diff --git a/kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch b/kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch new file mode 100644 index 0000000..3f4a9a4 --- /dev/null +++ b/kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch @@ -0,0 +1,131 @@ +From d6784a187e586d6e8541ab8b1ab41541c768e614 Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:32 -0500 +Subject: [PATCH 3/7] scsi: add error reporting to scsi_SG_IO() + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [2/5] 7858df3edd4f42eb1b9d3bf515b3470dd812327e (stefanha/centos-stream-qemu-kvm) + +Report the details of the SG_IO ioctl failure if an Error pointer is +provided. This information aids troubleshooting and will be used by the +SCSI Persistent Reservations migration code. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-3-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit 6302598fe538206fd02494007ab5d218524dc7a7) +Signed-off-by: Stefan Hajnoczi +--- + hw/scsi/scsi-disk.c | 2 +- + hw/scsi/scsi-generic.c | 33 ++++++++++++++++++++++++++++----- + include/hw/scsi/scsi.h | 2 +- + 3 files changed, 30 insertions(+), 7 deletions(-) + +diff --git a/hw/scsi/scsi-disk.c b/hw/scsi/scsi-disk.c +index ca0214c404..c736988ef6 100644 +--- a/hw/scsi/scsi-disk.c ++++ b/hw/scsi/scsi-disk.c +@@ -2749,7 +2749,7 @@ static int get_device_type(SCSIDiskState *s) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(s->qdev.conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), s->qdev.io_timeout); ++ buf, sizeof(buf), s->qdev.io_timeout, NULL); + if (ret < 0) { + return -1; + } +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index b27ad48f18..4f851186f6 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -526,10 +526,10 @@ static int read_naa_id(const uint8_t *p, uint64_t *p_wwn) + + int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, + uint8_t cmd_size, uint8_t *buf, uint8_t buf_size, +- uint32_t timeout) ++ uint32_t timeout, Error **errp) + { + sg_io_hdr_t io_header; +- uint8_t sensebuf[8]; ++ uint8_t sensebuf[8] = {}; + int ret; + + memset(&io_header, 0, sizeof(io_header)); +@@ -549,6 +549,29 @@ int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, + io_header.driver_status || io_header.host_status) { + trace_scsi_generic_ioctl_sgio_done(cmd[0], ret, io_header.status, + io_header.host_status); ++ if (ret < 0) { ++ error_setg_errno(errp, -ret, "SG_IO ioctl failed"); ++ } else { ++ g_autofree char *sensebuf_hex = ++ g_strdup_printf("%02x%02x%02x%02x%02x%02x%02x%02x", ++ sensebuf[0], ++ sensebuf[1], ++ sensebuf[2], ++ sensebuf[3], ++ sensebuf[4], ++ sensebuf[5], ++ sensebuf[6], ++ sensebuf[7]); ++ ++ error_setg(errp, "SG_IO SCSI command failed with status=0x%x " ++ "driver_status=0x%x host_status=0x%x sensebuf=%s " ++ "sb_len_wr=%u", ++ io_header.status, ++ io_header.driver_status, ++ io_header.host_status, ++ sensebuf_hex, ++ io_header.sb_len_wr); ++ } + return -1; + } + return 0; +@@ -575,7 +598,7 @@ static void scsi_generic_set_vpd_bl_emulation(SCSIDevice *s) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ buf, sizeof(buf), s->io_timeout, NULL); + if (ret < 0) { + /* + * Do not assume anything if we can't retrieve the +@@ -611,7 +634,7 @@ static void scsi_generic_read_device_identification(SCSIDevice *s) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ buf, sizeof(buf), s->io_timeout, NULL); + if (ret < 0) { + return; + } +@@ -663,7 +686,7 @@ static int get_stream_blocksize(BlockBackend *blk) + cmd[4] = sizeof(buf); + + ret = scsi_SG_IO(blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), +- buf, sizeof(buf), 6); ++ buf, sizeof(buf), 6, NULL); + if (ret < 0) { + return -1; + } +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index a79c3aadfc..977b7e4d55 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -237,7 +237,7 @@ void scsi_device_unit_attention_reported(SCSIDevice *dev); + void scsi_generic_read_device_inquiry(SCSIDevice *dev); + int scsi_device_get_sense(SCSIDevice *dev, uint8_t *buf, int len, bool fixed); + int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, uint8_t cmd_size, +- uint8_t *buf, uint8_t buf_size, uint32_t timeout); ++ uint8_t *buf, uint8_t buf_size, uint32_t timeout, Error **errp); + SCSIDevice *scsi_device_find(SCSIBus *bus, int channel, int target, int lun); + SCSIDevice *scsi_device_get(SCSIBus *bus, int channel, int target, int lun); + +-- +2.47.3 + diff --git a/kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch b/kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch new file mode 100644 index 0000000..3224362 --- /dev/null +++ b/kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch @@ -0,0 +1,117 @@ +From af75da80d87f9b2490368edb3818bf0692c66882 Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:31 -0500 +Subject: [PATCH 2/7] scsi: generalize scsi_SG_IO_FROM_DEV() to scsi_SG_IO() + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [1/5] 45295e0aa8da535c0fca388e6b8e3a0665a4cf7c (stefanha/centos-stream-qemu-kvm) + +Add a direction argument so that scsi_SG_IO() can be used for +SG_DXFER_FROM_DEV and SG_DXFER_TO_DEV transfers. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-2-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit 03396b9afcf93964bb4dbb9d0cd7387ba0f63aa3) +Signed-off-by: Stefan Hajnoczi +--- + hw/scsi/scsi-disk.c | 4 ++-- + hw/scsi/scsi-generic.c | 18 ++++++++++-------- + include/hw/scsi/scsi.h | 4 ++-- + 3 files changed, 14 insertions(+), 12 deletions(-) + +diff --git a/hw/scsi/scsi-disk.c b/hw/scsi/scsi-disk.c +index b4782c6248..ca0214c404 100644 +--- a/hw/scsi/scsi-disk.c ++++ b/hw/scsi/scsi-disk.c +@@ -2748,8 +2748,8 @@ static int get_device_type(SCSIDiskState *s) + cmd[0] = INQUIRY; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(s->qdev.conf.blk, cmd, sizeof(cmd), +- buf, sizeof(buf), s->qdev.io_timeout); ++ ret = scsi_SG_IO(s->qdev.conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->qdev.io_timeout); + if (ret < 0) { + return -1; + } +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index 9e380a2109..b27ad48f18 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -524,8 +524,9 @@ static int read_naa_id(const uint8_t *p, uint64_t *p_wwn) + return -EINVAL; + } + +-int scsi_SG_IO_FROM_DEV(BlockBackend *blk, uint8_t *cmd, uint8_t cmd_size, +- uint8_t *buf, uint8_t buf_size, uint32_t timeout) ++int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, ++ uint8_t cmd_size, uint8_t *buf, uint8_t buf_size, ++ uint32_t timeout) + { + sg_io_hdr_t io_header; + uint8_t sensebuf[8]; +@@ -533,7 +534,7 @@ int scsi_SG_IO_FROM_DEV(BlockBackend *blk, uint8_t *cmd, uint8_t cmd_size, + + memset(&io_header, 0, sizeof(io_header)); + io_header.interface_id = 'S'; +- io_header.dxfer_direction = SG_DXFER_FROM_DEV; ++ io_header.dxfer_direction = direction; + io_header.dxfer_len = buf_size; + io_header.dxferp = buf; + io_header.cmdp = cmd; +@@ -573,8 +574,8 @@ static void scsi_generic_set_vpd_bl_emulation(SCSIDevice *s) + cmd[2] = 0x00; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(s->conf.blk, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout); + if (ret < 0) { + /* + * Do not assume anything if we can't retrieve the +@@ -609,8 +610,8 @@ static void scsi_generic_read_device_identification(SCSIDevice *s) + cmd[2] = 0x83; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(s->conf.blk, cmd, sizeof(cmd), +- buf, sizeof(buf), s->io_timeout); ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout); + if (ret < 0) { + return; + } +@@ -661,7 +662,8 @@ static int get_stream_blocksize(BlockBackend *blk) + cmd[0] = MODE_SENSE; + cmd[4] = sizeof(buf); + +- ret = scsi_SG_IO_FROM_DEV(blk, cmd, sizeof(cmd), buf, sizeof(buf), 6); ++ ret = scsi_SG_IO(blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), 6); + if (ret < 0) { + return -1; + } +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index 90ee192b4d..a79c3aadfc 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -236,8 +236,8 @@ void scsi_device_report_change(SCSIDevice *dev, SCSISense sense); + void scsi_device_unit_attention_reported(SCSIDevice *dev); + void scsi_generic_read_device_inquiry(SCSIDevice *dev); + int scsi_device_get_sense(SCSIDevice *dev, uint8_t *buf, int len, bool fixed); +-int scsi_SG_IO_FROM_DEV(BlockBackend *blk, uint8_t *cmd, uint8_t cmd_size, +- uint8_t *buf, uint8_t buf_size, uint32_t timeout); ++int scsi_SG_IO(BlockBackend *blk, int direction, uint8_t *cmd, uint8_t cmd_size, ++ uint8_t *buf, uint8_t buf_size, uint32_t timeout); + SCSIDevice *scsi_device_find(SCSIBus *bus, int channel, int target, int lun); + SCSIDevice *scsi_device_get(SCSIBus *bus, int channel, int target, int lun); + +-- +2.47.3 + diff --git a/kvm-scsi-save-load-SCSI-reservation-state.patch b/kvm-scsi-save-load-SCSI-reservation-state.patch new file mode 100644 index 0000000..7660bff --- /dev/null +++ b/kvm-scsi-save-load-SCSI-reservation-state.patch @@ -0,0 +1,348 @@ +From 91e9a0fbb3d6ee8726ea0ef17ff17404f823577a Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:34 -0500 +Subject: [PATCH 5/7] scsi: save/load SCSI reservation state + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [4/5] 2ebcfedf1a9e54433ed1ba3593d4fbcbadcd644b (stefanha/centos-stream-qemu-kvm) + +Add a vmstate subsection to SCSIDiskState so that scsi-block devices can +transfer their reservation state during live migration. Upon loading the +subsection, the destination QEMU invokes the PERSISTENT RESERVE OUT +command's PREEMPT service action to atomically move the reservation from +the source I_T nexus to the destination I_T nexus. This results in +transparent live migration of SCSI reservations. + +This approach is incomplete since SCSI reservations are cooperative and +other hosts could interfere. Neither the source QEMU nor the destination +QEMU are aware of changes made by other hosts. The assumption is that +reservation is not taken over by a third host without cooperation from +the source host. + +I considered adding the vmstate subsection to SCSIDevice instead of +SCSIDiskState, since reservations are part of the SCSI Primary Commands +that other devices apart from disks could support. However, due to +fragility of migrating reservations, we will probably limit support to +scsi-block and maybe scsi-disk in the future. In the end, I think it +makes sense to place this within scsi-disk.c. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-5-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit ab57b51f1375b6a6f098a74c6f79207a9630948d) +Signed-off-by: Stefan Hajnoczi + +Conflicts: +- hw/core/machine.c + + Downstream does not have hw_compat_10_2. Make sure that machine types + prior to RHEL 10.2 default to migrate-pr=off. + +- hw/scsi/scsi-disk.c + + Downstream is missing commit 40de712a89d8f ("migration: Add + error-parameterized function variants in VMSD struct"), so add a + .post_load() wrapper function that reports the Error and returns + -EINVAL. +--- + hw/core/machine.c | 1 + + hw/scsi/scsi-disk.c | 99 +++++++++++++++++++++++++++++++++++++++++- + hw/scsi/scsi-generic.c | 83 +++++++++++++++++++++++++++++++++++ + hw/scsi/trace-events | 1 + + include/hw/scsi/scsi.h | 1 + + 5 files changed, 184 insertions(+), 1 deletion(-) + +diff --git a/hw/core/machine.c b/hw/core/machine.c +index 2a1a42cebc..2b339f6a13 100644 +--- a/hw/core/machine.c ++++ b/hw/core/machine.c +@@ -295,6 +295,7 @@ const char *rhel_old_machine_deprecation = + "machine types for previous major releases are deprecated"; + + GlobalProperty hw_compat_rhel_10_2[] = { ++ { "scsi-block", "migrate-pr", "off" }, + /* hw_compat_rhel_10_2 from hw_compat_10_0 */ + { "scsi-hd", "dpofua", "off" }, + /* hw_compat_rhel_10_2 from hw_compat_10_0 */ +diff --git a/hw/scsi/scsi-disk.c b/hw/scsi/scsi-disk.c +index c736988ef6..e7b821ad19 100644 +--- a/hw/scsi/scsi-disk.c ++++ b/hw/scsi/scsi-disk.c +@@ -28,6 +28,7 @@ + #include "qemu/hw-version.h" + #include "qemu/memalign.h" + #include "hw/scsi/scsi.h" ++#include "migration/misc.h" + #include "migration/qemu-file-types.h" + #include "migration/vmstate.h" + #include "hw/scsi/emulation.h" +@@ -122,6 +123,7 @@ struct SCSIDiskState { + */ + uint16_t rotation_rate; + bool migrate_emulated_scsi_request; ++ NotifierWithReturn migration_notifier; + }; + + static void scsi_free_request(SCSIRequest *req) +@@ -2737,6 +2739,29 @@ static SCSIRequest *scsi_new_request(SCSIDevice *d, uint32_t tag, uint32_t lun, + } + + #ifdef __linux__ ++/* ++ * Preempt on the SCSI Persistent Reservation on the source when migration ++ * fails because the destination may have already preempted and we need to get ++ * the reservation back. ++ */ ++static int scsi_block_migration_notifier(NotifierWithReturn *notifier, ++ MigrationEvent *e, Error **errp) ++{ ++ if (e->type == MIG_EVENT_PRECOPY_FAILED) { ++ SCSIDiskState *s = ++ container_of(notifier, SCSIDiskState, migration_notifier); ++ SCSIDevice *d = &s->qdev; ++ Error *local_err = NULL; ++ ++ if (!scsi_generic_pr_state_preempt(d, &local_err)) { ++ /* MIG_EVENT_PRECOPY_FAILED cannot fail, so just warn */ ++ error_prepend(&local_err, "scsi-block migration rollback: "); ++ warn_report_err(local_err); ++ } ++ } ++ return 0; ++} ++ + static int get_device_type(SCSIDiskState *s) + { + uint8_t cmd[16]; +@@ -2815,6 +2840,16 @@ static void scsi_block_realize(SCSIDevice *dev, Error **errp) + + scsi_realize(&s->qdev, errp); + scsi_generic_read_device_inquiry(&s->qdev); ++ ++ migration_add_notifier(&s->migration_notifier, ++ scsi_block_migration_notifier); ++} ++ ++static void scsi_block_unrealize(SCSIDevice *dev) ++{ ++ SCSIDiskState *s = DO_UPCAST(SCSIDiskState, qdev, dev); ++ ++ migration_remove_notifier(&s->migration_notifier); + } + + typedef struct SCSIBlockReq { +@@ -3209,6 +3244,60 @@ static const Property scsi_hd_properties[] = { + DEFINE_BLOCK_CHS_PROPERTIES(SCSIDiskState, qdev.conf), + }; + ++#ifdef __linux__ ++static bool scsi_disk_pr_state_post_load_errp(void *opaque, int version_id, ++ Error **errp) ++{ ++ SCSIDiskState *s = opaque; ++ SCSIDevice *dev = &s->qdev; ++ ++ return scsi_generic_pr_state_preempt(dev, errp); ++} ++ ++static int scsi_disk_pr_state_post_load(void *opaque, int version_id) ++{ ++ SCSIDiskState *s = opaque; ++ Error *errp = NULL; ++ ++ if (scsi_disk_pr_state_post_load_errp(s, version_id, &errp)) { ++ return 0; ++ } else { ++ error_report_err(errp); ++ return -EINVAL; ++ } ++} ++ ++static bool scsi_disk_pr_state_needed(void *opaque) ++{ ++ SCSIDiskState *s = opaque; ++ SCSIPRState *pr_state = &s->qdev.pr_state; ++ bool ret; ++ ++ if (!s->qdev.migrate_pr) { ++ return false; ++ } ++ ++ /* A reservation requires a key, so checking this field is enough */ ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ ret = pr_state->key; ++ } ++ return ret; ++} ++ ++static const VMStateDescription vmstate_scsi_disk_pr_state = { ++ .name = "scsi-disk/pr", ++ .version_id = 1, ++ .minimum_version_id = 1, ++ .post_load = scsi_disk_pr_state_post_load, ++ .needed = scsi_disk_pr_state_needed, ++ .fields = (const VMStateField[]) { ++ VMSTATE_UINT64(qdev.pr_state.key, SCSIDiskState), ++ VMSTATE_UINT8(qdev.pr_state.resv_type, SCSIDiskState), ++ VMSTATE_END_OF_LIST() ++ } ++}; ++#endif /* __linux__ */ ++ + static const VMStateDescription vmstate_scsi_disk_state = { + .name = "scsi-disk", + .version_id = 1, +@@ -3221,7 +3310,13 @@ static const VMStateDescription vmstate_scsi_disk_state = { + VMSTATE_BOOL(tray_open, SCSIDiskState), + VMSTATE_BOOL(tray_locked, SCSIDiskState), + VMSTATE_END_OF_LIST() +- } ++ }, ++ .subsections = (const VMStateDescription * const []) { ++#ifdef __linux__ ++ &vmstate_scsi_disk_pr_state, ++#endif ++ NULL ++ }, + }; + + static void scsi_hd_class_initfn(ObjectClass *klass, const void *data) +@@ -3301,6 +3396,7 @@ static const Property scsi_block_properties[] = { + -1), + DEFINE_PROP_UINT32("io_timeout", SCSIDiskState, qdev.io_timeout, + DEFAULT_IO_TIMEOUT), ++ DEFINE_PROP_BOOL("migrate-pr", SCSIDiskState, qdev.migrate_pr, true), + }; + + static void scsi_block_class_initfn(ObjectClass *klass, const void *data) +@@ -3310,6 +3406,7 @@ static void scsi_block_class_initfn(ObjectClass *klass, const void *data) + SCSIDiskClass *sdc = SCSI_DISK_BASE_CLASS(klass); + + sc->realize = scsi_block_realize; ++ sc->unrealize = scsi_block_unrealize; + sc->alloc_req = scsi_block_new_request; + sc->parse_cdb = scsi_block_parse_cdb; + sdc->dma_readv = scsi_block_dma_readv; +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index a6280eaa87..b8b3f399f0 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -423,6 +423,89 @@ static void scsi_handle_persistent_reserve_out_reply( + } + } + ++static bool scsi_generic_pr_register(SCSIDevice *s, uint64_t key, Error **errp) ++{ ++ uint8_t cmd[10] = {}; ++ uint8_t buf[24] = {}; ++ uint64_t key_be = cpu_to_be64(key); ++ int ret; ++ ++ cmd[0] = PERSISTENT_RESERVE_OUT; ++ cmd[1] = PRO_REGISTER; ++ cmd[8] = sizeof(buf); ++ memcpy(&buf[8], &key_be, sizeof(key_be)); ++ ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_TO_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout, errp); ++ if (ret < 0) { ++ error_prepend(errp, "PERSISTENT RESERVE OUT with REGISTER"); ++ return false; ++ } ++ return true; ++} ++ ++static bool scsi_generic_pr_preempt(SCSIDevice *s, uint64_t key, ++ uint8_t resv_type, Error **errp) ++{ ++ uint8_t cmd[10] = {}; ++ uint8_t buf[24] = {}; ++ uint64_t key_be = cpu_to_be64(key); ++ int ret; ++ ++ cmd[0] = PERSISTENT_RESERVE_OUT; ++ cmd[1] = PRO_PREEMPT; ++ cmd[2] = resv_type & 0xf; ++ cmd[8] = sizeof(buf); ++ memcpy(&buf[0], &key_be, sizeof(key_be)); ++ memcpy(&buf[8], &key_be, sizeof(key_be)); ++ ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_TO_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout, errp); ++ if (ret < 0) { ++ error_prepend(errp, "PERSISTENT RESERVE OUT with PREEMPT"); ++ return false; ++ } ++ return true; ++} ++ ++/* Register keys and preempt reservations after live migration */ ++bool scsi_generic_pr_state_preempt(SCSIDevice *s, Error **errp) ++{ ++ SCSIPRState *pr_state = &s->pr_state; ++ uint64_t key; ++ uint8_t resv_type; ++ ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ key = pr_state->key; ++ resv_type = pr_state->resv_type; ++ } ++ ++ trace_scsi_generic_pr_state_preempt(key, resv_type); ++ ++ if (key) { ++ if (!scsi_generic_pr_register(s, key, errp)) { ++ return false; ++ } ++ ++ /* ++ * Two cases: ++ * ++ * 1. There is no reservation (resv_type is 0) and the other I_T nexus ++ * will be unregistered. This is important so the source host does ++ * not leak registered keys across live migration. ++ * ++ * 2. There is a reservation (resv_type is not 0) and the other I_T ++ * nexus will be unregistered and its reservation is atomically ++ * taken over by us. This is the scenario where a reservation is ++ * migrated along with the guest. ++ */ ++ if (!scsi_generic_pr_preempt(s, key, resv_type, errp)) { ++ return false; ++ } ++ } ++ return true; ++} ++ + static void scsi_read_complete(void * opaque, int ret) + { + SCSIGenericReq *r = (SCSIGenericReq *)opaque; +diff --git a/hw/scsi/trace-events b/hw/scsi/trace-events +index fdb87a237f..ab7a6f4cea 100644 +--- a/hw/scsi/trace-events ++++ b/hw/scsi/trace-events +@@ -362,3 +362,4 @@ scsi_generic_aio_sgio_command(uint32_t tag, uint8_t cmd, uint32_t timeout) "gene + scsi_generic_ioctl_sgio_command(uint8_t cmd, uint32_t timeout) "generic ioctl sgio: cmd=0x%x timeout=%u" + scsi_generic_ioctl_sgio_done(uint8_t cmd, int ret, uint8_t status, uint8_t host_status) "generic ioctl sgio: cmd=0x%x ret=%d status=0x%x host_status=0x%x" + scsi_generic_persistent_reserve_out_reply(uint8_t service_action, uint8_t resv_type, uint64_t old_key, uint64_t new_key) "persistent reserve out reply service_action=%u resv_type=%u old_key=0x%" PRIx64 " new_key=0x%" PRIx64 ++scsi_generic_pr_state_preempt(uint64_t key, uint8_t resv_type) "key=0x%" PRIx64 " resv_type=%u" +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index 7e120fdd6a..f61c63c5ea 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -253,6 +253,7 @@ SCSIDevice *scsi_device_get(SCSIBus *bus, int channel, int target, int lun); + + /* scsi-generic.c. */ + extern const SCSIReqOps scsi_generic_req_ops; ++bool scsi_generic_pr_state_preempt(SCSIDevice *s, Error **errp); + + /* scsi-disk.c */ + #define SCSI_DISK_QUIRK_MODE_PAGE_APPLE_VENDOR 0 +-- +2.47.3 + diff --git a/kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch b/kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch new file mode 100644 index 0000000..82a3403 --- /dev/null +++ b/kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch @@ -0,0 +1,331 @@ +From 220f405f15c997b636048aae3d4df44c4518e71d Mon Sep 17 00:00:00 2001 +From: Stefan Hajnoczi +Date: Thu, 29 Jan 2026 16:20:33 -0500 +Subject: [PATCH 4/7] scsi: track SCSI reservation state for live migration + +RH-Author: Stefan Hajnoczi +RH-MergeRequest: 464: scsi: persistent reservation live migration +RH-Jira: RHEL-132749 +RH-Acked-by: Miroslav Rezanina +RH-Acked-by: Kevin Wolf +RH-Commit: [3/5] 563e4c4917709c65ca503898488131e3babf16e7 (stefanha/centos-stream-qemu-kvm) + +SCSI Persistent Reservations are stateful and external to the guest. In +order to transparently move reservations to the destination host during +live migration, it is necessary to track the state built up on the +source host before migration. Only then can the destination host ensure +an equivalent state is restored upon migration. + +Snoop on successful PERSISTENT RESERVE OUT commands and save the +reservation key and reservation type. This will allow registered keys +and reservations to be migrated. + +Also patch PERSISTENT RESERVE IN replies with the REPORT CAPABILITIES +service action since features that involve the physical SCSI bus target +ports must not be exposed to the guest (it sees a virtual SCSI bus). + +Usually this plays out as follows: +1. The guest invokes the REGISTER service action to register a + reservation key on its I_T nexus. +2. The guest invokes the RESERVE service action to create a reservation + using the previously-registered key. + +This commit implements the snooping and stores the reservation key and +type (if any) for each LUN. The snooped PR state and the migrate_pr flag +to enable PR migration will be used in later commits. + +Signed-off-by: Stefan Hajnoczi +Reviewed-by: Paolo Bonzini +Message-id: 20260129212035.219676-4-stefanha@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit 70f0e0cedb2e0d7511cdebbce9b21a01bba55b74) +Signed-off-by: Stefan Hajnoczi +--- + hw/scsi/scsi-bus.c | 3 + + hw/scsi/scsi-generic.c | 165 +++++++++++++++++++++++++++++++++++++++ + hw/scsi/trace-events | 1 + + include/hw/scsi/scsi.h | 10 +++ + include/scsi/constants.h | 21 +++++ + 5 files changed, 200 insertions(+) + +diff --git a/hw/scsi/scsi-bus.c b/hw/scsi/scsi-bus.c +index 9b12ee7f1c..878ccf62c9 100644 +--- a/hw/scsi/scsi-bus.c ++++ b/hw/scsi/scsi-bus.c +@@ -393,6 +393,7 @@ static void scsi_qdev_realize(DeviceState *qdev, Error **errp) + } + + qemu_mutex_init(&dev->requests_lock); ++ qemu_mutex_init(&dev->pr_state.mutex); + QTAILQ_INIT(&dev->requests); + scsi_device_realize(dev, &local_err); + if (local_err) { +@@ -417,6 +418,8 @@ static void scsi_qdev_unrealize(DeviceState *qdev) + + scsi_device_unrealize(dev); + ++ qemu_mutex_destroy(&dev->pr_state.mutex); ++ + blockdev_mark_auto_del(dev->conf.blk); + } + +diff --git a/hw/scsi/scsi-generic.c b/hw/scsi/scsi-generic.c +index 4f851186f6..a6280eaa87 100644 +--- a/hw/scsi/scsi-generic.c ++++ b/hw/scsi/scsi-generic.c +@@ -264,6 +264,165 @@ static int scsi_generic_emulate_block_limits(SCSIGenericReq *r, SCSIDevice *s) + return r->buflen; + } + ++/* ++ * Patch persistent reservation capabilities that are not emulated. ++ */ ++static void scsi_handle_persistent_reserve_in_reply(SCSIGenericReq *r, ++ SCSIDevice *s) ++{ ++ uint8_t service_action = r->req.cmd.buf[1] & 0x1f; ++ ++ if (!s->migrate_pr) { ++ return; /* when migration is disabled there is no need for patching */ ++ } ++ ++ if (service_action == PRI_REPORT_CAPABILITIES) { ++ assert(r->buflen >= 3); ++ ++ /* ++ * Clear specify initiator ports capable (SIP_C) and all target ports ++ * capable (ATC_C). ++ * ++ * SPEC_I_PT is not supported because the guest sees an emulated SCSI ++ * bus and does not have the underlying transport IDs needed to use ++ * SPEC_I_PT. ++ * ++ * ALL_TG_PT is not supported because we only track the state of this ++ * emulated I_T nexus, not the underlying device's target ports. ++ */ ++ r->buf[2] &= ~0xc; ++ } ++} ++ ++static int scsi_generic_read_reservation(SCSIDevice *s, uint64_t *key, ++ uint8_t *resv_type, Error **errp) ++{ ++ uint8_t cmd[10] = {}; ++ uint8_t buf[24] = {}; ++ uint32_t additional_length; ++ int ret; ++ ++ *key = 0; ++ *resv_type = 0; ++ ++ cmd[0] = PERSISTENT_RESERVE_IN; ++ cmd[1] = PRI_READ_RESERVATION; ++ cmd[8] = sizeof(buf); ++ ++ ret = scsi_SG_IO(s->conf.blk, SG_DXFER_FROM_DEV, cmd, sizeof(cmd), ++ buf, sizeof(buf), s->io_timeout, errp); ++ if (ret < 0) { ++ return ret; ++ } ++ ++ memcpy(&additional_length, &buf[4], sizeof(additional_length)); ++ be32_to_cpus(&additional_length); ++ ++ if (additional_length >= 0x10) { ++ memcpy(key, &buf[8], sizeof(*key)); ++ be64_to_cpus(key); ++ ++ *resv_type = buf[21] & 0xf; ++ } ++ return 0; ++} ++ ++/* ++ * Snoop changes to registered keys and reservations so that this information ++ * can be transferred during live migration. ++ */ ++static void scsi_handle_persistent_reserve_out_reply( ++ SCSIGenericReq *r, ++ SCSIDevice *s) ++{ ++ SCSIPRState *pr_state = &s->pr_state; ++ uint8_t service_action = r->req.cmd.buf[1] & 0x1f; ++ uint8_t resv_type = r->req.cmd.buf[2] & 0xf; ++ uint64_t old_key; ++ uint64_t new_key; ++ ++ assert(r->buflen >= 16); ++ memcpy(&old_key, &r->buf[0], sizeof(old_key)); ++ memcpy(&new_key, &r->buf[8], sizeof(new_key)); ++ be64_to_cpus(&old_key); ++ be64_to_cpus(&new_key); ++ ++ trace_scsi_generic_persistent_reserve_out_reply(service_action, resv_type, ++ old_key, new_key); ++ ++ switch (service_action) { ++ case PRO_REGISTER: /* fallthrough */ ++ case PRO_REGISTER_AND_IGNORE_EXISTING_KEY: ++ if (service_action == PRO_REGISTER && old_key == 0 && new_key == 0) { ++ /* Do nothing */ ++ } else { ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->key = new_key; ++ if (new_key == 0) { ++ pr_state->resv_type = 0; /* release reservation */ ++ } ++ } ++ } ++ break; ++ ++ case PRO_RESERVE: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->resv_type = resv_type; ++ } ++ break; ++ ++ case PRO_RELEASE: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->resv_type = 0; ++ } ++ break; ++ ++ case PRO_CLEAR: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->key = 0; ++ pr_state->resv_type = 0; ++ } ++ break; ++ ++ case PRO_REPLACE_LOST_RESERVATION: ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ pr_state->key = new_key; ++ pr_state->resv_type = resv_type; ++ } ++ break; ++ ++ case PRO_PREEMPT: /* fallthrough */ ++ case PRO_PREEMPT_AND_ABORT: { ++ uint64_t dev_key; ++ uint8_t dev_resv_type; ++ Error *local_err = NULL; ++ ++ /* Not enough information to know actual state, ask the device */ ++ if (!scsi_generic_read_reservation(s, &dev_key, &dev_resv_type, ++ &local_err)) { ++ WITH_QEMU_LOCK_GUARD(&pr_state->mutex) { ++ if (pr_state->key == dev_key) { ++ pr_state->resv_type = dev_resv_type; ++ } else { ++ pr_state->resv_type = 0; ++ } ++ } ++ } ++ if (local_err) { ++ warn_report_err(local_err); ++ } ++ break; ++ } ++ ++ /* ++ * PRO_REGISTER_AND_MOVE cannot be implemented since it involves the ++ * physical SCSI bus target ports. ++ */ ++ default: ++ break; /* do nothing */ ++ } ++} ++ + static void scsi_read_complete(void * opaque, int ret) + { + SCSIGenericReq *r = (SCSIGenericReq *)opaque; +@@ -346,6 +505,9 @@ static void scsi_read_complete(void * opaque, int ret) + if (r->req.cmd.buf[0] == INQUIRY) { + len = scsi_handle_inquiry_reply(r, s, len); + } ++ if (r->req.cmd.buf[0] == PERSISTENT_RESERVE_IN) { ++ scsi_handle_persistent_reserve_in_reply(r, s); ++ } + + req_complete: + scsi_req_data(&r->req, len); +@@ -395,6 +557,9 @@ static void scsi_write_complete(void * opaque, int ret) + s->blocksize = (r->buf[9] << 16) | (r->buf[10] << 8) | r->buf[11]; + trace_scsi_generic_write_complete_blocksize(s->blocksize); + } ++ if (r->req.cmd.buf[0] == PERSISTENT_RESERVE_OUT) { ++ scsi_handle_persistent_reserve_out_reply(r, s); ++ } + + scsi_command_complete_noio(r, ret); + } +diff --git a/hw/scsi/trace-events b/hw/scsi/trace-events +index 6c2788e202..fdb87a237f 100644 +--- a/hw/scsi/trace-events ++++ b/hw/scsi/trace-events +@@ -361,3 +361,4 @@ scsi_generic_realize_blocksize(int blocksize) "block size %d" + scsi_generic_aio_sgio_command(uint32_t tag, uint8_t cmd, uint32_t timeout) "generic aio sgio: tag=0x%x cmd=0x%x timeout=%u" + scsi_generic_ioctl_sgio_command(uint8_t cmd, uint32_t timeout) "generic ioctl sgio: cmd=0x%x timeout=%u" + scsi_generic_ioctl_sgio_done(uint8_t cmd, int ret, uint8_t status, uint8_t host_status) "generic ioctl sgio: cmd=0x%x ret=%d status=0x%x host_status=0x%x" ++scsi_generic_persistent_reserve_out_reply(uint8_t service_action, uint8_t resv_type, uint64_t old_key, uint64_t new_key) "persistent reserve out reply service_action=%u resv_type=%u old_key=0x%" PRIx64 " new_key=0x%" PRIx64 +diff --git a/include/hw/scsi/scsi.h b/include/hw/scsi/scsi.h +index 977b7e4d55..7e120fdd6a 100644 +--- a/include/hw/scsi/scsi.h ++++ b/include/hw/scsi/scsi.h +@@ -54,6 +54,13 @@ struct SCSIRequest { + QTAILQ_ENTRY(SCSIRequest) next; + }; + ++/* Per-SCSIDevice Persistent Reservation state */ ++typedef struct { ++ QemuMutex mutex; /* protects all fields (e.g. from multiple IOThreads) */ ++ uint64_t key; /* 0 if no registered key */ ++ uint8_t resv_type; /* 0 if no reservation */ ++} SCSIPRState; ++ + #define TYPE_SCSI_DEVICE "scsi-device" + OBJECT_DECLARE_TYPE(SCSIDevice, SCSIDeviceClass, SCSI_DEVICE) + +@@ -94,6 +101,9 @@ struct SCSIDevice + uint32_t io_timeout; + bool needs_vpd_bl_emulation; + bool hba_supports_iothread; ++ ++ bool migrate_pr; ++ SCSIPRState pr_state; + }; + + extern const VMStateDescription vmstate_scsi_device; +diff --git a/include/scsi/constants.h b/include/scsi/constants.h +index 9b98451912..cb97bdb636 100644 +--- a/include/scsi/constants.h ++++ b/include/scsi/constants.h +@@ -319,4 +319,25 @@ + #define IDENT_DESCR_TGT_DESCR_SIZE 32 + #define XCOPY_BLK2BLK_SEG_DESC_SIZE 28 + ++/* ++ * PERSISTENT RESERVATION IN service action codes ++ */ ++#define PRI_READ_KEYS 0x00 ++#define PRI_READ_RESERVATION 0x01 ++#define PRI_REPORT_CAPABILITIES 0x02 ++#define PRI_READ_FULL_STATUS 0x03 ++ ++/* ++ * PERSISTENT RESERVATION OUT service action codes ++ */ ++#define PRO_REGISTER 0x00 ++#define PRO_RESERVE 0x01 ++#define PRO_RELEASE 0x02 ++#define PRO_CLEAR 0x03 ++#define PRO_PREEMPT 0x04 ++#define PRO_PREEMPT_AND_ABORT 0x05 ++#define PRO_REGISTER_AND_IGNORE_EXISTING_KEY 0x06 ++#define PRO_REGISTER_AND_MOVE 0x07 ++#define PRO_REPLACE_LOST_RESERVATION 0x08 ++ + #endif +-- +2.47.3 + diff --git a/kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch b/kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch new file mode 100644 index 0000000..a6a8242 --- /dev/null +++ b/kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch @@ -0,0 +1,123 @@ +From 2643a61dd6de41945d714aac210173754a7b5a7f Mon Sep 17 00:00:00 2001 +From: German Maglione +Date: Wed, 22 Oct 2025 18:24:05 +0200 +Subject: [PATCH 1/7] vhost-user: make vhost_set_vring_file() synchronous +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Hanna Czenczek +RH-MergeRequest: 462: vhost-user: make vhost_set_vring_file() synchronous +RH-Jira: RHEL-147425 +RH-Acked-by: Stefano Garzarella +RH-Acked-by: Eugenio Pérez +RH-Commit: [1/1] 92476878c6fa5cb8ac1b308d2c8d4275767fd780 (hreitz/qemu-kvm-c-9-s) + +QEMU sends all of VHOST_USER_SET_VRING_KICK, _CALL, and _ERR without +setting the NEED_REPLY flag, i.e. by the time the respective +vhost_user_set_vring_*() function returns, it is completely up to chance +whether the back-end has already processed the request and switched over +to the new FD for interrupts. + +At least for vhost_user_set_vring_call(), that is a problem: It is +called through vhost_virtqueue_mask(), which is generally used in the +VirtioDeviceClass.guest_notifier_mask() implementation, which is in turn +called by virtio_pci_one_vector_unmask(). The fact that we do not wait +for the back-end to install the FD leads to a race there: + +Masking interrupts is implemented by redirecting interrupts to an +internal event FD that is not connected to the guest. Unmasking then +re-installs the guest-connected IRQ FD, then checks if there are pending +interrupts left on the masked event FD, and if so, issues an interrupt +to the guest. + +Because guest_notifier_mask() (through vhost_user_set_vring_call()) +doesn't wait for the back-end to switch over to the actual IRQ FD, it's +possible we check for pending interrupts while the back-end is still +using the masked event FD, and then we will lose interrupts that occur +before the back-end finally does switch over. + +Fix this by setting NEED_REPLY on those VHOST_USER_SET_VRING_* messages, +so when we get that reply, we know that the back-end is now using the +new FD. + +We have a few reports of a virtiofs mount hanging: +- https://gitlab.com/virtio-fs/virtiofsd/-/issues/101 +- https://gitlab.com/virtio-fs/virtiofsd/-/issues/133 +- https://gitlab.com/virtio-fs/virtiofsd/-/issues/213 + +This is quite difficult bug to reproduce, even for the reporters. +It only happens on production, every few weeks, and/or on 1 in 300 VMs. +So, we are not 100% sure this fixes that issue. However, we think this +is still a bug, and at least we have one report that claims this fixed +the issue: + +https://gitlab.com/virtio-fs/virtiofsd/-/issues/133#note_2743209419 + +Fixes: 5f6f6664bf24 ("Add vhost-user as a vhost backend.") +Signed-off-by: German Maglione +Signed-off-by: Hanna Czenczek +Reviewed-by: Eugenio Pérez +Reviewed-by: Stefano Garzarella +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +Message-Id: <20251022162405.318672-1-gmaglione@redhat.com> +(cherry picked from commit 1ba9a5220325dd5260a0c37b6299ce38364a5120) +Signed-off-by: Hanna Czenczek +--- + hw/virtio/vhost-user.c | 24 +++++++++++++++++++++++- + 1 file changed, 23 insertions(+), 1 deletion(-) + +diff --git a/hw/virtio/vhost-user.c b/hw/virtio/vhost-user.c +index 1e1d6b0d6e..3e1e1d8d7e 100644 +--- a/hw/virtio/vhost-user.c ++++ b/hw/virtio/vhost-user.c +@@ -1327,8 +1327,11 @@ static int vhost_set_vring_file(struct vhost_dev *dev, + VhostUserRequest request, + struct vhost_vring_file *file) + { ++ int ret; + int fds[VHOST_USER_MAX_RAM_SLOTS]; + size_t fd_num = 0; ++ bool reply_supported = virtio_has_feature(dev->protocol_features, ++ VHOST_USER_PROTOCOL_F_REPLY_ACK); + VhostUserMsg msg = { + .hdr.request = request, + .hdr.flags = VHOST_USER_VERSION, +@@ -1336,13 +1339,32 @@ static int vhost_set_vring_file(struct vhost_dev *dev, + .hdr.size = sizeof(msg.payload.u64), + }; + ++ if (reply_supported) { ++ msg.hdr.flags |= VHOST_USER_NEED_REPLY_MASK; ++ } ++ + if (file->fd > 0) { + fds[fd_num++] = file->fd; + } else { + msg.payload.u64 |= VHOST_USER_VRING_NOFD_MASK; + } + +- return vhost_user_write(dev, &msg, fds, fd_num); ++ ret = vhost_user_write(dev, &msg, fds, fd_num); ++ if (ret < 0) { ++ return ret; ++ } ++ ++ if (reply_supported) { ++ /* ++ * wait for the back-end's confirmation that the new FD is active, ++ * otherwise guest_notifier_mask() could check for pending interrupts ++ * while the back-end is still using the masked event FD, losing ++ * interrupts that occur before the back-end installs the FD ++ */ ++ return process_message_reply(dev, &msg); ++ } ++ ++ return 0; + } + + static int vhost_user_set_vring_kick(struct vhost_dev *dev, +-- +2.47.3 + diff --git a/qemu-kvm.spec b/qemu-kvm.spec index 154ae6d..7e1c473 100644 --- a/qemu-kvm.spec +++ b/qemu-kvm.spec @@ -143,7 +143,7 @@ Obsoletes: %{name}-block-ssh <= %{epoch}:%{version} \ Summary: QEMU is a machine emulator and virtualizer Name: qemu-kvm Version: 10.1.0 -Release: 12%{?rcrel}%{?dist}%{?cc_suffix} +Release: 13%{?rcrel}%{?dist}%{?cc_suffix} # Epoch because we pushed a qemu-1.0 package. AIUI this can't ever be dropped # Epoch 15 used for RHEL 8 # Epoch 17 used for RHEL 9 (due to release versioning offset in RHEL 8.5) @@ -393,6 +393,21 @@ Patch118: kvm-virtio-net-implement-extended-features-support.patch Patch119: kvm-net-implement-tunnel-probing.patch # For RHEL-143785 - backport support for GSO over UDP tunnel offload Patch120: kvm-net-implement-UDP-tunnel-features-offloading.patch +# For RHEL-147425 - virtiofs: processes become stuck in request_wait_answer on virtiofs mounts +Patch121: kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch122: kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch123: kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch124: kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch125: kvm-scsi-save-load-SCSI-reservation-state.patch +# For RHEL-132749 - Migrate SCSI PR state and preempt reservation upon live migration +Patch126: kvm-docs-add-SCSI-migrate-pr-documentation.patch +# For RHEL-134989 - Hotplugged interface device can not be shown in the guest +# For RHEL-146584 - [RHEL-10.2][ARM]: Unable to Check the mem prefetched size on Guest +Patch127: kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch %if %{have_clang} BuildRequires: clang @@ -1472,6 +1487,23 @@ useradd -r -u 107 -g qemu -G kvm -d / -s /sbin/nologin \ %endif %changelog +* Thu Feb 19 2026 Miroslav Rezanina - 10.1.0-13 +- kvm-vhost-user-make-vhost_set_vring_file-synchronous.patch [RHEL-147425] +- kvm-scsi-generalize-scsi_SG_IO_FROM_DEV-to-scsi_SG_IO.patch [RHEL-132749] +- kvm-scsi-add-error-reporting-to-scsi_SG_IO.patch [RHEL-132749] +- kvm-scsi-track-SCSI-reservation-state-for-live-migration.patch [RHEL-132749] +- kvm-scsi-save-load-SCSI-reservation-state.patch [RHEL-132749] +- kvm-docs-add-SCSI-migrate-pr-documentation.patch [RHEL-132749] +- kvm-Revert-hw-arm-virt-Use-ACPI-PCI-hotplug-by-default-f.patch [RHEL-134989 RHEL-146584] +- Resolves: RHEL-147425 + (virtiofs: processes become stuck in request_wait_answer on virtiofs mounts) +- Resolves: RHEL-132749 + (Migrate SCSI PR state and preempt reservation upon live migration) +- Resolves: RHEL-134989 + (Hotplugged interface device can not be shown in the guest) +- Resolves: RHEL-146584 + ([RHEL-10.2][ARM]: Unable to Check the mem prefetched size on Guest) + * Mon Feb 02 2026 Miroslav Rezanina - 10.1.0-12 - kvm-rbd-Run-co-BH-CB-in-the-coroutine-s-AioContext.patch [RHEL-79118] - kvm-curl-Fix-coroutine-waking.patch [RHEL-79118]