* [PATCH v11 01/16] s390x/pci: implement IOMMU replay
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 02/16] s390x/pci: Create function to contain translation status check Konstantin Shkolnyy
` (14 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
From: Matthew Rosato <mjrosato@linux.ibm.com>
There are a few scenarios where IOMMU replay can potentially be needed
for zPCI device, namely VFIO device reset scenarios where the guest
continues running and expects the contents of its IOMMU to be replayed
upon IOAT re-registration and migration scenarios where the destination
must reconstruct the IOMMU on the destination.
zPCI migration is not supported yet, but the IOMMU replay function is
implemented so that it can be called both from IOMMUMemoryRegionClass
now and migration post_load later.
Signed-off-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 71 +++++++++++++++++++++++++++++---
hw/s390x/s390-pci-inst.c | 4 +-
include/hw/s390x/s390-pci-inst.h | 1 +
3 files changed, 69 insertions(+), 7 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index 2eb4e8cec4..b3f7b59422 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -593,14 +593,75 @@ err:
return ret;
}
-static void s390_pci_iommu_replay(IOMMUMemoryRegion *iommu,
+static void s390_pci_ioat_replay(S390PCIIOMMU *iommu)
+{
+ S390PCIBusDevice *pbdev = iommu->pbdev;
+ S390IOTLBEntry entry;
+ uint16_t error = 0;
+ uint32_t dma_avail;
+ hwaddr curr, end;
+
+ curr = iommu->pba;
+ end = iommu->pal;
+
+ if (iommu->dm_mr || !pbdev) {
+ /* If direct mapping is used, there are no guest tables to replay */
+ return;
+ }
+
+ if (iommu->dma_limit) {
+ dma_avail = iommu->dma_limit->avail;
+ } else {
+ dma_avail = 1;
+ }
+
+ while (curr < end) {
+ error = s390_guest_io_table_walk(iommu->g_iota, curr, &entry);
+ if (error) {
+ pbdev->state = ZPCI_FS_ERROR;
+ s390_pci_generate_error_event(error, pbdev->fh, pbdev->fid, curr,
+ 0);
+ error_report("Failure to walk table during iommu remap");
+ return;
+ }
+
+ /* Advance to next frame boundary if start was not frame-aligned */
+ curr = QEMU_ALIGN_UP(curr + 1, entry.len);
+ if (entry.perm != IOMMU_NONE) {
+ while (entry.iova < curr && entry.iova < end) {
+ if (dma_avail > 0) {
+ dma_avail = s390_pci_update_iotlb(iommu, &entry);
+ } else {
+ /*
+ * There is no reliable method to request the guest to
+ * release mappings other than in response to a RPCIT
+ * instruction; generate a permanent error condition and
+ * require the device to be completely re-initialized from
+ * the guest side.
+ */
+ pbdev->state = ZPCI_FS_ERROR;
+ s390_pci_generate_error_event(ERR_EVENT_PERMERR, pbdev->fh,
+ pbdev->fid, 0, 0);
+ error_report("DMA mappings exhausted: iommu remap failed");
+ return;
+ }
+ entry.iova += TARGET_PAGE_SIZE;
+ entry.translated_addr += TARGET_PAGE_SIZE;
+ }
+ }
+ }
+}
+
+static void s390_pci_iommu_replay(IOMMUMemoryRegion *mr,
IOMMUNotifier *notifier)
{
- /* It's impossible to plug a pci device on s390x that already has iommu
- * mappings which need to be replayed, that is due to the "one iommu per
- * zpci device" construct. But when we support migration of vfio-pci
- * devices in future, we need to revisit this.
+ S390PCIIOMMU *iommu = container_of(mr, S390PCIIOMMU, iommu_mr);
+
+ /*
+ * The notifier argument is not used directly; s390_pci_ioat_replay()
+ * broadcasts via memory_region_notify_iommu() instead.
*/
+ s390_pci_ioat_replay(iommu);
}
static S390PCIIOMMU *s390_pci_get_iommu(S390pciState *s, PCIBus *bus,
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index 09026360d4..285ff0e089 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -630,8 +630,8 @@ int pcistg_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
return 0;
}
-static uint32_t s390_pci_update_iotlb(S390PCIIOMMU *iommu,
- S390IOTLBEntry *entry)
+uint32_t s390_pci_update_iotlb(S390PCIIOMMU *iommu,
+ S390IOTLBEntry *entry)
{
S390IOTLBEntry *cache = g_hash_table_lookup(iommu->iotlb, &entry->iova);
IOMMUTLBEvent event = {
diff --git a/include/hw/s390x/s390-pci-inst.h b/include/hw/s390x/s390-pci-inst.h
index 5cb8da540b..c782990e3b 100644
--- a/include/hw/s390x/s390-pci-inst.h
+++ b/include/hw/s390x/s390-pci-inst.h
@@ -111,6 +111,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
uintptr_t ra);
void fmb_timer_free(S390PCIBusDevice *pbdev);
+uint32_t s390_pci_update_iotlb(S390PCIIOMMU *iommu, S390IOTLBEntry *entry);
#define ZPCI_IO_BAR_MIN 0
#define ZPCI_IO_BAR_MAX 5
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 02/16] s390x/pci: Create function to contain translation status check
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 01/16] s390x/pci: implement IOMMU replay Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 03/16] s390x/pci: Move iommu_mr from S390PCIIOMMU to S390PCIBusDevice Konstantin Shkolnyy
` (13 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
Make it more clear what the bit means, and the new function will be called
from yet another place in the future.
Reviewed-by: Christian Borntraeger <borntraeger@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-inst.c | 7 ++++++-
include/hw/s390x/s390-pci-bus.h | 1 +
2 files changed, 7 insertions(+), 1 deletion(-)
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index 285ff0e089..c5d87aa615 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -1025,6 +1025,11 @@ int pci_dereg_irqs(S390PCIBusDevice *pbdev)
return 0;
}
+bool s390_pci_is_translation_enabled(uint64_t g_iota)
+{
+ return ((g_iota >> 11) & 0x1) != 0; /* "T" bit */
+}
+
static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
uintptr_t ra)
{
@@ -1033,7 +1038,7 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
uint64_t pal = ldq_be_p(&fib.pal);
uint64_t g_iota = ldq_be_p(&fib.iota);
uint8_t dt = (g_iota >> 2) & 0x7;
- uint8_t t = (g_iota >> 11) & 0x1;
+ bool t = s390_pci_is_translation_enabled(g_iota);
pba &= ~0xfff;
pal |= 0xfff;
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index f182bab246..a732ad652b 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -391,6 +391,7 @@ int pci_chsc_sei_nt2_get_event(void *res);
int pci_chsc_sei_nt2_have_event(void);
void s390_pci_sclp_configure(SCCB *sccb);
void s390_pci_sclp_deconfigure(SCCB *sccb);
+bool s390_pci_is_translation_enabled(uint64_t g_iota);
void s390_pci_iommu_enable(S390PCIIOMMU *iommu);
void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu);
void s390_pci_iommu_disable(S390PCIIOMMU *iommu);
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 03/16] s390x/pci: Move iommu_mr from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 01/16] s390x/pci: implement IOMMU replay Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 02/16] s390x/pci: Create function to contain translation status check Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 04/16] s390x/pci: Move dm_mr " Konstantin Shkolnyy
` (12 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the
IOMMU container information needed by the PCI layer for a given slot.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 35 +++++++++++++++++---------------
hw/s390x/s390-pci-inst.c | 28 +++++++++++++------------
include/hw/s390x/s390-pci-bus.h | 6 +++---
include/hw/s390x/s390-pci-inst.h | 4 ++--
4 files changed, 39 insertions(+), 34 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index b3f7b59422..e709d6e478 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -207,7 +207,7 @@ void s390_pci_sclp_deconfigure(SCCB *sccb)
pci_dereg_irqs(pbdev);
}
if (pbdev->iommu->enabled) {
- pci_dereg_ioat(pbdev->iommu);
+ pci_dereg_ioat(pbdev);
}
pbdev->state = ZPCI_FS_STANDBY;
rc = SCLP_RC_NORMAL_COMPLETION;
@@ -539,7 +539,8 @@ uint16_t s390_guest_io_table_walk(uint64_t g_iota, hwaddr addr,
static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
IOMMUAccessFlags flag, int iommu_idx)
{
- S390PCIIOMMU *iommu = container_of(mr, S390PCIIOMMU, iommu_mr);
+ S390PCIBusDevice *pbdev = container_of(mr, S390PCIBusDevice, iommu_mr);
+ S390PCIIOMMU *iommu = pbdev->iommu;
S390IOTLBEntry *entry;
uint64_t iova = addr & TARGET_PAGE_MASK;
uint16_t error = 0;
@@ -593,18 +594,18 @@ err:
return ret;
}
-static void s390_pci_ioat_replay(S390PCIIOMMU *iommu)
+static void s390_pci_ioat_replay(S390PCIBusDevice *pbdev)
{
- S390PCIBusDevice *pbdev = iommu->pbdev;
S390IOTLBEntry entry;
uint16_t error = 0;
uint32_t dma_avail;
hwaddr curr, end;
+ S390PCIIOMMU *iommu = pbdev->iommu;
curr = iommu->pba;
end = iommu->pal;
- if (iommu->dm_mr || !pbdev) {
+ if (iommu->dm_mr) {
/* If direct mapping is used, there are no guest tables to replay */
return;
}
@@ -630,7 +631,7 @@ static void s390_pci_ioat_replay(S390PCIIOMMU *iommu)
if (entry.perm != IOMMU_NONE) {
while (entry.iova < curr && entry.iova < end) {
if (dma_avail > 0) {
- dma_avail = s390_pci_update_iotlb(iommu, &entry);
+ dma_avail = s390_pci_update_iotlb(pbdev, &entry);
} else {
/*
* There is no reliable method to request the guest to
@@ -655,13 +656,13 @@ static void s390_pci_ioat_replay(S390PCIIOMMU *iommu)
static void s390_pci_iommu_replay(IOMMUMemoryRegion *mr,
IOMMUNotifier *notifier)
{
- S390PCIIOMMU *iommu = container_of(mr, S390PCIIOMMU, iommu_mr);
+ S390PCIBusDevice *pbdev = container_of(mr, S390PCIBusDevice, iommu_mr);
/*
* The notifier argument is not used directly; s390_pci_ioat_replay()
* broadcasts via memory_region_notify_iommu() instead.
*/
- s390_pci_ioat_replay(iommu);
+ s390_pci_ioat_replay(pbdev);
}
static S390PCIIOMMU *s390_pci_get_iommu(S390pciState *s, PCIBus *bus,
@@ -783,19 +784,20 @@ static const MemoryRegionOps s390_msi_ctrl_ops = {
.endianness = DEVICE_LITTLE_ENDIAN,
};
-void s390_pci_iommu_enable(S390PCIIOMMU *iommu)
+void s390_pci_iommu_enable(S390PCIBusDevice *pbdev)
{
+ S390PCIIOMMU *iommu = pbdev->iommu;
/*
* The iommu region is initialized against a 0-mapped address space,
* so the smallest IOMMU region we can define runs from 0 to the end
* of the PCI address space.
*/
char *name = g_strdup_printf("iommu-s390-%04x", iommu->pbdev->uid);
- memory_region_init_iommu(&iommu->iommu_mr, sizeof(iommu->iommu_mr),
+ memory_region_init_iommu(&pbdev->iommu_mr, sizeof(pbdev->iommu_mr),
TYPE_S390_IOMMU_MEMORY_REGION, OBJECT(&iommu->mr),
name, iommu->pal + 1);
iommu->enabled = true;
- memory_region_add_subregion(&iommu->mr, 0, MEMORY_REGION(&iommu->iommu_mr));
+ memory_region_add_subregion(&iommu->mr, 0, MEMORY_REGION(&pbdev->iommu_mr));
g_free(name);
}
@@ -821,8 +823,9 @@ void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu)
iommu->dm_mr);
}
-void s390_pci_iommu_disable(S390PCIIOMMU *iommu)
+void s390_pci_iommu_disable(S390PCIBusDevice *pbdev)
{
+ S390PCIIOMMU *iommu = pbdev->iommu;
iommu->enabled = false;
g_hash_table_remove_all(iommu->iotlb);
if (iommu->dm_mr) {
@@ -832,8 +835,8 @@ void s390_pci_iommu_disable(S390PCIIOMMU *iommu)
iommu->dm_mr = NULL;
} else {
memory_region_del_subregion(&iommu->mr,
- MEMORY_REGION(&iommu->iommu_mr));
- object_unparent(OBJECT(&iommu->iommu_mr));
+ MEMORY_REGION(&pbdev->iommu_mr));
+ object_unparent(OBJECT(&pbdev->iommu_mr));
}
}
@@ -1432,7 +1435,7 @@ static void s390_pcihost_reset(DeviceState *dev)
pci_dereg_irqs(pbdev);
}
if (pbdev->iommu->enabled) {
- pci_dereg_ioat(pbdev->iommu);
+ pci_dereg_ioat(pbdev);
}
pbdev->state = ZPCI_FS_STANDBY;
s390_pci_perform_unplug(pbdev);
@@ -1573,7 +1576,7 @@ static void s390_pci_device_reset(DeviceState *dev)
pci_dereg_irqs(pbdev);
}
if (pbdev->iommu->enabled) {
- pci_dereg_ioat(pbdev->iommu);
+ pci_dereg_ioat(pbdev);
}
fmb_timer_free(pbdev);
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index c5d87aa615..50514d02a6 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -630,9 +630,10 @@ int pcistg_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
return 0;
}
-uint32_t s390_pci_update_iotlb(S390PCIIOMMU *iommu,
+uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
S390IOTLBEntry *entry)
{
+ S390PCIIOMMU *iommu = pbdev->iommu;
S390IOTLBEntry *cache = g_hash_table_lookup(iommu->iotlb, &entry->iova);
IOMMUTLBEvent event = {
.type = entry->perm ? IOMMU_NOTIFIER_MAP : IOMMU_NOTIFIER_UNMAP,
@@ -663,7 +664,7 @@ uint32_t s390_pci_update_iotlb(S390PCIIOMMU *iommu,
event.type = IOMMU_NOTIFIER_UNMAP;
event.entry.perm = IOMMU_NONE;
- memory_region_notify_iommu(&iommu->iommu_mr, 0, event);
+ memory_region_notify_iommu(&pbdev->iommu_mr, 0, event);
event.type = IOMMU_NOTIFIER_MAP;
event.entry.perm = entry->perm;
} else {
@@ -683,13 +684,13 @@ uint32_t s390_pci_update_iotlb(S390PCIIOMMU *iommu,
* All associated iotlb entries have already been cleared, trigger the
* unmaps.
*/
- memory_region_notify_iommu(&iommu->iommu_mr, 0, event);
+ memory_region_notify_iommu(&pbdev->iommu_mr, 0, event);
out:
return iommu->dma_limit ? iommu->dma_limit->avail : 1;
}
-static void s390_pci_batch_unmap(S390PCIIOMMU *iommu, uint64_t iova,
+static void s390_pci_batch_unmap(S390PCIBusDevice *pbdev, uint64_t iova,
uint64_t len)
{
uint64_t remain = len, start = iova, end = start + len - 1, mask, size;
@@ -707,7 +708,7 @@ static void s390_pci_batch_unmap(S390PCIIOMMU *iommu, uint64_t iova,
size = mask + 1;
event.entry.iova = start;
event.entry.addr_mask = mask;
- memory_region_notify_iommu(&iommu->iommu_mr, 0, event);
+ memory_region_notify_iommu(&pbdev->iommu_mr, 0, event);
start += size;
remain -= size;
}
@@ -804,7 +805,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
coalesce += entry.len;
} else if (coalesce > 0) {
/* Unleash the coalesced unmap before processing a new map */
- s390_pci_batch_unmap(iommu, iova, coalesce);
+ s390_pci_batch_unmap(pbdev, iova, coalesce);
coalesce = 0;
}
@@ -812,7 +813,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
start = QEMU_ALIGN_UP(start + 1, entry.len);
while (entry.iova < start && entry.iova < end) {
if (dma_avail > 0 || entry.perm == IOMMU_NONE) {
- dma_avail = s390_pci_update_iotlb(iommu, &entry);
+ dma_avail = s390_pci_update_iotlb(pbdev, &entry);
entry.iova += TARGET_PAGE_SIZE;
entry.translated_addr += TARGET_PAGE_SIZE;
} else {
@@ -828,7 +829,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
}
if (coalesce) {
/* Unleash the coalesced unmap before finishing rpcit */
- s390_pci_batch_unmap(iommu, iova, coalesce);
+ s390_pci_batch_unmap(pbdev, iova, coalesce);
coalesce = 0;
}
if (again && dma_avail > 0)
@@ -1079,7 +1080,7 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
iommu->g_iota = g_iota;
if (t) {
- s390_pci_iommu_enable(iommu);
+ s390_pci_iommu_enable(pbdev);
} else {
s390_pci_iommu_direct_map_enable(iommu);
}
@@ -1087,9 +1088,10 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
return 0;
}
-void pci_dereg_ioat(S390PCIIOMMU *iommu)
+void pci_dereg_ioat(S390PCIBusDevice *pbdev)
{
- s390_pci_iommu_disable(iommu);
+ S390PCIIOMMU *iommu = pbdev->iommu;
+ s390_pci_iommu_disable(pbdev);
iommu->pba = 0;
iommu->pal = 0;
iommu->g_iota = 0;
@@ -1313,7 +1315,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_SEQUENCE);
} else {
- pci_dereg_ioat(pbdev->iommu);
+ pci_dereg_ioat(pbdev);
}
break;
case ZPCI_MOD_FC_REREG_IOAT:
@@ -1324,7 +1326,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_SEQUENCE);
} else {
- pci_dereg_ioat(pbdev->iommu);
+ pci_dereg_ioat(pbdev);
if (reg_ioat(env, pbdev, fib, ra)) {
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_INSUF_RES);
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index a732ad652b..74f559deaf 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -277,7 +277,6 @@ struct S390PCIIOMMU {
S390PCIBusDevice *pbdev;
AddressSpace as;
MemoryRegion mr;
- IOMMUMemoryRegion iommu_mr;
MemoryRegion *dm_mr;
bool enabled;
uint64_t g_iota;
@@ -355,6 +354,7 @@ struct S390PCIBusDevice {
S390MsixInfo msix;
AdapterRoutes routes;
S390PCIIOMMU *iommu;
+ IOMMUMemoryRegion iommu_mr;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
IndAddr *indicator;
@@ -392,9 +392,9 @@ int pci_chsc_sei_nt2_have_event(void);
void s390_pci_sclp_configure(SCCB *sccb);
void s390_pci_sclp_deconfigure(SCCB *sccb);
bool s390_pci_is_translation_enabled(uint64_t g_iota);
-void s390_pci_iommu_enable(S390PCIIOMMU *iommu);
+void s390_pci_iommu_enable(S390PCIBusDevice *pbdev);
void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu);
-void s390_pci_iommu_disable(S390PCIIOMMU *iommu);
+void s390_pci_iommu_disable(S390PCIBusDevice *pbdev);
void s390_pci_generate_error_event(uint16_t pec, uint32_t fh, uint32_t fid,
uint64_t faddr, uint32_t e);
uint16_t s390_guest_io_table_walk(uint64_t g_iota, hwaddr addr,
diff --git a/include/hw/s390x/s390-pci-inst.h b/include/hw/s390x/s390-pci-inst.h
index c782990e3b..38268c256e 100644
--- a/include/hw/s390x/s390-pci-inst.h
+++ b/include/hw/s390x/s390-pci-inst.h
@@ -99,7 +99,7 @@ typedef struct ZpciFib {
} QEMU_PACKED ZpciFib;
int pci_dereg_irqs(S390PCIBusDevice *pbdev);
-void pci_dereg_ioat(S390PCIIOMMU *iommu);
+void pci_dereg_ioat(S390PCIBusDevice *pbdev);
int clp_service_call(S390CPU *cpu, uint8_t r2, uintptr_t ra);
int pcilg_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra);
int pcistg_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra);
@@ -111,7 +111,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
uintptr_t ra);
void fmb_timer_free(S390PCIBusDevice *pbdev);
-uint32_t s390_pci_update_iotlb(S390PCIIOMMU *iommu, S390IOTLBEntry *entry);
+uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev, S390IOTLBEntry *entry);
#define ZPCI_IO_BAR_MIN 0
#define ZPCI_IO_BAR_MAX 5
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 04/16] s390x/pci: Move dm_mr from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (2 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 03/16] s390x/pci: Move iommu_mr from S390PCIIOMMU to S390PCIBusDevice Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 05/16] s390x/pci: Move iotlb " Konstantin Shkolnyy
` (11 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify to simplify S390PCIIOMMU towards a structure that
contains only the IOMMU container information needed by the PCI layer for
a given slot.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 21 +++++++++++----------
hw/s390x/s390-pci-inst.c | 2 +-
include/hw/s390x/s390-pci-bus.h | 4 ++--
3 files changed, 14 insertions(+), 13 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index e709d6e478..99122716d7 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -605,7 +605,7 @@ static void s390_pci_ioat_replay(S390PCIBusDevice *pbdev)
curr = iommu->pba;
end = iommu->pal;
- if (iommu->dm_mr) {
+ if (pbdev->dm_mr) {
/* If direct mapping is used, there are no guest tables to replay */
return;
}
@@ -801,8 +801,9 @@ void s390_pci_iommu_enable(S390PCIBusDevice *pbdev)
g_free(name);
}
-void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu)
+void s390_pci_iommu_direct_map_enable(S390PCIBusDevice *pbdev)
{
+ S390PCIIOMMU *iommu = pbdev->iommu;
MachineState *ms = MACHINE(qdev_get_machine());
S390CcwMachineState *s390ms = S390_CCW_MACHINE(ms);
@@ -814,13 +815,13 @@ void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu)
g_autofree char *name = g_strdup_printf("iommu-dm-s390-%04x",
iommu->pbdev->uid);
- iommu->dm_mr = g_malloc0(sizeof(*iommu->dm_mr));
- memory_region_init_alias(iommu->dm_mr, OBJECT(&iommu->mr), name,
+ pbdev->dm_mr = g_malloc0(sizeof(*pbdev->dm_mr));
+ memory_region_init_alias(pbdev->dm_mr, OBJECT(&iommu->mr), name,
get_system_memory(), 0,
s390_get_memory_limit(s390ms));
iommu->enabled = true;
memory_region_add_subregion(&iommu->mr, iommu->pbdev->zpci_fn.sdma,
- iommu->dm_mr);
+ pbdev->dm_mr);
}
void s390_pci_iommu_disable(S390PCIBusDevice *pbdev)
@@ -828,11 +829,11 @@ void s390_pci_iommu_disable(S390PCIBusDevice *pbdev)
S390PCIIOMMU *iommu = pbdev->iommu;
iommu->enabled = false;
g_hash_table_remove_all(iommu->iotlb);
- if (iommu->dm_mr) {
- memory_region_del_subregion(&iommu->mr, iommu->dm_mr);
- object_unparent(OBJECT(iommu->dm_mr));
- g_free(iommu->dm_mr);
- iommu->dm_mr = NULL;
+ if (pbdev->dm_mr) {
+ memory_region_del_subregion(&iommu->mr, pbdev->dm_mr);
+ object_unparent(OBJECT(pbdev->dm_mr));
+ g_free(pbdev->dm_mr);
+ pbdev->dm_mr = NULL;
} else {
memory_region_del_subregion(&iommu->mr,
MEMORY_REGION(&pbdev->iommu_mr));
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index 50514d02a6..f9907c7d5c 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -1082,7 +1082,7 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
if (t) {
s390_pci_iommu_enable(pbdev);
} else {
- s390_pci_iommu_direct_map_enable(iommu);
+ s390_pci_iommu_direct_map_enable(pbdev);
}
return 0;
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index 74f559deaf..b74cc654bb 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -277,7 +277,6 @@ struct S390PCIIOMMU {
S390PCIBusDevice *pbdev;
AddressSpace as;
MemoryRegion mr;
- MemoryRegion *dm_mr;
bool enabled;
uint64_t g_iota;
uint64_t pba;
@@ -355,6 +354,7 @@ struct S390PCIBusDevice {
AdapterRoutes routes;
S390PCIIOMMU *iommu;
IOMMUMemoryRegion iommu_mr;
+ MemoryRegion *dm_mr;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
IndAddr *indicator;
@@ -393,7 +393,7 @@ void s390_pci_sclp_configure(SCCB *sccb);
void s390_pci_sclp_deconfigure(SCCB *sccb);
bool s390_pci_is_translation_enabled(uint64_t g_iota);
void s390_pci_iommu_enable(S390PCIBusDevice *pbdev);
-void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu);
+void s390_pci_iommu_direct_map_enable(S390PCIBusDevice *pbdev);
void s390_pci_iommu_disable(S390PCIBusDevice *pbdev);
void s390_pci_generate_error_event(uint16_t pec, uint32_t fh, uint32_t fid,
uint64_t faddr, uint32_t e);
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 05/16] s390x/pci: Move iotlb from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (3 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 04/16] s390x/pci: Move dm_mr " Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 06/16] s390x/pci: Remove a ptr to S390PCIBusDevice from S390PCIIOMMU Konstantin Shkolnyy
` (10 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the
IOMMU container information needed by the PCI layer for a given slot.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 10 +++++-----
hw/s390x/s390-pci-inst.c | 6 +++---
include/hw/s390x/s390-pci-bus.h | 2 +-
3 files changed, 9 insertions(+), 9 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index 99122716d7..850b4452a5 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -570,7 +570,7 @@ static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
goto err;
}
- entry = g_hash_table_lookup(iommu->iotlb, &iova);
+ entry = g_hash_table_lookup(pbdev->iotlb, &iova);
if (entry) {
ret.iova = entry->iova;
ret.translated_addr = entry->translated_addr;
@@ -692,8 +692,6 @@ static S390PCIIOMMU *s390_pci_get_iommu(S390pciState *s, PCIBus *bus,
PCI_FUNC(devfn));
memory_region_init(&iommu->mr, OBJECT(iommu), mr_name, UINT64_MAX);
address_space_init(&iommu->as, &iommu->mr, as_name);
- iommu->iotlb = g_hash_table_new_full(g_int64_hash, g_int64_equal,
- NULL, g_free);
table->iommu[PCI_SLOT(devfn)] = iommu;
g_free(mr_name);
@@ -828,7 +826,7 @@ void s390_pci_iommu_disable(S390PCIBusDevice *pbdev)
{
S390PCIIOMMU *iommu = pbdev->iommu;
iommu->enabled = false;
- g_hash_table_remove_all(iommu->iotlb);
+ g_hash_table_remove_all(pbdev->iotlb);
if (pbdev->dm_mr) {
memory_region_del_subregion(&iommu->mr, pbdev->dm_mr);
object_unparent(OBJECT(pbdev->dm_mr));
@@ -852,7 +850,6 @@ static void s390_pci_iommu_free(S390pciState *s, PCIBus *bus, int32_t devfn)
}
table->iommu[PCI_SLOT(devfn)] = NULL;
- g_hash_table_destroy(iommu->iotlb);
/*
* An attached PCI device may have memory listeners, eg. VFIO PCI.
* The associated subregion will already have been unmapped in
@@ -1274,6 +1271,8 @@ static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *de
/* the allocated idx is actually getting used */
s->next_idx = (pbdev->idx + 1) & FH_MASK_INDEX;
pbdev->fh = pbdev->idx;
+ pbdev->iotlb = g_hash_table_new_full(g_int64_hash, g_int64_equal,
+ NULL, g_free);
QTAILQ_INSERT_TAIL(&s->zpci_devs, pbdev, link);
g_hash_table_insert(s->zpci_table, &pbdev->idx, pbdev);
} else {
@@ -1316,6 +1315,7 @@ static void s390_pcihost_unplug(const HotplugHandler *hotplug_dev, DeviceState *
if (pbdev->iommu && pbdev->iommu->dma_limit) {
s390_pci_end_dma_count(s, pbdev->iommu->dma_limit);
}
+ g_hash_table_destroy(pbdev->iotlb);
qdev_unrealize(dev);
}
}
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index f9907c7d5c..30114abd43 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -634,7 +634,7 @@ uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
S390IOTLBEntry *entry)
{
S390PCIIOMMU *iommu = pbdev->iommu;
- S390IOTLBEntry *cache = g_hash_table_lookup(iommu->iotlb, &entry->iova);
+ S390IOTLBEntry *cache = g_hash_table_lookup(pbdev->iotlb, &entry->iova);
IOMMUTLBEvent event = {
.type = entry->perm ? IOMMU_NOTIFIER_MAP : IOMMU_NOTIFIER_UNMAP,
.entry = {
@@ -650,7 +650,7 @@ uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
if (!cache) {
goto out;
}
- g_hash_table_remove(iommu->iotlb, &entry->iova);
+ g_hash_table_remove(pbdev->iotlb, &entry->iova);
inc_dma_avail(iommu);
/* Don't notify the iommu yet, maybe we can bundle contiguous unmaps */
goto out;
@@ -677,7 +677,7 @@ uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
cache->translated_addr = entry->translated_addr;
cache->len = TARGET_PAGE_SIZE;
cache->perm = entry->perm;
- g_hash_table_replace(iommu->iotlb, &cache->iova, cache);
+ g_hash_table_replace(pbdev->iotlb, &cache->iova, cache);
}
/*
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index b74cc654bb..7fbe80be16 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -282,7 +282,6 @@ struct S390PCIIOMMU {
uint64_t pba;
uint64_t pal;
uint64_t max_dma_limit;
- GHashTable *iotlb;
S390PCIDMACount *dma_limit;
};
@@ -355,6 +354,7 @@ struct S390PCIBusDevice {
S390PCIIOMMU *iommu;
IOMMUMemoryRegion iommu_mr;
MemoryRegion *dm_mr;
+ GHashTable *iotlb;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
IndAddr *indicator;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 06/16] s390x/pci: Remove a ptr to S390PCIBusDevice from S390PCIIOMMU
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (4 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 05/16] s390x/pci: Move iotlb " Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 07/16] s390x/pci: Move/rename enabled from S390PCIIOMMU to S390PCIBusDevice Konstantin Shkolnyy
` (9 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This pointer is no longer used, after fields were moved from S390PCIIOMMU
to S390PCIBusDevice.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 15 +++++++--------
include/hw/s390x/s390-pci-bus.h | 1 -
2 files changed, 7 insertions(+), 9 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index 850b4452a5..d74d0c4925 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -552,7 +552,7 @@ static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
.perm = IOMMU_NONE,
};
- switch (iommu->pbdev->state) {
+ switch (pbdev->state) {
case ZPCI_FS_ENABLED:
case ZPCI_FS_BLOCKED:
if (!iommu->enabled) {
@@ -587,9 +587,9 @@ static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
}
err:
if (error) {
- iommu->pbdev->state = ZPCI_FS_ERROR;
- s390_pci_generate_error_event(error, iommu->pbdev->fh,
- iommu->pbdev->fid, addr, 0);
+ pbdev->state = ZPCI_FS_ERROR;
+ s390_pci_generate_error_event(error, pbdev->fh,
+ pbdev->fid, addr, 0);
}
return ret;
}
@@ -790,7 +790,7 @@ void s390_pci_iommu_enable(S390PCIBusDevice *pbdev)
* so the smallest IOMMU region we can define runs from 0 to the end
* of the PCI address space.
*/
- char *name = g_strdup_printf("iommu-s390-%04x", iommu->pbdev->uid);
+ char *name = g_strdup_printf("iommu-s390-%04x", pbdev->uid);
memory_region_init_iommu(&pbdev->iommu_mr, sizeof(pbdev->iommu_mr),
TYPE_S390_IOMMU_MEMORY_REGION, OBJECT(&iommu->mr),
name, iommu->pal + 1);
@@ -811,14 +811,14 @@ void s390_pci_iommu_direct_map_enable(S390PCIBusDevice *pbdev)
* IOVA X + SDMA. VFIO will handle pinning via its memory listener.
*/
g_autofree char *name = g_strdup_printf("iommu-dm-s390-%04x",
- iommu->pbdev->uid);
+ pbdev->uid);
pbdev->dm_mr = g_malloc0(sizeof(*pbdev->dm_mr));
memory_region_init_alias(pbdev->dm_mr, OBJECT(&iommu->mr), name,
get_system_memory(), 0,
s390_get_memory_limit(s390ms));
iommu->enabled = true;
- memory_region_add_subregion(&iommu->mr, iommu->pbdev->zpci_fn.sdma,
+ memory_region_add_subregion(&iommu->mr, pbdev->zpci_fn.sdma,
pbdev->dm_mr);
}
@@ -1210,7 +1210,6 @@ static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *de
pbdev->pdev = pdev;
pbdev->iommu = s390_pci_get_iommu(s, pci_get_bus(pdev), pdev->devfn);
- pbdev->iommu->pbdev = pbdev;
pbdev->state = ZPCI_FS_DISABLED;
set_pbdev_info(pbdev);
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index 7fbe80be16..d67daeb34d 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -274,7 +274,6 @@ typedef struct S390PCIDMACount {
struct S390PCIIOMMU {
Object parent_obj;
- S390PCIBusDevice *pbdev;
AddressSpace as;
MemoryRegion mr;
bool enabled;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 07/16] s390x/pci: Move/rename enabled from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (5 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 06/16] s390x/pci: Remove a ptr to S390PCIBusDevice from S390PCIIOMMU Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 08/16] s390x/pci: Move dma_limit " Konstantin Shkolnyy
` (8 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the IOMMU
container information needed by the PCI layer for a given slot.
This also allows to save/restore this field during migration.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 14 +++++++-------
hw/s390x/s390-pci-inst.c | 8 ++++----
include/hw/s390x/s390-pci-bus.h | 2 +-
3 files changed, 12 insertions(+), 12 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index d74d0c4925..3a071eab54 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -206,7 +206,7 @@ void s390_pci_sclp_deconfigure(SCCB *sccb)
} else if (pbdev->summary_ind) {
pci_dereg_irqs(pbdev);
}
- if (pbdev->iommu->enabled) {
+ if (pbdev->iommu_enabled) {
pci_dereg_ioat(pbdev);
}
pbdev->state = ZPCI_FS_STANDBY;
@@ -555,7 +555,7 @@ static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
switch (pbdev->state) {
case ZPCI_FS_ENABLED:
case ZPCI_FS_BLOCKED:
- if (!iommu->enabled) {
+ if (!pbdev->iommu_enabled) {
return ret;
}
break;
@@ -794,7 +794,7 @@ void s390_pci_iommu_enable(S390PCIBusDevice *pbdev)
memory_region_init_iommu(&pbdev->iommu_mr, sizeof(pbdev->iommu_mr),
TYPE_S390_IOMMU_MEMORY_REGION, OBJECT(&iommu->mr),
name, iommu->pal + 1);
- iommu->enabled = true;
+ pbdev->iommu_enabled = true;
memory_region_add_subregion(&iommu->mr, 0, MEMORY_REGION(&pbdev->iommu_mr));
g_free(name);
}
@@ -817,7 +817,7 @@ void s390_pci_iommu_direct_map_enable(S390PCIBusDevice *pbdev)
memory_region_init_alias(pbdev->dm_mr, OBJECT(&iommu->mr), name,
get_system_memory(), 0,
s390_get_memory_limit(s390ms));
- iommu->enabled = true;
+ pbdev->iommu_enabled = true;
memory_region_add_subregion(&iommu->mr, pbdev->zpci_fn.sdma,
pbdev->dm_mr);
}
@@ -825,7 +825,7 @@ void s390_pci_iommu_direct_map_enable(S390PCIBusDevice *pbdev)
void s390_pci_iommu_disable(S390PCIBusDevice *pbdev)
{
S390PCIIOMMU *iommu = pbdev->iommu;
- iommu->enabled = false;
+ pbdev->iommu_enabled = false;
g_hash_table_remove_all(pbdev->iotlb);
if (pbdev->dm_mr) {
memory_region_del_subregion(&iommu->mr, pbdev->dm_mr);
@@ -1434,7 +1434,7 @@ static void s390_pcihost_reset(DeviceState *dev)
} else if (pbdev->summary_ind) {
pci_dereg_irqs(pbdev);
}
- if (pbdev->iommu->enabled) {
+ if (pbdev->iommu_enabled) {
pci_dereg_ioat(pbdev);
}
pbdev->state = ZPCI_FS_STANDBY;
@@ -1575,7 +1575,7 @@ static void s390_pci_device_reset(DeviceState *dev)
} else if (pbdev->summary_ind) {
pci_dereg_irqs(pbdev);
}
- if (pbdev->iommu->enabled) {
+ if (pbdev->iommu_enabled) {
pci_dereg_ioat(pbdev);
}
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index 30114abd43..c1d9be232a 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -1299,7 +1299,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
if (dmaas != 0) {
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_DMAAS_INVAL);
- } else if (pbdev->iommu->enabled) {
+ } else if (pbdev->iommu_enabled) {
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_SEQUENCE);
} else if (reg_ioat(env, pbdev, fib, ra)) {
@@ -1311,7 +1311,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
if (dmaas != 0) {
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_DMAAS_INVAL);
- } else if (!pbdev->iommu->enabled) {
+ } else if (!pbdev->iommu_enabled) {
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_SEQUENCE);
} else {
@@ -1322,7 +1322,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
if (dmaas != 0) {
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_DMAAS_INVAL);
- } else if (!pbdev->iommu->enabled) {
+ } else if (!pbdev->iommu_enabled) {
cc = ZPCI_PCI_LS_ERR;
s390_set_status_code(env, r1, ZPCI_MOD_ST_SEQUENCE);
} else {
@@ -1452,7 +1452,7 @@ int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
/* fallthrough */
case ZPCI_FS_ENABLED:
fib.fc |= 0x80;
- if (pbdev->iommu->enabled) {
+ if (pbdev->iommu_enabled) {
fib.fc |= 0x10;
}
if (!(fh & FH_MASK_ENABLE)) {
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index d67daeb34d..b851295b44 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -276,7 +276,6 @@ struct S390PCIIOMMU {
Object parent_obj;
AddressSpace as;
MemoryRegion mr;
- bool enabled;
uint64_t g_iota;
uint64_t pba;
uint64_t pal;
@@ -351,6 +350,7 @@ struct S390PCIBusDevice {
S390MsixInfo msix;
AdapterRoutes routes;
S390PCIIOMMU *iommu;
+ bool iommu_enabled;
IOMMUMemoryRegion iommu_mr;
MemoryRegion *dm_mr;
GHashTable *iotlb;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 08/16] s390x/pci: Move dma_limit from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (6 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 07/16] s390x/pci: Move/rename enabled from S390PCIIOMMU to S390PCIBusDevice Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 09/16] s390x/pci: Move g_iota " Konstantin Shkolnyy
` (7 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the
IOMMU container information needed by the PCI layer for a given slot.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 10 +++++-----
hw/s390x/s390-pci-inst.c | 23 +++++++++++------------
include/hw/s390x/s390-pci-bus.h | 2 +-
3 files changed, 17 insertions(+), 18 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index 3a071eab54..53af08018d 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -610,8 +610,8 @@ static void s390_pci_ioat_replay(S390PCIBusDevice *pbdev)
return;
}
- if (iommu->dma_limit) {
- dma_avail = iommu->dma_limit->avail;
+ if (pbdev->dma_limit) {
+ dma_avail = pbdev->dma_limit->avail;
} else {
dma_avail = 1;
}
@@ -1233,7 +1233,7 @@ static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *de
pbdev->forwarding_assist = false;
}
}
- pbdev->iommu->dma_limit = s390_pci_start_dma_count(s, pbdev);
+ pbdev->dma_limit = s390_pci_start_dma_count(s, pbdev);
/* Fill in CLP information passed via the vfio region */
s390_pci_get_clp_info(pbdev);
if (!pbdev->interp) {
@@ -1311,8 +1311,8 @@ static void s390_pcihost_unplug(const HotplugHandler *hotplug_dev, DeviceState *
pbdev->fid = 0;
QTAILQ_REMOVE(&s->zpci_devs, pbdev, link);
g_hash_table_remove(s->zpci_table, &pbdev->idx);
- if (pbdev->iommu && pbdev->iommu->dma_limit) {
- s390_pci_end_dma_count(s, pbdev->iommu->dma_limit);
+ if (pbdev->dma_limit) {
+ s390_pci_end_dma_count(s, pbdev->dma_limit);
}
g_hash_table_destroy(pbdev->iotlb);
qdev_unrealize(dev);
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index c1d9be232a..fa18ad2617 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -28,17 +28,17 @@
#include "trace.h"
-static inline void inc_dma_avail(S390PCIIOMMU *iommu)
+static inline void inc_dma_avail(S390PCIBusDevice *pbdev)
{
- if (iommu->dma_limit) {
- iommu->dma_limit->avail++;
+ if (pbdev->dma_limit) {
+ pbdev->dma_limit->avail++;
}
}
-static inline void dec_dma_avail(S390PCIIOMMU *iommu)
+static inline void dec_dma_avail(S390PCIBusDevice *pbdev)
{
- if (iommu->dma_limit) {
- iommu->dma_limit->avail--;
+ if (pbdev->dma_limit) {
+ pbdev->dma_limit->avail--;
}
}
@@ -633,7 +633,6 @@ int pcistg_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
S390IOTLBEntry *entry)
{
- S390PCIIOMMU *iommu = pbdev->iommu;
S390IOTLBEntry *cache = g_hash_table_lookup(pbdev->iotlb, &entry->iova);
IOMMUTLBEvent event = {
.type = entry->perm ? IOMMU_NOTIFIER_MAP : IOMMU_NOTIFIER_UNMAP,
@@ -651,7 +650,7 @@ uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
goto out;
}
g_hash_table_remove(pbdev->iotlb, &entry->iova);
- inc_dma_avail(iommu);
+ inc_dma_avail(pbdev);
/* Don't notify the iommu yet, maybe we can bundle contiguous unmaps */
goto out;
} else {
@@ -669,7 +668,7 @@ uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
event.entry.perm = entry->perm;
} else {
/* invalid->valid transitions consume a new DMA slot */
- dec_dma_avail(iommu);
+ dec_dma_avail(pbdev);
}
cache = g_new(S390IOTLBEntry, 1);
@@ -687,7 +686,7 @@ uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev,
memory_region_notify_iommu(&pbdev->iommu_mr, 0, event);
out:
- return iommu->dma_limit ? iommu->dma_limit->avail : 1;
+ return pbdev->dma_limit ? pbdev->dma_limit->avail : 1;
}
static void s390_pci_batch_unmap(S390PCIBusDevice *pbdev, uint64_t iova,
@@ -764,8 +763,8 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
}
iommu = pbdev->iommu;
- if (iommu->dma_limit) {
- dma_avail = iommu->dma_limit->avail;
+ if (pbdev->dma_limit) {
+ dma_avail = pbdev->dma_limit->avail;
} else {
dma_avail = 1;
}
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index b851295b44..00e7e456e7 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -280,7 +280,6 @@ struct S390PCIIOMMU {
uint64_t pba;
uint64_t pal;
uint64_t max_dma_limit;
- S390PCIDMACount *dma_limit;
};
typedef struct S390PCIIOMMUTable {
@@ -354,6 +353,7 @@ struct S390PCIBusDevice {
IOMMUMemoryRegion iommu_mr;
MemoryRegion *dm_mr;
GHashTable *iotlb;
+ S390PCIDMACount *dma_limit;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
IndAddr *indicator;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 09/16] s390x/pci: Move g_iota from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (7 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 08/16] s390x/pci: Move dma_limit " Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 10/16] s390x/pci: Move pba " Konstantin Shkolnyy
` (6 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the
IOMMU container information needed by the PCI layer for a given slot.
This also allows to save/restore this field during migration.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 2 +-
hw/s390x/s390-pci-inst.c | 10 +++++-----
include/hw/s390x/s390-pci-bus.h | 2 +-
3 files changed, 7 insertions(+), 7 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index 53af08018d..88ffa93977 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -617,7 +617,7 @@ static void s390_pci_ioat_replay(S390PCIBusDevice *pbdev)
}
while (curr < end) {
- error = s390_guest_io_table_walk(iommu->g_iota, curr, &entry);
+ error = s390_guest_io_table_walk(pbdev->g_iota, curr, &entry);
if (error) {
pbdev->state = ZPCI_FS_ERROR;
s390_pci_generate_error_event(error, pbdev->fh, pbdev->fid, curr,
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index fa18ad2617..3163539ead 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -768,7 +768,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
} else {
dma_avail = 1;
}
- if (!iommu->g_iota) {
+ if (!pbdev->g_iota) {
error = ERR_EVENT_INVALAS;
goto err;
}
@@ -788,7 +788,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
start = sstart;
again = false;
while (start < end) {
- error = s390_guest_io_table_walk(iommu->g_iota, start, &entry);
+ error = s390_guest_io_table_walk(pbdev->g_iota, start, &entry);
if (error) {
break;
}
@@ -1076,7 +1076,7 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
iommu->pba = pba;
iommu->pal = pal;
- iommu->g_iota = g_iota;
+ pbdev->g_iota = g_iota;
if (t) {
s390_pci_iommu_enable(pbdev);
@@ -1093,7 +1093,7 @@ void pci_dereg_ioat(S390PCIBusDevice *pbdev)
s390_pci_iommu_disable(pbdev);
iommu->pba = 0;
iommu->pal = 0;
- iommu->g_iota = 0;
+ pbdev->g_iota = 0;
}
void fmb_timer_free(S390PCIBusDevice *pbdev)
@@ -1466,7 +1466,7 @@ int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
stq_be_p(&fib.pba, pbdev->iommu->pba);
stq_be_p(&fib.pal, pbdev->iommu->pal);
- stq_be_p(&fib.iota, pbdev->iommu->g_iota);
+ stq_be_p(&fib.iota, pbdev->g_iota);
stq_be_p(&fib.aibv, pbdev->routes.adapter.ind_addr);
stq_be_p(&fib.aisb, pbdev->routes.adapter.summary_addr);
stq_be_p(&fib.fmb_addr, pbdev->fmb_addr);
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index 00e7e456e7..1bb1b87d52 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -276,7 +276,6 @@ struct S390PCIIOMMU {
Object parent_obj;
AddressSpace as;
MemoryRegion mr;
- uint64_t g_iota;
uint64_t pba;
uint64_t pal;
uint64_t max_dma_limit;
@@ -353,6 +352,7 @@ struct S390PCIBusDevice {
IOMMUMemoryRegion iommu_mr;
MemoryRegion *dm_mr;
GHashTable *iotlb;
+ uint64_t g_iota;
S390PCIDMACount *dma_limit;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 10/16] s390x/pci: Move pba from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (8 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 09/16] s390x/pci: Move g_iota " Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 11/16] s390x/pci: Move pal " Konstantin Shkolnyy
` (5 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the
IOMMU container information needed by the PCI layer for a given slot.
This also allows to save/restore this field during migration.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 4 ++--
hw/s390x/s390-pci-inst.c | 10 +++++-----
include/hw/s390x/s390-pci-bus.h | 2 +-
3 files changed, 8 insertions(+), 8 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index 88ffa93977..b0aaea49cf 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -565,7 +565,7 @@ static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
trace_s390_pci_iommu_xlate(addr);
- if (addr < iommu->pba || addr > iommu->pal) {
+ if (addr < pbdev->pba || addr > iommu->pal) {
error = ERR_EVENT_OORANGE;
goto err;
}
@@ -602,7 +602,7 @@ static void s390_pci_ioat_replay(S390PCIBusDevice *pbdev)
hwaddr curr, end;
S390PCIIOMMU *iommu = pbdev->iommu;
- curr = iommu->pba;
+ curr = pbdev->pba;
end = iommu->pal;
if (pbdev->dm_mr) {
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index 3163539ead..1006b409e4 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -773,7 +773,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
goto err;
}
- if (end < start || end < iommu->pba || start > iommu->pal) {
+ if (end < start || end < pbdev->pba || start > iommu->pal) {
error = ERR_EVENT_OORANGE;
goto err;
}
@@ -781,7 +781,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
* If the specified range at least partially overlaps the registered
* aperture, clamp the request to the aperture and ignore the rest.
*/
- sstart = MAX(start, iommu->pba);
+ sstart = MAX(start, pbdev->pba);
end = MIN(end, iommu->pal + 1);
retry:
@@ -1074,7 +1074,7 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
return -EINVAL;
}
- iommu->pba = pba;
+ pbdev->pba = pba;
iommu->pal = pal;
pbdev->g_iota = g_iota;
@@ -1091,7 +1091,7 @@ void pci_dereg_ioat(S390PCIBusDevice *pbdev)
{
S390PCIIOMMU *iommu = pbdev->iommu;
s390_pci_iommu_disable(pbdev);
- iommu->pba = 0;
+ pbdev->pba = 0;
iommu->pal = 0;
pbdev->g_iota = 0;
}
@@ -1464,7 +1464,7 @@ int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
return 0;
}
- stq_be_p(&fib.pba, pbdev->iommu->pba);
+ stq_be_p(&fib.pba, pbdev->pba);
stq_be_p(&fib.pal, pbdev->iommu->pal);
stq_be_p(&fib.iota, pbdev->g_iota);
stq_be_p(&fib.aibv, pbdev->routes.adapter.ind_addr);
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index 1bb1b87d52..82e79885e4 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -276,7 +276,6 @@ struct S390PCIIOMMU {
Object parent_obj;
AddressSpace as;
MemoryRegion mr;
- uint64_t pba;
uint64_t pal;
uint64_t max_dma_limit;
};
@@ -353,6 +352,7 @@ struct S390PCIBusDevice {
MemoryRegion *dm_mr;
GHashTable *iotlb;
uint64_t g_iota;
+ uint64_t pba;
S390PCIDMACount *dma_limit;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 11/16] s390x/pci: Move pal from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (9 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 10/16] s390x/pci: Move pba " Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 12/16] s390x/pci: Move max_dma_limit " Konstantin Shkolnyy
` (4 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the
IOMMU container information needed by the PCI layer for a given slot.
This also allows to save/restore this field during migration.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 8 +++-----
hw/s390x/s390-pci-inst.c | 14 +++++---------
include/hw/s390x/s390-pci-bus.h | 2 +-
3 files changed, 9 insertions(+), 15 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index b0aaea49cf..799e596fec 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -540,7 +540,6 @@ static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
IOMMUAccessFlags flag, int iommu_idx)
{
S390PCIBusDevice *pbdev = container_of(mr, S390PCIBusDevice, iommu_mr);
- S390PCIIOMMU *iommu = pbdev->iommu;
S390IOTLBEntry *entry;
uint64_t iova = addr & TARGET_PAGE_MASK;
uint16_t error = 0;
@@ -565,7 +564,7 @@ static IOMMUTLBEntry s390_translate_iommu(IOMMUMemoryRegion *mr, hwaddr addr,
trace_s390_pci_iommu_xlate(addr);
- if (addr < pbdev->pba || addr > iommu->pal) {
+ if (addr < pbdev->pba || addr > pbdev->pal) {
error = ERR_EVENT_OORANGE;
goto err;
}
@@ -600,10 +599,9 @@ static void s390_pci_ioat_replay(S390PCIBusDevice *pbdev)
uint16_t error = 0;
uint32_t dma_avail;
hwaddr curr, end;
- S390PCIIOMMU *iommu = pbdev->iommu;
curr = pbdev->pba;
- end = iommu->pal;
+ end = pbdev->pal;
if (pbdev->dm_mr) {
/* If direct mapping is used, there are no guest tables to replay */
@@ -793,7 +791,7 @@ void s390_pci_iommu_enable(S390PCIBusDevice *pbdev)
char *name = g_strdup_printf("iommu-s390-%04x", pbdev->uid);
memory_region_init_iommu(&pbdev->iommu_mr, sizeof(pbdev->iommu_mr),
TYPE_S390_IOMMU_MEMORY_REGION, OBJECT(&iommu->mr),
- name, iommu->pal + 1);
+ name, pbdev->pal + 1);
pbdev->iommu_enabled = true;
memory_region_add_subregion(&iommu->mr, 0, MEMORY_REGION(&pbdev->iommu_mr));
g_free(name);
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index 1006b409e4..98380526d1 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -720,7 +720,6 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
uint32_t fh;
uint16_t error = 0;
S390PCIBusDevice *pbdev;
- S390PCIIOMMU *iommu;
S390IOTLBEntry entry;
hwaddr start, end, sstart;
uint32_t dma_avail;
@@ -762,7 +761,6 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
break;
}
- iommu = pbdev->iommu;
if (pbdev->dma_limit) {
dma_avail = pbdev->dma_limit->avail;
} else {
@@ -773,7 +771,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
goto err;
}
- if (end < start || end < pbdev->pba || start > iommu->pal) {
+ if (end < start || end < pbdev->pba || start > pbdev->pal) {
error = ERR_EVENT_OORANGE;
goto err;
}
@@ -782,7 +780,7 @@ int rpcit_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra)
* aperture, clamp the request to the aperture and ignore the rest.
*/
sstart = MAX(start, pbdev->pba);
- end = MIN(end, iommu->pal + 1);
+ end = MIN(end, pbdev->pal + 1);
retry:
start = sstart;
@@ -1033,7 +1031,6 @@ bool s390_pci_is_translation_enabled(uint64_t g_iota)
static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
uintptr_t ra)
{
- S390PCIIOMMU *iommu = pbdev->iommu;
uint64_t pba = ldq_be_p(&fib.pba);
uint64_t pal = ldq_be_p(&fib.pal);
uint64_t g_iota = ldq_be_p(&fib.iota);
@@ -1075,7 +1072,7 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
}
pbdev->pba = pba;
- iommu->pal = pal;
+ pbdev->pal = pal;
pbdev->g_iota = g_iota;
if (t) {
@@ -1089,10 +1086,9 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
void pci_dereg_ioat(S390PCIBusDevice *pbdev)
{
- S390PCIIOMMU *iommu = pbdev->iommu;
s390_pci_iommu_disable(pbdev);
pbdev->pba = 0;
- iommu->pal = 0;
+ pbdev->pal = 0;
pbdev->g_iota = 0;
}
@@ -1465,7 +1461,7 @@ int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
}
stq_be_p(&fib.pba, pbdev->pba);
- stq_be_p(&fib.pal, pbdev->iommu->pal);
+ stq_be_p(&fib.pal, pbdev->pal);
stq_be_p(&fib.iota, pbdev->g_iota);
stq_be_p(&fib.aibv, pbdev->routes.adapter.ind_addr);
stq_be_p(&fib.aisb, pbdev->routes.adapter.summary_addr);
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index 82e79885e4..9332763522 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -276,7 +276,6 @@ struct S390PCIIOMMU {
Object parent_obj;
AddressSpace as;
MemoryRegion mr;
- uint64_t pal;
uint64_t max_dma_limit;
};
@@ -353,6 +352,7 @@ struct S390PCIBusDevice {
GHashTable *iotlb;
uint64_t g_iota;
uint64_t pba;
+ uint64_t pal;
S390PCIDMACount *dma_limit;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 12/16] s390x/pci: Move max_dma_limit from S390PCIIOMMU to S390PCIBusDevice
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (10 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 11/16] s390x/pci: Move pal " Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 13/16] s390x/pci: Add a comment explaining S390PCIIOMMU purpose Konstantin Shkolnyy
` (3 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
This field is only used when S390PCIBusDevice exists, so it can be moved
there to simplify S390PCIIOMMU towards a structure that contains only the
IOMMU container information needed by the PCI layer for a given slot.
Reviewed-by: Farhan Ali <alifm@linux.ibm.com>
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-vfio.c | 4 ++--
include/hw/s390x/s390-pci-bus.h | 2 +-
2 files changed, 3 insertions(+), 3 deletions(-)
diff --git a/hw/s390x/s390-pci-vfio.c b/hw/s390x/s390-pci-vfio.c
index db6de00bd2..a035ba5b2c 100644
--- a/hw/s390x/s390-pci-vfio.c
+++ b/hw/s390x/s390-pci-vfio.c
@@ -90,7 +90,7 @@ S390PCIDMACount *s390_pci_start_dma_count(S390pciState *s,
cnt->users = 1;
cnt->avail = avail;
QTAILQ_INSERT_TAIL(&s->zpci_dma_limit, cnt, link);
- pbdev->iommu->max_dma_limit = avail;
+ pbdev->max_dma_limit = avail;
return cnt;
}
@@ -152,7 +152,7 @@ static void s390_pci_read_base(S390PCIBusDevice *pbdev,
* to request that the guest free DMA mappings as necessary.
*/
if (!pbdev->rtr_avail) {
- vfio_size = pbdev->iommu->max_dma_limit << qemu_target_page_bits();
+ vfio_size = pbdev->max_dma_limit << qemu_target_page_bits();
if (vfio_size > 0 && vfio_size < cap->end_dma - cap->start_dma + 1) {
pbdev->zpci_fn.edma = cap->start_dma + vfio_size - 1;
}
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index 9332763522..c7fa7b9514 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -276,7 +276,6 @@ struct S390PCIIOMMU {
Object parent_obj;
AddressSpace as;
MemoryRegion mr;
- uint64_t max_dma_limit;
};
typedef struct S390PCIIOMMUTable {
@@ -353,6 +352,7 @@ struct S390PCIBusDevice {
uint64_t g_iota;
uint64_t pba;
uint64_t pal;
+ uint64_t max_dma_limit;
S390PCIDMACount *dma_limit;
MemoryRegion msix_notify_mr;
IndAddr *summary_ind;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 13/16] s390x/pci: Add a comment explaining S390PCIIOMMU purpose
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (11 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 12/16] s390x/pci: Move max_dma_limit " Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function Konstantin Shkolnyy
` (2 subsequent siblings)
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
Now that S390PCIIOMMU had been reduced to only the IOMMU container
information needed by the PCI layer for the associated slot, add a
comment explaining the purpose of the structure.
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
include/hw/s390x/s390-pci-bus.h | 8 ++++++++
1 file changed, 8 insertions(+)
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index c7fa7b9514..5c2d07e38e 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -272,6 +272,14 @@ typedef struct S390PCIDMACount {
QTAILQ_ENTRY(S390PCIDMACount) link;
} S390PCIDMACount;
+/*
+ * This structure holds the AddressSpace for a PCI device slot. It must be
+ * allocated before PCIDevice registration completes, specifically to satisfy
+ * the get_address_space IOMMU callback invoked by do_pci_register_device().
+ * The root MemoryRegion is otherwise empty; DMA translations only become
+ * functional once the guest configures the device, at which point
+ * s390_pci_iommu_enable() registers the zPCI translation subregions beneath it.
+ */
struct S390PCIIOMMU {
Object parent_obj;
AddressSpace as;
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (12 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 13/16] s390x/pci: Add a comment explaining S390PCIIOMMU purpose Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-10-02 18:28 ` Matthew Rosato
2026-09-30 14:52 ` [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
2026-09-30 14:52 ` [PATCH v11 16/16] s390x/pci: Create function to contain fmb_timer start Konstantin Shkolnyy
15 siblings, 1 reply; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
Extract the pba, pal and g_iota checks from reg_ioat() into a function.
Make it external so that it can also be used in a follow up change.
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-inst.c | 52 ++++++++++++++++++++------------
include/hw/s390x/s390-pci-inst.h | 2 ++
2 files changed, 34 insertions(+), 20 deletions(-)
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index 98380526d1..b0af15ff15 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -1028,32 +1028,22 @@ bool s390_pci_is_translation_enabled(uint64_t g_iota)
return ((g_iota >> 11) & 0x1) != 0; /* "T" bit */
}
-static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
- uintptr_t ra)
+bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
+ uint64_t pal, uint64_t g_iota, Error **errp)
{
- uint64_t pba = ldq_be_p(&fib.pba);
- uint64_t pal = ldq_be_p(&fib.pal);
- uint64_t g_iota = ldq_be_p(&fib.iota);
uint8_t dt = (g_iota >> 2) & 0x7;
bool t = s390_pci_is_translation_enabled(g_iota);
- pba &= ~0xfff;
- pal |= 0xfff;
if (pba > pal || pba < pbdev->zpci_fn.sdma || pal > pbdev->zpci_fn.edma) {
- s390_program_interrupt(env, PGM_OPERAND, ra);
- return -EINVAL;
+ return false;
}
-
/* currently we only support designation type 1 with translation */
if (t && dt != ZPCI_IOTA_RTTO) {
- qemu_log_mask(LOG_GUEST_ERROR,
- "unsupported ioat dt %d t %d\n", dt, t);
- s390_program_interrupt(env, PGM_OPERAND, ra);
- return -EINVAL;
+ error_setg(errp, "unsupported ioat dt %d t %d", dt, t);
+ return false;
} else if (!t && !pbdev->rtr_avail) {
- qemu_log_mask(LOG_GUEST_ERROR, "relaxed translation not allowed\n");
- s390_program_interrupt(env, PGM_OPERAND, ra);
- return -EINVAL;
+ error_setg(errp, "relaxed translation not allowed");
+ return false;
}
/*
@@ -1064,9 +1054,31 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
* to QEMU for additional IOAT regions.
*/
if (t && pal >= ZPCI_TABLE_SIZE_RT) {
- qemu_log_mask(LOG_GUEST_ERROR,
- "ioat pal 0x%"PRIx64" exceeds max translatable address\n",
- pal);
+ error_setg(errp,
+ "ioat pal 0x%"PRIx64" exceeds max translatable address",
+ pal);
+ return false;
+ }
+ return true;
+}
+
+static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
+ uintptr_t ra)
+{
+ Error *err = NULL;
+ uint64_t pba = ldq_be_p(&fib.pba);
+ uint64_t pal = ldq_be_p(&fib.pal);
+ uint64_t g_iota = ldq_be_p(&fib.iota);
+ bool t = s390_pci_is_translation_enabled(g_iota);
+
+ pba &= ~0xfff;
+ pal |= 0xfff;
+
+ if (!s390_pci_ioat_validate(pbdev, pba, pal, g_iota, &err)) {
+ if (err) {
+ qemu_log_mask(LOG_GUEST_ERROR, "%s\n", error_get_pretty(err));
+ error_free(err);
+ }
s390_program_interrupt(env, PGM_OPERAND, ra);
return -EINVAL;
}
diff --git a/include/hw/s390x/s390-pci-inst.h b/include/hw/s390x/s390-pci-inst.h
index 38268c256e..3493d2ced0 100644
--- a/include/hw/s390x/s390-pci-inst.h
+++ b/include/hw/s390x/s390-pci-inst.h
@@ -100,6 +100,8 @@ typedef struct ZpciFib {
int pci_dereg_irqs(S390PCIBusDevice *pbdev);
void pci_dereg_ioat(S390PCIBusDevice *pbdev);
+bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
+ uint64_t pal, uint64_t g_iota, Error **errp);
int clp_service_call(S390CPU *cpu, uint8_t r2, uintptr_t ra);
int pcilg_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra);
int pcistg_service_call(S390CPU *cpu, uint8_t r1, uint8_t r2, uintptr_t ra);
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* Re: [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function
2026-09-30 14:52 ` [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function Konstantin Shkolnyy
@ 2026-10-02 18:28 ` Matthew Rosato
2026-10-03 0:43 ` Konstantin Shkolnyy
0 siblings, 1 reply; 26+ messages in thread
From: Matthew Rosato @ 2026-10-02 18:28 UTC (permalink / raw)
To: Konstantin Shkolnyy
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel
> -static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
> - uintptr_t ra)
> +bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
> + uint64_t pal, uint64_t g_iota, Error **errp)
> {
> - uint64_t pba = ldq_be_p(&fib.pba);
> - uint64_t pal = ldq_be_p(&fib.pal);
> - uint64_t g_iota = ldq_be_p(&fib.iota);
> uint8_t dt = (g_iota >> 2) & 0x7;
> bool t = s390_pci_is_translation_enabled(g_iota);
>
> - pba &= ~0xfff;
> - pal |= 0xfff;
> if (pba > pal || pba < pbdev->zpci_fn.sdma || pal > pbdev->zpci_fn.edma) {
> - s390_program_interrupt(env, PGM_OPERAND, ra);
> - return -EINVAL;
> + return false;
This will now violate include/qapi/error.h because you're returning
false but not setting errp:
* - On success, the function should not touch *errp. On failure, it
* should set a new error, e.g. with error_setg(errp, ...), or
* propagate an existing one, e.g. with error_propagate(errp, ...).
A simple solution would be to add a new message for this case.
[...]
> +
> +static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
> + uintptr_t ra)
> +{
> + Error *err = NULL;
> + uint64_t pba = ldq_be_p(&fib.pba);
> + uint64_t pal = ldq_be_p(&fib.pal);
> + uint64_t g_iota = ldq_be_p(&fib.iota);
> + bool t = s390_pci_is_translation_enabled(g_iota);
> +
> + pba &= ~0xfff;
> + pal |= 0xfff;
> +
> + if (!s390_pci_ioat_validate(pbdev, pba, pal, g_iota, &err)) {
> + if (err) {
> + qemu_log_mask(LOG_GUEST_ERROR, "%s\n", error_get_pretty(err));
> + error_free(err);
> + }
With that change, you won't need the if (err) check anymore.
Thanks,
Matt
^ permalink raw reply [flat|nested] 26+ messages in thread* Re: [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function
2026-10-02 18:28 ` Matthew Rosato
@ 2026-10-03 0:43 ` Konstantin Shkolnyy
2026-10-05 13:23 ` Konstantin Shkolnyy
2026-10-05 13:30 ` Matthew Rosato
0 siblings, 2 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-10-03 0:43 UTC (permalink / raw)
To: Matthew Rosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel
On 261002 13:28, Matthew Rosato wrote:
>
>> -static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
>> - uintptr_t ra)
>> +bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
>> + uint64_t pal, uint64_t g_iota, Error **errp)
>> {
>> - uint64_t pba = ldq_be_p(&fib.pba);
>> - uint64_t pal = ldq_be_p(&fib.pal);
>> - uint64_t g_iota = ldq_be_p(&fib.iota);
>> uint8_t dt = (g_iota >> 2) & 0x7;
>> bool t = s390_pci_is_translation_enabled(g_iota);
>>
>> - pba &= ~0xfff;
>> - pal |= 0xfff;
>> if (pba > pal || pba < pbdev->zpci_fn.sdma || pal > pbdev->zpci_fn.edma) {
>> - s390_program_interrupt(env, PGM_OPERAND, ra);
>> - return -EINVAL;
>> + return false;
>
> This will now violate include/qapi/error.h because you're returning
> false but not setting errp:
>
> * - On success, the function should not touch *errp. On failure, it
> * should set a new error, e.g. with error_setg(errp, ...), or
> * propagate an existing one, e.g. with error_propagate(errp, ...).
>
> A simple solution would be to add a new message for this case.
But the original reg_ioat() also didn't print any message in this error
branch. I had to assume it was intentional, and preserved the behavior.
If you believe a new message should be introduced, I can insert a commit
for that - what would be the explanation?
>
> [...]
>> +
>> +static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib,
>> + uintptr_t ra)
>> +{
>> + Error *err = NULL;
>> + uint64_t pba = ldq_be_p(&fib.pba);
>> + uint64_t pal = ldq_be_p(&fib.pal);
>> + uint64_t g_iota = ldq_be_p(&fib.iota);
>> + bool t = s390_pci_is_translation_enabled(g_iota);
>> +
>> + pba &= ~0xfff;
>> + pal |= 0xfff;
>> +
>> + if (!s390_pci_ioat_validate(pbdev, pba, pal, g_iota, &err)) {
>> + if (err) {
>> + qemu_log_mask(LOG_GUEST_ERROR, "%s\n", error_get_pretty(err));
>> + error_free(err);
>> + }
>
> With that change, you won't need the if (err) check anymore.
>
> Thanks,
> Matt
^ permalink raw reply [flat|nested] 26+ messages in thread* Re: [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function
2026-10-03 0:43 ` Konstantin Shkolnyy
@ 2026-10-05 13:23 ` Konstantin Shkolnyy
2026-10-05 13:46 ` Matthew Rosato
2026-10-05 13:30 ` Matthew Rosato
1 sibling, 1 reply; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-10-05 13:23 UTC (permalink / raw)
To: Matthew Rosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel
On 261002 19:43, Konstantin Shkolnyy wrote:
> On 261002 13:28, Matthew Rosato wrote:
>>
>>> -static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev,
>>> ZpciFib fib,
>>> - uintptr_t ra)
>>> +bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
>>> + uint64_t pal, uint64_t g_iota, Error
>>> **errp)
>>> {
>>> - uint64_t pba = ldq_be_p(&fib.pba);
>>> - uint64_t pal = ldq_be_p(&fib.pal);
>>> - uint64_t g_iota = ldq_be_p(&fib.iota);
>>> uint8_t dt = (g_iota >> 2) & 0x7;
>>> bool t = s390_pci_is_translation_enabled(g_iota);
>>> - pba &= ~0xfff;
>>> - pal |= 0xfff;
>>> if (pba > pal || pba < pbdev->zpci_fn.sdma || pal > pbdev-
>>> >zpci_fn.edma) {
>>> - s390_program_interrupt(env, PGM_OPERAND, ra);
>>> - return -EINVAL;
>>> + return false;
>>
>> This will now violate include/qapi/error.h because you're returning
>> false but not setting errp:
>>
>> * - On success, the function should not touch *errp. On failure, it
>> * should set a new error, e.g. with error_setg(errp, ...), or
>> * propagate an existing one, e.g. with error_propagate(errp, ...).
>>
>> A simple solution would be to add a new message for this case.
>
> But the original reg_ioat() also didn't print any message in this error
> branch. I had to assume it was intentional, and preserved the behavior.
>
> If you believe a new message should be introduced, I can insert a commit
> for that - what would be the explanation?
Other code sites reporting PGM_OPERAND also don't generate error
messages - it appears, the current policy is not to do that in the cases
when the guest is at fault (gave invalid parameters). The case at hand
falls in that category.
>
>>
>> [...]
>>> +
>>> +static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev,
>>> ZpciFib fib,
>>> + uintptr_t ra)
>>> +{
>>> + Error *err = NULL;
>>> + uint64_t pba = ldq_be_p(&fib.pba);
>>> + uint64_t pal = ldq_be_p(&fib.pal);
>>> + uint64_t g_iota = ldq_be_p(&fib.iota);
>>> + bool t = s390_pci_is_translation_enabled(g_iota);
>>> +
>>> + pba &= ~0xfff;
>>> + pal |= 0xfff;
>>> +
>>> + if (!s390_pci_ioat_validate(pbdev, pba, pal, g_iota, &err)) {
>>> + if (err) {
>>> + qemu_log_mask(LOG_GUEST_ERROR, "%s\n",
>>> error_get_pretty(err));
>>> + error_free(err);
>>> + }
>>
>> With that change, you won't need the if (err) check anymore.
>>
>> Thanks,
>> Matt
>
^ permalink raw reply [flat|nested] 26+ messages in thread* Re: [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function
2026-10-05 13:23 ` Konstantin Shkolnyy
@ 2026-10-05 13:46 ` Matthew Rosato
2026-10-05 16:20 ` Konstantin Shkolnyy
0 siblings, 1 reply; 26+ messages in thread
From: Matthew Rosato @ 2026-10-05 13:46 UTC (permalink / raw)
To: Konstantin Shkolnyy
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel
On 10/5/26 9:23 AM, Konstantin Shkolnyy wrote:
> On 261002 19:43, Konstantin Shkolnyy wrote:
>> On 261002 13:28, Matthew Rosato wrote:
>>>
>>>> -static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev,
>>>> ZpciFib fib,
>>>> - uintptr_t ra)
>>>> +bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
>>>> + uint64_t pal, uint64_t g_iota, Error
>>>> **errp)
>>>> {
>>>> - uint64_t pba = ldq_be_p(&fib.pba);
>>>> - uint64_t pal = ldq_be_p(&fib.pal);
>>>> - uint64_t g_iota = ldq_be_p(&fib.iota);
>>>> uint8_t dt = (g_iota >> 2) & 0x7;
>>>> bool t = s390_pci_is_translation_enabled(g_iota);
>>>> - pba &= ~0xfff;
>>>> - pal |= 0xfff;
>>>> if (pba > pal || pba < pbdev->zpci_fn.sdma || pal > pbdev-
>>>> >zpci_fn.edma) {
>>>> - s390_program_interrupt(env, PGM_OPERAND, ra);
>>>> - return -EINVAL;
>>>> + return false;
>>>
>>> This will now violate include/qapi/error.h because you're returning
>>> false but not setting errp:
>>>
>>> * - On success, the function should not touch *errp. On failure, it
>>> * should set a new error, e.g. with error_setg(errp, ...), or
>>> * propagate an existing one, e.g. with error_propagate(errp, ...).
>>>
>>> A simple solution would be to add a new message for this case.
>>
>> But the original reg_ioat() also didn't print any message in this
>> error branch. I had to assume it was intentional, and preserved the
>> behavior.
>>
>> If you believe a new message should be introduced, I can insert a
>> commit for that - what would be the explanation?
>
> Other code sites reporting PGM_OPERAND also don't generate error
> messages - it appears, the current policy is not to do that in the cases
> when the guest is at fault (gave invalid parameters). The case at hand
> falls in that category.
>
The other 2 cases in this very function, where we are providing an error
message, are also due to invalid guest parameters. We advertised the
necessary information to make a correct decision via CLP query payloads,
but the guest gave us bad input anyway.
The fact that these messages are in response to guest parameters is why
it made sense to switch to qemu_log_mask(LOG_GUEST_ERROR, ...) -- the
guest can control if the message pops by purposely setting bad inputs.
With that in mind, there's probably an argument for adding more messages
and/or trace events to other PGM_OPERAND paths for debug purposes since
these aren't things expected to fail on a normally-behaving guest. But
that's beyond the scope of this series.
For this series: you need to either add a message to this case or, if
you feel that strongly about not reporting a message for it, then
another solution would be to remove it from the s390_pci_ioat_validate()
function and check it in 2 places so that a message is not required and
then everything in s390_pci_ioat_validate() will generate an errp for
the false case, satisfying include/qapi/error.h.
Thanks,
Matt
^ permalink raw reply [flat|nested] 26+ messages in thread* Re: [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function
2026-10-05 13:46 ` Matthew Rosato
@ 2026-10-05 16:20 ` Konstantin Shkolnyy
0 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-10-05 16:20 UTC (permalink / raw)
To: Matthew Rosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel
On 261005 08:46, Matthew Rosato wrote:
> On 10/5/26 9:23 AM, Konstantin Shkolnyy wrote:
>> On 261002 19:43, Konstantin Shkolnyy wrote:
>>> On 261002 13:28, Matthew Rosato wrote:
>>>>
>>>>> -static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev,
>>>>> ZpciFib fib,
>>>>> - uintptr_t ra)
>>>>> +bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
>>>>> + uint64_t pal, uint64_t g_iota, Error
>>>>> **errp)
>>>>> {
>>>>> - uint64_t pba = ldq_be_p(&fib.pba);
>>>>> - uint64_t pal = ldq_be_p(&fib.pal);
>>>>> - uint64_t g_iota = ldq_be_p(&fib.iota);
>>>>> uint8_t dt = (g_iota >> 2) & 0x7;
>>>>> bool t = s390_pci_is_translation_enabled(g_iota);
>>>>> - pba &= ~0xfff;
>>>>> - pal |= 0xfff;
>>>>> if (pba > pal || pba < pbdev->zpci_fn.sdma || pal > pbdev-
>>>>>> zpci_fn.edma) {
>>>>> - s390_program_interrupt(env, PGM_OPERAND, ra);
>>>>> - return -EINVAL;
>>>>> + return false;
>>>>
>>>> This will now violate include/qapi/error.h because you're returning
>>>> false but not setting errp:
>>>>
>>>> * - On success, the function should not touch *errp. On failure, it
>>>> * should set a new error, e.g. with error_setg(errp, ...), or
>>>> * propagate an existing one, e.g. with error_propagate(errp, ...).
>>>>
>>>> A simple solution would be to add a new message for this case.
>>>
>>> But the original reg_ioat() also didn't print any message in this
>>> error branch. I had to assume it was intentional, and preserved the
>>> behavior.
>>>
>>> If you believe a new message should be introduced, I can insert a
>>> commit for that - what would be the explanation?
>>
>> Other code sites reporting PGM_OPERAND also don't generate error
>> messages - it appears, the current policy is not to do that in the cases
>> when the guest is at fault (gave invalid parameters). The case at hand
>> falls in that category.
>>
> The other 2 cases in this very function, where we are providing an error
> message, are also due to invalid guest parameters. We advertised the
> necessary information to make a correct decision via CLP query payloads,
> but the guest gave us bad input anyway.
OK, since we already have a precedent right here, I'll add the message
you've suggested in the earlier reply.
>
> The fact that these messages are in response to guest parameters is why
> it made sense to switch to qemu_log_mask(LOG_GUEST_ERROR, ...) -- the
> guest can control if the message pops by purposely setting bad inputs.
>
> With that in mind, there's probably an argument for adding more messages
> and/or trace events to other PGM_OPERAND paths for debug purposes since
> these aren't things expected to fail on a normally-behaving guest. But
> that's beyond the scope of this series.
>
> For this series: you need to either add a message to this case or, if
> you feel that strongly about not reporting a message for it, then
> another solution would be to remove it from the s390_pci_ioat_validate()
> function and check it in 2 places so that a message is not required and
> then everything in s390_pci_ioat_validate() will generate an errp for
> the false case, satisfying include/qapi/error.h.
>
> Thanks,
> Matt
^ permalink raw reply [flat|nested] 26+ messages in thread
* Re: [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function
2026-10-03 0:43 ` Konstantin Shkolnyy
2026-10-05 13:23 ` Konstantin Shkolnyy
@ 2026-10-05 13:30 ` Matthew Rosato
1 sibling, 0 replies; 26+ messages in thread
From: Matthew Rosato @ 2026-10-05 13:30 UTC (permalink / raw)
To: Konstantin Shkolnyy
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel
On 10/2/26 8:43 PM, Konstantin Shkolnyy wrote:
> On 261002 13:28, Matthew Rosato wrote:
>>
>>> -static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev,
>>> ZpciFib fib,
>>> - uintptr_t ra)
>>> +bool s390_pci_ioat_validate(S390PCIBusDevice *pbdev, uint64_t pba,
>>> + uint64_t pal, uint64_t g_iota, Error
>>> **errp)
>>> {
>>> - uint64_t pba = ldq_be_p(&fib.pba);
>>> - uint64_t pal = ldq_be_p(&fib.pal);
>>> - uint64_t g_iota = ldq_be_p(&fib.iota);
>>> uint8_t dt = (g_iota >> 2) & 0x7;
>>> bool t = s390_pci_is_translation_enabled(g_iota);
>>> - pba &= ~0xfff;
>>> - pal |= 0xfff;
>>> if (pba > pal || pba < pbdev->zpci_fn.sdma || pal > pbdev-
>>> >zpci_fn.edma) {
>>> - s390_program_interrupt(env, PGM_OPERAND, ra);
>>> - return -EINVAL;
>>> + return false;
>>
>> This will now violate include/qapi/error.h because you're returning
>> false but not setting errp:
>>
>> * - On success, the function should not touch *errp. On failure, it
>> * should set a new error, e.g. with error_setg(errp, ...), or
>> * propagate an existing one, e.g. with error_propagate(errp, ...).
>>
>> A simple solution would be to add a new message for this case.
>
> But the original reg_ioat() also didn't print any message in this error
> branch. I had to assume it was intentional, and preserved the behavior.
>
Yes, but the original path was not passing an Error *errp back to a
caller, it was calling error_report (and then qemu_log_mask) directly;
your patch is changing this to a model where we return an Error *errp to
a caller. Which in and of itself is totally fine, but we should make
efforts to then adhere to the rules laid out in include/qapi/error.h
where all failure paths should have an error when passing back an errp.
> If you believe a new message should be introduced, I can insert a commit
> for that - what would be the explanation?
Separate commit not needed, it only becomes necessary as part of this
change as described above.
As for a message, something like the following?
"ioat pba 0x%"PRIx64" pal 0x%"PRIx64" out of device dma range
[0x%"PRIx64" 0x%"PRIx64"]", pba, pal, pbdev->zpci_fn.sdma,
pbdev->zpci_fn.edma
>
>>
>> [...]
>>> +
>>> +static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev,
>>> ZpciFib fib,
>>> + uintptr_t ra)
>>> +{
>>> + Error *err = NULL;
>>> + uint64_t pba = ldq_be_p(&fib.pba);
>>> + uint64_t pal = ldq_be_p(&fib.pal);
>>> + uint64_t g_iota = ldq_be_p(&fib.iota);
>>> + bool t = s390_pci_is_translation_enabled(g_iota);
>>> +
>>> + pba &= ~0xfff;
>>> + pal |= 0xfff;
>>> +
>>> + if (!s390_pci_ioat_validate(pbdev, pba, pal, g_iota, &err)) {
>>> + if (err) {
>>> + qemu_log_mask(LOG_GUEST_ERROR, "%s\n",
>>> error_get_pretty(err));
>>> + error_free(err);
>>> + }
>>
>> With that change, you won't need the if (err) check anymore.
>>
>> Thanks,
>> Matt
>
^ permalink raw reply [flat|nested] 26+ messages in thread
* [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (13 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 14/16] s390x/pci: Factor ioat sanity checks into a separate function Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
2026-10-05 19:39 ` Matthew Rosato
2026-10-05 22:20 ` Farhan Ali
2026-09-30 14:52 ` [PATCH v11 16/16] s390x/pci: Create function to contain fmb_timer start Konstantin Shkolnyy
15 siblings, 2 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
Implement zPCI device state migration, consequently enabling migration
of VMs that have emulated PCI devices, whether virtio or not.
Migration is allowed for devices whose function handle has the
FH_SHM_EMUL bit set. For these devices QEMU will save and restore the
state of its zPCI emulator.
This will enable emulated PCI migration starting with s390-ccw-virtio-11.2.
Passthrough devices will continue to block migration.
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 371 ++++++++++++++++++++++++++++++-
hw/s390x/s390-pci-inst.c | 2 +-
hw/s390x/s390-virtio-ccw.c | 4 +
include/hw/s390x/s390-pci-bus.h | 9 +-
include/hw/s390x/s390-pci-inst.h | 1 +
5 files changed, 378 insertions(+), 9 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index 799e596fec..a4469f529f 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -26,6 +26,8 @@
#include "hw/pci/pci_bridge.h"
#include "hw/pci/msi.h"
#include "exec/cpu-common.h"
+#include "migration/blocker.h"
+#include "migration/vmstate.h"
#include "qemu/error-report.h"
#include "qemu/module.h"
#include "system/physmem.h"
@@ -927,6 +929,8 @@ static void set_pbdev_info(S390PCIBusDevice *pbdev)
pbdev->pci_group = s390_group_find(ZPCI_DEFAULT_FN_GRP);
}
+static const VMStateDescription s390_pcihost_vmstate;
+
static void s390_pcihost_realize(DeviceState *dev, Error **errp)
{
PCIBus *b;
@@ -961,6 +965,7 @@ static void s390_pcihost_realize(DeviceState *dev, Error **errp)
s390_pci_init_default_group();
css_register_io_adapters(CSS_IO_ADAPTER_PCI, true, false,
S390_ADAPTER_SUPPRESSIBLE, errp);
+ vmstate_register(VMSTATE_IF(dev), 0, &s390_pcihost_vmstate, s);
s390_pcihost_kvm_realize();
}
@@ -1139,12 +1144,47 @@ static int s390_pci_interp_plug(S390pciState *s, S390PCIBusDevice *pbdev)
return 0;
}
+static int s390_set_zpci_migration_blocker(S390PCIBusDevice *pbdev,
+ S390pciState *s, Error **errp)
+{
+ if (s->zpci_migr_enabled) {
+ return 0;
+ }
+ error_setg(&pbdev->zpci_migr_blocker,
+ "Migration blocked on this machine type by zPCI device "
+ "uid %d", pbdev->uid);
+ return migrate_add_blocker(&pbdev->zpci_migr_blocker, errp);
+}
+
+static void s390_clear_zpci_migration_blocker(S390PCIBusDevice *pbdev)
+{
+ migrate_del_blocker(&pbdev->zpci_migr_blocker);
+}
+
+static int s390_set_passthrough_migration_blocker(S390PCIBusDevice *pbdev,
+ Error **errp)
+{
+ if (pbdev->fh & FH_SHM_EMUL) {
+ return 0;
+ }
+ error_setg(&pbdev->passthrough_migr_blocker,
+ "Migration blocked by passthrough zPCI device uid %d",
+ pbdev->uid);
+ return migrate_add_blocker(&pbdev->passthrough_migr_blocker, errp);
+}
+
+static void s390_clear_passthrough_migration_blocker(S390PCIBusDevice *pbdev)
+{
+ migrate_del_blocker(&pbdev->passthrough_migr_blocker);
+}
+
static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *dev,
Error **errp)
{
S390pciState *s = S390_PCI_HOST_BRIDGE(hotplug_dev);
PCIDevice *pdev = NULL;
S390PCIBusDevice *pbdev = NULL;
+ bool auto_pbdev = false;
int rc;
if (object_dynamic_cast(OBJECT(dev), TYPE_PCI_BRIDGE)) {
@@ -1204,6 +1244,7 @@ static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *de
if (!pbdev) {
return;
}
+ auto_pbdev = true;
}
pbdev->pdev = pdev;
@@ -1223,7 +1264,7 @@ static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *de
if (rc) {
error_setg(errp, "Plug failed for zPCI device in "
"interpretation mode: %d", rc);
- return;
+ goto err_unlink_pbdev;
}
} else {
trace_s390_pcihost("zPCI interpretation missing");
@@ -1252,16 +1293,46 @@ static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *de
pbdev->rtr_avail = false;
}
+ if (s390_set_passthrough_migration_blocker(pbdev, errp) != 0) {
+ goto err_unlink_pbdev;
+ }
+
if (s390_pci_msix_init(pbdev) && !pbdev->interp) {
error_setg(errp, "MSI-X support is mandatory "
"in the S390 architecture");
- return;
+ s390_clear_passthrough_migration_blocker(pbdev);
+ goto err_unlink_pbdev;
}
if (dev->hotplugged) {
s390_pci_generate_plug_event(HP_EVENT_TO_CONFIGURED ,
pbdev->fh, pbdev->fid);
}
+ return;
+
+err_unlink_pbdev:
+ if (pbdev->pft == ZPCI_PFT_ISM) {
+ notifier_remove(&pbdev->shutdown_notifier);
+ }
+ if (pbdev->dma_limit) {
+ s390_pci_end_dma_count(s, pbdev->dma_limit);
+ pbdev->dma_limit = NULL;
+ }
+ pbdev->fh &= ~FH_MASK_SHM;
+ pbdev->iommu = NULL;
+ pbdev->pdev = NULL;
+ pbdev->state = ZPCI_FS_RESERVED;
+ if (auto_pbdev) {
+ QTAILQ_REMOVE(&s->zpci_devs, pbdev, link);
+ if (g_hash_table_lookup(s->zpci_table, &pbdev->idx) == pbdev) {
+ g_hash_table_remove(s->zpci_table, &pbdev->idx);
+ }
+ g_hash_table_destroy(pbdev->iotlb);
+ s390_clear_zpci_migration_blocker(pbdev);
+ qdev_unrealize(DEVICE(pbdev));
+ object_unparent(OBJECT(pbdev));
+ }
+ return;
} else if (object_dynamic_cast(OBJECT(dev), TYPE_S390_PCI_DEVICE)) {
pbdev = S390_PCI_DEVICE(dev);
@@ -1272,6 +1343,12 @@ static void s390_pcihost_plug(const HotplugHandler *hotplug_dev, DeviceState *de
NULL, g_free);
QTAILQ_INSERT_TAIL(&s->zpci_devs, pbdev, link);
g_hash_table_insert(s->zpci_table, &pbdev->idx, pbdev);
+ if (s390_set_zpci_migration_blocker(pbdev, s, errp) != 0) {
+ g_hash_table_remove(s->zpci_table, &pbdev->idx);
+ QTAILQ_REMOVE(&s->zpci_devs, pbdev, link);
+ g_hash_table_destroy(pbdev->iotlb);
+ return;
+ }
} else {
g_assert_not_reached();
}
@@ -1294,6 +1371,8 @@ static void s390_pcihost_unplug(const HotplugHandler *hotplug_dev, DeviceState *
return;
}
+ s390_clear_passthrough_migration_blocker(pbdev);
+
s390_pci_generate_plug_event(HP_EVENT_STANDBY_TO_RESERVED,
pbdev->fh, pbdev->fid);
bus = pci_get_bus(pci_dev);
@@ -1313,6 +1392,7 @@ static void s390_pcihost_unplug(const HotplugHandler *hotplug_dev, DeviceState *
s390_pci_end_dma_count(s, pbdev->dma_limit);
}
g_hash_table_destroy(pbdev->iotlb);
+ s390_clear_zpci_migration_blocker(pbdev);
qdev_unrealize(dev);
}
}
@@ -1417,6 +1497,16 @@ void s390_pci_ism_reset(void)
}
}
+static void s390_pci_clear_pending_sei(S390pciState *s)
+{
+ SeiContainer *sei_cont;
+
+ while ((sei_cont = QTAILQ_FIRST(&s->pending_sei))) {
+ QTAILQ_REMOVE(&s->pending_sei, sei_cont, link);
+ g_free(sei_cont);
+ }
+}
+
static void s390_pcihost_reset(DeviceState *dev)
{
S390pciState *s = S390_PCI_HOST_BRIDGE(dev);
@@ -1440,6 +1530,8 @@ static void s390_pcihost_reset(DeviceState *dev)
}
}
+ s390_pci_clear_pending_sei(s);
+
/*
* When resetting a PCI bridge, the assigned numbers are set to 0. So
* on every system reset, we also have to reassign numbers.
@@ -1448,6 +1540,109 @@ static void s390_pcihost_reset(DeviceState *dev)
pci_for_each_device_under_bus(bus, s390_pci_enumerate_bridge, s);
}
+/*
+ * TYPE_S390_PCI_HOST_BRIDGE device state migration is registered via
+ * vmstate_register() rather than dc->vmsd because dc->vmsd is already assigned
+ * by our base, TYPE_PCI_HOST_BRIDGE, to &vmstate_pcihost which migrates
+ * PCIHostState.config_reg. There is no mechanism to add a subclass vmsd to the
+ * parent's.
+ */
+static bool s390_pcihost_vmstate_needed(void *opaque)
+{
+ S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
+ return s->zpci_migr_enabled;
+}
+
+static bool s390_pcihost_pending_sei_vmstate_needed(void *opaque)
+{
+ S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
+ return s->zpci_migr_enabled && !QTAILQ_EMPTY(&s->pending_sei);
+}
+
+/* Per-element descriptor for pending_sei list */
+static const VMStateDescription vmstate_sei_container = {
+ .name = "s390_sei_container",
+ .version_id = 1,
+ .minimum_version_id = 1,
+ .fields = (const VMStateField[]) {
+ VMSTATE_UINT32(fid, SeiContainer),
+ VMSTATE_UINT32(fh, SeiContainer),
+ VMSTATE_UINT8(cc, SeiContainer),
+ VMSTATE_UINT16(pec, SeiContainer),
+ VMSTATE_UINT64(faddr, SeiContainer),
+ VMSTATE_UINT32(e, SeiContainer),
+ VMSTATE_END_OF_LIST()
+ }
+};
+
+/*
+ * When pending_sei load is executed, pending_sei can already contain SEIs
+ * generated here in the target QEMU, for example, by state loads of zpci
+ * devices. A dumb load here would place the older SEIs from the source QEMU
+ * after the new SEIs. To fix this, we have pre_load() stash the new SEIs from
+ * pending_sei and post_load() unstash them after the old ones.
+ */
+static void s390_pci_move_sei_list(SeiContainerList *dst, SeiContainerList *src)
+{
+ SeiContainer *sei_cont;
+
+ while ((sei_cont = QTAILQ_FIRST(src))) {
+ QTAILQ_REMOVE(src, sei_cont, link);
+ QTAILQ_INSERT_TAIL(dst, sei_cont, link);
+ }
+}
+
+static int s390_pcihost_pending_sei_vmstate_pre_load(void *opaque)
+{
+ S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
+
+ QTAILQ_INIT(&s->pending_sei_stash);
+ s390_pci_move_sei_list(&s->pending_sei_stash, &s->pending_sei);
+ return 0;
+}
+
+static int s390_pcihost_pending_sei_vmstate_post_load(void *opaque,
+ int version_id)
+{
+ S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
+
+ s390_pci_move_sei_list(&s->pending_sei, &s->pending_sei_stash);
+ return 0;
+}
+
+static const VMStateDescription s390_pcihost_pending_sei_vmstate = {
+ .name = TYPE_S390_PCI_HOST_BRIDGE "/pending-sei",
+ .version_id = 1,
+ .minimum_version_id = 1,
+ .needed = s390_pcihost_pending_sei_vmstate_needed,
+ .pre_load = s390_pcihost_pending_sei_vmstate_pre_load,
+ .post_load = s390_pcihost_pending_sei_vmstate_post_load,
+ .fields = (const VMStateField[]) {
+ VMSTATE_QTAILQ_V(pending_sei, S390pciState, 1,
+ vmstate_sei_container, SeiContainer, link),
+ VMSTATE_END_OF_LIST()
+ }
+};
+
+static const VMStateDescription s390_pcihost_vmstate = {
+ .name = TYPE_S390_PCI_HOST_BRIDGE,
+ .version_id = 1,
+ .minimum_version_id = 1,
+ .needed = s390_pcihost_vmstate_needed,
+ .fields = (const VMStateField[]) {
+ VMSTATE_END_OF_LIST()
+ },
+ .subsections = (const VMStateDescription * const []) {
+ &s390_pcihost_pending_sei_vmstate,
+ NULL
+ }
+};
+
+static const Property phb_props[] = {
+ DEFINE_PROP_BOOL("x-zpci-migr-enabled", S390pciState,
+ zpci_migr_enabled, true),
+};
+
static void s390_pcihost_class_init(ObjectClass *klass, const void *data)
{
DeviceClass *dc = DEVICE_CLASS(klass);
@@ -1461,6 +1656,7 @@ static void s390_pcihost_class_init(ObjectClass *klass, const void *data)
hc->unplug_request = s390_pcihost_unplug_request;
hc->unplug = s390_pcihost_unplug;
msi_nonbroken = true;
+ device_class_set_props(dc, phb_props);
}
static const TypeInfo s390_pcihost_info = {
@@ -1474,10 +1670,33 @@ static const TypeInfo s390_pcihost_info = {
}
};
+/* Return a unique bus "path" for zpci device */
+static char *s390_pci_bus_get_dev_path(DeviceState *dev)
+{
+ S390PCIBusDevice *pbdev = S390_PCI_DEVICE(dev);
+ return g_strdup_printf("uid-%04x", pbdev->uid);
+}
+
+static void s390_pcibus_class_init(ObjectClass *oc, const void *data)
+{
+ BusClass *bc = BUS_CLASS(oc);
+ bc->get_dev_path = s390_pci_bus_get_dev_path;
+}
+
static const TypeInfo s390_pcibus_info = {
.name = TYPE_S390_PCI_BUS,
.parent = TYPE_BUS,
.instance_size = sizeof(S390PCIBus),
+ /*
+ * Implement get_dev_path() to provide each zpci device with a unique
+ * stable UID-based bus "path". The "path" is used as part of idstr in the
+ * migration stream, making idstr unique and instance_id always 0.
+ * For migration to succeed, (idstr+instance_id) must match those generated
+ * during QEMU start. Without unique idstr, QEMU will generate variable
+ * instance_id to distinguish devices, and that instance_id can change
+ * if a device is unplugged and plugged back, preventing migration.
+ */
+ .class_init = s390_pcibus_class_init,
};
static uint16_t s390_pci_generate_uid(S390pciState *s)
@@ -1623,13 +1842,151 @@ static const Property s390_pci_device_properties[] = {
true),
};
-static const VMStateDescription s390_pci_device_vmstate = {
- .name = TYPE_S390_PCI_DEVICE,
+static int s390_pci_device_pre_load(void *opaque)
+{
+ S390PCIBusDevice *pbdev = S390_PCI_DEVICE(opaque);
+ S390PCIBusDevice *found_pbdev;
+
/*
- * TODO: add state handling here, so migration works at least with
- * emulated pci devices on s390x
+ * Because state loading can change pbdev->idx make sure pbdev is removed
+ * from the table before that happens. The table type used stores a pointer
+ * to pbdev->idx and becomes corrupt if idx is changed from outside. But be
+ * careful to not remove instead another pbdev whose state might have been
+ * loaded earlier and that got assigned this idx value and had therefore
+ * already replaced our pbdev in the table. post_load() will reinsert our
+ * pbdev into the table. (found_pbdev could even be 0 if a previous
+ * vmstate_load_vmsd() on this device failed before reaching post_load().)
*/
- .unmigratable = 1,
+ found_pbdev = g_hash_table_lookup(s390_get_phb()->zpci_table, &pbdev->idx);
+ if (found_pbdev == pbdev) {
+ g_hash_table_remove(s390_get_phb()->zpci_table, &pbdev->idx);
+ }
+
+ return 0;
+}
+
+static bool s390_pci_device_post_load_errp(void *opaque, int version_id,
+ Error **errp)
+{
+ S390PCIBusDevice *pbdev = S390_PCI_DEVICE(opaque);
+
+ /*
+ * Guard against the scenario that s390_pcihost_plug() of the target PCI
+ * device succeeded in the source QEMU, but failed on this destination
+ * QEMU, before migration state load. In this case we'll find !pbdev->pdev
+ * but the pbdev->state != ZPCI_FS_RESERVED as just loaded from the stream.
+ * Such value combination is invalid and migration should fail.
+ */
+ if (pbdev->state != ZPCI_FS_RESERVED && !pbdev->pdev) {
+ error_setg(errp, "zpci device uid 0x%x state %d has no PCI device",
+ pbdev->uid, pbdev->state);
+ return false;
+ }
+
+ pbdev->zpci_fn.fid = pbdev->fid;
+ pbdev->zpci_fn.uid = pbdev->uid;
+
+ /*
+ * Now that pbdev->idx has been loaded, use it to place pbdev back into
+ * the table. This may replace a different not-yet-state-loaded pbdev,
+ * but pre_load() handles this case.
+ */
+ g_hash_table_replace(s390_get_phb()->zpci_table, &pbdev->idx, pbdev);
+
+ /*
+ * Regenerate IOMMU state, including IOTLB contents and QEMU memory regions.
+ */
+ if (pbdev->iommu_enabled) {
+ if (!pbdev->iommu) {
+ error_setg(errp, "iommu is NULL");
+ return false;
+ }
+ if (!s390_pci_ioat_validate(pbdev, pbdev->pba, pbdev->pal,
+ pbdev->g_iota, errp)) {
+ if (*errp) {
+ error_prepend(errp,
+ "invalid pba, pal or g_iota in migration stream: ");
+ } else {
+ error_setg(errp,
+ "invalid pba, pal or g_iota in migration stream");
+ }
+ return false;
+ }
+ if (s390_pci_is_translation_enabled(pbdev->g_iota)) {
+ s390_pci_iommu_enable(pbdev);
+ s390_pci_ioat_replay(pbdev);
+ } else {
+ /* TODO: unreachable until vfio passthrough migration is enabled */
+ s390_pci_iommu_direct_map_enable(pbdev);
+ }
+ }
+
+ /*
+ * Guest sets fmb_addr by mpcifc.ZPCI_MOD_FC_SET_MEASURE instruction,
+ * whose handler consequently starts fmb_timer. We may need to restart it.
+ */
+ if (pbdev->fmb_addr) {
+ if (pbdev->fmb_timer) {
+ error_setg(errp, "fmb_timer is not NULL");
+ return false;
+ }
+ if (!pbdev->pci_group) {
+ error_setg(errp, "pci_group is NULL");
+ return false;
+ }
+ pbdev->fmb_timer = timer_new_ms(QEMU_CLOCK_VIRTUAL,
+ fmb_update, pbdev);
+ timer_mod(pbdev->fmb_timer,
+ qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL) +
+ pbdev->pci_group->zpci_group.mui);
+ }
+ return true;
+}
+
+static const VMStateDescription s390_pci_device_vmstate = {
+ .name = TYPE_S390_PCI_DEVICE,
+ .version_id = 1,
+ .minimum_version_id = 1,
+ .priority = MIG_PRI_IOMMU,
+ .pre_load = s390_pci_device_pre_load,
+ .post_load_errp = s390_pci_device_post_load_errp,
+ .fields = (const VMStateField[]) {
+ VMSTATE_UINT32(state, S390PCIBusDevice),
+ VMSTATE_UINT16(uid, S390PCIBusDevice),
+ VMSTATE_UINT32(idx, S390PCIBusDevice),
+ VMSTATE_UINT32(fh, S390PCIBusDevice),
+ VMSTATE_UINT32(fid, S390PCIBusDevice),
+ VMSTATE_BOOL(fid_defined, S390PCIBusDevice),
+ VMSTATE_UINT64(fmb_addr, S390PCIBusDevice),
+ VMSTATE_UINT32(fmb.format, S390PCIBusDevice),
+ VMSTATE_UINT32(fmb.sample, S390PCIBusDevice),
+ VMSTATE_UINT64(fmb.last_update, S390PCIBusDevice),
+ VMSTATE_UINT64_ARRAY(fmb.counter, S390PCIBusDevice,
+ ARRAY_SIZE(((S390PCIBusDevice *)0)->fmb.counter)),
+ VMSTATE_UINT64(fmb.fmt0.dma_rbytes, S390PCIBusDevice),
+ VMSTATE_UINT64(fmb.fmt0.dma_wbytes, S390PCIBusDevice),
+ VMSTATE_UINT8(isc, S390PCIBusDevice),
+ VMSTATE_UINT16(noi, S390PCIBusDevice),
+ VMSTATE_UINT8(sum, S390PCIBusDevice),
+ VMSTATE_UINT8(pft, S390PCIBusDevice),
+ VMSTATE_UINT64(routes.adapter.ind_addr, S390PCIBusDevice),
+ VMSTATE_UINT64(routes.adapter.summary_addr, S390PCIBusDevice),
+ VMSTATE_UINT64(routes.adapter.ind_offset, S390PCIBusDevice),
+ VMSTATE_UINT32(routes.adapter.summary_offset, S390PCIBusDevice),
+ VMSTATE_UINT32(routes.adapter.adapter_id, S390PCIBusDevice),
+ VMSTATE_BOOL(iommu_enabled, S390PCIBusDevice),
+ VMSTATE_UINT64(g_iota, S390PCIBusDevice),
+ VMSTATE_UINT64(pba, S390PCIBusDevice),
+ VMSTATE_UINT64(pal, S390PCIBusDevice),
+ VMSTATE_PTR_TO_IND_ADDR(summary_ind, S390PCIBusDevice),
+ VMSTATE_PTR_TO_IND_ADDR(indicator, S390PCIBusDevice),
+ VMSTATE_BOOL(unplug_requested, S390PCIBusDevice),
+ VMSTATE_BOOL(interp, S390PCIBusDevice),
+ VMSTATE_BOOL(forwarding_assist, S390PCIBusDevice),
+ VMSTATE_BOOL(aif, S390PCIBusDevice),
+ VMSTATE_BOOL(rtr_avail, S390PCIBusDevice),
+ VMSTATE_END_OF_LIST()
+ }
};
static void s390_pci_device_class_init(ObjectClass *klass, const void *data)
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index b0af15ff15..c3bc5ee72b 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -1154,7 +1154,7 @@ static int fmb_do_update(S390PCIBusDevice *pbdev, int offset, uint64_t val,
return ret;
}
-static void fmb_update(void *opaque)
+void fmb_update(void *opaque)
{
S390PCIBusDevice *pbdev = opaque;
int64_t t = qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL);
diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c
index 9e2db872fb..fda5ecc9ad 100644
--- a/hw/s390x/s390-virtio-ccw.c
+++ b/hw/s390x/s390-virtio-ccw.c
@@ -1018,12 +1018,16 @@ static void ccw_machine_11_1_instance_options(MachineState *machine)
static void ccw_machine_11_1_class_options(MachineClass *mc)
{
S390CcwMachineClass *s390mc = S390_CCW_MACHINE_CLASS(mc);
+ static GlobalProperty compat[] = {
+ { TYPE_S390_PCI_HOST_BRIDGE, "x-zpci-migr-enabled", "off" },
+ };
s390mc->use_certs = false;
s390mc->use_secure = false;
ccw_machine_11_2_class_options(mc);
compat_props_add(mc->compat_props, hw_compat_11_1, hw_compat_11_1_len);
+ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat));
}
DEFINE_CCW_MACHINE(11, 1);
diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h
index 5c2d07e38e..71f956700b 100644
--- a/include/hw/s390x/s390-pci-bus.h
+++ b/include/hw/s390x/s390-pci-bus.h
@@ -338,6 +338,8 @@ struct S390PCIBusDevice {
uint16_t uid;
uint32_t idx;
uint32_t fh;
+ Error *zpci_migr_blocker; /* machines 11.1 or older */
+ Error *passthrough_migr_blocker;
uint32_t fid;
bool fid_defined;
uint64_t fmb_addr;
@@ -379,6 +381,8 @@ struct S390PCIBus {
BusState qbus;
};
+typedef QTAILQ_HEAD(SeiContainerList, SeiContainer) SeiContainerList;
+
struct S390pciState {
PCIHostState parent_obj;
uint32_t next_idx;
@@ -386,11 +390,14 @@ struct S390pciState {
S390PCIBus *bus;
GHashTable *iommu_table;
GHashTable *zpci_table;
- QTAILQ_HEAD(, SeiContainer) pending_sei;
+ SeiContainerList pending_sei;
+ /* Only used temporarily between migration pre_load and post_load. */
+ SeiContainerList pending_sei_stash;
QTAILQ_HEAD(, S390PCIBusDevice) zpci_devs;
QTAILQ_HEAD(, S390PCIDMACount) zpci_dma_limit;
QTAILQ_HEAD(, S390PCIGroup) zpci_groups;
uint8_t next_sim_grp;
+ bool zpci_migr_enabled;
};
S390pciState *s390_get_phb(void);
diff --git a/include/hw/s390x/s390-pci-inst.h b/include/hw/s390x/s390-pci-inst.h
index 3493d2ced0..e31497069b 100644
--- a/include/hw/s390x/s390-pci-inst.h
+++ b/include/hw/s390x/s390-pci-inst.h
@@ -113,6 +113,7 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
uintptr_t ra);
void fmb_timer_free(S390PCIBusDevice *pbdev);
+void fmb_update(void *opaque);
uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev, S390IOTLBEntry *entry);
#define ZPCI_IO_BAR_MIN 0
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread* Re: [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices
2026-09-30 14:52 ` [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
@ 2026-10-05 19:39 ` Matthew Rosato
2026-10-05 22:20 ` Farhan Ali
1 sibling, 0 replies; 26+ messages in thread
From: Matthew Rosato @ 2026-10-05 19:39 UTC (permalink / raw)
To: Konstantin Shkolnyy
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel
> +/* Return a unique bus "path" for zpci device */
> +static char *s390_pci_bus_get_dev_path(DeviceState *dev)
> +{
> + S390PCIBusDevice *pbdev = S390_PCI_DEVICE(dev);
> + return g_strdup_printf("uid-%04x", pbdev->uid);
> +}
> +
> +static void s390_pcibus_class_init(ObjectClass *oc, const void *data)
> +{
> + BusClass *bc = BUS_CLASS(oc);
> + bc->get_dev_path = s390_pci_bus_get_dev_path;
> +}
> +
> static const TypeInfo s390_pcibus_info = {
> .name = TYPE_S390_PCI_BUS,
> .parent = TYPE_BUS,
> .instance_size = sizeof(S390PCIBus),
> + /*
> + * Implement get_dev_path() to provide each zpci device with a unique
> + * stable UID-based bus "path". The "path" is used as part of idstr in the
> + * migration stream, making idstr unique and instance_id always 0.
> + * For migration to succeed, (idstr+instance_id) must match those generated
> + * during QEMU start. Without unique idstr, QEMU will generate variable
> + * instance_id to distinguish devices, and that instance_id can change
> + * if a device is unplugged and plugged back, preventing migration.
> + */
> + .class_init = s390_pcibus_class_init,
This patch is already pretty big. I wonder if adding
s390_pci_bus_get_dev_path + setting get_dev_path could be its own patch
prior to this one - up to you.
[...]
> +static bool s390_pci_device_post_load_errp(void *opaque, int version_id,
> + Error **errp)
> +{
> + S390PCIBusDevice *pbdev = S390_PCI_DEVICE(opaque);
> +
> + /*
> + * Guard against the scenario that s390_pcihost_plug() of the target PCI
> + * device succeeded in the source QEMU, but failed on this destination
> + * QEMU, before migration state load. In this case we'll find !pbdev->pdev
> + * but the pbdev->state != ZPCI_FS_RESERVED as just loaded from the stream.
> + * Such value combination is invalid and migration should fail.
> + */
> + if (pbdev->state != ZPCI_FS_RESERVED && !pbdev->pdev) {
> + error_setg(errp, "zpci device uid 0x%x state %d has no PCI device",
> + pbdev->uid, pbdev->state);
> + return false;
> + }
> +
> + pbdev->zpci_fn.fid = pbdev->fid;
> + pbdev->zpci_fn.uid = pbdev->uid;
> +
> + /*
> + * Now that pbdev->idx has been loaded, use it to place pbdev back into
> + * the table. This may replace a different not-yet-state-loaded pbdev,
> + * but pre_load() handles this case.
> + */
> + g_hash_table_replace(s390_get_phb()->zpci_table, &pbdev->idx, pbdev);
> +
> + /*
> + * Regenerate IOMMU state, including IOTLB contents and QEMU memory regions.
> + */
> + if (pbdev->iommu_enabled) {
> + if (!pbdev->iommu) {
> + error_setg(errp, "iommu is NULL");
> + return false;
> + }
> + if (!s390_pci_ioat_validate(pbdev, pbdev->pba, pbdev->pal,
> + pbdev->g_iota, errp)) {
> + if (*errp) {
> + error_prepend(errp,
> + "invalid pba, pal or g_iota in migration stream: ");
> + } else {
> + error_setg(errp,
> + "invalid pba, pal or g_iota in migration stream");
> + }
> + return false;
> + }
> + if (s390_pci_is_translation_enabled(pbdev->g_iota)) {
> + s390_pci_iommu_enable(pbdev);
> + s390_pci_ioat_replay(pbdev);
> + } else {
> + /* TODO: unreachable until vfio passthrough migration is enabled */
> + s390_pci_iommu_direct_map_enable(pbdev);
I would further clarify that this is because rtr_avail is always set to
false for emulated devices today.
Also, besides vfio migration, it could also be reachable if were to ever
allow rtr_avail for emulated devices.
But I wonder: if we can never reach this code today, should this then be
a g_assert_not_reached() and improve the comment above to indicate that
if/when either emulated devices are allowed to set rtr_avail or
migration of VFIO devices are added, this path needs to be updated to
call s390_pci_iommu_direct_map_enable(pbdev)?
If/when either of those things happen in the future it would already
need to be on a machine-version boundary whether this call were in-place
or not. And then we don't worry about devices driving this codepath
that should not be able to do so.
[...]
> struct S390pciState {
> PCIHostState parent_obj;
> uint32_t next_idx;
> @@ -386,11 +390,14 @@ struct S390pciState {
> S390PCIBus *bus;
> GHashTable *iommu_table;
> GHashTable *zpci_table;
> - QTAILQ_HEAD(, SeiContainer) pending_sei;
> + SeiContainerList pending_sei;
> + /* Only used temporarily between migration pre_load and post_load. */
> + SeiContainerList pending_sei_stash;
Hrm, I don't love that -- but I can't think of a better solution either.
I guess if we find anything else that needs this kind of treatment in
the future we should create a single object chained off of S390pciState
to hold all of them vs adding more of these to S390pciState.
Thanks,
Matt
^ permalink raw reply [flat|nested] 26+ messages in thread* Re: [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices
2026-09-30 14:52 ` [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
2026-10-05 19:39 ` Matthew Rosato
@ 2026-10-05 22:20 ` Farhan Ali
2026-10-05 22:31 ` Konstantin Shkolnyy
1 sibling, 1 reply; 26+ messages in thread
From: Farhan Ali @ 2026-10-05 22:20 UTC (permalink / raw)
To: Konstantin Shkolnyy, mjrosato
Cc: farman, richard.henderson, iii, david, cohuck, pasic, borntraeger,
qemu-s390x, qemu-devel
<..snip..>
On 9/30/2026 7:52 AM, Konstantin Shkolnyy wrote:
> +/*
> + * When pending_sei load is executed, pending_sei can already contain SEIs
> + * generated here in the target QEMU, for example, by state loads of zpci
> + * devices. A dumb load here would place the older SEIs from the source QEMU
> + * after the new SEIs. To fix this, we have pre_load() stash the new SEIs from
> + * pending_sei and post_load() unstash them after the old ones.
> + */
> +static void s390_pci_move_sei_list(SeiContainerList *dst, SeiContainerList *src)
> +{
> + SeiContainer *sei_cont;
> +
> + while ((sei_cont = QTAILQ_FIRST(src))) {
> + QTAILQ_REMOVE(src, sei_cont, link);
> + QTAILQ_INSERT_TAIL(dst, sei_cont, link);
> + }
> +}
> +
> +static int s390_pcihost_pending_sei_vmstate_pre_load(void *opaque)
> +{
> + S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
> +
> + QTAILQ_INIT(&s->pending_sei_stash);
> + s390_pci_move_sei_list(&s->pending_sei_stash, &s->pending_sei);
> + return 0;
> +}
Does this mean we can append SEIs to the list while we are still
migrating? In that case do we need any form of synchronization here when
accessing the pending_sei list?
Thanks
Farhan
> +
> +static int s390_pcihost_pending_sei_vmstate_post_load(void *opaque,
> + int version_id)
> +{
> + S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
> +
> + s390_pci_move_sei_list(&s->pending_sei, &s->pending_sei_stash);
> + return 0;
> +}
> +
> +static const VMStateDescription s390_pcihost_pending_sei_vmstate = {
> + .name = TYPE_S390_PCI_HOST_BRIDGE "/pending-sei",
> + .version_id = 1,
> + .minimum_version_id = 1,
> + .needed = s390_pcihost_pending_sei_vmstate_needed,
> + .pre_load = s390_pcihost_pending_sei_vmstate_pre_load,
> + .post_load = s390_pcihost_pending_sei_vmstate_post_load,
> + .fields = (const VMStateField[]) {
> + VMSTATE_QTAILQ_V(pending_sei, S390pciState, 1,
> + vmstate_sei_container, SeiContainer, link),
> + VMSTATE_END_OF_LIST()
> + }
> +};
> +
^ permalink raw reply [flat|nested] 26+ messages in thread* Re: [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices
2026-10-05 22:20 ` Farhan Ali
@ 2026-10-05 22:31 ` Konstantin Shkolnyy
0 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-10-05 22:31 UTC (permalink / raw)
To: Farhan Ali, mjrosato
Cc: farman, richard.henderson, iii, david, cohuck, pasic, borntraeger,
qemu-s390x, qemu-devel
On 261005 17:20, Farhan Ali wrote:
> <..snip..>
>
> On 9/30/2026 7:52 AM, Konstantin Shkolnyy wrote:
>> +/*
>> + * When pending_sei load is executed, pending_sei can already contain
>> SEIs
>> + * generated here in the target QEMU, for example, by state loads of
>> zpci
>> + * devices. A dumb load here would place the older SEIs from the
>> source QEMU
>> + * after the new SEIs. To fix this, we have pre_load() stash the new
>> SEIs from
>> + * pending_sei and post_load() unstash them after the old ones.
>> + */
>> +static void s390_pci_move_sei_list(SeiContainerList *dst,
>> SeiContainerList *src)
>> +{
>> + SeiContainer *sei_cont;
>> +
>> + while ((sei_cont = QTAILQ_FIRST(src))) {
>> + QTAILQ_REMOVE(src, sei_cont, link);
>> + QTAILQ_INSERT_TAIL(dst, sei_cont, link);
>> + }
>> +}
>> +
>> +static int s390_pcihost_pending_sei_vmstate_pre_load(void *opaque)
>> +{
>> + S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
>> +
>> + QTAILQ_INIT(&s->pending_sei_stash);
>> + s390_pci_move_sei_list(&s->pending_sei_stash, &s->pending_sei);
>> + return 0;
>> +}
>
> Does this mean we can append SEIs to the list while we are still
> migrating? In that case do we need any form of synchronization here when
> accessing the pending_sei list?
SEIs can be generated the migration process itself - by
s390_pci_ioat_replay() called from s390_pci_device_post_load_errp().
That happens before the migration process comes here to load
s390_pcihost state (with old SEIs).
Since the migration process is single-threaded I don't think we need to
synchronize between its stages.
>
> Thanks
>
> Farhan
>
>> +
>> +static int s390_pcihost_pending_sei_vmstate_post_load(void *opaque,
>> + int version_id)
>> +{
>> + S390pciState *s = S390_PCI_HOST_BRIDGE(opaque);
>> +
>> + s390_pci_move_sei_list(&s->pending_sei, &s->pending_sei_stash);
>> + return 0;
>> +}
>> +
>> +static const VMStateDescription s390_pcihost_pending_sei_vmstate = {
>> + .name = TYPE_S390_PCI_HOST_BRIDGE "/pending-sei",
>> + .version_id = 1,
>> + .minimum_version_id = 1,
>> + .needed = s390_pcihost_pending_sei_vmstate_needed,
>> + .pre_load = s390_pcihost_pending_sei_vmstate_pre_load,
>> + .post_load = s390_pcihost_pending_sei_vmstate_post_load,
>> + .fields = (const VMStateField[]) {
>> + VMSTATE_QTAILQ_V(pending_sei, S390pciState, 1,
>> + vmstate_sei_container, SeiContainer, link),
>> + VMSTATE_END_OF_LIST()
>> + }
>> +};
>> +
^ permalink raw reply [flat|nested] 26+ messages in thread
* [PATCH v11 16/16] s390x/pci: Create function to contain fmb_timer start
2026-09-30 14:52 [PATCH v11 00/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
` (14 preceding siblings ...)
2026-09-30 14:52 ` [PATCH v11 15/16] s390x/pci: Implement migration for emulated devices Konstantin Shkolnyy
@ 2026-09-30 14:52 ` Konstantin Shkolnyy
15 siblings, 0 replies; 26+ messages in thread
From: Konstantin Shkolnyy @ 2026-09-30 14:52 UTC (permalink / raw)
To: mjrosato
Cc: alifm, farman, richard.henderson, iii, david, cohuck, pasic,
borntraeger, qemu-s390x, qemu-devel, Konstantin Shkolnyy
fmb_timer is now started in 3 different places. The new function will
encapsulate that to make sure mui is added in all cases.
Reviewed-by: Matthew Rosato <mjrosato@linux.ibm.com>
Signed-off-by: Konstantin Shkolnyy <kshk@linux.ibm.com>
---
hw/s390x/s390-pci-bus.c | 5 ++---
hw/s390x/s390-pci-inst.c | 14 ++++++++++----
include/hw/s390x/s390-pci-inst.h | 1 +
3 files changed, 13 insertions(+), 7 deletions(-)
diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
index a4469f529f..e90d187e5f 100644
--- a/hw/s390x/s390-pci-bus.c
+++ b/hw/s390x/s390-pci-bus.c
@@ -1936,9 +1936,8 @@ static bool s390_pci_device_post_load_errp(void *opaque, int version_id,
}
pbdev->fmb_timer = timer_new_ms(QEMU_CLOCK_VIRTUAL,
fmb_update, pbdev);
- timer_mod(pbdev->fmb_timer,
- qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL) +
- pbdev->pci_group->zpci_group.mui);
+ s390_pci_schedule_fmb_timer(pbdev,
+ qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL));
}
return true;
}
diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c
index c3bc5ee72b..2a994bac43 100644
--- a/hw/s390x/s390-pci-inst.c
+++ b/hw/s390x/s390-pci-inst.c
@@ -1154,9 +1154,16 @@ static int fmb_do_update(S390PCIBusDevice *pbdev, int offset, uint64_t val,
return ret;
}
+void s390_pci_schedule_fmb_timer(S390PCIBusDevice *pbdev, uint64_t start)
+{
+ timer_mod(pbdev->fmb_timer, start + pbdev->pci_group->zpci_group.mui);
+}
+
void fmb_update(void *opaque)
{
S390PCIBusDevice *pbdev = opaque;
+
+ /* Must be read before updating U bit */
int64_t t = qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL);
int i;
@@ -1193,7 +1200,7 @@ void fmb_update(void *opaque)
sizeof(pbdev->fmb.last_update))) {
return;
}
- timer_mod(pbdev->fmb_timer, t + pbdev->pci_group->zpci_group.mui);
+ s390_pci_schedule_fmb_timer(pbdev, t);
}
static int mpcifc_reg_int_interp(S390PCIBusDevice *pbdev, ZpciFib *fib)
@@ -1386,9 +1393,8 @@ int mpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
timer_del(pbdev->fmb_timer);
}
pbdev->fmb_addr = fmb_addr;
- timer_mod(pbdev->fmb_timer,
- qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL) +
- pbdev->pci_group->zpci_group.mui);
+ s390_pci_schedule_fmb_timer(pbdev,
+ qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL));
break;
}
default:
diff --git a/include/hw/s390x/s390-pci-inst.h b/include/hw/s390x/s390-pci-inst.h
index e31497069b..0c27e5bcfb 100644
--- a/include/hw/s390x/s390-pci-inst.h
+++ b/include/hw/s390x/s390-pci-inst.h
@@ -114,6 +114,7 @@ int stpcifc_service_call(S390CPU *cpu, uint8_t r1, uint64_t fiba, uint8_t ar,
uintptr_t ra);
void fmb_timer_free(S390PCIBusDevice *pbdev);
void fmb_update(void *opaque);
+void s390_pci_schedule_fmb_timer(S390PCIBusDevice *pbdev, uint64_t start);
uint32_t s390_pci_update_iotlb(S390PCIBusDevice *pbdev, S390IOTLBEntry *entry);
#define ZPCI_IO_BAR_MIN 0
--
2.34.1
^ permalink raw reply related [flat|nested] 26+ messages in thread