From: Farhan Ali <alifm@linux.ibm.com>
To: "Cédric Le Goater" <clg@redhat.com>,
qemu-devel@nongnu.org, qemu-s390x@nongnu.org
Cc: mjrosato@linux.ibm.com, farman@linux.ibm.com, cohuck@redhat.com,
alex@shazbot.org, armbru@redhat.com
Subject: Re: [PATCH v5 2/3] s390x/pci: Add PCI error handling for vfio pci devices
Date: Wed, 16 Sep 2026 10:52:53 -0700 [thread overview]
Message-ID: <6071779c-9323-4217-9b57-10353ee781b3@linux.ibm.com> (raw)
In-Reply-To: <b71cc588-f0f9-4c1e-9207-5ddd103a002e@redhat.com>
Hi Cedric,
On 9/16/2026 1:13 AM, Cédric Le Goater wrote:
> On 9/14/26 19:44, Farhan Ali wrote:
>> Add an s390x specific handler for vfio error notifier. For s390x pci
>> devices,
>> we have platform specific error information. We need to retrieve this
>> error
>> information for passthrough devices. This is done via a
>> VFIO_DEVICE_FEATURE
>> ioctl which exposes that information.
>>
>> Once this error information is retrieved we can then inject an error
>> into
>> the guest, and let the guest drive the recovery.
>>
>> Signed-off-by: Farhan Ali <alifm@linux.ibm.com>
>> ---
>> hw/s390x/s390-pci-bus.c | 6 ++
>> hw/s390x/s390-pci-vfio-stubs.c | 6 ++
>> hw/s390x/s390-pci-vfio.c | 115 +++++++++++++++++++++++++++++++
>> include/hw/s390x/s390-pci-bus.h | 1 +
>> include/hw/s390x/s390-pci-vfio.h | 1 +
>> 5 files changed, 129 insertions(+)
>>
>> diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c
>> index 2eb4e8cec4..b2967dacba 100644
>> --- a/hw/s390x/s390-pci-bus.c
>> +++ b/hw/s390x/s390-pci-bus.c
>> @@ -1085,6 +1085,7 @@ static void s390_pcihost_plug(const
>> HotplugHandler *hotplug_dev, DeviceState *de
>> S390pciState *s = S390_PCI_HOST_BRIDGE(hotplug_dev);
>> PCIDevice *pdev = NULL;
>> S390PCIBusDevice *pbdev = NULL;
>> + Error *local_err = NULL;
>> int rc;
>> if (object_dynamic_cast(OBJECT(dev), TYPE_PCI_BRIDGE)) {
>> @@ -1175,6 +1176,11 @@ static void s390_pcihost_plug(const
>> HotplugHandler *hotplug_dev, DeviceState *de
>> pbdev->iommu->dma_limit = s390_pci_start_dma_count(s,
>> pbdev);
>> /* Fill in CLP information passed via the vfio region */
>> s390_pci_get_clp_info(pbdev);
>> + /* Setup error handler for error recovery */
>> + if (!s390_pci_setup_err_handler(pbdev, &local_err)) {
>> + warn_report_err(local_err);
>> + }
>> +
>> if (!pbdev->interp) {
>> /* Do vfio passthrough but intercept for I/O */
>> pbdev->fh |= FH_SHM_VFIO;
>> diff --git a/hw/s390x/s390-pci-vfio-stubs.c
>> b/hw/s390x/s390-pci-vfio-stubs.c
>> index d9882b7aad..9fc84ca135 100644
>> --- a/hw/s390x/s390-pci-vfio-stubs.c
>> +++ b/hw/s390x/s390-pci-vfio-stubs.c
>> @@ -30,3 +30,9 @@ bool s390_pci_get_host_fh(S390PCIBusDevice *pbdev,
>> uint32_t *fh)
>> void s390_pci_get_clp_info(S390PCIBusDevice *pbdev)
>> {
>> }
>> +
>> +bool s390_pci_setup_err_handler(S390PCIBusDevice *pbdev, Error **errp)
>> +{
>> + error_setg(errp, "VFIO not available, cannot setup error handler");
>> + return false;
>> +}
>> diff --git a/hw/s390x/s390-pci-vfio.c b/hw/s390x/s390-pci-vfio.c
>> index db6de00bd2..6c072005fd 100644
>> --- a/hw/s390x/s390-pci-vfio.c
>> +++ b/hw/s390x/s390-pci-vfio.c
>> @@ -10,6 +10,7 @@
>> */
>> #include "qemu/osdep.h"
>> +#include "qemu/error-report.h"
>> #include <sys/ioctl.h>
>> #include <linux/vfio.h>
>> @@ -105,6 +106,85 @@ void s390_pci_end_dma_count(S390pciState *s,
>> S390PCIDMACount *cnt)
>> }
>> }
>> +static bool s390_pci_get_feature_err(VFIOPCIDevice *vfio_pci,
>> + PciCcdfErr *ccdf,
>> + uint32_t ccdf_err_length,
>> + Error **errp)
>> +{
>> + int ret;
>> + size_t total_size;
>> + struct vfio_device_feature_zpci_err *err;
>> + g_autofree void *buf = NULL;
>
> could be buf = g_malloc(ccdf_err_length);
okay, will change.
>
>
>> + g_autofree struct vfio_device_feature *feature = NULL;
>> +
>> + total_size = sizeof(*feature) + sizeof(*err);
>> + feature = g_malloc(total_size);
>> + feature->argsz = total_size;
>> + feature->flags = VFIO_DEVICE_FEATURE_GET |
>> VFIO_DEVICE_FEATURE_ZPCI_ERROR;
>> +
>> + buf = g_malloc(ccdf_err_length);
>> + err = (void *)feature->data;
>> + err->data = (uint64_t)buf;
>> + ret = vfio_device_get_feature(&vfio_pci->vbasedev, feature);
>> +
>> + if (ret) {
>> + if (ret != -ENOMSG) {
>> + error_setg(errp, "Failed feature get
>> VFIO_DEVICE_FEATURE_ZPCI_ERROR"
>> + " (rc=%d)", ret);
>> + }
>> + return false;
>
> returning false without setting errp :/
An ENOMSG would indicate there are no more pending errors to be handled
for the device. It doesn't indicate a critical error. So I don't think
we want to set the errp? I am open to suggestion on this, should errp be
set to warn in this case?
>
>> + }
>> +
>> + memcpy(ccdf, (PciCcdfErr *) err->data, ccdf_err_length);
>> +
>> + return true;
>> +}
>> +
>> +static void s390_pci_err_handler(void *opaque)
>> +{
>> + VFIOPCIDevice *vfio_pci;
>> + S390PCIBusDevice *pbdev;
>> + Error *errp = NULL;
>
> a 'local_err' name would be preferred.
okay, will change.
>
>> + PciCcdfErr ccdf;
>> + bool ret = true;
>> +
>> + vfio_pci = opaque;
>> + if (!event_notifier_test_and_clear(&vfio_pci->err_notifier)) {
>
> This means that the vfio_pci->err_notifier eventfd was initialized.
> IOW, pci_aer is true.
>
> Is that the case for Z ? If not, this needs its own eventfd setup.
Yes, it is. AFAIU the pci_aer is set if VFIO_PCI_ERR_IRQ_INDEX is
supported, which it is on Z. The recent kernel change [1] enables it on
all supported devices on Z.
[1]
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git/commit/?id=4e3c1fc8abcb8eff062150b4340fa4569696d645
>
>
>> + return;
>> + }
>> +
>> + pbdev = s390_pci_find_dev_by_target(s390_get_phb(),
>> + DEVICE(&vfio_pci->parent_obj)->id);
>> +
>> + if (!pbdev) {
>> + error_report("No matching zpci device found");
>> + return;
>> + }
>> + pbdev->state = ZPCI_FS_ERROR;
>> +
>> + if (sizeof(ccdf) != pbdev->ccdf_err_length) {
>> + error_report(
>> + "CCDF size mismatch expected size=%zu, provided
>> size=%d",
>> + sizeof(ccdf), pbdev->ccdf_err_length);
>> + return;
>> + }
>> +
>> + while (ret) {
>> + ret = s390_pci_get_feature_err(vfio_pci, &ccdf,
>> + pbdev->ccdf_err_length, &errp);
>> + if (!ret) {
>> + if (errp) {
>> + error_report_err(errp);
>> + }
>> + break;
>
> The 'local_err' not being set with a 'false' returned value is not
> following the qapi/error.h guidelines. This is unexpected.
> It is also wrong. ERRP_GUARD() is needed. please read qapi/error.h.
Should the ERRP_GUARD() be added here for local_err or in
s390_pci_get_feature_err()?
I am open to suggestions on how we can handle the case of ENOMSG which
is not a critical error.
>
>
>
>> + }
>> + s390_pci_generate_error_event(ccdf.pec, pbdev->fh, pbdev->fid,
>> + ccdf.faddr, ccdf.e);
>> + }
>> +
>> + return;
>> +}
>> +
>> static void s390_pci_read_base(S390PCIBusDevice *pbdev,
>> struct vfio_device_info *info)
>> {
>> @@ -134,6 +214,10 @@ static void s390_pci_read_base(S390PCIBusDevice
>> *pbdev,
>> /* Store function type separately for type-specific behavior */
>> pbdev->pft = cap->pft;
>> + if (hdr->version >= 3) {
>> + pbdev->ccdf_err_length = cap->ccdf_err_length;
>
> So ccdf_err_length can be 0. Is that expected ?
On kernels that don't support the vfio device feature, the kernel
doesn't provide ccdf_err_length. So in that case pbdev->ccdf_err_length
can be 0.
>
>> + }
>> +
>> /*
>> * If the device is a passthrough ISM device, disallow relaxed
>> * translation.
>> @@ -371,3 +455,34 @@ void s390_pci_get_clp_info(S390PCIBusDevice *pbdev)
>> s390_pci_read_util(pbdev, info);
>> s390_pci_read_pfip(pbdev, info);
>> }
>> +
>> +bool s390_pci_setup_err_handler(S390PCIBusDevice *pbdev, Error **errp)
>> +{
>> + int ret;
>> + int32_t fd;
>> + VFIOPCIDevice *vfio_pci = VFIO_PCI_DEVICE(pbdev->pdev);
>> + uint64_t buf[DIV_ROUND_UP(sizeof(struct vfio_device_feature),
>> + sizeof(uint64_t))] = {};
>> + struct vfio_device_feature *feature = (struct
>> vfio_device_feature *)buf;
>> +
>> + feature->argsz = sizeof(buf);
>> + feature->flags = VFIO_DEVICE_FEATURE_PROBE |
>> VFIO_DEVICE_FEATURE_ZPCI_ERROR;
>> +
>> + ret = vfio_device_get_feature(&vfio_pci->vbasedev, feature);
>> +
>> + if (ret != 0) {
>> + if (ret == -ENOTTY) {
>> + error_setg(errp, "Automated error recovery unavailable
>> for device");
>> + } else {
>> + error_setg(errp,
>> + "Failed to probe for
>> VFIO_DEVICE_FEATURE_ZPCI_ERROR (ret=%d)",
>> + ret);
>> + }
>> + return false;
>> + }
>> +
>> + fd = event_notifier_get_fd(&vfio_pci->err_notifier);
>> + qemu_set_fd_handler(fd, s390_pci_err_handler, NULL, vfio_pci);
>
> Shouldn't we check ccdf_err_length before installing the handler ?
> because
> it won't run cleanly anyhow.
>
> Thanks,
>
> C.
>
I can move the ccdf check before installing the handler.
Thanks
Farhan
>
>> +
>> + return true;
>> +}
>> diff --git a/include/hw/s390x/s390-pci-bus.h
>> b/include/hw/s390x/s390-pci-bus.h
>> index 9228523ce8..c2348ede86 100644
>> --- a/include/hw/s390x/s390-pci-bus.h
>> +++ b/include/hw/s390x/s390-pci-bus.h
>> @@ -364,6 +364,7 @@ struct S390PCIBusDevice {
>> bool forwarding_assist;
>> bool aif;
>> bool rtr_avail;
>> + uint32_t ccdf_err_length;
>> QTAILQ_ENTRY(S390PCIBusDevice) link;
>> };
>> diff --git a/include/hw/s390x/s390-pci-vfio.h
>> b/include/hw/s390x/s390-pci-vfio.h
>> index f7d6149daf..c7886b63ea 100644
>> --- a/include/hw/s390x/s390-pci-vfio.h
>> +++ b/include/hw/s390x/s390-pci-vfio.h
>> @@ -20,5 +20,6 @@ S390PCIDMACount
>> *s390_pci_start_dma_count(S390pciState *s,
>> void s390_pci_end_dma_count(S390pciState *s, S390PCIDMACount *cnt);
>> bool s390_pci_get_host_fh(S390PCIBusDevice *pbdev, uint32_t *fh);
>> void s390_pci_get_clp_info(S390PCIBusDevice *pbdev);
>> +bool s390_pci_setup_err_handler(S390PCIBusDevice *pbdev, Error **errp);
>> #endif
>
next prev parent reply other threads:[~2026-09-16 17:54 UTC|newest]
Thread overview: 9+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-14 17:44 [PATCH v5 0/3] Error recovery for zPCI passthrough devices Farhan Ali
2026-09-14 17:44 ` [PATCH v5 1/3] linux-headers: Update Linux header to 7.3-rc3 Farhan Ali
2026-09-15 22:32 ` Farhan Ali
2026-09-14 17:44 ` [PATCH v5 2/3] s390x/pci: Add PCI error handling for vfio pci devices Farhan Ali
2026-09-16 8:13 ` Cédric Le Goater
2026-09-16 17:52 ` Farhan Ali [this message]
2026-09-17 1:21 ` Matthew Rosato
2026-09-17 6:51 ` Cédric Le Goater
2026-09-14 17:44 ` [PATCH v5 3/3] s390x/pci: Reset a device in error state Farhan Ali
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=6071779c-9323-4217-9b57-10353ee781b3@linux.ibm.com \
--to=alifm@linux.ibm.com \
--cc=alex@shazbot.org \
--cc=armbru@redhat.com \
--cc=clg@redhat.com \
--cc=cohuck@redhat.com \
--cc=farman@linux.ibm.com \
--cc=mjrosato@linux.ibm.com \
--cc=qemu-devel@nongnu.org \
--cc=qemu-s390x@nongnu.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.