Currently QEMU unconditionally stops VM upon receiving any unrecoverable error from the device. This prevents recovery of the device within the guest.
Upon receiving the AER error from device, an error is injected to guest if the device supports AER. If the device does not support AER, check whether the upstream PCIe bridge supports it and forward the error to the bridge. The error forward is gated with an optional extension parameter and can be enabled with -device vfio-pci,host=<BDF>,...,x-forward-aer=on Signed-off-by: Satyanarayana K V P <[email protected]> Cc: Michał Winiarski <[email protected]> Cc: Michal Wajdeczko <[email protected]> Cc: Matthew Brost <[email protected]> Cc: Alex Williamson <[email protected]> Cc: Cédric Le Goater <[email protected]> --- hw/vfio/pci.c | 80 ++++++++++++++++++++++++++++++++++++++++++++------- hw/vfio/pci.h | 1 + 2 files changed, 70 insertions(+), 11 deletions(-) diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c index 9c06b25e63..1913d18ff5 100644 --- a/hw/vfio/pci.c +++ b/hw/vfio/pci.c @@ -2628,6 +2628,14 @@ static void vfio_add_ext_cap(VFIOPCIDevice *vdev) pcie_add_capability(pdev, cap_id, cap_ver, next, size); } break; + case PCI_EXT_CAP_ID_ERR: + if (pcie_aer_init(pdev, cap_ver, next, size, &err) < 0) { + warn_report_err(err); + err = NULL; + /* Mark capability as absent on failure to initialize */ + pdev->exp.aer_cap = 0; + } + break; /* * VFIO kernel does not expose the PASID CAP today. We may synthesize * one later through IOMMUFD APIs. If VFIO ever starts exposing it, @@ -3124,26 +3132,68 @@ void vfio_pci_put_device(VFIOPCIDevice *vdev) g_free(vdev->msix); } -static void vfio_err_notifier_handler(void *opaque) +static void vfio_err_notify_guest(VFIOPCIDevice *vdev) { - VFIOPCIDevice *vdev = opaque; + PCIDevice *pdev = PCI_DEVICE(vdev); + PCIDevice *pbridge_dev; + PCIEAERErr aer_err = { + .status = PCI_ERR_UNC_MALF_TLP, + .flags = 0, + }; + int ret; - if (!event_notifier_test_and_clear(&vdev->err_notifier)) { - return; + if (!vdev->forward_aer) { + error_report("%s(%s)" + "Unrecoverable error detected and could not notify guest. " + "Please collect any data possible and then kill the guest", + __func__, vdev->vbasedev.name); + goto vm_stop_out; } /* - * TBD. Retrieve the error details and decide what action - * needs to be taken. One of the actions could be to pass - * the error to the guest and have the guest driver recover - * from the error. This requires that PCIe capabilities be - * exposed to the guest. For now, we just terminate the - * guest to contain the error. + * If the device does not have AER capability and still configured to + * forward error to guest, try to find a PCIe bridge with AER capability and + * forward the error to it. If neither the device nor the PCIe bridge supports + * AER, the VM is immediately stopped when the error is reported. */ + if (!pdev->exp.aer_cap) { + pbridge_dev = pci_bridge_get_device(pci_get_bus(pdev)); + if (pbridge_dev && pbridge_dev->exp.aer_cap) { + warn_report("Forwarding error to PCIe bridge due to lack of AER capability on device"); + pdev = pbridge_dev; + } else { + error_report("%s(%s) No AER capable device is found, stopping VM", + __func__, vdev->vbasedev.name); + goto vm_stop_out; + } + } + + aer_err.source_id = pci_requester_id(pdev); + + ret = pcie_aer_inject_error(pdev, &aer_err); + if (ret < 0) { + error_report("%s(%s): AER injection failed (%d)", __func__, vdev->vbasedev.name, ret); + goto vm_stop_out; + } - error_report("%s(%s) Unrecoverable error detected. Please collect any data possible and then kill the guest", __func__, vdev->vbasedev.name); + return; +vm_stop_out: vm_stop(RUN_STATE_INTERNAL_ERROR); + return; +} + +static void vfio_err_notifier_handler(void *opaque) +{ + VFIOPCIDevice *vdev = opaque; + + if (!event_notifier_test_and_clear(&vdev->err_notifier)) { + return; + } + + error_report("%s(%s) Unrecoverable error detected for the device", + __func__, vdev->vbasedev.name); + vfio_err_notify_guest(vdev); } /* @@ -3624,6 +3674,9 @@ static void vfio_exitfn(PCIDevice *pdev) VFIOPCIDevice *vdev = VFIO_PCI_DEVICE(pdev); VFIODevice *vbasedev = &vdev->vbasedev; + if (pdev->exp.aer_cap) { + pcie_aer_exit(pdev); + } vfio_unregister_req_notifier(vdev); vfio_unregister_err_notifier(vdev); pci_device_set_intx_routing_notifier(pdev, NULL); @@ -3815,6 +3868,7 @@ static const Property vfio_pci_properties[] = { DEFINE_PROP_BOOL("skip-vsc-check", VFIOPCIDevice, skip_vsc_check, true), DEFINE_PROP_UINT16("x-vpasid-cap-offset", VFIOPCIDevice, vpasid_cap_offset, 0), + DEFINE_PROP_BOOL("x-forward-aer", VFIOPCIDevice, forward_aer, false), }; static void vfio_pci_set_fd(Object *obj, const char *str, Error **errp) @@ -3973,6 +4027,10 @@ static void vfio_pci_class_init(ObjectClass *klass, const void *data) "a vIOMMU. A value of 0 (default) places the capability at the " "end of the extended configuration space. The offset must be " "4-byte aligned and within the PCIe extended configuration space"); + object_class_property_set_description(klass, + "x-forward-aer", + "Forward Advanced Error Reporting (AER) events to the guest " + "with generic error status"); } static const TypeInfo vfio_pci_info = { diff --git a/hw/vfio/pci.h b/hw/vfio/pci.h index c3a1f53d35..b1d19c4c9e 100644 --- a/hw/vfio/pci.h +++ b/hw/vfio/pci.h @@ -187,6 +187,7 @@ struct VFIOPCIDevice { bool defer_kvm_irq_routing; bool clear_parent_atomics_on_exit; bool skip_vsc_check; + bool forward_aer; uint16_t vpasid_cap_offset; VFIODisplay *dpy; Notifier irqchip_change_notifier; -- 2.43.0
