On Wed, 2026-09-09 at 15:28 +0800, Zhenzhong Duan wrote:
> Caution: External email. Do not open attachments or click links, unless this
> email comes from a known sender and you know the content is safe.
>
>
> When the guest enables the PRQ in vIOMMU, allocate a FAULTQ object so that
> host-side recoverable fault events can be received and propagated back to
> the guest.
>
> Install an event handler on the FAULTQ fd to read and propagate host
> generated recoverable fault events to the guest.
>
> The handler runs in QEMU's main loop, using a non-blocking fd registered
> via qemu_set_fd_handler().
>
> Signed-off-by: Zhenzhong Duan
> <[[email protected]](mailto:[email protected])>
> Tested-by: Xudong Hao <[[email protected]](mailto:[email protected])>
> ---
> hw/i386/intel_iommu_accel.h | 2 +
> hw/i386/intel_iommu_internal.h | 3 +
> hw/i386/intel_iommu.c | 14 ++-
> hw/i386/intel_iommu_accel.c | 162 +++++++++++++++++++++++++++++++--
> hw/i386/trace-events | 1 +
> 5 files changed, 171 insertions(+), 11 deletions(-)
>
> diff --git a/hw/i386/intel_iommu_accel.h b/hw/i386/intel_iommu_accel.h
> index 46c1a29409..e0319749a5 100644
> --- a/hw/i386/intel_iommu_accel.h
> +++ b/hw/i386/intel_iommu_accel.h
> @@ -18,6 +18,8 @@ typedef struct VTDAccelPASIDCacheEntry {
> VTDPASIDEntry pasid_entry;
> uint32_t pasid;
> uint32_t fs_hwpt_id;
> + uint32_t fault_id;
> + int fault_fd;
> QLIST_ENTRY(VTDAccelPASIDCacheEntry) next;
> } VTDAccelPASIDCacheEntry;
>
> diff --git a/hw/i386/intel_iommu_internal.h b/hw/i386/intel_iommu_internal.h
> index 924e91cb8a..5ebc204e9a 100644
> --- a/hw/i386/intel_iommu_internal.h
> +++ b/hw/i386/intel_iommu_internal.h
> @@ -786,4 +786,7 @@ int vtd_dev_to_context_entry(IntelIOMMUState *s, uint8_t
> bus_num,
> VTDAddressSpace *vtd_get_as_by_sid(IntelIOMMUState *s, uint16_t sid);
> int vtd_dev_get_pe_from_pasid(IntelIOMMUState *s, PCIBus *bus, uint8_t
> devfn,
> uint32_t pasid, VTDPASIDEntry *pe);
> +int vtd_pri_request_page(PCIBus *bus, void *opaque, int devfn, uint32_t
> pasid,
> + bool priv_req, bool exec_req, hwaddr addr, bool
> lpig,
> + uint16_t prgi, bool is_read, bool is_write);
> #endif
> diff --git a/hw/i386/intel_iommu.c b/hw/i386/intel_iommu.c
> index 82c3c3b2c3..022923a058 100644
> --- a/hw/i386/intel_iommu.c
> +++ b/hw/i386/intel_iommu.c
> @@ -5369,11 +5369,15 @@ static int
> vtd_pri_perform_implicit_invalidation(VTDAddressSpace *vtd_as,
> return ret;
> }
>
> -/* Page Request Descriptor : 7.4.1.1 */
> -static int vtd_pri_request_page(PCIBus *bus, void *opaque, int devfn,
> - uint32_t pasid, bool priv_req, bool
> exec_req,
> - hwaddr addr, bool lpig, uint16_t prgi,
> - bool is_read, bool is_write)
> +/*
> + * Page Request Descriptor : 7.4.1.1
> + *
> + * Because it is facing the emulated device models, so this should be PCI
> + * PASID instead of the pasids used within vIOMMU.
> + */
> +int vtd_pri_request_page(PCIBus *bus, void *opaque, int devfn, uint32_t
> pasid,
> + bool priv_req, bool exec_req, hwaddr addr, bool
> lpig,
> + uint16_t prgi, bool is_read, bool is_write)
> {
> IntelIOMMUState *s = opaque;
> VTDAddressSpace *vtd_as;
> diff --git a/hw/i386/intel_iommu_accel.c b/hw/i386/intel_iommu_accel.c
> index d2f41f18f1..c0bd249c4a 100644
> --- a/hw/i386/intel_iommu_accel.c
> +++ b/hw/i386/intel_iommu_accel.c
> @@ -9,6 +9,7 @@
> */
>
> #include "qemu/osdep.h"
> +#include "qemu/error-report.h"
> #include "system/iommufd.h"
> #include "intel_iommu_internal.h"
> #include "intel_iommu_accel.h"
> @@ -77,14 +78,140 @@ VTDHostIOMMUDevice
> *vtd_find_hiod_iommufd(VTDAddressSpace *as)
> return NULL;
> }
>
> -static bool vtd_create_fs_hwpt(VTDHostIOMMUDevice *vtd_hiod,
> - VTDPASIDEntry *pe, uint32_t *fs_hwpt_id,
> - Error **errp)
> +static void vtd_propagate_recoverable_fault(VTDAccelPASIDCacheEntry
> *vtd_pce,
> + struct iommu_hwpt_pgfault
> *fault,
> + unsigned cnt)
Why not a trailing "s" to vtd_propagate_recoverable_fault?
> +{
> + VTDHostIOMMUDevice *vtd_hiod = vtd_pce->vtd_hiod;
> + uint32_t pasid =
> + vtd_pce->pasid == IOMMU_NO_PASID ? PCI_NO_PASID : vtd_pce->pasid;
> +
> + for (; cnt--; fault++) {
> + bool last_page = fault->flags & IOMMU_PGFAULT_FLAGS_LAST_PAGE;
> +
> + vtd_pri_request_page(vtd_hiod->bus, vtd_hiod->iommu_state,
> + vtd_hiod->devfn, pasid,
> + fault->perm & IOMMU_PGFAULT_PERM_PRIV,
> + fault->perm & IOMMU_PGFAULT_PERM_EXEC,
> + fault->addr, last_page, fault->grpid,
> + fault->perm & IOMMU_PGFAULT_PERM_READ,
> + fault->perm & IOMMU_PGFAULT_PERM_WRITE);
> + }
> +}
> +
> +/* Batch size per read(); remaining faults trigger another callback */
> +#define FAULTQ_BUF_SIZE 2048
> +
> +static void vtd_read_fs_faultq(void *opaque)
> +{
> + VTDAccelPASIDCacheEntry *vtd_pce = opaque;
> + struct iommu_hwpt_pgfault fault[FAULTQ_BUF_SIZE];
s/fault/faults/
> + uint32_t id = vtd_pce->fault_id;
> + int fd = vtd_pce->fault_fd;
> + ssize_t bytes, last_bytes;
> +
> + bytes = read(fd, fault, sizeof(fault));
> + trace_vtd_read_fs_faultq(id, fd, bytes);
> + if (bytes < 0) {
> + if (errno != EAGAIN && errno != EINTR) {
> + error_report_once("FAULTQ(id %u): read failed (%m)", id);
> + }
> + return;
> + } else if (!bytes) {
> + error_report_once("FAULTQ(id %u): fault group empty unexpectedly",
> id);
> + return;
> + }
> +
> + last_bytes = bytes % sizeof(fault[0]);
> + if (last_bytes) {
> + error_report_once("FAULTQ(id %u): discard partial fault data:
> %zd/%zu",
> + id, last_bytes, sizeof(fault));
> + }
> +
> + vtd_propagate_recoverable_fault(vtd_pce, fault, bytes /
> sizeof(fault[0]));
> +}
> +
> +static void vtd_destroy_fs_faultq(VTDHostIOMMUDevice *vtd_hiod,
> + uint32_t fault_id, int fault_fd)
> +{
> + HostIOMMUDeviceIOMMUFD *hiodi =
> HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod);
> +
> + if (fault_fd < 0) {
> + return;
> + }
> +
> + close(fault_fd);
> + iommufd_backend_free_id(hiodi->iommufd, fault_id);
> +}
> +
> +static bool vtd_create_fs_faultq(VTDHostIOMMUDevice *vtd_hiod,
> + uint32_t *fault_id_p, int *fault_fd_p,
> + Error **errp)
> +{
> + HostIOMMUDeviceIOMMUFD *hiodi =
> HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod);
> + uint32_t fault_id, fault_fd;
> + int flags;
> +
> + if (!iommufd_backend_alloc_faultq(hiodi->iommufd, &fault_id, &fault_fd,
> + errp)) {
> + return false;
> + }
> +
> + flags = fcntl(fault_fd, F_GETFL);
> + if (flags < 0) {
> + error_setg_errno(errp, errno, "Failed to get flags for FAULTQ fd");
> + goto free_faultq;
> + }
> +
> + if (fcntl(fault_fd, F_SETFL, flags | O_NONBLOCK) < 0) {
> + error_setg_errno(errp, errno, "Failed to set O_NONBLOCK on FAULTQ
> fd");
> + goto free_faultq;
> + }
> +
> + *fault_id_p = fault_id;
> + *fault_fd_p = fault_fd;
> + return true;
> +
> +free_faultq:
> + vtd_destroy_fs_faultq(vtd_hiod, fault_id, fault_fd);
> + return false;
> +}
> +
> +static void vtd_destroy_old_fs_faultq(VTDAccelPASIDCacheEntry *vtd_pce)
> +{
> + if (vtd_pce->fault_fd < 0) {
> + return;
> + }
> +
> + qemu_set_fd_handler(vtd_pce->fault_fd, NULL, NULL, NULL);
> + vtd_destroy_fs_faultq(vtd_pce->vtd_hiod, vtd_pce->fault_id,
> + vtd_pce->fault_fd);
> + vtd_pce->fault_id = 0;
> + vtd_pce->fault_fd = -1;
> +}
> +
> +static void vtd_setup_fs_faultq(VTDAccelPASIDCacheEntry *vtd_pce,
> + uint32_t fault_id, int fault_fd)
> +{
> + if (fault_fd < 0) {
> + return;
> + }
> +
> + vtd_pce->fault_id = fault_id;
> + vtd_pce->fault_fd = fault_fd;
> + qemu_set_fd_handler(fault_fd, vtd_read_fs_faultq, NULL, vtd_pce);
> +}
> +
> +static bool vtd_create_fs_hwpt(VTDHostIOMMUDevice *vtd_hiod, VTDPASIDEntry
> *pe,
> + bool has_fault_id, uint32_t fault_id,
> + uint32_t *fs_hwpt_id, Error **errp)
> {
> HostIOMMUDeviceIOMMUFD *hiodi =
> HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod);
> struct iommu_hwpt_vtd_s1 vtd = {};
> uint32_t flags = vtd_hiod->iommu_state->pasid ? IOMMU_HWPT_ALLOC_PASID :
> 0;
>
> + flags |= has_fault_id ? IOMMU_HWPT_FAULT_ID_VALID : 0;
> +
> vtd.flags = (VTD_SM_PASID_ENTRY_SRE(pe) ? IOMMU_VTD_S1_SRE : 0) |
> (VTD_SM_PASID_ENTRY_WPE(pe) ? IOMMU_VTD_S1_WPE : 0) |
> (VTD_SM_PASID_ENTRY_EAFE(pe) ? IOMMU_VTD_S1_EAFE : 0);
> @@ -94,7 +221,7 @@ static bool vtd_create_fs_hwpt(VTDHostIOMMUDevice
> *vtd_hiod,
> return iommufd_backend_alloc_hwpt(hiodi->iommufd, hiodi->devid,
> hiodi->hwpt_id, flags,
> IOMMU_HWPT_DATA_VTD_S1, sizeof(vtd),
> &vtd,
> - 0, fs_hwpt_id, errp);
> + fault_id, fs_hwpt_id, errp);
> }
>
> static void vtd_destroy_old_fs_hwpt(VTDAccelPASIDCacheEntry *vtd_pce)
> @@ -115,7 +242,8 @@ static bool
> vtd_device_attach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce,
> VTDHostIOMMUDevice *vtd_hiod = vtd_pce->vtd_hiod;
> HostIOMMUDeviceIOMMUFD *hiodi =
> HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod);
> VTDPASIDEntry *pe = &vtd_pce->pasid_entry;
> - uint32_t hwpt_id = hiodi->hwpt_id, pasid = vtd_pce->pasid;
> + uint32_t hwpt_id = hiodi->hwpt_id, pasid = vtd_pce->pasid, fault_id = 0;
>
> + int fault_fd = -1;
> bool ret;
>
> /*
> @@ -130,7 +258,24 @@ static bool
> vtd_device_attach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce,
> }
>
> if (vtd_pe_pgtt_is_fst(pe)) {
> - if (!vtd_create_fs_hwpt(vtd_hiod, pe, &hwpt_id, errp)) {
> + IntelIOMMUState *s = vtd_hiod->iommu_state;
> + VTDContextEntry ce;
> + uint8_t bus_n = pci_bus_num(vtd_hiod->bus);
> + bool is_pre = false;
> +
> + if (s->svm &&
> + !vtd_dev_to_context_entry(s, bus_n, vtd_hiod->devfn, &ce)) {
> + is_pre = !!VTD_CE_GET_PRE(&ce);
> +
> + if (is_pre &&
> + !vtd_create_fs_faultq(vtd_hiod, &fault_id, &fault_fd, errp))
> {
> + return false;
> + }
> + }
> +
> + if (!vtd_create_fs_hwpt(vtd_hiod, pe, is_pre, fault_id, &hwpt_id,
> + errp)) {
> + vtd_destroy_fs_faultq(vtd_hiod, fault_id, fault_fd);
> return false;
> }
> }
> @@ -140,11 +285,14 @@ static bool
> vtd_device_attach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce,
> if (ret) {
> /* Destroy old fs_hwpt if it's a replacement */
> vtd_destroy_old_fs_hwpt(vtd_pce);
> + vtd_destroy_old_fs_faultq(vtd_pce);
> if (vtd_pe_pgtt_is_fst(pe)) {
> vtd_pce->fs_hwpt_id = hwpt_id;
> + vtd_setup_fs_faultq(vtd_pce, fault_id, fault_fd);
> }
> } else if (vtd_pe_pgtt_is_fst(pe)) {
> iommufd_backend_free_id(hiodi->iommufd, hwpt_id);
> + vtd_destroy_fs_faultq(vtd_hiod, fault_id, fault_fd);
> }
>
> return ret;
> @@ -177,6 +325,7 @@ static bool
> vtd_device_detach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce,
>
> if (ret) {
> vtd_destroy_old_fs_hwpt(vtd_pce);
> + vtd_destroy_old_fs_faultq(vtd_pce);
> }
>
> return ret;
> @@ -276,6 +425,7 @@ static void vtd_accel_fill_pc(VTDHostIOMMUDevice
> *vtd_hiod, uint32_t pasid,
> vtd_pce->vtd_hiod = vtd_hiod;
> vtd_pce->pasid = pasid;
> vtd_pce->pasid_entry = *pe;
> + vtd_pce->fault_fd = -1;
> QLIST_INSERT_HEAD(&vtd_hiod->pasid_cache_list, vtd_pce, next);
>
> if (!vtd_device_attach_iommufd(vtd_pce, &local_err)) {
> diff --git a/hw/i386/trace-events b/hw/i386/trace-events
> index a1a50d0910..fcd3f33f0a 100644
> --- a/hw/i386/trace-events
> +++ b/hw/i386/trace-events
> @@ -77,6 +77,7 @@ vtd_reset_exit(void) ""
> vtd_device_attach_hwpt(uint32_t dev_id, uint32_t pasid, uint32_t hwpt_id,
> int ret) "dev_id %d pasid %d hwpt_id %d, ret: %d"
> vtd_device_detach_hwpt(uint32_t dev_id, uint32_t pasid, int ret) "dev_id %d
> pasid %d ret: %d"
> vtd_device_reattach_def_hwpt(uint32_t dev_id, uint32_t pasid, uint32_t
> hwpt_id, int ret) "dev_id %d pasid %d hwpt_id %d, ret: %d"
> +vtd_read_fs_faultq(uint32_t fault_id, uint32_t fault_fd, ssize_t bytes)
> "fault_id %d fault_fd %d ret: %zd"
>
> # amd_iommu.c
> amdvi_evntlog_fail(uint64_t addr, uint32_t head) "error: fail to write at
> addr 0x%"PRIx64" + offset 0x%"PRIx32
> --
> 2.52.0
>