Hi Zhenzhong, Reviewed-by: Clement Mathieu--Drif <[email protected]>
Thanks On Wed, 2026-09-16 at 18:04 +0800, Zhenzhong Duan wrote: > Caution: External email. Do not open attachments or click links, unless this > email comes from a known sender and you know the content is safe. > > > When the guest enables the PRQ in vIOMMU, allocate a FAULTQ object so that > host-side recoverable fault events can be received and propagated back to > the guest. > > Install an event handler on the FAULTQ fd to read and propagate host > generated recoverable fault events to the guest. > > The handler runs in QEMU's main loop, using a non-blocking fd registered > via qemu_set_fd_handler(). > > Signed-off-by: Zhenzhong Duan > <[[email protected]](mailto:[email protected])> > Tested-by: Xudong Hao <[[email protected]](mailto:[email protected])> > --- > hw/i386/intel_iommu_accel.h | 2 + > hw/i386/intel_iommu_internal.h | 3 + > hw/i386/intel_iommu.c | 14 ++- > hw/i386/intel_iommu_accel.c | 163 +++++++++++++++++++++++++++++++-- > hw/i386/trace-events | 1 + > 5 files changed, 172 insertions(+), 11 deletions(-) > > diff --git a/hw/i386/intel_iommu_accel.h b/hw/i386/intel_iommu_accel.h > index 46c1a29409..e0319749a5 100644 > --- a/hw/i386/intel_iommu_accel.h > +++ b/hw/i386/intel_iommu_accel.h > @@ -18,6 +18,8 @@ typedef struct VTDAccelPASIDCacheEntry { > VTDPASIDEntry pasid_entry; > uint32_t pasid; > uint32_t fs_hwpt_id; > + uint32_t fault_id; > + int fault_fd; > QLIST_ENTRY(VTDAccelPASIDCacheEntry) next; > } VTDAccelPASIDCacheEntry; > > diff --git a/hw/i386/intel_iommu_internal.h b/hw/i386/intel_iommu_internal.h > index df7a0efa6e..4792035668 100644 > --- a/hw/i386/intel_iommu_internal.h > +++ b/hw/i386/intel_iommu_internal.h > @@ -787,4 +787,7 @@ int vtd_dev_to_context_entry(IntelIOMMUState *s, uint8_t > bus_num, > VTDAddressSpace *vtd_get_as_by_sid(IntelIOMMUState *s, uint16_t sid); > int vtd_dev_get_pe_from_pasid(IntelIOMMUState *s, PCIBus *bus, uint8_t > devfn, > uint32_t pasid, VTDPASIDEntry *pe); > +int vtd_pri_request_page(PCIBus *bus, void *opaque, int devfn, uint32_t > pasid, > + bool priv_req, bool exec_req, hwaddr addr, bool > lpig, > + uint16_t prgi, bool is_read, bool is_write); > #endif > diff --git a/hw/i386/intel_iommu.c b/hw/i386/intel_iommu.c > index 350d2b7753..81c53c6f5a 100644 > --- a/hw/i386/intel_iommu.c > +++ b/hw/i386/intel_iommu.c > @@ -5383,11 +5383,15 @@ static int > vtd_pri_perform_implicit_invalidation(VTDAddressSpace *vtd_as, > return ret; > } > > -/* Page Request Descriptor : 7.4.1.1 */ > -static int vtd_pri_request_page(PCIBus *bus, void *opaque, int devfn, > - uint32_t pasid, bool priv_req, bool > exec_req, > - hwaddr addr, bool lpig, uint16_t prgi, > - bool is_read, bool is_write) > +/* > + * Page Request Descriptor : 7.4.1.1 > + * > + * Because it is facing the emulated device models, so this should be PCI > + * PASID instead of the pasids used within vIOMMU. > + */ > +int vtd_pri_request_page(PCIBus *bus, void *opaque, int devfn, uint32_t > pasid, > + bool priv_req, bool exec_req, hwaddr addr, bool > lpig, > + uint16_t prgi, bool is_read, bool is_write) > { > IntelIOMMUState *s = opaque; > VTDAddressSpace *vtd_as; > diff --git a/hw/i386/intel_iommu_accel.c b/hw/i386/intel_iommu_accel.c > index c8f2835204..7baefe30cc 100644 > --- a/hw/i386/intel_iommu_accel.c > +++ b/hw/i386/intel_iommu_accel.c > @@ -9,6 +9,7 @@ > */ > > #include "qemu/osdep.h" > +#include "qemu/error-report.h" > #include "system/iommufd.h" > #include "intel_iommu_internal.h" > #include "intel_iommu_accel.h" > @@ -83,14 +84,141 @@ VTDHostIOMMUDevice > *vtd_find_hiod_iommufd(VTDAddressSpace *as) > return NULL; > } > > -static bool vtd_create_fs_hwpt(VTDHostIOMMUDevice *vtd_hiod, > - VTDPASIDEntry *pe, uint32_t *fs_hwpt_id, > - Error **errp) > +static void vtd_propagate_recoverable_faults(VTDAccelPASIDCacheEntry > *vtd_pce, > + struct iommu_hwpt_pgfault > *fault, > + unsigned cnt) > +{ > + VTDHostIOMMUDevice *vtd_hiod = vtd_pce->vtd_hiod; > + uint32_t pasid = > + vtd_pce->pasid == IOMMU_NO_PASID ? PCI_NO_PASID : vtd_pce->pasid; > + > + for (; cnt--; fault++) { > + bool last_page = fault->flags & IOMMU_PGFAULT_FLAGS_LAST_PAGE; > + > + vtd_pri_request_page(vtd_hiod->bus, vtd_hiod->iommu_state, > + vtd_hiod->devfn, pasid, > + fault->perm & IOMMU_PGFAULT_PERM_PRIV, > + fault->perm & IOMMU_PGFAULT_PERM_EXEC, > + fault->addr, last_page, fault->grpid, > + fault->perm & IOMMU_PGFAULT_PERM_READ, > + fault->perm & IOMMU_PGFAULT_PERM_WRITE); > + } > +} > + > +/* Batch size per read(); remaining faults trigger another callback */ > +#define FAULTQ_BUF_SIZE 2048 > + > +static void vtd_read_fs_faultq(void *opaque) > +{ > + VTDAccelPASIDCacheEntry *vtd_pce = opaque; > + struct iommu_hwpt_pgfault faults[FAULTQ_BUF_SIZE]; > + uint32_t id = vtd_pce->fault_id; > + int fd = vtd_pce->fault_fd; > + ssize_t bytes, last_bytes; > + > + bytes = read(fd, faults, sizeof(faults)); > + trace_vtd_read_fs_faultq(id, fd, bytes); > + if (bytes < 0) { > + if (errno != EAGAIN && errno != EINTR) { > + error_report_once("FAULTQ(id %u): read failed (%m)", id); > + } > + return; > + } else if (!bytes) { > + error_report_once("FAULTQ(id %u): fault group empty unexpectedly", > id); > + return; > + } > + > + last_bytes = bytes % sizeof(faults[0]); > + if (last_bytes) { > + error_report_once("FAULTQ(id %u): discard partial fault data: > %zd/%zu", > + id, last_bytes, sizeof(faults[0])); > + } > + > + vtd_propagate_recoverable_faults(vtd_pce, faults, > + bytes / sizeof(faults[0])); > +} > + > +static void vtd_destroy_fs_faultq(VTDHostIOMMUDevice *vtd_hiod, > + uint32_t fault_id, int fault_fd) > +{ > + HostIOMMUDeviceIOMMUFD *hiodi = > HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod); > + > + if (fault_fd < 0) { > + return; > + } > + > + close(fault_fd); > + iommufd_backend_free_id(hiodi->iommufd, fault_id); > +} > + > +static bool vtd_create_fs_faultq(VTDHostIOMMUDevice *vtd_hiod, > + uint32_t *fault_id_p, int *fault_fd_p, > + Error **errp) > +{ > + HostIOMMUDeviceIOMMUFD *hiodi = > HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod); > + uint32_t fault_id, fault_fd; > + int flags; > + > + if (!iommufd_backend_alloc_faultq(hiodi->iommufd, &fault_id, &fault_fd, > + errp)) { > + return false; > + } > + > + flags = fcntl(fault_fd, F_GETFL); > + if (flags < 0) { > + error_setg_errno(errp, errno, "Failed to get flags for FAULTQ fd"); > + goto free_faultq; > + } > + > + if (fcntl(fault_fd, F_SETFL, flags | O_NONBLOCK) < 0) { > + error_setg_errno(errp, errno, "Failed to set O_NONBLOCK on FAULTQ > fd"); > + goto free_faultq; > + } > + > + *fault_id_p = fault_id; > + *fault_fd_p = fault_fd; > + return true; > + > +free_faultq: > + vtd_destroy_fs_faultq(vtd_hiod, fault_id, fault_fd); > + return false; > +} > + > +static void vtd_destroy_old_fs_faultq(VTDAccelPASIDCacheEntry *vtd_pce) > +{ > + if (vtd_pce->fault_fd < 0) { > + return; > + } > + > + qemu_set_fd_handler(vtd_pce->fault_fd, NULL, NULL, NULL); > + vtd_destroy_fs_faultq(vtd_pce->vtd_hiod, vtd_pce->fault_id, > + vtd_pce->fault_fd); > + vtd_pce->fault_id = 0; > + vtd_pce->fault_fd = -1; > +} > + > +static void vtd_setup_fs_faultq(VTDAccelPASIDCacheEntry *vtd_pce, > + uint32_t fault_id, int fault_fd) > +{ > + if (fault_fd < 0) { > + return; > + } > + > + vtd_pce->fault_id = fault_id; > + vtd_pce->fault_fd = fault_fd; > + qemu_set_fd_handler(fault_fd, vtd_read_fs_faultq, NULL, vtd_pce); > +} > + > +static bool vtd_create_fs_hwpt(VTDHostIOMMUDevice *vtd_hiod, VTDPASIDEntry > *pe, > + bool has_fault_id, uint32_t fault_id, > + uint32_t *fs_hwpt_id, Error **errp) > { > HostIOMMUDeviceIOMMUFD *hiodi = > HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod); > struct iommu_hwpt_vtd_s1 vtd = {}; > uint32_t flags = vtd_hiod->iommu_state->pasid ? IOMMU_HWPT_ALLOC_PASID : > 0; > > + flags |= has_fault_id ? IOMMU_HWPT_FAULT_ID_VALID : 0; > + > vtd.flags = (VTD_SM_PASID_ENTRY_SRE(pe) ? IOMMU_VTD_S1_SRE : 0) | > (VTD_SM_PASID_ENTRY_WPE(pe) ? IOMMU_VTD_S1_WPE : 0) | > (VTD_SM_PASID_ENTRY_EAFE(pe) ? IOMMU_VTD_S1_EAFE : 0); > @@ -100,7 +228,7 @@ static bool vtd_create_fs_hwpt(VTDHostIOMMUDevice > *vtd_hiod, > return iommufd_backend_alloc_hwpt(hiodi->iommufd, hiodi->devid, > hiodi->hwpt_id, flags, > IOMMU_HWPT_DATA_VTD_S1, sizeof(vtd), > &vtd, > - 0, fs_hwpt_id, errp); > + fault_id, fs_hwpt_id, errp); > } > > static void vtd_destroy_old_fs_hwpt(VTDAccelPASIDCacheEntry *vtd_pce) > @@ -121,7 +249,8 @@ static bool > vtd_device_attach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce, > VTDHostIOMMUDevice *vtd_hiod = vtd_pce->vtd_hiod; > HostIOMMUDeviceIOMMUFD *hiodi = > HOST_IOMMU_DEVICE_IOMMUFD(vtd_hiod->hiod); > VTDPASIDEntry *pe = &vtd_pce->pasid_entry; > - uint32_t hwpt_id = hiodi->hwpt_id, pasid = vtd_pce->pasid; > + uint32_t hwpt_id = hiodi->hwpt_id, pasid = vtd_pce->pasid, fault_id = 0; > > + int fault_fd = -1; > bool ret; > > /* > @@ -136,7 +265,24 @@ static bool > vtd_device_attach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce, > } > > if (vtd_pe_pgtt_is_fst(pe)) { > - if (!vtd_create_fs_hwpt(vtd_hiod, pe, &hwpt_id, errp)) { > + IntelIOMMUState *s = vtd_hiod->iommu_state; > + VTDContextEntry ce; > + uint8_t bus_n = pci_bus_num(vtd_hiod->bus); > + bool is_pre = false; > + > + if (s->svm && > + !vtd_dev_to_context_entry(s, bus_n, vtd_hiod->devfn, &ce)) { > + is_pre = !!VTD_CE_GET_PRE(&ce); > + > + if (is_pre && > + !vtd_create_fs_faultq(vtd_hiod, &fault_id, &fault_fd, errp)) > { > + return false; > + } > + } > + > + if (!vtd_create_fs_hwpt(vtd_hiod, pe, is_pre, fault_id, &hwpt_id, > + errp)) { > + vtd_destroy_fs_faultq(vtd_hiod, fault_id, fault_fd); > return false; > } > } > @@ -146,11 +292,14 @@ static bool > vtd_device_attach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce, > if (ret) { > /* Destroy old fs_hwpt if it's a replacement */ > vtd_destroy_old_fs_hwpt(vtd_pce); > + vtd_destroy_old_fs_faultq(vtd_pce); > if (vtd_pe_pgtt_is_fst(pe)) { > vtd_pce->fs_hwpt_id = hwpt_id; > + vtd_setup_fs_faultq(vtd_pce, fault_id, fault_fd); > } > } else if (vtd_pe_pgtt_is_fst(pe)) { > iommufd_backend_free_id(hiodi->iommufd, hwpt_id); > + vtd_destroy_fs_faultq(vtd_hiod, fault_id, fault_fd); > } > > return ret; > @@ -183,6 +332,7 @@ static bool > vtd_device_detach_iommufd(VTDAccelPASIDCacheEntry *vtd_pce, > > if (ret) { > vtd_destroy_old_fs_hwpt(vtd_pce); > + vtd_destroy_old_fs_faultq(vtd_pce); > } > > return ret; > @@ -282,6 +432,7 @@ static void vtd_accel_fill_pc(VTDHostIOMMUDevice > *vtd_hiod, uint32_t pasid, > vtd_pce->vtd_hiod = vtd_hiod; > vtd_pce->pasid = pasid; > vtd_pce->pasid_entry = *pe; > + vtd_pce->fault_fd = -1; > QLIST_INSERT_HEAD(&vtd_hiod->pasid_cache_list, vtd_pce, next); > > if (!vtd_device_attach_iommufd(vtd_pce, &local_err)) { > diff --git a/hw/i386/trace-events b/hw/i386/trace-events > index a1a50d0910..ad84d106a4 100644 > --- a/hw/i386/trace-events > +++ b/hw/i386/trace-events > @@ -77,6 +77,7 @@ vtd_reset_exit(void) "" > vtd_device_attach_hwpt(uint32_t dev_id, uint32_t pasid, uint32_t hwpt_id, > int ret) "dev_id %d pasid %d hwpt_id %d, ret: %d" > vtd_device_detach_hwpt(uint32_t dev_id, uint32_t pasid, int ret) "dev_id %d > pasid %d ret: %d" > vtd_device_reattach_def_hwpt(uint32_t dev_id, uint32_t pasid, uint32_t > hwpt_id, int ret) "dev_id %d pasid %d hwpt_id %d, ret: %d" > +vtd_read_fs_faultq(uint32_t fault_id, uint32_t fault_fd, ssize_t ret) > "fault_id %d fault_fd %d ret: %zd" > > # amd_iommu.c > amdvi_evntlog_fail(uint64_t addr, uint32_t head) "error: fail to write at > addr 0x%"PRIx64" + offset 0x%"PRIx32 > -- > 2.52.0 >
