From: Manish Honap <[email protected]> At bind, build the CXL memory device for the passed-through Type-2 accelerator so it joins the CXL topology and its HDM region resolves to a host physical range. A Type-2 device has no mailbox, so there is no media-ready register to poll: set media ready directly once the component registers validate (mirroring drivers/net/ethernet/sfc/efx_cxl.c)
As per current vfio-cxl support, reject a device with: - more than one HDM decoder - interleaving enabled - whose reset the host cannot service The CXL-core allocations are grouped with devres so a failed bind unwinds them: init failure falls back to plain vfio-pci with the device still bound, so devm would otherwise hold them until unbind. A low-power transition would reset the CXL Type-2 function and lose its CXL.mem contents, so keep it in D0 while it is assigned. Assisted-by: LLM Signed-off-by: Manish Honap <[email protected]> --- drivers/vfio/pci/cxl/vfio_cxl_core.c | 118 +++++++++++++++++++++++++++ drivers/vfio/pci/vfio_pci_core.c | 16 ++++ 2 files changed, 134 insertions(+) diff --git a/drivers/vfio/pci/cxl/vfio_cxl_core.c b/drivers/vfio/pci/cxl/vfio_cxl_core.c index cd5d41856404..0d92e6a409c1 100644 --- a/drivers/vfio/pci/cxl/vfio_cxl_core.c +++ b/drivers/vfio/pci/cxl/vfio_cxl_core.c @@ -6,15 +6,132 @@ */ #include <linux/module.h> +#include <linux/pci.h> +#include <linux/range.h> #include <linux/vfio_pci_core.h> +#include <cxl/cxl.h> +#include <cxl/pci.h> + +/** + * struct vfio_cxl_state - per-device state for a vfio-cxl device + * @cxlds: CXL device state; kept first for devm_cxl_dev_state_create() + * @cxlmd: memory device joined to the CXL topology at bind + * @hpa_range: host physical range of the HDM region + */ +struct vfio_cxl_state { + struct cxl_dev_state cxlds; + struct cxl_memdev *cxlmd; + struct range hpa_range; +}; static int vfio_cxl_init_device(struct vfio_pci_core_device *vdev) { + struct pci_dev *pdev = vdev->pdev; + struct vfio_cxl_state *cxl; + struct cxl_memdev *cxlmd; + u64 hdm_size, serial; + u16 dvsec; + int ret; + + /* + * pdev->hdm is cached at PCI enumeration, before any driver binds, so a + * device without it has no usable HDM decoder. Fall back to plain + * vfio-pci rather than deferring the bind forever. + */ + if (!pdev->hdm) + return -ENODEV; + + /* The guest drives one virtual decoder; multiple are unsupported. */ + if (pdev->hdm->decoder_count != 1) + return -EOPNOTSUPP; + + /* An interleaved decoder cannot be mapped 1:1 to the guest. */ + if (pdev->hdm->settings[0].interleave_ways != 1) + return -EOPNOTSUPP; + + /* + * The guest drives resets through the CXL Device DVSEC and polls the + * shadow for completion. If the host cannot service a function-scoped + * CXL reset, that request could never complete, so refuse the device + * rather than advertise a reset the guest would poll on forever. + */ + if (!cxl_reset_capable(pdev)) + return -EOPNOTSUPP; + + hdm_size = range_len(&pdev->hdm->settings[0].hpa_range); + if (!hdm_size) + return -ENXIO; + + dvsec = pci_find_dvsec_capability(pdev, PCI_VENDOR_ID_CXL, + PCI_DVSEC_CXL_DEVICE); + serial = pci_get_dsn(pdev); + + /* + * Group the CXL-core allocations so a later failure unwinds them here. + * A failed init falls back to plain vfio-pci with the device still + * bound, so devm would otherwise hold them until unbind. + */ + if (!devres_open_group(&pdev->dev, NULL, GFP_KERNEL)) + return -ENOMEM; + + cxl = devm_cxl_dev_state_create(&pdev->dev, CXL_DEVTYPE_DEVMEM, serial, + dvsec, struct vfio_cxl_state, cxlds, + false); + if (!cxl) { + ret = -ENOMEM; + goto err; + } + + ret = cxl_pci_setup_regs(pdev, CXL_REGLOC_RBI_COMPONENT, + &cxl->cxlds.reg_map); + if (ret) { + pci_err(pdev, "vfio-cxl: no component registers\n"); + goto err; + } + + if (!cxl->cxlds.reg_map.component_map.hdm_decoder.valid) { + pci_err(pdev, "vfio-cxl: HDM decoder registers not found\n"); + ret = -ENODEV; + goto err; + } + + /* + * A Type-2 accelerator has no mailbox and no media-ready register, so + * set media ready directly. + */ + cxl->cxlds.media_ready = true; + + ret = cxl_set_capacity(&cxl->cxlds, hdm_size); + if (ret) + goto err; + + cxlmd = devm_cxl_probe_mem(&cxl->cxlds, &cxl->hpa_range); + if (IS_ERR(cxlmd)) { + ret = PTR_ERR(cxlmd); + goto err; + } + + cxl->cxlmd = cxlmd; + devres_close_group(&pdev->dev, NULL); + + /* + * Powering a CXL Type-2 function down and back up reinitializes its + * device state and discards the contents of its coherent memory. Pin + * it in D0 for as long as it is assigned so CXL.mem stays intact. + */ + vdev->disable_idle_d3 = true; + vdev->cxl = cxl; + return 0; + +err: + devres_release_group(&pdev->dev, NULL); + return ret; } static void vfio_cxl_release_device(struct vfio_pci_core_device *vdev) { + vdev->cxl = NULL; } static int vfio_cxl_open_device(struct vfio_pci_core_device *vdev) @@ -59,3 +176,4 @@ module_exit(vfio_cxl_exit); MODULE_LICENSE("GPL"); MODULE_DESCRIPTION("VFIO support for CXL Type-2 devices"); +MODULE_IMPORT_NS("CXL"); diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c index a91e07181847..ddd6807893fd 100644 --- a/drivers/vfio/pci/vfio_pci_core.c +++ b/drivers/vfio/pci/vfio_pci_core.c @@ -325,6 +325,14 @@ int vfio_pci_set_power_state(struct vfio_pci_core_device *vdev, pci_power_t stat bool needs_restore = false, needs_save = false; int ret; + /* + * A low-power transition would reset the CXL Type-2 function and lose + * its CXL.mem contents, so keep it in D0 regardless of the guest or + * VMM request. + */ + if (vdev->cxl_ops && state > PCI_D0) + state = PCI_D0; + /* Prevent changing power state for PFs with VFs enabled */ if (state > PCI_D0) { lockdep_assert_held_write(&vdev->memory_lock); @@ -373,6 +381,14 @@ int vfio_pci_set_power_state(struct vfio_pci_core_device *vdev, pci_power_t stat static int vfio_pci_runtime_pm_entry(struct vfio_pci_core_device *vdev, struct eventfd_ctx *efdctx) { + /* + * Low power entry lets the PCI core autosuspend the device to D3hot, + * which would soft-reset a CXL Type-2 device and lose its coherent HDM + * memory. Refuse it for CXL; the device stays in D0. + */ + if (vdev->cxl_ops) + return -EINVAL; + /* * The vdev power related flags are protected with 'memory_lock' * semaphore. -- 2.25.1

