On a link-down or impending PF reset, in-flight Tx descriptors were left pending when the queues were torn down, which could trigger Malicious Driver Detection (MDD) events and leak descriptors.
Added iavf_dev_tx_drain() to let already-posted Tx bursts complete and flush the rings within a bounded budget, and call it on link-down and reset-impending events before teardown, preventing MDD events and descriptor leaks. The drain selects the cleanup routine that matches the active Tx path: the scalar path uses ci_tx_xmit_cleanup(), while the vector and CTX paths use ci_tx_free_bufs_vec(). This matters because the scalar and vector paths track their software rings differently (ci_tx_entry vs ci_tx_entry_vec) and using the scalar routine on a vector queue would walk the wrong ring and free the wrong mbufs. Signed-off-by: Anurag Mandal <[email protected]> --- drivers/net/intel/iavf/iavf_rxtx.c | 104 ++++++++++++++++++++++++++++ drivers/net/intel/iavf/iavf_rxtx.h | 6 ++ drivers/net/intel/iavf/iavf_vchnl.c | 7 ++ 3 files changed, 117 insertions(+) diff --git a/drivers/net/intel/iavf/iavf_rxtx.c b/drivers/net/intel/iavf/iavf_rxtx.c index 4f2ffe6188..931bb8420d 100644 --- a/drivers/net/intel/iavf/iavf_rxtx.c +++ b/drivers/net/intel/iavf/iavf_rxtx.c @@ -32,6 +32,7 @@ #include "iavf.h" #include "iavf_rxtx.h" +#include "iavf_rxtx_vec_common.h" #include "iavf_ipsec_crypto.h" #include "rte_pmd_iavf.h" @@ -4025,6 +4026,109 @@ iavf_tx_done_cleanup_full(struct ci_tx_queue *txq, return (int)pkt_cnt; } +/* + * Reclaim completed Tx descriptors for a single queue using the cleanup + * routine that matches the active Tx path. + * The scalar and vector paths track their software rings differently + * (ci_tx_entry vs ci_tx_entry_vec) and keep separate completion + * bookkeeping, so using the scalar routine on a vector queue + * (or vice versa) would free the wrong mbufs. + * Returns true if any descriptors were reclaimed. + */ +static bool +iavf_tx_drain_cleanup(struct ci_tx_queue *txq, + enum iavf_tx_func_type tx_func_type) +{ + switch (tx_func_type) { + case IAVF_TX_AVX2_CTX: + case IAVF_TX_AVX2_CTX_OFFLOAD: + case IAVF_TX_AVX512_CTX: + case IAVF_TX_AVX512_CTX_OFFLOAD: + return ci_tx_free_bufs_vec(txq, iavf_tx_desc_done, true) != 0; + case IAVF_TX_NEON: + case IAVF_TX_AVX2: + case IAVF_TX_AVX2_OFFLOAD: + case IAVF_TX_AVX512: + case IAVF_TX_AVX512_OFFLOAD: + return ci_tx_free_bufs_vec(txq, iavf_tx_desc_done, false) != 0; + case IAVF_TX_DEFAULT: + default: + return ci_tx_xmit_cleanup(txq) == 0; + } +} + +/* + * iavf_dev_tx_drain - drain in-flight Tx descriptors after a link-down or + * impending PF reset event. + */ +void +iavf_dev_tx_drain(struct rte_eth_dev *dev) +{ + struct iavf_adapter *adapter = + IAVF_DEV_PRIVATE_TO_ADAPTER(dev->data->dev_private); + enum iavf_tx_func_type tx_func_type = adapter->tx_func_type; + struct ci_tx_queue *txq; + uint64_t hz, deadline; + int idle_iters = 0; + uint16_t qid; + + /* + * Allow any Tx burst already in flight on a data-plane lcore to + * write its remaining descriptors and notify. After + * this window, the no_poll gate set by the caller is observed at + * the next burst-entry and no new descriptors will be posted. + */ + rte_delay_us_block(IAVF_TX_DRAIN_SETTLE_US); + + hz = rte_get_timer_hz(); + deadline = rte_get_timer_cycles() + + (hz * IAVF_TX_DRAIN_TIMEOUT_US) / 1000000ULL; + + while (rte_get_timer_cycles() < deadline) { + bool any_pending = false; + bool any_progress = false; + + for (qid = 0; qid < dev->data->nb_tx_queues; qid++) { + txq = dev->data->tx_queues[qid]; + if (txq == NULL || + dev->data->tx_queue_state[qid] != + RTE_ETH_QUEUE_STATE_STARTED) + continue; + + /* + * nb_tx_free == nb_tx_desc - 1 means the ring is + * empty (one descriptor is always reserved). + */ + if (txq->nb_tx_free >= txq->nb_tx_desc - 1) + continue; + + any_pending = true; + if (iavf_tx_drain_cleanup(txq, tx_func_type)) + any_progress = true; + } + + if (!any_pending) + return; + + if (any_progress) { + idle_iters = 0; + } else if (++idle_iters >= IAVF_TX_DRAIN_IDLE_MAX) { + /* + * HW has not advanced the RS-bit write-back for + * several polling intervals; either the queue is + * quiescent except for the sub-rs_thresh tail + * (which we cannot observe here) or HW is no + * longer fetching. Further polling is unlikely to + * help, and the PF teardown path has its own + * grace period for the remainder. + */ + break; + } + + rte_delay_us_block(IAVF_TX_DRAIN_POLL_US); + } +} + int iavf_dev_tx_done_cleanup(void *txq, uint32_t free_cnt) { diff --git a/drivers/net/intel/iavf/iavf_rxtx.h b/drivers/net/intel/iavf/iavf_rxtx.h index 22ea415f44..4088bc421c 100644 --- a/drivers/net/intel/iavf/iavf_rxtx.h +++ b/drivers/net/intel/iavf/iavf_rxtx.h @@ -506,6 +506,11 @@ enum iavf_tx_ctx_desc_tunnel_l4_tunnel_type { /* Valid indicator bit for the time_stamp_low field */ #define IAVF_RX_FLX_DESC_TS_VALID (0x1UL) +#define IAVF_TX_DRAIN_TIMEOUT_US 10000 /* total drain budget: 10 ms */ +#define IAVF_TX_DRAIN_SETTLE_US 100 /* let in-flight burst land */ +#define IAVF_TX_DRAIN_POLL_US 50 /* poll interval */ +#define IAVF_TX_DRAIN_IDLE_MAX 20 /* ~1 ms of no RS write-back */ + int iavf_dev_rx_queue_setup(struct rte_eth_dev *dev, uint16_t queue_idx, uint16_t nb_desc, @@ -641,6 +646,7 @@ void iavf_set_default_ptype_table(struct rte_eth_dev *dev); void iavf_rx_queue_release_mbufs_vec(struct ci_rx_queue *rxq); void iavf_rx_queue_release_mbufs_neon(struct ci_rx_queue *rxq); enum rte_vect_max_simd iavf_get_max_simd_bitwidth(void); +void iavf_dev_tx_drain(struct rte_eth_dev *dev); static inline void iavf_dump_rx_descriptor(struct ci_rx_queue *rxq, diff --git a/drivers/net/intel/iavf/iavf_vchnl.c b/drivers/net/intel/iavf/iavf_vchnl.c index 8e102b02aa..d22990a524 100644 --- a/drivers/net/intel/iavf/iavf_vchnl.c +++ b/drivers/net/intel/iavf/iavf_vchnl.c @@ -269,6 +269,8 @@ iavf_handle_link_change_event(struct rte_eth_dev *dev, iavf_set_no_poll(adapter, true); PMD_DRV_LOG(DEBUG, "VF no poll turned %s", adapter->no_poll ? "on" : "off"); + if (!vf->link_up) + iavf_dev_tx_drain(dev); } /* @@ -341,6 +343,8 @@ iavf_read_msg_from_pf(struct iavf_adapter *adapter, uint16_t buf_len, if (!vf->vf_reset) { vf->vf_reset = true; iavf_set_no_poll(adapter, false); + if (adapter->devargs.no_poll_on_link_down) + iavf_dev_tx_drain(vf->eth_dev); iavf_dev_event_post(vf->eth_dev, RTE_ETH_EVENT_INTR_RESET, NULL, 0); @@ -579,6 +583,9 @@ iavf_handle_pf_event_msg(struct rte_eth_dev *dev, uint8_t *msg, if (!vf->vf_reset) { vf->vf_reset = true; iavf_set_no_poll(adapter, false); + iavf_dev_watchdog_enable(adapter); + if (adapter->devargs.no_poll_on_link_down) + iavf_dev_tx_drain(dev); iavf_dev_event_post(dev, RTE_ETH_EVENT_INTR_RESET, NULL, 0); } -- 2.34.1

