The combined Tx burst path suffers from severe cache misses because the compact CQE and normal paths are interleaved. Split out a dedicated compact CQE entry point and select it directly from the runtime flag instead of the dynamic tx_ops indirection.
Add a tx_burst_mode_get callback so applications can query the active Tx burst mode and its offloads for each queue. Signed-off-by: Jiacheng Ye <[email protected]> --- doc/guides/rel_notes/release_26_11.rst | 1 + drivers/net/hinic3/hinic3_ethdev.c | 109 +++-- drivers/net/hinic3/hinic3_ethdev.h | 10 +- drivers/net/hinic3/hinic3_tx.c | 528 +++++++++++++++++++++---- drivers/net/hinic3/hinic3_tx.h | 58 ++- 5 files changed, 574 insertions(+), 132 deletions(-) diff --git a/doc/guides/rel_notes/release_26_11.rst b/doc/guides/rel_notes/release_26_11.rst index 19ecce010f..216f2b297e 100644 --- a/doc/guides/rel_notes/release_26_11.rst +++ b/doc/guides/rel_notes/release_26_11.rst @@ -102,6 +102,7 @@ New Features * Added device parameters for runtime configuration. * Added a compact CQE receive data path. * Extended TCP segmentation offload (TSO) to 255 segments. + * Added a compact CQE transmit data path. * **Updated Intel iavf driver.** diff --git a/drivers/net/hinic3/hinic3_ethdev.c b/drivers/net/hinic3/hinic3_ethdev.c index c472b8fae1..ffbd080ede 100644 --- a/drivers/net/hinic3/hinic3_ethdev.c +++ b/drivers/net/hinic3/hinic3_ethdev.c @@ -288,6 +288,8 @@ static const struct hinic3_xstats_name_off hinic3_txq_stats_strings[] = { HINIC3_TXQ_STAT(burst_pkts), HINIC3_TXQ_STAT(sge_len0), HINIC3_TXQ_STAT(mbuf_null), + HINIC3_TXQ_STAT(cpy_pkts), + HINIC3_TXQ_STAT(sge_len_too_large), #ifdef HINIC3_XSTAT_PROF_TX HINIC3_TXQ_STAT(app_tsc), @@ -1399,6 +1401,14 @@ hinic3_tx_queue_setup(struct rte_eth_dev *dev, uint16_t qid, uint16_t nb_desc, nic_dev = HINIC3_ETH_DEV_TO_PRIVATE_NIC_DEV(dev); hwdev = nic_dev->hwdev; + /* Queue depth must be equal to first queue */ + if (nic_dev->txq_depth == 0) { + nic_dev->txq_depth = nb_desc; + } else if (nb_desc != nic_dev->txq_depth) { + PMD_DRV_LOG(WARNING, "txq%u depth:%u is not equal to first queue depth:%u.", + qid, nb_desc, nic_dev->txq_depth); + nb_desc = nic_dev->txq_depth; + } /* Queue depth must be power of 2, otherwise will be aligned up. */ sq_depth = (nb_desc & (nb_desc - 1)) @@ -1456,6 +1466,7 @@ hinic3_tx_queue_setup(struct rte_eth_dev *dev, uint16_t qid, uint16_t nb_desc, txq->owner = 1; txq->non_tso_max_pkt_len = HINIC3_IS_SP230_NIC(nic_dev) ? MAX_SINGLE_SGE_SIZE : HINIC3_MAX_JUMBO_FRAME_SIZE; + txq->tx_free_loop = nic_dev->config.tx_free_loop; if (!ODD_NUMBER_QUEUE_ID(qid) && hinic3_cmd_vf_lag(nic_dev->hwdev, hinic3_global_func_id(nic_dev->hwdev), @@ -1464,9 +1475,11 @@ hinic3_tx_queue_setup(struct rte_eth_dev *dev, uint16_t qid, uint16_t nb_desc, else txq->cos = nic_dev->default_cos; - txq->cos = nic_dev->cos_map[(int)(txq->cos) & nic_dev->cos_mask]; + if (HINIC3_IS_SP230_NIC(nic_dev)) + txq->cos = nic_dev->cos_map[(int)(txq->cos) & nic_dev->cos_mask]; txq->tx_deferred_start = tx_conf->tx_deferred_start; + txq->tx_wqe_compact_task = HINIC3_SUPPORT_TX_WQE_COMPACT_TASK(nic_dev); ci_mz = hinic3_dma_zone_reserve(dev, "hinic3_sq_ci", qid, @@ -2300,7 +2313,13 @@ hinic3_dev_stop(struct rte_eth_dev *dev) /* Free mempool. */ hinic3_copy_mempool_uninit(nic_dev); + for (uint16_t i = 0; i < dev->data->nb_rx_queues; i++) + dev->data->rx_queue_state[i] = RTE_ETH_QUEUE_STATE_STOPPED; + for (uint16_t i = 0; i < dev->data->nb_tx_queues; i++) + dev->data->tx_queue_state[i] = RTE_ETH_QUEUE_STATE_STOPPED; + nic_dev->rxq_depth = 0; + nic_dev->txq_depth = 0; return 0; } @@ -2356,6 +2375,9 @@ hinic3_dev_close(struct rte_eth_dev *eth_dev) { struct hinic3_nic_dev *nic_dev = HINIC3_ETH_DEV_TO_PRIVATE_NIC_DEV(eth_dev); + + if (rte_eal_process_type() != RTE_PROC_PRIMARY) + return 0; int ret; if (hinic3_test_and_set_bit(HINIC3_DEV_CLOSE, &nic_dev->dev_status)) { @@ -3012,6 +3034,26 @@ hinic3_dev_stats_get(struct rte_eth_dev *dev, struct rte_eth_stats *stats, } } + if (HINIC3_IS_VF(nic_dev->hwdev) && HINIC3_IS_SP230_NIC(nic_dev)) { + for (uint32_t i = 0; i < nic_dev->num_rqs; i++) { + struct hinic3_rxq *rxq = nic_dev->rxqs[i]; + if (rxq == NULL) + continue; + stats->ipackets += rxq->rxq_stats.packets; + stats->ibytes += rxq->rxq_stats.bytes; + stats->imissed += rxq->rxq_stats.dropped; + } + + for (uint32_t i = 0; i < nic_dev->num_sqs; i++) { + struct hinic3_txq *txq = nic_dev->txqs[i]; + if (txq == NULL) + continue; + stats->opackets += txq->txq_stats.packets; + stats->obytes += txq->txq_stats.bytes; + } + return 0; + } + /* Vport stats. */ stats->oerrors += vport_stats.tx_discard_vport; @@ -3082,14 +3124,16 @@ get_port_cir_drop(struct hinic3_nic_dev *nic_dev, memset(&port_stats, 0, sizeof(port_stats)); - err = hinic3_get_cir_drop(nic_dev->hwdev, &port_stats); - if (err) { - PMD_DRV_LOG(ERR, "Failed to get CPB cir drops from fw."); + if (HINIC3_IS_SP620_NIC(nic_dev)) { + err = hinic3_get_cir_drop(nic_dev->hwdev, &port_stats); + if (err) { + PMD_DRV_LOG(ERR, "Failed to get CPB cir drops from fw."); - for (i = 0; i < RTE_DIM(hinic3_cir_drop_stats_strings); i++) - xstats[i].value = 0; + for (i = 0; i < RTE_DIM(hinic3_cir_drop_stats_strings); i++) + xstats[i].value = 0; - return RTE_DIM(hinic3_cir_drop_stats_strings); + return RTE_DIM(hinic3_cir_drop_stats_strings); + } } for (i = 0; i < RTE_DIM(hinic3_cir_drop_stats_strings); i++) { @@ -3695,6 +3739,7 @@ static const struct eth_dev_ops hinic3_pmd_ops = { .mac_addr_add = hinic3_mac_addr_add, .set_mc_addr_list = hinic3_set_mc_addr_list, .flow_ops_get = hinic3_dev_filter_ctrl, + .tx_burst_mode_get = hinic3_tx_burst_mode_get, .fec_get_capability = hinic3_fec_capability_get, .fec_get = hinic3_fec_get, .fec_set = hinic3_fec_set, @@ -3747,16 +3792,9 @@ static const struct eth_dev_ops hinic3_pmd_vf_ops = { .mac_addr_add = hinic3_mac_addr_add, .set_mc_addr_list = hinic3_set_mc_addr_list, .flow_ops_get = hinic3_dev_filter_ctrl, + .tx_burst_mode_get = hinic3_tx_burst_mode_get, }; -static void hinic3_nic_tx_rx_ops_init(struct hinic3_nic_dev *nic_dev) -{ - if (HINIC3_SUPPORT_TX_WQE_COMPACT_TASK(nic_dev)) - nic_dev->tx_ops->nic_tx_set_wqe_offload = hinic3_tx_set_compact_task_offload; - else - nic_dev->tx_ops->nic_tx_set_wqe_offload = hinic3_tx_set_normal_task_offload; -} - /** * Initialize the network function, including hardware configuration, memory * allocation for data structures, MAC address setup, and interrupt enabling. @@ -3785,6 +3823,25 @@ hinic3_func_init(struct rte_eth_dev *eth_dev) PMD_DRV_LOG(INFO, "Initialize %s in secondary process", eth_dev->data->name); + char name[RTE_ETH_NAME_MAX_LEN]; + snprintf(name, sizeof(name), "%s", eth_dev->data->name); + eth_dev = rte_eth_dev_attach_secondary(name); + if (eth_dev == NULL) { + PMD_DRV_LOG(ERR, "can not attach rte ethdev, dev_name: %s", name); + return -ENOMEM; + } + + nic_dev = HINIC3_ETH_DEV_TO_PRIVATE_NIC_DEV(eth_dev); + if (nic_dev == NULL) { + PMD_DRV_LOG(ERR, "nic_dev hwdev is NULL, dev_name: %s", name); + return -ENOMEM; + } + + if (HINIC3_FUNC_TYPE(nic_dev->hwdev) == TYPE_VF) + eth_dev->dev_ops = &hinic3_pmd_vf_ops; + else + eth_dev->dev_ops = &hinic3_pmd_ops; + return 0; } @@ -3806,13 +3863,6 @@ hinic3_func_init(struct rte_eth_dev *eth_dev) goto alloc_eth_addr_fail; } - nic_dev->tx_ops = rte_zmalloc("tx_ops", sizeof(struct hinic3_nic_tx_ops), 0); - if (!nic_dev->tx_ops) { - PMD_DRV_LOG(ERR, "Allocate tx_ops memory failed"); - err = -ENOMEM; - goto alloc_tx_ops_fail; - } - nic_dev->mc_list = rte_zmalloc("hinic3_mc", HINIC3_MAX_MC_MAC_ADDRS * sizeof(struct rte_ether_addr), 0); if (!nic_dev->mc_list) { @@ -3866,8 +3916,6 @@ hinic3_func_init(struct rte_eth_dev *eth_dev) goto get_cap_fail; } - hinic3_nic_tx_rx_ops_init(nic_dev); - /* Read wqe type from kernel parameter. */ if (hinic3_read_module_param(nic_dev, "rq_wqe_type", &compact_cqe) != 0) { err = -EINVAL; @@ -3970,10 +4018,6 @@ hinic3_func_init(struct rte_eth_dev *eth_dev) nic_dev->mc_list = NULL; alloc_mc_list_fail: - rte_free(nic_dev->tx_ops); - nic_dev->tx_ops = NULL; - -alloc_tx_ops_fail: rte_free(eth_dev->data->mac_addrs); eth_dev->data->mac_addrs = NULL; @@ -4113,12 +4157,13 @@ hinic3_dev_init(struct rte_eth_dev *eth_dev) hinic3_nic_feature_init(nic_dev); - if (nic_dev->config.rx_cqe_compact_en != HINIC3_RX_CQE_COMPACT_EN) + if (nic_dev->config.rx_cqe_compact_en != HINIC3_RX_CQE_COMPACT_EN) { eth_dev->rx_pkt_burst = hinic3_recv_pkts; - else + eth_dev->tx_pkt_burst = hinic3_xmit_pkts; + } else { eth_dev->rx_pkt_burst = hinic3_recv_pkts_compact_cqe; - - eth_dev->tx_pkt_burst = hinic3_xmit_pkts; + eth_dev->tx_pkt_burst = hinic3_xmit_pkts_compact_cqe; + } return err; } diff --git a/drivers/net/hinic3/hinic3_ethdev.h b/drivers/net/hinic3/hinic3_ethdev.h index 41f6738645..399c9c9342 100644 --- a/drivers/net/hinic3/hinic3_ethdev.h +++ b/drivers/net/hinic3/hinic3_ethdev.h @@ -15,6 +15,11 @@ #define HINIC3_PMD_DRV_VERSION "B106" +#define HINIC3_MAX_QUEUE_DEPTH 16384 +#define HINIC3_MIN_QUEUE_DEPTH 128 + +#define HINIC3_QUEUE_STAT_CNTRS 256 + #define PCI_DEV_TO_INTR_HANDLE(pci_dev) ((pci_dev)->intr_handle) #define HINIC3_PKT_RX_L4_CKSUM_BAD RTE_MBUF_F_RX_L4_CKSUM_BAD @@ -112,7 +117,7 @@ enum nic_feature_cap { }; -#define DEFAULT_DRV_FEATURE 0x3FC3FFF +#define DEFAULT_DRV_FEATURE 0x0BFC3FFF TAILQ_HEAD(hinic3_ethertype_filter_list, rte_flow); TAILQ_HEAD(hinic3_fdir_rule_filter_list, rte_flow); @@ -152,6 +157,7 @@ struct hinic3_nic_dev { uint16_t max_rqs; uint16_t rxq_depth; + uint16_t txq_depth; uint16_t rx_buff_len; uint16_t mtu_size; @@ -189,7 +195,6 @@ struct hinic3_nic_dev { struct hinic3_fdir_rule_filter_list filter_fdir_rule_list; uint8_t cos_map[HINIC3_COS_NUM_MAX]; struct hinic3_ptype_table *ptype_tbl; - struct hinic3_nic_tx_ops *tx_ops; uint32_t fec_mode; /**< Current FEC mode for ethdev. */ enum nic_type nic_type; /**< NIC type. */ @@ -204,6 +209,7 @@ struct hinic3_nic_dev { extern const struct rte_flow_ops hinic3_flow_ops; +bool is_sp230_pci_dev(struct rte_pci_device *pci_dev); /** * Enable interrupt for the specified RX queue. * diff --git a/drivers/net/hinic3/hinic3_tx.c b/drivers/net/hinic3/hinic3_tx.c index 566e22315d..c3f053f869 100644 --- a/drivers/net/hinic3/hinic3_tx.c +++ b/drivers/net/hinic3/hinic3_tx.c @@ -78,6 +78,30 @@ hinic3_get_and_update_sq_owner(struct hinic3_txq *sq, uint16_t curr_pi, uint16_t return owner; } +static void * +hinic3_get_sq_wqe(struct hinic3_txq *sq, + struct hinic3_wqe_info *wqe_info) +{ + uint16_t cur_pi = MASKED_QUEUE_IDX(sq, sq->prod_idx); + uint32_t end_pi; + + end_pi = cur_pi + wqe_info->wqebb_cnt; + sq->prod_idx += wqe_info->wqebb_cnt; + + wqe_info->owner = (uint8_t)(sq->owner); + wqe_info->pi = cur_pi; + wqe_info->wrapped = 0; + + if (unlikely(end_pi >= sq->q_depth)) { + sq->owner = !sq->owner; + + if (likely(end_pi > sq->q_depth)) + wqe_info->wrapped = (uint8_t)(sq->q_depth - cur_pi); + } + + return NIC_WQE_ADDR(sq, cur_pi); +} + static inline void hinic3_put_sq_wqe(struct hinic3_txq *sq, struct hinic3_wqe_info *wqe_info) { @@ -98,9 +122,9 @@ hinic3_put_sq_wqe(struct hinic3_txq *sq, struct hinic3_wqe_info *wqe_info) * Point to wqe_info of send queue(SQ). */ static void -hinic3_set_wqe_combo(struct hinic3_txq *sq, - struct hinic3_sq_wqe_combo *wqe_combo, - struct hinic3_wqe_info *wqe_info) +hinic3_set_wqe_combo_compact_cqe(struct hinic3_txq *sq, + struct hinic3_sq_wqe_combo *wqe_combo, + struct hinic3_wqe_info *wqe_info) { uint16_t tmp_pi; @@ -125,6 +149,62 @@ hinic3_set_wqe_combo(struct hinic3_txq *sq, wqe_info->owner = hinic3_get_and_update_sq_owner(sq, wqe_info->pi, wqe_info->wqebb_cnt); } +/** + * Sets the WQE combination information in the transmit queue (SQ). + * + * @param[in] sq + * Point to send queue. + * @param[out] wqe_combo + * Point to wqe_combo of send queue(SQ). + * @param[in] wqe_info + * Point to wqe_info of send queue(SQ). + */ +static void +hinic3_set_wqe_combo(struct hinic3_txq *sq, + struct hinic3_sq_wqe_combo *wqe_combo, + struct hinic3_sq_wqe *wqe, + struct hinic3_wqe_info *wqe_info) +{ + wqe_combo->hdr = &wqe->compact_wqe.wqe_desc; + + if (wqe_info->offload) { + if (wqe_info->wrapped == HINIC3_TX_TASK_WRAPPED) { + wqe_combo->task = (struct hinic3_sq_task *) + (void *)sq->sq_head_addr; + wqe_combo->bds_head = (struct hinic3_sq_bufdesc *) + (void *)(sq->sq_head_addr + sq->wqebb_size); + } else if (wqe_info->wrapped == HINIC3_TX_BD_DESC_WRAPPED) { + wqe_combo->task = &wqe->extend_wqe.task; + wqe_combo->bds_head = (struct hinic3_sq_bufdesc *) + (void *)(sq->sq_head_addr); + } else { + wqe_combo->task = &wqe->extend_wqe.task; + wqe_combo->bds_head = wqe->extend_wqe.buf_desc; + } + + wqe_combo->wqe_type = SQ_WQE_EXTENDED_TYPE; + wqe_combo->task_type = SQ_WQE_TASKSECT_16BYTES; + return; + } + + if (wqe_info->wrapped == HINIC3_TX_TASK_WRAPPED) { + wqe_combo->bds_head = (struct hinic3_sq_bufdesc *) + (void *)(sq->sq_head_addr); + } else { + wqe_combo->bds_head = + (struct hinic3_sq_bufdesc *)(&wqe->extend_wqe.task); + } + + if (wqe_info->wqebb_cnt > 1) { + wqe_combo->wqe_type = SQ_WQE_EXTENDED_TYPE; + wqe_combo->task_type = SQ_WQE_TASKSECT_4BYTES; + /* This section used as vlan insert, needs to clear */ + wqe_combo->bds_head->rsvd = 0; + } else { + wqe_combo->wqe_type = SQ_WQE_COMPACT_TYPE; + } +} + int hinic3_start_all_sqs(struct rte_eth_dev *eth_dev) { @@ -344,33 +424,15 @@ hinic3_tx_offload_pkt_prepare(struct hinic3_nic_dev *nic_dev, struct rte_mbuf *m return 0; } -void -hinic3_tx_set_normal_task_offload(struct hinic3_wqe_info *wqe_info, - struct hinic3_sq_wqe_combo *wqe_combo) +static inline void hinic3_set_vlan_tx_offload(struct hinic3_sq_task *task, + uint16_t vlan_tag, uint8_t vlan_type) { - struct hinic3_sq_task *task = wqe_combo->task; - struct hinic3_offload_info *offload_info = &wqe_info->offload_info; - - task->pkt_info0 = 0; - task->pkt_info0 |= SQ_TASK_INFO0_SET(offload_info->inner_l4_en, INNER_L4_EN); - task->pkt_info0 |= SQ_TASK_INFO0_SET(offload_info->inner_l3_en, INNER_L3_EN); - task->pkt_info0 |= SQ_TASK_INFO0_SET(offload_info->encapsulation, TUNNEL_FLAG); - task->pkt_info0 |= SQ_TASK_INFO0_SET(offload_info->out_l3_en, OUT_L3_EN); - task->pkt_info0 |= SQ_TASK_INFO0_SET(offload_info->out_l4_en, OUT_L4_EN); - task->pkt_info0 = hinic3_hw_be32(task->pkt_info0); - - if (wqe_combo->task_type == SQ_WQE_TASKSECT_16BYTES) { - task->ip_identify = 0; - task->pkt_info2 = 0; - task->vlan_offload = 0; - task->vlan_offload = SQ_TASK_INFO3_SET(offload_info->vlan_tag, VLAN_TAG) | - SQ_TASK_INFO3_SET(offload_info->vlan_sel, VLAN_TYPE) | - SQ_TASK_INFO3_SET(offload_info->vlan_valid, VLAN_TAG_VALID); - task->vlan_offload = hinic3_hw_be32(task->vlan_offload); - } + task->vlan_offload = SQ_TASK_INFO3_SET(vlan_tag, VLAN_TAG) | + SQ_TASK_INFO3_SET(vlan_type, VLAN_TYPE) | + SQ_TASK_INFO3_SET(1U, VLAN_TAG_VALID); } -void +static void hinic3_tx_set_compact_task_offload(struct hinic3_wqe_info *wqe_info, struct hinic3_sq_wqe_combo *wqe_combo) { @@ -391,10 +453,9 @@ hinic3_tx_set_compact_task_offload(struct hinic3_wqe_info *wqe_info, } static int -hinic3_set_tx_offload(struct hinic3_nic_dev *nic_dev, - struct rte_mbuf *mbuf, - struct hinic3_sq_wqe_combo *wqe_combo, - struct hinic3_wqe_info *wqe_info) +hinic3_set_tx_offload_compact_cqe(struct rte_mbuf *mbuf, + struct hinic3_sq_wqe_combo *wqe_combo, + struct hinic3_wqe_info *wqe_info) { uint64_t ol_flags = mbuf->ol_flags; struct hinic3_offload_info *offload_info = &wqe_info->offload_info; @@ -443,6 +504,9 @@ hinic3_set_tx_offload(struct hinic3_nic_dev *nic_dev, offload_info->encapsulation = 1; wqe_info->queue_info.udp_dp_en = 1; break; + case HINIC3_PKT_TX_TUNNEL_IPIP: + offload_info->encapsulation = 1; + break; case 0: break; @@ -458,8 +522,110 @@ hinic3_set_tx_offload(struct hinic3_nic_dev *nic_dev, offload_info->out_l4_en = 1; set_tx_wqe_offload: - nic_dev->tx_ops->nic_tx_set_wqe_offload(wqe_info, wqe_combo); + hinic3_tx_set_compact_task_offload(wqe_info, wqe_combo); + + return 0; +} + +static int +hinic3_set_tx_offload(struct rte_mbuf *mbuf, + struct hinic3_sq_task *task, + struct hinic3_wqe_info *wqe_info, + enum nic_type nic_type) +{ + uint64_t ol_flags = mbuf->ol_flags; + uint16_t pld_offset = 0; + uint32_t queue_info = 0; + uint16_t vlan_tag; + + task->pkt_info0 = 0; + task->ip_identify = 0; + task->pkt_info2 = 0; + task->vlan_offload = 0; + + /* Vlan offload */ + if (unlikely(ol_flags & HINIC3_PKT_TX_VLAN_PKT)) { + vlan_tag = mbuf->vlan_tci; + hinic3_set_vlan_tx_offload(task, vlan_tag, HINIC3_TX_TPID0); + task->vlan_offload = hinic3_hw_be32(task->vlan_offload); + } + + if (!(ol_flags & HINIC3_TX_CKSUM_OFFLOAD_MASK)) + return 0; + + /* Tso offload */ + if (ol_flags & HINIC3_PKT_TX_TCP_SEG) { + if (((ol_flags & HINIC3_PKT_TX_TUNNEL_IPIP) == HINIC3_PKT_TX_TUNNEL_IPIP) && + nic_type == NIC_SP620) { + PMD_DRV_LOG(ERR, "IPinIP not support TSO"); + return -EINVAL; + } + if ((ol_flags & HINIC3_PKT_TX_TUNNEL_MASK) == HINIC3_PKT_TX_TUNNEL_VXLAN_GPE) { + PMD_DRV_LOG(ERR, "VXLAN_GPE not support TSO"); + return -EINVAL; + } + pld_offset = wqe_info->payload_offset; + if ((pld_offset >> 1) > MAX_PAYLOAD_OFFSET) + return -EINVAL; + + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, INNER_L4_EN); + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, INNER_L3_EN); + + queue_info |= SQ_CTRL_QUEUE_INFO_SET(1U, TSO); + queue_info |= SQ_CTRL_QUEUE_INFO_SET(pld_offset >> 1, PLDOFF); + + /* Set MSS value */ + queue_info = SQ_CTRL_QUEUE_INFO_CLEAR(queue_info, MSS); + queue_info |= SQ_CTRL_QUEUE_INFO_SET(mbuf->tso_segsz, MSS); + } else { + if (ol_flags & HINIC3_PKT_TX_IP_CKSUM) + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, INNER_L3_EN); + + switch (ol_flags & HINIC3_PKT_TX_L4_MASK) { + case HINIC3_PKT_TX_TCP_CKSUM: + case HINIC3_PKT_TX_UDP_CKSUM: + case HINIC3_PKT_TX_SCTP_CKSUM: + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, INNER_L4_EN); + break; + case HINIC3_PKT_TX_L4_NO_CKSUM: + break; + default: + PMD_DRV_LOG(INFO, "not support pkt type"); + return -EINVAL; + } + } + + /* For vxlan, also can support PKT_TX_TUNNEL_GENEVE/GRE, etc */ + switch (ol_flags & HINIC3_PKT_TX_TUNNEL_MASK) { + case HINIC3_PKT_TX_TUNNEL_VXLAN: + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, TUNNEL_FLAG); + break; + case HINIC3_PKT_TX_TUNNEL_VXLAN_GPE: + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, TUNNEL_FLAG); + break; + case HINIC3_PKT_TX_TUNNEL_GENEVE: + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, TUNNEL_FLAG); + break; + case HINIC3_PKT_TX_TUNNEL_IPIP: + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, TUNNEL_FLAG); + break; + case 0: + break; + default: + /* For non UDP/GRE tunneling, drop the tunnel packet */ + PMD_DRV_LOG(INFO, "not support tunnel pkt type"); + return -EINVAL; + } + + if (ol_flags & HINIC3_PKT_TX_OUTER_IP_CKSUM) + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, OUT_L3_EN); + + if (ol_flags & HINIC3_PKT_TX_OUTER_UDP_CKSUM) + task->pkt_info0 |= SQ_TASK_INFO0_SET(1U, OUT_L4_EN); + task->pkt_info0 = hinic3_hw_be32(task->pkt_info0); + task->pkt_info2 = hinic3_hw_be32(task->pkt_info2); + wqe_info->queue_info_sp600 = queue_info; return 0; } @@ -602,6 +768,8 @@ hinic3_non_tso_pkt_pre_process(struct rte_mbuf *mbuf, struct hinic3_wqe_info *wq * Point to the mbuf to send. * @param[in] wqe_info * Point to wqe_info of send queue(SQ). + * @param[in] non_tso_max_pkt_len + * Number of non_tso_max_pkt_len(SP230 is 9600, SP560/620 is 4K) * @return * 0 as success, -EINVAL as failure. */ @@ -806,6 +974,14 @@ hinic3_mbuf_dma_map_sge(struct hinic3_txq *txq, struct rte_mbuf *mbuf, hinic3_set_buf_desc(buf_desc, dma_addr, mbuf->data_len); buf_desc++; } + + /* + * SP620: For wqe compact type, no need to prepare + * sq ctrl info. Need set queue_info = 0. + */ + if (txq->nic_dev->nic_type == NIC_SP620) + wqe_desc->queue_info = 0; + mbuf = mbuf->next; } @@ -853,8 +1029,8 @@ hinic3_mbuf_dma_map_sge(struct hinic3_txq *txq, struct rte_mbuf *mbuf, * Point to wqe_info of send queue. */ static void -hinic3_prepare_sq_ctrl(struct hinic3_sq_wqe_combo *wqe_combo, - struct hinic3_wqe_info *wqe_info) +hinic3_prepare_sq_ctrl_compact_cqe(struct hinic3_sq_wqe_combo *wqe_combo, + struct hinic3_wqe_info *wqe_info) { struct hinic3_queue_info *queue_info = &wqe_info->queue_info; struct hinic3_sq_wqe_desc *wqe_desc = wqe_combo->hdr; @@ -867,7 +1043,7 @@ hinic3_prepare_sq_ctrl(struct hinic3_sq_wqe_combo *wqe_combo, if (wqe_combo->wqe_type == SQ_WQE_EXTENDED_TYPE) { wqe_desc->ctrl_len |= SQ_CTRL_SET(wqe_info->sge_cnt, BUFDESC_NUM) | SQ_CTRL_SET(wqe_combo->task_type, TASKSECT_LEN) | - SQ_CTRL_SET(SQ_NORMAL_WQE, DATA_FORMAT); + SQ_CTRL_SET(SQ_WQE_SGL, DATA_FORMAT); *qsf = SQ_CTRL_QUEUE_INFO_SET(1, UC) | SQ_CTRL_QUEUE_INFO_SET(queue_info->sctp, SCTP) | @@ -888,14 +1064,56 @@ hinic3_prepare_sq_ctrl(struct hinic3_sq_wqe_combo *wqe_combo, *qsf = hinic3_hw_be32(*qsf); } else { wqe_desc->ctrl_len |= SQ_CTRL_COMPACT_QUEUE_INFO_SET(queue_info->sctp, SCTP) | - SQ_CTRL_COMPACT_QUEUE_INFO_SET(queue_info->udp_dp_en, UDP_DP_EN) | - SQ_CTRL_COMPACT_QUEUE_INFO_SET(queue_info->ufo, UFO) | - SQ_CTRL_COMPACT_QUEUE_INFO_SET(queue_info->pkt_type, PKT_TYPE); + SQ_CTRL_COMPACT_QUEUE_INFO_SET(queue_info->udp_dp_en, + UDP_DP_EN) | + SQ_CTRL_COMPACT_QUEUE_INFO_SET(queue_info->ufo, UFO) | + SQ_CTRL_COMPACT_QUEUE_INFO_SET(queue_info->pkt_type, + PKT_TYPE); } wqe_desc->ctrl_len = hinic3_hw_be32(wqe_desc->ctrl_len); } +/** + * Sets and configures fields in the transmit queue control descriptor based on + * the WQE type. + * + * @param[out] wqe_combo + * Point to wqe_combo of send queue. + * @param[in] wqe_info + * Point to wqe_info of send queue. + */ +static void +hinic3_prepare_sq_ctrl(struct hinic3_sq_wqe_combo *wqe_combo, + struct hinic3_wqe_info *wqe_info) +{ + struct hinic3_sq_wqe_desc *wqe_desc = wqe_combo->hdr; + + wqe_desc->ctrl_len |= SQ_CTRL_SET(wqe_info->sge_cnt, BUFDESC_NUM) | + SQ_CTRL_SET(wqe_combo->task_type, TASKSECT_LEN) | + SQ_CTRL_SET(SQ_NORMAL_WQE, DATA_FORMAT) | + SQ_CTRL_SET(wqe_combo->wqe_type, EXTENDED) | + SQ_CTRL_SET(wqe_info->owner, OWNER); + + wqe_desc->ctrl_len = hinic3_hw_be32(wqe_desc->ctrl_len); + + wqe_desc->queue_info = wqe_info->queue_info_sp600; + wqe_desc->queue_info |= SQ_CTRL_QUEUE_INFO_SET(1U, UC); + + if (!SQ_CTRL_QUEUE_INFO_GET(wqe_desc->queue_info, MSS)) { + wqe_desc->queue_info |= + SQ_CTRL_QUEUE_INFO_SET(TX_MSS_DEFAULT, MSS); + } else if (SQ_CTRL_QUEUE_INFO_GET(wqe_desc->queue_info, MSS) < + TX_MSS_MIN) { + /* Mss should not less than 80 */ + wqe_desc->queue_info = + SQ_CTRL_QUEUE_INFO_CLEAR(wqe_desc->queue_info, MSS); + wqe_desc->queue_info |= SQ_CTRL_QUEUE_INFO_SET(TX_MSS_MIN, MSS); + } + + wqe_desc->queue_info = hinic3_hw_be32(wqe_desc->queue_info); +} + /** * It is responsible for sending data packets. * @@ -912,13 +1130,16 @@ uint16_t hinic3_xmit_pkts(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkts) { struct hinic3_txq *txq = tx_queue; + struct hinic3_nic_dev *nic_dev = txq->nic_dev; struct hinic3_tx_info *tx_info = NULL; struct rte_mbuf *mbuf_pkt = NULL; struct hinic3_sq_wqe_combo wqe_combo = {0}; + struct hinic3_sq_wqe *sq_wqe = NULL; struct hinic3_wqe_info wqe_info = {0}; uint32_t offload_err, free_cnt; uint64_t tx_bytes = 0; uint16_t free_wqebb_cnt, nb_tx; + uint64_t tx_free_loop = 0; int err; #ifdef HINIC3_XSTAT_PROF_TX @@ -929,6 +1150,7 @@ hinic3_xmit_pkts(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkts) if (unlikely(!HINIC3_TXQ_IS_STARTED(txq))) return 0; + rte_prefetch0(tx_pkts[0]); free_cnt = txq->tx_free_thresh; /* Reclaim tx mbuf before xmit new packets. */ if (hinic3_get_sq_free_wqebbs(txq) < txq->tx_free_thresh) @@ -936,53 +1158,64 @@ hinic3_xmit_pkts(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkts) /* Tx loop routine. */ for (nb_tx = 0; nb_tx < nb_pkts; nb_tx++) { - mbuf_pkt = *tx_pkts++; + mbuf_pkt = *tx_pkts; + rte_prefetch0(++tx_pkts); if (unlikely(hinic3_get_tx_offload(txq->nic_dev, mbuf_pkt, &wqe_info, txq->non_tso_max_pkt_len))) { txq->txq_stats.offload_errors++; break; } - wqe_info.wqebb_cnt = wqe_info.sge_cnt; - if (likely(wqe_info.offload || wqe_info.wqebb_cnt > 1)) { - if (txq->tx_wqe_compact_task) { - /** - * One more wqebb is needed for compact task under two situations: - * 1. TSO: MSS field is needed, no available space for - * compact task in compact wqe. - * 2. SGE number > 1: wqe is handlerd as extended wqe by nic. - */ - if (mbuf_pkt->ol_flags & HINIC3_PKT_TX_TCP_SEG || - wqe_info.wqebb_cnt > 1) - wqe_info.wqebb_cnt++; - } else { - /* Use extended sq wqe with normal TS */ - wqe_info.wqebb_cnt++; - } - } + if (!wqe_info.offload) + /* + * Use extended sq wqe with small TS, which can include + * multi sges, or compact sq normal wqe, which just + * supports one sge + */ + wqe_info.wqebb_cnt = wqe_info.sge_cnt; + else + /* Use extended sq wqe with normal TS */ + wqe_info.wqebb_cnt = wqe_info.sge_cnt + 1; free_wqebb_cnt = hinic3_get_sq_free_wqebbs(txq); - if (unlikely(wqe_info.wqebb_cnt > free_wqebb_cnt)) { - /* Reclaim again. */ + while (wqe_info.wqebb_cnt > free_wqebb_cnt) { + /* + * Try to reclaim completed Tx WQEs when SQ space is + * insufficient. This gives hardware a chance to free + * descriptors and helps prevent packet drops caused + * by transient SQ congestion. + */ hinic3_xmit_mbuf_cleanup(txq, free_cnt); free_wqebb_cnt = hinic3_get_sq_free_wqebbs(txq); - if (unlikely(wqe_info.wqebb_cnt > free_wqebb_cnt)) { + if ((tx_free_loop++) > nic_dev->config.tx_free_loop) { txq->txq_stats.tx_busy += (nb_pkts - nb_tx); - break; + goto end; } } - /* Task or bd section maybe wrapped for one wqe. */ - hinic3_set_wqe_combo(txq, &wqe_combo, &wqe_info); + /* Get sq wqe address from wqe_page */ + sq_wqe = hinic3_get_sq_wqe(txq, &wqe_info); + if (unlikely(!sq_wqe)) { + txq->txq_stats.tx_busy++; + break; + } - /* Fill tx packet offload into qsf and task field. */ - offload_err = hinic3_set_tx_offload(txq->nic_dev, mbuf_pkt, &wqe_combo, &wqe_info); + /* Task or bd section maybe wrapped for one wqe */ + hinic3_set_wqe_combo(txq, &wqe_combo, sq_wqe, &wqe_info); + + wqe_info.queue_info_sp600 = 0; + /* Fill tx packet offload into qsf and task field */ + if (wqe_info.offload) { + offload_err = hinic3_set_tx_offload(mbuf_pkt, + wqe_combo.task, + &wqe_info, + txq->nic_dev->nic_type); if (unlikely(offload_err)) { hinic3_put_sq_wqe(txq, &wqe_info); txq->txq_stats.offload_errors++; break; } - + } /* Fill sq_wqe buf_desc and bd_desc. */ err = hinic3_mbuf_dma_map_sge(txq, mbuf_pkt, &wqe_combo, &wqe_info); @@ -1006,7 +1239,7 @@ hinic3_xmit_pkts(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkts) tx_bytes += mbuf_pkt->pkt_len; } - +end: /* Update txq stats. */ if (nb_tx) { hinic3_write_db(txq->db_addr, txq->q_id, (int)(txq->cos), @@ -1081,3 +1314,166 @@ hinic3_flush_txqs(struct hinic3_nic_dev *nic_dev) PMD_DRV_LOG(ERR, "Stop sq%d failed", qid); } } + +int +hinic3_tx_burst_mode_get(struct rte_eth_dev *dev, + uint16_t tx_queue_id, + struct rte_eth_burst_mode *mode) +{ + uint16_t tx_offloads = dev->data->dev_conf.txmode.offloads; + struct hinic3_nic_dev *nic_dev; + struct hinic3_txq *txq = NULL; + + nic_dev = HINIC3_ETH_DEV_TO_PRIVATE_NIC_DEV(dev); + txq = nic_dev->txqs[tx_queue_id]; + if (!txq) + return -EINVAL; + + snprintf(mode->info, sizeof(mode->info), + "Scalar%s%s%s%s%s", + (tx_offloads & RTE_ETH_TX_OFFLOAD_MULTI_SEGS) ? " + MULTI" : "", + (tx_offloads & (RTE_ETH_TX_OFFLOAD_TCP_TSO | + RTE_ETH_TX_OFFLOAD_VXLAN_TNL_TSO)) ? " + TSO" : "", + (tx_offloads & (RTE_ETH_TX_OFFLOAD_IPV4_CKSUM | + RTE_ETH_TX_OFFLOAD_UDP_CKSUM | + RTE_ETH_TX_OFFLOAD_TCP_CKSUM | + RTE_ETH_TX_OFFLOAD_SCTP_CKSUM | + RTE_ETH_TX_OFFLOAD_OUTER_IPV4_CKSUM)) ? " + CKSUM" : "", + (tx_offloads & RTE_ETH_TX_OFFLOAD_VLAN_INSERT) ? " + VLAN" : "", + (tx_offloads & RTE_ETH_TX_OFFLOAD_QINQ_INSERT) ? " + QINQ" : ""); + + return 0; +} + +/** + * It is responsible for sending data packets. + * + * @param[in] tx_queue + * Point to send queue. + * @param[in] tx_pkts + * Pointer to the array of data packets to be sent. + * @param[in] nb_pkts + * Number of sent packets. + * @return + * Number of actually sent packets. + */ +uint16_t +hinic3_xmit_pkts_compact_cqe(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkts) +{ + struct hinic3_txq *txq = tx_queue; + struct hinic3_nic_dev *nic_dev = txq->nic_dev; + struct hinic3_tx_info *tx_info = NULL; + struct rte_mbuf *mbuf_pkt = NULL; + struct hinic3_sq_wqe_combo wqe_combo = {0}; + struct hinic3_wqe_info wqe_info = {0}; + uint32_t offload_err, free_cnt; + uint64_t tx_bytes = 0; + uint16_t free_wqebb_cnt, nb_tx; + uint64_t tx_free_loop = 0; + int err; + +#ifdef HINIC3_XSTAT_PROF_TX + uint64_t t1, t2; + t1 = rte_get_tsc_cycles(); +#endif + + if (unlikely(!HINIC3_TXQ_IS_STARTED(txq))) + return 0; + + free_cnt = txq->tx_free_thresh; + /* Reclaim tx mbuf before xmit new packets. */ + if (hinic3_get_sq_free_wqebbs(txq) < txq->tx_free_thresh) + hinic3_xmit_mbuf_cleanup(txq, free_cnt); + + /* Tx loop routine. */ + for (nb_tx = 0; nb_tx < nb_pkts; nb_tx++) { + mbuf_pkt = *tx_pkts++; + if (unlikely(hinic3_get_tx_offload(nic_dev, mbuf_pkt, &wqe_info, + txq->non_tso_max_pkt_len))) { + txq->txq_stats.offload_errors++; + break; + } + + wqe_info.wqebb_cnt = wqe_info.sge_cnt; + if (likely(wqe_info.offload || wqe_info.wqebb_cnt > 1)) { + if (txq->tx_wqe_compact_task) { + /** + * One more wqebb is needed for compact task under two situations: + * 1. TSO: MSS field is needed, no available space for + * compact task in compact wqe. + * 2. SGE number > 1: wqe is handlerd as extended wqe by nic. + */ + if (mbuf_pkt->ol_flags & HINIC3_PKT_TX_TCP_SEG || + wqe_info.wqebb_cnt > 1) + wqe_info.wqebb_cnt++; + } else { + /* Use extended sq wqe with normal TS */ + wqe_info.wqebb_cnt++; + } + } + + free_wqebb_cnt = hinic3_get_sq_free_wqebbs(txq); + while (wqe_info.wqebb_cnt > free_wqebb_cnt) { + /* Reclaim again */ + hinic3_xmit_mbuf_cleanup(txq, free_cnt); + free_wqebb_cnt = hinic3_get_sq_free_wqebbs(txq); + if ((tx_free_loop++) > nic_dev->config.tx_free_loop) { + txq->txq_stats.tx_busy += (nb_pkts - nb_tx); + goto end; + } + } + + /* Task or bd section maybe wrapped for one wqe. */ + hinic3_set_wqe_combo_compact_cqe(txq, &wqe_combo, &wqe_info); + + /* Fill tx packet offload into qsf and task field. */ + offload_err = hinic3_set_tx_offload_compact_cqe(mbuf_pkt, &wqe_combo, &wqe_info); + if (unlikely(offload_err)) { + hinic3_put_sq_wqe(txq, &wqe_info); + txq->txq_stats.offload_errors++; + break; + } + + /* Fill sq_wqe buf_desc and bd_desc. */ + err = hinic3_mbuf_dma_map_sge(txq, mbuf_pkt, &wqe_combo, &wqe_info); + + if (err) { + hinic3_put_sq_wqe(txq, &wqe_info); + txq->txq_stats.offload_errors++; + break; + } + + /* Record tx info. */ + tx_info = &txq->tx_info[wqe_info.pi]; + tx_info->mbuf = mbuf_pkt; + tx_info->wqebb_cnt = wqe_info.wqebb_cnt; + + /* + * For wqe compact type, no need to prepare + * sq ctrl info. + */ + hinic3_prepare_sq_ctrl_compact_cqe(&wqe_combo, &wqe_info); + + tx_bytes += mbuf_pkt->pkt_len; + } +end: + /* Update txq stats. */ + if (nb_tx) { + hinic3_write_db(txq->db_addr, txq->q_id, (int)(txq->cos), + SQ_CFLAG_DP, + MASKED_QUEUE_IDX(txq, txq->prod_idx)); + txq->txq_stats.packets += nb_tx; + txq->txq_stats.bytes += tx_bytes; + } + txq->txq_stats.burst_pkts = nb_tx; + +#ifdef HINIC3_XSTAT_PROF_TX + t2 = rte_get_tsc_cycles(); + txq->txq_stats.app_tsc = t1 - txq->prof_tx_end_tsc; + txq->prof_tx_end_tsc = t2; + txq->txq_stats.pmd_tsc = t2 - t1; + txq->txq_stats.burst_pkts = nb_tx; +#endif + + return nb_tx; +} diff --git a/drivers/net/hinic3/hinic3_tx.h b/drivers/net/hinic3/hinic3_tx.h index 5d4101630e..10ea174ea2 100644 --- a/drivers/net/hinic3/hinic3_tx.h +++ b/drivers/net/hinic3/hinic3_tx.h @@ -18,6 +18,26 @@ (HINIC3_NONTSO_PKT_MAX_SGE - HINIC3_NON_COPY_SGE_NUM) #define HINIC3_TSO_MBUF_NUM_MAX (HINIC3_TSO_PKT_MAX_SGE - HINIC3_NON_COPY_SGE_NUM) +/* Tx offload info */ +struct hinic3_tx_offload_info { + uint8_t outer_l2_len; + uint8_t outer_l3_type; + uint16_t outer_l3_len; + + uint8_t inner_l2_len; + uint8_t inner_l3_type; + uint16_t inner_l3_len; + + uint8_t tunnel_length; + uint8_t tunnel_type; + uint8_t inner_l4_type; + uint8_t inner_l4_len; + + uint16_t payload_offset; + uint8_t inner_l4_tcp_udp; + uint8_t rsvd0; +}; + /* Tx wqe queue info */ struct hinic3_queue_info { uint8_t pri; @@ -55,10 +75,10 @@ struct hinic3_wqe_info { uint16_t sge_cnt; uint8_t offload; - uint8_t rsvd0; /**< Reserved field 0. */ + uint8_t wrapped; uint16_t payload_offset; - uint8_t rsvd1; /**< Reserved field 1. */ + uint32_t queue_info_sp600; uint8_t owner; uint16_t pi; @@ -381,6 +401,7 @@ struct __rte_cache_aligned hinic3_txq { uint64_t sq_bot_sge_addr; uint32_t cos; uint8_t tx_wqe_compact_task; + uint16_t tx_free_loop; uint32_t non_tso_max_pkt_len; struct hinic3_txq_stats txq_stats; #ifdef HINIC3_XSTAT_PROF_TX @@ -388,41 +409,14 @@ struct __rte_cache_aligned hinic3_txq { #endif }; -/* Tx WQE offload set callback function */ -typedef void (*nic_tx_set_wqe_offload_t)(struct hinic3_wqe_info *wqe_info, - struct hinic3_sq_wqe_combo *wqe_combo); - -struct hinic3_nic_tx_ops { - nic_tx_set_wqe_offload_t nic_tx_set_wqe_offload; -}; - void hinic3_flush_txqs(struct hinic3_nic_dev *nic_dev); void hinic3_free_txq_mbufs(struct hinic3_txq *txq); void hinic3_free_all_txq_mbufs(struct hinic3_nic_dev *nic_dev); +uint16_t hinic3_xmit_pkts_compact_cqe(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkts); uint16_t hinic3_xmit_pkts(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkts); int hinic3_stop_sq(struct hinic3_txq *txq); int hinic3_start_all_sqs(struct rte_eth_dev *eth_dev); int hinic3_tx_done_cleanup(void *txq, uint32_t free_cnt); - -/** - * Set wqe task section - * - * @param[in] wqe_info - * Packet info parsed according to mbuf - * @param[in] wqe_combo - * Wqe need to format - */ -void hinic3_tx_set_normal_task_offload(struct hinic3_wqe_info *wqe_info, - struct hinic3_sq_wqe_combo *wqe_combo); - -/** - * Set compact wqe task section - * - * @param[in] wqe_info - * Packet info parsed according to mbuf - * @param[in] wqe_combo - * Wqe need to format - */ -void hinic3_tx_set_compact_task_offload(struct hinic3_wqe_info *wqe_info, - struct hinic3_sq_wqe_combo *wqe_combo); +int hinic3_tx_burst_mode_get(struct rte_eth_dev *dev, uint16_t tx_queue_id, + struct rte_eth_burst_mode *mode); #endif /**< _HINIC3_TX_H_ */ -- 2.33.0

