1. Rework the Tx completion flush to walk descriptors one at a time and free each completed segment with rte_pktmbuf_free_seg(), add prefetch hints on the next descriptor cache line, switch the ring-held mbufs in zxdh_dev_free_mbufs() to rte_pktmbuf_free_seg(), and drop the unused uint64_t offloads field from struct zxdh_virtnet_rx along with the dead cache-line macros.
2. The flush relies on the device reporting per-descriptor id (desc[k].id == k) in used descriptors, matching what the enqueue paths set. The per-descriptor id field is the packed-ring spec convention; no reset of the id field is performed in the flush loop. 3. Also update the 26.07 release notes to document this series' user-visible changes (queue interrupt fix, single-segment Rx fast path, Rx/Tx packed-ring optimizations, and removed xstats counters). Signed-off-by: Junlong Wang <[email protected]> --- doc/guides/rel_notes/release_26_07.rst | 11 +++ drivers/net/zxdh/zxdh_ethdev.c | 4 +- drivers/net/zxdh/zxdh_rxtx.c | 128 +++++++++---------------- drivers/net/zxdh/zxdh_rxtx.h | 1 - 4 files changed, 60 insertions(+), 84 deletions(-) diff --git a/doc/guides/rel_notes/release_26_07.rst b/doc/guides/rel_notes/release_26_07.rst index 7227db439f..d22978bacd 100644 --- a/doc/guides/rel_notes/release_26_07.rst +++ b/doc/guides/rel_notes/release_26_07.rst @@ -122,6 +122,17 @@ New Features Added AGENTS.md file for AI review and supporting scripts to review patches and documentation. +* **Updated ZTE zxdh ethernet driver.** + + * Fixed an issue that prevented enabling queue interrupts. + * Added a fast single-segment Rx path (``zxdh_recv_single_pkts``) that + is selected when the MTU fits in a single buffer. + * Optimized the packed-ring Rx recv path. + * Optimized the packed-ring Tx xmit path with per-descriptor mbuf + free (``rte_pktmbuf_free_seg``) and prefetch hints. + * Removed unused xstats counters (``full``, ``norefill``, + ``multicast_packets``, ``broadcast_packets``) from both Rx and Tx + queues. Removed Items ------------- diff --git a/drivers/net/zxdh/zxdh_ethdev.c b/drivers/net/zxdh/zxdh_ethdev.c index aa931b346a..6d38ef46a3 100644 --- a/drivers/net/zxdh/zxdh_ethdev.c +++ b/drivers/net/zxdh/zxdh_ethdev.c @@ -489,7 +489,7 @@ zxdh_dev_free_mbufs(struct rte_eth_dev *dev) if (!vq) continue; while ((buf = zxdh_queue_detach_unused(vq)) != NULL) - rte_pktmbuf_free(buf); + rte_pktmbuf_free_seg(buf); PMD_DRV_LOG(DEBUG, "freeing %s[%d] used and unused buf", "rxq", i * 2); } @@ -498,7 +498,7 @@ zxdh_dev_free_mbufs(struct rte_eth_dev *dev) if (!vq) continue; while ((buf = zxdh_queue_detach_unused(vq)) != NULL) - rte_pktmbuf_free(buf); + rte_pktmbuf_free_seg(buf); PMD_DRV_LOG(DEBUG, "freeing %s[%d] used and unused buf", "txq", i * 2 + 1); } diff --git a/drivers/net/zxdh/zxdh_rxtx.c b/drivers/net/zxdh/zxdh_rxtx.c index 45fb2448a8..52d6d9726e 100644 --- a/drivers/net/zxdh/zxdh_rxtx.c +++ b/drivers/net/zxdh/zxdh_rxtx.c @@ -114,6 +114,14 @@ RTE_MBUF_F_TX_SEC_OFFLOAD | \ RTE_MBUF_F_TX_UDP_SEG) +#if RTE_CACHE_LINE_SIZE == 128 +#define NEXT_CACHELINE_OFF_16B 8 +#elif RTE_CACHE_LINE_SIZE == 64 +#define NEXT_CACHELINE_OFF_16B 4 +#else +#define NEXT_CACHELINE_OFF_16B (RTE_CACHE_LINE_SIZE / 16) +#endif + uint32_t zxdh_outer_l2_type[16] = { 0, RTE_PTYPE_L2_ETHER, @@ -201,43 +209,6 @@ uint32_t zxdh_inner_l4_type[16] = { 0, }; -static void -zxdh_xmit_cleanup_inorder_packed(struct zxdh_virtqueue *vq, int32_t num) -{ - uint16_t used_idx = 0; - uint16_t id = 0; - uint16_t curr_id = 0; - uint16_t free_cnt = 0; - uint16_t size = vq->vq_nentries; - struct zxdh_vring_packed_desc *desc = vq->vq_packed.ring.desc; - struct zxdh_vq_desc_extra *dxp = NULL; - - used_idx = vq->vq_used_cons_idx; - /* desc_is_used has a load-acquire or rte_io_rmb inside - * and wait for used desc in virtqueue. - */ - while (num > 0 && desc_is_used(&desc[used_idx], vq)) { - id = desc[used_idx].id; - do { - curr_id = used_idx; - dxp = &vq->vq_descx[used_idx]; - used_idx += dxp->ndescs; - free_cnt += dxp->ndescs; - num -= dxp->ndescs; - if (used_idx >= size) { - used_idx -= size; - vq->used_wrap_counter ^= 1; - } - if (dxp->cookie != NULL) { - rte_pktmbuf_free(dxp->cookie); - dxp->cookie = NULL; - } - } while (curr_id != id); - } - vq->vq_used_cons_idx = used_idx; - vq->vq_free_cnt += free_cnt; -} - static inline uint16_t zxdh_get_mtu(struct zxdh_virtqueue *vq) { @@ -334,7 +305,7 @@ zxdh_xmit_fill_net_hdr(struct zxdh_virtqueue *vq, struct rte_mbuf *cookie, } static inline void -zxdh_enqueue_xmit_packed_fast(struct zxdh_virtnet_tx *txvq, +zxdh_xmit_enqueue_push(struct zxdh_virtnet_tx *txvq, struct rte_mbuf *cookie) { struct zxdh_virtqueue *vq = txvq->vq; @@ -345,7 +316,6 @@ zxdh_enqueue_xmit_packed_fast(struct zxdh_virtnet_tx *txvq, uint8_t hdr_len = vq->hw->dl_net_hdr_len; struct zxdh_vring_packed_desc *dp = &vq->vq_packed.ring.desc[id]; - dxp->ndescs = 1; dxp->cookie = cookie; hdr = rte_pktmbuf_mtod_offset(cookie, struct zxdh_net_hdr_dl *, -hdr_len); zxdh_xmit_fill_net_hdr(vq, cookie, hdr); @@ -362,51 +332,56 @@ zxdh_enqueue_xmit_packed_fast(struct zxdh_virtnet_tx *txvq, } static inline void -zxdh_enqueue_xmit_packed(struct zxdh_virtnet_tx *txvq, +zxdh_xmit_enqueue_append(struct zxdh_virtnet_tx *txvq, struct rte_mbuf *cookie, uint16_t needed) { struct zxdh_tx_region *txr = txvq->zxdh_net_hdr_mz->addr; struct zxdh_virtqueue *vq = txvq->vq; - uint16_t id = vq->vq_avail_idx; - struct zxdh_vq_desc_extra *dxp = &vq->vq_descx[id]; uint16_t head_idx = vq->vq_avail_idx; uint16_t idx = head_idx; struct zxdh_vring_packed_desc *start_dp = vq->vq_packed.ring.desc; struct zxdh_vring_packed_desc *head_dp = &vq->vq_packed.ring.desc[idx]; struct zxdh_net_hdr_dl *hdr = NULL; - uint16_t head_flags = cookie->next ? ZXDH_VRING_DESC_F_NEXT : 0; + uint16_t id = vq->vq_avail_idx; + struct zxdh_vq_desc_extra *dxp = &vq->vq_descx[id]; uint8_t hdr_len = vq->hw->dl_net_hdr_len; + uint16_t head_flags = 0; - dxp->ndescs = needed; - dxp->cookie = cookie; - head_flags |= vq->cached_flags; + dxp->cookie = NULL; + /* + * Head descriptor has no mbuf cookie. Per-segment cookies are + * stored on the segment descs so zxdh_xmit_fast_flush() can free + * each via rte_pktmbuf_free_seg(). zxdh_queue_detach_unused() and + * zxdh_queue_rxvq_flush() both skip NULL cookies and are the only + * expected readers of head cookies. + */ + /* setup first tx ring slot to point to header stored in reserved region. */ start_dp[idx].addr = txvq->zxdh_net_hdr_mem + RTE_PTR_DIFF(&txr[idx].tx_hdr, txr); start_dp[idx].len = hdr_len; - head_flags |= ZXDH_VRING_DESC_F_NEXT; + start_dp[idx].id = idx; + head_flags |= vq->cached_flags | ZXDH_VRING_DESC_F_NEXT; hdr = (void *)&txr[idx].tx_hdr; - rte_prefetch1(hdr); + zxdh_xmit_fill_net_hdr(vq, cookie, hdr); + idx++; if (idx >= vq->vq_nentries) { idx -= vq->vq_nentries; vq->cached_flags ^= ZXDH_VRING_PACKED_DESC_F_AVAIL_USED; } - zxdh_xmit_fill_net_hdr(vq, cookie, hdr); - do { start_dp[idx].addr = rte_pktmbuf_iova(cookie); start_dp[idx].len = cookie->data_len; - start_dp[idx].id = id; - if (likely(idx != head_idx)) { - uint16_t flags = cookie->next ? ZXDH_VRING_DESC_F_NEXT : 0; + start_dp[idx].id = idx; - flags |= vq->cached_flags; - start_dp[idx].flags = flags; - } + vq->vq_descx[idx].cookie = cookie; + uint16_t flags = cookie->next ? ZXDH_VRING_DESC_F_NEXT : 0; + flags |= vq->cached_flags; + start_dp[idx].flags = flags; idx++; if (idx >= vq->vq_nentries) { @@ -456,7 +431,7 @@ zxdh_update_packet_stats(struct zxdh_virtnet_stats *stats, struct rte_mbuf *mbuf } static void -zxdh_xmit_flush(struct zxdh_virtqueue *vq) +zxdh_xmit_fast_flush(struct zxdh_virtqueue *vq) { uint16_t id = 0; uint16_t curr_id = 0; @@ -472,20 +447,22 @@ zxdh_xmit_flush(struct zxdh_virtqueue *vq) * for a used descriptor in the virtqueue. */ while (desc_is_used(&desc[used_idx], vq)) { + rte_prefetch0(&desc[used_idx + NEXT_CACHELINE_OFF_16B]); id = desc[used_idx].id; do { + desc[used_idx].id = used_idx; curr_id = used_idx; dxp = &vq->vq_descx[used_idx]; - used_idx += dxp->ndescs; - free_cnt += dxp->ndescs; - if (used_idx >= size) { - used_idx -= size; - vq->used_wrap_counter ^= 1; - } if (dxp->cookie != NULL) { - rte_pktmbuf_free(dxp->cookie); + rte_pktmbuf_free_seg(dxp->cookie); dxp->cookie = NULL; } + used_idx += 1; + free_cnt += 1; + if (unlikely(used_idx == size)) { + used_idx = 0; + vq->used_wrap_counter ^= 1; + } } while (curr_id != id); } vq->vq_used_cons_idx = used_idx; @@ -499,13 +476,12 @@ zxdh_xmit_pkts_packed(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkt struct zxdh_virtqueue *vq = txvq->vq; uint16_t nb_tx = 0; - zxdh_xmit_flush(vq); + zxdh_xmit_fast_flush(vq); for (nb_tx = 0; nb_tx < nb_pkts; nb_tx++) { struct rte_mbuf *txm = tx_pkts[nb_tx]; int32_t can_push = 0; int32_t slots = 0; - int32_t need = 0; rte_prefetch0(txm); /* optimize ring usage */ @@ -522,26 +498,16 @@ zxdh_xmit_pkts_packed(void *tx_queue, struct rte_mbuf **tx_pkts, uint16_t nb_pkt * default => number of segments + 1 **/ slots = txm->nb_segs + !can_push; - need = slots - vq->vq_free_cnt; /* Positive value indicates it need free vring descriptors */ - if (unlikely(need > 0)) { - zxdh_xmit_cleanup_inorder_packed(vq, need); - need = slots - vq->vq_free_cnt; - if (unlikely(need > 0)) { - PMD_TX_LOG(ERR, - " No enough %d free tx descriptors to transmit." - "freecnt %d", - need, - vq->vq_free_cnt); - break; - } - } + + if (unlikely(slots > vq->vq_free_cnt)) + break; /* Enqueue Packet buffers */ if (can_push) - zxdh_enqueue_xmit_packed_fast(txvq, txm); + zxdh_xmit_enqueue_push(txvq, txm); else - zxdh_enqueue_xmit_packed(txvq, txm, slots); + zxdh_xmit_enqueue_append(txvq, txm, slots); zxdh_update_packet_stats(&txvq->stats, txm); } txvq->stats.packets += nb_tx; diff --git a/drivers/net/zxdh/zxdh_rxtx.h b/drivers/net/zxdh/zxdh_rxtx.h index bf84c4796f..d119b62e16 100644 --- a/drivers/net/zxdh/zxdh_rxtx.h +++ b/drivers/net/zxdh/zxdh_rxtx.h @@ -48,7 +48,6 @@ struct __rte_cache_aligned zxdh_virtnet_rx { struct rte_mempool *mpool; /* mempool for mbuf allocation */ struct zxdh_virtnet_stats stats; const struct rte_memzone *mz; /* mem zone to populate RX ring. */ - uint64_t offloads; uint16_t queue_id; /* DPDK queue index. */ uint16_t port_id; /* Device port identifier. */ }; -- 2.27.0

