1. Rework the Tx completion flush to walk descriptors one at a time and
   free each completed segment with rte_pktmbuf_free_seg(), add prefetch
   hints on the next descriptor cache line, switch the ring-held mbufs
   in zxdh_dev_free_mbufs() to rte_pktmbuf_free_seg(), and drop the
   unused uint64_t offloads field from struct zxdh_virtnet_rx along
   with the dead cache-line macros.

2. The flush relies on the device reporting per-descriptor id
   (desc[k].id == k) in used descriptors, matching what the enqueue
   paths set. The per-descriptor id field is the packed-ring spec
   convention; no reset of the id field is performed in the flush loop.

3. Also update the 26.07 release notes to document this series'
   user-visible changes (queue interrupt fix, single-segment Rx fast
   path, Rx/Tx packed-ring optimizations, and removed xstats counters).

Signed-off-by: Junlong Wang <[email protected]>
---
 doc/guides/rel_notes/release_26_07.rst |  11 +++
 drivers/net/zxdh/zxdh_ethdev.c         |   4 +-
 drivers/net/zxdh/zxdh_rxtx.c           | 128 +++++++++----------------
 drivers/net/zxdh/zxdh_rxtx.h           |   1 -
 4 files changed, 60 insertions(+), 84 deletions(-)

diff --git a/doc/guides/rel_notes/release_26_07.rst 
b/doc/guides/rel_notes/release_26_07.rst
index 7227db439f..d22978bacd 100644
--- a/doc/guides/rel_notes/release_26_07.rst
+++ b/doc/guides/rel_notes/release_26_07.rst
@@ -122,6 +122,17 @@ New Features
   Added AGENTS.md file for AI review
   and supporting scripts to review patches and documentation.
 
+* **Updated ZTE zxdh ethernet driver.**
+
+  * Fixed an issue that prevented enabling queue interrupts.
+  * Added a fast single-segment Rx path (``zxdh_recv_single_pkts``) that
+    is selected when the MTU fits in a single buffer.
+  * Optimized the packed-ring Rx recv path.
+  * Optimized the packed-ring Tx xmit path with per-descriptor mbuf
+    free (``rte_pktmbuf_free_seg``) and prefetch hints.
+  * Removed unused xstats counters (``full``, ``norefill``,
+    ``multicast_packets``, ``broadcast_packets``) from both Rx and Tx
+    queues.
 
 Removed Items
 -------------
diff --git a/drivers/net/zxdh/zxdh_ethdev.c b/drivers/net/zxdh/zxdh_ethdev.c
index aa931b346a..6d38ef46a3 100644
--- a/drivers/net/zxdh/zxdh_ethdev.c
+++ b/drivers/net/zxdh/zxdh_ethdev.c
@@ -489,7 +489,7 @@ zxdh_dev_free_mbufs(struct rte_eth_dev *dev)
                if (!vq)
                        continue;
                while ((buf = zxdh_queue_detach_unused(vq)) != NULL)
-                       rte_pktmbuf_free(buf);
+                       rte_pktmbuf_free_seg(buf);
                PMD_DRV_LOG(DEBUG, "freeing %s[%d] used and unused buf",
                "rxq", i * 2);
        }
@@ -498,7 +498,7 @@ zxdh_dev_free_mbufs(struct rte_eth_dev *dev)
                if (!vq)
                        continue;
                while ((buf = zxdh_queue_detach_unused(vq)) != NULL)
-                       rte_pktmbuf_free(buf);
+                       rte_pktmbuf_free_seg(buf);
                PMD_DRV_LOG(DEBUG, "freeing %s[%d] used and unused buf",
                "txq", i * 2 + 1);
        }
diff --git a/drivers/net/zxdh/zxdh_rxtx.c b/drivers/net/zxdh/zxdh_rxtx.c
index 45fb2448a8..52d6d9726e 100644
--- a/drivers/net/zxdh/zxdh_rxtx.c
+++ b/drivers/net/zxdh/zxdh_rxtx.c
@@ -114,6 +114,14 @@
                RTE_MBUF_F_TX_SEC_OFFLOAD |     \
                RTE_MBUF_F_TX_UDP_SEG)
 
+#if RTE_CACHE_LINE_SIZE == 128
+#define NEXT_CACHELINE_OFF_16B   8
+#elif RTE_CACHE_LINE_SIZE == 64
+#define NEXT_CACHELINE_OFF_16B   4
+#else
+#define NEXT_CACHELINE_OFF_16B  (RTE_CACHE_LINE_SIZE / 16)
+#endif
+
 uint32_t zxdh_outer_l2_type[16] = {
        0,
        RTE_PTYPE_L2_ETHER,
@@ -201,43 +209,6 @@ uint32_t zxdh_inner_l4_type[16] = {
        0,
 };
 
-static void
-zxdh_xmit_cleanup_inorder_packed(struct zxdh_virtqueue *vq, int32_t num)
-{
-       uint16_t used_idx = 0;
-       uint16_t id       = 0;
-       uint16_t curr_id  = 0;
-       uint16_t free_cnt = 0;
-       uint16_t size     = vq->vq_nentries;
-       struct zxdh_vring_packed_desc *desc = vq->vq_packed.ring.desc;
-       struct zxdh_vq_desc_extra     *dxp  = NULL;
-
-       used_idx = vq->vq_used_cons_idx;
-       /* desc_is_used has a load-acquire or rte_io_rmb inside
-        * and wait for used desc in virtqueue.
-        */
-       while (num > 0 && desc_is_used(&desc[used_idx], vq)) {
-               id = desc[used_idx].id;
-               do {
-                       curr_id = used_idx;
-                       dxp = &vq->vq_descx[used_idx];
-                       used_idx += dxp->ndescs;
-                       free_cnt += dxp->ndescs;
-                       num -= dxp->ndescs;
-                       if (used_idx >= size) {
-                               used_idx -= size;
-                               vq->used_wrap_counter ^= 1;
-                       }
-                       if (dxp->cookie != NULL) {
-                               rte_pktmbuf_free(dxp->cookie);
-                               dxp->cookie = NULL;
-                       }
-               } while (curr_id != id);
-       }
-       vq->vq_used_cons_idx = used_idx;
-       vq->vq_free_cnt += free_cnt;
-}
-
 static inline uint16_t
 zxdh_get_mtu(struct zxdh_virtqueue *vq)
 {
@@ -334,7 +305,7 @@ zxdh_xmit_fill_net_hdr(struct zxdh_virtqueue *vq, struct 
rte_mbuf *cookie,
 }
 
 static inline void
-zxdh_enqueue_xmit_packed_fast(struct zxdh_virtnet_tx *txvq,
+zxdh_xmit_enqueue_push(struct zxdh_virtnet_tx *txvq,
                                                struct rte_mbuf *cookie)
 {
        struct zxdh_virtqueue *vq = txvq->vq;
@@ -345,7 +316,6 @@ zxdh_enqueue_xmit_packed_fast(struct zxdh_virtnet_tx *txvq,
        uint8_t hdr_len = vq->hw->dl_net_hdr_len;
        struct zxdh_vring_packed_desc *dp = &vq->vq_packed.ring.desc[id];
 
-       dxp->ndescs = 1;
        dxp->cookie = cookie;
        hdr = rte_pktmbuf_mtod_offset(cookie, struct zxdh_net_hdr_dl *, 
-hdr_len);
        zxdh_xmit_fill_net_hdr(vq, cookie, hdr);
@@ -362,51 +332,56 @@ zxdh_enqueue_xmit_packed_fast(struct zxdh_virtnet_tx 
*txvq,
 }
 
 static inline void
-zxdh_enqueue_xmit_packed(struct zxdh_virtnet_tx *txvq,
+zxdh_xmit_enqueue_append(struct zxdh_virtnet_tx *txvq,
                                                struct rte_mbuf *cookie,
                                                uint16_t needed)
 {
        struct zxdh_tx_region *txr = txvq->zxdh_net_hdr_mz->addr;
        struct zxdh_virtqueue *vq = txvq->vq;
-       uint16_t id = vq->vq_avail_idx;
-       struct zxdh_vq_desc_extra *dxp = &vq->vq_descx[id];
        uint16_t head_idx = vq->vq_avail_idx;
        uint16_t idx = head_idx;
        struct zxdh_vring_packed_desc *start_dp = vq->vq_packed.ring.desc;
        struct zxdh_vring_packed_desc *head_dp = &vq->vq_packed.ring.desc[idx];
        struct zxdh_net_hdr_dl *hdr = NULL;
 
-       uint16_t head_flags = cookie->next ? ZXDH_VRING_DESC_F_NEXT : 0;
+       uint16_t id = vq->vq_avail_idx;
+       struct zxdh_vq_desc_extra *dxp = &vq->vq_descx[id];
        uint8_t hdr_len = vq->hw->dl_net_hdr_len;
+       uint16_t head_flags = 0;
 
-       dxp->ndescs = needed;
-       dxp->cookie = cookie;
-       head_flags |= vq->cached_flags;
+       dxp->cookie = NULL;
+       /*
+        * Head descriptor has no mbuf cookie. Per-segment cookies are
+        * stored on the segment descs so zxdh_xmit_fast_flush() can free
+        * each via rte_pktmbuf_free_seg(). zxdh_queue_detach_unused() and
+        * zxdh_queue_rxvq_flush() both skip NULL cookies and are the only
+        * expected readers of head cookies.
+        */
 
+       /* setup first tx ring slot to point to header stored in reserved 
region. */
        start_dp[idx].addr = txvq->zxdh_net_hdr_mem + 
RTE_PTR_DIFF(&txr[idx].tx_hdr, txr);
        start_dp[idx].len  = hdr_len;
-       head_flags |= ZXDH_VRING_DESC_F_NEXT;
+       start_dp[idx].id = idx;
+       head_flags |= vq->cached_flags | ZXDH_VRING_DESC_F_NEXT;
        hdr = (void *)&txr[idx].tx_hdr;
 
-       rte_prefetch1(hdr);
+       zxdh_xmit_fill_net_hdr(vq, cookie, hdr);
+
        idx++;
        if (idx >= vq->vq_nentries) {
                idx -= vq->vq_nentries;
                vq->cached_flags ^= ZXDH_VRING_PACKED_DESC_F_AVAIL_USED;
        }
 
-       zxdh_xmit_fill_net_hdr(vq, cookie, hdr);
-
        do {
                start_dp[idx].addr = rte_pktmbuf_iova(cookie);
                start_dp[idx].len  = cookie->data_len;
-               start_dp[idx].id = id;
-               if (likely(idx != head_idx)) {
-                       uint16_t flags = cookie->next ? ZXDH_VRING_DESC_F_NEXT 
: 0;
+               start_dp[idx].id = idx;
 
-                       flags |= vq->cached_flags;
-                       start_dp[idx].flags = flags;
-               }
+               vq->vq_descx[idx].cookie = cookie;
+               uint16_t flags = cookie->next ? ZXDH_VRING_DESC_F_NEXT : 0;
+               flags |= vq->cached_flags;
+               start_dp[idx].flags = flags;
 
                idx++;
                if (idx >= vq->vq_nentries) {
@@ -456,7 +431,7 @@ zxdh_update_packet_stats(struct zxdh_virtnet_stats *stats, 
struct rte_mbuf *mbuf
 }
 
 static void
-zxdh_xmit_flush(struct zxdh_virtqueue *vq)
+zxdh_xmit_fast_flush(struct zxdh_virtqueue *vq)
 {
        uint16_t id       = 0;
        uint16_t curr_id  = 0;
@@ -472,20 +447,22 @@ zxdh_xmit_flush(struct zxdh_virtqueue *vq)
         * for a used descriptor in the virtqueue.
         */
        while (desc_is_used(&desc[used_idx], vq)) {
+               rte_prefetch0(&desc[used_idx + NEXT_CACHELINE_OFF_16B]);
                id = desc[used_idx].id;
                do {
+                       desc[used_idx].id = used_idx;
                        curr_id = used_idx;
                        dxp = &vq->vq_descx[used_idx];
-                       used_idx += dxp->ndescs;
-                       free_cnt += dxp->ndescs;
-                       if (used_idx >= size) {
-                               used_idx -= size;
-                               vq->used_wrap_counter ^= 1;
-                       }
                        if (dxp->cookie != NULL) {
-                               rte_pktmbuf_free(dxp->cookie);
+                               rte_pktmbuf_free_seg(dxp->cookie);
                                dxp->cookie = NULL;
                        }
+                       used_idx += 1;
+                       free_cnt += 1;
+                       if (unlikely(used_idx == size)) {
+                               used_idx = 0;
+                               vq->used_wrap_counter ^= 1;
+                       }
                } while (curr_id != id);
        }
        vq->vq_used_cons_idx = used_idx;
@@ -499,13 +476,12 @@ zxdh_xmit_pkts_packed(void *tx_queue, struct rte_mbuf 
**tx_pkts, uint16_t nb_pkt
        struct zxdh_virtqueue  *vq   = txvq->vq;
        uint16_t nb_tx = 0;
 
-       zxdh_xmit_flush(vq);
+       zxdh_xmit_fast_flush(vq);
 
        for (nb_tx = 0; nb_tx < nb_pkts; nb_tx++) {
                struct rte_mbuf *txm = tx_pkts[nb_tx];
                int32_t can_push     = 0;
                int32_t slots        = 0;
-               int32_t need         = 0;
 
                rte_prefetch0(txm);
                /* optimize ring usage */
@@ -522,26 +498,16 @@ zxdh_xmit_pkts_packed(void *tx_queue, struct rte_mbuf 
**tx_pkts, uint16_t nb_pkt
                 * default    => number of segments + 1
                 **/
                slots = txm->nb_segs + !can_push;
-               need = slots - vq->vq_free_cnt;
                /* Positive value indicates it need free vring descriptors */
-               if (unlikely(need > 0)) {
-                       zxdh_xmit_cleanup_inorder_packed(vq, need);
-                       need = slots - vq->vq_free_cnt;
-                       if (unlikely(need > 0)) {
-                               PMD_TX_LOG(ERR,
-                                               " No enough %d free tx 
descriptors to transmit."
-                                               "freecnt %d",
-                                               need,
-                                               vq->vq_free_cnt);
-                               break;
-                       }
-               }
+
+               if (unlikely(slots >  vq->vq_free_cnt))
+                       break;
 
                /* Enqueue Packet buffers */
                if (can_push)
-                       zxdh_enqueue_xmit_packed_fast(txvq, txm);
+                       zxdh_xmit_enqueue_push(txvq, txm);
                else
-                       zxdh_enqueue_xmit_packed(txvq, txm, slots);
+                       zxdh_xmit_enqueue_append(txvq, txm, slots);
                zxdh_update_packet_stats(&txvq->stats, txm);
        }
        txvq->stats.packets += nb_tx;
diff --git a/drivers/net/zxdh/zxdh_rxtx.h b/drivers/net/zxdh/zxdh_rxtx.h
index bf84c4796f..d119b62e16 100644
--- a/drivers/net/zxdh/zxdh_rxtx.h
+++ b/drivers/net/zxdh/zxdh_rxtx.h
@@ -48,7 +48,6 @@ struct __rte_cache_aligned zxdh_virtnet_rx {
        struct rte_mempool       *mpool;            /* mempool for mbuf 
allocation */
        struct zxdh_virtnet_stats      stats;
        const struct rte_memzone *mz;               /* mem zone to populate RX 
ring. */
-       uint64_t offloads;
        uint16_t                  queue_id;         /* DPDK queue index. */
        uint16_t                  port_id;          /* Device port identifier. 
*/
 };
-- 
2.27.0

Reply via email to