On a link-down or impending PF reset, in-flight Tx descriptors
were left pending when the queues were torn down, which could
trigger Malicious Driver Detection (MDD) events and
leak descriptors.

Added iavf_dev_tx_drain() to let already-posted Tx bursts
complete and flush the rings within a bounded budget,
and call it on link-down and reset-impending events
before teardown, preventing MDD events and descriptor leaks.
The drain selects the cleanup routine that matches the
active Tx path: the scalar path uses ci_tx_xmit_cleanup(),
while the vector and CTX paths use ci_tx_free_bufs_vec().
This matters because the scalar and vector paths track
their software rings differently
(ci_tx_entry vs ci_tx_entry_vec) and using the scalar
routine on a vector queue would walk the wrong ring
and free the wrong mbufs.

Signed-off-by: Anurag Mandal <[email protected]>
---
 drivers/net/intel/iavf/iavf_rxtx.c  | 104 ++++++++++++++++++++++++++++
 drivers/net/intel/iavf/iavf_rxtx.h  |   6 ++
 drivers/net/intel/iavf/iavf_vchnl.c |   7 ++
 3 files changed, 117 insertions(+)

diff --git a/drivers/net/intel/iavf/iavf_rxtx.c 
b/drivers/net/intel/iavf/iavf_rxtx.c
index 4f2ffe6188..931bb8420d 100644
--- a/drivers/net/intel/iavf/iavf_rxtx.c
+++ b/drivers/net/intel/iavf/iavf_rxtx.c
@@ -32,6 +32,7 @@
 
 #include "iavf.h"
 #include "iavf_rxtx.h"
+#include "iavf_rxtx_vec_common.h"
 #include "iavf_ipsec_crypto.h"
 #include "rte_pmd_iavf.h"
 
@@ -4025,6 +4026,109 @@ iavf_tx_done_cleanup_full(struct ci_tx_queue *txq,
        return (int)pkt_cnt;
 }
 
+/*
+ * Reclaim completed Tx descriptors for a single queue using the cleanup
+ * routine that matches the active Tx path.
+ * The scalar and vector paths track their software rings differently
+ * (ci_tx_entry vs ci_tx_entry_vec) and keep separate completion
+ * bookkeeping, so using the scalar routine on a vector queue
+ * (or vice versa) would free the wrong mbufs.
+ * Returns true if any descriptors were reclaimed.
+ */
+static bool
+iavf_tx_drain_cleanup(struct ci_tx_queue *txq,
+                     enum iavf_tx_func_type tx_func_type)
+{
+       switch (tx_func_type) {
+       case IAVF_TX_AVX2_CTX:
+       case IAVF_TX_AVX2_CTX_OFFLOAD:
+       case IAVF_TX_AVX512_CTX:
+       case IAVF_TX_AVX512_CTX_OFFLOAD:
+               return ci_tx_free_bufs_vec(txq, iavf_tx_desc_done, true) != 0;
+       case IAVF_TX_NEON:
+       case IAVF_TX_AVX2:
+       case IAVF_TX_AVX2_OFFLOAD:
+       case IAVF_TX_AVX512:
+       case IAVF_TX_AVX512_OFFLOAD:
+               return ci_tx_free_bufs_vec(txq, iavf_tx_desc_done, false) != 0;
+       case IAVF_TX_DEFAULT:
+       default:
+               return ci_tx_xmit_cleanup(txq) == 0;
+       }
+}
+
+/*
+ * iavf_dev_tx_drain - drain in-flight Tx descriptors after a link-down or
+ * impending PF reset event.
+ */
+void
+iavf_dev_tx_drain(struct rte_eth_dev *dev)
+{
+       struct iavf_adapter *adapter =
+               IAVF_DEV_PRIVATE_TO_ADAPTER(dev->data->dev_private);
+       enum iavf_tx_func_type tx_func_type = adapter->tx_func_type;
+       struct ci_tx_queue *txq;
+       uint64_t hz, deadline;
+       int idle_iters = 0;
+       uint16_t qid;
+
+       /*
+        * Allow any Tx burst already in flight on a data-plane lcore to
+        * write its remaining descriptors and notify. After
+        * this window, the no_poll gate set by the caller is observed at
+        * the next burst-entry and no new descriptors will be posted.
+        */
+       rte_delay_us_block(IAVF_TX_DRAIN_SETTLE_US);
+
+       hz = rte_get_timer_hz();
+       deadline = rte_get_timer_cycles() +
+               (hz * IAVF_TX_DRAIN_TIMEOUT_US) / 1000000ULL;
+
+       while (rte_get_timer_cycles() < deadline) {
+               bool any_pending = false;
+               bool any_progress = false;
+
+               for (qid = 0; qid < dev->data->nb_tx_queues; qid++) {
+                       txq = dev->data->tx_queues[qid];
+                       if (txq == NULL ||
+                           dev->data->tx_queue_state[qid] !=
+                               RTE_ETH_QUEUE_STATE_STARTED)
+                               continue;
+
+                       /*
+                        * nb_tx_free == nb_tx_desc - 1 means the ring is
+                        * empty (one descriptor is always reserved).
+                        */
+                       if (txq->nb_tx_free >= txq->nb_tx_desc - 1)
+                               continue;
+
+                       any_pending = true;
+                       if (iavf_tx_drain_cleanup(txq, tx_func_type))
+                               any_progress = true;
+               }
+
+               if (!any_pending)
+                       return;
+
+               if (any_progress) {
+                       idle_iters = 0;
+               } else if (++idle_iters >= IAVF_TX_DRAIN_IDLE_MAX) {
+                       /*
+                        * HW has not advanced the RS-bit write-back for
+                        * several polling intervals; either the queue is
+                        * quiescent except for the sub-rs_thresh tail
+                        * (which we cannot observe here) or HW is no
+                        * longer fetching. Further polling is unlikely to
+                        * help, and the PF teardown path has its own
+                        * grace period for the remainder.
+                        */
+                       break;
+               }
+
+               rte_delay_us_block(IAVF_TX_DRAIN_POLL_US);
+       }
+}
+
 int
 iavf_dev_tx_done_cleanup(void *txq, uint32_t free_cnt)
 {
diff --git a/drivers/net/intel/iavf/iavf_rxtx.h 
b/drivers/net/intel/iavf/iavf_rxtx.h
index 22ea415f44..4088bc421c 100644
--- a/drivers/net/intel/iavf/iavf_rxtx.h
+++ b/drivers/net/intel/iavf/iavf_rxtx.h
@@ -506,6 +506,11 @@ enum iavf_tx_ctx_desc_tunnel_l4_tunnel_type {
 /* Valid indicator bit for the time_stamp_low field */
 #define IAVF_RX_FLX_DESC_TS_VALID      (0x1UL)
 
+#define IAVF_TX_DRAIN_TIMEOUT_US       10000   /* total drain budget: 10 ms */
+#define IAVF_TX_DRAIN_SETTLE_US                100     /* let in-flight burst 
land  */
+#define IAVF_TX_DRAIN_POLL_US          50      /* poll interval             */
+#define IAVF_TX_DRAIN_IDLE_MAX         20      /* ~1 ms of no RS write-back */
+
 int iavf_dev_rx_queue_setup(struct rte_eth_dev *dev,
                           uint16_t queue_idx,
                           uint16_t nb_desc,
@@ -641,6 +646,7 @@ void iavf_set_default_ptype_table(struct rte_eth_dev *dev);
 void iavf_rx_queue_release_mbufs_vec(struct ci_rx_queue *rxq);
 void iavf_rx_queue_release_mbufs_neon(struct ci_rx_queue *rxq);
 enum rte_vect_max_simd iavf_get_max_simd_bitwidth(void);
+void iavf_dev_tx_drain(struct rte_eth_dev *dev);
 
 static inline
 void iavf_dump_rx_descriptor(struct ci_rx_queue *rxq,
diff --git a/drivers/net/intel/iavf/iavf_vchnl.c 
b/drivers/net/intel/iavf/iavf_vchnl.c
index 8e102b02aa..d22990a524 100644
--- a/drivers/net/intel/iavf/iavf_vchnl.c
+++ b/drivers/net/intel/iavf/iavf_vchnl.c
@@ -269,6 +269,8 @@ iavf_handle_link_change_event(struct rte_eth_dev *dev,
                iavf_set_no_poll(adapter, true);
                PMD_DRV_LOG(DEBUG, "VF no poll turned %s",
                            adapter->no_poll ? "on" : "off");
+               if (!vf->link_up)
+                       iavf_dev_tx_drain(dev);
        }
 
        /*
@@ -341,6 +343,8 @@ iavf_read_msg_from_pf(struct iavf_adapter *adapter, 
uint16_t buf_len,
                        if (!vf->vf_reset) {
                                vf->vf_reset = true;
                                iavf_set_no_poll(adapter, false);
+                               if (adapter->devargs.no_poll_on_link_down)
+                                       iavf_dev_tx_drain(vf->eth_dev);
                                iavf_dev_event_post(vf->eth_dev,
                                        RTE_ETH_EVENT_INTR_RESET,
                                        NULL, 0);
@@ -579,6 +583,9 @@ iavf_handle_pf_event_msg(struct rte_eth_dev *dev, uint8_t 
*msg,
                if (!vf->vf_reset) {
                        vf->vf_reset = true;
                        iavf_set_no_poll(adapter, false);
+                       iavf_dev_watchdog_enable(adapter);
+                       if (adapter->devargs.no_poll_on_link_down)
+                               iavf_dev_tx_drain(dev);
                        iavf_dev_event_post(dev, RTE_ETH_EVENT_INTR_RESET,
                                NULL, 0);
                }
-- 
2.34.1

Reply via email to