The ethtool reconfiguration paths (channel count, ring size, private
flags), mana_change_mtu() and mana_xdp_set() rebuild the queues with
mana_detach() then mana_attach(). That tears the vport down, so RDMA can
claim it while released, and a failed mana_attach() leaves the port down
with no way back but manual intervention.

Add the data model and helpers for pre-allocate and swap: a queue set
built, published and torn down independently of the vport, against a
scratch port context so the live one never points at queues still being
built or freed. Building in place is not an option: mana_start_xmit()
dereferences apc->tx_qp[] guarded only by apc->port_is_up.

The TX drain moves out of mana_dealloc_queues() so the new teardown path
gets it too, and its fallback reset becomes pci_try_reset_function()
rather than an open-coded pcie_flr(), which does not save and restore
config space. Trylock because this runs under RTNL while removal takes
the device lock first.

No functional change otherwise: nothing calls the new helpers yet.

Signed-off-by: Long Li <[email protected]>
---
 .../net/ethernet/microsoft/mana/mana_bpf.c    |  31 ++
 drivers/net/ethernet/microsoft/mana/mana_en.c | 477 ++++++++++++++++--
 .../ethernet/microsoft/mana/mana_ethtool.c    |   9 +-
 include/net/mana/mana.h                       |  63 +++
 4 files changed, 538 insertions(+), 42 deletions(-)

diff --git a/drivers/net/ethernet/microsoft/mana/mana_bpf.c 
b/drivers/net/ethernet/microsoft/mana/mana_bpf.c
index 
53308e139cbe917b074dd381c83546fc74d7b79f..ca602e27044f92b87295cbc2de924adc71efa780
 100644
--- a/drivers/net/ethernet/microsoft/mana/mana_bpf.c
+++ b/drivers/net/ethernet/microsoft/mana/mana_bpf.c
@@ -265,3 +265,34 @@ int mana_bpf(struct net_device *ndev, struct netdev_bpf 
*bpf)
 
        return ret;
 }
+
+/* Read the XDP program a queue set is running, without changing anything. */
+struct bpf_prog *mana_chn_xdp_peek(struct mana_port_context *apc)
+{
+       ASSERT_RTNL();
+
+       if (!apc->rxqs || !apc->rxqs[0])
+               return NULL;
+
+       return rtnl_dereference(apc->rxqs[0]->bpf_prog);
+}
+
+/* Drop the per-queue references a retiring set holds on @prog.
+ *
+ * Kept separate from mana_chn_setxdp() so the pointers can stay in place
+ * until the queues stop polling: clearing them up front would let packets
+ * already sitting in a retiring RQ take the pass path and reach the stack
+ * without the program ever seeing them.
+ */
+void mana_chn_xdp_release(struct bpf_prog *prog, unsigned int num_queues)
+{
+       unsigned int i;
+
+       ASSERT_RTNL();
+
+       if (!prog)
+               return;
+
+       for (i = 0; i < num_queues; i++)
+               bpf_prog_put(prog);
+}
diff --git a/drivers/net/ethernet/microsoft/mana/mana_en.c 
b/drivers/net/ethernet/microsoft/mana/mana_en.c
index 
3c96e6fc3d81dc16853cc458ef620b5150aa8988..60b1fc93d453b16bec37924e32ea54c64f179f1c
 100644
--- a/drivers/net/ethernet/microsoft/mana/mana_en.c
+++ b/drivers/net/ethernet/microsoft/mana/mana_en.c
@@ -2015,7 +2015,8 @@ static void mana_poll_tx_cq(struct mana_cq *cq)
        /* Ensure checking txq_stopped before apc->port_is_up. */
        smp_rmb();
 
-       if (txq_stopped && apc->port_is_up && avail_space >= MAX_TX_WQE_SIZE) {
+       if (txq_stopped && !READ_ONCE(txq->retiring) && apc->port_is_up &&
+           avail_space >= MAX_TX_WQE_SIZE) {
                netif_tx_wake_queue(net_txq);
                apc->eth_stats.wake_queue++;
        }
@@ -2751,6 +2752,7 @@ static int mana_create_txq(struct mana_port_context *apc,
                u64_stats_init(&txq->stats.syncp);
                txq->ndev = net;
                txq->net_txq = netdev_get_tx_queue(net, i);
+               txq->reset_gen = READ_ONCE(apc->ac->reset_gen);
                txq->vp_offset = apc->tx_vp_offset;
                txq->napi_initialized = false;
                skb_queue_head_init(&txq->pending_skbs);
@@ -3006,11 +3008,14 @@ static int mana_push_wqe(struct mana_rxq *rxq)
 
 static int mana_create_page_pool(struct mana_rxq *rxq, struct gdma_context *gc)
 {
-       struct mana_port_context *mpc = netdev_priv(rxq->ndev);
        struct page_pool_params pprm = {};
        int ret;
 
-       pprm.pool_size = mpc->rx_queue_size / rxq->frag_count + 1;
+       /* Size the recycle ring from the queue being built, not from the live
+        * port context: during a swap the queue may be sized for a ring the
+        * running configuration does not use yet.
+        */
+       pprm.pool_size = rxq->num_rx_buf / rxq->frag_count + 1;
        pprm.nid = gc->numa_node;
        pprm.napi = &rxq->rx_cq.napi;
        pprm.netdev = rxq->ndev;
@@ -3676,15 +3681,114 @@ int mana_attach(struct net_device *ndev)
        return 0;
 }
 
-static int mana_dealloc_queues(struct net_device *ndev)
+/* Drain a set about to be destroyed: nothing new can reach it, so wait for the
+ * hardware to finish what it owns, then release every mapped SKB.
+ *
+ * The 120s budget is shared across all queues. On timeout the device is reset,
+ * since its buffers are about to be freed while it may still DMA into them; if
+ * that fails too they are leaked.
+ *
+ * Returns true only if a reset happened, taking every queue on the function
+ * down with it.
+ */
+static bool mana_drain_txqs(struct mana_port_context *apc)
 {
-       struct mana_port_context *apc = netdev_priv(ndev);
        unsigned long timeout = jiffies + 120 * HZ;
        struct gdma_dev *gd = apc->ac->gdma_dev;
+       bool quiesced = true;
+       bool reset = false;
        struct mana_txq *txq;
        struct sk_buff *skb;
-       int i, err;
        u32 tsleep;
+       int i, err;
+
+       if (!apc->tx_qp)
+               return false;
+
+       for (i = 0; i < apc->num_queues; i++) {
+               if (!apc->tx_qp[i])
+                       continue;
+
+               txq = &apc->tx_qp[i]->txq;
+
+               /* The function was reset after this queue was created, so the
+                * device has stopped touching its buffers and the completions
+                * waited for below can never arrive. Without this the port
+                * would burn the full timeout under RTNL, then reset the
+                * function again on the way out.
+                */
+               if (READ_ONCE(apc->ac->reset_gen) != txq->reset_gen)
+                       continue;
+
+               tsleep = 1000;
+               while (atomic_read(&txq->pending_sends) > 0 &&
+                      time_before(jiffies, timeout)) {
+                       usleep_range(tsleep, tsleep + 1000);
+                       tsleep <<= 1;
+               }
+               if (atomic_read(&txq->pending_sends)) {
+                       /* The device still owns these buffers, so reset it
+                        * before they are freed. pci_try_reset_function()
+                        * rather than pcie_flr(): it saves and restores config
+                        * space, which a bare FLR wipes behind the PCI core's
+                        * back. Trylock because RTNL is held here while the
+                        * remove path takes the device lock first.
+                        */
+                       err = 
pci_try_reset_function(to_pci_dev(gd->gdma_context->dev));
+                       if (err) {
+                               netdev_err(apc->ndev,
+                                          "function reset failed: %d, %d pkts 
pending in txq %u\n",
+                                          err, 
atomic_read(&txq->pending_sends),
+                                          txq->gdma_txq_id);
+                               quiesced = false;
+                       } else {
+                               /* Every queue on the function is dead now,
+                                * including the ones this loop has not reached
+                                * and those of the other ports.
+                                */
+                               WRITE_ONCE(apc->ac->reset_gen,
+                                          apc->ac->reset_gen + 1);
+
+                               /* Only a reset that actually happened takes the
+                                * other ports down with it; reporting a failed
+                                * one would rebuild them for nothing.
+                                */
+                               reset = true;
+                       }
+                       break;
+               }
+       }
+
+       /* Only a reset that actually happened makes freeing these safe; without
+        * one the device still owns them. Leak instead, bounded at one SQ ring
+        * of skbs per queue.
+        */
+       if (!quiesced) {
+               netdev_err(apc->ndev,
+                          "device not quiesced, leaking pending TX buffers 
instead of unmapping memory it can still DMA from\n");
+               return reset;
+       }
+
+       for (i = 0; i < apc->num_queues; i++) {
+               if (!apc->tx_qp[i])
+                       continue;
+
+               txq = &apc->tx_qp[i]->txq;
+               while ((skb = skb_dequeue(&txq->pending_skbs))) {
+                       mana_unmap_skb(skb, apc);
+                       dev_kfree_skb_any(skb);
+               }
+               atomic_set(&txq->pending_sends, 0);
+       }
+
+       return reset;
+}
+
+static int mana_dealloc_queues(struct net_device *ndev)
+{
+       struct mana_port_context *apc = netdev_priv(ndev);
+       struct gdma_dev *gd = apc->ac->gdma_dev;
+       int err;
 
        if (apc->port_is_up)
                return -EINVAL;
@@ -3702,41 +3806,27 @@ static int mana_dealloc_queues(struct net_device *ndev)
         * new packets due to apc->port_is_up being false.
         *
         * Drain all the in-flight TX packets.
-        * A timeout of 120 seconds for all the queues is used.
-        * This will break the while loop when h/w is not responding.
-        * This value of 120 has been decided here considering max
-        * number of queues.
+        *
+        * If the drain had to reset the function to get there, every other
+        * port on the adapter lost its queues too, so schedule them for a
+        * rebuild. This port is being torn down here and needs no such
+        * treatment, and a down port stays down: with port_st_save false,
+        * detach and attach both skip the queue work.
         */
+       if (mana_drain_txqs(apc)) {
+               struct mana_context *ac = apc->ac;
+               unsigned int i;
 
-       if (apc->tx_qp) {
-               for (i = 0; i < apc->num_queues; i++) {
-                       txq = &apc->tx_qp[i]->txq;
-                       tsleep = 1000;
-                       while (atomic_read(&txq->pending_sends) > 0 &&
-                              time_before(jiffies, timeout)) {
-                               usleep_range(tsleep, tsleep + 1000);
-                               tsleep <<= 1;
-                       }
-                       if (atomic_read(&txq->pending_sends)) {
-                               err =
-                                   pcie_flr(to_pci_dev(gd->gdma_context->dev));
-                               if (err) {
-                                       netdev_err(ndev, "flr failed %d with %d 
pkts pending in txq %u\n",
-                                                  err,
-                                           atomic_read(&txq->pending_sends),
-                                           txq->gdma_txq_id);
-                               }
-                               break;
-                       }
-               }
+               for (i = 0; i < ac->num_ports; i++) {
+                       struct mana_port_context *sib;
 
-               for (i = 0; i < apc->num_queues; i++) {
-                       txq = &apc->tx_qp[i]->txq;
-                       while ((skb = skb_dequeue(&txq->pending_skbs))) {
-                               mana_unmap_skb(skb, apc);
-                               dev_kfree_skb_any(skb);
-                       }
-                       atomic_set(&txq->pending_sends, 0);
+                       if (!ac->ports[i] || ac->ports[i] == ndev)
+                               continue;
+                       sib = netdev_priv(ac->ports[i]);
+                       netdev_err(ac->ports[i],
+                                  "queues reset by a sibling port, scheduling 
rebuild\n");
+                       queue_work(ac->per_port_queue_reset_wq,
+                                  &sib->queue_reset_work);
                }
        }
 
@@ -3760,6 +3850,310 @@ static int mana_dealloc_queues(struct net_device *ndev)
        return 0;
 }
 
+/*
+ * ---------------------------------------------------------------------------
+ * Pre-allocate + swap reconfiguration path.
+ *
+ * The detach/attach reconfigure path tears the vport down and rebuilds it,
+ * which lets RDMA grab the vport mid-flight and, if attach fails, leaves the
+ * port permanently broken.
+ *
+ * The swap path builds a *new* set of EQs/TXQs/RXQs while the current set
+ * keeps serving traffic. If allocation fails the current qset is untouched
+ * and we return the error; the user's requested value is never silently
+ * replaced by a fallback. Publishing a new set onto the live port
+ * context is added separately. The vport is never torn down: vport_use_count
+ * stays at 1 throughout, so RDMA cannot hijack it.
+ *
+ * Allocation and teardown run against a *scratch* mana_port_context rather
+ * than the live one. This is essential, not cosmetic: an earlier revision
+ * temporarily NULLed apc->tx_qp so the allocators could
+ * build into the live context, which reliably panicked in mana_start_xmit()
+ * under traffic (it dereferences apc->tx_qp[] guarded only by port_is_up).
+ * The live apc is now mutated only inside mana_publish_qset(), with TX
+ * disabled.
+ *
+ * Note that both sets are live between publish and free, so this peaks at
+ * old+new queues, and therefore at old+new MSI-X vectors. A later patch
+ * gives the port a shared EQ pool so only the queues, not the interrupts,
+ * are doubled up.
+ *
+ * Per-queue debugfs is suppressed for a set while it is being built or torn
+ * down (see mana_qset_scratch_alloc()): the directory names are derived from
+ * the queue index, so the incoming set would collide with the outgoing one
+ * under vport%d. Restoring it needs per-set subdirectories or a
+ * debugfs_rename() once the swap has completed.
+ * ---------------------------------------------------------------------------
+ */
+
+/* Snapshot the queue-set fields of @ctx into @out. */
+static void mana_qset_snapshot(const struct mana_port_context *ctx,
+                              struct mana_qset *out)
+{
+       out->eqs                = ctx->eqs;
+       out->tx_qp              = ctx->tx_qp;
+       out->rxqs               = ctx->rxqs;
+       out->indir_table        = ctx->indir_table;
+       out->indir_table_sz     = ctx->indir_table_sz;
+       out->rxobj_table        = ctx->rxobj_table;
+       out->default_rxobj      = ctx->default_rxobj;
+       out->num_queues         = ctx->num_queues;
+       out->rx_queue_size      = ctx->rx_queue_size;
+       out->tx_queue_size      = ctx->tx_queue_size;
+       out->priv_flags         = ctx->priv_flags;
+       out->mana_eqs_debugfs   = ctx->mana_eqs_debugfs;
+}
+
+/* Install @qset's fields onto @ctx. The vport (port_handle,
+ * vport_use_count) and the port-level debugfs dir are deliberately not
+ * touched: they outlive any individual queue set.
+ */
+static void mana_qset_install(struct mana_port_context *ctx,
+                             const struct mana_qset *qset)
+{
+       ctx->eqs                = qset->eqs;
+       ctx->tx_qp              = qset->tx_qp;
+       ctx->rxqs               = qset->rxqs;
+       ctx->indir_table        = qset->indir_table;
+       ctx->indir_table_sz     = qset->indir_table_sz;
+       ctx->rxobj_table        = qset->rxobj_table;
+       ctx->default_rxobj      = qset->default_rxobj;
+       ctx->num_queues         = qset->num_queues;
+       ctx->rx_queue_size      = qset->rx_queue_size;
+       ctx->tx_queue_size      = qset->tx_queue_size;
+       ctx->priv_flags         = qset->priv_flags;
+       ctx->mana_eqs_debugfs   = qset->mana_eqs_debugfs;
+}
+
+/**
+ * mana_qset_scratch_alloc - build a scratch port context for queue work
+ * @apc: the live port context to shadow
+ *
+ * Returns a heap copy of @apc that shares its vport identity but owns no
+ * queues, so the existing allocators and destroyers can run against it
+ * without touching the live context.
+ */
+struct mana_port_context *mana_qset_scratch_alloc(struct mana_port_context 
*apc)
+{
+       struct mana_port_context *scratch;
+
+       scratch = kvzalloc(sizeof(*scratch), GFP_KERNEL);
+       if (!scratch)
+               return NULL;
+
+       *scratch = *apc;
+
+       /* Owns no queues yet. */
+       scratch->eqs            = NULL;
+       scratch->tx_qp          = NULL;
+       scratch->rxqs           = NULL;
+       scratch->indir_table    = NULL;
+       scratch->rxobj_table    = NULL;
+       scratch->default_rxobj  = INVALID_MANA_HANDLE;
+       scratch->mana_eqs_debugfs = NULL;
+
+       /* Never consume the live set's pre-allocated RX buffers;
+        * mana_get_rxbuf() falls back to normal allocation when these
+        * are NULL, which is what we want since the swap path no longer
+        * needs to de-risk post-teardown allocation.
+        */
+       scratch->rxbufs_pre     = NULL;
+       scratch->das_pre        = NULL;
+       scratch->rxbpre_total   = 0;
+
+       /* Suppress debugfs for queues built through the scratch context:
+        * two sets are alive at once and would collide on the same names
+        * under vport%d. debugfs_start_creating() returns early on an
+        * IS_ERR() parent, and debugfs_remove() ignores IS_ERR_OR_NULL,
+        * so this makes every create/remove a clean no-op.
+        */
+       scratch->mana_port_debugfs = ERR_PTR(-ENODEV);
+
+       return scratch;
+}
+
+void mana_qset_scratch_free(struct mana_port_context *scratch)
+{
+       kvfree(scratch);
+}
+
+/* Rebuild the queues at the current count in @scratch, for callers changing a
+ * per-queue property; a count change goes through mana_split_qset() or
+ * mana_grow_qset(), so this never has to add an EQ. The installed set keeps
+ * serving traffic meanwhile. On error nothing is left allocated.
+ */
+int mana_alloc_qset(struct mana_port_context *scratch, unsigned int num_queues,
+                   unsigned int rx_queue_size, unsigned int tx_queue_size,
+                   u32 priv_flags, struct mana_qset *out)
+{
+       struct net_device *ndev = scratch->ndev;
+       int err;
+
+       ASSERT_RTNL();
+
+       scratch->num_queues     = num_queues;
+       scratch->rx_queue_size  = rx_queue_size;
+       scratch->tx_queue_size  = tx_queue_size;
+       scratch->priv_flags     = priv_flags;
+
+       err = mana_init_port_context(scratch);
+       if (err)
+               goto out_err;
+
+       err = mana_rss_table_alloc(scratch);
+       if (err)
+               goto cleanup_rxq_array;
+
+       err = mana_create_eq(scratch);
+       if (err)
+               goto cleanup_rss;
+
+       err = mana_create_txq(scratch, ndev);
+       if (err)
+               goto cleanup_eq;
+
+       err = mana_add_rx_queues(scratch, ndev);
+       if (err)
+               goto cleanup_rxq;
+
+       mana_rss_table_init(scratch);
+
+       mana_qset_snapshot(scratch, out);
+       return 0;
+
+cleanup_rxq:
+       /* mana_add_rx_queues() may have created queues before failing; they
+        * own RQ/CQ objects, NAPI state and page pools, so tear down whatever
+        * made it into scratch->rxqs[] before dropping the array.
+        */
+       mana_destroy_rxqs(scratch);
+       mana_destroy_txq(scratch);
+cleanup_eq:
+       mana_destroy_eq(scratch);
+cleanup_rss:
+       mana_cleanup_indir_table(scratch);
+cleanup_rxq_array:
+       kfree(scratch->rxqs);
+       scratch->rxqs = NULL;
+out_err:
+       netdev_err(ndev, "mana_alloc_qset(num_queues=%u) failed: %d\n",
+                  num_queues, err);
+       return err;
+}
+
+/* Tear down @qset, no longer installed on @apc, against @scratch so the live
+ * context never points at queues being freed.
+ */
+void mana_free_qset(struct mana_port_context *scratch, struct mana_qset *qset)
+{
+       struct bpf_prog *retiring_prog;
+       unsigned int retiring_queues;
+
+       ASSERT_RTNL();
+
+       if (!qset->rxqs && !qset->tx_qp && !qset->eqs)
+               return;
+
+       /* These queues are leaving. Stop their completions from touching the
+        * shared netdev queues: net_txq is shared with whatever replaced them
+        * at the same index, and a queue that is only draining always looks
+        * like it has room, so it would wake a live queue that stopped itself
+        * because its ring was full. The synchronize_net() below then retires
+        * any poll that has not seen the flag yet.
+        */
+       if (qset->tx_qp) {
+               unsigned int q;
+
+               for (q = 0; q < qset->num_queues; q++) {
+                       if (qset->tx_qp[q])
+                               WRITE_ONCE(qset->tx_qp[q]->txq.retiring, true);
+               }
+       }
+
+       /* The datapath gates on apc->port_is_up and then dereferences
+        * apc->tx_qp[] / apc->rxqs[] with no lock. mana_publish_qset() drains
+        * those readers before it installs the incoming set, which cannot
+        * cover one that sampled the retiring pointers between that install
+        * and the gate reopening. mana_xdp_xmit() is the case that matters:
+        * it runs from a redirecting device's NAPI, so the napi_synchronize()
+        * that mana_destroy_txq()/mana_destroy_rxq() do on this port's own
+        * NAPIs never waits for it. Give any such reader a grace period to
+        * finish before its queues are torn down under it. Every caller is a
+        * reconfiguration path holding RTNL, so this is expedited.
+        */
+       synchronize_net();
+
+       mana_qset_install(scratch, qset);
+
+       /* Note what this set owes the XDP program, but leave the queues
+        * pointing at it. They are still polling, and a packet already in a
+        * retiring RQ has to keep running the program rather than slip past
+        * it into the stack. The references are dropped once the queues are
+        * gone, below. XDP_TX from those polls is harmless here: it goes
+        * through mana_start_xmit() on the live port context, so it reaches
+        * the queue set that replaced this one, not the one being drained.
+        */
+       retiring_prog = mana_chn_xdp_peek(scratch);
+       retiring_queues = scratch->num_queues;
+
+       /* The retiring TX queues may still hold packets the device has not
+        * completed. Drain them before the SQs and the SKB queues go away,
+        * or those SKBs and their DMA mappings are leaked.
+        *
+        * This runs before any RX teardown, the order mana_dealloc_queues()
+        * uses. A device wedged badly enough to need the reset below is also
+        * one whose RQ teardown will not complete, and unmapping RX buffers
+        * first would leave it free to keep writing into them for as long as
+        * the drain takes.
+        */
+       if (mana_drain_txqs(scratch)) {
+               /* The drain had to reset the function to stop the device
+                * touching those buffers. A function reset takes down every
+                * port on the adapter, not just this one, so rebuild them all
+                * - the same recovery mana_tx_timeout() relies on. A port that
+                * is already down has nothing to rebuild and its handler
+                * leaves it down.
+                */
+               struct mana_port_context *apc = netdev_priv(scratch->ndev);
+               struct mana_context *ac = apc->ac;
+               unsigned int i;
+
+               netdev_err(scratch->ndev,
+                          "device reset while retiring a queue set, scheduling 
port reset\n");
+
+               for (i = 0; i < ac->num_ports; i++) {
+                       if (!ac->ports[i])
+                               continue;
+                       queue_work(ac->per_port_queue_reset_wq,
+                                  &((struct mana_port_context *)
+                                    
netdev_priv(ac->ports[i]))->queue_reset_work);
+               }
+       }
+
+       /* Traffic was still being steered at these queues moments ago, so
+        * fence each retiring RQ before its buffers are unmapped, again the
+        * order mana_dealloc_queues() uses. mana_destroy_rxq() does destroy
+        * the hardware RQ before unmapping anything, but the fence is what
+        * makes the device confirm it is done with the buffers first.
+        */
+       mana_fence_rqs(scratch);
+
+       mana_destroy_rxqs(scratch);
+
+       /* The queues are gone, so nothing can run the program any more. */
+       mana_chn_xdp_release(retiring_prog, retiring_queues);
+
+       mana_destroy_txq(scratch);
+       mana_destroy_eq(scratch);
+       mana_cleanup_indir_table(scratch);
+       kfree(scratch->rxqs);
+       scratch->rxqs = NULL;
+
+       memset(qset, 0, sizeof(*qset));
+}
+
+/* --- end of pre-allocate + swap reconfiguration path ---------------------- 
*/
+
 int mana_detach(struct net_device *ndev, bool from_close)
 {
        struct mana_port_context *apc = netdev_priv(ndev);
@@ -4237,6 +4631,13 @@ void mana_remove(struct gdma_dev *gd, bool suspending)
                unregister_netdevice(ndev);
                mana_cleanup_indir_table(apc);
 
+               /* Clear the slot before the netdev goes away. A later port
+                * whose teardown has to reset the function walks ac->ports[]
+                * to schedule the rebuild, and would otherwise reach into the
+                * port freed here.
+                */
+               ac->ports[i] = NULL;
+
                rtnl_unlock();
 
                free_netdev(ndev);
diff --git a/drivers/net/ethernet/microsoft/mana/mana_ethtool.c 
b/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
index 
ece7ff9cc409a806b6a6de70a85b44874bfa6dad..04b7a5c0fdabc9abc693065c32d4f60fc9ac6809
 100644
--- a/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
+++ b/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
@@ -648,10 +648,11 @@ static int mana_set_coalesce(struct net_device *ndev,
        return 0;
 }
 
-/* mana_set_channels - change the number of queues on a port
- *
- * Returns -EBUSY if RDMA holds the vport with EQs sized to the
- * current num_queues.
+/* A count change leaves every surviving queue configured as it was, so
+ * neither direction rebuilds: a reduction retires the tail, an increase
+ * builds only the queues added. On failure the existing queues keep running
+ * and the requested value is never replaced by a fallback. The vport is never
+ * torn down, so RDMA cannot take it mid-reconfiguration.
  */
 static int mana_set_channels(struct net_device *ndev,
                             struct ethtool_channels *channels)
diff --git a/include/net/mana/mana.h b/include/net/mana/mana.h
index 
83b7eff4646ead7aef1382c6ce565a573a940af4..a7b7a00a57f176e9dc889f7b09b6ef56d5608a8e
 100644
--- a/include/net/mana/mana.h
+++ b/include/net/mana/mana.h
@@ -143,6 +143,16 @@ struct mana_txq {
 
        bool napi_initialized;
 
+       /* Value of mana_context.reset_gen when this queue was created. */
+       u32 reset_gen;
+
+       /* Set once this queue has been unpublished and is on its way out.
+        * Its completions must not touch flow control any more: net_txq is
+        * shared with the queue that replaced it at the same index, and a
+        * draining queue always looks like it has room.
+        */
+       bool retiring;
+
        struct mana_stats_tx stats;
 };
 
@@ -537,6 +547,13 @@ struct mana_context {
        u8 bm_hostmode;
 
        struct mana_ethtool_hc_stats hc_stats;
+
+       /* Bumped on every PCI function reset. A queue created before the
+        * current value can no longer be reached by the device, so its buffers
+        * need no drain. Written under RTNL, read locklessly.
+        */
+       u32 reset_gen;
+
        struct workqueue_struct *per_port_queue_reset_wq;
        /* Workqueue for querying hardware stats */
        struct delayed_work gf_stats_work;
@@ -661,6 +678,39 @@ struct mana_port_context {
        u32 steer_cqe_coalescing;
 };
 
+/* struct mana_qset - a self-contained snapshot of the queue-related
+ * fields inside mana_port_context that can be swapped atomically.
+ *
+ * Prototype for the "pre-allocate + swap" reconfiguration path (as
+ * suggested by netdev maintainers): a new qset is allocated while the
+ * current one keeps serving traffic, then apc's queue fields are
+ * atomically switched to the new set and the old set is torn down.
+ * The vport (port_handle / vport_use_count) is *not* touched, so RDMA
+ * can never race in during reconfiguration.
+ */
+struct mana_qset {
+       struct mana_eq          *eqs;
+       struct mana_tx_qp       **tx_qp;
+       struct mana_rxq         **rxqs;
+
+       u32                     *indir_table;
+       u32                     indir_table_sz;
+       mana_handle_t           *rxobj_table;
+       mana_handle_t           default_rxobj;
+
+       unsigned int            num_queues;
+       unsigned int            rx_queue_size;
+       unsigned int            tx_queue_size;
+       u32                     priv_flags;
+
+       /* Per-queue-set debugfs root ("EQs"). Owned by the qset: it is
+        * recreated by mana_create_eq() for each new set and torn down
+        * with that set, so it must travel with the qset rather than
+        * staying on apc.
+        */
+       struct dentry           *mana_eqs_debugfs;
+};
+
 netdev_tx_t mana_start_xmit(struct sk_buff *skb, struct net_device *ndev);
 int mana_config_rss(struct mana_port_context *ac, enum TRI_STATE rx,
                    bool update_hash, bool update_tab);
@@ -670,6 +720,17 @@ int mana_alloc_queues(struct net_device *ndev);
 int mana_attach(struct net_device *ndev);
 int mana_detach(struct net_device *ndev, bool from_close);
 
+/* Pre-allocate + swap reconfiguration. Allocation and teardown run against a
+ * scratch context, so the live port context is mutated only inside
+ * mana_publish_qset() with TX disabled. Both sets share a port-owned EQ pool.
+ */
+struct mana_port_context *mana_qset_scratch_alloc(struct mana_port_context 
*apc);
+void mana_qset_scratch_free(struct mana_port_context *scratch);
+int mana_alloc_qset(struct mana_port_context *scratch, unsigned int num_queues,
+                   unsigned int rx_queue_size, unsigned int tx_queue_size,
+                   u32 priv_flags, struct mana_qset *out);
+void mana_free_qset(struct mana_port_context *scratch, struct mana_qset *qset);
+
 void mana_dim_change(struct mana_cq *cq, bool enable);
 
 int mana_probe(struct gdma_dev *gd, bool resuming);
@@ -685,6 +746,8 @@ u32 mana_run_xdp(struct net_device *ndev, struct mana_rxq 
*rxq,
                 struct xdp_buff *xdp, void *buf_va, uint pkt_len);
 struct bpf_prog *mana_xdp_get(struct mana_port_context *apc);
 void mana_chn_setxdp(struct mana_port_context *apc, struct bpf_prog *prog);
+struct bpf_prog *mana_chn_xdp_peek(struct mana_port_context *apc);
+void mana_chn_xdp_release(struct bpf_prog *prog, unsigned int num_queues);
 int mana_bpf(struct net_device *ndev, struct netdev_bpf *bpf);
 int mana_query_gf_stats(struct mana_context *ac);
 int mana_query_link_cfg(struct mana_port_context *apc);
-- 
2.43.0


Reply via email to