[PATCH net-next v3 06/13] net: mana: swap queue sets in mana_change_mtu
From: Long Li <longli@microsoft.com>
Date: 2026-09-01 01:45:29
Also in:
linux-hyperv, linux-rdma, lkml
Subsystem:
hyper-v/azure core and drivers, networking drivers, networking [general], the rest · Maintainers:
"K. Y. Srinivasan", Haiyang Zhang, Wei Liu, Dexuan Cui, Long Li, Andrew Lunn, "David S. Miller", Eric Dumazet, Jakub Kicinski, Paolo Abeni, Linus Torvalds
The RX buffer layout depends on the MTU, so changing it rebuilds the queues. Convert mana_change_mtu() to pre-allocate and swap. The MTU becomes part of the queue-set configuration, so a new set can be built for the new MTU while the running one still serves traffic at the old one, and ndev->mtu is updated only once the new set is live. Previously it was written before mana_attach() and rolled back on failure, so a failed change was briefly visible to the stack. Signed-off-by: Long Li <longli@microsoft.com> --- drivers/net/ethernet/microsoft/mana/mana_en.c | 69 ++++++++++++++----- .../ethernet/microsoft/mana/mana_ethtool.c | 10 +-- include/net/mana/mana.h | 12 +++- 3 files changed, 67 insertions(+), 24 deletions(-)
diff --git a/drivers/net/ethernet/microsoft/mana/mana_en.c b/drivers/net/ethernet/microsoft/mana/mana_en.c
index 880a3ba37fd3e872dbfeb101c46ac8f527c4bbdd..2c5aa5e5d1a114e1492b99b5213bb532153d1147 100644
--- a/drivers/net/ethernet/microsoft/mana/mana_en.c
+++ b/drivers/net/ethernet/microsoft/mana/mana_en.c@@ -917,35 +917,49 @@ int mana_pre_alloc_rxbufs(struct mana_port_context *mpc, int new_mtu, int num_qu return -ENOMEM; } +/* ndev->mtu is updated only once the new set is live (mana_publish_qset), so + * a failed allocation leaves the queues and the advertised MTU untouched. + */ static int mana_change_mtu(struct net_device *ndev, int new_mtu) { struct mana_port_context *mpc = netdev_priv(ndev); - unsigned int old_mtu = ndev->mtu; + struct mana_port_context *scratch; + struct mana_qset newq, oldq; int err; - /* Pre-allocate buffers to prevent failure in mana_attach later */ - err = mana_pre_alloc_rxbufs(mpc, new_mtu, mpc->num_queues); - if (err) { - netdev_err(ndev, "Insufficient memory for new MTU\n"); - return err; + /* Port is down: no queues to rebuild, just record the new MTU. + * mana_open() will size the RX buffers accordingly. + */ + if (!mpc->port_is_up) { + mpc->configured_mtu = new_mtu; + WRITE_ONCE(ndev->mtu, new_mtu); + return 0; } - err = mana_detach(ndev, false); - if (err) { - netdev_err(ndev, "mana_detach failed: %d\n", err); - goto out; - } + scratch = mana_qset_scratch_alloc(mpc); + if (!scratch) + return -ENOMEM; - WRITE_ONCE(ndev->mtu, new_mtu); + err = mana_alloc_qset(mpc, scratch, mpc->num_queues, + mpc->rx_queue_size, mpc->tx_queue_size, + mpc->priv_flags, new_mtu, &newq); + if (err) + goto free_scratch; /* current qset and ndev->mtu untouched */ - err = mana_attach(ndev); + err = mana_publish_qset(mpc, &newq, &oldq); if (err) { - netdev_err(ndev, "mana_attach failed: %d\n", err); - WRITE_ONCE(ndev->mtu, old_mtu); + mana_free_qset(scratch, &newq); + goto free_scratch; } -out: - mana_pre_dealloc_rxbufs(mpc); + mana_free_qset(scratch, &oldq); + +free_scratch: + /* After the caller-side cleanup above, so the EQ pool outlives the + * CQs that reference it. + */ + mana_publish_close_if_needed(mpc); + mana_qset_scratch_free(scratch); return err; }
@@ -3195,7 +3209,8 @@ static struct mana_rxq *mana_create_rxq(struct mana_port_context *apc, rxq->rxq_idx = rxq_idx; rxq->rxobj = INVALID_MANA_HANDLE; - mana_get_rxbuf_cfg(apc, ndev->mtu, &rxq->datasize, &rxq->alloc_size, + mana_get_rxbuf_cfg(apc, apc->configured_mtu, &rxq->datasize, + &rxq->alloc_size, &rxq->headroom, &rxq->frag_count); /* Create page pool for RX queue */ err = mana_create_page_pool(rxq, gc);
@@ -4006,6 +4021,7 @@ static void mana_qset_snapshot(const struct mana_port_context *ctx, out->rx_queue_size = ctx->rx_queue_size; out->tx_queue_size = ctx->tx_queue_size; out->priv_flags = ctx->priv_flags; + out->mtu = ctx->configured_mtu; } /* Install @qset's fields onto @ctx. The vport (port_handle,
@@ -4025,6 +4041,7 @@ static void mana_qset_install(struct mana_port_context *ctx, ctx->rx_queue_size = qset->rx_queue_size; ctx->tx_queue_size = qset->tx_queue_size; ctx->priv_flags = qset->priv_flags; + ctx->configured_mtu = qset->mtu; } /**
@@ -4083,7 +4100,7 @@ void mana_qset_scratch_free(struct mana_port_context *scratch) int mana_alloc_qset(struct mana_port_context *apc, struct mana_port_context *scratch, unsigned int num_queues, unsigned int rx_queue_size, unsigned int tx_queue_size, - u32 priv_flags, struct mana_qset *out) + u32 priv_flags, int mtu, struct mana_qset *out) { struct net_device *ndev = scratch->ndev; int err;
@@ -4095,6 +4112,12 @@ int mana_alloc_qset(struct mana_port_context *apc, scratch->tx_queue_size = tx_queue_size; scratch->priv_flags = priv_flags; + /* mana_get_rxbuf_cfg() reads this when sizing RX buffers, so the + * new set is built for the requested MTU without disturbing the + * running set. + */ + scratch->configured_mtu = mtu; + err = mana_init_port_context(scratch); if (err) goto out_err;
@@ -4335,6 +4358,11 @@ int mana_publish_qset(struct mana_port_context *apc, struct mana_qset *newq, if (err) goto rollback; + /* The new set is serving traffic, so advertise its MTU. A no-op unless + * the caller is changing it. + */ + WRITE_ONCE(ndev->mtu, apc->configured_mtu); + /* Pair with the queue-state stores above: a datapath reader that sees * the gate open must also see the queue set it is about to index. */
@@ -4389,6 +4417,8 @@ int mana_publish_qset(struct mana_port_context *apc, struct mana_qset *newq, return err; } + WRITE_ONCE(ndev->mtu, apc->configured_mtu); + /* Same pairing as the success path: the restored queue set has to be * visible before the gate reopens on it. */
@@ -4600,6 +4630,7 @@ static int mana_probe_port(struct mana_context *ac, int port_idx, apc->port_handle = INVALID_MANA_HANDLE; apc->pf_filter_handle = INVALID_MANA_HANDLE; apc->port_idx = port_idx; + apc->configured_mtu = ndev->mtu; apc->link_cfg_error = 1; apc->cqe_coalescing_enable = 0; apc->cqe8_coalescing_enable = 0;
diff --git a/drivers/net/ethernet/microsoft/mana/mana_ethtool.c b/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
index eab7df3fb888b3e0cc2e0657965da2ba4190ecd8..d01add523576f97bea7c02dfdab1073cb24cbca1 100644
--- a/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
+++ b/drivers/net/ethernet/microsoft/mana/mana_ethtool.c@@ -722,7 +722,8 @@ static int mana_set_channels(struct net_device *ndev, } err = mana_alloc_qset(apc, scratch, new_count, apc->rx_queue_size, - apc->tx_queue_size, apc->priv_flags, &newq); + apc->tx_queue_size, apc->priv_flags, + apc->configured_mtu, &newq); if (err) goto free_scratch; /* current qset untouched, nothing to undo */
@@ -816,7 +817,7 @@ static int mana_set_ringparam(struct net_device *ndev, } err = mana_alloc_qset(apc, scratch, apc->num_queues, new_rx, new_tx, - apc->priv_flags, &newq); + apc->priv_flags, apc->configured_mtu, &newq); if (err) { NL_SET_ERR_MSG_FMT(extack, "failed to change ring params: %d", err);
@@ -913,8 +914,9 @@ static int mana_set_priv_flags(struct net_device *ndev, u32 priv_flags) goto clear_flag; } - err = mana_alloc_qset(apc, scratch, apc->num_queues, apc->rx_queue_size, - apc->tx_queue_size, priv_flags, &newq); + err = mana_alloc_qset(apc, scratch, apc->num_queues, + apc->rx_queue_size, apc->tx_queue_size, + priv_flags, apc->configured_mtu, &newq); if (err) goto free_scratch; /* current qset and priv_flags untouched */
diff --git a/include/net/mana/mana.h b/include/net/mana/mana.h
index dd767ab623912a149b4445cea7453034c8d69e37..765eb5358e9ca2b9977096631db9aab37cedacfd 100644
--- a/include/net/mana/mana.h
+++ b/include/net/mana/mana.h@@ -626,6 +626,11 @@ struct mana_port_context { unsigned int rx_queue_size; unsigned int tx_queue_size; + /* MTU the RX queues were built for. Equal to ndev->mtu except during a + * swap, when the new set is built before ndev->mtu is updated. + */ + int configured_mtu; + mana_handle_t port_handle; mana_handle_t pf_filter_handle;
@@ -717,6 +722,11 @@ struct mana_qset { unsigned int tx_queue_size; u32 priv_flags; + /* MTU the RX buffers of this set were sized for. It feeds + * mana_get_rxbuf_cfg(), so it is part of the queue-set + * configuration and must be swapped atomically with the queues. + */ + int mtu; }; netdev_tx_t mana_start_xmit(struct sk_buff *skb, struct net_device *ndev);
@@ -738,7 +748,7 @@ void mana_qset_scratch_free(struct mana_port_context *scratch); int mana_alloc_qset(struct mana_port_context *apc, struct mana_port_context *scratch, unsigned int num_queues, unsigned int rx_queue_size, unsigned int tx_queue_size, - u32 priv_flags, struct mana_qset *out); + u32 priv_flags, int mtu, struct mana_qset *out); int mana_publish_qset(struct mana_port_context *apc, struct mana_qset *newq, struct mana_qset *out_old); void mana_publish_close_if_needed(struct mana_port_context *apc);
--
2.43.0