Thread (26 messages) 26 messages, 4 authors, 11d ago

[PATCH net-next 08/13] net/mlx5e: TC, track peer flows in a vhca_id xarray

flat view
COOLING11d IN LINUX-NEXT: 2 (0M)

From: Tariq Toukan <tariqt@nvidia.com>
Date: 2026-09-23 10:41:13
Also in: linux-rdma, lkml
Subsystem: mellanox ethernet driver (mlx5e), mellanox mlx5 core vpi driver, networking drivers, the rest · Maintainers: Saeed Mahameed, Tariq Toukan, Mark Bloch, Leon Romanovsky, Andrew Lunn, "David S. Miller", Eric Dumazet, Jakub Kicinski, Paolo Abeni, Linus Torvalds

2 review trailers; queued in linux-next as 9d608a40c2d4 on 2026-09-29.

From: Shay Drory <redacted>

The per-peer reverse index lived in a fixed esw->offloads.peer_flows[]
array of list heads indexed by the peer's LAG sequence number, capping a
shared FDB at MLX5_MAX_PORTS members.

Replace the array with an xarray keyed by the peer vhca_id. Each entry
is a heap-allocated list head that chains the flows duplicated to that
peer via their peer node. This lifts the MLX5_MAX_PORTS cap: the number
of peers is bounded only by the number of distinct vhca_ids.

The lifetime of the duplicated peer flows list_head follows the devcom
pairing of the two eswitches:
  - pair   (mlx5_esw_offloads_pair): kzalloc_obj + INIT_LIST_HEAD +
	   xa_store the dup_peer_flows under the peer's vhca_id.
  - add    (mlx5e_tc_add_fdb_peer_flow): xa_load the dup_peer_flows and
	   list_add the duplicated flow's peer node.
  - del    (mlx5e_tc_del_fdb_peer_flow): list_del that node.
  - flush  (mlx5e_tc_clean_fdb_peer_flows): xa_for_each dup_peer_flows,
	   drop every duplicated flow on it; called from unpair.
  - unpair (mlx5_esw_offloads_unpair):  xa_erase + kfree the
	   dup_peer_flows.

So a dup_peer_flows is created when the local eswitch pairs with a peer
over devcom and freed when they unpair.

No functional change.

Signed-off-by: Shay Drory <redacted>
Reviewed-by: Moshe Shemesh <redacted>
Reviewed-by: Akiva Goldberger <redacted>
Signed-off-by: Tariq Toukan <tariqt@nvidia.com>
---
 .../net/ethernet/mellanox/mlx5/core/en_tc.c   | 22 +++++++---------
 .../net/ethernet/mellanox/mlx5/core/eswitch.h |  2 +-
 .../mellanox/mlx5/core/eswitch_offloads.c     | 26 ++++++++++++++++---
 3 files changed, 34 insertions(+), 16 deletions(-)
diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c b/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c
index fae4f8625da4..c99e824b4c6d 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c
@@ -4596,12 +4596,13 @@ static int mlx5e_tc_add_fdb_peer_flow(struct flow_cls_offload *f,
 				      unsigned long flow_flags,
 				      struct mlx5_eswitch *peer_esw)
 {
+	u16 peer_vhca_id = MLX5_CAP_GEN(peer_esw->dev, vhca_id);
 	struct mlx5e_priv *priv = flow->priv, *peer_priv;
 	struct mlx5_eswitch *esw = priv->mdev->priv.eswitch;
 	struct mlx5_esw_flow_attr *attr = flow->attr->esw_attr;
 	struct mlx5e_tc_flow_parse_attr *parse_attr;
-	int i = mlx5_lag_get_dev_seq(peer_esw->dev);
 	struct mlx5e_rep_priv *peer_urpriv;
+	struct list_head *dup_peer_flows;
 	struct mlx5e_tc_flow *peer_flow;
 	struct mlx5_core_dev *in_mdev;
 	int err = 0;
@@ -4634,8 +4635,11 @@ static int mlx5e_tc_add_fdb_peer_flow(struct flow_cls_offload *f,
 	peer_flow->peer_orig = flow;
 	list_add_tail(&peer_flow->peer_flows, &flow->peer_flows);
 	flow_flag_set(flow, DUP);
+	dup_peer_flows = xa_load(&esw->offloads.peer_flows, peer_vhca_id);
+	if (!dup_peer_flows)
+		return -ENODEV;
 	mutex_lock(&esw->offloads.peer_mutex);
-	list_add_tail(&peer_flow->peer, &esw->offloads.peer_flows[i]);
+	list_add_tail(&peer_flow->peer, dup_peer_flows);
 	mutex_unlock(&esw->offloads.peer_mutex);
 
 out:
@@ -5530,20 +5534,14 @@ int mlx5e_tc_num_filters(struct mlx5e_priv *priv, unsigned long flags)
 
 void mlx5e_tc_clean_fdb_peer_flows(struct mlx5_eswitch *esw)
 {
-	struct mlx5_devcom_comp_dev *devcom = esw->devcom, *pos;
 	struct mlx5e_tc_flow *peer_flow, *tmp_peer_flow;
-	struct mlx5_eswitch *peer_esw;
-	int i;
-
-	mlx5_devcom_for_each_peer_entry(devcom, peer_esw, pos) {
-		i = mlx5_lag_get_dev_seq(peer_esw->dev);
-		if (i < 0)
-			continue;
+	struct list_head *dup_peer_flows;
+	unsigned long index;
 
+	xa_for_each(&esw->offloads.peer_flows, index, dup_peer_flows)
 		list_for_each_entry_safe(peer_flow, tmp_peer_flow,
-					 &esw->offloads.peer_flows[i], peer)
+					 dup_peer_flows, peer)
 			mlx5e_tc_del_fdb_peer_flow(peer_flow);
-	}
 }
 
 void mlx5e_tc_reoffload_flows_work(struct work_struct *work)
diff --git a/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h b/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h
index 8b1f93b13ea9..e9cbcd73b23f 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h
+++ b/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h
@@ -319,7 +319,7 @@ struct mlx5_esw_offload {
 	struct mlx5_flow_handle *vport_rx_drop_rule;
 	struct mlx5_flow_table *ft_ipsec_tx_pol;
 	struct xarray vport_reps;
-	struct list_head peer_flows[MLX5_MAX_PORTS];
+	struct xarray peer_flows;
 	struct mutex peer_mutex;
 	struct mutex encap_tbl_lock; /* protects encap_tbl */
 	DECLARE_HASHTABLE(encap_tbl, 8);
diff --git a/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c b/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c
index e7d92d9bde16..c712848b202d 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c
@@ -3383,10 +3383,15 @@ static void mlx5_esw_offloads_rep_event_unpair(struct mlx5_eswitch *esw,
 static void mlx5_esw_offloads_unpair(struct mlx5_eswitch *esw,
 				     struct mlx5_eswitch *peer_esw)
 {
+	struct list_head *dup_peer_flows;
+
 #if IS_ENABLED(CONFIG_MLX5_CLS_ACT)
 	mlx5e_tc_clean_fdb_peer_flows(esw);
 #endif
 	mlx5_esw_offloads_rep_event_unpair(esw, peer_esw);
+	dup_peer_flows = xa_erase(&esw->offloads.peer_flows,
+				  MLX5_CAP_GEN(peer_esw->dev, vhca_id));
+	kfree(dup_peer_flows);
 	esw_del_fdb_peer_miss_rules(esw, peer_esw->dev);
 }
 
@@ -3394,6 +3399,7 @@ static int mlx5_esw_offloads_pair(struct mlx5_eswitch *esw,
 				  struct mlx5_eswitch *peer_esw)
 {
 	const struct mlx5_eswitch_rep_ops *ops;
+	struct list_head *dup_peer_flows;
 	struct mlx5_eswitch_rep *rep;
 	unsigned long i;
 	u8 rep_type;
@@ -3403,6 +3409,20 @@ static int mlx5_esw_offloads_pair(struct mlx5_eswitch *esw,
 	if (err)
 		return err;
 
+	dup_peer_flows = kzalloc_obj(*dup_peer_flows);
+	if (!dup_peer_flows) {
+		err = -ENOMEM;
+		goto err_out;
+	}
+	INIT_LIST_HEAD(dup_peer_flows);
+	err = xa_err(xa_store(&esw->offloads.peer_flows,
+			      MLX5_CAP_GEN(peer_esw->dev, vhca_id),
+			      dup_peer_flows, GFP_KERNEL));
+	if (err) {
+		kfree(dup_peer_flows);
+		goto err_out;
+	}
+
 	mlx5_esw_for_each_rep(esw, i, rep) {
 		for (rep_type = 0; rep_type < NUM_REP_TYPES; rep_type++) {
 			ops = esw->offloads.rep_ops[rep_type];
@@ -3549,10 +3569,8 @@ void mlx5_esw_offloads_devcom_init(struct mlx5_eswitch *esw,
 				   const struct mlx5_devcom_match_attr *attr)
 {
 	int err;
-	int i;
 
-	for (i = 0; i < MLX5_MAX_PORTS; i++)
-		INIT_LIST_HEAD(&esw->offloads.peer_flows[i]);
+	xa_init(&esw->offloads.peer_flows);
 	mutex_init(&esw->offloads.peer_mutex);
 
 	if (!MLX5_CAP_ESW(esw->dev, merged_eswitch))
@@ -3595,6 +3613,8 @@ void mlx5_esw_offloads_devcom_cleanup(struct mlx5_eswitch *esw)
 	xa_destroy(&esw->paired);
 	xa_destroy(&esw->fdb_table.offloads.peer_miss_rules);
 	esw->devcom = NULL;
+	WARN_ON_ONCE(!xa_empty(&esw->offloads.peer_flows));
+	xa_destroy(&esw->offloads.peer_flows);
 }
 
 bool mlx5_esw_offloads_devcom_is_ready(struct mlx5_eswitch *esw)
-- 
2.44.0
Keyboard shortcuts
hback out one level
jnext message in thread
kprevious message in thread
ldrill in
Escclose help / fold thread tree
?toggle this help