[PATCH net-next 08/13] net/mlx5e: TC, track peer flows in a vhca_id xarray

From: Tariq Toukan

Date: Wed Sep 23 2026 - 06:55:35 EST


From: Shay Drory <shayd@xxxxxxxxxx>

The per-peer reverse index lived in a fixed esw->offloads.peer_flows[]
array of list heads indexed by the peer's LAG sequence number, capping a
shared FDB at MLX5_MAX_PORTS members.

Replace the array with an xarray keyed by the peer vhca_id. Each entry
is a heap-allocated list head that chains the flows duplicated to that
peer via their peer node. This lifts the MLX5_MAX_PORTS cap: the number
of peers is bounded only by the number of distinct vhca_ids.

The lifetime of the duplicated peer flows list_head follows the devcom
pairing of the two eswitches:
- pair (mlx5_esw_offloads_pair): kzalloc_obj + INIT_LIST_HEAD +
xa_store the dup_peer_flows under the peer's vhca_id.
- add (mlx5e_tc_add_fdb_peer_flow): xa_load the dup_peer_flows and
list_add the duplicated flow's peer node.
- del (mlx5e_tc_del_fdb_peer_flow): list_del that node.
- flush (mlx5e_tc_clean_fdb_peer_flows): xa_for_each dup_peer_flows,
drop every duplicated flow on it; called from unpair.
- unpair (mlx5_esw_offloads_unpair): xa_erase + kfree the
dup_peer_flows.

So a dup_peer_flows is created when the local eswitch pairs with a peer
over devcom and freed when they unpair.

No functional change.

Signed-off-by: Shay Drory <shayd@xxxxxxxxxx>
Reviewed-by: Moshe Shemesh <moshe@xxxxxxxxxx>
Reviewed-by: Akiva Goldberger <agoldberger@xxxxxxxxxx>
Signed-off-by: Tariq Toukan <tariqt@xxxxxxxxxx>
---
.../net/ethernet/mellanox/mlx5/core/en_tc.c | 22 +++++++---------
.../net/ethernet/mellanox/mlx5/core/eswitch.h | 2 +-
.../mellanox/mlx5/core/eswitch_offloads.c | 26 ++++++++++++++++---
3 files changed, 34 insertions(+), 16 deletions(-)

diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c b/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c
index fae4f8625da4..c99e824b4c6d 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/en_tc.c
@@ -4596,12 +4596,13 @@ static int mlx5e_tc_add_fdb_peer_flow(struct flow_cls_offload *f,
unsigned long flow_flags,
struct mlx5_eswitch *peer_esw)
{
+ u16 peer_vhca_id = MLX5_CAP_GEN(peer_esw->dev, vhca_id);
struct mlx5e_priv *priv = flow->priv, *peer_priv;
struct mlx5_eswitch *esw = priv->mdev->priv.eswitch;
struct mlx5_esw_flow_attr *attr = flow->attr->esw_attr;
struct mlx5e_tc_flow_parse_attr *parse_attr;
- int i = mlx5_lag_get_dev_seq(peer_esw->dev);
struct mlx5e_rep_priv *peer_urpriv;
+ struct list_head *dup_peer_flows;
struct mlx5e_tc_flow *peer_flow;
struct mlx5_core_dev *in_mdev;
int err = 0;
@@ -4634,8 +4635,11 @@ static int mlx5e_tc_add_fdb_peer_flow(struct flow_cls_offload *f,
peer_flow->peer_orig = flow;
list_add_tail(&peer_flow->peer_flows, &flow->peer_flows);
flow_flag_set(flow, DUP);
+ dup_peer_flows = xa_load(&esw->offloads.peer_flows, peer_vhca_id);
+ if (!dup_peer_flows)
+ return -ENODEV;
mutex_lock(&esw->offloads.peer_mutex);
- list_add_tail(&peer_flow->peer, &esw->offloads.peer_flows[i]);
+ list_add_tail(&peer_flow->peer, dup_peer_flows);
mutex_unlock(&esw->offloads.peer_mutex);

out:
@@ -5530,20 +5534,14 @@ int mlx5e_tc_num_filters(struct mlx5e_priv *priv, unsigned long flags)

void mlx5e_tc_clean_fdb_peer_flows(struct mlx5_eswitch *esw)
{
- struct mlx5_devcom_comp_dev *devcom = esw->devcom, *pos;
struct mlx5e_tc_flow *peer_flow, *tmp_peer_flow;
- struct mlx5_eswitch *peer_esw;
- int i;
-
- mlx5_devcom_for_each_peer_entry(devcom, peer_esw, pos) {
- i = mlx5_lag_get_dev_seq(peer_esw->dev);
- if (i < 0)
- continue;
+ struct list_head *dup_peer_flows;
+ unsigned long index;

+ xa_for_each(&esw->offloads.peer_flows, index, dup_peer_flows)
list_for_each_entry_safe(peer_flow, tmp_peer_flow,
- &esw->offloads.peer_flows[i], peer)
+ dup_peer_flows, peer)
mlx5e_tc_del_fdb_peer_flow(peer_flow);
- }
}

void mlx5e_tc_reoffload_flows_work(struct work_struct *work)
diff --git a/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h b/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h
index 8b1f93b13ea9..e9cbcd73b23f 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h
+++ b/drivers/net/ethernet/mellanox/mlx5/core/eswitch.h
@@ -319,7 +319,7 @@ struct mlx5_esw_offload {
struct mlx5_flow_handle *vport_rx_drop_rule;
struct mlx5_flow_table *ft_ipsec_tx_pol;
struct xarray vport_reps;
- struct list_head peer_flows[MLX5_MAX_PORTS];
+ struct xarray peer_flows;
struct mutex peer_mutex;
struct mutex encap_tbl_lock; /* protects encap_tbl */
DECLARE_HASHTABLE(encap_tbl, 8);
diff --git a/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c b/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c
index e7d92d9bde16..c712848b202d 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/eswitch_offloads.c
@@ -3383,10 +3383,15 @@ static void mlx5_esw_offloads_rep_event_unpair(struct mlx5_eswitch *esw,
static void mlx5_esw_offloads_unpair(struct mlx5_eswitch *esw,
struct mlx5_eswitch *peer_esw)
{
+ struct list_head *dup_peer_flows;
+
#if IS_ENABLED(CONFIG_MLX5_CLS_ACT)
mlx5e_tc_clean_fdb_peer_flows(esw);
#endif
mlx5_esw_offloads_rep_event_unpair(esw, peer_esw);
+ dup_peer_flows = xa_erase(&esw->offloads.peer_flows,
+ MLX5_CAP_GEN(peer_esw->dev, vhca_id));
+ kfree(dup_peer_flows);
esw_del_fdb_peer_miss_rules(esw, peer_esw->dev);
}

@@ -3394,6 +3399,7 @@ static int mlx5_esw_offloads_pair(struct mlx5_eswitch *esw,
struct mlx5_eswitch *peer_esw)
{
const struct mlx5_eswitch_rep_ops *ops;
+ struct list_head *dup_peer_flows;
struct mlx5_eswitch_rep *rep;
unsigned long i;
u8 rep_type;
@@ -3403,6 +3409,20 @@ static int mlx5_esw_offloads_pair(struct mlx5_eswitch *esw,
if (err)
return err;

+ dup_peer_flows = kzalloc_obj(*dup_peer_flows);
+ if (!dup_peer_flows) {
+ err = -ENOMEM;
+ goto err_out;
+ }
+ INIT_LIST_HEAD(dup_peer_flows);
+ err = xa_err(xa_store(&esw->offloads.peer_flows,
+ MLX5_CAP_GEN(peer_esw->dev, vhca_id),
+ dup_peer_flows, GFP_KERNEL));
+ if (err) {
+ kfree(dup_peer_flows);
+ goto err_out;
+ }
+
mlx5_esw_for_each_rep(esw, i, rep) {
for (rep_type = 0; rep_type < NUM_REP_TYPES; rep_type++) {
ops = esw->offloads.rep_ops[rep_type];
@@ -3549,10 +3569,8 @@ void mlx5_esw_offloads_devcom_init(struct mlx5_eswitch *esw,
const struct mlx5_devcom_match_attr *attr)
{
int err;
- int i;

- for (i = 0; i < MLX5_MAX_PORTS; i++)
- INIT_LIST_HEAD(&esw->offloads.peer_flows[i]);
+ xa_init(&esw->offloads.peer_flows);
mutex_init(&esw->offloads.peer_mutex);

if (!MLX5_CAP_ESW(esw->dev, merged_eswitch))
@@ -3595,6 +3613,8 @@ void mlx5_esw_offloads_devcom_cleanup(struct mlx5_eswitch *esw)
xa_destroy(&esw->paired);
xa_destroy(&esw->fdb_table.offloads.peer_miss_rules);
esw->devcom = NULL;
+ WARN_ON_ONCE(!xa_empty(&esw->offloads.peer_flows));
+ xa_destroy(&esw->offloads.peer_flows);
}

bool mlx5_esw_offloads_devcom_is_ready(struct mlx5_eswitch *esw)
--
2.44.0