[PATCH net-next v2 7/7] net: bcmgenet: reassemble jumbo frames from status block fragments
From: Nicolai Buchwitz
Date: Mon Oct 05 2026 - 18:29:47 EST
The hardware does not truncate a frame longer than the packet ready
threshold. It splits the frame across descriptors and writes a status block
at the start of each one. The first fragment then arrives with SOP and no
EOP and is dropped as fragmented. This caps the MTU.
Strip the status blocks and reassemble the fragments. Only the last block
holds the checksum of the whole frame. Broadcom confirmed from the RTL that
every GENET revision splits long frames this way, not just the v5 this was
tested on.
The MAC only checksums a frame it holds in full, so anything longer than
the threshold falls back to software. At jumbo sizes the larger frame saves
more per packet overhead than the checksum costs.
UMAC_MAX_FRAME_LEN is 14 bit and counts the FCS. That puts the maximum MTU
at 16347.
Suggested-by: Justin Chen <justin.chen@xxxxxxxxxxxx>
Signed-off-by: Nicolai Buchwitz <nb@xxxxxxxxxxx>
Tested-by: Pierre-Marin Leclercq <pierremarinleclercq88@xxxxxxxxx>
---
drivers/net/ethernet/broadcom/genet/bcmgenet.c | 106 +++++++++++++++++++++----
drivers/net/ethernet/broadcom/genet/bcmgenet.h | 2 +
2 files changed, 94 insertions(+), 14 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/genet/bcmgenet.c b/drivers/net/ethernet/broadcom/genet/bcmgenet.c
index e8f86374c7cd..faa13f12ce7e 100644
--- a/drivers/net/ethernet/broadcom/genet/bcmgenet.c
+++ b/drivers/net/ethernet/broadcom/genet/bcmgenet.c
@@ -84,11 +84,8 @@
ENET_THLD_MAX * ENET_THLD_UNIT, \
ENET_THLD_PAGE_LEN)
-/* Largest MTU that fits one descriptor, with room for a VLAN tag so a VLAN
- * interface can use the parent MTU.
- */
-#define ENET_MAX_MTU (ENET_THLD_MAX_LEN - GENET_RBUF_ALIGN - \
- ETH_HLEN - VLAN_HLEN)
+/* UMAC_MAX_FRAME_LEN is 14 bits wide and counts the FCS */
+#define ENET_MAX_JUMBO_MTU (GENMASK(13, 0) - ENET_FRAME_OVERHEAD)
/* Tx/Rx DMA register offset, skip 256 descriptors */
#define WORDS_PER_BD(p) (p->hw_params->words_per_bd)
@@ -2178,8 +2175,8 @@ static netdev_tx_t bcmgenet_xmit(struct sk_buff *skb, struct net_device *dev)
goto out;
}
- /* The MAC only inserts a checksum into a frame it holds in full, and
- * silently drops a longer one, so fall back to software.
+ /* The MAC holds a frame to insert its checksum, but only as much as
+ * its FIFO takes. Longer frames are dropped silently.
*/
if (unlikely(skb->len > priv->tx_thld_len) &&
skb->ip_summed == CHECKSUM_PARTIAL) {
@@ -2333,6 +2330,54 @@ static int bcmgenet_rx_refill(struct bcmgenet_rx_ring *ring,
return 0;
}
+/* Drop the frame being collected. Its remaining descriptors carry no SOP,
+ * so they are dropped quietly until the next one does.
+ */
+static void bcmgenet_discard_frags(struct bcmgenet_rx_ring *ring)
+{
+ ring->frag_drop = true;
+
+ if (!ring->frag_head)
+ return;
+
+ dev_kfree_skb_any(ring->frag_head);
+ ring->frag_head = NULL;
+}
+
+/* A frame longer than the threshold arrives in several descriptors, each with
+ * its own status block. Only the first one carries a header, so hand the page
+ * of every later one to the frame already being collected. Returns the frame
+ * once EOP is in, NULL while more descriptors are expected or once the frame
+ * had to be dropped.
+ */
+static struct sk_buff *bcmgenet_add_frag(struct bcmgenet_rx_ring *ring,
+ struct page *page,
+ unsigned int offset,
+ unsigned int size,
+ unsigned int dma_flag,
+ unsigned int len)
+{
+ struct sk_buff *head = ring->frag_head;
+
+ if (unlikely(skb_shinfo(head)->nr_frags >= MAX_SKB_FRAGS)) {
+ BCMGENET_STATS64_INC((&ring->stats64), fragmented_errors);
+ bcmgenet_discard_frags(ring);
+ page_pool_put_full_page(ring->page_pool, page, true);
+ return NULL;
+ }
+
+ skb_add_rx_frag(head, skb_shinfo(head)->nr_frags, page,
+ offset + sizeof(struct status_64),
+ len - sizeof(struct status_64), size);
+
+ if (!(dma_flag & DMA_EOP))
+ return NULL;
+
+ ring->frag_head = NULL;
+
+ return head;
+}
+
/* bcmgenet_desc_rx - descriptor based rx process.
* this could be called from bottom half, or from NAPI polling method.
*/
@@ -2397,6 +2442,7 @@ static unsigned int bcmgenet_desc_rx(struct bcmgenet_rx_ring *ring,
if (bcmgenet_rx_refill(ring, cb)) {
BCMGENET_STATS64_INC(stats, dropped);
+ bcmgenet_discard_frags(ring);
goto next;
}
@@ -2430,15 +2476,25 @@ static unsigned int bcmgenet_desc_rx(struct bcmgenet_rx_ring *ring,
netif_err(priv, rx_status, dev,
"invalid packet length %d\n", len);
BCMGENET_STATS64_INC(stats, length_errors);
+ bcmgenet_discard_frags(ring);
page_pool_put_full_page(ring->page_pool, rx_page,
true);
goto next;
}
- if (unlikely(!(dma_flag & DMA_EOP) || !(dma_flag & DMA_SOP))) {
- netif_err(priv, rx_status, dev,
- "dropping fragmented packet!\n");
- BCMGENET_STATS64_INC(stats, fragmented_errors);
+ /* A new SOP resynchronizes after an incomplete frame */
+ if (dma_flag & DMA_SOP) {
+ if (ring->frag_head) {
+ BCMGENET_STATS64_INC(stats, fragmented_errors);
+ bcmgenet_discard_frags(ring);
+ }
+ ring->frag_drop = false;
+ } else if (unlikely(!ring->frag_head)) {
+ /* Rest of a dropped frame, or no SOP seen yet */
+ if (!ring->frag_drop) {
+ BCMGENET_STATS64_INC(stats, fragmented_errors);
+ ring->frag_drop = true;
+ }
page_pool_put_full_page(ring->page_pool, rx_page,
true);
goto next;
@@ -2468,17 +2524,27 @@ static unsigned int bcmgenet_desc_rx(struct bcmgenet_rx_ring *ring,
DMA_RX_RXER)) == DMA_RX_RXER)
u64_stats_inc(&stats->errors);
u64_stats_update_end(&stats->syncp);
+ bcmgenet_discard_frags(ring);
page_pool_put_full_page(ring->page_pool, rx_page,
true);
goto next;
} /* error packet */
+ if (!(dma_flag & DMA_SOP)) {
+ skb = bcmgenet_add_frag(ring, rx_page, rx_offset,
+ rx_size, dma_flag, len);
+ if (!skb)
+ goto next;
+ goto deliver;
+ }
+
/* Build SKB from the page - data starts at hard_start,
* frame begins after RSB(64) + pad(2) = 66 bytes.
*/
skb = napi_build_skb(hard_start, rx_size);
if (unlikely(!skb)) {
BCMGENET_STATS64_INC(stats, dropped);
+ bcmgenet_discard_frags(ring);
page_pool_put_full_page(ring->page_pool, rx_page,
true);
goto next;
@@ -2490,8 +2556,18 @@ static unsigned int bcmgenet_desc_rx(struct bcmgenet_rx_ring *ring,
skb_reserve(skb, GENET_RSB_PAD);
__skb_put(skb, len - GENET_RSB_PAD);
- if (priv->crc_fwd_en) {
- skb_trim(skb, skb->len - ETH_FCS_LEN);
+ if (unlikely(!(dma_flag & DMA_EOP))) {
+ ring->frag_head = skb;
+ goto next;
+ }
+
+deliver:
+
+ if (priv->crc_fwd_en &&
+ unlikely(pskb_trim(skb, skb->len - ETH_FCS_LEN))) {
+ BCMGENET_STATS64_INC(stats, dropped);
+ dev_kfree_skb_any(skb);
+ goto next;
}
/* Set up checksum offload */
@@ -2608,6 +2684,8 @@ static void bcmgenet_free_rx_buffers(struct bcmgenet_priv *priv)
cb = ring->cbs + i;
bcmgenet_free_rx_cb(cb, ring->page_pool);
}
+ /* a partial frame still holds pages of this pool */
+ bcmgenet_discard_frags(ring);
}
}
@@ -4293,7 +4371,7 @@ static int bcmgenet_probe(struct platform_device *pdev)
/* v1 cannot program the thresholds, so it stays at the default MTU */
priv->rx_buf_len = bcmgenet_rx_buf_len(dev->mtu);
if (!GENET_IS_V1(priv))
- dev->max_mtu = ENET_MAX_MTU;
+ dev->max_mtu = ENET_MAX_JUMBO_MTU;
INIT_WORK(&priv->bcmgenet_irq_work, bcmgenet_irq_task);
priv->clk_wol = devm_clk_get_optional(&priv->pdev->dev, "enet-wol");
diff --git a/drivers/net/ethernet/broadcom/genet/bcmgenet.h b/drivers/net/ethernet/broadcom/genet/bcmgenet.h
index a4933a5d3823..97c27b7920d5 100644
--- a/drivers/net/ethernet/broadcom/genet/bcmgenet.h
+++ b/drivers/net/ethernet/broadcom/genet/bcmgenet.h
@@ -581,6 +581,8 @@ struct bcmgenet_rx_ring {
unsigned int cb_ptr; /* Rx ring initial CB ptr */
unsigned int end_ptr; /* Rx ring end CB ptr */
unsigned int old_discards;
+ struct sk_buff *frag_head; /* frame being reassembled */
+ bool frag_drop; /* discarding until the next SOP */
struct bcmgenet_net_dim dim;
u32 rx_max_coalesced_frames;
u32 rx_coalesce_usecs;
--
2.53.0