Re: [REGRESSION][BISECTED] tun/tap & vhost-net: multi-threaded network performance

From: Simon Schippers

Date: Fri Jul 03 2026 - 06:49:41 EST


On 7/3/26 00:44, Michael S. Tsirkin wrote:
> Well, the issue was with host to guest right?
> Then testing what does bql do might be interesting.
> Might help.
> Something like this? Lightly tested.

Your approach calls netdev_tx_completed_queue() per individual packet
which is wrong and will cause a constant BQL limit of 2 as we have seen
in [1], causing a regression *100%*.
Citing the documentation of netdev_tx_completed_queue() in netdevice.h:

* Must be called at most once per TX completion round (and not per
* individual packet), so that BQL can adjust its limits appropriately.

[1] Link: https://lore.kernel.org/all/e8cdba04-aa9a-45c6-9807-8274b62920df@xxxxxxxxxxxxxx/

>
>
> diff --git a/drivers/net/tun.c b/drivers/net/tun.c
> index bfa49fa9e3a1..abc46354c107 100644
> --- a/drivers/net/tun.c
> +++ b/drivers/net/tun.c
> @@ -1076,6 +1076,7 @@ static netdev_tx_t tun_net_xmit(struct sk_buff *skb, struct net_device *dev)
> queue = netdev_get_tx_queue(dev, txq);
>
> spin_lock(&tfile->tx_ring.producer_lock);
> + netdev_tx_sent_queue(queue, len);
> ret = __ptr_ring_produce(&tfile->tx_ring, skb);
> if (!qdisc_txq_has_no_queue(queue) &&
> __ptr_ring_check_produce(&tfile->tx_ring) == -ENOSPC) {
> @@ -1088,6 +1089,7 @@ static netdev_tx_t tun_net_xmit(struct sk_buff *skb, struct net_device *dev)
> spin_unlock(&tfile->tx_ring.producer_lock);
>
> if (ret) {
> + netdev_tx_completed_queue(queue, 1, len);
> /* This should be a rare case if a qdisc is present, but
> * can happen due to lltx.
> * Since skb_tx_timestamp(), skb_orphan(),
> @@ -2148,15 +2150,19 @@ static ssize_t tun_put_user(struct tun_struct *tun,
>
> /* Callers must hold ring.consumer_lock */
> static void __tun_wake_queue(struct tun_struct *tun,
> - struct tun_file *tfile, int consumed)
> + struct tun_file *tfile,
> + unsigned int pkts, unsigned int bytes)
> {
> struct netdev_queue *txq = netdev_get_tx_queue(tun->dev,
> tfile->queue_index);
>
> + if (bytes)
> + netdev_tx_completed_queue(txq, pkts, bytes);
> +


Right here.



> /* Paired with smp_mb__after_atomic() in tun_net_xmit() */
> smp_mb();
> if (netif_tx_queue_stopped(txq)) {
> - tfile->cons_cnt += consumed;
> + tfile->cons_cnt += pkts;
> if (tfile->cons_cnt >= tfile->tx_ring.size / 2 ||
> __ptr_ring_empty(&tfile->tx_ring)) {
> netif_tx_wake_queue(txq);
> @@ -2167,12 +2173,16 @@ static void __tun_wake_queue(struct tun_struct *tun,
>
> static void *tun_ring_consume(struct tun_struct *tun, struct tun_file *tfile)
> {
> + unsigned int bytes = 0;
> void *ptr;
>
> spin_lock(&tfile->tx_ring.consumer_lock);
> ptr = __ptr_ring_consume(&tfile->tx_ring);
> - if (ptr)
> - __tun_wake_queue(tun, tfile, 1);
> + if (ptr) {
> + if (!tun_is_xdp_frame(ptr))
> + bytes = ((struct sk_buff *)ptr)->len;
> + __tun_wake_queue(tun, tfile, 1, bytes);
> + }
>
> spin_unlock(&tfile->tx_ring.consumer_lock);
> return ptr;
> @@ -3805,7 +3815,7 @@ struct ptr_ring *tun_get_tx_ring(struct file *file)
> EXPORT_SYMBOL_GPL(tun_get_tx_ring);
>
> /* Callers must hold ring.consumer_lock */
> -void tun_wake_queue(struct file *file, int consumed)
> +void tun_wake_queue(struct file *file, unsigned int pkts, unsigned int bytes)
> {
> struct tun_file *tfile;
> struct tun_struct *tun;
> @@ -3821,7 +3831,7 @@ void tun_wake_queue(struct file *file, int consumed)
>
> tun = rcu_dereference(tfile->tun);
> if (tun)
> - __tun_wake_queue(tun, tfile, consumed);
> + __tun_wake_queue(tun, tfile, pkts, bytes);
>
> rcu_read_unlock();
> }
> diff --git a/drivers/vhost/net.c b/drivers/vhost/net.c
> index db341c922673..5267b323bd59 100644
> --- a/drivers/vhost/net.c
> +++ b/drivers/vhost/net.c
> @@ -181,14 +181,23 @@ static int vhost_net_buf_produce(struct sock *sk,
> {
> struct file *file = sk->sk_socket->file;
> struct vhost_net_buf *rxq = &nvq->rxq;
> + unsigned int bytes = 0;
> + int i;
>
> rxq->head = 0;
> spin_lock(&nvq->rx_ring->consumer_lock);
> rxq->tail = __ptr_ring_consume_batched(nvq->rx_ring, rxq->queue,
> VHOST_NET_BATCH);
>
> - if (rxq->tail)
> - tun_wake_queue(file, rxq->tail);
> + if (rxq->tail) {
> + for (i = 0; i < rxq->tail; i++) {
> + void *ptr = rxq->queue[i];
> +
> + if (!tun_is_xdp_frame(ptr))
> + bytes += ((struct sk_buff *)ptr)->len;
> + }
> + tun_wake_queue(file, rxq->tail, bytes);
> + }
>
> spin_unlock(&nvq->rx_ring->consumer_lock);
> return rxq->tail;
> diff --git a/include/linux/if_tun.h b/include/linux/if_tun.h
> index 5f3e206c7a73..49b85bf4f828 100644
> --- a/include/linux/if_tun.h
> +++ b/include/linux/if_tun.h
> @@ -22,7 +22,7 @@ struct tun_msg_ctl {
> #if defined(CONFIG_TUN) || defined(CONFIG_TUN_MODULE)
> struct socket *tun_get_socket(struct file *);
> struct ptr_ring *tun_get_tx_ring(struct file *file);
> -void tun_wake_queue(struct file *file, int consumed);
> +void tun_wake_queue(struct file *file, unsigned int pkts, unsigned int bytes);
>
> static inline bool tun_is_xdp_frame(void *ptr)
> {
> @@ -56,7 +56,8 @@ static inline struct ptr_ring *tun_get_tx_ring(struct file *f)
> return ERR_PTR(-EINVAL);
> }
>
> -static inline void tun_wake_queue(struct file *f, int consumed) {}
> +static inline void tun_wake_queue(struct file *f,
> + unsigned int pkts, unsigned int bytes) {}
>
> static inline bool tun_is_xdp_frame(void *ptr)
> {
>