mirror of
https://github.com/torvalds/linux.git
synced 2026-07-27 09:36:22 +02:00
Merge branch 'tun-tap-vhost-net-apply-qdisc-backpressure-on-full-ptr_ring-to-reduce-tx-drops'
Simon Schippers says: ==================== tun/tap & vhost-net: apply qdisc backpressure on full ptr_ring to reduce TX drops This patch series deals with tun/tap & vhost-net which drop incoming SKBs whenever their internal ptr_ring buffer is full. Instead, with this patch series, the associated netdev queue is stopped - but only when a qdisc is attached. If no qdisc is present the existing behavior is preserved. The XDP transmit path is not affected. This patch series touches tun/tap and vhost-net, as they share common logic and must be updated together. Modifying only one of them would break the other. By applying proper backpressure, this change allows the connected qdisc to operate correctly, as reported in [1], and significantly improves performance in real-world scenarios, as demonstrated in our paper [2]. For example, we observed a 36% TCP throughput improvement for an OpenVPN connection between Germany and the USA. Synthetic pktgen benchmarks indicate a slight regression, and packet loss is reduced to near zero. Pktgen benchmarks are provided per commit, with the final commit showing the overall performance. [1] Link: https://unix.stackexchange.com/questions/762935/traffic-shaping-ineffective-on-tun-device [2] Link: https://cni.etit.tu-dortmund.de/storages/cni-etit/r/Research/Publications/2025/Gebauer_2025_VTCFall/Gebauer_VTCFall2025_AuthorsVersion.pdf ==================== Link: https://patch.msgid.link/20260510151529.43895-1-simon.schippers@tu-dortmund.de Signed-off-by: Jakub Kicinski <kuba@kernel.org>
This commit is contained in:
commit
3803065cd6
|
|
@ -145,6 +145,8 @@ struct tun_file {
|
|||
struct list_head next;
|
||||
struct tun_struct *detached;
|
||||
struct ptr_ring tx_ring;
|
||||
/* Protected by tx_ring.consumer_lock */
|
||||
int cons_cnt;
|
||||
struct xdp_rxq_info xdp_rxq;
|
||||
};
|
||||
|
||||
|
|
@ -588,8 +590,13 @@ static void __tun_detach(struct tun_file *tfile, bool clean)
|
|||
rcu_assign_pointer(tun->tfiles[index],
|
||||
tun->tfiles[tun->numqueues - 1]);
|
||||
ntfile = rtnl_dereference(tun->tfiles[index]);
|
||||
spin_lock(&ntfile->tx_ring.consumer_lock);
|
||||
ntfile->queue_index = index;
|
||||
ntfile->xdp_rxq.queue_index = index;
|
||||
ntfile->cons_cnt = 0;
|
||||
if (__ptr_ring_empty(&ntfile->tx_ring))
|
||||
netif_wake_subqueue(tun->dev, index);
|
||||
spin_unlock(&ntfile->tx_ring.consumer_lock);
|
||||
rcu_assign_pointer(tun->tfiles[tun->numqueues - 1],
|
||||
NULL);
|
||||
|
||||
|
|
@ -730,6 +737,9 @@ static int tun_attach(struct tun_struct *tun, struct file *file,
|
|||
goto out;
|
||||
}
|
||||
|
||||
spin_lock(&tfile->tx_ring.consumer_lock);
|
||||
tfile->cons_cnt = 0;
|
||||
spin_unlock(&tfile->tx_ring.consumer_lock);
|
||||
tfile->queue_index = tun->numqueues;
|
||||
tfile->socket.sk->sk_shutdown &= ~RCV_SHUTDOWN;
|
||||
|
||||
|
|
@ -1008,6 +1018,7 @@ static netdev_tx_t tun_net_xmit(struct sk_buff *skb, struct net_device *dev)
|
|||
struct netdev_queue *queue;
|
||||
struct tun_file *tfile;
|
||||
int len = skb->len;
|
||||
int ret;
|
||||
|
||||
rcu_read_lock();
|
||||
tfile = rcu_dereference(tun->tfiles[txq]);
|
||||
|
|
@ -1062,13 +1073,33 @@ static netdev_tx_t tun_net_xmit(struct sk_buff *skb, struct net_device *dev)
|
|||
|
||||
nf_reset_ct(skb);
|
||||
|
||||
if (ptr_ring_produce(&tfile->tx_ring, skb)) {
|
||||
queue = netdev_get_tx_queue(dev, txq);
|
||||
|
||||
spin_lock(&tfile->tx_ring.producer_lock);
|
||||
ret = __ptr_ring_produce(&tfile->tx_ring, skb);
|
||||
if (!qdisc_txq_has_no_queue(queue) &&
|
||||
__ptr_ring_check_produce(&tfile->tx_ring) == -ENOSPC) {
|
||||
netif_tx_stop_queue(queue);
|
||||
/* Paired with smp_mb() in __tun_wake_queue() */
|
||||
smp_mb__after_atomic();
|
||||
if (!__ptr_ring_check_produce(&tfile->tx_ring))
|
||||
netif_tx_wake_queue(queue);
|
||||
}
|
||||
spin_unlock(&tfile->tx_ring.producer_lock);
|
||||
|
||||
if (ret) {
|
||||
/* This should be a rare case if a qdisc is present, but
|
||||
* can happen due to lltx.
|
||||
* Since skb_tx_timestamp(), skb_orphan(),
|
||||
* run_ebpf_filter() and pskb_trim() could have tinkered
|
||||
* with the SKB, returning NETDEV_TX_BUSY is unsafe and
|
||||
* we must drop instead.
|
||||
*/
|
||||
drop_reason = SKB_DROP_REASON_FULL_RING;
|
||||
goto drop;
|
||||
}
|
||||
|
||||
/* dev->lltx requires to do our own update of trans_start */
|
||||
queue = netdev_get_tx_queue(dev, txq);
|
||||
txq_trans_cond_update(queue);
|
||||
|
||||
/* Notify and wake up reader process */
|
||||
|
|
@ -2115,13 +2146,46 @@ static ssize_t tun_put_user(struct tun_struct *tun,
|
|||
return total;
|
||||
}
|
||||
|
||||
static void *tun_ring_recv(struct tun_file *tfile, int noblock, int *err)
|
||||
/* Callers must hold ring.consumer_lock */
|
||||
static void __tun_wake_queue(struct tun_struct *tun,
|
||||
struct tun_file *tfile, int consumed)
|
||||
{
|
||||
struct netdev_queue *txq = netdev_get_tx_queue(tun->dev,
|
||||
tfile->queue_index);
|
||||
|
||||
/* Paired with smp_mb__after_atomic() in tun_net_xmit() */
|
||||
smp_mb();
|
||||
if (netif_tx_queue_stopped(txq)) {
|
||||
tfile->cons_cnt += consumed;
|
||||
if (tfile->cons_cnt >= tfile->tx_ring.size / 2 ||
|
||||
__ptr_ring_empty(&tfile->tx_ring)) {
|
||||
netif_tx_wake_queue(txq);
|
||||
tfile->cons_cnt = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void *tun_ring_consume(struct tun_struct *tun, struct tun_file *tfile)
|
||||
{
|
||||
void *ptr;
|
||||
|
||||
spin_lock(&tfile->tx_ring.consumer_lock);
|
||||
ptr = __ptr_ring_consume(&tfile->tx_ring);
|
||||
if (ptr)
|
||||
__tun_wake_queue(tun, tfile, 1);
|
||||
|
||||
spin_unlock(&tfile->tx_ring.consumer_lock);
|
||||
return ptr;
|
||||
}
|
||||
|
||||
static void *tun_ring_recv(struct tun_struct *tun, struct tun_file *tfile,
|
||||
int noblock, int *err)
|
||||
{
|
||||
DECLARE_WAITQUEUE(wait, current);
|
||||
void *ptr = NULL;
|
||||
int error = 0;
|
||||
|
||||
ptr = ptr_ring_consume(&tfile->tx_ring);
|
||||
ptr = tun_ring_consume(tun, tfile);
|
||||
if (ptr)
|
||||
goto out;
|
||||
if (noblock) {
|
||||
|
|
@ -2133,7 +2197,7 @@ static void *tun_ring_recv(struct tun_file *tfile, int noblock, int *err)
|
|||
|
||||
while (1) {
|
||||
set_current_state(TASK_INTERRUPTIBLE);
|
||||
ptr = ptr_ring_consume(&tfile->tx_ring);
|
||||
ptr = tun_ring_consume(tun, tfile);
|
||||
if (ptr)
|
||||
break;
|
||||
if (signal_pending(current)) {
|
||||
|
|
@ -2170,7 +2234,7 @@ static ssize_t tun_do_read(struct tun_struct *tun, struct tun_file *tfile,
|
|||
|
||||
if (!ptr) {
|
||||
/* Read frames from ring */
|
||||
ptr = tun_ring_recv(tfile, noblock, &err);
|
||||
ptr = tun_ring_recv(tun, tfile, noblock, &err);
|
||||
if (!ptr)
|
||||
return err;
|
||||
}
|
||||
|
|
@ -3622,6 +3686,16 @@ static int tun_queue_resize(struct tun_struct *tun)
|
|||
dev->tx_queue_len, GFP_KERNEL,
|
||||
tun_ptr_free);
|
||||
|
||||
if (!ret) {
|
||||
for (i = 0; i < tun->numqueues; i++) {
|
||||
tfile = rtnl_dereference(tun->tfiles[i]);
|
||||
spin_lock(&tfile->tx_ring.consumer_lock);
|
||||
netif_wake_subqueue(tun->dev, tfile->queue_index);
|
||||
tfile->cons_cnt = 0;
|
||||
spin_unlock(&tfile->tx_ring.consumer_lock);
|
||||
}
|
||||
}
|
||||
|
||||
kfree(rings);
|
||||
return ret;
|
||||
}
|
||||
|
|
@ -3730,6 +3804,29 @@ struct ptr_ring *tun_get_tx_ring(struct file *file)
|
|||
}
|
||||
EXPORT_SYMBOL_GPL(tun_get_tx_ring);
|
||||
|
||||
/* Callers must hold ring.consumer_lock */
|
||||
void tun_wake_queue(struct file *file, int consumed)
|
||||
{
|
||||
struct tun_file *tfile;
|
||||
struct tun_struct *tun;
|
||||
|
||||
if (file->f_op != &tun_fops)
|
||||
return;
|
||||
|
||||
tfile = file->private_data;
|
||||
if (!tfile)
|
||||
return;
|
||||
|
||||
rcu_read_lock();
|
||||
|
||||
tun = rcu_dereference(tfile->tun);
|
||||
if (tun)
|
||||
__tun_wake_queue(tun, tfile, consumed);
|
||||
|
||||
rcu_read_unlock();
|
||||
}
|
||||
EXPORT_SYMBOL_GPL(tun_wake_queue);
|
||||
|
||||
module_init(tun_init);
|
||||
module_exit(tun_cleanup);
|
||||
MODULE_DESCRIPTION(DRV_DESCRIPTION);
|
||||
|
|
|
|||
|
|
@ -176,13 +176,21 @@ static void *vhost_net_buf_consume(struct vhost_net_buf *rxq)
|
|||
return ret;
|
||||
}
|
||||
|
||||
static int vhost_net_buf_produce(struct vhost_net_virtqueue *nvq)
|
||||
static int vhost_net_buf_produce(struct sock *sk,
|
||||
struct vhost_net_virtqueue *nvq)
|
||||
{
|
||||
struct file *file = sk->sk_socket->file;
|
||||
struct vhost_net_buf *rxq = &nvq->rxq;
|
||||
|
||||
rxq->head = 0;
|
||||
rxq->tail = ptr_ring_consume_batched(nvq->rx_ring, rxq->queue,
|
||||
VHOST_NET_BATCH);
|
||||
spin_lock(&nvq->rx_ring->consumer_lock);
|
||||
rxq->tail = __ptr_ring_consume_batched(nvq->rx_ring, rxq->queue,
|
||||
VHOST_NET_BATCH);
|
||||
|
||||
if (rxq->tail)
|
||||
tun_wake_queue(file, rxq->tail);
|
||||
|
||||
spin_unlock(&nvq->rx_ring->consumer_lock);
|
||||
return rxq->tail;
|
||||
}
|
||||
|
||||
|
|
@ -209,14 +217,15 @@ static int vhost_net_buf_peek_len(void *ptr)
|
|||
return __skb_array_len_with_tag(ptr);
|
||||
}
|
||||
|
||||
static int vhost_net_buf_peek(struct vhost_net_virtqueue *nvq)
|
||||
static int vhost_net_buf_peek(struct sock *sk,
|
||||
struct vhost_net_virtqueue *nvq)
|
||||
{
|
||||
struct vhost_net_buf *rxq = &nvq->rxq;
|
||||
|
||||
if (!vhost_net_buf_is_empty(rxq))
|
||||
goto out;
|
||||
|
||||
if (!vhost_net_buf_produce(nvq))
|
||||
if (!vhost_net_buf_produce(sk, nvq))
|
||||
return 0;
|
||||
|
||||
out:
|
||||
|
|
@ -995,7 +1004,7 @@ static int peek_head_len(struct vhost_net_virtqueue *rvq, struct sock *sk)
|
|||
unsigned long flags;
|
||||
|
||||
if (rvq->rx_ring)
|
||||
return vhost_net_buf_peek(rvq);
|
||||
return vhost_net_buf_peek(sk, rvq);
|
||||
|
||||
spin_lock_irqsave(&sk->sk_receive_queue.lock, flags);
|
||||
head = skb_peek(&sk->sk_receive_queue);
|
||||
|
|
|
|||
|
|
@ -22,6 +22,7 @@ struct tun_msg_ctl {
|
|||
#if defined(CONFIG_TUN) || defined(CONFIG_TUN_MODULE)
|
||||
struct socket *tun_get_socket(struct file *);
|
||||
struct ptr_ring *tun_get_tx_ring(struct file *file);
|
||||
void tun_wake_queue(struct file *file, int consumed);
|
||||
|
||||
static inline bool tun_is_xdp_frame(void *ptr)
|
||||
{
|
||||
|
|
@ -55,6 +56,8 @@ static inline struct ptr_ring *tun_get_tx_ring(struct file *f)
|
|||
return ERR_PTR(-EINVAL);
|
||||
}
|
||||
|
||||
static inline void tun_wake_queue(struct file *f, int consumed) {}
|
||||
|
||||
static inline bool tun_is_xdp_frame(void *ptr)
|
||||
{
|
||||
return false;
|
||||
|
|
|
|||
|
|
@ -96,6 +96,20 @@ static inline bool ptr_ring_full_bh(struct ptr_ring *r)
|
|||
return ret;
|
||||
}
|
||||
|
||||
/* Note: callers invoking this in a loop must use a compiler barrier,
|
||||
* for example cpu_relax(). Callers must hold producer_lock.
|
||||
*/
|
||||
static inline int __ptr_ring_check_produce(struct ptr_ring *r)
|
||||
{
|
||||
if (unlikely(!r->size))
|
||||
return -EINVAL;
|
||||
|
||||
if (data_race(r->queue[r->producer]))
|
||||
return -ENOSPC;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Note: callers invoking this in a loop must use a compiler barrier,
|
||||
* for example cpu_relax(). Callers must hold producer_lock.
|
||||
* Callers are responsible for making sure pointer that is being queued
|
||||
|
|
@ -103,8 +117,10 @@ static inline bool ptr_ring_full_bh(struct ptr_ring *r)
|
|||
*/
|
||||
static inline int __ptr_ring_produce(struct ptr_ring *r, void *ptr)
|
||||
{
|
||||
if (unlikely(!r->size) || data_race(r->queue[r->producer]))
|
||||
return -ENOSPC;
|
||||
int p = __ptr_ring_check_produce(r);
|
||||
|
||||
if (p)
|
||||
return p;
|
||||
|
||||
/* Make sure the pointer we are storing points to a valid data. */
|
||||
/* Pairs with the dependency ordering in __ptr_ring_consume. */
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user