io_uring: get rid of tw_pending for !DEFER task work

The normal task_work path used a tw_pending bit to ensure the callback
was only added once: the mpscq drains incrementally, so a single
tctx_task_work() run can take the queue through empty -> non-empty
several times, and each transition would otherwise re-add the already
pending callback_head. This corrupts the task_work list, and is what
tw_pending protects again.

This can go away, if we stop running the task_work as soon as the queue
empties.

Suggested-by: Caleb Sander Mateos <csander@purestorage.com>
Reviewed-by: Caleb Sander Mateos <csander@purestorage.com>
Signed-off-by: Jens Axboe <axboe@kernel.dk>
This commit is contained in:
Jens Axboe 2026-06-15 13:43:16 -06:00
parent c554246ff4
commit ca4aa97194
3 changed files with 17 additions and 13 deletions

View File

@ -149,8 +149,6 @@ struct io_uring_task {
struct { /* task_work */
struct mpscq task_list;
/* BIT(0) guards adding tw only once */
unsigned long tw_pending;
struct callback_head task_work;
} ____cacheline_aligned_in_smp;
};

View File

@ -122,4 +122,13 @@ static inline struct llist_node *mpscq_pop(struct mpscq *q,
return NULL;
}
/*
* Returns true if the most recent mpscq_pop() that returned a node also
* emptied the queue. Consumer must be serialized.
*/
static inline bool mpscq_pop_emptied(struct mpscq *q, struct llist_node *head)
{
return head == &q->stub;
}
#endif /* IOU_MPSCQ_H */

View File

@ -34,10 +34,6 @@ void io_tctx_fallback_work(struct work_struct *work)
fallback_work);
unsigned int count = 0;
/* see tctx_task_work() - a set bit must always have a run coming */
clear_bit(0, &tctx->tw_pending);
smp_mb__after_atomic();
/*
* Run the entries directly. We're in PF_KTHRED context, hence
* io_should_terminate_tw() is true and they will be marked as
@ -101,6 +97,13 @@ void tctx_task_work_run(struct io_uring_task *tctx, unsigned int max_entries,
io_poll_task_func, io_req_rw_complete,
(struct io_tw_req){req}, ts);
(*count)++;
/*
* Break if most recent pop emptied the queue. This helps
* bound task_work run, and also protects the regular
* task_work addition.
*/
if (mpscq_pop_emptied(&tctx->task_list, tctx->task_head))
break;
if (unlikely(need_resched())) {
ctx_flush_and_put(ctx, ts);
ctx = NULL;
@ -127,8 +130,6 @@ void tctx_task_work(struct callback_head *cb)
unsigned int count = 0;
tctx = container_of(cb, struct io_uring_task, task_work);
clear_bit(0, &tctx->tw_pending);
smp_mb__after_atomic();
tctx_task_work_run(tctx, UINT_MAX, &count);
}
@ -206,7 +207,7 @@ void io_req_normal_work_add(struct io_kiocb *req)
struct io_uring_task *tctx = req->tctx;
struct io_ring_ctx *ctx = req->ctx;
/* task_work already pending, we're done */
/* tw run already pending, nothing else to do */
if (!mpscq_push(&tctx->task_list, &req->io_task_work.node))
return;
@ -223,10 +224,6 @@ void io_req_normal_work_add(struct io_kiocb *req)
return;
}
/* task_work must only be added once */
if (test_and_set_bit(0, &tctx->tw_pending))
return;
if (likely(!task_work_add(tctx->task, &tctx->task_work, ctx->notify_method)))
return;