eventpoll: extract ep_deliver_event() from ep_send_events()

ep_send_events()'s body covered two concerns: per-item work (PM
wakeup-source bookkeeping, re-poll, copy_to_user, level-trigger
re-queue, EPOLLONESHOT mask clear) and the scan-level accumulator
(maxevents cap, EFAULT preservation, txlist/rdllist splice).

Extract the per-item work as ep_deliver_event(), which returns a
tri-state int:

  1       one event was delivered; caller advances the counter,
  0       re-poll produced no caller-requested events (item drops
          out of the ready list; a future callback will re-queue),
 -EFAULT  copy_to_user() faulted; item is already re-inserted at
          the head of the txlist so ep_done_scan() splices it back
          to rdllist.

The per-item comments (PM ordering, the "sole writer to rdllist"
invariant for the LT re-queue, the EFAULT semantics) move into
ep_deliver_event(). ep_send_events() reduces to the fatal-signal
short-circuit, scan bracket, and a short txlist walk that accumulates
the deliveries and preserves the "first error wins" EFAULT contract
(res = delivered only if no event was previously delivered; otherwise
the success count is returned and -EFAULT is reported on the next
call).

No functional change.

Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
Link: https://patch.msgid.link/20260424-work-epoll-rework-v1-12-249ed00a20f3@kernel.org
Signed-off-by: Christian Brauner <brauner@kernel.org>
This commit is contained in:
Christian Brauner 2026-04-24 15:46:43 +02:00
parent 08fe3679a8
commit 499a5e7f4a
No known key found for this signature in database
GPG Key ID: 91C61BC06578DCA2

View File

@ -1979,6 +1979,82 @@ static int ep_modify(struct eventpoll *ep, struct epitem *epi,
return 0;
}
/*
* Attempt to deliver one event for @epi into @*uevents.
*
* Returns 1 if an event was delivered (with *uevents advanced to the
* next slot), 0 if the re-poll reported no caller-requested events
* (@epi drops out of the ready list; a future callback will re-add
* it), or -EFAULT if copy_to_user() faulted (in which case @epi is
* re-inserted at the head of @txlist so ep_done_scan() merges it
* back to rdllist for the next attempt).
*
* PM bookkeeping and level-triggered re-queue are handled here.
* Caller holds ep->mtx and the scan is active.
*/
static int ep_deliver_event(struct eventpoll *ep, struct epitem *epi,
poll_table *pt,
struct epoll_event __user **uevents,
struct list_head *txlist)
{
struct epoll_event __user *next;
struct wakeup_source *ws;
__poll_t revents;
/*
* Activate ep->ws before deactivating epi->ws to prevent
* triggering auto-suspend here (in case we reactivate epi->ws
* below). Rearranging to delay the deactivation would let
* epi->ws drift out of sync with ep_is_linked().
*/
ws = ep_wakeup_source(epi);
if (ws) {
if (ws->active)
__pm_stay_awake(ep->ws);
__pm_relax(ws);
}
list_del_init(&epi->rdllink);
/*
* Re-poll under ep->mtx so userspace cannot change the item
* out from under us. If no caller-requested events remain,
* @epi stays off the ready list; the poll callback will
* re-queue it when events next appear.
*/
revents = ep_item_poll(epi, pt, 1);
if (!revents)
return 0;
next = epoll_put_uevent(revents, epi->event.data, *uevents);
if (!next) {
/*
* copy_to_user() faulted: put the item back so
* ep_done_scan() splices it onto rdllist for the next
* attempt.
*/
list_add(&epi->rdllink, txlist);
ep_pm_stay_awake(epi);
return -EFAULT;
}
*uevents = next;
if (epi->event.events & EPOLLONESHOT) {
epi->event.events &= EP_PRIVATE_BITS;
} else if (!(epi->event.events & EPOLLET)) {
/*
* Level-triggered: re-queue so the next epoll_wait()
* rechecks availability. We are the sole writer to
* rdllist here -- epoll_ctl() callers are locked out
* by ep->mtx, and the poll callback queues to ovflist
* during scans.
*/
list_add_tail(&epi->rdllink, &ep->rdllist);
ep_pm_stay_awake(epi);
}
return 1;
}
static int ep_send_events(struct eventpoll *ep,
struct epoll_event __user *events, int maxevents)
{
@ -2001,70 +2077,24 @@ static int ep_send_events(struct eventpoll *ep,
ep_start_scan(ep, &txlist);
/*
* We can loop without lock because we are passed a task private list.
* Items cannot vanish during the loop we are holding ep->mtx.
* We can loop without lock because we are passed a task-private
* txlist; items cannot vanish while we hold ep->mtx.
*/
list_for_each_entry_safe(epi, tmp, &txlist, rdllink) {
struct wakeup_source *ws;
__poll_t revents;
int delivered;
if (res >= maxevents)
break;
/*
* Activate ep->ws before deactivating epi->ws to prevent
* triggering auto-suspend here (in case we reactive epi->ws
* below).
*
* This could be rearranged to delay the deactivation of epi->ws
* instead, but then epi->ws would temporarily be out of sync
* with ep_is_linked().
*/
ws = ep_wakeup_source(epi);
if (ws) {
if (ws->active)
__pm_stay_awake(ep->ws);
__pm_relax(ws);
}
list_del_init(&epi->rdllink);
/*
* If the event mask intersect the caller-requested one,
* deliver the event to userspace. Again, we are holding ep->mtx,
* so no operations coming from userspace can change the item.
*/
revents = ep_item_poll(epi, &pt, 1);
if (!revents)
continue;
events = epoll_put_uevent(revents, epi->event.data, events);
if (!events) {
list_add(&epi->rdllink, &txlist);
ep_pm_stay_awake(epi);
delivered = ep_deliver_event(ep, epi, &pt, &events, &txlist);
if (delivered < 0) {
if (!res)
res = -EFAULT;
res = delivered;
break;
}
res++;
if (epi->event.events & EPOLLONESHOT)
epi->event.events &= EP_PRIVATE_BITS;
else if (!(epi->event.events & EPOLLET)) {
/*
* If this file has been added with Level
* Trigger mode, we need to insert back inside
* the ready list, so that the next call to
* epoll_wait() will check again the events
* availability. At this point, no one can insert
* into ep->rdllist besides us. The epoll_ctl()
* callers are locked out by
* ep_send_events() holding "mtx" and the
* poll callback will queue them in ep->ovflist.
*/
list_add_tail(&epi->rdllink, &ep->rdllist);
ep_pm_stay_awake(epi);
}
res += delivered;
}
ep_done_scan(ep, &txlist);
mutex_unlock(&ep->mtx);