From: Jens Axboe <axboe@kernel.dk> mainline inclusion from mainline-v7.3-rc1 commit cd305ee3633a45fcf5f3a5d83f99f3cb77d87b6e category: bugfix bugzilla: https://atomgit.com/src-openeuler/kernel/issues/18742 CVE: CVE-2026-80920 Reference: https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git/commit/?id=... -------------------------------- io_req_local_work_add() signals the CQ ring eventfd inline when it is the one to push the first entry onto ->work_list. For DEFER_TASKRUN rings that add is frequently done from a waitqueue wakeup handler, where an arbitrary waitqueue lock is held. eventfd_signal_mask() only refuses to recurse when current->in_eventfd is set, but that bit is set by eventfd_signal_mask() itself. If the wake chain starts somewhere else, signal goes out inline and can feed back into epoll. Add IOU_F_TWQ_IN_WAKE, set it on the task_work add done from the three waitqueue callbacks, and use it to force io_eventfd_signal() down the existing call_rcu_hurry() deferral instead of signaling inline. Fixes: 21a091b970cd ("io_uring: signal registered eventfd to process deferred task work") Cc: stable@vger.kernel.org Link: https://lore.kernel.org/all/20260813133843.2933127-1-4ncienth@gmail.com/ Signed-off-by: Jens Axboe <axboe@kernel.dk> Conflicts: include/linux/io_uring/cmd.h io_uring/io_uring.c io_uring/poll.c [On 6.6 the IOU_F_TWQ flags live in include/linux/io_uring/cmd.h and io_eventfd_signal() lives in io_uring/io_uring.c, so add IOU_F_TWQ_IN_WAKE there and thread the defer flag through the poll wakeup paths; io_uring/{eventfd,futex,tw,waitid}.c do not exist in this tree, and io_req_local_work_add() is in io_uring.c.] Co-authored-by: BackportAgent@a:deepseek-v4.1-flash Signed-off-by: Hulk Robot <hulkrobot@huawei.com> Signed-off-by: Zhang Xiaoxu <zhangxiaoxu5@huawei.com> Signed-off-by: Yang Erkun <yangerkun@huawei.com> --- include/linux/io_uring/cmd.h | 8 ++++++++ io_uring/io_uring.c | 8 ++++---- io_uring/poll.c | 21 +++++++++++---------- 3 files changed, 23 insertions(+), 14 deletions(-) diff --git a/include/linux/io_uring/cmd.h b/include/linux/io_uring/cmd.h index cdcda6486534..58c05530442e 100644 --- a/include/linux/io_uring/cmd.h +++ b/include/linux/io_uring/cmd.h @@ -17,6 +17,14 @@ enum { * It's also ignored unless IORING_SETUP_DEFER_TASKRUN is set. */ IOU_F_TWQ_LAZY_WAKE = 1, + + /* + * Set when task_work is queued from a waitqueue wakeup handler, where + * an arbitrary provider waitqueue lock is held. Signaling the CQ ring + * eventfd inline from there can recurse back into that lock through + * epoll, so the eventfd signal must be deferred. + */ + IOU_F_TWQ_IN_WAKE = 2, }; enum io_uring_cmd_flags { diff --git a/io_uring/io_uring.c b/io_uring/io_uring.c index d0431aa9b754..63dcd425a645 100644 --- a/io_uring/io_uring.c +++ b/io_uring/io_uring.c @@ -545,7 +545,7 @@ static void io_eventfd_ops(struct rcu_head *rcu) call_rcu(&ev_fd->rcu, io_eventfd_free); } -static void io_eventfd_signal(struct io_ring_ctx *ctx) +static void io_eventfd_signal(struct io_ring_ctx *ctx, bool defer) { struct io_ev_fd *ev_fd = NULL; @@ -568,7 +568,7 @@ static void io_eventfd_signal(struct io_ring_ctx *ctx) if (ev_fd->eventfd_async && !io_wq_current_is_worker()) goto out; - if (likely(eventfd_signal_allowed())) { + if (!defer && likely(eventfd_signal_allowed())) { eventfd_signal_mask(ev_fd->cq_ev_fd, 1, EPOLL_URING_WAKE); } else { atomic_inc(&ev_fd->refs); @@ -602,7 +602,7 @@ static void io_eventfd_flush_signal(struct io_ring_ctx *ctx) if (skip) return; - io_eventfd_signal(ctx); + io_eventfd_signal(ctx, false); } void __io_commit_cqring_flush(struct io_ring_ctx *ctx) @@ -1242,7 +1242,7 @@ static inline void io_req_local_work_add(struct io_kiocb *req, unsigned flags) if (ctx->flags & IORING_SETUP_TASKRUN_FLAG) atomic_or(IORING_SQ_TASKRUN, &ctx->rings->sq_flags); if (ctx->has_evfd) - io_eventfd_signal(ctx); + io_eventfd_signal(ctx, flags & IOU_F_TWQ_IN_WAKE); } nr_wait = atomic_read(&ctx->cq_wait_nr); diff --git a/io_uring/poll.c b/io_uring/poll.c index 704321fbceea..9071122ef5fd 100644 --- a/io_uring/poll.c +++ b/io_uring/poll.c @@ -229,19 +229,20 @@ enum { IOU_POLL_REQUEUE = 4, }; -static void __io_poll_execute(struct io_kiocb *req, int mask) +static void __io_poll_execute(struct io_kiocb *req, int mask, unsigned tw_flags) { io_req_set_res(req, mask, 0); req->io_task_work.func = io_poll_task_func; trace_io_uring_task_add(req, mask); - io_req_task_work_add(req); + __io_req_task_work_add(req, tw_flags); } -static inline void io_poll_execute(struct io_kiocb *req, int res) +static inline void io_poll_execute(struct io_kiocb *req, int res, + unsigned tw_flags) { if (io_poll_get_ownership(req)) - __io_poll_execute(req, res); + __io_poll_execute(req, res, tw_flags); } /* @@ -358,7 +359,7 @@ void io_poll_task_func(struct io_kiocb *req, struct io_tw_state *ts) return; } else if (ret == IOU_POLL_REQUEUE) { io_kbuf_recycle(req, 0); - __io_poll_execute(req, 0); + __io_poll_execute(req, 0, 0); return; } io_poll_remove_entries(req); @@ -396,7 +397,7 @@ static void io_poll_cancel_req(struct io_kiocb *req) { io_poll_mark_cancelled(req); /* kick tw, which should complete the request */ - io_poll_execute(req, 0); + io_poll_execute(req, 0, 0); } #define IO_ASYNC_POLL_COMMON (EPOLLONESHOT | EPOLLPRI) @@ -405,7 +406,7 @@ static __cold int io_pollfree_wake(struct io_kiocb *req, struct io_poll *poll) { io_poll_mark_cancelled(req); /* we have to kick tw in case it's not already */ - io_poll_execute(req, 0); + io_poll_execute(req, 0, IOU_F_TWQ_IN_WAKE); /* * If the waitqueue is being freed early but someone is already @@ -460,7 +461,7 @@ static int io_poll_wake(struct wait_queue_entry *wait, unsigned mode, int sync, else req->flags &= ~REQ_F_SINGLE_POLL; } - __io_poll_execute(req, mask); + __io_poll_execute(req, mask, IOU_F_TWQ_IN_WAKE); } return 1; } @@ -643,7 +644,7 @@ static int __io_arm_poll_handler(struct io_kiocb *req, if (mask && (poll->events & EPOLLET) && io_poll_can_finish_inline(req, ipt)) { - __io_poll_execute(req, mask); + __io_poll_execute(req, mask, 0); return 0; } @@ -653,7 +654,7 @@ static int __io_arm_poll_handler(struct io_kiocb *req, * poll was waken up, queue up a tw, it'll deal with it. */ if (atomic_cmpxchg(&req->poll_refs, 1, 0) != 1) - __io_poll_execute(req, 0); + __io_poll_execute(req, 0, 0); } return 0; } -- 2.52.0