From 56582c9ccc36003c41e558796ae2bb8ea1093e10 Mon Sep 17 00:00:00 2001 From: Jens Axboe Date: Tue, 8 Sep 2026 17:03:12 +0000 Subject: io_uring: wait for in-flight requests on ring release With cancelations now run at release time, what's left in-flight on the ring afterwards is mostly I/O that has already been issued to a device and just needs to finish. Until that happens, the files from those requests pin the files they were using. Wait for those. Signed-off-by: Jens Axboe --- io_uring/io_uring.c | 74 ++++++++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 73 insertions(+), 1 deletion(-) diff --git a/io_uring/io_uring.c b/io_uring/io_uring.c index 428f88c968c3..a67b2adeda36 100644 --- a/io_uring/io_uring.c +++ b/io_uring/io_uring.c @@ -2388,6 +2388,76 @@ static __cold void io_ring_ctx_cancel(struct io_ring_ctx *ctx) io_req_caches_free(ctx); } +/* Number of requests that should be waited for */ +static __cold unsigned int io_ring_ctx_inflight(struct io_ring_ctx *ctx) +{ + guard(mutex)(&ctx->uring_lock); + __io_req_caches_free(ctx); + return ctx->nr_req_allocated - ctx->nr_notifs; +} + +/* + * Run task_work completions for current. Only do so if the io_uring callback + * itself can get pruned first, otherwise we risk recursing. + */ +static __cold bool io_ring_run_own_completions(struct io_uring_task *tctx) +{ + unsigned int count = 0; + + if (!tctx || mpscq_empty(&tctx->task_list)) + return true; + if (!task_work_cancel(current, &tctx->task_work)) + return false; + tctx_task_work_run(tctx, UINT_MAX, &count); + return true; +} + +/* + * Requests may remain after cancelations have been run, as not all requests + * are cancelable. Storage I/O is an example. Wait for those so that once + * close(2) returns, files pinned by these requests have been released. + */ +static __cold void io_ring_ctx_wait_inflight(struct io_ring_ctx *ctx) +{ + struct io_uring_task *tctx = current->io_uring; + bool ran_own = true; + + if (current->flags & (PF_KTHREAD | PF_EXITING)) + return; + if (tctx && atomic_read(&tctx->in_cancel)) + return; + + while (io_ring_ctx_inflight(ctx) && !fatal_signal_pending(current)) { + unsigned int state; + + if (test_thread_flag(TIF_NOTIFY_SIGNAL)) { + clear_notify_signal(); + if (task_work_pending(current)) + set_notify_resume(current); + } + state = TASK_INTERRUPTIBLE; + if (signal_pending(current)) + state = TASK_KILLABLE; + set_current_state(state | TASK_FREEZABLE); + /* don't sleep on work that's already there and that we can run */ + if (ran_own && ((tctx && !mpscq_empty(&tctx->task_list)) || + io_local_work_pending(ctx))) + __set_current_state(TASK_RUNNING); + else + schedule_timeout(1); + + /* completions may be queued behind us */ + if (!io_ring_run_own_completions(tctx)) { + if (!ran_own) + break; + ran_own = false; + } else { + ran_own = true; + } + io_ring_ctx_cancel(ctx); + } +} + static __cold void io_ring_exit_work(struct work_struct *work) { struct io_ring_ctx *ctx = container_of(work, struct io_ring_ctx, exit_work); @@ -2478,8 +2548,10 @@ static __cold void io_ring_ctx_wait_and_kill(struct io_ring_ctx *ctx) * out, and for requests owned by the task closing the ring, this * ensures any held files are put before close(2) returns. */ - if (!(current->flags & PF_IO_WORKER)) + if (!(current->flags & PF_IO_WORKER)) { io_ring_ctx_cancel(ctx); + io_ring_ctx_wait_inflight(ctx); + } INIT_WORK(&ctx->exit_work, io_ring_exit_work); /* -- cgit v1.2.3