mirror of
https://github.com/linux-msm/laptops-kernel.git
synced 2026-08-13 14:19:53 -07:00
Merge tag 'for-7.0/io_uring-20260206' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux
Pull io_uring updates from Jens Axboe: - Clean up the IORING_SETUP_R_DISABLED and submitter task checking, mostly just in preparation for relaxing the locking for SINGLE_ISSUER in the future. - Improve IOPOLL by using a doubly linked list to manage completions. Previously it was singly listed, which meant that to complete request N in the chain 0..N-1 had to have completed first. With a doubly linked list we can complete whatever request completes in that order, rather than need to wait for a consecutive range to be available. This reduces latencies. - Improve the restriction setup and checking. Mostly in preparation for adding further features on top of that. Coming in a separate pull request. - Split out task_work and wait handling into separate files. These are mostly nicely abstracted already, but still remained in the io_uring.c file which is on the larger side. - Use GFP_KERNEL_ACCOUNT in a few more spots, where appropriate. - Ensure even the idle io-wq worker exits if a task no longer has any rings open. - Add support for a non-circular submission queue. By default, the SQ ring keeps moving around, even if only a few entries are used for each submission. This can be wasteful in terms of cachelines. If IORING_SETUP_SQ_REWIND is set for the ring when created, each submission will start at offset 0 instead of where we last left off doing submissions. - Various little cleanups * tag 'for-7.0/io_uring-20260206' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux: (30 commits) io_uring/kbuf: fix memory leak if io_buffer_add_list fails io_uring: Add SPDX id lines to remaining source files io_uring: allow io-wq workers to exit when unused io_uring/io-wq: add exit-on-idle state io_uring/net: don't continue send bundle if poll was required for retry io_uring/rsrc: use GFP_KERNEL_ACCOUNT consistently io_uring/futex: use GFP_KERNEL_ACCOUNT for futex data allocation io_uring/io-wq: handle !sysctl_hung_task_timeout_secs io_uring: fix bad indentation for setup flags if statement io_uring/rsrc: take unsigned index in io_rsrc_node_lookup() io_uring: introduce non-circular SQ io_uring: split out CQ waiting code into wait.c io_uring: split out task work code into tw.c io_uring/io-wq: don't trigger hung task for syzbot craziness io_uring: add IO_URING_EXIT_WAIT_MAX definition io_uring/sync: validate passed in offset io_uring/eventfd: remove unused ctx->evfd_last_cq_tail member io_uring/timeout: annotate data race in io_flush_timeouts() io_uring/uring_cmd: explicitly disallow cancelations for IOPOLL io_uring: fix IOPOLL with passthrough I/O ...
This commit is contained in:
@@ -224,7 +224,10 @@ struct io_restriction {
|
||||
DECLARE_BITMAP(sqe_op, IORING_OP_LAST);
|
||||
u8 sqe_flags_allowed;
|
||||
u8 sqe_flags_required;
|
||||
bool registered;
|
||||
/* IORING_OP_* restrictions exist */
|
||||
bool op_registered;
|
||||
/* IORING_REGISTER_* restrictions exist */
|
||||
bool reg_registered;
|
||||
};
|
||||
|
||||
struct io_submit_link {
|
||||
@@ -259,7 +262,8 @@ struct io_ring_ctx {
|
||||
struct {
|
||||
unsigned int flags;
|
||||
unsigned int drain_next: 1;
|
||||
unsigned int restricted: 1;
|
||||
unsigned int op_restricted: 1;
|
||||
unsigned int reg_restricted: 1;
|
||||
unsigned int off_timeout_used: 1;
|
||||
unsigned int drain_active: 1;
|
||||
unsigned int has_evfd: 1;
|
||||
@@ -316,7 +320,7 @@ struct io_ring_ctx {
|
||||
* manipulate the list, hence no extra locking is needed there.
|
||||
*/
|
||||
bool poll_multi_queue;
|
||||
struct io_wq_work_list iopoll_list;
|
||||
struct list_head iopoll_list;
|
||||
|
||||
struct io_file_table file_table;
|
||||
struct io_rsrc_data buf_table;
|
||||
@@ -444,6 +448,9 @@ struct io_ring_ctx {
|
||||
struct list_head defer_list;
|
||||
unsigned nr_drained;
|
||||
|
||||
/* protected by ->completion_lock */
|
||||
unsigned nr_req_allocated;
|
||||
|
||||
#ifdef CONFIG_NET_RX_BUSY_POLL
|
||||
struct list_head napi_list; /* track busy poll napi_id */
|
||||
spinlock_t napi_lock; /* napi_list lock */
|
||||
@@ -456,10 +463,6 @@ struct io_ring_ctx {
|
||||
DECLARE_HASHTABLE(napi_ht, 4);
|
||||
#endif
|
||||
|
||||
/* protected by ->completion_lock */
|
||||
unsigned evfd_last_cq_tail;
|
||||
unsigned nr_req_allocated;
|
||||
|
||||
/*
|
||||
* Protection for resize vs mmap races - both the mmap and resize
|
||||
* side will need to grab this lock, to prevent either side from
|
||||
@@ -714,15 +717,21 @@ struct io_kiocb {
|
||||
|
||||
atomic_t refs;
|
||||
bool cancel_seq_set;
|
||||
struct io_task_work io_task_work;
|
||||
|
||||
union {
|
||||
struct io_task_work io_task_work;
|
||||
/* For IOPOLL setup queues, with hybrid polling */
|
||||
u64 iopoll_start;
|
||||
};
|
||||
|
||||
union {
|
||||
/*
|
||||
* for polled requests, i.e. IORING_OP_POLL_ADD and async armed
|
||||
* poll
|
||||
*/
|
||||
struct hlist_node hash_node;
|
||||
/* For IOPOLL setup queues, with hybrid polling */
|
||||
u64 iopoll_start;
|
||||
/* IOPOLL completion handling */
|
||||
struct list_head iopoll_node;
|
||||
/* for private io_kiocb freeing */
|
||||
struct rcu_head rcu_head;
|
||||
};
|
||||
|
||||
@@ -237,6 +237,18 @@ enum io_uring_sqe_flags_bit {
|
||||
*/
|
||||
#define IORING_SETUP_SQE_MIXED (1U << 19)
|
||||
|
||||
/*
|
||||
* When set, io_uring ignores SQ head and tail and fetches SQEs to submit
|
||||
* starting from index 0 instead from the index stored in the head pointer.
|
||||
* IOW, the user should place all SQE at the beginning of the SQ memory
|
||||
* before issuing a submission syscall.
|
||||
*
|
||||
* It requires IORING_SETUP_NO_SQARRAY and is incompatible with
|
||||
* IORING_SETUP_SQPOLL. The user must also never change the SQ head and tail
|
||||
* values and keep it set to 0. Any other value is undefined behaviour.
|
||||
*/
|
||||
#define IORING_SETUP_SQ_REWIND (1U << 20)
|
||||
|
||||
enum io_uring_op {
|
||||
IORING_OP_NOP,
|
||||
IORING_OP_READV,
|
||||
|
||||
+8
-6
@@ -8,12 +8,14 @@ endif
|
||||
|
||||
obj-$(CONFIG_IO_URING) += io_uring.o opdef.o kbuf.o rsrc.o notif.o \
|
||||
tctx.o filetable.o rw.o poll.o \
|
||||
eventfd.o uring_cmd.o openclose.o \
|
||||
sqpoll.o xattr.o nop.o fs.o splice.o \
|
||||
sync.o msg_ring.o advise.o openclose.o \
|
||||
statx.o timeout.o cancel.o \
|
||||
waitid.o register.o truncate.o \
|
||||
memmap.o alloc_cache.o query.o
|
||||
tw.o wait.o eventfd.o uring_cmd.o \
|
||||
openclose.o sqpoll.o xattr.o nop.o \
|
||||
fs.o splice.o sync.o msg_ring.o \
|
||||
advise.o openclose.o statx.o timeout.o \
|
||||
cancel.o waitid.o register.o \
|
||||
truncate.o memmap.o alloc_cache.o \
|
||||
query.o
|
||||
|
||||
obj-$(CONFIG_IO_URING_ZCRX) += zcrx.o
|
||||
obj-$(CONFIG_IO_WQ) += io-wq.o
|
||||
obj-$(CONFIG_FUTEX) += futex.o
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
/* SPDX-License-Identifier: GPL-2.0 */
|
||||
#ifndef IOU_ALLOC_CACHE_H
|
||||
#define IOU_ALLOC_CACHE_H
|
||||
|
||||
#include <linux/io_uring_types.h>
|
||||
#include <linux/kasan.h>
|
||||
|
||||
/*
|
||||
* Don't allow the cache to grow beyond this size.
|
||||
|
||||
+2
-3
@@ -2,10 +2,8 @@
|
||||
#include <linux/kernel.h>
|
||||
#include <linux/errno.h>
|
||||
#include <linux/fs.h>
|
||||
#include <linux/file.h>
|
||||
#include <linux/mm.h>
|
||||
#include <linux/slab.h>
|
||||
#include <linux/namei.h>
|
||||
#include <linux/nospec.h>
|
||||
#include <linux/io_uring.h>
|
||||
|
||||
@@ -21,6 +19,7 @@
|
||||
#include "waitid.h"
|
||||
#include "futex.h"
|
||||
#include "cancel.h"
|
||||
#include "wait.h"
|
||||
|
||||
struct io_cancel {
|
||||
struct file *file;
|
||||
@@ -539,7 +538,7 @@ __cold bool io_uring_try_cancel_requests(struct io_ring_ctx *ctx,
|
||||
/* SQPOLL thread does its own polling */
|
||||
if ((!(ctx->flags & IORING_SETUP_SQPOLL) && cancel_all) ||
|
||||
is_sqpoll_thread) {
|
||||
while (!wq_list_empty(&ctx->iopoll_list)) {
|
||||
while (!list_empty(&ctx->iopoll_list)) {
|
||||
io_iopoll_try_reap_events(ctx);
|
||||
ret = true;
|
||||
cond_resched();
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: GPL-2.0
|
||||
#include <asm/ioctls.h>
|
||||
#include <linux/io_uring/net.h>
|
||||
#include <linux/errqueue.h>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
/* SPDX-License-Identifier: GPL-2.0 */
|
||||
|
||||
struct io_ring_ctx;
|
||||
int io_eventfd_register(struct io_ring_ctx *ctx, void __user *arg,
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
#ifndef IOU_FILE_TABLE_H
|
||||
#define IOU_FILE_TABLE_H
|
||||
|
||||
#include <linux/file.h>
|
||||
#include <linux/io_uring_types.h>
|
||||
#include "rsrc.h"
|
||||
|
||||
|
||||
+1
-1
@@ -186,7 +186,7 @@ int io_futexv_prep(struct io_kiocb *req, const struct io_uring_sqe *sqe)
|
||||
return -EINVAL;
|
||||
|
||||
ifd = kzalloc(struct_size_t(struct io_futexv_data, futexv, iof->futex_nr),
|
||||
GFP_KERNEL);
|
||||
GFP_KERNEL_ACCOUNT);
|
||||
if (!ifd)
|
||||
return -ENOMEM;
|
||||
|
||||
|
||||
+48
-3
@@ -17,6 +17,7 @@
|
||||
#include <linux/task_work.h>
|
||||
#include <linux/audit.h>
|
||||
#include <linux/mmu_context.h>
|
||||
#include <linux/sched/sysctl.h>
|
||||
#include <uapi/linux/io_uring.h>
|
||||
|
||||
#include "io-wq.h"
|
||||
@@ -34,6 +35,7 @@ enum {
|
||||
|
||||
enum {
|
||||
IO_WQ_BIT_EXIT = 0, /* wq exiting */
|
||||
IO_WQ_BIT_EXIT_ON_IDLE = 1, /* allow all workers to exit on idle */
|
||||
};
|
||||
|
||||
enum {
|
||||
@@ -706,9 +708,13 @@ static int io_wq_worker(void *data)
|
||||
raw_spin_lock(&acct->workers_lock);
|
||||
/*
|
||||
* Last sleep timed out. Exit if we're not the last worker,
|
||||
* or if someone modified our affinity.
|
||||
* or if someone modified our affinity. If wq is marked
|
||||
* idle-exit, drop the worker as well. This is used to avoid
|
||||
* keeping io-wq workers around for tasks that no longer have
|
||||
* any active io_uring instances.
|
||||
*/
|
||||
if (last_timeout && (exit_mask || acct->nr_workers > 1)) {
|
||||
if ((last_timeout && (exit_mask || acct->nr_workers > 1)) ||
|
||||
test_bit(IO_WQ_BIT_EXIT_ON_IDLE, &wq->state)) {
|
||||
acct->nr_workers--;
|
||||
raw_spin_unlock(&acct->workers_lock);
|
||||
__set_current_state(TASK_RUNNING);
|
||||
@@ -963,6 +969,24 @@ static bool io_wq_worker_wake(struct io_worker *worker, void *data)
|
||||
return false;
|
||||
}
|
||||
|
||||
void io_wq_set_exit_on_idle(struct io_wq *wq, bool enable)
|
||||
{
|
||||
if (!wq->task)
|
||||
return;
|
||||
|
||||
if (!enable) {
|
||||
clear_bit(IO_WQ_BIT_EXIT_ON_IDLE, &wq->state);
|
||||
return;
|
||||
}
|
||||
|
||||
if (test_and_set_bit(IO_WQ_BIT_EXIT_ON_IDLE, &wq->state))
|
||||
return;
|
||||
|
||||
rcu_read_lock();
|
||||
io_wq_for_each_worker(wq, io_wq_worker_wake, NULL);
|
||||
rcu_read_unlock();
|
||||
}
|
||||
|
||||
static void io_run_cancel(struct io_wq_work *work, struct io_wq *wq)
|
||||
{
|
||||
do {
|
||||
@@ -1313,6 +1337,8 @@ static void io_wq_cancel_tw_create(struct io_wq *wq)
|
||||
|
||||
static void io_wq_exit_workers(struct io_wq *wq)
|
||||
{
|
||||
unsigned long timeout, warn_timeout;
|
||||
|
||||
if (!wq->task)
|
||||
return;
|
||||
|
||||
@@ -1322,7 +1348,26 @@ static void io_wq_exit_workers(struct io_wq *wq)
|
||||
io_wq_for_each_worker(wq, io_wq_worker_wake, NULL);
|
||||
rcu_read_unlock();
|
||||
io_worker_ref_put(wq);
|
||||
wait_for_completion(&wq->worker_done);
|
||||
|
||||
/*
|
||||
* Shut up hung task complaint, see for example
|
||||
*
|
||||
* https://lore.kernel.org/all/696fc9e7.a70a0220.111c58.0006.GAE@google.com/
|
||||
*
|
||||
* where completely overloading the system with tons of long running
|
||||
* io-wq items can easily trigger the hung task timeout. Only sleep
|
||||
* uninterruptibly for half that time, and warn if we exceeded end
|
||||
* up waiting more than IO_URING_EXIT_WAIT_MAX.
|
||||
*/
|
||||
timeout = sysctl_hung_task_timeout_secs * HZ / 2;
|
||||
if (!timeout)
|
||||
timeout = MAX_SCHEDULE_TIMEOUT;
|
||||
warn_timeout = jiffies + IO_URING_EXIT_WAIT_MAX;
|
||||
do {
|
||||
if (wait_for_completion_timeout(&wq->worker_done, timeout))
|
||||
break;
|
||||
WARN_ON_ONCE(time_after(jiffies, warn_timeout));
|
||||
} while (1);
|
||||
|
||||
spin_lock_irq(&wq->hash->wait.lock);
|
||||
list_del_init(&wq->wait.entry);
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
/* SPDX-License-Identifier: GPL-2.0 */
|
||||
#ifndef INTERNAL_IO_WQ_H
|
||||
#define INTERNAL_IO_WQ_H
|
||||
|
||||
@@ -41,6 +42,7 @@ struct io_wq_data {
|
||||
struct io_wq *io_wq_create(unsigned bounded, struct io_wq_data *data);
|
||||
void io_wq_exit_start(struct io_wq *wq);
|
||||
void io_wq_put_and_exit(struct io_wq *wq);
|
||||
void io_wq_set_exit_on_idle(struct io_wq *wq, bool enable);
|
||||
|
||||
void io_wq_enqueue(struct io_wq *wq, struct io_wq_work *work);
|
||||
void io_wq_hash_work(struct io_wq_work *work, void *val);
|
||||
|
||||
+43
-739
File diff suppressed because it is too large
Load Diff
+12
-78
@@ -1,16 +1,17 @@
|
||||
/* SPDX-License-Identifier: GPL-2.0 */
|
||||
#ifndef IOU_CORE_H
|
||||
#define IOU_CORE_H
|
||||
|
||||
#include <linux/errno.h>
|
||||
#include <linux/lockdep.h>
|
||||
#include <linux/resume_user_mode.h>
|
||||
#include <linux/kasan.h>
|
||||
#include <linux/poll.h>
|
||||
#include <linux/io_uring_types.h>
|
||||
#include <uapi/linux/eventpoll.h>
|
||||
#include "alloc_cache.h"
|
||||
#include "io-wq.h"
|
||||
#include "slist.h"
|
||||
#include "tw.h"
|
||||
#include "opdef.h"
|
||||
|
||||
#ifndef CREATE_TRACE_POINTS
|
||||
@@ -69,7 +70,8 @@ struct io_ctx_config {
|
||||
IORING_SETUP_NO_SQARRAY |\
|
||||
IORING_SETUP_HYBRID_IOPOLL |\
|
||||
IORING_SETUP_CQE_MIXED |\
|
||||
IORING_SETUP_SQE_MIXED)
|
||||
IORING_SETUP_SQE_MIXED |\
|
||||
IORING_SETUP_SQ_REWIND)
|
||||
|
||||
#define IORING_ENTER_FLAGS (IORING_ENTER_GETEVENTS |\
|
||||
IORING_ENTER_SQ_WAKEUP |\
|
||||
@@ -89,6 +91,14 @@ struct io_ctx_config {
|
||||
IOSQE_BUFFER_SELECT |\
|
||||
IOSQE_CQE_SKIP_SUCCESS)
|
||||
|
||||
#define IO_REQ_LINK_FLAGS (REQ_F_LINK | REQ_F_HARDLINK)
|
||||
|
||||
/*
|
||||
* Complaint timeout for io_uring cancelation exits, and for io-wq exit
|
||||
* worker waiting.
|
||||
*/
|
||||
#define IO_URING_EXIT_WAIT_MAX (HZ * 60 * 5)
|
||||
|
||||
enum {
|
||||
IOU_COMPLETE = 0,
|
||||
|
||||
@@ -151,8 +161,6 @@ static inline bool io_should_wake(struct io_wait_queue *iowq)
|
||||
int io_prepare_config(struct io_ctx_config *config);
|
||||
|
||||
bool io_cqe_cache_refill(struct io_ring_ctx *ctx, bool overflow, bool cqe32);
|
||||
int io_run_task_work_sig(struct io_ring_ctx *ctx);
|
||||
int io_run_local_work(struct io_ring_ctx *ctx, int min_events, int max_events);
|
||||
void io_req_defer_failed(struct io_kiocb *req, s32 res);
|
||||
bool io_post_aux_cqe(struct io_ring_ctx *ctx, u64 user_data, s32 res, u32 cflags);
|
||||
void io_add_aux_cqe(struct io_ring_ctx *ctx, u64 user_data, s32 res, u32 cflags);
|
||||
@@ -166,15 +174,10 @@ struct file *io_file_get_normal(struct io_kiocb *req, int fd);
|
||||
struct file *io_file_get_fixed(struct io_kiocb *req, int fd,
|
||||
unsigned issue_flags);
|
||||
|
||||
void __io_req_task_work_add(struct io_kiocb *req, unsigned flags);
|
||||
void io_req_task_work_add_remote(struct io_kiocb *req, unsigned flags);
|
||||
void io_req_task_queue(struct io_kiocb *req);
|
||||
void io_req_task_complete(struct io_tw_req tw_req, io_tw_token_t tw);
|
||||
void io_req_task_queue_fail(struct io_kiocb *req, int ret);
|
||||
void io_req_task_submit(struct io_tw_req tw_req, io_tw_token_t tw);
|
||||
struct llist_node *io_handle_tw_list(struct llist_node *node, unsigned int *count, unsigned int max_entries);
|
||||
struct llist_node *tctx_task_work_run(struct io_uring_task *tctx, unsigned int max_entries, unsigned int *count);
|
||||
void tctx_task_work(struct callback_head *cb);
|
||||
__cold void io_uring_drop_tctx_refs(struct task_struct *task);
|
||||
|
||||
int io_ring_add_registered_file(struct io_uring_task *tctx, struct file *file,
|
||||
@@ -227,11 +230,6 @@ static inline bool io_is_compat(struct io_ring_ctx *ctx)
|
||||
return IS_ENABLED(CONFIG_COMPAT) && unlikely(ctx->compat);
|
||||
}
|
||||
|
||||
static inline void io_req_task_work_add(struct io_kiocb *req)
|
||||
{
|
||||
__io_req_task_work_add(req, 0);
|
||||
}
|
||||
|
||||
static inline void io_submit_flush_completions(struct io_ring_ctx *ctx)
|
||||
{
|
||||
if (!wq_list_empty(&ctx->submit_state.compl_reqs) ||
|
||||
@@ -456,59 +454,6 @@ static inline unsigned int io_sqring_entries(struct io_ring_ctx *ctx)
|
||||
return min(entries, ctx->sq_entries);
|
||||
}
|
||||
|
||||
static inline int io_run_task_work(void)
|
||||
{
|
||||
bool ret = false;
|
||||
|
||||
/*
|
||||
* Always check-and-clear the task_work notification signal. With how
|
||||
* signaling works for task_work, we can find it set with nothing to
|
||||
* run. We need to clear it for that case, like get_signal() does.
|
||||
*/
|
||||
if (test_thread_flag(TIF_NOTIFY_SIGNAL))
|
||||
clear_notify_signal();
|
||||
/*
|
||||
* PF_IO_WORKER never returns to userspace, so check here if we have
|
||||
* notify work that needs processing.
|
||||
*/
|
||||
if (current->flags & PF_IO_WORKER) {
|
||||
if (test_thread_flag(TIF_NOTIFY_RESUME)) {
|
||||
__set_current_state(TASK_RUNNING);
|
||||
resume_user_mode_work(NULL);
|
||||
}
|
||||
if (current->io_uring) {
|
||||
unsigned int count = 0;
|
||||
|
||||
__set_current_state(TASK_RUNNING);
|
||||
tctx_task_work_run(current->io_uring, UINT_MAX, &count);
|
||||
if (count)
|
||||
ret = true;
|
||||
}
|
||||
}
|
||||
if (task_work_pending(current)) {
|
||||
__set_current_state(TASK_RUNNING);
|
||||
task_work_run();
|
||||
ret = true;
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline bool io_local_work_pending(struct io_ring_ctx *ctx)
|
||||
{
|
||||
return !llist_empty(&ctx->work_llist) || !llist_empty(&ctx->retry_llist);
|
||||
}
|
||||
|
||||
static inline bool io_task_work_pending(struct io_ring_ctx *ctx)
|
||||
{
|
||||
return task_work_pending(current) || io_local_work_pending(ctx);
|
||||
}
|
||||
|
||||
static inline void io_tw_lock(struct io_ring_ctx *ctx, io_tw_token_t tw)
|
||||
{
|
||||
lockdep_assert_held(&ctx->uring_lock);
|
||||
}
|
||||
|
||||
/*
|
||||
* Don't complete immediately but use deferred completion infrastructure.
|
||||
* Protected by ->uring_lock and can only be used either with
|
||||
@@ -566,17 +511,6 @@ static inline bool io_alloc_req(struct io_ring_ctx *ctx, struct io_kiocb **req)
|
||||
return true;
|
||||
}
|
||||
|
||||
static inline bool io_allowed_defer_tw_run(struct io_ring_ctx *ctx)
|
||||
{
|
||||
return likely(ctx->submitter_task == current);
|
||||
}
|
||||
|
||||
static inline bool io_allowed_run_tw(struct io_ring_ctx *ctx)
|
||||
{
|
||||
return likely(!(ctx->flags & IORING_SETUP_DEFER_TASKRUN) ||
|
||||
ctx->submitter_task == current);
|
||||
}
|
||||
|
||||
static inline void io_req_queue_tw_complete(struct io_kiocb *req, s32 res)
|
||||
{
|
||||
io_req_set_res(req, res, 0);
|
||||
|
||||
+3
-2
@@ -669,8 +669,9 @@ int io_register_pbuf_ring(struct io_ring_ctx *ctx, void __user *arg)
|
||||
bl->buf_ring = br;
|
||||
if (reg.flags & IOU_PBUF_RING_INC)
|
||||
bl->flags |= IOBL_INC;
|
||||
io_buffer_add_list(ctx, bl, reg.bgid);
|
||||
return 0;
|
||||
ret = io_buffer_add_list(ctx, bl, reg.bgid);
|
||||
if (!ret)
|
||||
return 0;
|
||||
fail:
|
||||
io_free_region(ctx->user, &bl->region);
|
||||
kfree(bl);
|
||||
|
||||
+1
-1
@@ -56,7 +56,7 @@ struct page **io_pin_pages(unsigned long uaddr, unsigned long len, int *npages)
|
||||
if (WARN_ON_ONCE(nr_pages > INT_MAX))
|
||||
return ERR_PTR(-EOVERFLOW);
|
||||
|
||||
pages = kvmalloc_array(nr_pages, sizeof(struct page *), GFP_KERNEL);
|
||||
pages = kvmalloc_array(nr_pages, sizeof(struct page *), GFP_KERNEL_ACCOUNT);
|
||||
if (!pages)
|
||||
return ERR_PTR(-ENOMEM);
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
/* SPDX-License-Identifier: GPL-2.0 */
|
||||
#ifndef IO_URING_MEMMAP_H
|
||||
#define IO_URING_MEMMAP_H
|
||||
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: GPL-2.0
|
||||
#include <linux/device.h>
|
||||
#include <linux/init.h>
|
||||
#include <linux/kernel.h>
|
||||
|
||||
+14
-14
@@ -80,13 +80,9 @@ static void io_msg_tw_complete(struct io_tw_req tw_req, io_tw_token_t tw)
|
||||
percpu_ref_put(&ctx->refs);
|
||||
}
|
||||
|
||||
static int io_msg_remote_post(struct io_ring_ctx *ctx, struct io_kiocb *req,
|
||||
static void io_msg_remote_post(struct io_ring_ctx *ctx, struct io_kiocb *req,
|
||||
int res, u32 cflags, u64 user_data)
|
||||
{
|
||||
if (!READ_ONCE(ctx->submitter_task)) {
|
||||
kfree_rcu(req, rcu_head);
|
||||
return -EOWNERDEAD;
|
||||
}
|
||||
req->opcode = IORING_OP_NOP;
|
||||
req->cqe.user_data = user_data;
|
||||
io_req_set_res(req, res, cflags);
|
||||
@@ -95,7 +91,6 @@ static int io_msg_remote_post(struct io_ring_ctx *ctx, struct io_kiocb *req,
|
||||
req->tctx = NULL;
|
||||
req->io_task_work.func = io_msg_tw_complete;
|
||||
io_req_task_work_add_remote(req, IOU_F_TWQ_LAZY_WAKE);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int io_msg_data_remote(struct io_ring_ctx *target_ctx,
|
||||
@@ -111,8 +106,8 @@ static int io_msg_data_remote(struct io_ring_ctx *target_ctx,
|
||||
if (msg->flags & IORING_MSG_RING_FLAGS_PASS)
|
||||
flags = msg->cqe_flags;
|
||||
|
||||
return io_msg_remote_post(target_ctx, target, msg->len, flags,
|
||||
msg->user_data);
|
||||
io_msg_remote_post(target_ctx, target, msg->len, flags, msg->user_data);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int __io_msg_ring_data(struct io_ring_ctx *target_ctx,
|
||||
@@ -125,7 +120,11 @@ static int __io_msg_ring_data(struct io_ring_ctx *target_ctx,
|
||||
return -EINVAL;
|
||||
if (!(msg->flags & IORING_MSG_RING_FLAGS_PASS) && msg->dst_fd)
|
||||
return -EINVAL;
|
||||
if (target_ctx->flags & IORING_SETUP_R_DISABLED)
|
||||
/*
|
||||
* Keep IORING_SETUP_R_DISABLED check before submitter_task load
|
||||
* in io_msg_data_remote() -> io_req_task_work_add_remote()
|
||||
*/
|
||||
if (smp_load_acquire(&target_ctx->flags) & IORING_SETUP_R_DISABLED)
|
||||
return -EBADFD;
|
||||
|
||||
if (io_msg_need_remote(target_ctx))
|
||||
@@ -223,10 +222,7 @@ static int io_msg_fd_remote(struct io_kiocb *req)
|
||||
{
|
||||
struct io_ring_ctx *ctx = req->file->private_data;
|
||||
struct io_msg *msg = io_kiocb_to_cmd(req, struct io_msg);
|
||||
struct task_struct *task = READ_ONCE(ctx->submitter_task);
|
||||
|
||||
if (unlikely(!task))
|
||||
return -EOWNERDEAD;
|
||||
struct task_struct *task = ctx->submitter_task;
|
||||
|
||||
init_task_work(&msg->tw, io_msg_tw_fd_complete);
|
||||
if (task_work_add(task, &msg->tw, TWA_SIGNAL))
|
||||
@@ -245,7 +241,11 @@ static int io_msg_send_fd(struct io_kiocb *req, unsigned int issue_flags)
|
||||
return -EINVAL;
|
||||
if (target_ctx == ctx)
|
||||
return -EINVAL;
|
||||
if (target_ctx->flags & IORING_SETUP_R_DISABLED)
|
||||
/*
|
||||
* Keep IORING_SETUP_R_DISABLED check before submitter_task load
|
||||
* in io_msg_fd_remote()
|
||||
*/
|
||||
if (smp_load_acquire(&target_ctx->flags) & IORING_SETUP_R_DISABLED)
|
||||
return -EBADFD;
|
||||
if (!msg->src_file) {
|
||||
int ret = io_msg_grab_file(req, issue_flags);
|
||||
|
||||
+5
-1
@@ -515,7 +515,11 @@ static inline bool io_send_finish(struct io_kiocb *req,
|
||||
|
||||
cflags = io_put_kbufs(req, sel->val, sel->buf_list, io_bundle_nbufs(kmsg, sel->val));
|
||||
|
||||
if (bundle_finished || req->flags & REQ_F_BL_EMPTY)
|
||||
/*
|
||||
* Don't start new bundles if the buffer list is empty, or if the
|
||||
* current operation needed to go through polling to complete.
|
||||
*/
|
||||
if (bundle_finished || req->flags & (REQ_F_BL_EMPTY | REQ_F_POLLED))
|
||||
goto finish;
|
||||
|
||||
/*
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
// SPDX-License-Identifier: GPL-2.0
|
||||
#include <linux/kernel.h>
|
||||
#include <linux/errno.h>
|
||||
#include <linux/file.h>
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user