mirror of
https://github.com/torvalds/linux
synced 2026-07-22 10:10:47 +09:00
The local (DEFER_TASKRUN) task_work list is an llist, which is LIFO
ordered, and hence __io_run_local_work() has to restore the right
running order with an O(n) llist_reverse_order() pass first. On top of
that, a batch that gets capped by max_events needs the leftover entries
parked on a separate ->retry_llist, as they can't be pushed back to the
shared list.
Switch it to the FIFO mpscq. Adds are wait-free instead of a cmpxchg
retry loop, entries are popped in queue order with no reversal pass,
capping a run simply leaves the remainder on the queue, and
->retry_llist goes away entirely. The consumer cursor, ->work_head,
lives with the rest of the ->uring_lock protected state rather than
next to the queue, so that popping entries doesn't dirty the producer
side cacheline.
For low amounts of task_work, this ends up being a bit more efficient
than the existing scheme. As an example of that, doing multishot
receives for 8 clients has the following task_work overhead:
1.02% sock-test [kernel.kallsyms] [k] io_req_local_work_add
0.88% sock-test [kernel.kallsyms] [k] __io_run_local_work_loop
0.60% sock-test [kernel.kallsyms] [k] llist_reverse_order
0.14% sock-test [kernel.kallsyms] [k] __io_run_local_work
2.64% at ~46Gb/sec
and after this change:
1.08% sock-test [kernel.kallsyms] [k] io_req_local_work_add
1.03% sock-test [kernel.kallsyms] [k] __io_run_local_work
2.11% at ~53Gb/sec
which has less overhead even though that test run was faster. For a case
of having 1024 clients on a single ring:
2.22% sock-test [kernel.kallsyms] [k] llist_reverse_order
0.84% sock-test [kernel.kallsyms] [k] __io_run_local_work_loop
0.42% sock-test [kernel.kallsyms] [k] io_req_local_work_add
0.02% sock-test [kernel.kallsyms] [k] __io_run_local_work
3.50% at ~24Gb/sec
we start to see the llist reversing taking a considerable amount of
time, and the total add+run task_work overhead is around 3.5%. After
the change:
0.90% sock-test [kernel.kallsyms] [k] __io_run_local_work
0.42% sock-test [kernel.kallsyms] [k] io_req_local_work_add
1.32% at ~26Gb/sec
most of that overhead is gone, and performance is better as well.
Caleb Sander Mateos <csander@purestorage.com> reports that it improves
the performance of a ublk 4kb workload by 4% [1], while testing v1 of
this patchset.
[1] https://lore.kernel.org/io-uring/CADUfDZr-MMYBaP-e+y9+xuRhuiunO2sBTUCmwZyd7AgT8sVtiQ@mail.gmail.com/
Signed-off-by: Jens Axboe <axboe@kernel.dk>
56 lines
1.5 KiB
C
56 lines
1.5 KiB
C
// SPDX-License-Identifier: GPL-2.0
|
|
#ifndef IOU_WAIT_H
|
|
#define IOU_WAIT_H
|
|
|
|
#include <linux/io_uring_types.h>
|
|
|
|
/*
|
|
* ->cq_wait_nr is armed with the number of lazy task_work adds the waiter
|
|
* still needs, and counted down by the add side, with the add reaching zero
|
|
* issuing the (single) wake up for this wait cycle. Zero and below means no
|
|
* wake up is to be issued: IO_CQ_WAKE_INIT when no task is waiting (also
|
|
* what a forced wake up resets it to when claiming one), zero once the
|
|
* countdown has fired.
|
|
*/
|
|
#define IO_CQ_WAKE_INIT (-1)
|
|
|
|
struct ext_arg {
|
|
size_t argsz;
|
|
struct timespec64 ts;
|
|
const sigset_t __user *sig;
|
|
ktime_t min_time;
|
|
bool ts_set;
|
|
bool iowait;
|
|
};
|
|
|
|
int io_cqring_wait(struct io_ring_ctx *ctx, int min_events, u32 flags,
|
|
struct ext_arg *ext_arg);
|
|
int io_run_task_work_sig(struct io_ring_ctx *ctx);
|
|
void io_cqring_do_overflow_flush(struct io_ring_ctx *ctx);
|
|
void io_cqring_overflow_flush_locked(struct io_ring_ctx *ctx);
|
|
|
|
static inline unsigned int __io_cqring_events(struct io_ring_ctx *ctx)
|
|
{
|
|
struct io_rings *rings = io_get_rings(ctx);
|
|
return ctx->cached_cq_tail - READ_ONCE(rings->cq.head);
|
|
}
|
|
|
|
static inline unsigned int __io_cqring_events_user(struct io_ring_ctx *ctx)
|
|
{
|
|
struct io_rings *rings = io_get_rings(ctx);
|
|
|
|
return READ_ONCE(rings->cq.tail) - READ_ONCE(rings->cq.head);
|
|
}
|
|
|
|
/*
|
|
* Reads the tail/head of the CQ ring while providing an acquire ordering,
|
|
* see comment at top of io_uring.c.
|
|
*/
|
|
static inline unsigned io_cqring_events(struct io_ring_ctx *ctx)
|
|
{
|
|
smp_rmb();
|
|
return __io_cqring_events(ctx);
|
|
}
|
|
|
|
#endif
|