xor: improve the runtime selection benchmark

Use plain ktime_get_ns for the timing, use 4 + 1 disks for a realistic
load, and report the throughput on the data disks instead of the that on
the parity disk, which isn't all that useful.

Link: https://lore.kernel.org/20260715144825.95432-3-hch@lst.de
Signed-off-by: Christoph Hellwig <hch@lst.de>
Cc: Eric Biggers <ebiggers@kernel.org>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
This commit is contained in:
Christoph Hellwig 2026-07-15 16:47:33 +02:00 committed by Andrew Morton
parent b6a1359bbe
commit 2bfd85fd81

View File

@ -10,7 +10,6 @@
#include <linux/gfp.h>
#include <linux/slab.h>
#include <linux/raid/xor.h>
#include <linux/jiffies.h>
#include <linux/preempt.h>
#include <linux/static_call.h>
#include "xor_impl.h"
@ -73,59 +72,56 @@ void __init xor_force(struct xor_block_template *tmpl)
forced_template = tmpl;
}
#define BENCH_SIZE 4096
#define BENCH_SIZE SZ_4K
#define NR_SRCS 4
#define REPS 800U
static void __init
do_xor_speed(struct xor_block_template *tmpl, void *b1, void *b2)
static void __init do_xor_speed(struct xor_block_template *tmpl, void *dest,
void *srcs[NR_SRCS])
{
int speed;
unsigned long reps;
ktime_t min, start, t0;
void *srcs[1] = { b2 };
u64 t;
int i;
preempt_disable();
reps = 0;
t0 = ktime_get();
/* delay start until time has advanced */
while ((start = ktime_get()) == t0)
cpu_relax();
do {
t = ktime_get_ns();
for (i = 0; i < REPS; i++) {
mb(); /* prevent loop optimization */
tmpl->xor_gen(b1, srcs, 1, BENCH_SIZE);
tmpl->xor_gen(dest, srcs, NR_SRCS, BENCH_SIZE);
mb();
} while (reps++ < REPS || (t0 = ktime_get()) == start);
min = ktime_sub(t0, start);
}
t = max(ktime_get_ns() - t, 1);
preempt_enable();
// bytes/ns == GB/s, multiply by 1000 to get MB/s [not MiB/s]
speed = (1000 * reps * BENCH_SIZE) / (unsigned int)ktime_to_ns(min);
tmpl->speed = speed;
/* bytes/ns == GB/s, multiply by 1000 to get MB/s [not MiB/s] */
tmpl->speed = div64_u64((u64)BENCH_SIZE * REPS * NR_SRCS * 1000, t);
pr_info(" %-16s: %5d MB/sec\n", tmpl->name, speed);
pr_info(" %-16s: %5d MB/sec\n", tmpl->name, tmpl->speed);
}
static int __init calibrate_xor_blocks(void)
{
void *b1, *b2;
struct xor_block_template *f, *fastest;
void *srcs[NR_SRCS];
void *buf, *dest;
int i;
if (forced_template)
return 0;
b1 = kmalloc(PAGE_SIZE * 4, GFP_KERNEL);
if (!b1) {
buf = kmalloc(BENCH_SIZE * (NR_SRCS + 1), GFP_KERNEL);
if (!buf) {
pr_warn("xor: Yikes! No memory available.\n");
return -ENOMEM;
}
b2 = b1 + 2*PAGE_SIZE + BENCH_SIZE;
get_random_bytes(buf, BENCH_SIZE * (NR_SRCS + 1));
dest = buf;
for (i = 0; i < NR_SRCS; i++)
srcs[i] = buf + (i + 1) * BENCH_SIZE;
pr_info("xor: measuring software checksum speed\n");
fastest = template_list;
for (f = template_list; f; f = f->next) {
do_xor_speed(f, b1, b2);
do_xor_speed(f, dest, srcs);
if (f->speed > fastest->speed)
fastest = f;
}
@ -133,9 +129,10 @@ static int __init calibrate_xor_blocks(void)
pr_info("xor: using function: %s (%d MB/sec)\n",
fastest->name, fastest->speed);
kfree(b1);
kfree(buf);
return 0;
}
#undef NR_SRCS
#ifdef CONFIG_XOR_BLOCKS_ARCH
#include "xor_arch.h" /* $SRCARCH/xor_arch.h */