Files
linux-stable-mirror/tools/testing/selftests/sched_ext/total_bw.c
T
Andrea RighiandPeter Zijlstra c6aa079968 selftests/sched_ext: Validate dl_server attach/detach in total_bw test
Extend the total_bw selftest to validate the fair/ext dl_server
auto-attach/detach operations.

After the existing consistency checks, the test now doubles the
fair_server's runtime on every CPU via debugfs and verifies that:
 1. total_bw grew after the customization (proves fair_server was
    attached and apply_params() honored the dl_bw_attached flag),
 2. with the minimal BPF scheduler loaded, total_bw drops back to the
    baseline value (proves fair_server was detached and ext_server was
    attached at its own default runtime),
 3. after unload total_bw matches the doubled value from step 1 (proves
    fair_server was re-attached with the runtime customization preserved
    across the load/unload cycle).

Signed-off-by: Andrea Righi <arighi@nvidia.com>
Signed-off-by: Peter Zijlstra (Intel) <peterz@infradead.org>
Acked-by: Juri Lelli <juri.lelli@redhat.com>
Link: https://patch.msgid.link/20260526164420.638711-3-arighi@nvidia.com
2026-05-29 12:43:16 +02:00

481 lines
12 KiB
C

// SPDX-License-Identifier: GPL-2.0
/*
* Test to verify that total_bw value remains consistent across all CPUs
* in different BPF program states.
*
* Copyright (C) 2025 NVIDIA Corporation.
*/
#include <bpf/bpf.h>
#include <errno.h>
#include <pthread.h>
#include <scx/common.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/wait.h>
#include <unistd.h>
#include "minimal.bpf.skel.h"
#include "scx_test.h"
#define MAX_CPUS 512
#define STRESS_DURATION_SEC 5
struct total_bw_ctx {
struct minimal *skel;
long baseline_bw[MAX_CPUS];
int nr_cpus;
};
static void *cpu_stress_thread(void *arg)
{
volatile int i;
time_t end_time = time(NULL) + STRESS_DURATION_SEC;
while (time(NULL) < end_time)
for (i = 0; i < 1000000; i++)
;
return NULL;
}
/*
* The first enqueue on a CPU causes the DL server to start, for that
* reason run stressor threads in the hopes it schedules on all CPUs.
*/
static int run_cpu_stress(int nr_cpus)
{
pthread_t *threads;
int i, ret = 0;
threads = calloc(nr_cpus, sizeof(pthread_t));
if (!threads)
return -ENOMEM;
/* Create threads to run on each CPU */
for (i = 0; i < nr_cpus; i++) {
if (pthread_create(&threads[i], NULL, cpu_stress_thread, NULL)) {
ret = -errno;
fprintf(stderr, "Failed to create thread %d: %s\n", i, strerror(-ret));
break;
}
}
/* Wait for all threads to complete */
for (i = 0; i < nr_cpus; i++) {
if (threads[i])
pthread_join(threads[i], NULL);
}
free(threads);
return ret;
}
static int read_total_bw_values(long *bw_values, int max_cpus)
{
FILE *fp;
char line[256];
int cpu_count = 0;
fp = fopen("/sys/kernel/debug/sched/debug", "r");
if (!fp) {
SCX_ERR("Failed to open debug file");
return -1;
}
while (fgets(line, sizeof(line), fp)) {
char *bw_str = strstr(line, "total_bw");
if (bw_str) {
bw_str = strchr(bw_str, ':');
if (bw_str) {
/* Only store up to max_cpus values */
if (cpu_count < max_cpus)
bw_values[cpu_count] = atol(bw_str + 1);
cpu_count++;
}
}
}
fclose(fp);
return cpu_count;
}
/*
* Read a per-CPU dl_server param (runtime or period) from debugfs.
* Returns the value in nanoseconds, or -1 on failure.
*/
static long read_server_param(const char *server, const char *param, int cpu)
{
char path[128];
long value = -1;
FILE *fp;
snprintf(path, sizeof(path),
"/sys/kernel/debug/sched/%s_server/cpu%d/%s",
server, cpu, param);
fp = fopen(path, "r");
if (!fp)
return -1;
if (fscanf(fp, "%ld", &value) != 1)
value = -1;
fclose(fp);
return value;
}
/*
* Write a per-CPU dl_server param to debugfs. Returns 0 on success.
*/
static int write_server_param(const char *server, const char *param,
int cpu, long value)
{
char path[128];
FILE *fp;
int ret = 0;
snprintf(path, sizeof(path),
"/sys/kernel/debug/sched/%s_server/cpu%d/%s",
server, cpu, param);
fp = fopen(path, "w");
if (!fp)
return -1;
if (fprintf(fp, "%ld", value) < 0)
ret = -1;
if (fclose(fp) != 0)
ret = -1;
return ret;
}
static int read_fair_runtime_all(int nr_cpus, long *runtimes)
{
int i;
for (i = 0; i < nr_cpus; i++) {
runtimes[i] = read_server_param("fair", "runtime", i);
if (runtimes[i] <= 0)
return -1;
}
return 0;
}
static int write_fair_runtime_all(int nr_cpus, long value)
{
int i;
for (i = 0; i < nr_cpus; i++) {
if (write_server_param("fair", "runtime", i, value) < 0) {
SCX_ERR("Failed to write fair_server runtime on CPU %d", i);
return -1;
}
}
return 0;
}
/*
* Restore per-CPU fair_server runtimes.
*/
static int restore_fair_runtime_all(int nr_cpus, const long *runtimes)
{
int ret = 0;
int i;
for (i = 0; i < nr_cpus; i++) {
if (write_server_param("fair", "runtime", i, runtimes[i]) < 0) {
SCX_ERR("Failed to restore fair_server runtime on CPU %d", i);
ret = -1;
}
}
return ret;
}
static bool verify_total_bw_consistency(long *bw_values, int count)
{
int i;
long first_value;
if (count <= 0)
return false;
first_value = bw_values[0];
for (i = 1; i < count; i++) {
if (bw_values[i] != first_value) {
SCX_ERR("Inconsistent total_bw: CPU0=%ld, CPU%d=%ld",
first_value, i, bw_values[i]);
return false;
}
}
return true;
}
static int fetch_verify_total_bw(long *bw_values, int nr_cpus)
{
int attempts = 0;
int max_attempts = 10;
int count;
/*
* The first enqueue on a CPU causes the DL server to start, for that
* reason run stressor threads in the hopes it schedules on all CPUs.
*/
if (run_cpu_stress(nr_cpus) < 0) {
SCX_ERR("Failed to run CPU stress");
return -1;
}
/* Try multiple times to get stable values */
while (attempts < max_attempts) {
count = read_total_bw_values(bw_values, nr_cpus);
fprintf(stderr, "Read %d total_bw values (testing %d CPUs)\n", count, nr_cpus);
/* If system has more CPUs than we're testing, that's OK */
if (count < nr_cpus) {
SCX_ERR("Expected at least %d CPUs, got %d", nr_cpus, count);
attempts++;
sleep(1);
continue;
}
/* Only verify the CPUs we're testing */
if (verify_total_bw_consistency(bw_values, nr_cpus)) {
fprintf(stderr, "Values are consistent: %ld\n", bw_values[0]);
return 0;
}
attempts++;
sleep(1);
}
return -1;
}
static enum scx_test_status setup(void **ctx)
{
struct total_bw_ctx *test_ctx;
if (access("/sys/kernel/debug/sched/debug", R_OK) != 0) {
fprintf(stderr, "Skipping test: debugfs sched/debug not accessible\n");
return SCX_TEST_SKIP;
}
test_ctx = calloc(1, sizeof(*test_ctx));
if (!test_ctx)
return SCX_TEST_FAIL;
test_ctx->nr_cpus = sysconf(_SC_NPROCESSORS_ONLN);
if (test_ctx->nr_cpus <= 0) {
free(test_ctx);
return SCX_TEST_FAIL;
}
/* If system has more CPUs than MAX_CPUS, just test the first MAX_CPUS */
if (test_ctx->nr_cpus > MAX_CPUS)
test_ctx->nr_cpus = MAX_CPUS;
/* Test scenario 1: BPF program not loaded */
/* Read and verify baseline total_bw before loading BPF program */
fprintf(stderr, "BPF prog initially not loaded, reading total_bw values\n");
if (fetch_verify_total_bw(test_ctx->baseline_bw, test_ctx->nr_cpus) < 0) {
SCX_ERR("Failed to get stable baseline values");
free(test_ctx);
return SCX_TEST_FAIL;
}
/* Load the BPF skeleton */
test_ctx->skel = minimal__open();
if (!test_ctx->skel) {
free(test_ctx);
return SCX_TEST_FAIL;
}
SCX_ENUM_INIT(test_ctx->skel);
if (minimal__load(test_ctx->skel)) {
minimal__destroy(test_ctx->skel);
free(test_ctx);
return SCX_TEST_FAIL;
}
*ctx = test_ctx;
return SCX_TEST_PASS;
}
static enum scx_test_status run(void *ctx)
{
struct total_bw_ctx *test_ctx = ctx;
struct bpf_link *link;
long loaded_bw[MAX_CPUS];
long unloaded_bw[MAX_CPUS];
long doubled_bw[MAX_CPUS];
long original_runtime[MAX_CPUS], doubled_runtime;
enum scx_test_status ret;
int i;
/* Test scenario 2: BPF program loaded */
link = bpf_map__attach_struct_ops(test_ctx->skel->maps.minimal_ops);
if (!link) {
SCX_ERR("Failed to attach scheduler");
return SCX_TEST_FAIL;
}
fprintf(stderr, "BPF program loaded, reading total_bw values\n");
if (fetch_verify_total_bw(loaded_bw, test_ctx->nr_cpus) < 0) {
SCX_ERR("Failed to get stable values with BPF loaded");
bpf_link__destroy(link);
return SCX_TEST_FAIL;
}
bpf_link__destroy(link);
/* Test scenario 3: BPF program unloaded */
fprintf(stderr, "BPF program unloaded, reading total_bw values\n");
if (fetch_verify_total_bw(unloaded_bw, test_ctx->nr_cpus) < 0) {
SCX_ERR("Failed to get stable values after BPF unload");
return SCX_TEST_FAIL;
}
/* Verify all three scenarios have the same total_bw values */
for (i = 0; i < test_ctx->nr_cpus; i++) {
if (test_ctx->baseline_bw[i] != loaded_bw[i]) {
SCX_ERR("CPU%d: baseline_bw=%ld != loaded_bw=%ld",
i, test_ctx->baseline_bw[i], loaded_bw[i]);
return SCX_TEST_FAIL;
}
if (test_ctx->baseline_bw[i] != unloaded_bw[i]) {
SCX_ERR("CPU%d: baseline_bw=%ld != unloaded_bw=%ld",
i, test_ctx->baseline_bw[i], unloaded_bw[i]);
return SCX_TEST_FAIL;
}
}
fprintf(stderr, "All total_bw values are consistent across all scenarios\n");
/*
* Validate auto-register/unregister of dl_server bandwidth reservations.
*
* Doubling fair_server's runtime doubles its bw contribution. With a
* full-mode BPF scheduler (minimal_ops), the kernel should detach
* fair_server and attach ext_server, dropping total_bw back to its
* pre-customization (default ext_server-only) value. On unload, the
* fair_server reservation should come back with its customized runtime
* preserved, so total_bw doubles again.
*/
if (read_fair_runtime_all(test_ctx->nr_cpus, original_runtime) < 0) {
fprintf(stderr, "Skipping attach/detach validation: debugfs not accessible\n");
return SCX_TEST_PASS;
}
doubled_runtime = original_runtime[0] * 2;
fprintf(stderr,
"Setting fair_server runtime to %ld ns on all CPUs (orig %ld)\n",
doubled_runtime, original_runtime[0]);
if (write_fair_runtime_all(test_ctx->nr_cpus, doubled_runtime) < 0) {
ret = SCX_TEST_FAIL;
goto restore;
}
if (fetch_verify_total_bw(doubled_bw, test_ctx->nr_cpus) < 0) {
SCX_ERR("Failed to get stable values after doubling fair runtime");
ret = SCX_TEST_FAIL;
goto restore;
}
/*
* After doubling the runtime, fair_server's bw contribution must grow.
* We don't assert exactly 2x, because the kernel's to_ratio() truncates
* the value, so 2 * to_ratio(period, runtime) and
* to_ratio(period, 2 * runtime) can differ.
*/
for (i = 0; i < test_ctx->nr_cpus; i++) {
if (doubled_bw[i] <= test_ctx->baseline_bw[i]) {
SCX_ERR("CPU%d: fair did not increase total_bw (baseline=%ld, doubled=%ld)",
i, test_ctx->baseline_bw[i], doubled_bw[i]);
ret = SCX_TEST_FAIL;
goto restore;
}
}
link = bpf_map__attach_struct_ops(test_ctx->skel->maps.minimal_ops);
if (!link) {
SCX_ERR("Failed to attach scheduler for detach test");
ret = SCX_TEST_FAIL;
goto restore;
}
if (fetch_verify_total_bw(loaded_bw, test_ctx->nr_cpus) < 0) {
SCX_ERR("Failed to get stable values with BPF loaded (detach test)");
bpf_link__destroy(link);
ret = SCX_TEST_FAIL;
goto restore;
}
/*
* In full mode the customized fair_server is detached and ext_server is
* attached at its default runtime, total_bw must match baseline.
*/
for (i = 0; i < test_ctx->nr_cpus; i++) {
if (loaded_bw[i] != test_ctx->baseline_bw[i]) {
SCX_ERR("CPU%d: expected bw %ld (fair detached, ext default), got %ld",
i, test_ctx->baseline_bw[i], loaded_bw[i]);
bpf_link__destroy(link);
ret = SCX_TEST_FAIL;
goto restore;
}
}
bpf_link__destroy(link);
if (fetch_verify_total_bw(unloaded_bw, test_ctx->nr_cpus) < 0) {
SCX_ERR("Failed to get stable values after BPF unload (detach test)");
ret = SCX_TEST_FAIL;
goto restore;
}
/*
* After unload, fair_server is re-attached with its preserved 2x
* runtime, so total_bw should return to the doubled value.
*/
for (i = 0; i < test_ctx->nr_cpus; i++) {
if (unloaded_bw[i] != doubled_bw[i]) {
SCX_ERR("CPU%d: BPF unloaded: expected %ld (fair restored at 2x), got %ld",
i, doubled_bw[i], unloaded_bw[i]);
ret = SCX_TEST_FAIL;
goto restore;
}
}
fprintf(stderr,
"dl_server attach/detach with customized fair runtime verified\n");
ret = SCX_TEST_PASS;
restore:
if (restore_fair_runtime_all(test_ctx->nr_cpus, original_runtime) < 0)
SCX_ERR("Failed to fully restore per-CPU fair_server runtimes");
return ret;
}
static void cleanup(void *ctx)
{
struct total_bw_ctx *test_ctx = ctx;
if (test_ctx) {
if (test_ctx->skel)
minimal__destroy(test_ctx->skel);
free(test_ctx);
}
}
struct scx_test total_bw = {
.name = "total_bw",
.description = "Verify total_bw consistency across BPF program states",
.setup = setup,
.run = run,
.cleanup = cleanup,
};
REGISTER_SCX_TEST(&total_bw)