1 // SPDX-License-Identifier: GPL-2.0 2 3 #define _GNU_SOURCE 4 #include <assert.h> 5 #include <linux/limits.h> 6 #include <pthread.h> 7 #include <sched.h> 8 #include <signal.h> 9 #include <sys/syscall.h> 10 #include <unistd.h> 11 12 #include "kselftest.h" 13 #include "cgroup_util.h" 14 15 static int idle_process_fn(const char *cgroup, void *arg) 16 { 17 (void)pause(); 18 return 0; 19 } 20 21 static int do_migration_fn(const char *cgroup, void *arg) 22 { 23 int object_pid = (int)(size_t)arg; 24 25 if (setuid(TEST_UID)) 26 return EXIT_FAILURE; 27 28 // XXX checking /proc/$pid/cgroup would be quicker than wait 29 if (cg_enter(cgroup, object_pid) || 30 cg_wait_for_proc_count(cgroup, 1)) 31 return EXIT_FAILURE; 32 33 return EXIT_SUCCESS; 34 } 35 36 static int do_controller_fn(const char *cgroup, void *arg) 37 { 38 const char *child = cgroup; 39 const char *parent = arg; 40 41 if (setuid(TEST_UID)) 42 return EXIT_FAILURE; 43 44 if (!cg_read_strstr(child, "cgroup.controllers", "cpuset")) 45 return EXIT_FAILURE; 46 47 if (cg_write(parent, "cgroup.subtree_control", "+cpuset")) 48 return EXIT_FAILURE; 49 50 if (cg_read_strstr(child, "cgroup.controllers", "cpuset")) 51 return EXIT_FAILURE; 52 53 if (cg_write(parent, "cgroup.subtree_control", "-cpuset")) 54 return EXIT_FAILURE; 55 56 if (!cg_read_strstr(child, "cgroup.controllers", "cpuset")) 57 return EXIT_FAILURE; 58 59 return EXIT_SUCCESS; 60 } 61 62 /* 63 * Migrate a process between two sibling cgroups. 64 * The success should only depend on the parent cgroup permissions and not the 65 * migrated process itself (cpuset controller is in place because it uses 66 * security_task_setscheduler() in cgroup v1). 67 * 68 * Deliberately don't set cpuset.cpus in children to avoid definining migration 69 * permissions between two different cpusets. 70 */ 71 static int test_cpuset_perms_object(const char *root, bool allow) 72 { 73 char *parent = NULL, *child_src = NULL, *child_dst = NULL; 74 char *parent_procs = NULL, *child_src_procs = NULL, *child_dst_procs = NULL; 75 const uid_t test_euid = TEST_UID; 76 int object_pid = 0; 77 int ret = KSFT_FAIL; 78 79 parent = cg_name(root, "cpuset_test_0"); 80 if (!parent) 81 goto cleanup; 82 parent_procs = cg_name(parent, "cgroup.procs"); 83 if (!parent_procs) 84 goto cleanup; 85 if (cg_create(parent)) 86 goto cleanup; 87 88 child_src = cg_name(parent, "cpuset_test_1"); 89 if (!child_src) 90 goto cleanup; 91 child_src_procs = cg_name(child_src, "cgroup.procs"); 92 if (!child_src_procs) 93 goto cleanup; 94 if (cg_create(child_src)) 95 goto cleanup; 96 97 child_dst = cg_name(parent, "cpuset_test_2"); 98 if (!child_dst) 99 goto cleanup; 100 child_dst_procs = cg_name(child_dst, "cgroup.procs"); 101 if (!child_dst_procs) 102 goto cleanup; 103 if (cg_create(child_dst)) 104 goto cleanup; 105 106 if (cg_write(parent, "cgroup.subtree_control", "+cpuset")) 107 goto cleanup; 108 109 if (cg_read_strstr(child_src, "cgroup.controllers", "cpuset") || 110 cg_read_strstr(child_dst, "cgroup.controllers", "cpuset")) 111 goto cleanup; 112 113 /* Enable permissions along src->dst tree path */ 114 if (chown(child_src_procs, test_euid, -1) || 115 chown(child_dst_procs, test_euid, -1)) 116 goto cleanup; 117 118 if (allow && chown(parent_procs, test_euid, -1)) 119 goto cleanup; 120 121 /* Fork a privileged child as a test object */ 122 object_pid = cg_run_nowait(child_src, idle_process_fn, NULL); 123 if (object_pid < 0) 124 goto cleanup; 125 126 /* Carry out migration in a child process that can drop all privileges 127 * (including capabilities), the main process must remain privileged for 128 * cleanup. 129 * Child process's cgroup is irrelevant but we place it into child_dst 130 * as hacky way to pass information about migration target to the child. 131 */ 132 if (allow ^ (cg_run(child_dst, do_migration_fn, (void *)(size_t)object_pid) == EXIT_SUCCESS)) 133 goto cleanup; 134 135 ret = KSFT_PASS; 136 137 cleanup: 138 if (object_pid > 0) { 139 (void)kill(object_pid, SIGTERM); 140 (void)clone_reap(object_pid, WEXITED); 141 } 142 143 cg_destroy(child_dst); 144 free(child_dst_procs); 145 free(child_dst); 146 147 cg_destroy(child_src); 148 free(child_src_procs); 149 free(child_src); 150 151 cg_destroy(parent); 152 free(parent_procs); 153 free(parent); 154 155 return ret; 156 } 157 158 static int test_cpuset_perms_object_allow(const char *root) 159 { 160 return test_cpuset_perms_object(root, true); 161 } 162 163 static int test_cpuset_perms_object_deny(const char *root) 164 { 165 return test_cpuset_perms_object(root, false); 166 } 167 168 /* 169 * Migrate a process between parent and child implicitely 170 * Implicit migration happens when a controller is enabled/disabled. 171 * 172 */ 173 static int test_cpuset_perms_subtree(const char *root) 174 { 175 char *parent = NULL, *child = NULL; 176 char *parent_procs = NULL, *parent_subctl = NULL, *child_procs = NULL; 177 const uid_t test_euid = TEST_UID; 178 int object_pid = 0; 179 int ret = KSFT_FAIL; 180 181 parent = cg_name(root, "cpuset_test_0"); 182 if (!parent) 183 goto cleanup; 184 parent_procs = cg_name(parent, "cgroup.procs"); 185 if (!parent_procs) 186 goto cleanup; 187 parent_subctl = cg_name(parent, "cgroup.subtree_control"); 188 if (!parent_subctl) 189 goto cleanup; 190 if (cg_create(parent)) 191 goto cleanup; 192 193 child = cg_name(parent, "cpuset_test_1"); 194 if (!child) 195 goto cleanup; 196 child_procs = cg_name(child, "cgroup.procs"); 197 if (!child_procs) 198 goto cleanup; 199 if (cg_create(child)) 200 goto cleanup; 201 202 /* Enable permissions as in a delegated subtree */ 203 if (chown(parent_procs, test_euid, -1) || 204 chown(parent_subctl, test_euid, -1) || 205 chown(child_procs, test_euid, -1)) 206 goto cleanup; 207 208 /* Put a privileged child in the subtree and modify controller state 209 * from an unprivileged process, the main process remains privileged 210 * for cleanup. 211 * The unprivileged child runs in subtree too to avoid parent and 212 * internal-node constraing violation. 213 */ 214 object_pid = cg_run_nowait(child, idle_process_fn, NULL); 215 if (object_pid < 0) 216 goto cleanup; 217 218 if (cg_run(child, do_controller_fn, parent) != EXIT_SUCCESS) 219 goto cleanup; 220 221 ret = KSFT_PASS; 222 223 cleanup: 224 if (object_pid > 0) { 225 (void)kill(object_pid, SIGTERM); 226 (void)clone_reap(object_pid, WEXITED); 227 } 228 229 cg_destroy(child); 230 free(child_procs); 231 free(child); 232 233 cg_destroy(parent); 234 free(parent_subctl); 235 free(parent_procs); 236 free(parent); 237 238 return ret; 239 } 240 241 static int get_cpu_affinity(cpu_set_t *mask) 242 { 243 CPU_ZERO(mask); 244 return sched_getaffinity(0, sizeof(*mask), mask); 245 } 246 247 static int cpu_set_equal(cpu_set_t *dst, unsigned long mask) 248 { 249 cpu_set_t expected; 250 251 CPU_ZERO(&expected); 252 assert(sizeof(mask) < CPU_SETSIZE); 253 254 for (int cpu = 0; cpu < sizeof(mask) * 8; ++cpu) 255 if ((1UL << cpu) & mask) 256 CPU_SET(cpu, &expected); 257 258 return CPU_EQUAL(&expected, dst); 259 } 260 261 enum test_phase { 262 AFFINITY_SETUP, 263 AFFINITY_CONTROLLER_DISABLED, 264 AFFINITY_COMPLETE, 265 AFFINITY_ERROR 266 }; 267 268 struct thread_args { 269 const char *cgroup; 270 cpu_set_t *affinity_before; 271 cpu_set_t *affinity_after; 272 int affinity_before_ready; 273 }; 274 275 static pthread_mutex_t test_mutex = PTHREAD_MUTEX_INITIALIZER; 276 static pthread_cond_t test_cond = PTHREAD_COND_INITIALIZER; 277 static enum test_phase test_phase; 278 279 static void *affinity_thread_fn(void *arg) 280 { 281 struct thread_args *args = (struct thread_args *)arg; 282 283 if (cg_enter_current_thread(args->cgroup)) 284 goto fail; 285 286 if (get_cpu_affinity(args->affinity_before) != 0) 287 goto fail; 288 289 pthread_mutex_lock(&test_mutex); 290 args->affinity_before_ready = 1; 291 pthread_cond_broadcast(&test_cond); 292 293 while (test_phase < AFFINITY_CONTROLLER_DISABLED) 294 pthread_cond_wait(&test_cond, &test_mutex); 295 pthread_mutex_unlock(&test_mutex); 296 297 if (get_cpu_affinity(args->affinity_after) != 0) 298 goto fail; 299 300 301 return NULL; 302 303 fail: 304 pthread_mutex_lock(&test_mutex); 305 test_phase = AFFINITY_ERROR; 306 pthread_cond_broadcast(&test_cond); 307 pthread_mutex_unlock(&test_mutex); 308 return NULL; 309 } 310 311 /* 312 * Test that disabling cpuset controller properly updates thread affinity. 313 * 314 * This test exposes a bug in cpuset_attach() where threads in child cgroups 315 * don't get their affinity updated when the cpuset controller is disabled. 316 * 317 * Setup: 318 * - Create parent cgroup with cpuset.cpus=0-1 319 * - Create child A with cpuset.cpus=0-1 320 * - Create child B with cpuset.cpus=1 321 * - Place multithreaded process: group leader + thread_a in A, thread_b in B 322 * - Disable cpuset controller on parent 323 * 324 * Expected: thread_b's affinity should expand from {1} to {0-1} 325 * Buggy: thread_b's affinity remains {1} 326 */ 327 static int test_cpuset_affinity_on_controller_disable(const char *root) 328 { 329 char *parent = NULL, *child_a = NULL, *child_b = NULL; 330 pthread_t thread_a, thread_b; 331 int thread_a_created = 0, thread_b_created = 0; 332 cpu_set_t affinity_a_before, affinity_a_after; 333 cpu_set_t affinity_b_before, affinity_b_after; 334 int ret = KSFT_FAIL; 335 336 parent = cg_name(root, "cpuset_affinity_test"); 337 if (!parent) 338 goto cleanup; 339 if (cg_create(parent)) 340 goto cleanup; 341 if (cg_write(parent, "cgroup.type", "threaded")) 342 goto cleanup; 343 344 child_a = cg_name(parent, "A"); 345 if (!child_a) 346 goto cleanup; 347 if (cg_create(child_a)) 348 goto cleanup; 349 if (cg_write(child_a, "cgroup.type", "threaded")) 350 goto cleanup; 351 352 child_b = cg_name(parent, "B"); 353 if (!child_b) 354 goto cleanup; 355 if (cg_create(child_b)) 356 goto cleanup; 357 if (cg_write(child_b, "cgroup.type", "threaded")) 358 goto cleanup; 359 360 /* Now enable cpuset controller in parent */ 361 if (cg_write(parent, "cgroup.subtree_control", "+cpuset")) 362 goto skip; 363 364 /* 365 * Set CPU affinity constraints 366 * Skip the test if the setting of "cpuset.cpus" fails as the test 367 * system may not have CPU 1. 368 */ 369 if (cg_write(parent, "cpuset.cpus", "0-1")) 370 goto skip; 371 if (cg_write(child_a, "cpuset.cpus", "0-1")) 372 goto skip; 373 if (cg_write(child_b, "cpuset.cpus", "1")) 374 goto skip; 375 376 /* Move group leader (main thread) to child A */ 377 if (cg_enter_current(child_a)) 378 goto cleanup; 379 380 /* Create threads - they will move themselves to their respective cgroups */ 381 test_phase = AFFINITY_SETUP; 382 383 struct thread_args args_a = { 384 .cgroup = child_a, 385 .affinity_before = &affinity_a_before, 386 .affinity_after = &affinity_a_after, 387 .affinity_before_ready = 0, 388 }; 389 if (pthread_create(&thread_a, NULL, affinity_thread_fn, &args_a)) 390 goto cleanup; 391 thread_a_created = 1; 392 393 struct thread_args args_b = { 394 .cgroup = child_b, 395 .affinity_before = &affinity_b_before, 396 .affinity_after = &affinity_b_after, 397 .affinity_before_ready = 0, 398 }; 399 if (pthread_create(&thread_b, NULL, affinity_thread_fn, &args_b)) 400 goto cleanup_threads; 401 thread_b_created = 1; 402 403 pthread_mutex_lock(&test_mutex); 404 while ((test_phase < AFFINITY_ERROR) && 405 (args_a.affinity_before_ready + args_b.affinity_before_ready < 2)) 406 pthread_cond_wait(&test_cond, &test_mutex); 407 408 /* If a thread failed during setup, bail out */ 409 if (test_phase == AFFINITY_ERROR) { 410 pthread_mutex_unlock(&test_mutex); 411 goto cleanup_threads; 412 } 413 pthread_mutex_unlock(&test_mutex); 414 415 if (!cpu_set_equal(&affinity_a_before, 0x3)) { 416 ksft_print_msg("FAIL: thread_a initial affinity incorrect\n"); 417 goto cleanup_threads; 418 } 419 420 if (!cpu_set_equal(&affinity_b_before, 0x2)) { 421 ksft_print_msg("FAIL: thread_b initial affinity incorrect\n"); 422 goto cleanup_threads; 423 } 424 425 /* Disable cpuset controller - this should trigger affinity update */ 426 if (cg_write(parent, "cgroup.subtree_control", "-cpuset")) 427 goto cleanup_threads; 428 429 /* Signal threads to save their final affinity and exit */ 430 pthread_mutex_lock(&test_mutex); 431 test_phase = AFFINITY_CONTROLLER_DISABLED; 432 pthread_cond_broadcast(&test_cond); 433 pthread_mutex_unlock(&test_mutex); 434 435 pthread_join(thread_a, NULL); 436 pthread_join(thread_b, NULL); 437 438 /* Verify thread affinities AFTER disabling controller */ 439 if (!cpu_set_equal(&affinity_a_after, 0x3)) { 440 ksft_print_msg("FAIL: thread_a final affinity incorrect\n"); 441 goto cleanup; 442 } 443 444 if (!cpu_set_equal(&affinity_b_after, 0x3)) { 445 ksft_print_msg("FAIL: thread_b affinity did not expand to {0-1}\n"); 446 goto cleanup; 447 } 448 449 ret = KSFT_PASS; 450 goto cleanup; 451 452 skip: 453 ret = KSFT_SKIP; 454 goto cleanup; 455 456 cleanup_threads: 457 pthread_mutex_lock(&test_mutex); 458 test_phase = AFFINITY_COMPLETE; 459 pthread_cond_broadcast(&test_cond); 460 pthread_mutex_unlock(&test_mutex); 461 462 if (thread_a_created) 463 pthread_join(thread_a, NULL); 464 if (thread_b_created) 465 pthread_join(thread_b, NULL); 466 467 cleanup: 468 /* Move back to root before cleanup */ 469 cg_enter_current(root); 470 471 cg_destroy(child_b); 472 free(child_b); 473 cg_destroy(child_a); 474 free(child_a); 475 cg_destroy(parent); 476 free(parent); 477 478 return ret; 479 } 480 481 482 #define T(x) { x, #x } 483 struct cpuset_test { 484 int (*fn)(const char *root); 485 const char *name; 486 } tests[] = { 487 T(test_cpuset_perms_object_allow), 488 T(test_cpuset_perms_object_deny), 489 T(test_cpuset_perms_subtree), 490 T(test_cpuset_affinity_on_controller_disable), 491 }; 492 #undef T 493 494 int main(int argc, char *argv[]) 495 { 496 char root[PATH_MAX]; 497 int i; 498 499 ksft_print_header(); 500 if (cg_find_unified_root(root, sizeof(root), NULL)) 501 ksft_exit_skip("cgroup v2 isn't mounted\n"); 502 503 if (cg_read_strstr(root, "cgroup.subtree_control", "cpuset")) 504 if (cg_write(root, "cgroup.subtree_control", "+cpuset")) 505 ksft_exit_skip("Failed to set cpuset controller\n"); 506 507 ksft_set_plan(ARRAY_SIZE(tests)); 508 for (i = 0; i < ARRAY_SIZE(tests); i++) { 509 switch (tests[i].fn(root)) { 510 case KSFT_PASS: 511 ksft_test_result_pass("%s\n", tests[i].name); 512 break; 513 case KSFT_SKIP: 514 ksft_test_result_skip("%s\n", tests[i].name); 515 break; 516 default: 517 ksft_test_result_fail("%s\n", tests[i].name); 518 break; 519 } 520 } 521 522 ksft_finished(); 523 } 524