1 /* $NetBSD: kern_mutex.c,v 1.113 2026/08/16 21:52:42 riastradh Exp $ */ 2 3 /*- 4 * Copyright (c) 2002, 2006, 2007, 2008, 2019, 2023 5 * The NetBSD Foundation, Inc. 6 * All rights reserved. 7 * 8 * This code is derived from software contributed to The NetBSD Foundation 9 * by Jason R. Thorpe and Andrew Doran. 10 * 11 * Redistribution and use in source and binary forms, with or without 12 * modification, are permitted provided that the following conditions 13 * are met: 14 * 1. Redistributions of source code must retain the above copyright 15 * notice, this list of conditions and the following disclaimer. 16 * 2. Redistributions in binary form must reproduce the above copyright 17 * notice, this list of conditions and the following disclaimer in the 18 * documentation and/or other materials provided with the distribution. 19 * 20 * THIS SOFTWARE IS PROVIDED BY THE NETBSD FOUNDATION, INC. AND CONTRIBUTORS 21 * ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED 22 * TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR 23 * PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR CONTRIBUTORS 24 * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR 25 * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF 26 * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS 27 * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN 28 * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) 29 * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE 30 * POSSIBILITY OF SUCH DAMAGE. 31 */ 32 33 /* 34 * Kernel mutex implementation, modeled after those found in Solaris, 35 * a description of which can be found in: 36 * 37 * Solaris Internals: Core Kernel Architecture, Jim Mauro and 38 * Richard McDougall. 39 */ 40 41 #define __MUTEX_PRIVATE 42 43 #include <sys/cdefs.h> 44 __KERNEL_RCSID(0, "$NetBSD: kern_mutex.c,v 1.113 2026/08/16 21:52:42 riastradh Exp $"); 45 46 #include <sys/param.h> 47 48 #include <sys/atomic.h> 49 #include <sys/cpu.h> 50 #include <sys/intr.h> 51 #include <sys/kernel.h> 52 #include <sys/lock.h> 53 #include <sys/lockdebug.h> 54 #include <sys/mutex.h> 55 #include <sys/proc.h> 56 #include <sys/pserialize.h> 57 #include <sys/sched.h> 58 #include <sys/sleepq.h> 59 #include <sys/syncobj.h> 60 #include <sys/systm.h> 61 #include <sys/types.h> 62 63 #include <dev/lockstat.h> 64 65 #include <machine/lock.h> 66 67 /* 68 * When not running a debug kernel, spin mutexes are not much 69 * more than an splraiseipl() and splx() pair. 70 */ 71 72 #if defined(DIAGNOSTIC) || defined(MULTIPROCESSOR) || defined(LOCKDEBUG) 73 #define FULL 74 #endif 75 76 /* 77 * Debugging support. 78 */ 79 80 #define MUTEX_WANTLOCK(mtx, owantedp) \ 81 LOCKDEBUG_WANTLOCK(MUTEX_DEBUG_P(mtx), (mtx), \ 82 (uintptr_t)__builtin_return_address(0), 0, owantedp) 83 #define MUTEX_TESTLOCK(mtx) \ 84 LOCKDEBUG_WANTLOCK(MUTEX_DEBUG_P(mtx), (mtx), \ 85 (uintptr_t)__builtin_return_address(0), -1, NULL) 86 #define MUTEX_LOCKED(mtx, owantedp) \ 87 LOCKDEBUG_LOCKED(MUTEX_DEBUG_P(mtx), (mtx), NULL, \ 88 (uintptr_t)__builtin_return_address(0), 0, owantedp) 89 #define MUTEX_UNLOCKED(mtx) \ 90 LOCKDEBUG_UNLOCKED(MUTEX_DEBUG_P(mtx), (mtx), \ 91 (uintptr_t)__builtin_return_address(0), 0) 92 #define MUTEX_ABORT(mtx, msg) \ 93 mutex_abort(__func__, __LINE__, mtx, msg) 94 95 #if defined(LOCKDEBUG) 96 97 #define MUTEX_DASSERT(mtx, cond) \ 98 do { \ 99 if (__predict_false(!(cond))) \ 100 MUTEX_ABORT(mtx, "assertion failed: " #cond); \ 101 } while (/* CONSTCOND */ 0) 102 103 #else /* LOCKDEBUG */ 104 105 #define MUTEX_DASSERT(mtx, cond) /* nothing */ 106 107 #endif /* LOCKDEBUG */ 108 109 #if defined(DIAGNOSTIC) 110 111 #define MUTEX_ASSERT(mtx, cond) \ 112 do { \ 113 if (__predict_false(!(cond))) \ 114 MUTEX_ABORT(mtx, "assertion failed: " #cond); \ 115 } while (/* CONSTCOND */ 0) 116 117 #else /* DIAGNOSTIC */ 118 119 #define MUTEX_ASSERT(mtx, cond) /* nothing */ 120 121 #endif /* DIAGNOSTIC */ 122 123 /* 124 * Some architectures can't use __cpu_simple_lock as is so allow a way 125 * for them to use an alternate definition. 126 */ 127 #ifndef MUTEX_SPINBIT_LOCK_INIT 128 #define MUTEX_SPINBIT_LOCK_INIT(mtx) __cpu_simple_lock_init(&(mtx)->mtx_lock) 129 #endif 130 #ifndef MUTEX_SPINBIT_LOCKED_P 131 #define MUTEX_SPINBIT_LOCKED_P(mtx) __SIMPLELOCK_LOCKED_P(&(mtx)->mtx_lock) 132 #endif 133 #ifndef MUTEX_SPINBIT_LOCK_TRY 134 #define MUTEX_SPINBIT_LOCK_TRY(mtx) __cpu_simple_lock_try(&(mtx)->mtx_lock) 135 #endif 136 #ifndef MUTEX_SPINBIT_LOCK_UNLOCK 137 #define MUTEX_SPINBIT_LOCK_UNLOCK(mtx) __cpu_simple_unlock(&(mtx)->mtx_lock) 138 #endif 139 140 #ifndef MUTEX_INITIALIZE_SPIN_IPL 141 #define MUTEX_INITIALIZE_SPIN_IPL(mtx, ipl) \ 142 ((mtx)->mtx_ipl = makeiplcookie((ipl))) 143 #endif 144 145 /* 146 * Spin mutex SPL save / restore. 147 */ 148 149 #define MUTEX_SPIN_SPLRAISE(mtx) \ 150 do { \ 151 const int s = splraiseipl(MUTEX_SPIN_IPL(mtx)); \ 152 struct cpu_info * const x__ci = curcpu(); \ 153 const int x__cnt = x__ci->ci_mtx_count--; \ 154 __insn_barrier(); \ 155 if (x__cnt == 0) \ 156 x__ci->ci_mtx_oldspl = s; \ 157 } while (/* CONSTCOND */ 0) 158 159 #define MUTEX_SPIN_SPLRESTORE(mtx) \ 160 do { \ 161 struct cpu_info * const x__ci = curcpu(); \ 162 const int s = x__ci->ci_mtx_oldspl; \ 163 __insn_barrier(); \ 164 if (++(x__ci->ci_mtx_count) == 0) \ 165 splx(s); \ 166 } while (/* CONSTCOND */ 0) 167 168 /* 169 * Memory barriers. 170 */ 171 #ifdef __HAVE_ATOMIC_AS_MEMBAR 172 #define MUTEX_MEMBAR_ENTER() 173 #else 174 #define MUTEX_MEMBAR_ENTER() membar_enter() 175 #endif 176 177 /* 178 * For architectures that provide 'simple' mutexes: they provide a 179 * CAS function that is either MP-safe, or does not need to be MP 180 * safe. Adaptive mutexes on these architectures do not require an 181 * additional interlock. 182 */ 183 184 #ifdef __HAVE_SIMPLE_MUTEXES 185 186 #define MUTEX_OWNER(owner) \ 187 (owner & MUTEX_THREAD) 188 #define MUTEX_HAS_WAITERS(mtx) \ 189 (((int)(mtx)->mtx_owner & MUTEX_BIT_WAITERS) != 0) 190 191 #define MUTEX_INITIALIZE_ADAPTIVE(mtx, dodebug) \ 192 do { \ 193 if (!dodebug) \ 194 (mtx)->mtx_owner |= MUTEX_BIT_NODEBUG; \ 195 } while (/* CONSTCOND */ 0) 196 197 #define MUTEX_INITIALIZE_SPIN(mtx, dodebug, ipl) \ 198 do { \ 199 (mtx)->mtx_owner = MUTEX_BIT_SPIN; \ 200 if (!dodebug) \ 201 (mtx)->mtx_owner |= MUTEX_BIT_NODEBUG; \ 202 MUTEX_INITIALIZE_SPIN_IPL((mtx), (ipl)); \ 203 MUTEX_SPINBIT_LOCK_INIT((mtx)); \ 204 } while (/* CONSTCOND */ 0) 205 206 #define MUTEX_DESTROY(mtx) \ 207 do { \ 208 (mtx)->mtx_owner = MUTEX_THREAD; \ 209 } while (/* CONSTCOND */ 0) 210 211 #define MUTEX_SPIN_P(owner) \ 212 (((owner) & MUTEX_BIT_SPIN) != 0) 213 #define MUTEX_ADAPTIVE_P(owner) \ 214 (((owner) & MUTEX_BIT_SPIN) == 0) 215 216 #ifndef MUTEX_CAS 217 #define MUTEX_CAS(p, o, n) \ 218 (atomic_cas_ulong((volatile unsigned long *)(p), (o), (n)) == (o)) 219 #endif /* MUTEX_CAS */ 220 221 #define MUTEX_DEBUG_P(mtx) (((mtx)->mtx_owner & MUTEX_BIT_NODEBUG) == 0) 222 #if defined(LOCKDEBUG) 223 #define MUTEX_OWNED(owner) (((owner) & ~MUTEX_BIT_NODEBUG) != 0) 224 #define MUTEX_INHERITDEBUG(n, o) (n) |= (o) & MUTEX_BIT_NODEBUG 225 #else /* defined(LOCKDEBUG) */ 226 #define MUTEX_OWNED(owner) ((owner) != 0) 227 #define MUTEX_INHERITDEBUG(n, o) /* nothing */ 228 #endif /* defined(LOCKDEBUG) */ 229 230 static inline int 231 MUTEX_ACQUIRE(kmutex_t *mtx, uintptr_t curthread) 232 { 233 int rv; 234 uintptr_t oldown = 0; 235 uintptr_t newown = curthread; 236 237 MUTEX_INHERITDEBUG(oldown, mtx->mtx_owner); 238 MUTEX_INHERITDEBUG(newown, oldown); 239 rv = MUTEX_CAS(&mtx->mtx_owner, oldown, newown); 240 membar_acquire(); 241 return rv; 242 } 243 244 static inline int 245 MUTEX_SET_WAITERS(kmutex_t *mtx, uintptr_t owner) 246 { 247 int rv; 248 249 rv = MUTEX_CAS(&mtx->mtx_owner, owner, owner | MUTEX_BIT_WAITERS); 250 MUTEX_MEMBAR_ENTER(); 251 return rv; 252 } 253 254 static inline void 255 MUTEX_RELEASE(kmutex_t *mtx) 256 { 257 uintptr_t newown; 258 259 newown = 0; 260 MUTEX_INHERITDEBUG(newown, mtx->mtx_owner); 261 atomic_store_release(&mtx->mtx_owner, newown); 262 } 263 #endif /* __HAVE_SIMPLE_MUTEXES */ 264 265 /* 266 * Patch in stubs via strong alias where they are not available. 267 */ 268 269 #if defined(LOCKDEBUG) 270 #undef __HAVE_MUTEX_STUBS 271 #undef __HAVE_SPIN_MUTEX_STUBS 272 #endif 273 274 #ifndef __HAVE_MUTEX_STUBS 275 __strong_alias(mutex_enter,mutex_vector_enter); 276 __strong_alias(mutex_exit,mutex_vector_exit); 277 #endif 278 279 #ifndef __HAVE_SPIN_MUTEX_STUBS 280 __strong_alias(mutex_spin_enter,mutex_vector_enter); 281 __strong_alias(mutex_spin_exit,mutex_vector_exit); 282 #endif 283 284 static void mutex_abort(const char *, size_t, volatile const kmutex_t *, 285 const char *); 286 static void mutex_dump(const volatile void *, lockop_printer_t); 287 static lwp_t *mutex_owner(wchan_t); 288 289 lockops_t mutex_spin_lockops = { 290 .lo_name = "Mutex", 291 .lo_type = LOCKOPS_SPIN, 292 .lo_dump = mutex_dump, 293 }; 294 295 lockops_t mutex_adaptive_lockops = { 296 .lo_name = "Mutex", 297 .lo_type = LOCKOPS_SLEEP, 298 .lo_dump = mutex_dump, 299 }; 300 301 syncobj_t mutex_syncobj = { 302 .sobj_name = "mutex", 303 .sobj_flag = SOBJ_SLEEPQ_SORTED, 304 .sobj_boostpri = PRI_KERNEL, 305 .sobj_unsleep = turnstile_unsleep, 306 .sobj_changepri = turnstile_changepri, 307 .sobj_lendpri = sleepq_lendpri, 308 .sobj_owner = mutex_owner, 309 }; 310 311 /* 312 * mutex_dump: 313 * 314 * Dump the contents of a mutex structure. 315 */ 316 static void 317 mutex_dump(const volatile void *cookie, lockop_printer_t pr) 318 { 319 const volatile kmutex_t *mtx = cookie; 320 uintptr_t owner = mtx->mtx_owner; 321 322 pr("owner field : %#018lx wait/spin: %16d/%d\n", 323 (long)MUTEX_OWNER(owner), MUTEX_HAS_WAITERS(mtx), 324 MUTEX_SPIN_P(owner)); 325 } 326 327 /* 328 * mutex_abort: 329 * 330 * Dump information about an error and panic the system. This 331 * generates a lot of machine code in the DIAGNOSTIC case, so 332 * we ask the compiler to not inline it. 333 */ 334 static void __noinline 335 mutex_abort(const char *func, size_t line, volatile const kmutex_t *mtx, 336 const char *msg) 337 { 338 339 LOCKDEBUG_ABORT(func, line, mtx, (MUTEX_SPIN_P(mtx->mtx_owner) ? 340 &mutex_spin_lockops : &mutex_adaptive_lockops), msg); 341 } 342 343 /* 344 * mutex_init: 345 * 346 * Initialize a mutex for use. Note that adaptive mutexes are in 347 * essence spin mutexes that can sleep to avoid deadlock and wasting 348 * CPU time. We can't easily provide a type of mutex that always 349 * sleeps - see comments in mutex_vector_enter() about releasing 350 * mutexes unlocked. 351 */ 352 void 353 _mutex_init(kmutex_t *mtx, kmutex_type_t type, int ipl, 354 uintptr_t return_address) 355 { 356 lockops_t *lockops __unused; 357 bool dodebug; 358 359 memset(mtx, 0, sizeof(*mtx)); 360 361 if (ipl == IPL_NONE || ipl == IPL_SOFTCLOCK || 362 ipl == IPL_SOFTBIO || ipl == IPL_SOFTNET || 363 ipl == IPL_SOFTSERIAL) { 364 lockops = (type == MUTEX_NODEBUG ? 365 NULL : &mutex_adaptive_lockops); 366 dodebug = LOCKDEBUG_ALLOC(mtx, lockops, return_address); 367 MUTEX_INITIALIZE_ADAPTIVE(mtx, dodebug); 368 } else { 369 lockops = (type == MUTEX_NODEBUG ? 370 NULL : &mutex_spin_lockops); 371 dodebug = LOCKDEBUG_ALLOC(mtx, lockops, return_address); 372 MUTEX_INITIALIZE_SPIN(mtx, dodebug, ipl); 373 } 374 } 375 376 void 377 mutex_init(kmutex_t *mtx, kmutex_type_t type, int ipl) 378 { 379 380 _mutex_init(mtx, type, ipl, (uintptr_t)__builtin_return_address(0)); 381 } 382 383 /* 384 * mutex_destroy: 385 * 386 * Tear down a mutex. 387 */ 388 void 389 mutex_destroy(kmutex_t *mtx) 390 { 391 uintptr_t owner = mtx->mtx_owner; 392 393 if (MUTEX_ADAPTIVE_P(owner)) { 394 MUTEX_ASSERT(mtx, !MUTEX_OWNED(owner)); 395 MUTEX_ASSERT(mtx, !MUTEX_HAS_WAITERS(mtx)); 396 } else { 397 MUTEX_ASSERT(mtx, !MUTEX_SPINBIT_LOCKED_P(mtx)); 398 } 399 400 LOCKDEBUG_FREE(MUTEX_DEBUG_P(mtx), mtx); 401 MUTEX_DESTROY(mtx); 402 } 403 404 #ifdef MULTIPROCESSOR 405 /* 406 * mutex_oncpu: 407 * 408 * Return true if an adaptive mutex owner is running on a CPU in the 409 * system. If the target is waiting on the kernel big lock, then we 410 * must release it. This is necessary to avoid deadlock. 411 */ 412 static bool 413 mutex_oncpu(uintptr_t owner) 414 { 415 struct cpu_info *ci; 416 lwp_t *l; 417 418 KASSERT(kpreempt_disabled()); 419 420 if (!MUTEX_OWNED(owner)) { 421 return false; 422 } 423 424 /* 425 * See lwp_dtor() why dereference of the LWP pointer is safe. 426 * We must have kernel preemption disabled for that. 427 */ 428 l = (lwp_t *)MUTEX_OWNER(owner); 429 ci = l->l_cpu; 430 431 if (ci && ci->ci_curlwp == l) { 432 /* Target is running; do we need to block? */ 433 return (atomic_load_relaxed(&ci->ci_biglock_wanted) != l); 434 } 435 436 /* Not running. It may be safe to block now. */ 437 return false; 438 } 439 #endif /* MULTIPROCESSOR */ 440 441 /* 442 * mutex_vector_enter: 443 * 444 * Support routine for mutex_enter() that must handle all cases. In 445 * the LOCKDEBUG case, mutex_enter() is always aliased here, even if 446 * fast-path stubs are available. If a mutex_spin_enter() stub is 447 * not available, then it is also aliased directly here. 448 */ 449 void 450 mutex_vector_enter(kmutex_t *mtx) 451 { 452 uintptr_t owner, curthread; 453 turnstile_t *ts; 454 #ifdef MULTIPROCESSOR 455 u_int count; 456 #endif 457 volatile void *owanted; 458 LOCKSTAT_COUNTER(spincnt); 459 LOCKSTAT_COUNTER(slpcnt); 460 LOCKSTAT_TIMER(spintime); 461 LOCKSTAT_TIMER(slptime); 462 LOCKSTAT_FLAG(lsflag); 463 464 /* 465 * Handle spin mutexes. 466 */ 467 KPREEMPT_DISABLE(curlwp); 468 owner = mtx->mtx_owner; 469 if (MUTEX_SPIN_P(owner)) { 470 #if defined(LOCKDEBUG) && defined(MULTIPROCESSOR) 471 u_int spins = 0; 472 #endif 473 KPREEMPT_ENABLE(curlwp); 474 MUTEX_SPIN_SPLRAISE(mtx); 475 MUTEX_WANTLOCK(mtx, &owanted); 476 #ifdef FULL 477 if (MUTEX_SPINBIT_LOCK_TRY(mtx)) { 478 MUTEX_LOCKED(mtx, &owanted); 479 return; 480 } 481 #if !defined(MULTIPROCESSOR) 482 MUTEX_ABORT(mtx, "locking against myself"); 483 #else /* !MULTIPROCESSOR */ 484 485 LOCKSTAT_ENTER(lsflag); 486 LOCKSTAT_START_TIMER(lsflag, spintime); 487 count = SPINLOCK_BACKOFF_MIN; 488 489 /* 490 * Spin testing the lock word and do exponential backoff 491 * to reduce cache line ping-ponging between CPUs. 492 */ 493 do { 494 while (MUTEX_SPINBIT_LOCKED_P(mtx)) { 495 SPINLOCK_SPIN_HOOK; 496 SPINLOCK_BACKOFF(count); 497 #ifdef LOCKDEBUG 498 if (SPINLOCK_SPINOUT(spins)) 499 MUTEX_ABORT(mtx, "spinout"); 500 #endif /* LOCKDEBUG */ 501 } 502 } while (!MUTEX_SPINBIT_LOCK_TRY(mtx)); 503 504 if (count != SPINLOCK_BACKOFF_MIN) { 505 LOCKSTAT_STOP_TIMER(lsflag, spintime); 506 LOCKSTAT_EVENT(lsflag, mtx, 507 LB_SPIN_MUTEX | LB_SPIN, 1, spintime); 508 } 509 LOCKSTAT_EXIT(lsflag); 510 #endif /* !MULTIPROCESSOR */ 511 #endif /* FULL */ 512 MUTEX_LOCKED(mtx, &owanted); 513 return; 514 } 515 516 curthread = (uintptr_t)curlwp; 517 518 MUTEX_DASSERT(mtx, MUTEX_ADAPTIVE_P(owner)); 519 MUTEX_ASSERT(mtx, curthread != 0); 520 MUTEX_ASSERT(mtx, !cpu_intr_p()); 521 MUTEX_WANTLOCK(mtx, &owanted); 522 523 if (__predict_true(panicstr == NULL)) { 524 KDASSERT(pserialize_not_in_read_section()); 525 LOCKDEBUG_BARRIER(&kernel_lock, 1); 526 } 527 528 LOCKSTAT_ENTER(lsflag); 529 530 /* 531 * Adaptive mutex; spin trying to acquire the mutex. If we 532 * determine that the owner is not running on a processor, 533 * then we stop spinning, and sleep instead. 534 */ 535 for (;;) { 536 if (!MUTEX_OWNED(owner)) { 537 /* 538 * Mutex owner clear could mean two things: 539 * 540 * * The mutex has been released. 541 * * The owner field hasn't been set yet. 542 * 543 * Try to acquire it again. If that fails, 544 * we'll just loop again. 545 */ 546 if (MUTEX_ACQUIRE(mtx, curthread)) 547 break; 548 owner = mtx->mtx_owner; 549 continue; 550 } 551 if (__predict_false(MUTEX_OWNER(owner) == curthread)) { 552 MUTEX_ABORT(mtx, "locking against myself"); 553 } 554 #ifdef MULTIPROCESSOR 555 /* 556 * Check to see if the owner is running on a processor. 557 * If so, then we should just spin, as the owner will 558 * likely release the lock very soon. 559 */ 560 if (mutex_oncpu(owner)) { 561 LOCKSTAT_START_TIMER(lsflag, spintime); 562 count = SPINLOCK_BACKOFF_MIN; 563 do { 564 KPREEMPT_ENABLE(curlwp); 565 SPINLOCK_BACKOFF(count); 566 KPREEMPT_DISABLE(curlwp); 567 owner = mtx->mtx_owner; 568 } while (mutex_oncpu(owner)); 569 LOCKSTAT_STOP_TIMER(lsflag, spintime); 570 LOCKSTAT_COUNT(spincnt, 1); 571 if (!MUTEX_OWNED(owner)) 572 continue; 573 } 574 #endif 575 576 ts = turnstile_lookup(mtx); 577 578 /* 579 * Once we have the turnstile chain interlock, mark the 580 * mutex as having waiters. If that fails, spin again: 581 * chances are that the mutex has been released. 582 */ 583 if (!MUTEX_SET_WAITERS(mtx, owner)) { 584 turnstile_exit(mtx); 585 owner = mtx->mtx_owner; 586 continue; 587 } 588 589 #ifdef MULTIPROCESSOR 590 /* 591 * mutex_exit() is permitted to release the mutex without 592 * any interlocking instructions, and the following can 593 * occur as a result: 594 * 595 * CPU 1: MUTEX_SET_WAITERS() CPU2: mutex_exit() 596 * ---------------------------- ---------------------------- 597 * .. load mtx->mtx_owner 598 * .. see has-waiters bit clear 599 * set has-waiters bit .. 600 * .. store mtx->mtx_owner := 0 601 * return success 602 * 603 * There is another race that can occur: a third CPU could 604 * acquire the mutex as soon as it is released. Since 605 * adaptive mutexes are primarily spin mutexes, this is not 606 * something that we need to worry about too much. What we 607 * do need to ensure is that the waiters bit gets set. 608 * 609 * To allow the unlocked release, we need to make some 610 * assumptions here: 611 * 612 * o Release is the only non-atomic/unlocked operation 613 * that can be performed on the mutex. (It must still 614 * be atomic on the local CPU, e.g. in case interrupted 615 * or preempted). 616 * 617 * o At any given time on each mutex, MUTEX_SET_WAITERS() 618 * can only ever be in progress on one CPU in the 619 * system - guaranteed by the turnstile chain lock. 620 * 621 * o No other operations other than MUTEX_SET_WAITERS() 622 * and release can modify a mutex with a non-zero 623 * owner field. 624 * 625 * o If the holding LWP switches away, it posts a store 626 * fence before changing curlwp, ensuring that any 627 * overwrite of the mutex waiters flag by mutex_exit() 628 * completes before the modification of curlwp becomes 629 * visible to this CPU. 630 * 631 * o cpu_switchto() posts a store fence after setting curlwp 632 * and before resuming execution of an LWP. 633 * 634 * o _kernel_lock() posts a store fence before setting 635 * curcpu()->ci_biglock_wanted, and after clearing it. 636 * This ensures that any overwrite of the mutex waiters 637 * flag by mutex_exit() completes before the modification 638 * of ci_biglock_wanted becomes visible. 639 * 640 * After MUTEX_SET_WAITERS() succeeds, simultaneously 641 * confirming that the same LWP still holds the mutex 642 * since we took the turnstile lock and notifying it that 643 * we're waiting, we check the lock holder's status again. 644 * Some of the possible outcomes (not an exhaustive list; 645 * XXX this should be made exhaustive): 646 * 647 * 1. The on-CPU check returns true: the holding LWP is 648 * running again. The lock may be released soon and 649 * we should spin. Importantly, we can't trust the 650 * value of the waiters flag. 651 * 652 * 2. The on-CPU check returns false: the holding LWP is 653 * not running. We now have the opportunity to check 654 * if mutex_exit() has blatted the modifications made 655 * by MUTEX_SET_WAITERS(). 656 * 657 * 3. The on-CPU check returns false: the holding LWP may 658 * or may not be running. It has context switched at 659 * some point during our check. Again, we have the 660 * chance to see if the waiters bit is still set or 661 * has been overwritten. 662 * 663 * 4. The on-CPU check returns false: the holding LWP is 664 * running on a CPU, but wants the big lock. It's OK 665 * to check the waiters field in this case. 666 * 667 * 5. The has-waiters check fails: the mutex has been 668 * released, the waiters flag cleared and another LWP 669 * now owns the mutex. 670 * 671 * 6. The has-waiters check fails: the mutex has been 672 * released. 673 * 674 * If the waiters bit is not set it's unsafe to go asleep, 675 * as we might never be awoken. 676 */ 677 if (mutex_oncpu(owner)) { 678 turnstile_exit(mtx); 679 owner = mtx->mtx_owner; 680 continue; 681 } 682 membar_consumer(); 683 if (!MUTEX_HAS_WAITERS(mtx)) { 684 turnstile_exit(mtx); 685 owner = mtx->mtx_owner; 686 continue; 687 } 688 #endif /* MULTIPROCESSOR */ 689 690 LOCKSTAT_START_TIMER(lsflag, slptime); 691 692 turnstile_block(ts, TS_WRITER_Q, mtx, &mutex_syncobj); 693 694 LOCKSTAT_STOP_TIMER(lsflag, slptime); 695 LOCKSTAT_COUNT(slpcnt, 1); 696 697 owner = mtx->mtx_owner; 698 } 699 KPREEMPT_ENABLE(curlwp); 700 701 LOCKSTAT_EVENT(lsflag, mtx, LB_ADAPTIVE_MUTEX | LB_SLEEP1, 702 slpcnt, slptime); 703 LOCKSTAT_EVENT(lsflag, mtx, LB_ADAPTIVE_MUTEX | LB_SPIN, 704 spincnt, spintime); 705 LOCKSTAT_EXIT(lsflag); 706 707 MUTEX_DASSERT(mtx, MUTEX_OWNER(mtx->mtx_owner) == curthread); 708 MUTEX_LOCKED(mtx, &owanted); 709 } 710 711 /* 712 * mutex_vector_exit: 713 * 714 * Support routine for mutex_exit() that handles all cases. 715 */ 716 void 717 mutex_vector_exit(kmutex_t *mtx) 718 { 719 turnstile_t *ts; 720 uintptr_t curthread; 721 722 if (MUTEX_SPIN_P(mtx->mtx_owner)) { 723 #ifdef FULL 724 if (__predict_false(!MUTEX_SPINBIT_LOCKED_P(mtx))) { 725 MUTEX_ABORT(mtx, "exiting unheld spin mutex"); 726 } 727 MUTEX_UNLOCKED(mtx); 728 MUTEX_SPINBIT_LOCK_UNLOCK(mtx); 729 #endif 730 MUTEX_SPIN_SPLRESTORE(mtx); 731 return; 732 } 733 734 #ifndef __HAVE_MUTEX_STUBS 735 /* 736 * On some architectures without mutex stubs, we can enter here to 737 * release mutexes before interrupts and whatnot are up and running. 738 * We need this hack to keep them sweet. 739 */ 740 if (__predict_false(cold)) { 741 MUTEX_UNLOCKED(mtx); 742 MUTEX_RELEASE(mtx); 743 return; 744 } 745 #endif 746 747 curthread = (uintptr_t)curlwp; 748 MUTEX_DASSERT(mtx, curthread != 0); 749 MUTEX_ASSERT(mtx, MUTEX_OWNER(mtx->mtx_owner) == curthread); 750 MUTEX_UNLOCKED(mtx); 751 #if !defined(LOCKDEBUG) 752 __USE(curthread); 753 #endif 754 755 #ifdef LOCKDEBUG 756 /* 757 * Avoid having to take the turnstile chain lock every time 758 * around. Raise the priority level to splhigh() in order 759 * to disable preemption and so make the following atomic. 760 * This also blocks out soft interrupts that could set the 761 * waiters bit. 762 */ 763 { 764 int s = splhigh(); 765 if (!MUTEX_HAS_WAITERS(mtx)) { 766 MUTEX_RELEASE(mtx); 767 splx(s); 768 return; 769 } 770 splx(s); 771 } 772 #endif 773 774 /* 775 * Get this lock's turnstile. This gets the interlock on 776 * the sleep queue. Once we have that, we can clear the 777 * lock. If there was no turnstile for the lock, there 778 * were no waiters remaining. 779 */ 780 ts = turnstile_lookup(mtx); 781 782 if (ts == NULL) { 783 MUTEX_RELEASE(mtx); 784 turnstile_exit(mtx); 785 } else { 786 MUTEX_RELEASE(mtx); 787 turnstile_wakeup(ts, TS_WRITER_Q, 788 TS_WAITERS(ts, TS_WRITER_Q), NULL); 789 } 790 } 791 792 #ifndef __HAVE_SIMPLE_MUTEXES 793 /* 794 * mutex_wakeup: 795 * 796 * Support routine for mutex_exit() that wakes up all waiters. 797 * We assume that the mutex has been released, but it need not 798 * be. 799 */ 800 void 801 mutex_wakeup(kmutex_t *mtx) 802 { 803 turnstile_t *ts; 804 805 ts = turnstile_lookup(mtx); 806 if (ts == NULL) { 807 turnstile_exit(mtx); 808 return; 809 } 810 MUTEX_CLEAR_WAITERS(mtx); 811 turnstile_wakeup(ts, TS_WRITER_Q, TS_WAITERS(ts, TS_WRITER_Q), NULL); 812 } 813 #endif /* !__HAVE_SIMPLE_MUTEXES */ 814 815 /* 816 * mutex_owned: 817 * 818 * Return true if the current LWP (adaptive) or CPU (spin) 819 * holds the mutex. 820 */ 821 int 822 mutex_owned(const kmutex_t *mtx) 823 { 824 825 if (mtx == NULL) 826 return 0; 827 if (MUTEX_ADAPTIVE_P(mtx->mtx_owner)) 828 return MUTEX_OWNER(mtx->mtx_owner) == (uintptr_t)curlwp; 829 #ifdef FULL 830 return MUTEX_SPINBIT_LOCKED_P(mtx); 831 #else 832 return 1; 833 #endif 834 } 835 836 /* 837 * mutex_owner: 838 * 839 * Return the current owner of an adaptive mutex. Used for 840 * priority inheritance. 841 */ 842 static lwp_t * 843 mutex_owner(wchan_t wchan) 844 { 845 volatile const kmutex_t *mtx = wchan; 846 847 MUTEX_ASSERT(mtx, MUTEX_ADAPTIVE_P(mtx->mtx_owner)); 848 return (struct lwp *)MUTEX_OWNER(mtx->mtx_owner); 849 } 850 851 /* 852 * mutex_ownable: 853 * 854 * When compiled with DEBUG and LOCKDEBUG defined, ensure that 855 * the mutex is available. We cannot use !mutex_owned() since 856 * that won't work correctly for spin mutexes. 857 */ 858 int 859 mutex_ownable(const kmutex_t *mtx) 860 { 861 862 #ifdef LOCKDEBUG 863 MUTEX_TESTLOCK(mtx); 864 #endif 865 return 1; 866 } 867 868 /* 869 * mutex_tryenter: 870 * 871 * Try to acquire the mutex; return non-zero if we did. 872 */ 873 int 874 mutex_tryenter(kmutex_t *mtx) 875 { 876 uintptr_t curthread; 877 878 /* 879 * Handle spin mutexes. 880 */ 881 if (MUTEX_SPIN_P(mtx->mtx_owner)) { 882 MUTEX_SPIN_SPLRAISE(mtx); 883 #ifdef FULL 884 if (MUTEX_SPINBIT_LOCK_TRY(mtx)) { 885 MUTEX_WANTLOCK(mtx, NULL); 886 MUTEX_LOCKED(mtx, NULL); 887 return 1; 888 } 889 MUTEX_SPIN_SPLRESTORE(mtx); 890 #else 891 MUTEX_WANTLOCK(mtx, NULL); 892 MUTEX_LOCKED(mtx, NULL); 893 return 1; 894 #endif 895 } else { 896 curthread = (uintptr_t)curlwp; 897 MUTEX_ASSERT(mtx, curthread != 0); 898 if (MUTEX_ACQUIRE(mtx, curthread)) { 899 MUTEX_WANTLOCK(mtx, NULL); 900 MUTEX_LOCKED(mtx, NULL); 901 MUTEX_DASSERT(mtx, 902 MUTEX_OWNER(mtx->mtx_owner) == curthread); 903 return 1; 904 } 905 } 906 907 return 0; 908 } 909 910 #if defined(__HAVE_SPIN_MUTEX_STUBS) || defined(FULL) 911 /* 912 * mutex_spin_retry: 913 * 914 * Support routine for mutex_spin_enter(). Assumes that the caller 915 * has already raised the SPL, and adjusted counters. 916 */ 917 void 918 mutex_spin_retry(kmutex_t *mtx) 919 { 920 #ifdef MULTIPROCESSOR 921 u_int count; 922 LOCKSTAT_TIMER(spintime); 923 LOCKSTAT_FLAG(lsflag); 924 #ifdef LOCKDEBUG 925 u_int spins = 0; 926 #endif /* LOCKDEBUG */ 927 volatile void *owanted; 928 929 MUTEX_WANTLOCK(mtx, &owanted); 930 931 LOCKSTAT_ENTER(lsflag); 932 LOCKSTAT_START_TIMER(lsflag, spintime); 933 count = SPINLOCK_BACKOFF_MIN; 934 935 /* 936 * Spin testing the lock word and do exponential backoff 937 * to reduce cache line ping-ponging between CPUs. 938 */ 939 do { 940 while (MUTEX_SPINBIT_LOCKED_P(mtx)) { 941 SPINLOCK_BACKOFF(count); 942 #ifdef LOCKDEBUG 943 if (SPINLOCK_SPINOUT(spins)) 944 MUTEX_ABORT(mtx, "spinout"); 945 #endif /* LOCKDEBUG */ 946 } 947 } while (!MUTEX_SPINBIT_LOCK_TRY(mtx)); 948 949 LOCKSTAT_STOP_TIMER(lsflag, spintime); 950 LOCKSTAT_EVENT(lsflag, mtx, LB_SPIN_MUTEX | LB_SPIN, 1, spintime); 951 LOCKSTAT_EXIT(lsflag); 952 953 MUTEX_LOCKED(mtx, &owanted); 954 #else /* MULTIPROCESSOR */ 955 MUTEX_ABORT(mtx, "locking against myself"); 956 #endif /* MULTIPROCESSOR */ 957 } 958 #endif /* defined(__HAVE_SPIN_MUTEX_STUBS) || defined(FULL) */ 959