1 #include "jemalloc/internal/jemalloc_preamble.h" 2 #include "jemalloc/internal/jemalloc_internal_includes.h" 3 4 #include "jemalloc/internal/assert.h" 5 #include "jemalloc/internal/base.h" 6 #include "jemalloc/internal/mutex.h" 7 #include "jemalloc/internal/safety_check.h" 8 #include "jemalloc/internal/san.h" 9 #include "jemalloc/internal/sc.h" 10 11 /******************************************************************************/ 12 /* Data. */ 13 14 bool opt_tcache = true; 15 16 /* global_do_not_change_tcache_maxclass is set to 32KB by default. */ 17 size_t opt_tcache_max = ((size_t)1) << 15; 18 19 /* Reasonable defaults for min and max values. */ 20 unsigned opt_tcache_nslots_small_min = 20; 21 unsigned opt_tcache_nslots_small_max = 200; 22 unsigned opt_tcache_nslots_large = 20; 23 24 /* 25 * We attempt to make the number of slots in a tcache bin for a given size class 26 * equal to the number of objects in a slab times some multiplier. By default, 27 * the multiplier is 2 (i.e. we set the maximum number of objects in the tcache 28 * to twice the number of objects in a slab). 29 * This is bounded by some other constraints as well, like the fact that it 30 * must be even, must be less than opt_tcache_nslots_small_max, etc.. 31 */ 32 ssize_t opt_lg_tcache_nslots_mul = 1; 33 34 /* 35 * Number of allocation bytes between tcache incremental GCs. Again, this 36 * default just seems to work well; more tuning is possible. 37 */ 38 size_t opt_tcache_gc_incr_bytes = 65536; 39 40 /* 41 * With default settings, we may end up flushing small bins frequently with 42 * small flush amounts. To limit this tendency, we can set a number of bytes to 43 * "delay" by. If we try to flush N M-byte items, we decrease that size-class's 44 * delay by N * M. So, if delay is 1024 and we're looking at the 64-byte size 45 * class, we won't do any flushing until we've been asked to flush 1024/64 == 16 46 * items. This can happen in any configuration (i.e. being asked to flush 16 47 * items once, or 4 items 4 times). 48 * 49 * Practically, this is stored as a count of items in a uint8_t, so the 50 * effective maximum value for a size class is 255 * sz. 51 */ 52 size_t opt_tcache_gc_delay_bytes = 0; 53 54 /* 55 * When a cache bin is flushed because it's full, how much of it do we flush? 56 * By default, we flush half the maximum number of items. 57 */ 58 unsigned opt_lg_tcache_flush_small_div = 1; 59 unsigned opt_lg_tcache_flush_large_div = 1; 60 61 /* 62 * Number of cache bins enabled, including both large and small. This value 63 * is only used to initialize tcache_nbins in the per-thread tcache. 64 * Directly modifying it will not affect threads already launched. 65 */ 66 unsigned global_do_not_change_tcache_nbins; 67 /* 68 * Max size class to be cached (can be small or large). This value is only used 69 * to initialize tcache_max in the per-thread tcache. Directly modifying it 70 * will not affect threads already launched. 71 */ 72 size_t global_do_not_change_tcache_maxclass; 73 74 /* 75 * Default bin info for each bin. Will be initialized in malloc_conf_init 76 * and tcache_boot and should not be modified after that. 77 */ 78 static cache_bin_info_t opt_tcache_ncached_max[TCACHE_NBINS_MAX] = {{0}}; 79 /* 80 * Marks whether a bin's info is set already. This is used in 81 * tcache_bin_info_compute to avoid overwriting ncached_max specified by 82 * malloc_conf. It should be set only when parsing malloc_conf. 83 */ 84 static bool opt_tcache_ncached_max_set[TCACHE_NBINS_MAX] = {0}; 85 86 tcaches_t *tcaches; 87 88 /* Index of first element within tcaches that has never been used. */ 89 static unsigned tcaches_past; 90 91 /* Head of singly linked list tracking available tcaches elements. */ 92 static tcaches_t *tcaches_avail; 93 94 /* Protects tcaches{,_past,_avail}. */ 95 static malloc_mutex_t tcaches_mtx; 96 97 /******************************************************************************/ 98 99 size_t 100 tcache_salloc(tsdn_t *tsdn, const void *ptr) { 101 return arena_salloc(tsdn, ptr); 102 } 103 104 static uint64_t 105 tcache_gc_new_event_wait(tsd_t *tsd) { 106 return opt_tcache_gc_incr_bytes; 107 } 108 109 static uint64_t 110 tcache_gc_postponed_event_wait(tsd_t *tsd) { 111 return TE_MIN_START_WAIT; 112 } 113 114 static inline void 115 tcache_bin_fill_ctl_init(tcache_slow_t *tcache_slow, szind_t szind) { 116 assert(szind < SC_NBINS); 117 cache_bin_fill_ctl_t *ctl = 118 &tcache_slow->bin_fill_ctl_do_not_access_directly[szind]; 119 ctl->base = 1; 120 ctl->offset = 0; 121 } 122 123 static inline cache_bin_fill_ctl_t * 124 tcache_bin_fill_ctl_get(tcache_slow_t *tcache_slow, szind_t szind) { 125 assert(szind < SC_NBINS); 126 cache_bin_fill_ctl_t *ctl = 127 &tcache_slow->bin_fill_ctl_do_not_access_directly[szind]; 128 assert(ctl->base > ctl->offset); 129 return ctl; 130 } 131 132 /* 133 * The number of items to be filled at a time for a given small bin is 134 * calculated by (ncached_max >> lg_fill_div). 135 * The actual ctl struct consists of two fields, i.e. base and offset, 136 * and the difference between the two(base - offset) is the final lg_fill_div. 137 * The base is adjusted during GC based on the traffic within a period of time, 138 * while the offset is updated in real time to handle the immediate traffic. 139 */ 140 static inline uint8_t 141 tcache_nfill_small_lg_div_get(tcache_slow_t *tcache_slow, szind_t szind) { 142 cache_bin_fill_ctl_t *ctl = tcache_bin_fill_ctl_get(tcache_slow, szind); 143 return (ctl->base - (opt_experimental_tcache_gc ? ctl->offset : 0)); 144 } 145 146 /* 147 * When we want to fill more items to respond to burst load, 148 * offset is increased so that (base - offset) is decreased, 149 * which in return increases the number of items to be filled. 150 */ 151 static inline void 152 tcache_nfill_small_burst_prepare(tcache_slow_t *tcache_slow, szind_t szind) { 153 cache_bin_fill_ctl_t *ctl = tcache_bin_fill_ctl_get(tcache_slow, szind); 154 if (ctl->offset + 1 < ctl->base) { 155 ctl->offset++; 156 } 157 } 158 159 static inline void 160 tcache_nfill_small_burst_reset(tcache_slow_t *tcache_slow, szind_t szind) { 161 cache_bin_fill_ctl_t *ctl = tcache_bin_fill_ctl_get(tcache_slow, szind); 162 ctl->offset = 0; 163 } 164 165 /* 166 * limit == 0: indicating that the fill count should be increased, 167 * i.e. lg_div(base) should be decreased. 168 * 169 * limit != 0: limit is set to ncached_max, indicating that the fill 170 * count should be decreased, i.e. lg_div(base) should be increased. 171 */ 172 static inline void 173 tcache_nfill_small_gc_update( 174 tcache_slow_t *tcache_slow, szind_t szind, cache_bin_sz_t limit) { 175 cache_bin_fill_ctl_t *ctl = tcache_bin_fill_ctl_get(tcache_slow, szind); 176 if (!limit && ctl->base > 1) { 177 /* 178 * Increase fill count by 2X for small bins. Make sure 179 * lg_fill_div stays greater than 1. 180 */ 181 ctl->base--; 182 } else if (limit && (limit >> ctl->base) > 1) { 183 /* 184 * Reduce fill count by 2X. Limit lg_fill_div such that 185 * the fill count is always at least 1. 186 */ 187 ctl->base++; 188 } 189 /* Reset the offset for the next GC period. */ 190 ctl->offset = 0; 191 } 192 193 static uint8_t 194 tcache_gc_item_delay_compute(szind_t szind) { 195 assert(szind < SC_NBINS); 196 size_t sz = sz_index2size(szind); 197 size_t item_delay = opt_tcache_gc_delay_bytes / sz; 198 size_t delay_max = ZU(1) 199 << (sizeof(((tcache_slow_t *)NULL)->bin_flush_delay_items[0]) * 8); 200 if (item_delay >= delay_max) { 201 item_delay = delay_max - 1; 202 } 203 return (uint8_t)item_delay; 204 } 205 206 static inline void * 207 tcache_gc_small_heuristic_addr_get( 208 tsd_t *tsd, tcache_slow_t *tcache_slow, szind_t szind) { 209 assert(szind < SC_NBINS); 210 tsdn_t *tsdn = tsd_tsdn(tsd); 211 bin_t *bin = bin_choose(tsdn, tcache_slow->arena, szind, NULL); 212 assert(bin != NULL); 213 214 malloc_mutex_lock(tsdn, &bin->lock); 215 edata_t *slab = (bin->slabcur == NULL) 216 ? edata_heap_first(&bin->slabs_nonfull) 217 : bin->slabcur; 218 assert(slab != NULL || edata_heap_empty(&bin->slabs_nonfull)); 219 void *ret = (slab != NULL) ? edata_addr_get(slab) : NULL; 220 assert(ret != NULL || slab == NULL); 221 malloc_mutex_unlock(tsdn, &bin->lock); 222 223 return ret; 224 } 225 226 static inline bool 227 tcache_gc_is_addr_remote(void *addr, uintptr_t min, uintptr_t max) { 228 assert(addr != NULL); 229 return ((uintptr_t)addr < min || (uintptr_t)addr >= max); 230 } 231 232 static inline cache_bin_sz_t 233 tcache_gc_small_nremote_get(cache_bin_t *cache_bin, void *addr, 234 uintptr_t *addr_min, uintptr_t *addr_max, szind_t szind, size_t nflush) { 235 assert(addr != NULL && addr_min != NULL && addr_max != NULL); 236 /* The slab address range that the provided addr belongs to. */ 237 uintptr_t slab_min = (uintptr_t)addr; 238 uintptr_t slab_max = slab_min + bin_infos[szind].slab_size; 239 /* 240 * When growing retained virtual memory, it's increased exponentially, 241 * starting from 2M, so that the total number of disjoint virtual 242 * memory ranges retained by each shard is limited. 243 */ 244 uintptr_t neighbor_min = ((uintptr_t)addr > TCACHE_GC_NEIGHBOR_LIMIT) 245 ? ((uintptr_t)addr - TCACHE_GC_NEIGHBOR_LIMIT) 246 : 0; 247 uintptr_t neighbor_max = ((uintptr_t)addr 248 < (UINTPTR_MAX - TCACHE_GC_NEIGHBOR_LIMIT)) 249 ? ((uintptr_t)addr + TCACHE_GC_NEIGHBOR_LIMIT) 250 : UINTPTR_MAX; 251 252 /* Scan the entire bin to count the number of remote pointers. */ 253 void **head = cache_bin->stack_head; 254 cache_bin_sz_t n_remote_slab = 0, n_remote_neighbor = 0; 255 cache_bin_sz_t ncached = cache_bin_ncached_get_local(cache_bin); 256 for (void **cur = head; cur < head + ncached; cur++) { 257 n_remote_slab += (cache_bin_sz_t)tcache_gc_is_addr_remote( 258 *cur, slab_min, slab_max); 259 n_remote_neighbor += (cache_bin_sz_t)tcache_gc_is_addr_remote( 260 *cur, neighbor_min, neighbor_max); 261 } 262 /* 263 * Note: since slab size is dynamic and can be larger than 2M, i.e. 264 * TCACHE_GC_NEIGHBOR_LIMIT, there is no guarantee as to which of 265 * n_remote_slab and n_remote_neighbor is greater. 266 */ 267 assert(n_remote_slab <= ncached && n_remote_neighbor <= ncached); 268 /* 269 * We first consider keeping ptrs from the neighboring addr range, 270 * since in most cases the range is greater than the slab range. 271 * So if the number of non-neighbor ptrs is more than the intended 272 * flush amount, we use it as the anchor for flushing. 273 */ 274 if (n_remote_neighbor >= nflush) { 275 *addr_min = neighbor_min; 276 *addr_max = neighbor_max; 277 return n_remote_neighbor; 278 } 279 /* 280 * We then consider only keeping ptrs from the local slab, and in most 281 * cases this is stricter, assuming that slab < 2M is the common case. 282 */ 283 *addr_min = slab_min; 284 *addr_max = slab_max; 285 return n_remote_slab; 286 } 287 288 /* Shuffle the ptrs in the bin to put the remote pointers at the bottom. */ 289 static inline void 290 tcache_gc_small_bin_shuffle(cache_bin_t *cache_bin, cache_bin_sz_t nremote, 291 uintptr_t addr_min, uintptr_t addr_max) { 292 void **swap = NULL; 293 cache_bin_sz_t ncached = cache_bin_ncached_get_local(cache_bin); 294 cache_bin_sz_t ntop = ncached - nremote, cnt = 0; 295 assert(ntop > 0 && ntop < ncached); 296 /* 297 * Scan the [head, head + ntop) part of the cache bin, during which 298 * bubbling the non-remote ptrs to the top of the bin. 299 * After this, the [head, head + cnt) part of the bin contains only 300 * non-remote ptrs, and they're in the same relative order as before. 301 * While the [head + cnt, head + ntop) part contains only remote ptrs. 302 */ 303 void **head = cache_bin->stack_head; 304 for (void **cur = head; cur < head + ntop; cur++) { 305 if (!tcache_gc_is_addr_remote(*cur, addr_min, addr_max)) { 306 /* Tracks the number of non-remote ptrs seen so far. */ 307 cnt++; 308 /* 309 * There is remote ptr before the current non-remote ptr, 310 * swap the current non-remote ptr with the remote ptr, 311 * and increment the swap pointer so that it's still 312 * pointing to the top remote ptr in the bin. 313 */ 314 if (swap != NULL) { 315 assert(swap < cur); 316 assert(tcache_gc_is_addr_remote( 317 *swap, addr_min, addr_max)); 318 void *tmp = *cur; 319 *cur = *swap; 320 *swap = tmp; 321 swap++; 322 assert(swap <= cur); 323 assert(tcache_gc_is_addr_remote( 324 *swap, addr_min, addr_max)); 325 } 326 continue; 327 } else if (swap == NULL) { 328 /* Swap always points to the top remote ptr in the bin. */ 329 swap = cur; 330 } 331 } 332 /* 333 * Scan the [head + ntop, head + ncached) part of the cache bin, 334 * after which it should only contain remote ptrs. 335 */ 336 for (void **cur = head + ntop; cur < head + ncached; cur++) { 337 /* Early break if all non-remote ptrs have been moved. */ 338 if (cnt == ntop) { 339 break; 340 } 341 if (!tcache_gc_is_addr_remote(*cur, addr_min, addr_max)) { 342 assert(tcache_gc_is_addr_remote( 343 *(head + cnt), addr_min, addr_max)); 344 void *tmp = *cur; 345 *cur = *(head + cnt); 346 *(head + cnt) = tmp; 347 cnt++; 348 } 349 } 350 assert(cnt == ntop); 351 /* Sanity check to make sure the shuffle is done correctly. */ 352 for (void **cur = head; cur < head + ncached; cur++) { 353 assert(*cur != NULL); 354 assert( 355 ((cur < head + ntop) 356 && !tcache_gc_is_addr_remote(*cur, addr_min, addr_max)) 357 || ((cur >= head + ntop) 358 && tcache_gc_is_addr_remote(*cur, addr_min, addr_max))); 359 } 360 } 361 362 static bool 363 tcache_gc_small( 364 tsd_t *tsd, tcache_slow_t *tcache_slow, tcache_t *tcache, szind_t szind) { 365 /* 366 * Aim to flush 3/4 of items below low-water, with remote pointers being 367 * prioritized for flushing. 368 */ 369 assert(szind < SC_NBINS); 370 371 cache_bin_t *cache_bin = &tcache->bins[szind]; 372 assert(!tcache_bin_disabled(szind, cache_bin, tcache->tcache_slow)); 373 cache_bin_sz_t ncached = cache_bin_ncached_get_local(cache_bin); 374 cache_bin_sz_t low_water = cache_bin_low_water_get(cache_bin); 375 if (low_water > 0) { 376 /* 377 * There is unused items within the GC period => reduce fill count. 378 * limit field != 0 is borrowed to indicate that the fill count 379 * should be reduced. 380 */ 381 tcache_nfill_small_gc_update(tcache_slow, szind, 382 /* limit */ cache_bin_ncached_max_get(cache_bin)); 383 } else if (tcache_slow->bin_refilled[szind]) { 384 /* 385 * There has been refills within the GC period => increase fill count. 386 * limit field set to 0 is borrowed to indicate that the fill count 387 * should be increased. 388 */ 389 tcache_nfill_small_gc_update(tcache_slow, szind, /* limit */ 0); 390 tcache_slow->bin_refilled[szind] = false; 391 } 392 assert(!tcache_slow->bin_refilled[szind]); 393 394 cache_bin_sz_t nflush = low_water - (low_water >> 2); 395 /* 396 * When the new tcache gc is not enabled, keep the flush delay logic, 397 * and directly flush the bottom nflush items if needed. 398 */ 399 if (!opt_experimental_tcache_gc) { 400 if (nflush < tcache_slow->bin_flush_delay_items[szind]) { 401 /* Workaround for a conversion warning. */ 402 uint8_t nflush_uint8 = (uint8_t)nflush; 403 assert(sizeof(tcache_slow->bin_flush_delay_items[0]) 404 == sizeof(nflush_uint8)); 405 tcache_slow->bin_flush_delay_items[szind] -= 406 nflush_uint8; 407 return false; 408 } 409 410 tcache_slow->bin_flush_delay_items[szind] = 411 tcache_gc_item_delay_compute(szind); 412 goto label_flush; 413 } 414 415 /* Directly goto the flush path when the entire bin needs to be flushed. */ 416 if (nflush == ncached) { 417 goto label_flush; 418 } 419 420 /* Query arena binshard to get heuristic locality info. */ 421 void *addr = tcache_gc_small_heuristic_addr_get( 422 tsd, tcache_slow, szind); 423 if (addr == NULL) { 424 goto label_flush; 425 } 426 427 /* 428 * Use the queried addr above to get the number of remote ptrs in the 429 * bin, and the min/max of the local addr range. 430 */ 431 uintptr_t addr_min, addr_max; 432 cache_bin_sz_t nremote = tcache_gc_small_nremote_get( 433 cache_bin, addr, &addr_min, &addr_max, szind, nflush); 434 435 /* 436 * Update the nflush to the larger value between the intended flush count 437 * and the number of remote ptrs. 438 */ 439 if (nremote > nflush) { 440 nflush = nremote; 441 } 442 /* 443 * When entering the locality check, nflush should be less than ncached, 444 * otherwise the entire bin should be flushed regardless. The only case 445 * when nflush gets updated to ncached after locality check is, when all 446 * the items in the bin are remote, in which case the entire bin should 447 * also be flushed. 448 */ 449 assert(nflush < ncached || nremote == ncached); 450 if (nremote == 0 || nremote == ncached) { 451 goto label_flush; 452 } 453 454 /* 455 * Move the remote points to the bottom of the bin for flushing. 456 * As long as moved to the bottom, the order of these nremote ptrs 457 * does not matter, since they are going to be flushed anyway. 458 * The rest of the ptrs are moved to the top of the bin, and their 459 * relative order is maintained. 460 */ 461 tcache_gc_small_bin_shuffle(cache_bin, nremote, addr_min, addr_max); 462 463 label_flush: 464 if (nflush == 0) { 465 assert(low_water == 0); 466 return false; 467 } 468 assert(nflush <= ncached); 469 tcache_bin_flush_small( 470 tsd, tcache, cache_bin, szind, (unsigned)(ncached - nflush)); 471 return true; 472 } 473 474 static bool 475 tcache_gc_large( 476 tsd_t *tsd, tcache_slow_t *tcache_slow, tcache_t *tcache, szind_t szind) { 477 /* 478 * Like the small GC, flush 3/4 of untouched items. However, simply flush 479 * the bottom nflush items, without any locality check. 480 */ 481 assert(szind >= SC_NBINS); 482 cache_bin_t *cache_bin = &tcache->bins[szind]; 483 assert(!tcache_bin_disabled(szind, cache_bin, tcache->tcache_slow)); 484 cache_bin_sz_t low_water = cache_bin_low_water_get(cache_bin); 485 if (low_water == 0) { 486 return false; 487 } 488 unsigned nrem = (unsigned)(cache_bin_ncached_get_local(cache_bin) 489 - low_water + (low_water >> 2)); 490 tcache_bin_flush_large(tsd, tcache, cache_bin, szind, nrem); 491 return true; 492 } 493 494 /* Try to gc one bin by szind, return true if there is item flushed. */ 495 static bool 496 tcache_try_gc_bin( 497 tsd_t *tsd, tcache_slow_t *tcache_slow, tcache_t *tcache, szind_t szind) { 498 assert(tcache != NULL); 499 cache_bin_t *cache_bin = &tcache->bins[szind]; 500 if (tcache_bin_disabled(szind, cache_bin, tcache_slow)) { 501 return false; 502 } 503 504 bool is_small = (szind < SC_NBINS); 505 tcache_bin_flush_stashed(tsd, tcache, cache_bin, szind, is_small); 506 bool ret = is_small ? tcache_gc_small(tsd, tcache_slow, tcache, szind) 507 : tcache_gc_large(tsd, tcache_slow, tcache, szind); 508 cache_bin_low_water_set(cache_bin); 509 return ret; 510 } 511 512 static void 513 tcache_gc_event(tsd_t *tsd) { 514 tcache_t *tcache = tcache_get(tsd); 515 if (tcache == NULL) { 516 return; 517 } 518 519 tcache_slow_t *tcache_slow = tsd_tcache_slowp_get(tsd); 520 assert(tcache_slow != NULL); 521 522 /* When the new tcache gc is not enabled, GC one bin at a time. */ 523 if (!opt_experimental_tcache_gc) { 524 szind_t szind = tcache_slow->next_gc_bin; 525 tcache_try_gc_bin(tsd, tcache_slow, tcache, szind); 526 tcache_slow->next_gc_bin++; 527 if (tcache_slow->next_gc_bin == tcache_nbins_get(tcache_slow)) { 528 tcache_slow->next_gc_bin = 0; 529 } 530 return; 531 } 532 533 nstime_t now; 534 nstime_copy(&now, &tcache_slow->last_gc_time); 535 nstime_update(&now); 536 assert(nstime_compare(&now, &tcache_slow->last_gc_time) >= 0); 537 538 if (nstime_ns(&now) - nstime_ns(&tcache_slow->last_gc_time) 539 < TCACHE_GC_INTERVAL_NS) { 540 // time interval is too short, skip this event. 541 return; 542 } 543 /* Update last_gc_time to now. */ 544 nstime_copy(&tcache_slow->last_gc_time, &now); 545 546 unsigned gc_small_nbins = 0, gc_large_nbins = 0; 547 unsigned tcache_nbins = tcache_nbins_get(tcache_slow); 548 unsigned small_nbins = tcache_nbins > SC_NBINS ? SC_NBINS 549 : tcache_nbins; 550 szind_t szind_small = tcache_slow->next_gc_bin_small; 551 szind_t szind_large = tcache_slow->next_gc_bin_large; 552 553 /* Flush at most TCACHE_GC_SMALL_NBINS_MAX small bins at a time. */ 554 for (unsigned i = 0; 555 i < small_nbins && gc_small_nbins < TCACHE_GC_SMALL_NBINS_MAX; 556 i++) { 557 assert(szind_small < SC_NBINS); 558 if (tcache_try_gc_bin(tsd, tcache_slow, tcache, szind_small)) { 559 gc_small_nbins++; 560 } 561 if (++szind_small == small_nbins) { 562 szind_small = 0; 563 } 564 } 565 tcache_slow->next_gc_bin_small = szind_small; 566 567 if (tcache_nbins <= SC_NBINS) { 568 return; 569 } 570 571 /* Flush at most TCACHE_GC_LARGE_NBINS_MAX large bins at a time. */ 572 for (unsigned i = SC_NBINS; 573 i < tcache_nbins && gc_large_nbins < TCACHE_GC_LARGE_NBINS_MAX; 574 i++) { 575 assert(szind_large >= SC_NBINS && szind_large < tcache_nbins); 576 if (tcache_try_gc_bin(tsd, tcache_slow, tcache, szind_large)) { 577 gc_large_nbins++; 578 } 579 if (++szind_large == tcache_nbins) { 580 szind_large = SC_NBINS; 581 } 582 } 583 tcache_slow->next_gc_bin_large = szind_large; 584 } 585 586 void * 587 tcache_alloc_small_hard(tsdn_t *tsdn, arena_t *arena, tcache_t *tcache, 588 cache_bin_t *cache_bin, szind_t binind, bool *tcache_success) { 589 tcache_slow_t *tcache_slow = tcache->tcache_slow; 590 void *ret; 591 592 assert(tcache_slow->arena != NULL); 593 assert(!tcache_bin_disabled(binind, cache_bin, tcache_slow)); 594 assert(cache_bin_ncached_get_local(cache_bin) == 0); 595 cache_bin_sz_t nfill = cache_bin_ncached_max_get(cache_bin) 596 >> tcache_nfill_small_lg_div_get(tcache_slow, binind); 597 if (nfill == 0) { 598 nfill = 1; 599 } 600 cache_bin_sz_t nfill_min = opt_experimental_tcache_gc 601 ? ((nfill >> 1) + 1) 602 : nfill; 603 cache_bin_sz_t nfill_max = nfill; 604 CACHE_BIN_PTR_ARRAY_DECLARE(ptrs, nfill_max); 605 cache_bin_init_ptr_array_for_fill(cache_bin, &ptrs, nfill_max); 606 607 cache_bin_sz_t filled = arena_ptr_array_fill_small(tsdn, arena, binind, 608 &ptrs, /* nfill_min */ nfill_min, /* nfill_max */ nfill_max, 609 cache_bin->tstats); 610 cache_bin_finish_fill(cache_bin, &ptrs, filled); 611 assert(filled >= nfill_min && filled <= nfill_max); 612 assert(cache_bin_ncached_get_local(cache_bin) == filled); 613 614 tcache_slow->bin_refilled[binind] = true; 615 tcache_nfill_small_burst_prepare(tcache_slow, binind); 616 ret = cache_bin_alloc(cache_bin, tcache_success); 617 618 return ret; 619 } 620 621 JEMALLOC_ALWAYS_INLINE void 622 tcache_bin_flush_bottom(tsd_t *tsd, tcache_t *tcache, cache_bin_t *cache_bin, 623 szind_t binind, unsigned rem, bool small) { 624 assert(rem <= cache_bin_ncached_max_get(cache_bin)); 625 assert(!tcache_bin_disabled(binind, cache_bin, tcache->tcache_slow)); 626 cache_bin_sz_t orig_nstashed = cache_bin_nstashed_get_local(cache_bin); 627 tcache_bin_flush_stashed(tsd, tcache, cache_bin, binind, small); 628 629 cache_bin_sz_t ncached = cache_bin_ncached_get_local(cache_bin); 630 assert((cache_bin_sz_t)rem <= ncached + orig_nstashed); 631 if ((cache_bin_sz_t)rem > ncached) { 632 /* 633 * The flush_stashed above could have done enough flushing, if 634 * there were many items stashed. Validate that: 1) non zero 635 * stashed, and 2) bin stack has available space now. 636 */ 637 assert(orig_nstashed > 0); 638 assert(ncached + cache_bin_nstashed_get_local(cache_bin) 639 < cache_bin_ncached_max_get(cache_bin)); 640 /* Still go through the flush logic for stats purpose only. */ 641 rem = ncached; 642 } 643 cache_bin_sz_t nflush = ncached - (cache_bin_sz_t)rem; 644 645 CACHE_BIN_PTR_ARRAY_DECLARE(ptrs, nflush); 646 cache_bin_init_ptr_array_for_flush(cache_bin, &ptrs, nflush); 647 648 arena_ptr_array_flush(tsd, binind, &ptrs, nflush, small, 649 tcache->tcache_slow->arena, cache_bin->tstats); 650 651 cache_bin_finish_flush(cache_bin, &ptrs, nflush); 652 } 653 654 void 655 tcache_bin_flush_small(tsd_t *tsd, tcache_t *tcache, cache_bin_t *cache_bin, 656 szind_t binind, unsigned rem) { 657 tcache_nfill_small_burst_reset(tcache->tcache_slow, binind); 658 tcache_bin_flush_bottom(tsd, tcache, cache_bin, binind, rem, 659 /* small */ true); 660 } 661 662 void 663 tcache_bin_flush_large(tsd_t *tsd, tcache_t *tcache, cache_bin_t *cache_bin, 664 szind_t binind, unsigned rem) { 665 tcache_bin_flush_bottom(tsd, tcache, cache_bin, binind, rem, 666 /* small */ false); 667 } 668 669 /* 670 * Flushing stashed happens when 1) tcache fill, 2) tcache flush, or 3) tcache 671 * GC event. This makes sure that the stashed items do not hold memory for too 672 * long, and new buffers can only be allocated when nothing is stashed. 673 * 674 * The downside is, the time between stash and flush may be relatively short, 675 * especially when the request rate is high. It lowers the chance of detecting 676 * write-after-free -- however that is a delayed detection anyway, and is less 677 * of a focus than the memory overhead. 678 */ 679 void 680 tcache_bin_flush_stashed(tsd_t *tsd, tcache_t *tcache, cache_bin_t *cache_bin, 681 szind_t binind, bool is_small) { 682 assert(!tcache_bin_disabled(binind, cache_bin, tcache->tcache_slow)); 683 /* 684 * The two below are for assertion only. The content of original cached 685 * items remain unchanged -- the stashed items reside on the other end 686 * of the stack. Checking the stack head and ncached to verify. 687 */ 688 void *head_content = *cache_bin->stack_head; 689 cache_bin_sz_t orig_cached = cache_bin_ncached_get_local(cache_bin); 690 691 cache_bin_sz_t nstashed = cache_bin_nstashed_get_local(cache_bin); 692 assert(orig_cached + nstashed <= cache_bin_ncached_max_get(cache_bin)); 693 if (nstashed == 0) { 694 return; 695 } 696 697 CACHE_BIN_PTR_ARRAY_DECLARE(ptrs, nstashed); 698 cache_bin_init_ptr_array_for_stashed( 699 cache_bin, binind, &ptrs, nstashed); 700 san_check_stashed_ptrs(ptrs.ptr, nstashed, sz_index2size(binind)); 701 arena_ptr_array_flush(tsd, binind, &ptrs, nstashed, is_small, 702 tcache->tcache_slow->arena, cache_bin->tstats); 703 cache_bin_finish_flush_stashed(cache_bin); 704 705 assert(cache_bin_nstashed_get_local(cache_bin) == 0); 706 assert(cache_bin_ncached_get_local(cache_bin) == orig_cached); 707 assert(head_content == *cache_bin->stack_head); 708 } 709 710 JET_EXTERN bool 711 tcache_get_default_ncached_max_set(szind_t ind) { 712 return opt_tcache_ncached_max_set[ind]; 713 } 714 715 JET_EXTERN const cache_bin_info_t * 716 tcache_get_default_ncached_max(void) { 717 return opt_tcache_ncached_max; 718 } 719 720 bool 721 tcache_bin_ncached_max_read( 722 tsd_t *tsd, size_t bin_size, cache_bin_sz_t *ncached_max) { 723 if (bin_size > TCACHE_MAXCLASS_LIMIT) { 724 return true; 725 } 726 727 if (!tcache_available(tsd)) { 728 *ncached_max = 0; 729 return false; 730 } 731 732 tcache_t *tcache = tsd_tcachep_get(tsd); 733 assert(tcache != NULL); 734 szind_t bin_ind = sz_size2index(bin_size); 735 736 cache_bin_t *bin = &tcache->bins[bin_ind]; 737 *ncached_max = tcache_bin_disabled(bin_ind, bin, tcache->tcache_slow) 738 ? 0 739 : cache_bin_ncached_max_get(bin); 740 return false; 741 } 742 743 void 744 tcache_arena_associate(tsdn_t *tsdn, tcache_slow_t *tcache_slow, 745 tcache_t *tcache, arena_t *arena) { 746 assert(tcache_slow->arena == NULL); 747 tcache_slow->arena = arena; 748 749 if (config_stats) { 750 /* Link into list of extant tcaches. */ 751 malloc_mutex_lock(tsdn, &arena->tcache_ql_mtx); 752 753 ql_elm_new(tcache_slow, link); 754 ql_tail_insert(&arena->tcache_ql, tcache_slow, link); 755 cache_bin_array_descriptor_init( 756 &tcache_slow->cache_bin_array_descriptor, tcache->bins); 757 ql_tail_insert(&arena->cache_bin_array_descriptor_ql, 758 &tcache_slow->cache_bin_array_descriptor, link); 759 760 malloc_mutex_unlock(tsdn, &arena->tcache_ql_mtx); 761 } 762 } 763 764 static void 765 tcache_arena_dissociate( 766 tsdn_t *tsdn, tcache_slow_t *tcache_slow, tcache_t *tcache) { 767 arena_t *arena = tcache_slow->arena; 768 assert(arena != NULL); 769 if (config_stats) { 770 /* Unlink from list of extant tcaches. */ 771 malloc_mutex_lock(tsdn, &arena->tcache_ql_mtx); 772 if (config_debug) { 773 bool in_ql = false; 774 tcache_slow_t *iter; 775 ql_foreach (iter, &arena->tcache_ql, link) { 776 if (iter == tcache_slow) { 777 in_ql = true; 778 break; 779 } 780 } 781 assert(in_ql); 782 } 783 ql_remove(&arena->tcache_ql, tcache_slow, link); 784 ql_remove(&arena->cache_bin_array_descriptor_ql, 785 &tcache_slow->cache_bin_array_descriptor, link); 786 tcache_stats_merge(tsdn, tcache_slow->tcache, arena); 787 malloc_mutex_unlock(tsdn, &arena->tcache_ql_mtx); 788 } 789 tcache_slow->arena = NULL; 790 } 791 792 void 793 tcache_arena_reassociate(tsdn_t *tsdn, tcache_slow_t *tcache_slow, 794 tcache_t *tcache, arena_t *arena) { 795 tcache_arena_dissociate(tsdn, tcache_slow, tcache); 796 tcache_arena_associate(tsdn, tcache_slow, tcache, arena); 797 } 798 799 static void 800 tcache_default_settings_init(tcache_slow_t *tcache_slow) { 801 assert(tcache_slow != NULL); 802 assert(global_do_not_change_tcache_maxclass != 0); 803 assert(global_do_not_change_tcache_nbins != 0); 804 tcache_slow->tcache_nbins = global_do_not_change_tcache_nbins; 805 } 806 807 static void 808 tcache_init(tsd_t *tsd, tcache_slow_t *tcache_slow, tcache_t *tcache, void *mem, 809 const cache_bin_info_t *tcache_bin_info) { 810 tcache->tcache_slow = tcache_slow; 811 tcache_slow->tcache = tcache; 812 813 memset(&tcache_slow->link, 0, sizeof(ql_elm(tcache_t))); 814 nstime_init_zero(&tcache_slow->last_gc_time); 815 tcache_slow->next_gc_bin = 0; 816 tcache_slow->next_gc_bin_small = 0; 817 tcache_slow->next_gc_bin_large = SC_NBINS; 818 tcache_slow->arena = NULL; 819 tcache_slow->dyn_alloc = mem; 820 821 /* 822 * We reserve cache bins for all small size classes, even if some may 823 * not get used (i.e. bins higher than tcache_nbins). This allows 824 * the fast and common paths to access cache bin metadata safely w/o 825 * worrying about which ones are disabled. 826 */ 827 unsigned tcache_nbins = tcache_nbins_get(tcache_slow); 828 size_t cur_offset = 0; 829 cache_bin_preincrement(tcache_bin_info, tcache_nbins, mem, &cur_offset); 830 for (unsigned i = 0; i < tcache_nbins; i++) { 831 if (i < SC_NBINS) { 832 tcache_bin_fill_ctl_init(tcache_slow, i); 833 tcache_slow->bin_refilled[i] = false; 834 tcache_slow->bin_flush_delay_items[i] = 835 tcache_gc_item_delay_compute(i); 836 } 837 cache_bin_t *cache_bin = &tcache->bins[i]; 838 if (tcache_bin_info[i].ncached_max > 0) { 839 cache_bin_init( 840 cache_bin, &tcache_bin_info[i], mem, &cur_offset); 841 } else { 842 cache_bin_init_disabled( 843 cache_bin, tcache_bin_info[i].ncached_max); 844 } 845 } 846 /* 847 * Initialize all disabled bins to a state that can safely and 848 * efficiently fail all fastpath alloc / free, so that no additional 849 * check around tcache_nbins is needed on fastpath. Yet we still 850 * store the ncached_max in the bin_info for future usage. 851 */ 852 for (unsigned i = tcache_nbins; i < TCACHE_NBINS_MAX; i++) { 853 cache_bin_t *cache_bin = &tcache->bins[i]; 854 cache_bin_init_disabled( 855 cache_bin, tcache_bin_info[i].ncached_max); 856 assert(tcache_bin_disabled(i, cache_bin, tcache->tcache_slow)); 857 } 858 859 cache_bin_postincrement(mem, &cur_offset); 860 if (config_debug) { 861 /* Sanity check that the whole stack is used. */ 862 size_t size, alignment; 863 cache_bin_info_compute_alloc( 864 tcache_bin_info, tcache_nbins, &size, &alignment); 865 assert(cur_offset == size); 866 } 867 } 868 869 static inline unsigned 870 tcache_ncached_max_compute(szind_t szind) { 871 if (szind >= SC_NBINS) { 872 return opt_tcache_nslots_large; 873 } 874 unsigned slab_nregs = bin_infos[szind].nregs; 875 876 /* We may modify these values; start with the opt versions. */ 877 unsigned nslots_small_min = opt_tcache_nslots_small_min; 878 unsigned nslots_small_max = opt_tcache_nslots_small_max; 879 880 /* 881 * Clamp values to meet our constraints -- even, nonzero, min < max, and 882 * suitable for a cache bin size. 883 */ 884 if (opt_tcache_nslots_small_max > CACHE_BIN_NCACHED_MAX) { 885 nslots_small_max = CACHE_BIN_NCACHED_MAX; 886 } 887 if (nslots_small_min % 2 != 0) { 888 nslots_small_min++; 889 } 890 if (nslots_small_max % 2 != 0) { 891 nslots_small_max--; 892 } 893 if (nslots_small_min < 2) { 894 nslots_small_min = 2; 895 } 896 if (nslots_small_max < 2) { 897 nslots_small_max = 2; 898 } 899 if (nslots_small_min > nslots_small_max) { 900 nslots_small_min = nslots_small_max; 901 } 902 903 unsigned candidate; 904 if (opt_lg_tcache_nslots_mul < 0) { 905 candidate = slab_nregs >> (-opt_lg_tcache_nslots_mul); 906 } else { 907 candidate = slab_nregs << opt_lg_tcache_nslots_mul; 908 } 909 if (candidate % 2 != 0) { 910 /* 911 * We need the candidate size to be even -- we assume that we 912 * can divide by two and get a positive number (e.g. when 913 * flushing). 914 */ 915 ++candidate; 916 } 917 if (candidate <= nslots_small_min) { 918 return nslots_small_min; 919 } else if (candidate <= nslots_small_max) { 920 return candidate; 921 } else { 922 return nslots_small_max; 923 } 924 } 925 926 JET_EXTERN void 927 tcache_bin_info_compute(cache_bin_info_t tcache_bin_info[TCACHE_NBINS_MAX]) { 928 /* 929 * Compute the values for each bin, but for bins with indices larger 930 * than tcache_nbins, no items will be cached. 931 */ 932 for (szind_t i = 0; i < TCACHE_NBINS_MAX; i++) { 933 unsigned ncached_max = tcache_get_default_ncached_max_set(i) 934 ? (unsigned)tcache_get_default_ncached_max()[i].ncached_max 935 : tcache_ncached_max_compute(i); 936 assert(ncached_max <= CACHE_BIN_NCACHED_MAX); 937 cache_bin_info_init( 938 &tcache_bin_info[i], (cache_bin_sz_t)ncached_max); 939 } 940 } 941 942 static void * 943 tcache_stack_alloc_impl(tsdn_t *tsdn, size_t size, size_t alignment) { 944 if (cache_bin_stack_use_thp()) { 945 /* Alignment is ignored since it comes from THP. */ 946 assert(alignment == QUANTUM); 947 return b0_alloc_tcache_stack(tsdn, size); 948 } 949 size = sz_sa2u(size, alignment); 950 return ipallocztm(tsdn, size, alignment, true, NULL, 951 true, arena_get(TSDN_NULL, 0, true)); 952 } 953 954 void *(*JET_MUTABLE tcache_stack_alloc)(tsdn_t *tsdn, size_t size, 955 size_t alignment) = tcache_stack_alloc_impl; 956 957 static bool 958 tsd_tcache_data_init_impl( 959 tsd_t *tsd, arena_t *arena, const cache_bin_info_t *tcache_bin_info) { 960 tcache_slow_t *tcache_slow = tsd_tcache_slowp_get_unsafe(tsd); 961 tcache_t *tcache = tsd_tcachep_get_unsafe(tsd); 962 963 assert(cache_bin_still_zero_initialized(&tcache->bins[0])); 964 unsigned tcache_nbins = tcache_nbins_get(tcache_slow); 965 size_t size, alignment; 966 cache_bin_info_compute_alloc( 967 tcache_bin_info, tcache_nbins, &size, &alignment); 968 969 void *mem = tcache_stack_alloc(tsd_tsdn(tsd), size, alignment); 970 if (mem == NULL) { 971 return true; 972 } 973 974 tcache_init(tsd, tcache_slow, tcache, mem, tcache_bin_info); 975 /* 976 * Initialization is a bit tricky here. After malloc init is done, all 977 * threads can rely on arena_choose and associate tcache accordingly. 978 * However, the thread that does actual malloc bootstrapping relies on 979 * functional tsd, and it can only rely on a0. In that case, we 980 * associate its tcache to a0 temporarily, and later on 981 * arena_choose_hard() will re-associate properly. 982 */ 983 tcache_slow->arena = NULL; 984 if (!malloc_initialized()) { 985 /* If in initialization, assign to a0. */ 986 arena = arena_get(tsd_tsdn(tsd), 0, false); 987 tcache_arena_associate( 988 tsd_tsdn(tsd), tcache_slow, tcache, arena); 989 } else { 990 if (arena == NULL) { 991 arena = arena_choose(tsd, NULL); 992 } 993 /* This may happen if thread.tcache.enabled is used. */ 994 if (tcache_slow->arena == NULL) { 995 tcache_arena_associate( 996 tsd_tsdn(tsd), tcache_slow, tcache, arena); 997 } 998 } 999 assert(arena == tcache_slow->arena); 1000 1001 return false; 1002 } 1003 1004 /* Initialize auto tcache (embedded in TSD). */ 1005 static bool 1006 tsd_tcache_data_init(tsd_t *tsd, arena_t *arena, 1007 const cache_bin_info_t tcache_bin_info[TCACHE_NBINS_MAX]) { 1008 assert(tcache_bin_info != NULL); 1009 bool err = tsd_tcache_data_init_impl(tsd, arena, tcache_bin_info); 1010 if (unlikely(err)) { 1011 /* 1012 * Disable the tcache before calling malloc_write to 1013 * avoid recursive allocations through libc hooks. 1014 */ 1015 tsd_tcache_enabled_set(tsd, false); 1016 tsd_slow_update(tsd); 1017 malloc_write("<jemalloc>: Failed to allocate tcache data\n"); 1018 if (opt_abort) { 1019 abort(); 1020 } 1021 } 1022 return err; 1023 } 1024 1025 /* Created manual tcache for tcache.create mallctl. */ 1026 tcache_t * 1027 tcache_create_explicit(tsd_t *tsd) { 1028 /* 1029 * We place the cache bin stacks, then the tcache_t, then a pointer to 1030 * the beginning of the whole allocation (for freeing). The makes sure 1031 * the cache bins have the requested alignment. 1032 */ 1033 unsigned tcache_nbins = global_do_not_change_tcache_nbins; 1034 size_t tcache_size, alignment; 1035 cache_bin_info_compute_alloc(tcache_get_default_ncached_max(), 1036 tcache_nbins, &tcache_size, &alignment); 1037 1038 size_t size = tcache_size + sizeof(tcache_t) + sizeof(tcache_slow_t); 1039 /* Naturally align the pointer stacks. */ 1040 size = PTR_CEILING(size); 1041 size = sz_sa2u(size, alignment); 1042 1043 void *mem = ipallocztm(tsd_tsdn(tsd), size, alignment, true, NULL, true, 1044 arena_get(TSDN_NULL, 0, true)); 1045 if (mem == NULL) { 1046 return NULL; 1047 } 1048 tcache_t *tcache = (void *)((byte_t *)mem + tcache_size); 1049 tcache_slow_t *tcache_slow = (void *)((byte_t *)mem + tcache_size 1050 + sizeof(tcache_t)); 1051 tcache_default_settings_init(tcache_slow); 1052 tcache_init( 1053 tsd, tcache_slow, tcache, mem, tcache_get_default_ncached_max()); 1054 1055 tcache_arena_associate( 1056 tsd_tsdn(tsd), tcache_slow, tcache, arena_ichoose(tsd, NULL)); 1057 1058 return tcache; 1059 } 1060 1061 bool 1062 tsd_tcache_enabled_data_init(tsd_t *tsd) { 1063 /* Called upon tsd initialization. */ 1064 tsd_tcache_enabled_set(tsd, opt_tcache); 1065 /* 1066 * tcache is not available yet, but we need to set up its tcache_nbins 1067 * in advance. 1068 */ 1069 tcache_default_settings_init(tsd_tcache_slowp_get(tsd)); 1070 tsd_slow_update(tsd); 1071 1072 if (opt_tcache) { 1073 /* Trigger tcache init. */ 1074 return tsd_tcache_data_init( 1075 tsd, NULL, tcache_get_default_ncached_max()); 1076 } 1077 1078 return false; 1079 } 1080 1081 void 1082 tcache_enabled_set(tsd_t *tsd, bool enabled) { 1083 bool was_enabled = tsd_tcache_enabled_get(tsd); 1084 1085 if (!was_enabled && enabled) { 1086 if (tsd_tcache_data_init( 1087 tsd, NULL, tcache_get_default_ncached_max())) { 1088 return; 1089 } 1090 } else if (was_enabled && !enabled) { 1091 tcache_cleanup(tsd); 1092 } 1093 /* Commit the state last. Above calls check current state. */ 1094 tsd_tcache_enabled_set(tsd, enabled); 1095 tsd_slow_update(tsd); 1096 } 1097 1098 bool 1099 thread_tcache_max_set(tsd_t *tsd, size_t tcache_max) { 1100 assert(tcache_max <= TCACHE_MAXCLASS_LIMIT); 1101 assert(tcache_max == sz_s2u(tcache_max)); 1102 tcache_t *tcache = tsd_tcachep_get(tsd); 1103 tcache_slow_t *tcache_slow = tcache->tcache_slow; 1104 cache_bin_info_t tcache_bin_info[TCACHE_NBINS_MAX] = {{0}}; 1105 bool ret = false; 1106 assert(tcache != NULL && tcache_slow != NULL); 1107 1108 bool enabled = tcache_available(tsd); 1109 arena_t *assigned_arena JEMALLOC_CLANG_ANALYZER_SILENCE_INIT(NULL); 1110 if (enabled) { 1111 assigned_arena = tcache_slow->arena; 1112 /* Carry over the bin settings during the reboot. */ 1113 tcache_bin_settings_backup(tcache, tcache_bin_info); 1114 /* Shutdown and reboot the tcache for a clean slate. */ 1115 tcache_cleanup(tsd); 1116 } 1117 1118 /* 1119 * Still set tcache_nbins of the tcache even if the tcache is not 1120 * available yet because the values are stored in tsd_t and are 1121 * always available for changing. 1122 */ 1123 tcache_max_set(tcache_slow, tcache_max); 1124 1125 if (enabled) { 1126 ret = tsd_tcache_data_init(tsd, assigned_arena, tcache_bin_info); 1127 } 1128 1129 assert(tcache_nbins_get(tcache_slow) == sz_size2index(tcache_max) + 1); 1130 return ret; 1131 } 1132 1133 static bool 1134 tcache_bin_info_settings_parse(const char *bin_settings_segment_cur, 1135 size_t len_left, cache_bin_info_t tcache_bin_info[TCACHE_NBINS_MAX], 1136 bool bin_info_is_set[TCACHE_NBINS_MAX]) { 1137 do { 1138 size_t size_start, size_end; 1139 size_t ncached_max; 1140 bool err = multi_setting_parse_next(&bin_settings_segment_cur, 1141 &len_left, &size_start, &size_end, &ncached_max); 1142 if (err) { 1143 return true; 1144 } 1145 if (size_end > TCACHE_MAXCLASS_LIMIT) { 1146 size_end = TCACHE_MAXCLASS_LIMIT; 1147 } 1148 if (size_start > TCACHE_MAXCLASS_LIMIT 1149 || size_start > size_end) { 1150 continue; 1151 } 1152 /* May get called before sz_init (during malloc_conf_init). */ 1153 szind_t bin_start = sz_size2index_compute(size_start); 1154 szind_t bin_end = sz_size2index_compute(size_end); 1155 if (ncached_max > CACHE_BIN_NCACHED_MAX) { 1156 ncached_max = (size_t)CACHE_BIN_NCACHED_MAX; 1157 } 1158 for (szind_t i = bin_start; i <= bin_end; i++) { 1159 cache_bin_info_init( 1160 &tcache_bin_info[i], (cache_bin_sz_t)ncached_max); 1161 if (bin_info_is_set != NULL) { 1162 bin_info_is_set[i] = true; 1163 } 1164 } 1165 } while (len_left > 0); 1166 1167 return false; 1168 } 1169 1170 bool 1171 tcache_bin_info_default_init( 1172 const char *bin_settings_segment_cur, size_t len_left) { 1173 return tcache_bin_info_settings_parse(bin_settings_segment_cur, 1174 len_left, opt_tcache_ncached_max, opt_tcache_ncached_max_set); 1175 } 1176 1177 bool 1178 tcache_bins_ncached_max_write(tsd_t *tsd, char *settings, size_t len) { 1179 assert(tcache_available(tsd)); 1180 assert(len != 0); 1181 tcache_t *tcache = tsd_tcachep_get(tsd); 1182 assert(tcache != NULL); 1183 cache_bin_info_t tcache_bin_info[TCACHE_NBINS_MAX]; 1184 tcache_bin_settings_backup(tcache, tcache_bin_info); 1185 1186 if (tcache_bin_info_settings_parse( 1187 settings, len, tcache_bin_info, NULL)) { 1188 return true; 1189 } 1190 1191 arena_t *assigned_arena = tcache->tcache_slow->arena; 1192 tcache_cleanup(tsd); 1193 return tsd_tcache_data_init(tsd, assigned_arena, tcache_bin_info); 1194 } 1195 1196 static void 1197 tcache_flush_cache(tsd_t *tsd, tcache_t *tcache) { 1198 tcache_slow_t *tcache_slow = tcache->tcache_slow; 1199 assert(tcache_slow->arena != NULL); 1200 1201 for (unsigned i = 0; i < tcache_nbins_get(tcache_slow); i++) { 1202 cache_bin_t *cache_bin = &tcache->bins[i]; 1203 if (tcache_bin_disabled(i, cache_bin, tcache_slow)) { 1204 continue; 1205 } 1206 if (i < SC_NBINS) { 1207 tcache_bin_flush_small(tsd, tcache, cache_bin, i, 0); 1208 } else { 1209 tcache_bin_flush_large(tsd, tcache, cache_bin, i, 0); 1210 } 1211 if (config_stats) { 1212 assert(cache_bin->tstats.nrequests == 0); 1213 } 1214 } 1215 } 1216 1217 void 1218 tcache_flush(tsd_t *tsd) { 1219 assert(tcache_available(tsd)); 1220 tcache_flush_cache(tsd, tsd_tcachep_get(tsd)); 1221 } 1222 1223 static void 1224 tcache_destroy(tsd_t *tsd, tcache_t *tcache, bool tsd_tcache) { 1225 tcache_slow_t *tcache_slow = tcache->tcache_slow; 1226 tcache_flush_cache(tsd, tcache); 1227 arena_t *arena = tcache_slow->arena; 1228 tcache_arena_dissociate(tsd_tsdn(tsd), tcache_slow, tcache); 1229 1230 if (tsd_tcache) { 1231 cache_bin_t *cache_bin = &tcache->bins[0]; 1232 cache_bin_assert_empty(cache_bin); 1233 } 1234 if (tsd_tcache && cache_bin_stack_use_thp()) { 1235 b0_dalloc_tcache_stack(tsd_tsdn(tsd), tcache_slow->dyn_alloc); 1236 } else { 1237 idalloctm(tsd_tsdn(tsd), tcache_slow->dyn_alloc, NULL, NULL, 1238 true, true); 1239 } 1240 1241 /* 1242 * The deallocation and tcache flush above may not trigger decay since 1243 * we are on the tcache shutdown path (potentially with non-nominal 1244 * tsd). Manually trigger decay to avoid pathological cases. Also 1245 * include arena 0 because the tcache array is allocated from it. 1246 */ 1247 arena_decay( 1248 tsd_tsdn(tsd), arena_get(tsd_tsdn(tsd), 0, false), false, false); 1249 1250 if (arena_nthreads_get(arena, false) == 0 1251 && !background_thread_enabled()) { 1252 /* Force purging when no threads assigned to the arena anymore. */ 1253 arena_decay(tsd_tsdn(tsd), arena, 1254 /* is_background_thread */ false, /* all */ true); 1255 } else { 1256 arena_decay(tsd_tsdn(tsd), arena, 1257 /* is_background_thread */ false, /* all */ false); 1258 } 1259 } 1260 1261 /* For auto tcache (embedded in TSD) only. */ 1262 void 1263 tcache_cleanup(tsd_t *tsd) { 1264 tcache_t *tcache = tsd_tcachep_get(tsd); 1265 if (!tcache_available(tsd)) { 1266 assert(tsd_tcache_enabled_get(tsd) == false); 1267 assert(cache_bin_still_zero_initialized(&tcache->bins[0])); 1268 return; 1269 } 1270 assert(tsd_tcache_enabled_get(tsd)); 1271 assert(!cache_bin_still_zero_initialized(&tcache->bins[0])); 1272 1273 tcache_destroy(tsd, tcache, true); 1274 /* Make sure all bins used are reinitialized to the clean state. */ 1275 memset(tcache->bins, 0, sizeof(cache_bin_t) * TCACHE_NBINS_MAX); 1276 } 1277 1278 void 1279 tcache_stats_merge(tsdn_t *tsdn, tcache_t *tcache, arena_t *arena) { 1280 cassert(config_stats); 1281 1282 /* Merge and reset tcache stats. */ 1283 for (unsigned i = 0; i < tcache_nbins_get(tcache->tcache_slow); i++) { 1284 cache_bin_t *cache_bin = &tcache->bins[i]; 1285 if (tcache_bin_disabled(i, cache_bin, tcache->tcache_slow)) { 1286 continue; 1287 } 1288 if (i < SC_NBINS) { 1289 bin_t *bin = bin_choose(tsdn, arena, i, NULL); 1290 malloc_mutex_lock(tsdn, &bin->lock); 1291 bin->stats.nrequests += cache_bin->tstats.nrequests; 1292 malloc_mutex_unlock(tsdn, &bin->lock); 1293 } else { 1294 arena_stats_large_flush_nrequests_add(tsdn, 1295 &arena->stats, i, cache_bin->tstats.nrequests); 1296 } 1297 cache_bin->tstats.nrequests = 0; 1298 } 1299 } 1300 1301 static bool 1302 tcaches_create_prep(tsd_t *tsd, base_t *base) { 1303 bool err; 1304 1305 malloc_mutex_assert_owner(tsd_tsdn(tsd), &tcaches_mtx); 1306 1307 if (tcaches == NULL) { 1308 tcaches = base_alloc(tsd_tsdn(tsd), base, 1309 sizeof(tcache_t *) * (MALLOCX_TCACHE_MAX + 1), CACHELINE); 1310 if (tcaches == NULL) { 1311 err = true; 1312 goto label_return; 1313 } 1314 } 1315 1316 if (tcaches_avail == NULL && tcaches_past > MALLOCX_TCACHE_MAX) { 1317 err = true; 1318 goto label_return; 1319 } 1320 1321 err = false; 1322 label_return: 1323 return err; 1324 } 1325 1326 bool 1327 tcaches_create(tsd_t *tsd, base_t *base, unsigned *r_ind) { 1328 witness_assert_depth(tsdn_witness_tsdp_get(tsd_tsdn(tsd)), 0); 1329 1330 bool err; 1331 1332 malloc_mutex_lock(tsd_tsdn(tsd), &tcaches_mtx); 1333 1334 if (tcaches_create_prep(tsd, base)) { 1335 err = true; 1336 goto label_return; 1337 } 1338 1339 tcache_t *tcache = tcache_create_explicit(tsd); 1340 if (tcache == NULL) { 1341 err = true; 1342 goto label_return; 1343 } 1344 1345 tcaches_t *elm; 1346 if (tcaches_avail != NULL) { 1347 elm = tcaches_avail; 1348 tcaches_avail = tcaches_avail->next; 1349 elm->tcache = tcache; 1350 *r_ind = (unsigned)(elm - tcaches); 1351 } else { 1352 elm = &tcaches[tcaches_past]; 1353 elm->tcache = tcache; 1354 *r_ind = tcaches_past; 1355 tcaches_past++; 1356 } 1357 1358 err = false; 1359 label_return: 1360 malloc_mutex_unlock(tsd_tsdn(tsd), &tcaches_mtx); 1361 witness_assert_depth(tsdn_witness_tsdp_get(tsd_tsdn(tsd)), 0); 1362 return err; 1363 } 1364 1365 static tcache_t * 1366 tcaches_elm_remove(tsd_t *tsd, tcaches_t *elm, bool allow_reinit) { 1367 malloc_mutex_assert_owner(tsd_tsdn(tsd), &tcaches_mtx); 1368 1369 if (elm->tcache == NULL) { 1370 return NULL; 1371 } 1372 tcache_t *tcache = elm->tcache; 1373 if (allow_reinit) { 1374 elm->tcache = TCACHES_ELM_NEED_REINIT; 1375 } else { 1376 elm->tcache = NULL; 1377 } 1378 1379 if (tcache == TCACHES_ELM_NEED_REINIT) { 1380 return NULL; 1381 } 1382 return tcache; 1383 } 1384 1385 void 1386 tcaches_flush(tsd_t *tsd, unsigned ind) { 1387 malloc_mutex_lock(tsd_tsdn(tsd), &tcaches_mtx); 1388 tcache_t *tcache = tcaches_elm_remove(tsd, &tcaches[ind], true); 1389 malloc_mutex_unlock(tsd_tsdn(tsd), &tcaches_mtx); 1390 if (tcache != NULL) { 1391 /* Destroy the tcache; recreate in tcaches_get() if needed. */ 1392 tcache_destroy(tsd, tcache, false); 1393 } 1394 } 1395 1396 void 1397 tcaches_destroy(tsd_t *tsd, unsigned ind) { 1398 malloc_mutex_lock(tsd_tsdn(tsd), &tcaches_mtx); 1399 tcaches_t *elm = &tcaches[ind]; 1400 tcache_t *tcache = tcaches_elm_remove(tsd, elm, false); 1401 elm->next = tcaches_avail; 1402 tcaches_avail = elm; 1403 malloc_mutex_unlock(tsd_tsdn(tsd), &tcaches_mtx); 1404 if (tcache != NULL) { 1405 tcache_destroy(tsd, tcache, false); 1406 } 1407 } 1408 1409 bool 1410 tcache_boot(tsdn_t *tsdn, base_t *base) { 1411 global_do_not_change_tcache_maxclass = sz_s2u(opt_tcache_max); 1412 assert(global_do_not_change_tcache_maxclass <= TCACHE_MAXCLASS_LIMIT); 1413 global_do_not_change_tcache_nbins = 1414 sz_size2index(global_do_not_change_tcache_maxclass) + 1; 1415 /* 1416 * Pre-compute default bin info and store the results in 1417 * opt_tcache_ncached_max. After the changes here, 1418 * opt_tcache_ncached_max should not be modified and should always be 1419 * accessed using tcache_get_default_ncached_max. 1420 */ 1421 tcache_bin_info_compute(opt_tcache_ncached_max); 1422 1423 if (malloc_mutex_init(&tcaches_mtx, "tcaches", WITNESS_RANK_TCACHES, 1424 malloc_mutex_rank_exclusive)) { 1425 return true; 1426 } 1427 1428 return false; 1429 } 1430 1431 void 1432 tcache_prefork(tsdn_t *tsdn) { 1433 malloc_mutex_prefork(tsdn, &tcaches_mtx); 1434 } 1435 1436 void 1437 tcache_postfork_parent(tsdn_t *tsdn) { 1438 malloc_mutex_postfork_parent(tsdn, &tcaches_mtx); 1439 } 1440 1441 void 1442 tcache_postfork_child(tsdn_t *tsdn) { 1443 malloc_mutex_postfork_child(tsdn, &tcaches_mtx); 1444 } 1445 1446 void 1447 tcache_assert_initialized(tcache_t *tcache) { 1448 assert(!cache_bin_still_zero_initialized(&tcache->bins[0])); 1449 } 1450 1451 static te_enabled_t 1452 tcache_gc_enabled(void) { 1453 return (opt_tcache_gc_incr_bytes > 0) ? te_enabled_yes : te_enabled_no; 1454 } 1455 1456 /* Handles alloc and dalloc the same way */ 1457 te_base_cb_t tcache_gc_te_handler = { 1458 .enabled = &tcache_gc_enabled, 1459 .new_event_wait = &tcache_gc_new_event_wait, 1460 .postponed_event_wait = &tcache_gc_postponed_event_wait, 1461 .event_handler = &tcache_gc_event, 1462 }; 1463