1 #ifndef JEMALLOC_INTERNAL_PAC_H 2 #define JEMALLOC_INTERNAL_PAC_H 3 4 #include "jemalloc/internal/jemalloc_preamble.h" 5 #include "jemalloc/internal/decay.h" 6 #include "jemalloc/internal/ecache.h" 7 #include "jemalloc/internal/edata.h" 8 #include "jemalloc/internal/edata_cache.h" 9 #include "jemalloc/internal/exp_grow.h" 10 #include "jemalloc/internal/lockedint.h" 11 #include "jemalloc/internal/sec.h" 12 #include "jemalloc/internal/tsd_types.h" 13 #include "san_bump.h" 14 15 /* 16 * Page allocator classic (PAC), a page-level allocator that: 17 * - Can be used for arenas with custom extent hooks. 18 * - Can always satisfy any allocation request (including highly-fragmentary 19 * ones). 20 * - Can use efficient OS-level zeroing primitives for demand-filled pages. 21 */ 22 23 /* How "eager" decay/purging should be. */ 24 enum pac_purge_eagerness_e { 25 PAC_PURGE_ALWAYS, 26 PAC_PURGE_NEVER, 27 PAC_PURGE_ON_EPOCH_ADVANCE 28 }; 29 typedef enum pac_purge_eagerness_e pac_purge_eagerness_t; 30 31 /* 32 * When a decay sweep would purge more than this many pages, its purge is 33 * treated as due (worth waking the background thread for) instead of left 34 * to accumulate. PAC decay policy; the HPA defers on time intervals, not 35 * on a page count, so this constant is PAC-only. 36 */ 37 #define PAC_DECAY_PURGE_NPAGES_THRESHOLD UINT64_C(1024) 38 39 typedef struct pac_decay_stats_s pac_decay_stats_t; 40 struct pac_decay_stats_s { 41 /* Total number of purge sweeps. */ 42 locked_u64_t npurge; 43 /* Total number of madvise calls made. */ 44 locked_u64_t nmadvise; 45 /* Total number of pages purged. */ 46 locked_u64_t purged; 47 }; 48 49 typedef struct pac_estats_s pac_estats_t; 50 struct pac_estats_s { 51 /* 52 * Stats for a given index in the range [0, SC_NPSIZES] in the various 53 * ecache_ts. 54 * We track both bytes and # of extents: two extents in the same bucket 55 * may have different sizes if adjacent size classes differ by more than 56 * a page, so bytes cannot always be derived from # of extents. 57 */ 58 size_t ndirty; 59 size_t dirty_bytes; 60 size_t nmuzzy; 61 size_t muzzy_bytes; 62 size_t nretained; 63 size_t retained_bytes; 64 size_t npinned; 65 size_t pinned_bytes; 66 }; 67 68 typedef struct pac_stats_s pac_stats_t; 69 struct pac_stats_s { 70 pac_decay_stats_t decay_dirty; 71 pac_decay_stats_t decay_muzzy; 72 73 /* 74 * Number of unused virtual memory bytes currently retained. Retained 75 * bytes are technically mapped (though always decommitted or purged), 76 * but they are excluded from pac_mapped. 77 */ 78 size_t retained; /* Derived. */ 79 /* 80 * Number of bytes in pinned (non-reclaimable) extents currently 81 * cached. Unlike retained, pinned bytes count toward pac_mapped. 82 */ 83 size_t pinned; /* Derived. */ 84 85 /* 86 * Number of bytes currently mapped, excluding retained memory (and any 87 * base-allocated memory, which is tracked by the arena stats). 88 * 89 * We name this "pac_mapped" to avoid confusion with the arena_stats 90 * "mapped". 91 */ 92 atomic_zu_t pac_mapped; 93 94 /* VM space had to be leaked (undocumented). Normally 0. */ 95 atomic_zu_t abandoned_vm; 96 97 /* PAC SEC stats. Derived. */ 98 sec_stats_t pac_sec_stats; 99 }; 100 101 typedef struct pac_s pac_t; 102 struct pac_s { 103 /* Small extent cache in front of PAC ecaches to reduce contention. */ 104 sec_t sec; 105 /* 106 * Runtime gate for PAC SEC. 0 disables (when SEC is not configured or 107 * dirty_decay_ms == 0); otherwise mirrors sec.opts.max_alloc. 108 */ 109 atomic_zu_t sec_max_alloc; 110 111 /* True once pinned memory has been seen. */ 112 atomic_b_t has_pinned; 113 114 /* 115 * Collections of extents that were previously allocated. These are 116 * used when allocating extents, in an attempt to re-use address space. 117 * 118 * Synchronization: internal. 119 */ 120 ecache_t ecache_dirty; 121 ecache_t ecache_muzzy; 122 ecache_t ecache_retained; 123 ecache_t ecache_pinned; 124 125 base_t *base; 126 emap_t *emap; 127 edata_cache_t *edata_cache; 128 129 /* The grow info for the retained ecache. */ 130 exp_grow_t exp_grow; 131 malloc_mutex_t grow_mtx; 132 133 /* Special allocator for guarded frequently reused extents. */ 134 san_bump_alloc_t sba; 135 136 /* How large extents should be before getting auto-purged. */ 137 atomic_zu_t oversize_threshold; 138 139 /* 140 * Decay-based purging state, responsible for scheduling extent state 141 * transitions. 142 * 143 * Synchronization: via the internal mutex. 144 */ 145 decay_t decay_dirty; /* dirty --> muzzy */ 146 decay_t decay_muzzy; /* muzzy --> retained */ 147 148 malloc_mutex_t *stats_mtx; 149 pac_stats_t *stats; 150 151 /* Extent serial number generator state. */ 152 atomic_zu_t extent_sn_next; 153 }; 154 155 typedef struct pac_thp_s pac_thp_t; 156 struct pac_thp_s { 157 /* 158 * opt_thp controls THP for user requested allocations. Settings 159 * "always", "never" and "default" are available if THP is supported 160 * by the OS and the default extent hooks are used: 161 * - "always" and "never" are covered by pages_set_thp_state() in 162 * ehooks_default_alloc_impl(). 163 * - "default" makes no change for all the other auto arenas except 164 * the huge arena. For the huge arena, we might also look at 165 * opt_metadata_thp to decide whether to use THP or not. 166 * This is a temporary remedy before HPA is fully supported. 167 */ 168 bool thp_madvise; 169 /* Below fields are protected by the lock. */ 170 malloc_mutex_t lock; 171 bool auto_thp_switched; 172 atomic_u_t n_thp_lazy; 173 /* 174 * List that tracks HUGEPAGE aligned regions that're lazily hugified 175 * in auto thp mode. 176 */ 177 edata_list_active_t thp_lazy_list; 178 }; 179 180 bool pac_init(tsdn_t *tsdn, pac_t *pac, base_t *base, emap_t *emap, 181 edata_cache_t *edata_cache, nstime_t *cur_time, size_t oversize_threshold, 182 ssize_t dirty_decay_ms, ssize_t muzzy_decay_ms, pac_stats_t *pac_stats, 183 malloc_mutex_t *stats_mtx); 184 185 edata_t *pac_alloc(tsdn_t *tsdn, pac_t *pac, size_t size, size_t alignment, 186 bool zero, bool guarded, bool frequent_reuse, 187 bool *deferred_work_generated); 188 bool pac_expand(tsdn_t *tsdn, pac_t *pac, edata_t *edata, size_t old_size, 189 size_t new_size, bool zero, bool *deferred_work_generated); 190 bool pac_shrink(tsdn_t *tsdn, pac_t *pac, edata_t *edata, size_t old_size, 191 size_t new_size, bool *deferred_work_generated); 192 void pac_dalloc(tsdn_t *tsdn, pac_t *pac, edata_t *edata, 193 bool *deferred_work_generated); 194 uint64_t pac_time_until_deferred_work(tsdn_t *tsdn, pac_t *pac); 195 196 static inline size_t 197 pac_mapped(const pac_t *pac) { 198 return atomic_load_zu(&pac->stats->pac_mapped, ATOMIC_RELAXED); 199 } 200 201 void extent_record(tsdn_t *tsdn, pac_t *pac, ehooks_t *ehooks, 202 ecache_t *ecache, edata_t *edata); 203 204 static inline void 205 pac_record_grown(tsdn_t *tsdn, pac_t *pac, ehooks_t *ehooks, 206 edata_t *edata) { 207 bool pinned = edata_pinned_get(edata); 208 if (pinned && config_stats) { 209 atomic_fetch_add_zu(&pac->stats->pac_mapped, 210 edata_size_get(edata), ATOMIC_RELAXED); 211 } 212 extent_record(tsdn, pac, ehooks, 213 pinned ? &pac->ecache_pinned : &pac->ecache_retained, edata); 214 } 215 216 static inline ehooks_t * 217 pac_ehooks_get(const pac_t *pac) { 218 return base_ehooks_get(pac->base); 219 } 220 221 /* 222 * All purging functions require holding decay->mtx. This is one of the few 223 * places external modules are allowed to peek inside pa_shard_t internals. 224 */ 225 226 /* 227 * Decays the number of pages currently in the ecache. This might not leave the 228 * ecache empty if other threads are inserting dirty objects into it 229 * concurrently with the call. 230 */ 231 void pac_decay_all(tsdn_t *tsdn, pac_t *pac, decay_t *decay, 232 pac_decay_stats_t *decay_stats, ecache_t *ecache, bool fully_decay); 233 /* 234 * Updates decay settings for the current time, and conditionally purges in 235 * response (depending on decay_purge_setting). Returns whether or not the 236 * epoch advanced. 237 */ 238 bool pac_maybe_decay_purge(tsdn_t *tsdn, pac_t *pac, decay_t *decay, 239 pac_decay_stats_t *decay_stats, ecache_t *ecache, 240 pac_purge_eagerness_t eagerness); 241 242 /* 243 * Result of a pac_decay_deferred() pass. Reported per decay state so 244 * pac_do_deferred_work can decide, per state, whether to notify the background 245 * thread. npages_new is the per-state epoch backlog delta, only meaningful when 246 * the corresponding *_epoch_advanced is true. 247 */ 248 typedef struct pac_deferred_work_result_s pac_deferred_work_result_t; 249 struct pac_deferred_work_result_s { 250 size_t dirty_npages_new; 251 size_t muzzy_npages_new; 252 bool dirty_epoch_advanced; 253 bool muzzy_epoch_advanced; 254 }; 255 256 /* 257 * All deferred decay-purge work for a PAC shard: decide the eagerness from the 258 * caller context, run pac_decay_deferred, and (application path only) notify the 259 * background thread for any decay epoch that advanced. The bg-thread driver 260 * passes is_background_thread=true and is never self-notified. 261 */ 262 void pac_do_deferred_work( 263 tsdn_t *tsdn, pac_t *pac, bool is_background_thread); 264 265 /* 266 * Application-path hook (after deferred_work_generated): wake the background 267 * thread if it is sleeping idle. Runs the same early-wake logic as the notify 268 * path (pac_maybe_wake_bg), gated on the thread being idle. 269 */ 270 void pac_wake_bg_on_deferred(tsdn_t *tsdn, pac_t *pac); 271 272 /* 273 * Fully decay the extents of the given state, acquiring decay->mtx internally. 274 */ 275 void pac_decay_all_now(tsdn_t *tsdn, pac_t *pac, extent_state_t state); 276 277 /* 278 * Gets / sets the maximum amount that we'll grow an arena down the 279 * grow-retained pathways (unless forced to by an allocaction request). 280 * 281 * Set new_limit to NULL if it's just a query, or old_limit to NULL if you don't 282 * care about the previous value. 283 * 284 * Returns true on error (if the new limit is not valid). 285 */ 286 bool pac_retain_grow_limit_get_set( 287 tsdn_t *tsdn, pac_t *pac, size_t *old_limit, size_t *new_limit); 288 289 bool pac_decay_ms_set(tsdn_t *tsdn, pac_t *pac, extent_state_t state, 290 ssize_t decay_ms); 291 ssize_t pac_decay_ms_get(pac_t *pac, extent_state_t state); 292 293 /* Whether the muzzy decay path is relevant (muzzy pages exist, or decay is on). */ 294 static inline bool 295 pac_should_decay_muzzy(pac_t *pac) { 296 return ecache_npages_get(&pac->ecache_muzzy) != 0 297 || pac_decay_ms_get(pac, extent_state_muzzy) > 0; 298 } 299 300 /* Whether dirty decay is immediate (dirty_decay_ms == 0). */ 301 static inline bool 302 pac_decay_immediately(pac_t *pac) { 303 return decay_immediately(&pac->decay_dirty); 304 } 305 306 void pac_destroy(tsdn_t *tsdn, pac_t *pac); 307 308 void pac_sec_flush(tsdn_t *tsdn, pac_t *pac); 309 310 #endif /* JEMALLOC_INTERNAL_PAC_H */ 311