Home | History | Annotate | Line # | Download | only in internal
      1 #ifndef JEMALLOC_INTERNAL_PAC_H
      2 #define JEMALLOC_INTERNAL_PAC_H
      3 
      4 #include "jemalloc/internal/jemalloc_preamble.h"
      5 #include "jemalloc/internal/decay.h"
      6 #include "jemalloc/internal/ecache.h"
      7 #include "jemalloc/internal/edata.h"
      8 #include "jemalloc/internal/edata_cache.h"
      9 #include "jemalloc/internal/exp_grow.h"
     10 #include "jemalloc/internal/lockedint.h"
     11 #include "jemalloc/internal/sec.h"
     12 #include "jemalloc/internal/tsd_types.h"
     13 #include "san_bump.h"
     14 
     15 /*
     16  * Page allocator classic (PAC), a page-level allocator that:
     17  * - Can be used for arenas with custom extent hooks.
     18  * - Can always satisfy any allocation request (including highly-fragmentary
     19  *   ones).
     20  * - Can use efficient OS-level zeroing primitives for demand-filled pages.
     21  */
     22 
     23 /* How "eager" decay/purging should be. */
     24 enum pac_purge_eagerness_e {
     25 	PAC_PURGE_ALWAYS,
     26 	PAC_PURGE_NEVER,
     27 	PAC_PURGE_ON_EPOCH_ADVANCE
     28 };
     29 typedef enum pac_purge_eagerness_e pac_purge_eagerness_t;
     30 
     31 /*
     32  * When a decay sweep would purge more than this many pages, its purge is
     33  * treated as due (worth waking the background thread for) instead of left
     34  * to accumulate.  PAC decay policy; the HPA defers on time intervals, not
     35  * on a page count, so this constant is PAC-only.
     36  */
     37 #define PAC_DECAY_PURGE_NPAGES_THRESHOLD UINT64_C(1024)
     38 
     39 typedef struct pac_decay_stats_s pac_decay_stats_t;
     40 struct pac_decay_stats_s {
     41 	/* Total number of purge sweeps. */
     42 	locked_u64_t npurge;
     43 	/* Total number of madvise calls made. */
     44 	locked_u64_t nmadvise;
     45 	/* Total number of pages purged. */
     46 	locked_u64_t purged;
     47 };
     48 
     49 typedef struct pac_estats_s pac_estats_t;
     50 struct pac_estats_s {
     51 	/*
     52 	 * Stats for a given index in the range [0, SC_NPSIZES] in the various
     53 	 * ecache_ts.
     54 	 * We track both bytes and # of extents: two extents in the same bucket
     55 	 * may have different sizes if adjacent size classes differ by more than
     56 	 * a page, so bytes cannot always be derived from # of extents.
     57 	 */
     58 	size_t ndirty;
     59 	size_t dirty_bytes;
     60 	size_t nmuzzy;
     61 	size_t muzzy_bytes;
     62 	size_t nretained;
     63 	size_t retained_bytes;
     64 	size_t npinned;
     65 	size_t pinned_bytes;
     66 };
     67 
     68 typedef struct pac_stats_s pac_stats_t;
     69 struct pac_stats_s {
     70 	pac_decay_stats_t decay_dirty;
     71 	pac_decay_stats_t decay_muzzy;
     72 
     73 	/*
     74 	 * Number of unused virtual memory bytes currently retained.  Retained
     75 	 * bytes are technically mapped (though always decommitted or purged),
     76 	 * but they are excluded from pac_mapped.
     77 	 */
     78 	size_t retained; /* Derived. */
     79 	/*
     80 	 * Number of bytes in pinned (non-reclaimable) extents currently
     81 	 * cached.  Unlike retained, pinned bytes count toward pac_mapped.
     82 	 */
     83 	size_t pinned;   /* Derived. */
     84 
     85 	/*
     86 	 * Number of bytes currently mapped, excluding retained memory (and any
     87 	 * base-allocated memory, which is tracked by the arena stats).
     88 	 *
     89 	 * We name this "pac_mapped" to avoid confusion with the arena_stats
     90 	 * "mapped".
     91 	 */
     92 	atomic_zu_t pac_mapped;
     93 
     94 	/* VM space had to be leaked (undocumented).  Normally 0. */
     95 	atomic_zu_t abandoned_vm;
     96 
     97 	/* PAC SEC stats.  Derived. */
     98 	sec_stats_t pac_sec_stats;
     99 };
    100 
    101 typedef struct pac_s pac_t;
    102 struct pac_s {
    103 	/* Small extent cache in front of PAC ecaches to reduce contention. */
    104 	sec_t sec;
    105 	/*
    106 	 * Runtime gate for PAC SEC.  0 disables (when SEC is not configured or
    107 	 * dirty_decay_ms == 0); otherwise mirrors sec.opts.max_alloc.
    108 	 */
    109 	atomic_zu_t sec_max_alloc;
    110 
    111 	/* True once pinned memory has been seen. */
    112 	atomic_b_t has_pinned;
    113 
    114 	/*
    115 	 * Collections of extents that were previously allocated.  These are
    116 	 * used when allocating extents, in an attempt to re-use address space.
    117 	 *
    118 	 * Synchronization: internal.
    119 	 */
    120 	ecache_t ecache_dirty;
    121 	ecache_t ecache_muzzy;
    122 	ecache_t ecache_retained;
    123 	ecache_t ecache_pinned;
    124 
    125 	base_t        *base;
    126 	emap_t        *emap;
    127 	edata_cache_t *edata_cache;
    128 
    129 	/* The grow info for the retained ecache. */
    130 	exp_grow_t     exp_grow;
    131 	malloc_mutex_t grow_mtx;
    132 
    133 	/* Special allocator for guarded frequently reused extents. */
    134 	san_bump_alloc_t sba;
    135 
    136 	/* How large extents should be before getting auto-purged. */
    137 	atomic_zu_t oversize_threshold;
    138 
    139 	/*
    140 	 * Decay-based purging state, responsible for scheduling extent state
    141 	 * transitions.
    142 	 *
    143 	 * Synchronization: via the internal mutex.
    144 	 */
    145 	decay_t decay_dirty; /* dirty --> muzzy */
    146 	decay_t decay_muzzy; /* muzzy --> retained */
    147 
    148 	malloc_mutex_t *stats_mtx;
    149 	pac_stats_t    *stats;
    150 
    151 	/* Extent serial number generator state. */
    152 	atomic_zu_t extent_sn_next;
    153 };
    154 
    155 typedef struct pac_thp_s pac_thp_t;
    156 struct pac_thp_s {
    157 	/*
    158 	 * opt_thp controls THP for user requested allocations. Settings
    159 	 * "always", "never" and "default" are available if THP is supported
    160 	 * by the OS and the default extent hooks are used:
    161 	 * - "always" and "never" are covered by pages_set_thp_state() in
    162 	 *   ehooks_default_alloc_impl().
    163 	 * - "default" makes no change for all the other auto arenas except
    164 	 *   the huge arena. For the huge arena, we might also look at
    165 	 *   opt_metadata_thp to decide whether to use THP or not.
    166 	 *   This is a temporary remedy before HPA is fully supported.
    167 	 */
    168 	bool thp_madvise;
    169 	/* Below fields are protected by the lock. */
    170 	malloc_mutex_t lock;
    171 	bool           auto_thp_switched;
    172 	atomic_u_t     n_thp_lazy;
    173 	/*
    174 	 * List that tracks HUGEPAGE aligned regions that're lazily hugified
    175 	 * in auto thp mode.
    176 	 */
    177 	edata_list_active_t thp_lazy_list;
    178 };
    179 
    180 bool pac_init(tsdn_t *tsdn, pac_t *pac, base_t *base, emap_t *emap,
    181     edata_cache_t *edata_cache, nstime_t *cur_time, size_t oversize_threshold,
    182     ssize_t dirty_decay_ms, ssize_t muzzy_decay_ms, pac_stats_t *pac_stats,
    183     malloc_mutex_t *stats_mtx);
    184 
    185 edata_t *pac_alloc(tsdn_t *tsdn, pac_t *pac, size_t size, size_t alignment,
    186     bool zero, bool guarded, bool frequent_reuse,
    187     bool *deferred_work_generated);
    188 bool pac_expand(tsdn_t *tsdn, pac_t *pac, edata_t *edata, size_t old_size,
    189     size_t new_size, bool zero, bool *deferred_work_generated);
    190 bool pac_shrink(tsdn_t *tsdn, pac_t *pac, edata_t *edata, size_t old_size,
    191     size_t new_size, bool *deferred_work_generated);
    192 void pac_dalloc(tsdn_t *tsdn, pac_t *pac, edata_t *edata,
    193     bool *deferred_work_generated);
    194 uint64_t pac_time_until_deferred_work(tsdn_t *tsdn, pac_t *pac);
    195 
    196 static inline size_t
    197 pac_mapped(const pac_t *pac) {
    198 	return atomic_load_zu(&pac->stats->pac_mapped, ATOMIC_RELAXED);
    199 }
    200 
    201 void extent_record(tsdn_t *tsdn, pac_t *pac, ehooks_t *ehooks,
    202     ecache_t *ecache, edata_t *edata);
    203 
    204 static inline void
    205 pac_record_grown(tsdn_t *tsdn, pac_t *pac, ehooks_t *ehooks,
    206     edata_t *edata) {
    207 	bool pinned = edata_pinned_get(edata);
    208 	if (pinned && config_stats) {
    209 		atomic_fetch_add_zu(&pac->stats->pac_mapped,
    210 		    edata_size_get(edata), ATOMIC_RELAXED);
    211 	}
    212 	extent_record(tsdn, pac, ehooks,
    213 	    pinned ? &pac->ecache_pinned : &pac->ecache_retained, edata);
    214 }
    215 
    216 static inline ehooks_t *
    217 pac_ehooks_get(const pac_t *pac) {
    218 	return base_ehooks_get(pac->base);
    219 }
    220 
    221 /*
    222  * All purging functions require holding decay->mtx.  This is one of the few
    223  * places external modules are allowed to peek inside pa_shard_t internals.
    224  */
    225 
    226 /*
    227  * Decays the number of pages currently in the ecache.  This might not leave the
    228  * ecache empty if other threads are inserting dirty objects into it
    229  * concurrently with the call.
    230  */
    231 void pac_decay_all(tsdn_t *tsdn, pac_t *pac, decay_t *decay,
    232     pac_decay_stats_t *decay_stats, ecache_t *ecache, bool fully_decay);
    233 /*
    234  * Updates decay settings for the current time, and conditionally purges in
    235  * response (depending on decay_purge_setting).  Returns whether or not the
    236  * epoch advanced.
    237  */
    238 bool pac_maybe_decay_purge(tsdn_t *tsdn, pac_t *pac, decay_t *decay,
    239     pac_decay_stats_t *decay_stats, ecache_t *ecache,
    240     pac_purge_eagerness_t eagerness);
    241 
    242 /*
    243  * Result of a pac_decay_deferred() pass.  Reported per decay state so
    244  * pac_do_deferred_work can decide, per state, whether to notify the background
    245  * thread.  npages_new is the per-state epoch backlog delta, only meaningful when
    246  * the corresponding *_epoch_advanced is true.
    247  */
    248 typedef struct pac_deferred_work_result_s pac_deferred_work_result_t;
    249 struct pac_deferred_work_result_s {
    250 	size_t dirty_npages_new;
    251 	size_t muzzy_npages_new;
    252 	bool   dirty_epoch_advanced;
    253 	bool   muzzy_epoch_advanced;
    254 };
    255 
    256 /*
    257  * All deferred decay-purge work for a PAC shard: decide the eagerness from the
    258  * caller context, run pac_decay_deferred, and (application path only) notify the
    259  * background thread for any decay epoch that advanced.  The bg-thread driver
    260  * passes is_background_thread=true and is never self-notified.
    261  */
    262 void pac_do_deferred_work(
    263     tsdn_t *tsdn, pac_t *pac, bool is_background_thread);
    264 
    265 /*
    266  * Application-path hook (after deferred_work_generated): wake the background
    267  * thread if it is sleeping idle.  Runs the same early-wake logic as the notify
    268  * path (pac_maybe_wake_bg), gated on the thread being idle.
    269  */
    270 void pac_wake_bg_on_deferred(tsdn_t *tsdn, pac_t *pac);
    271 
    272 /*
    273  * Fully decay the extents of the given state, acquiring decay->mtx internally.
    274  */
    275 void pac_decay_all_now(tsdn_t *tsdn, pac_t *pac, extent_state_t state);
    276 
    277 /*
    278  * Gets / sets the maximum amount that we'll grow an arena down the
    279  * grow-retained pathways (unless forced to by an allocaction request).
    280  *
    281  * Set new_limit to NULL if it's just a query, or old_limit to NULL if you don't
    282  * care about the previous value.
    283  *
    284  * Returns true on error (if the new limit is not valid).
    285  */
    286 bool pac_retain_grow_limit_get_set(
    287     tsdn_t *tsdn, pac_t *pac, size_t *old_limit, size_t *new_limit);
    288 
    289 bool    pac_decay_ms_set(tsdn_t *tsdn, pac_t *pac, extent_state_t state,
    290        ssize_t decay_ms);
    291 ssize_t pac_decay_ms_get(pac_t *pac, extent_state_t state);
    292 
    293 /* Whether the muzzy decay path is relevant (muzzy pages exist, or decay is on). */
    294 static inline bool
    295 pac_should_decay_muzzy(pac_t *pac) {
    296 	return ecache_npages_get(&pac->ecache_muzzy) != 0
    297 	    || pac_decay_ms_get(pac, extent_state_muzzy) > 0;
    298 }
    299 
    300 /* Whether dirty decay is immediate (dirty_decay_ms == 0). */
    301 static inline bool
    302 pac_decay_immediately(pac_t *pac) {
    303 	return decay_immediately(&pac->decay_dirty);
    304 }
    305 
    306 void pac_destroy(tsdn_t *tsdn, pac_t *pac);
    307 
    308 void pac_sec_flush(tsdn_t *tsdn, pac_t *pac);
    309 
    310 #endif /* JEMALLOC_INTERNAL_PAC_H */
    311