mirror of
https://github.com/jemalloc/jemalloc.git
synced 2026-07-23 13:13:06 +00:00
Add batching to arena bins.
This adds a fast-path for threads freeing a small number of allocations to bins which are not their "home-base" and which encounter lock contention in attempting to do so. In producer-consumer workflows, such small lock hold times can cause lock convoying that greatly increases overall bin mutex contention.
This commit is contained in:
committed by
David Goldblatt
parent
44d91cf243
commit
fc615739cb
@@ -563,10 +563,11 @@ arena_dalloc_bin_locked_begin(arena_dalloc_bin_locked_info_t *info,
|
||||
* stats updates, which happen during finish (this lets running counts get left
|
||||
* in a register).
|
||||
*/
|
||||
JEMALLOC_ALWAYS_INLINE bool
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
arena_dalloc_bin_locked_step(tsdn_t *tsdn, arena_t *arena, bin_t *bin,
|
||||
arena_dalloc_bin_locked_info_t *info, szind_t binind, edata_t *slab,
|
||||
void *ptr) {
|
||||
void *ptr, edata_t **dalloc_slabs, unsigned ndalloc_slabs,
|
||||
unsigned *dalloc_slabs_count, edata_list_active_t *dalloc_slabs_extra) {
|
||||
const bin_info_t *bin_info = &bin_infos[binind];
|
||||
size_t regind = arena_slab_regind(info, binind, slab, ptr);
|
||||
slab_data_t *slab_data = edata_slab_data_get(slab);
|
||||
@@ -586,12 +587,17 @@ arena_dalloc_bin_locked_step(tsdn_t *tsdn, arena_t *arena, bin_t *bin,
|
||||
if (nfree == bin_info->nregs) {
|
||||
arena_dalloc_bin_locked_handle_newly_empty(tsdn, arena, slab,
|
||||
bin);
|
||||
return true;
|
||||
|
||||
if (*dalloc_slabs_count < ndalloc_slabs) {
|
||||
dalloc_slabs[*dalloc_slabs_count] = slab;
|
||||
(*dalloc_slabs_count)++;
|
||||
} else {
|
||||
edata_list_active_append(dalloc_slabs_extra, slab);
|
||||
}
|
||||
} else if (nfree == 1 && slab != bin->slabcur) {
|
||||
arena_dalloc_bin_locked_handle_newly_nonempty(tsdn, arena, slab,
|
||||
bin);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
@@ -604,11 +610,129 @@ arena_dalloc_bin_locked_finish(tsdn_t *tsdn, arena_t *arena, bin_t *bin,
|
||||
}
|
||||
}
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
arena_bin_flush_batch_impl(tsdn_t *tsdn, arena_t *arena, bin_t *bin,
|
||||
arena_dalloc_bin_locked_info_t *dalloc_bin_info, unsigned binind,
|
||||
edata_t **dalloc_slabs, unsigned ndalloc_slabs, unsigned *dalloc_count,
|
||||
edata_list_active_t *dalloc_slabs_extra) {
|
||||
assert(binind < bin_info_nbatched_sizes);
|
||||
bin_with_batch_t *batched_bin = (bin_with_batch_t *)bin;
|
||||
size_t nelems_to_pop = batcher_pop_begin(tsdn,
|
||||
&batched_bin->remote_frees);
|
||||
|
||||
bin_batching_test_mid_pop(nelems_to_pop);
|
||||
if (nelems_to_pop == BATCHER_NO_IDX) {
|
||||
malloc_mutex_assert_not_owner(tsdn,
|
||||
&batched_bin->remote_frees.mtx);
|
||||
return;
|
||||
} else {
|
||||
malloc_mutex_assert_owner(tsdn,
|
||||
&batched_bin->remote_frees.mtx);
|
||||
}
|
||||
|
||||
bin_remote_free_data_t remote_free_data[BIN_REMOTE_FREE_ELEMS_MAX];
|
||||
for (size_t i = 0; i < nelems_to_pop; i++) {
|
||||
remote_free_data[i] = batched_bin->remote_free_data[i];
|
||||
}
|
||||
batcher_pop_end(tsdn, &batched_bin->remote_frees);
|
||||
|
||||
for (size_t i = 0; i < nelems_to_pop; i++) {
|
||||
arena_dalloc_bin_locked_step(tsdn, arena, bin, dalloc_bin_info,
|
||||
binind, remote_free_data[i].slab, remote_free_data[i].ptr,
|
||||
dalloc_slabs, ndalloc_slabs, dalloc_count,
|
||||
dalloc_slabs_extra);
|
||||
}
|
||||
}
|
||||
|
||||
typedef struct arena_bin_flush_batch_state_s arena_bin_flush_batch_state_t;
|
||||
struct arena_bin_flush_batch_state_s {
|
||||
arena_dalloc_bin_locked_info_t info;
|
||||
|
||||
/*
|
||||
* Bin batching is subtle in that there are unusual edge cases in which
|
||||
* it can trigger the deallocation of more slabs than there were items
|
||||
* flushed (say, if every original deallocation triggered a slab
|
||||
* deallocation, and so did every batched one). So we keep a small
|
||||
* backup array for any "extra" slabs, as well as a a list to allow a
|
||||
* dynamic number of ones exceeding that array.
|
||||
*/
|
||||
edata_t *dalloc_slabs[8];
|
||||
unsigned dalloc_slab_count;
|
||||
edata_list_active_t dalloc_slabs_extra;
|
||||
};
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE unsigned
|
||||
arena_bin_batch_get_ndalloc_slabs(unsigned preallocated_slabs) {
|
||||
if (preallocated_slabs > bin_batching_test_ndalloc_slabs_max) {
|
||||
return bin_batching_test_ndalloc_slabs_max;
|
||||
}
|
||||
return preallocated_slabs;
|
||||
}
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
arena_bin_flush_batch_after_lock(tsdn_t *tsdn, arena_t *arena, bin_t *bin,
|
||||
unsigned binind, arena_bin_flush_batch_state_t *state) {
|
||||
if (binind >= bin_info_nbatched_sizes) {
|
||||
return;
|
||||
}
|
||||
|
||||
arena_dalloc_bin_locked_begin(&state->info, binind);
|
||||
state->dalloc_slab_count = 0;
|
||||
edata_list_active_init(&state->dalloc_slabs_extra);
|
||||
|
||||
unsigned preallocated_slabs = (unsigned)(sizeof(state->dalloc_slabs)
|
||||
/ sizeof(state->dalloc_slabs[0]));
|
||||
unsigned ndalloc_slabs = arena_bin_batch_get_ndalloc_slabs(
|
||||
preallocated_slabs);
|
||||
|
||||
arena_bin_flush_batch_impl(tsdn, arena, bin, &state->info, binind,
|
||||
state->dalloc_slabs, ndalloc_slabs,
|
||||
&state->dalloc_slab_count, &state->dalloc_slabs_extra);
|
||||
}
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
arena_bin_flush_batch_before_unlock(tsdn_t *tsdn, arena_t *arena, bin_t *bin,
|
||||
unsigned binind, arena_bin_flush_batch_state_t *state) {
|
||||
if (binind >= bin_info_nbatched_sizes) {
|
||||
return;
|
||||
}
|
||||
|
||||
arena_dalloc_bin_locked_finish(tsdn, arena, bin, &state->info);
|
||||
}
|
||||
|
||||
static inline bool
|
||||
arena_bin_has_batch(szind_t binind) {
|
||||
return binind < bin_info_nbatched_sizes;
|
||||
}
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
arena_bin_flush_batch_after_unlock(tsdn_t *tsdn, arena_t *arena, bin_t *bin,
|
||||
unsigned binind, arena_bin_flush_batch_state_t *state) {
|
||||
if (!arena_bin_has_batch(binind)) {
|
||||
return;
|
||||
}
|
||||
/*
|
||||
* The initialization of dalloc_slabs_extra is guarded by an
|
||||
* arena_bin_has_batch check higher up the stack. But the clang
|
||||
* analyzer forgets this down the stack, triggering a spurious error
|
||||
* reported here.
|
||||
*/
|
||||
JEMALLOC_CLANG_ANALYZER_SUPPRESS {
|
||||
bin_batching_test_after_unlock(state->dalloc_slab_count,
|
||||
edata_list_active_empty(&state->dalloc_slabs_extra));
|
||||
}
|
||||
for (unsigned i = 0; i < state->dalloc_slab_count; i++) {
|
||||
edata_t *slab = state->dalloc_slabs[i];
|
||||
arena_slab_dalloc(tsdn, arena_get_from_edata(slab), slab);
|
||||
}
|
||||
while (!edata_list_active_empty(&state->dalloc_slabs_extra)) {
|
||||
edata_t *slab = edata_list_active_first(
|
||||
&state->dalloc_slabs_extra);
|
||||
edata_list_active_remove(&state->dalloc_slabs_extra, slab);
|
||||
arena_slab_dalloc(tsdn, arena_get_from_edata(slab), slab);
|
||||
}
|
||||
}
|
||||
|
||||
static inline bin_t *
|
||||
arena_get_bin(arena_t *arena, szind_t binind, unsigned binshard) {
|
||||
bin_t *shard0 = (bin_t *)((byte_t *)arena + arena_bin_offsets[binind]);
|
||||
|
||||
@@ -11,6 +11,51 @@
|
||||
|
||||
#define BIN_REMOTE_FREE_ELEMS_MAX 16
|
||||
|
||||
#ifdef JEMALLOC_JET
|
||||
extern void (*bin_batching_test_after_push_hook)(size_t idx);
|
||||
extern void (*bin_batching_test_mid_pop_hook)(size_t elems_to_pop);
|
||||
extern void (*bin_batching_test_after_unlock_hook)(unsigned slab_dalloc_count,
|
||||
bool list_empty);
|
||||
#endif
|
||||
|
||||
#ifdef JEMALLOC_JET
|
||||
extern unsigned bin_batching_test_ndalloc_slabs_max;
|
||||
#else
|
||||
static const unsigned bin_batching_test_ndalloc_slabs_max = (unsigned)-1;
|
||||
#endif
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
bin_batching_test_after_push(size_t idx) {
|
||||
(void)idx;
|
||||
#ifdef JEMALLOC_JET
|
||||
if (bin_batching_test_after_push_hook != NULL) {
|
||||
bin_batching_test_after_push_hook(idx);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
bin_batching_test_mid_pop(size_t elems_to_pop) {
|
||||
(void)elems_to_pop;
|
||||
#ifdef JEMALLOC_JET
|
||||
if (bin_batching_test_mid_pop_hook != NULL) {
|
||||
bin_batching_test_mid_pop_hook(elems_to_pop);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
JEMALLOC_ALWAYS_INLINE void
|
||||
bin_batching_test_after_unlock(unsigned slab_dalloc_count, bool list_empty) {
|
||||
(void)slab_dalloc_count;
|
||||
(void)list_empty;
|
||||
#ifdef JEMALLOC_JET
|
||||
if (bin_batching_test_after_unlock_hook != NULL) {
|
||||
bin_batching_test_after_unlock_hook(slab_dalloc_count,
|
||||
list_empty);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
/*
|
||||
* A bin contains a set of extents that are currently being used for slab
|
||||
* allocations.
|
||||
@@ -70,7 +115,7 @@ bool bin_update_shard_size(unsigned bin_shards[SC_NBINS], size_t start_size,
|
||||
size_t end_size, size_t nshards);
|
||||
|
||||
/* Initializes a bin to empty. Returns true on error. */
|
||||
bool bin_init(bin_t *bin);
|
||||
bool bin_init(bin_t *bin, unsigned binind);
|
||||
|
||||
/* Forking. */
|
||||
void bin_prefork(tsdn_t *tsdn, bin_t *bin, bool has_batch);
|
||||
|
||||
@@ -48,6 +48,8 @@ struct bin_info_s {
|
||||
extern size_t opt_bin_info_max_batched_size;
|
||||
/* The number of batches per batched size class. */
|
||||
extern size_t opt_bin_info_remote_free_max_batch;
|
||||
// The max number of pending elems (across all batches)
|
||||
extern size_t opt_bin_info_remote_free_max;
|
||||
|
||||
extern szind_t bin_info_nbatched_sizes;
|
||||
extern unsigned bin_info_nbatched_bins;
|
||||
|
||||
Reference in New Issue
Block a user