Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -878,7 +878,7 @@ if (MI_BUILD_TESTS)
enable_testing()

# static link tests
set(mi_static_tests api api-fill stress-heaps stress-subprocs stress heap-mt heap-teardown heap-delete-race heap-churn heap-release-mt heap-aba heap-burst-destroy abandoned-lazy fork-user-heap snapshot prof prof-adversarial purge-zero park-handoff free-before-init)
set(mi_static_tests api api-fill stress-heaps stress-subprocs stress heap-mt heap-teardown heap-delete-race heap-churn heap-release-mt heap-aba heap-burst-destroy abandoned-lazy fork-user-heap snapshot prof prof-adversarial purge-zero park-handoff forced-purge free-before-init)
if(NOT (MI_DEBUG_TSAN OR MI_TRACK_ASAN OR MI_DEBUG_UBSAN))
list(APPEND mi_static_tests thp-optout) # counts madvise calls by interposing it, which a sanitizer runtime does first
endif()
Expand Down
3 changes: 2 additions & 1 deletion include/mimalloc/internal.h
Original file line number Diff line number Diff line change
Expand Up @@ -268,7 +268,8 @@ void* _mi_arenas_alloc_aligned(mi_heap_t* heap, size_t size, size_t alig
void _mi_arenas_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t memid);
void _mi_arenas_collect(bool force_purge, bool visit_all, mi_tld_t* tld);
void _mi_arenas_purge_abandoned_holes(mi_heap_t* heap, mi_tld_t* tld);
void _mi_arenas_try_purge(bool force, bool visit_all, mi_subproc_t* subproc, size_t tseq);
bool _mi_arenas_try_purge(bool force, bool visit_all, mi_subproc_t* subproc, size_t tseq); // false: dropped, another thread is purging
void _mi_arenas_forked_child(void);
void _mi_arenas_unsafe_destroy_all(mi_subproc_t* subproc);

mi_page_t* _mi_arenas_page_alloc(mi_theap_t* theap, size_t block_size, size_t page_alignment);
Expand Down
34 changes: 26 additions & 8 deletions src/arena.c
Original file line number Diff line number Diff line change
Expand Up @@ -1544,7 +1544,15 @@ void _mi_arenas_free(mi_subproc_t* subproc, void* p, size_t size, mi_memid_t mem

// Purge the arenas; if `force_purge` is true, amenable parts are purged even if not yet expired
void _mi_arenas_collect(bool force_purge, bool visit_all, mi_tld_t* tld) {
_mi_arenas_try_purge(force_purge, visit_all, tld->subproc, tld->thread_seq);
// A forced purge is `mi_collect(true)`, and whoever asks for that reads the footprint next. Only one thread purges at
// a time, and the pass that holds the guard does not stand in for ours: the scavenger's is never forced, so it leaves
// every arena whose delay has not passed, and it does not go back for what was freed behind it. So take our turn
// after it. The holder takes no lock and waits for no one; it is in there for its madvise calls.
size_t spin = 0;
while (!_mi_arenas_try_purge(force_purge, visit_all, tld->subproc, tld->thread_seq) && force_purge) {
if (spin < 256) { mi_atomic_pause(); spin++; }
else { _mi_prim_thread_yield(); }
}
}


Expand Down Expand Up @@ -2545,24 +2553,33 @@ void _mi_arenas_purge_now(mi_subproc_t* subproc) {
}
}

void _mi_arenas_try_purge(bool force, bool visit_all, mi_subproc_t* subproc, size_t tseq)
// allow only one thread to purge at a time (todo: allow concurrent purging?)
static mi_atomic_guard_t mi_arenas_purge_guard;

// The thread that held the guard across fork() is not in the child, so nothing there would ever release it.
void _mi_arenas_forked_child(void) {
mi_atomic_store_release(&mi_arenas_purge_guard, (uintptr_t)0);
}

// Returns false if another thread was purging and this pass was dropped because of it.
bool _mi_arenas_try_purge(bool force, bool visit_all, mi_subproc_t* subproc, size_t tseq)
{
// try purge can be called often so try to only run when needed
const long delay = mi_arena_purge_delay();
if (_mi_preloading() || delay <= 0) return; // nothing will be scheduled
if (_mi_preloading() || delay <= 0) return true; // nothing will be scheduled

// check if any arena needs purging?
const mi_msecs_t now = _mi_clock_now();
const mi_msecs_t arenas_expire = mi_atomic_loadi64_acquire(&subproc->purge_expire);
if (!visit_all && !force && (arenas_expire == 0 || arenas_expire > now)) return;
if (!visit_all && !force && (arenas_expire == 0 || arenas_expire > now)) return true;

const size_t max_arena = mi_arenas_get_count(subproc);
if (max_arena == 0) return;
if (max_arena == 0) return true;

// allow only one thread to purge at a time (todo: allow concurrent purging?)
static mi_atomic_guard_t purge_guard;
mi_atomic_guard(&purge_guard)
bool entered = false;
mi_atomic_guard(&mi_arenas_purge_guard)
{
entered = true;
// increase global expire: at most one purge per delay cycle
if (arenas_expire > now) { mi_atomic_storei64_release(&subproc->purge_expire, now + (delay/10)); }
const size_t arena_start = tseq % max_arena;
Expand Down Expand Up @@ -2598,6 +2615,7 @@ void _mi_arenas_try_purge(bool force, bool visit_all, mi_subproc_t* subproc, siz
mi_atomic_casi64_strong_acq_rel(&subproc->purge_expire, &expected, next_expire);
}
}
return entered;
}


Expand Down
1 change: 1 addition & 0 deletions src/subproc.c
Original file line number Diff line number Diff line change
Expand Up @@ -410,6 +410,7 @@ void _mi_process_fork_child(void) {
if (mi_atomic_exchange_acq_rel(&mi_fork_depth, 0) == 0) return;
_mi_process_is_forked_child = true;
_mi_scavenger_forked_child(); // the scavenger thread did not survive the fork; clear the state that says it did
_mi_arenas_forked_child(); // nor did a thread that was purging; release the guard it held
mi_lock_init(&mi_subprocs_lock);
for (mi_subproc_t* sp = mi_subprocs; sp != NULL; sp = sp->next) {
mi_lock_init(&sp->arena_reserve_lock);
Expand Down
86 changes: 86 additions & 0 deletions test/test-forced-purge.c
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
/* ----------------------------------------------------------------------------
Copyright (c) 2026, Microsoft Research, Daan Leijen
This is free software; you can redistribute it and/or modify it under the
terms of the MIT license. A copy of the license can be found in the file
"LICENSE" at the root of this distribution.
-----------------------------------------------------------------------------*/

// `mi_collect(true)` purges every free arena range before it returns.
//
// Only one thread runs a purge pass at a time, and with a scavenger thread there often is one in a
// pass: it takes the freed ranges `purge_delay` after they were freed. Its pass is not forced, so it
// is no substitute for the caller's, and a caller of `mi_collect(true)` that found the scavenger in
// there used to return with nothing purged at all. Whoever forces a collect reads the footprint next.
//
// Each round frees a few hundred MiB, waits for the scavenger to start on them (`arena_purges` counts
// the arenas a pass went into), and forces a collect while the scavenger is in the middle of that.
// Everything the round freed has to be counted in `purged` when the collect returns. Counters, not
// RSS: they read the same on every platform and in every build.

#include "mimalloc.h"
#include "mimalloc-stats.h"
#include <stdbool.h>
#include <stdio.h>
#include <string.h>
#include <time.h>

static int failures = 0;

static void check(const char* name, bool ok) {
fprintf(stderr, "test: %s... %s\n", name, ok ? "ok." : "FAILED");
if (!ok) failures++;
}

#define MAX_BLOCKS (32)
#define BLOCK_SIZE ((size_t)8 * 1024 * 1024)
#define ROUNDS (4)

// the counters of the subprocess itself: read without a lock, and without touching our theap
static void counters(size_t* purged, size_t* arena_purges) {
mi_stats_t_decl(stats);
mi_subproc_stats_get_exclusive(mi_subproc_main(), &stats);
*purged = (size_t)stats.purged.total;
*arena_purges = (size_t)stats.arena_purges.total;
}

int main(void) {
if (!mi_option_is_enabled(mi_option_scavenger) || mi_option_get(mi_option_purge_delay) <= 0) {
fprintf(stderr, "test-forced-purge: skipped (no scavenger, or no purge delay)\n");
return 0;
}
// the scavenger starts when a second thread initializes or a thread first parks
void* warm = mi_malloc(64); mi_free(warm);
if (mi_on_thread_idle_start()) { mi_on_thread_idle_end(); }

// 256 MiB a round, so that a pass over it holds the purge guard for milliseconds (64 MiB where address space is scarce)
const int BLOCKS = (sizeof(void*) >= 8 ? MAX_BLOCKS : MAX_BLOCKS / 4);
static void* blocks[MAX_BLOCKS];
int in_a_pass = 0;
for (int round = 0; round < ROUNDS; round++) {
for (int i = 0; i < BLOCKS; i++) {
blocks[i] = mi_malloc(BLOCK_SIZE);
if (blocks[i] == NULL) { fprintf(stderr, "test-forced-purge: out of memory\n"); return 1; }
memset(blocks[i], 1, BLOCK_SIZE); // resident, so purging it takes the scavenger a while
}
size_t purged0, passes0;
counters(&purged0, &passes0);
for (int i = 0; i < BLOCKS; i++) { mi_free(blocks[i]); } // back to the arena, purge scheduled

// Wait for the scavenger to go into an arena. If it never does (bounded), the forced collect
// below has all of it to purge by itself, which has to work just as well.
size_t purged1, passes1;
const time_t deadline = time(NULL) + 10;
do { counters(&purged1, &passes1); } while (passes1 == passes0 && time(NULL) < deadline);
if (passes1 != passes0) { in_a_pass++; }

mi_collect(true);

counters(&purged1, &passes1);
char name[128];
snprintf(name, sizeof(name), "round %d: a forced collect leaves nothing that was freed unpurged (%zu of %zu MiB)",
round, (purged1 - purged0) / (1024 * 1024), ((size_t)BLOCKS * BLOCK_SIZE) / (1024 * 1024));
check(name, purged1 - purged0 >= (size_t)BLOCKS * BLOCK_SIZE);
}
fprintf(stderr, "test-forced-purge: %d of %d forced collects came while the scavenger was purging\n", in_a_pass, ROUNDS);
return failures;
}
Loading