Defer heap release trim under demand and skip small page types (#419)

Trimming a heap when it was released delayed the exit of every thread
with madvise calls, which are slow with many threads since each
decommit flushes the TLB on every core running the process. Threads
started in the meantime found no queued heap and mapped new ones, up to
218 heaps instead of 129 in mimalloc-bench larson, and the pages of the
extra heaps added to the peak. Trimming small pages also restarted
their decommit delay on every release, so with short lived threads the
delay never expired and heaps kept up to four times the overflow
threshold instead of decommitting down to the retain count.

The trim is now done on release only when other heaps are already
queued, which also covers a program dropping from a high to a low
thread count where most released heaps are never adopted again. With
the queue empty the heap is flagged and the thread adopting it trims it
first. The trim is limited to the medium-large and large page types,
which hold most of the retained memory, and now also retires the empty
medium-large pages kept for their size class. Small and medium-small
pages are left to the decommit delay.

Measured with mimalloc-bench on a 64 core Threadripper PRO 3995WX with
128 threads, median of 5 runs:

  larson         1204 -> 1121 MiB peak RSS, 0.586 -> 0.581 s
  larson-sized   1176 -> 1120 MiB peak RSS, 0.582 -> 0.579 s
  mstress        2587 -> 2559 MiB peak RSS, 2.90 -> 2.84 s

With 128 threads allocating across all page types and exiting together,
the released heaps hold 2973 MiB instead of 3824 MiB.
diff --git a/CHANGELOG b/CHANGELOG
index dc0f68e..ae4b5ae 100644
--- a/CHANGELOG
+++ b/CHANGELOG
@@ -8,8 +8,10 @@
 allocate and free in phases. There is no background thread, the owning thread checks the delay when
 it frees a page, and rpmalloc_thread_collect now decommits the excess immediately.
 
-A released heap trims the memory it retains, retiring the empty large pages kept for their size
-class and decommitting free pages to below the overflow threshold.
+A released heap retires the empty medium-large and large pages kept for their size class and
+decommits its free pages of those types to below the overflow threshold. The trim is done on release
+when other heaps are already queued for reuse, and otherwise by the thread that adopts the heap.
+Small and medium-small pages are left to the decommit delay.
 
 The last available page of a size class is kept when its final block is freed, so alternating
 allocation and free of a single block no longer reinitializes the page on every pair.
diff --git a/rpmalloc/rpmalloc.c b/rpmalloc/rpmalloc.c
index bac093f..7ea2646 100644
--- a/rpmalloc/rpmalloc.c
+++ b/rpmalloc/rpmalloc.c
@@ -636,6 +636,8 @@
 	uint32_t id;
 	//! Finalization state flag
 	uint32_t finalize;
+	//! Set when the heap was released without being trimmed, the thread adopting it trims it first
+	uint32_t release_trim;
 	//! Memory map region offset
 	uint32_t offset;
 	//! Memory map size
@@ -785,6 +787,9 @@
 static heap_t*
 heap_allocate(int first_class);
 
+static void
+heap_release_free_pages(heap_t* heap);
+
 static NOINLINE void
 heap_page_free_overflow(heap_t* heap, uint32_t page_type);
 
@@ -1952,6 +1957,10 @@
 			heap_t* heap = global_heap_queue;
 			global_heap_queue = heap->next;
 			heap_lock_release();
+			if (heap->release_trim) {
+				heap->release_trim = 0;
+				heap_release_free_pages(heap);
+			}
 			return heap;
 		}
 		if (global_heap_pristine) {
@@ -2109,31 +2118,49 @@
 	heap_page_free_decommit(heap, page_type, global_page_free_retain[page_type]);
 }
 
-//! Trims the memory a heap holds for reuse when it is released, by its exiting thread or as a first
-//  class heap. Until another thread adopts the heap nothing reuses that memory or runs the decommit
-//  delay, so a heap that many short lived threads pass along would otherwise keep it all. The empty
-//  large pages kept as the last available page of their size class are retired, and free pages are
-//  decommitted down to below the overflow threshold, the most the heap holds without a delay
+//! Trims the memory a released heap holds for reuse. Until another thread adopts the heap nothing
+//  reuses that memory or runs the decommit delay, so a heap that many short lived threads pass along
+//  would otherwise keep it all. The empty medium-large and large pages kept as the last available
+//  page of their size class are retired, and free pages of those types are decommitted down to below
+//  the overflow threshold, the most the heap holds without a delay. Small and medium-small pages are
+//  cheap to keep and costly to fault back in, they are left to the decommit delay, which keeps
+//  running across heap owners
 static void
 heap_release_free_pages(heap_t* heap) {
 	for (uint32_t iclass = 0; iclass < SIZE_CLASS_COUNT; ++iclass) {
 		page_t* page = heap->page_available[iclass];
-		if (!page || page->next || page->block_used || (page->page_type != PAGE_LARGE))
+		if (!page || page->next || page->block_used || (page->page_type < PAGE_MEDIUM_LARGE))
 			continue;
 		heap->page_available[iclass] = 0;
 		page->is_free = 1;
 		page->is_zero = 0;
-		page->next = heap->page_free[PAGE_LARGE];
-		heap->page_free[PAGE_LARGE] = page;
-		++heap->page_free_commit_count[PAGE_LARGE];
+		page->next = heap->page_free[page->page_type];
+		heap->page_free[page->page_type] = page;
+		++heap->page_free_commit_count[page->page_type];
 	}
-	for (uint32_t itype = 0; itype < 4; ++itype) {
+	for (uint32_t itype = PAGE_MEDIUM_LARGE; itype <= PAGE_LARGE; ++itype) {
 		heap->page_free_overflow_ms[itype] = 0;
 		if (heap->page_free_commit_count[itype] >= global_page_free_overflow[itype])
 			heap_page_free_decommit(heap, itype, global_page_free_overflow[itype] - 1);
 	}
 }
 
+//! Releases a heap, by its exiting thread or as a first class heap, for reuse by another thread.
+//  With other heaps already queued the heap is likely to stay unused for a while and is trimmed
+//  right away. With the queue empty a thread is likely to adopt it shortly, so the trim is left to
+//  the adopter rather than delaying the release while threads wait for a heap
+static void
+heap_release_trim(heap_t* heap) {
+	heap_lock_acquire();
+	int queued = (global_heap_queue != 0);
+	heap_lock_release();
+	if (queued)
+		heap_release_free_pages(heap);
+	else
+		heap->release_trim = 1;
+	heap_release(heap);
+}
+
 static inline int
 heap_make_free_page_available(heap_t* heap, uint32_t size_class, page_t* page) {
 	page->size_class = size_class;
@@ -3318,8 +3345,7 @@
 rpmalloc_thread_finalize(void) {
 	heap_t* heap = get_thread_heap();
 	if (heap != global_heap_default) {
-		heap_release_free_pages(heap);
-		heap_release(heap);
+		heap_release_trim(heap);
 		set_thread_heap(global_heap_default);
 	}
 }
@@ -3482,10 +3508,8 @@
 
 void
 rpmalloc_heap_release(rpmalloc_heap_t* heap) {
-	if (heap) {
-		heap_release_free_pages(heap);
-		heap_release(heap);
-	}
+	if (heap)
+		heap_release_trim(heap);
 }
 
 RPMALLOC_ALLOCATOR void*