hakmem/core/hakmem_shared_pool_acquire.c

#include "hakmem_shared_pool_internal.h"
#include "hakmem_debug_master.h"
#include "hakmem_stats_master.h"
#include "box/ss_slab_meta_box.h"
#include "box/ss_hot_cold_box.h"
#include "box/pagefault_telemetry_box.h"
#include "box/tls_sll_drain_box.h"
#include "box/tls_slab_reuse_guard_box.h"
#include "box/ss_tier_box.h"  // P-Tier: Tier filtering support
#include "hakmem_policy.h"
#include "hakmem_env_cache.h"  // Priority-2: ENV cache
#include "front/tiny_warm_pool.h"  // Warm Pool: Prefill during registry scans

#include <stdlib.h>
#include <stdio.h>
#include <stdatomic.h>

// ============================================================================
// Performance Measurement: Shared Pool Lock Contention (ENV-gated)
// ============================================================================
// Global atomic counters for lock contention measurement
// ENV: HAKMEM_MEASURE_UNIFIED_CACHE=1 to enable (default: OFF)
_Atomic uint64_t g_sp_stage2_lock_acquired_global = 0;
_Atomic uint64_t g_sp_stage3_lock_acquired_global = 0;
_Atomic uint64_t g_sp_alloc_lock_contention_global = 0;

// Per-class lock acquisition statistics（Tiny クラス別の lock 負荷観測用）
_Atomic uint64_t g_sp_stage2_lock_acquired_by_class[TINY_NUM_CLASSES_SS] = {0};
_Atomic uint64_t g_sp_stage3_lock_acquired_by_class[TINY_NUM_CLASSES_SS] = {0};

// Check if measurement is enabled (cached)
static inline int sp_measure_enabled(void) {
    static int g_measure = -1;
    if (__builtin_expect(g_measure == -1, 0)) {
        const char* e = getenv("HAKMEM_MEASURE_UNIFIED_CACHE");
        g_measure = (e && *e && *e != '0') ? 1 : 0;
        if (g_measure == 1) {
            // Measurement が ON のときは per-class stage stats も有効化する
            // （Stage1/2/3 ヒット数は g_sp_stage*_hits に集計される）
            extern int g_sp_stage_stats_enabled;
            g_sp_stage_stats_enabled = 1;
        }
    }
    return g_measure;
}

// Print statistics function
void shared_pool_print_measurements(void);

// Stage 0.5: EMPTY slab direct scan（registry ベースの EMPTY 再利用）
// Scan existing SuperSlabs for EMPTY slabs (highest reuse priority) to
// avoid Stage 3 (mmap) when freed slabs are available.
//
// WARM POOL OPTIMIZATION:
// - During the registry scan, prefill warm pool with HOT SuperSlabs
// - This eliminates future registry scans for cache misses
// - Expected gain: +40-50% by reducing O(N) scan overhead
static inline int
sp_acquire_from_empty_scan(int class_idx, SuperSlab** ss_out, int* slab_idx_out, int dbg_acquire)
{
    // Priority-2: Use cached ENV
    int empty_reuse_enabled = HAK_ENV_SS_EMPTY_REUSE();
    if (!empty_reuse_enabled) {
        return -1;
    }

    extern SuperSlab* g_super_reg_by_class[TINY_NUM_CLASSES][SUPER_REG_PER_CLASS];
    extern int g_super_reg_class_size[TINY_NUM_CLASSES];

    int reg_size = (class_idx < TINY_NUM_CLASSES) ? g_super_reg_class_size[class_idx] : 0;
    // Priority-2: Use cached ENV
    int scan_limit = HAK_ENV_SS_EMPTY_SCAN_LIMIT();
    if (scan_limit > reg_size) scan_limit = reg_size;

    // Stage 0.5 hit counter for visualization
    static _Atomic uint64_t stage05_hits = 0;
    static _Atomic uint64_t stage05_attempts = 0;
    atomic_fetch_add_explicit(&stage05_attempts, 1, memory_order_relaxed);

    // Initialize warm pool on first use (per-thread, one-time)
    tiny_warm_pool_init_once();

    // Track SuperSlabs scanned during this acquire call for warm pool prefill
    SuperSlab* primary_result = NULL;
    int primary_slab_idx = -1;

    for (int i = 0; i < scan_limit; i++) {
        SuperSlab* ss = g_super_reg_by_class[class_idx][i];
        if (!(ss && ss->magic == SUPERSLAB_MAGIC)) continue;
        // P-Tier: Skip DRAINING tier SuperSlabs
        if (!ss_tier_is_hot(ss)) continue;
        if (ss->empty_count == 0) continue;  // No EMPTY slabs in this SS

        // WARM POOL PREFILL: Add HOT SuperSlabs to warm pool (if not already primary result)
        // This is low-cost during registry scan and avoids future expensive scans
        // Phase 1: Increase threshold from 4 to 12 to match TINY_WARM_POOL_MAX_PER_CLASS
        if (ss != primary_result && tiny_warm_pool_count(class_idx) < 12) {
            tiny_warm_pool_push(class_idx, ss);
            // Track prefilled SuperSlabs for metrics
            g_warm_pool_stats[class_idx].prefilled++;
        }

        uint32_t mask = ss->empty_mask;
        while (mask) {
            int empty_idx = __builtin_ctz(mask);
            mask &= (mask - 1);  // clear lowest bit

            TinySlabMeta* meta = &ss->slabs[empty_idx];
            if (meta->capacity > 0 && meta->used == 0) {
                tiny_tls_slab_reuse_guard(ss);
                ss_clear_slab_empty(ss, empty_idx);

                meta->class_idx = (uint8_t)class_idx;
                ss->class_map[empty_idx] = (uint8_t)class_idx;

#if !HAKMEM_BUILD_RELEASE
                if (dbg_acquire == 1) {
                    fprintf(stderr,
                            "[SP_ACQUIRE_STAGE0.5_EMPTY] class=%d reusing EMPTY slab (ss=%p slab=%d empty_count=%u warm_pool_size=%d)\n",
                            class_idx, (void*)ss, empty_idx, ss->empty_count, tiny_warm_pool_count(class_idx));
                }
#else
                (void)dbg_acquire;
#endif

                // Store primary result but continue scanning to prefill warm pool
                if (primary_result == NULL) {
                    primary_result = ss;
                    primary_slab_idx = empty_idx;
                    *ss_out = ss;
                    *slab_idx_out = empty_idx;
                    sp_stage_stats_init();
                    if (g_sp_stage_stats_enabled) {
                        atomic_fetch_add(&g_sp_stage1_hits[class_idx], 1);
                    }
                    atomic_fetch_add_explicit(&stage05_hits, 1, memory_order_relaxed);
                }
            }
        }
    }

    if (primary_result != NULL) {
        // Stage 0.5 hit rate visualization (every 100 hits)
        uint64_t hits = atomic_load_explicit(&stage05_hits, memory_order_relaxed);
        if (hits % 100 == 1) {
            uint64_t attempts = atomic_load_explicit(&stage05_attempts, memory_order_relaxed);
            fprintf(stderr, "[STAGE0.5_STATS] hits=%lu attempts=%lu rate=%.1f%% (scan_limit=%d warm_pool=%d)\n",
                    hits, attempts, (double)hits * 100.0 / attempts, scan_limit, tiny_warm_pool_count(class_idx));
        }
        return 0;
    }
    return -1;
}

int
shared_pool_acquire_slab(int class_idx, SuperSlab** ss_out, int* slab_idx_out)
{
    // Phase 12: SP-SLOT Box - 3-Stage Acquire Logic
    //
    // Stage 1: Reuse EMPTY slots from per-class free list (EMPTY→ACTIVE)
    // Stage 2: Find UNUSED slots in existing SuperSlabs
    // Stage 3: Get new SuperSlab (LRU pop or mmap)
    //
    // Invariants:
    //  - On success: *ss_out != NULL, 0 <= *slab_idx_out < total_slots
    //  - The chosen slab has meta->class_idx == class_idx

    if (!ss_out || !slab_idx_out) {
        return -1;
    }
    if (class_idx < 0 || class_idx >= TINY_NUM_CLASSES_SS) {
        return -1;
    }

    shared_pool_init();

    // Debug logging / stage stats
#if !HAKMEM_BUILD_RELEASE
    // Priority-2: Use cached ENV
    int dbg_acquire = HAK_ENV_SS_ACQUIRE_DEBUG();
#else
    static const int dbg_acquire = 0;
#endif
    sp_stage_stats_init();

stage1_retry_after_tension_drain:
    // ========== Stage 0.5 (Phase 12-1.1): EMPTY slab direct scan ==========
    // Scan existing SuperSlabs for EMPTY slabs (highest reuse priority) to
    // avoid Stage 3 (mmap) when freed slabs are available.
    if (sp_acquire_from_empty_scan(class_idx, ss_out, slab_idx_out, dbg_acquire) == 0) {
        return 0;
    }

    // ========== Stage 1 (Lock-Free): Try to reuse EMPTY slots ==========
    // P0-4: Lock-free pop from per-class free list (no mutex needed!)
    // Best case: Same class freed a slot, reuse immediately (cache-hot)
    SharedSSMeta* reuse_meta = NULL;
    int reuse_slot_idx = -1;

    if (sp_freelist_pop_lockfree(class_idx, &reuse_meta, &reuse_slot_idx)) {
        // Found EMPTY slot from lock-free list!
        // Now acquire mutex ONLY for slot activation and metadata update

        // P0 instrumentation: count lock acquisitions
        lock_stats_init();
        if (g_lock_stats_enabled == 1) {
            atomic_fetch_add(&g_lock_acquire_count, 1);
            atomic_fetch_add(&g_lock_acquire_slab_count, 1);
        }

        pthread_mutex_lock(&g_shared_pool.alloc_lock);

        // P0.3: Guard against TLS SLL orphaned pointers before reusing slab
        // RACE FIX: Load SuperSlab pointer atomically BEFORE guard (consistency)
        SuperSlab* ss_guard = atomic_load_explicit(&reuse_meta->ss, memory_order_relaxed);
        if (ss_guard) {
            tiny_tls_slab_reuse_guard(ss_guard);

            // P-Tier: Skip DRAINING tier SuperSlabs
            if (!ss_tier_is_hot(ss_guard)) {
                // DRAINING SuperSlab - skip this slot and fall through to Stage 2
                if (g_lock_stats_enabled == 1) {
                    atomic_fetch_add(&g_lock_release_count, 1);
                }
                pthread_mutex_unlock(&g_shared_pool.alloc_lock);
                goto stage2_fallback;
            }
        }

        // Activate slot under mutex (slot state transition requires protection)
        if (sp_slot_mark_active(reuse_meta, reuse_slot_idx, class_idx) == 0) {
            // RACE FIX: Load SuperSlab pointer atomically (consistency)
            SuperSlab* ss = atomic_load_explicit(&reuse_meta->ss, memory_order_relaxed);

            // RACE FIX: Check if SuperSlab was freed (NULL pointer)
            // This can happen if Thread A freed the SuperSlab after pushing slot to freelist,
            // but Thread B popped the stale slot before the freelist was cleared.
            if (!ss) {
                // SuperSlab freed - skip and fall through to Stage 2/3
                if (g_lock_stats_enabled == 1) {
                    atomic_fetch_add(&g_lock_release_count, 1);
                }
                pthread_mutex_unlock(&g_shared_pool.alloc_lock);
                goto stage2_fallback;
            }

            #if !HAKMEM_BUILD_RELEASE
            if (dbg_acquire == 1) {
                fprintf(stderr, "[SP_ACQUIRE_STAGE1_LOCKFREE] class=%d reusing EMPTY slot (ss=%p slab=%d)\n",
                        class_idx, (void*)ss, reuse_slot_idx);
            }
            #endif

            // Update SuperSlab metadata
            ss->slab_bitmap |= (1u << reuse_slot_idx);
            ss_slab_meta_class_idx_set(ss, reuse_slot_idx, (uint8_t)class_idx);

            if (ss->active_slabs == 0) {
                // Was empty, now active again
                ss->active_slabs = 1;
                g_shared_pool.active_count++;
            }
            // Track per-class active slots (approximate, under alloc_lock)
            if (class_idx < TINY_NUM_CLASSES_SS) {
                g_shared_pool.class_active_slots[class_idx]++;
            }

            // Update hint
            g_shared_pool.class_hints[class_idx] = ss;

            *ss_out = ss;
            *slab_idx_out = reuse_slot_idx;

            if (g_lock_stats_enabled == 1) {
                atomic_fetch_add(&g_lock_release_count, 1);
            }
            pthread_mutex_unlock(&g_shared_pool.alloc_lock);
	            if (g_sp_stage_stats_enabled) {
	                atomic_fetch_add(&g_sp_stage1_hits[class_idx], 1);
	            }
            return 0;  // ✅ Stage 1 (lock-free) success
        }

        // Slot activation failed (race condition?) - release lock and fall through
        if (g_lock_stats_enabled == 1) {
            atomic_fetch_add(&g_lock_release_count, 1);
        }
        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
    }

stage2_fallback:
    // ========== Stage 2 (Lock-Free): Try to claim UNUSED slots ==========
    // P0 Optimization: Try class hint FIRST for fast path (same class locality)
    // This reduces metadata scan from 100% to ~10% when hints are effective
    {
        SuperSlab* hint_ss = g_shared_pool.class_hints[class_idx];
        if (__builtin_expect(hint_ss != NULL, 1)) {
            // P-Tier: Skip DRAINING tier SuperSlabs
            if (!ss_tier_is_hot(hint_ss)) {
                // Clear stale hint pointing to DRAINING SuperSlab
                g_shared_pool.class_hints[class_idx] = NULL;
                goto stage2_scan;
            }

            // P0 Optimization: O(1) lookup via cached pointer (avoids metadata scan)
            SharedSSMeta* hint_meta = hint_ss->shared_meta;
            if (__builtin_expect(hint_meta != NULL, 1)) {
                // Try lock-free claiming on hint SuperSlab first
                int claimed_idx = sp_slot_claim_lockfree(hint_meta, class_idx);
                if (__builtin_expect(claimed_idx >= 0, 1)) {
                    // Fast path success! No need to scan all metadata
                    SuperSlab* ss = atomic_load_explicit(&hint_meta->ss, memory_order_acquire);
                    if (__builtin_expect(ss != NULL, 1)) {
                        #if !HAKMEM_BUILD_RELEASE
                        if (dbg_acquire == 1) {
                            fprintf(stderr, "[SP_ACQUIRE_STAGE2_HINT] class=%d claimed UNUSED slot from hint (ss=%p slab=%d)\n",
                                    class_idx, (void*)ss, claimed_idx);
                        }
                        #endif

                        // P0 instrumentation: count lock acquisitions
                        lock_stats_init();
                        if (g_lock_stats_enabled == 1) {
                            atomic_fetch_add(&g_lock_acquire_count, 1);
                            atomic_fetch_add(&g_lock_acquire_slab_count, 1);
                        }

                        pthread_mutex_lock(&g_shared_pool.alloc_lock);

                        // Performance measurement: count Stage 2 lock acquisitions
                        if (__builtin_expect(sp_measure_enabled(), 0)) {
                            atomic_fetch_add_explicit(&g_sp_stage2_lock_acquired_global,
                                                      1, memory_order_relaxed);
                            atomic_fetch_add_explicit(&g_sp_alloc_lock_contention_global,
                                                      1, memory_order_relaxed);
                            atomic_fetch_add_explicit(
                                &g_sp_stage2_lock_acquired_by_class[class_idx],
                                1, memory_order_relaxed);
                        }

                        // Update SuperSlab metadata under mutex
                        ss->slab_bitmap |= (1u << claimed_idx);
                        ss_slab_meta_class_idx_set(ss, claimed_idx, (uint8_t)class_idx);

                        if (ss->active_slabs == 0) {
                            ss->active_slabs = 1;
                            g_shared_pool.active_count++;
                        }
                        if (class_idx < TINY_NUM_CLASSES_SS) {
                            g_shared_pool.class_active_slots[class_idx]++;
                        }

                        // Hint is still good, no need to update
                        *ss_out = ss;
                        *slab_idx_out = claimed_idx;
                        sp_fix_geometry_if_needed(ss, claimed_idx, class_idx);

                        if (g_lock_stats_enabled == 1) {
                            atomic_fetch_add(&g_lock_release_count, 1);
                        }
                        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
                        if (g_sp_stage_stats_enabled) {
                            atomic_fetch_add(&g_sp_stage2_hits[class_idx], 1);
                        }
                        return 0;  // ✅ Stage 2 (hint fast path) success
                    }
                }
            }
        }
    }

stage2_scan:
    // P0-5: Lock-free atomic CAS claiming (no mutex needed for slot state transition!)
    // RACE FIX: Read ss_meta_count atomically (now properly declared as _Atomic)
    // No cast needed! memory_order_acquire synchronizes with release in sp_meta_find_or_create
    uint32_t meta_count = atomic_load_explicit(
        &g_shared_pool.ss_meta_count,
        memory_order_acquire
    );

    for (uint32_t i = 0; i < meta_count; i++) {
        SharedSSMeta* meta = &g_shared_pool.ss_metadata[i];

        // RACE FIX: Load SuperSlab pointer atomically BEFORE claiming
        // Use memory_order_acquire to synchronize with release in sp_meta_find_or_create
        SuperSlab* ss_preflight = atomic_load_explicit(&meta->ss, memory_order_acquire);
        if (!ss_preflight) {
            // SuperSlab was freed - skip this entry
            continue;
        }

        // P-Tier: Skip DRAINING tier SuperSlabs
        if (!ss_tier_is_hot(ss_preflight)) {
            continue;
        }

        // Try lock-free claiming (UNUSED → ACTIVE via CAS)
        int claimed_idx = sp_slot_claim_lockfree(meta, class_idx);
        if (claimed_idx >= 0) {
            // RACE FIX: Load SuperSlab pointer atomically again after claiming
            // Use memory_order_acquire to synchronize with release in sp_meta_find_or_create
            SuperSlab* ss = atomic_load_explicit(&meta->ss, memory_order_acquire);
            if (!ss) {
                // SuperSlab was freed between claiming and loading - skip this entry
                continue;
            }

            #if !HAKMEM_BUILD_RELEASE
            if (dbg_acquire == 1) {
                fprintf(stderr, "[SP_ACQUIRE_STAGE2_LOCKFREE] class=%d claimed UNUSED slot (ss=%p slab=%d)\n",
                        class_idx, (void*)ss, claimed_idx);
            }
            #endif

            // P0 instrumentation: count lock acquisitions
            lock_stats_init();
            if (g_lock_stats_enabled == 1) {
                atomic_fetch_add(&g_lock_acquire_count, 1);
                atomic_fetch_add(&g_lock_acquire_slab_count, 1);
            }

            pthread_mutex_lock(&g_shared_pool.alloc_lock);

            // Performance measurement: count Stage 2 scan lock acquisitions
            if (__builtin_expect(sp_measure_enabled(), 0)) {
                atomic_fetch_add_explicit(&g_sp_stage2_lock_acquired_global,
                                          1, memory_order_relaxed);
                atomic_fetch_add_explicit(&g_sp_alloc_lock_contention_global,
                                          1, memory_order_relaxed);
                atomic_fetch_add_explicit(
                    &g_sp_stage2_lock_acquired_by_class[class_idx],
                    1, memory_order_relaxed);
            }

            // Update SuperSlab metadata under mutex
            ss->slab_bitmap |= (1u << claimed_idx);
            ss_slab_meta_class_idx_set(ss, claimed_idx, (uint8_t)class_idx);

            if (ss->active_slabs == 0) {
                ss->active_slabs = 1;
                g_shared_pool.active_count++;
            }
            if (class_idx < TINY_NUM_CLASSES_SS) {
                g_shared_pool.class_active_slots[class_idx]++;
            }

            // Update hint
            g_shared_pool.class_hints[class_idx] = ss;

            *ss_out = ss;
            *slab_idx_out = claimed_idx;
            sp_fix_geometry_if_needed(ss, claimed_idx, class_idx);

            if (g_lock_stats_enabled == 1) {
                atomic_fetch_add(&g_lock_release_count, 1);
            }
            pthread_mutex_unlock(&g_shared_pool.alloc_lock);
	            if (g_sp_stage_stats_enabled) {
	                atomic_fetch_add(&g_sp_stage2_hits[class_idx], 1);
	            }
            return 0;  // ✅ Stage 2 (lock-free) success
        }

        // Claim failed (no UNUSED slots in this meta) - continue to next SuperSlab
    }

    // ========== Tension-Based Drain: Try to create EMPTY slots before Stage 3 ==========
    // If TLS SLL has accumulated blocks, drain them to enable EMPTY slot detection
    // This can avoid allocating new SuperSlabs by reusing EMPTY slots in Stage 1
    // ENV: HAKMEM_TINY_TENSION_DRAIN_ENABLE=0 to disable (default=1)
    // ENV: HAKMEM_TINY_TENSION_DRAIN_THRESHOLD=N to set threshold (default=1024)
    {
        // Priority-2: Use cached ENV
        int tension_drain_enabled = HAK_ENV_TINY_TENSION_DRAIN_ENABLE();
        uint32_t tension_threshold = (uint32_t)HAK_ENV_TINY_TENSION_DRAIN_THRESHOLD();

        if (tension_drain_enabled) {
            extern __thread TinyTLSSLL g_tls_sll[TINY_NUM_CLASSES];
            extern uint32_t tiny_tls_sll_drain(int class_idx, uint32_t batch_size);

            uint32_t sll_count = (class_idx < TINY_NUM_CLASSES) ? g_tls_sll[class_idx].count : 0;

            if (sll_count >= tension_threshold) {
                // Drain all blocks to maximize EMPTY slot creation
                uint32_t drained = tiny_tls_sll_drain(class_idx, 0);  // 0 = drain all

                if (drained > 0) {
                    // Retry Stage 1 (EMPTY reuse) after drain
                    // Some slabs might have become EMPTY (meta->used == 0)
                    goto stage1_retry_after_tension_drain;
                }
            }
        }
    }

    // ========== Stage 3: Mutex-protected fallback (new SuperSlab allocation) ==========
    // All existing SuperSlabs have no UNUSED slots → need new SuperSlab
    // P0 instrumentation: count lock acquisitions
    lock_stats_init();
    if (g_lock_stats_enabled == 1) {
        atomic_fetch_add(&g_lock_acquire_count, 1);
        atomic_fetch_add(&g_lock_acquire_slab_count, 1);
    }

    pthread_mutex_lock(&g_shared_pool.alloc_lock);

    // Performance measurement: count Stage 3 lock acquisitions
    if (__builtin_expect(sp_measure_enabled(), 0)) {
        atomic_fetch_add_explicit(&g_sp_stage3_lock_acquired_global,
                                  1, memory_order_relaxed);
        atomic_fetch_add_explicit(&g_sp_alloc_lock_contention_global,
                                  1, memory_order_relaxed);
        atomic_fetch_add_explicit(&g_sp_stage3_lock_acquired_by_class[class_idx],
                                  1, memory_order_relaxed);
    }

    // ========== Stage 3: Get new SuperSlab ==========
    // Try LRU cache first, then mmap
    SuperSlab* new_ss = NULL;

    // Stage 3a: Try LRU cache
    extern SuperSlab* hak_ss_lru_pop(uint8_t size_class);
    new_ss = hak_ss_lru_pop((uint8_t)class_idx);

    int from_lru = (new_ss != NULL);

    // Stage 3b: If LRU miss, allocate new SuperSlab
    if (!new_ss) {
        // Release the alloc_lock to avoid deadlock with registry during superslab_allocate
        if (g_lock_stats_enabled == 1) {
            atomic_fetch_add(&g_lock_release_count, 1);
        }
        pthread_mutex_unlock(&g_shared_pool.alloc_lock);

        SuperSlab* allocated_ss = sp_internal_allocate_superslab(class_idx);

        // Re-acquire the alloc_lock
        if (g_lock_stats_enabled == 1) {
            atomic_fetch_add(&g_lock_acquire_count, 1);
            atomic_fetch_add(&g_lock_acquire_slab_count, 1); // This is part of acquisition path
        }
        pthread_mutex_lock(&g_shared_pool.alloc_lock);

        if (!allocated_ss) {
            // Allocation failed; return now.
            if (g_lock_stats_enabled == 1) {
                atomic_fetch_add(&g_lock_release_count, 1);
            }
            pthread_mutex_unlock(&g_shared_pool.alloc_lock);
            return -1; // Out of memory
        }

        new_ss = allocated_ss;

        // Add newly allocated SuperSlab to the shared pool's internal array
        if (g_shared_pool.total_count >= g_shared_pool.capacity) {
            shared_pool_ensure_capacity_unlocked(g_shared_pool.total_count + 1);
            if (g_shared_pool.total_count >= g_shared_pool.capacity) {
                // Pool table expansion failed; leave ss alive (registry-owned),
                // but do not treat it as part of shared_pool.
                // This is a critical error, return early.
                if (g_lock_stats_enabled == 1) {
                    atomic_fetch_add(&g_lock_release_count, 1);
                }
                pthread_mutex_unlock(&g_shared_pool.alloc_lock);
                return -1;
            }
        }
        g_shared_pool.slabs[g_shared_pool.total_count] = new_ss;
        g_shared_pool.total_count++;
    }

    #if !HAKMEM_BUILD_RELEASE
    if (dbg_acquire == 1 && new_ss) {
        fprintf(stderr, "[SP_ACQUIRE_STAGE3] class=%d new SuperSlab (ss=%p from_lru=%d)\n",
                class_idx, (void*)new_ss, from_lru);
    }
    #endif

    if (!new_ss) {
        if (g_lock_stats_enabled == 1) {
            atomic_fetch_add(&g_lock_release_count, 1);
        }
        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
        return -1;  // ❌ Out of memory
    }

    // Before creating a new SuperSlab, consult learning-layer soft cap.
    // Phase 9-2: Soft Cap removed to allow Shared Pool to fully replace Legacy Backend.
    // We now rely on LRU eviction and EMPTY recycling to manage memory pressure.

    // Create metadata for this new SuperSlab
    SharedSSMeta* new_meta = sp_meta_find_or_create(new_ss);
    if (!new_meta) {
        if (g_lock_stats_enabled == 1) {
            atomic_fetch_add(&g_lock_release_count, 1);
        }
        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
        return -1;  // ❌ Metadata allocation failed
    }

    // Assign first slot to this class
    int first_slot = 0;
    if (sp_slot_mark_active(new_meta, first_slot, class_idx) != 0) {
        if (g_lock_stats_enabled == 1) {
            atomic_fetch_add(&g_lock_release_count, 1);
        }
        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
        return -1;  // ❌ Should not happen
    }

    // Update SuperSlab metadata
    new_ss->slab_bitmap |= (1u << first_slot);
    ss_slab_meta_class_idx_set(new_ss, first_slot, (uint8_t)class_idx);
    new_ss->active_slabs = 1;
    g_shared_pool.active_count++;
    if (class_idx < TINY_NUM_CLASSES_SS) {
        g_shared_pool.class_active_slots[class_idx]++;
    }

    // Update hint
    g_shared_pool.class_hints[class_idx] = new_ss;

    *ss_out = new_ss;
    *slab_idx_out = first_slot;
    sp_fix_geometry_if_needed(new_ss, first_slot, class_idx);

    if (g_lock_stats_enabled == 1) {
        atomic_fetch_add(&g_lock_release_count, 1);
    }
    pthread_mutex_unlock(&g_shared_pool.alloc_lock);
	    if (g_sp_stage_stats_enabled) {
	        atomic_fetch_add(&g_sp_stage3_hits[class_idx], 1);
	    }
    return 0;  // ✅ Stage 3 success
}

// ============================================================================
// Performance Measurement: Print Statistics
// ============================================================================
void shared_pool_print_measurements(void) {
    if (!sp_measure_enabled()) {
        return;  // Measurement disabled
    }

    uint64_t stage2 = atomic_load_explicit(&g_sp_stage2_lock_acquired_global,
                                           memory_order_relaxed);
    uint64_t stage3 = atomic_load_explicit(&g_sp_stage3_lock_acquired_global,
                                           memory_order_relaxed);
    uint64_t total_locks = atomic_load_explicit(&g_sp_alloc_lock_contention_global,
                                                memory_order_relaxed);

    if (total_locks == 0) {
        fprintf(stderr, "\n========================================\n");
        fprintf(stderr, "Shared Pool Contention Statistics\n");
        fprintf(stderr, "========================================\n");
        fprintf(stderr, "No lock acquisitions recorded\n");
        fprintf(stderr, "========================================\n\n");
        return;
    }

    double stage2_pct = (100.0 * stage2) / total_locks;
    double stage3_pct = (100.0 * stage3) / total_locks;

    fprintf(stderr, "\n========================================\n");
    fprintf(stderr, "Shared Pool Contention Statistics\n");
    fprintf(stderr, "========================================\n");
    fprintf(stderr, "Stage 2 Locks:    %llu (%.1f%%)\n",
            (unsigned long long)stage2, stage2_pct);
    fprintf(stderr, "Stage 3 Locks:    %llu (%.1f%%)\n",
            (unsigned long long)stage3, stage3_pct);
    fprintf(stderr, "Total Contention: %llu lock acquisitions\n",
            (unsigned long long)total_locks);

    // Per-class breakdown（Tiny 用クラス 0-7、特に C5–C7 を観測）
    fprintf(stderr, "\nPer-class Shared Pool Locks (Stage2/Stage3):\n");
    for (int cls = 0; cls < TINY_NUM_CLASSES_SS; cls++) {
        uint64_t s2c = atomic_load_explicit(
            &g_sp_stage2_lock_acquired_by_class[cls],
            memory_order_relaxed);
        uint64_t s3c = atomic_load_explicit(
            &g_sp_stage3_lock_acquired_by_class[cls],
            memory_order_relaxed);
        uint64_t tc = s2c + s3c;
        if (tc == 0) {
            continue;  // ロック取得のないクラスは省略
        }
        fprintf(stderr,
                "  C%d: Stage2=%llu Stage3=%llu Total=%llu\n",
                cls,
                (unsigned long long)s2c,
                (unsigned long long)s3c,
                (unsigned long long)tc);
    }

    fprintf(stderr, "========================================\n\n");
}
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								#include "hakmem_shared_pool_internal.h"
 								#include "hakmem_debug_master.h"
 								#include "hakmem_stats_master.h"
 								#include "box/ss_slab_meta_box.h"
 								#include "box/ss_hot_cold_box.h"
 								#include "box/pagefault_telemetry_box.h"
 								#include "box/tls_sll_drain_box.h"
 								#include "box/tls_slab_reuse_guard_box.h"
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
+								#include "box/ss_tier_box.h"  // P-Tier: Tier filtering support
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								#include "hakmem_policy.h"
-												Priority-2 ENV Cache: Shared Pool Acquire (5変数追加、5箇所置換)

【追加ENV変数】
- HAKMEM_SS_EMPTY_REUSE (default: 1)
- HAKMEM_SS_EMPTY_SCAN_LIMIT (default: 32)
- HAKMEM_SS_ACQUIRE_DEBUG (default: 0)
- HAKMEM_TINY_TENSION_DRAIN_ENABLE (default: 1)
- HAKMEM_TINY_TENSION_DRAIN_THRESHOLD (default: 1024)

【置換ファイル】
- core/hakmem_shared_pool_acquire.c (5箇所 → ENV Cache)

【変更詳細】
1. ENV Cache (hakmem_env_cache.h):
   - 構造体に5変数追加 (41→46変数)
   - hakmem_env_cache_init()に初期化追加
   - アクセサマクロ5個追加
   - カウント更新: 41→46

2. hakmem_shared_pool_acquire.c:
   - getenv("HAKMEM_SS_EMPTY_REUSE") → HAK_ENV_SS_EMPTY_REUSE()
   - getenv("HAKMEM_SS_EMPTY_SCAN_LIMIT") → HAK_ENV_SS_EMPTY_SCAN_LIMIT()
   - getenv("HAKMEM_SS_ACQUIRE_DEBUG") → HAK_ENV_SS_ACQUIRE_DEBUG()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_ENABLE") → HAK_ENV_TINY_TENSION_DRAIN_ENABLE()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_THRESHOLD") → HAK_ENV_TINY_TENSION_DRAIN_THRESHOLD()
   - #include "hakmem_env_cache.h" 追加

【効果】
- Shared Pool Acquire warm pathからgetenv()呼び出しを完全排除
- Lock-free Stage2のgetenv()オーバーヘッド削減

【テスト】
✅ make shared → 成功
✅ /tmp/test_mixed3_final → PASSED

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-02 20:51:50 +09:00
+								#include "hakmem_env_cache.h"  // Priority-2: ENV cache
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								#include "front/tiny_warm_pool.h"  // Warm Pool: Prefill during registry scans
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
 								#include <stdlib.h>
 								#include <stdio.h>
 								#include <stdatomic.h>
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								// ============================================================================
 								// Performance Measurement: Shared Pool Lock Contention (ENV-gated)
 								// ============================================================================
 								// Global atomic counters for lock contention measurement
 								// ENV: HAKMEM_MEASURE_UNIFIED_CACHE=1 to enable (default: OFF)
 								_Atomic uint64_t g_sp_stage2_lock_acquired_global = 0;
 								_Atomic uint64_t g_sp_stage3_lock_acquired_global = 0;
 								_Atomic uint64_t g_sp_alloc_lock_contention_global = 0;
-												Add Page Box layer for C7 class optimization

- Implement tiny_page_box.c/h: per-thread page cache between UC and Shared Pool
- Integrate Page Box into Unified Cache refill path
- Remove legacy SuperSlab implementation (merged into smallmid)
- Add HAKMEM_TINY_PAGE_BOX_CLASSES env var for selective class enabling
- Update bench_random_mixed.c with Page Box statistics

Current status: Implementation safe, no regressions.
Page Box ON/OFF shows minimal difference - pool strategy needs tuning.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 15:31:44 +09:00
+								// Per-class lock acquisition statistics（Tiny クラス別の lock 負荷観測用）
 								_Atomic uint64_t g_sp_stage2_lock_acquired_by_class[TINY_NUM_CLASSES_SS] = {0};
 								_Atomic uint64_t g_sp_stage3_lock_acquired_by_class[TINY_NUM_CLASSES_SS] = {0};
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								// Check if measurement is enabled (cached)
 								static inline int sp_measure_enabled(void) {
 								    static int g_measure = -1;
 								    if (__builtin_expect(g_measure == -1, 0)) {
 								        const char* e = getenv("HAKMEM_MEASURE_UNIFIED_CACHE");
 								        g_measure = (e && *e && *e != '0') ? 1 : 0;
-												Add Page Box layer for C7 class optimization

- Implement tiny_page_box.c/h: per-thread page cache between UC and Shared Pool
- Integrate Page Box into Unified Cache refill path
- Remove legacy SuperSlab implementation (merged into smallmid)
- Add HAKMEM_TINY_PAGE_BOX_CLASSES env var for selective class enabling
- Update bench_random_mixed.c with Page Box statistics

Current status: Implementation safe, no regressions.
Page Box ON/OFF shows minimal difference - pool strategy needs tuning.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 15:31:44 +09:00
+								        if (g_measure == 1) {
 								            // Measurement が ON のときは per-class stage stats も有効化する
 								            // （Stage1/2/3 ヒット数は g_sp_stage*_hits に集計される）
 								            extern int g_sp_stage_stats_enabled;
 								            g_sp_stage_stats_enabled = 1;
 								        }
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								    }
 								    return g_measure;
 								}
 								// Print statistics function
 								void shared_pool_print_measurements(void);
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								// Stage 0.5: EMPTY slab direct scan（registry ベースの EMPTY 再利用）
 								// Scan existing SuperSlabs for EMPTY slabs (highest reuse priority) to
 								// avoid Stage 3 (mmap) when freed slabs are available.
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								//
 								// WARM POOL OPTIMIZATION:
 								// - During the registry scan, prefill warm pool with HOT SuperSlabs
 								// - This eliminates future registry scans for cache misses
 								// - Expected gain: +40-50% by reducing O(N) scan overhead
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								static inline int
 								sp_acquire_from_empty_scan(int class_idx, SuperSlab** ss_out, int* slab_idx_out, int dbg_acquire)
 								{
-												Priority-2 ENV Cache: Shared Pool Acquire (5変数追加、5箇所置換)

【追加ENV変数】
- HAKMEM_SS_EMPTY_REUSE (default: 1)
- HAKMEM_SS_EMPTY_SCAN_LIMIT (default: 32)
- HAKMEM_SS_ACQUIRE_DEBUG (default: 0)
- HAKMEM_TINY_TENSION_DRAIN_ENABLE (default: 1)
- HAKMEM_TINY_TENSION_DRAIN_THRESHOLD (default: 1024)

【置換ファイル】
- core/hakmem_shared_pool_acquire.c (5箇所 → ENV Cache)

【変更詳細】
1. ENV Cache (hakmem_env_cache.h):
   - 構造体に5変数追加 (41→46変数)
   - hakmem_env_cache_init()に初期化追加
   - アクセサマクロ5個追加
   - カウント更新: 41→46

2. hakmem_shared_pool_acquire.c:
   - getenv("HAKMEM_SS_EMPTY_REUSE") → HAK_ENV_SS_EMPTY_REUSE()
   - getenv("HAKMEM_SS_EMPTY_SCAN_LIMIT") → HAK_ENV_SS_EMPTY_SCAN_LIMIT()
   - getenv("HAKMEM_SS_ACQUIRE_DEBUG") → HAK_ENV_SS_ACQUIRE_DEBUG()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_ENABLE") → HAK_ENV_TINY_TENSION_DRAIN_ENABLE()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_THRESHOLD") → HAK_ENV_TINY_TENSION_DRAIN_THRESHOLD()
   - #include "hakmem_env_cache.h" 追加

【効果】
- Shared Pool Acquire warm pathからgetenv()呼び出しを完全排除
- Lock-free Stage2のgetenv()オーバーヘッド削減

【テスト】
✅ make shared → 成功
✅ /tmp/test_mixed3_final → PASSED

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-02 20:51:50 +09:00
+								    // Priority-2: Use cached ENV
 								    int empty_reuse_enabled = HAK_ENV_SS_EMPTY_REUSE();
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								    if (!empty_reuse_enabled) {
 								        return -1;
 								    }
 								    extern SuperSlab* g_super_reg_by_class[TINY_NUM_CLASSES][SUPER_REG_PER_CLASS];
 								    extern int g_super_reg_class_size[TINY_NUM_CLASSES];
 								    int reg_size = (class_idx < TINY_NUM_CLASSES) ? g_super_reg_class_size[class_idx] : 0;
-												Priority-2 ENV Cache: Shared Pool Acquire (5変数追加、5箇所置換)

【追加ENV変数】
- HAKMEM_SS_EMPTY_REUSE (default: 1)
- HAKMEM_SS_EMPTY_SCAN_LIMIT (default: 32)
- HAKMEM_SS_ACQUIRE_DEBUG (default: 0)
- HAKMEM_TINY_TENSION_DRAIN_ENABLE (default: 1)
- HAKMEM_TINY_TENSION_DRAIN_THRESHOLD (default: 1024)

【置換ファイル】
- core/hakmem_shared_pool_acquire.c (5箇所 → ENV Cache)

【変更詳細】
1. ENV Cache (hakmem_env_cache.h):
   - 構造体に5変数追加 (41→46変数)
   - hakmem_env_cache_init()に初期化追加
   - アクセサマクロ5個追加
   - カウント更新: 41→46

2. hakmem_shared_pool_acquire.c:
   - getenv("HAKMEM_SS_EMPTY_REUSE") → HAK_ENV_SS_EMPTY_REUSE()
   - getenv("HAKMEM_SS_EMPTY_SCAN_LIMIT") → HAK_ENV_SS_EMPTY_SCAN_LIMIT()
   - getenv("HAKMEM_SS_ACQUIRE_DEBUG") → HAK_ENV_SS_ACQUIRE_DEBUG()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_ENABLE") → HAK_ENV_TINY_TENSION_DRAIN_ENABLE()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_THRESHOLD") → HAK_ENV_TINY_TENSION_DRAIN_THRESHOLD()
   - #include "hakmem_env_cache.h" 追加

【効果】
- Shared Pool Acquire warm pathからgetenv()呼び出しを完全排除
- Lock-free Stage2のgetenv()オーバーヘッド削減

【テスト】
✅ make shared → 成功
✅ /tmp/test_mixed3_final → PASSED

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-02 20:51:50 +09:00
+								    // Priority-2: Use cached ENV
 								    int scan_limit = HAK_ENV_SS_EMPTY_SCAN_LIMIT();
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								    if (scan_limit > reg_size) scan_limit = reg_size;
 								    // Stage 0.5 hit counter for visualization
 								    static _Atomic uint64_t stage05_hits = 0;
 								    static _Atomic uint64_t stage05_attempts = 0;
 								    atomic_fetch_add_explicit(&stage05_attempts, 1, memory_order_relaxed);
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								    // Initialize warm pool on first use (per-thread, one-time)
 								    tiny_warm_pool_init_once();
 								    // Track SuperSlabs scanned during this acquire call for warm pool prefill
 								    SuperSlab* primary_result = NULL;
 								    int primary_slab_idx = -1;
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								    for (int i = 0; i < scan_limit; i++) {
 								        SuperSlab* ss = g_super_reg_by_class[class_idx][i];
 								        if (!(ss && ss->magic == SUPERSLAB_MAGIC)) continue;
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
+								        // P-Tier: Skip DRAINING tier SuperSlabs
 								        if (!ss_tier_is_hot(ss)) continue;
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								        if (ss->empty_count == 0) continue;  // No EMPTY slabs in this SS
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								        // WARM POOL PREFILL: Add HOT SuperSlabs to warm pool (if not already primary result)
 								        // This is low-cost during registry scan and avoids future expensive scans
-												Phase 1: Warm Pool Capacity Increase (16 → 12 with matching threshold)

Key Changes:
- Reduced static capacity from 16 to 12 SuperSlabs per class
- Fixed prefill threshold from hardcoded 4 to match capacity (12)
- Updated environment variable clamping to [1,12]
- This allows warm pool to actually utilize its full capacity

Performance:
- Baseline (post-unified-cache-opt): 4.76M ops/s
- After Phase 1: 4.84M ops/s
- Improvement: +1.6% (expected +15-20%)

Note: Actual improvement lower than expected because the warm pool
bottleneck is only part of the overall allocation path. Unified cache
optimization (+14.9%) already addressed much of the registry scan overhead.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 12:16:39 +09:00
+								        // Phase 1: Increase threshold from 4 to 12 to match TINY_WARM_POOL_MAX_PER_CLASS
 								        if (ss != primary_result && tiny_warm_pool_count(class_idx) < 12) {
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								            tiny_warm_pool_push(class_idx, ss);
 								            // Track prefilled SuperSlabs for metrics
 								            g_warm_pool_stats[class_idx].prefilled++;
 								        }
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								        uint32_t mask = ss->empty_mask;
 								        while (mask) {
 								            int empty_idx = __builtin_ctz(mask);
 								            mask &= (mask - 1);  // clear lowest bit
 								            TinySlabMeta* meta = &ss->slabs[empty_idx];
 								            if (meta->capacity > 0 && meta->used == 0) {
 								                tiny_tls_slab_reuse_guard(ss);
 								                ss_clear_slab_empty(ss, empty_idx);
 								                meta->class_idx = (uint8_t)class_idx;
 								                ss->class_map[empty_idx] = (uint8_t)class_idx;
 								#if !HAKMEM_BUILD_RELEASE
 								                if (dbg_acquire == 1) {
 								                    fprintf(stderr,
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								                            "[SP_ACQUIRE_STAGE0.5_EMPTY] class=%d reusing EMPTY slab (ss=%p slab=%d empty_count=%u warm_pool_size=%d)\n",
 								                            class_idx, (void*)ss, empty_idx, ss->empty_count, tiny_warm_pool_count(class_idx));
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								                }
 								#else
 								                (void)dbg_acquire;
 								#endif
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								                // Store primary result but continue scanning to prefill warm pool
 								                if (primary_result == NULL) {
 								                    primary_result = ss;
 								                    primary_slab_idx = empty_idx;
 								                    *ss_out = ss;
 								                    *slab_idx_out = empty_idx;
 								                    sp_stage_stats_init();
 								                    if (g_sp_stage_stats_enabled) {
 								                        atomic_fetch_add(&g_sp_stage1_hits[class_idx], 1);
 								                    }
 								                    atomic_fetch_add_explicit(&stage05_hits, 1, memory_order_relaxed);
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								                }
 								            }
 								        }
 								    }
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
 								    if (primary_result != NULL) {
 								        // Stage 0.5 hit rate visualization (every 100 hits)
 								        uint64_t hits = atomic_load_explicit(&stage05_hits, memory_order_relaxed);
 								        if (hits % 100 == 1) {
 								            uint64_t attempts = atomic_load_explicit(&stage05_attempts, memory_order_relaxed);
 								            fprintf(stderr, "[STAGE0.5_STATS] hits=%lu attempts=%lu rate=%.1f%% (scan_limit=%d warm_pool=%d)\n",
 								                    hits, attempts, (double)hits * 100.0 / attempts, scan_limit, tiny_warm_pool_count(class_idx));
 								        }
 								        return 0;
 								    }
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								    return -1;
 								}
 								int
 								shared_pool_acquire_slab(int class_idx, SuperSlab** ss_out, int* slab_idx_out)
 								{
 								    // Phase 12: SP-SLOT Box - 3-Stage Acquire Logic
 								    //
 								    // Stage 1: Reuse EMPTY slots from per-class free list (EMPTY→ACTIVE)
 								    // Stage 2: Find UNUSED slots in existing SuperSlabs
 								    // Stage 3: Get new SuperSlab (LRU pop or mmap)
 								    //
 								    // Invariants:
 								    //  - On success: *ss_out != NULL, 0 <= *slab_idx_out < total_slots
 								    //  - The chosen slab has meta->class_idx == class_idx
 								    if (!ss_out || !slab_idx_out) {
 								        return -1;
 								    }
 								    if (class_idx < 0 || class_idx >= TINY_NUM_CLASSES_SS) {
 								        return -1;
 								    }
 								    shared_pool_init();
 								    // Debug logging / stage stats
 								#if !HAKMEM_BUILD_RELEASE
-												Priority-2 ENV Cache: Shared Pool Acquire (5変数追加、5箇所置換)

【追加ENV変数】
- HAKMEM_SS_EMPTY_REUSE (default: 1)
- HAKMEM_SS_EMPTY_SCAN_LIMIT (default: 32)
- HAKMEM_SS_ACQUIRE_DEBUG (default: 0)
- HAKMEM_TINY_TENSION_DRAIN_ENABLE (default: 1)
- HAKMEM_TINY_TENSION_DRAIN_THRESHOLD (default: 1024)

【置換ファイル】
- core/hakmem_shared_pool_acquire.c (5箇所 → ENV Cache)

【変更詳細】
1. ENV Cache (hakmem_env_cache.h):
   - 構造体に5変数追加 (41→46変数)
   - hakmem_env_cache_init()に初期化追加
   - アクセサマクロ5個追加
   - カウント更新: 41→46

2. hakmem_shared_pool_acquire.c:
   - getenv("HAKMEM_SS_EMPTY_REUSE") → HAK_ENV_SS_EMPTY_REUSE()
   - getenv("HAKMEM_SS_EMPTY_SCAN_LIMIT") → HAK_ENV_SS_EMPTY_SCAN_LIMIT()
   - getenv("HAKMEM_SS_ACQUIRE_DEBUG") → HAK_ENV_SS_ACQUIRE_DEBUG()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_ENABLE") → HAK_ENV_TINY_TENSION_DRAIN_ENABLE()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_THRESHOLD") → HAK_ENV_TINY_TENSION_DRAIN_THRESHOLD()
   - #include "hakmem_env_cache.h" 追加

【効果】
- Shared Pool Acquire warm pathからgetenv()呼び出しを完全排除
- Lock-free Stage2のgetenv()オーバーヘッド削減

【テスト】
✅ make shared → 成功
✅ /tmp/test_mixed3_final → PASSED

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-02 20:51:50 +09:00
+								    // Priority-2: Use cached ENV
 								    int dbg_acquire = HAK_ENV_SS_ACQUIRE_DEBUG();
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								#else
 								    static const int dbg_acquire = 0;
 								#endif
 								    sp_stage_stats_init();
 								stage1_retry_after_tension_drain:
 								    // ========== Stage 0.5 (Phase 12-1.1): EMPTY slab direct scan ==========
 								    // Scan existing SuperSlabs for EMPTY slabs (highest reuse priority) to
 								    // avoid Stage 3 (mmap) when freed slabs are available.
 								    if (sp_acquire_from_empty_scan(class_idx, ss_out, slab_idx_out, dbg_acquire) == 0) {
 								        return 0;
 								    }
 								    // ========== Stage 1 (Lock-Free): Try to reuse EMPTY slots ==========
 								    // P0-4: Lock-free pop from per-class free list (no mutex needed!)
 								    // Best case: Same class freed a slot, reuse immediately (cache-hot)
 								    SharedSSMeta* reuse_meta = NULL;
 								    int reuse_slot_idx = -1;
 								    if (sp_freelist_pop_lockfree(class_idx, &reuse_meta, &reuse_slot_idx)) {
 								        // Found EMPTY slot from lock-free list!
 								        // Now acquire mutex ONLY for slot activation and metadata update
 								        // P0 instrumentation: count lock acquisitions
 								        lock_stats_init();
 								        if (g_lock_stats_enabled == 1) {
 								            atomic_fetch_add(&g_lock_acquire_count, 1);
 								            atomic_fetch_add(&g_lock_acquire_slab_count, 1);
 								        }
 								        pthread_mutex_lock(&g_shared_pool.alloc_lock);
 								        // P0.3: Guard against TLS SLL orphaned pointers before reusing slab
 								        // RACE FIX: Load SuperSlab pointer atomically BEFORE guard (consistency)
 								        SuperSlab* ss_guard = atomic_load_explicit(&reuse_meta->ss, memory_order_relaxed);
 								        if (ss_guard) {
 								            tiny_tls_slab_reuse_guard(ss_guard);
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
-												Implement Warm Pool Secondary Prefill Optimization (Phase B-2c Complete)

Problem: Warm pool had 0% hit rate (only 1 hit per 3976 misses) despite being
implemented, causing all cache misses to go through expensive superslab_refill
registry scans.

Root Cause Analysis:
- Warm pool was initialized once and pushed a single slab after each refill
- When that slab was exhausted, it was discarded (not pushed back)
- Next refill would push another single slab, which was immediately exhausted
- Pool would oscillate between 0 and 1 items, yielding 0% hit rate

Solution: Secondary Prefill on Cache Miss
When warm pool becomes empty, we now do multiple superslab_refills and prefill
the pool with 3 additional HOT superlslabs before attempting to carve. This
builds a working set of slabs that can sustain allocation pressure.

Implementation Details:
- Modified unified_cache_refill() cold path to detect empty pool
- Added prefill loop: when pool count == 0, load 3 extra superlslabs
- Store extra slabs in warm pool, keep 1 in TLS for immediate carving
- Track prefill events in g_warm_pool_stats[].prefilled counter

Results (1M Random Mixed 256B allocations):
- Before: C7 hits=1, misses=3976, hit_rate=0.0%
- After:  C7 hits=3929, misses=3143, hit_rate=55.6%
- Throughput: 4.055M ops/s (maintained vs 4.07M baseline)
- Stability: Consistent 55.6% hit rate at 5M allocations (4.102M ops/s)

Performance Impact:
- No regression: throughput remained stable at ~4.1M ops/s
- Registry scan avoided in 55.6% of cache misses (significant savings)
- Warm pool now functioning as intended with strong locality

Configuration:
- TINY_WARM_POOL_MAX_PER_CLASS increased from 4 to 16 to support prefill
- Prefill budget hardcoded to 3 (tunable via env var if needed later)
- All statistics always compiled, ENV-gated printing via HAKMEM_WARM_POOL_STATS=1

Next Steps:
- Monitor for further optimization opportunities (prefill budget tuning)
- Consider adaptive prefill budget based on class-specific hit rates
- Validate at larger allocation counts (10M+ pending registry size fix)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 23:31:54 +09:00
+								            // P-Tier: Skip DRAINING tier SuperSlabs
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
+								            if (!ss_tier_is_hot(ss_guard)) {
 								                // DRAINING SuperSlab - skip this slot and fall through to Stage 2
 								                if (g_lock_stats_enabled == 1) {
 								                    atomic_fetch_add(&g_lock_release_count, 1);
 								                }
 								                pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								                goto stage2_fallback;
 								            }
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								        }
 								        // Activate slot under mutex (slot state transition requires protection)
 								        if (sp_slot_mark_active(reuse_meta, reuse_slot_idx, class_idx) == 0) {
 								            // RACE FIX: Load SuperSlab pointer atomically (consistency)
 								            SuperSlab* ss = atomic_load_explicit(&reuse_meta->ss, memory_order_relaxed);
 								            // RACE FIX: Check if SuperSlab was freed (NULL pointer)
 								            // This can happen if Thread A freed the SuperSlab after pushing slot to freelist,
 								            // but Thread B popped the stale slot before the freelist was cleared.
 								            if (!ss) {
 								                // SuperSlab freed - skip and fall through to Stage 2/3
 								                if (g_lock_stats_enabled == 1) {
 								                    atomic_fetch_add(&g_lock_release_count, 1);
 								                }
 								                pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								                goto stage2_fallback;
 								            }
 								            #if !HAKMEM_BUILD_RELEASE
 								            if (dbg_acquire == 1) {
 								                fprintf(stderr, "[SP_ACQUIRE_STAGE1_LOCKFREE] class=%d reusing EMPTY slot (ss=%p slab=%d)\n",
 								                        class_idx, (void*)ss, reuse_slot_idx);
 								            }
 								            #endif
 								            // Update SuperSlab metadata
 								            ss->slab_bitmap |= (1u << reuse_slot_idx);
 								            ss_slab_meta_class_idx_set(ss, reuse_slot_idx, (uint8_t)class_idx);
 								            if (ss->active_slabs == 0) {
 								                // Was empty, now active again
 								                ss->active_slabs = 1;
 								                g_shared_pool.active_count++;
 								            }
 								            // Track per-class active slots (approximate, under alloc_lock)
 								            if (class_idx < TINY_NUM_CLASSES_SS) {
 								                g_shared_pool.class_active_slots[class_idx]++;
 								            }
 								            // Update hint
 								            g_shared_pool.class_hints[class_idx] = ss;
 								            *ss_out = ss;
 								            *slab_idx_out = reuse_slot_idx;
 								            if (g_lock_stats_enabled == 1) {
 								                atomic_fetch_add(&g_lock_release_count, 1);
 								            }
 								            pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 									            if (g_sp_stage_stats_enabled) {
 									                atomic_fetch_add(&g_sp_stage1_hits[class_idx], 1);
 									            }
 								            return 0;  // ✅ Stage 1 (lock-free) success
 								        }
 								        // Slot activation failed (race condition?) - release lock and fall through
 								        if (g_lock_stats_enabled == 1) {
 								            atomic_fetch_add(&g_lock_release_count, 1);
 								        }
 								        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								    }
 								stage2_fallback:
 								    // ========== Stage 2 (Lock-Free): Try to claim UNUSED slots ==========
-												P0 Optimization: Shared Pool fast path with O(1) metadata lookup

Performance Results:
- Throughput: 2.66M ops/s → 3.8M ops/s (+43% improvement)
- sp_meta_find_or_create: O(N) linear scan → O(1) direct pointer
- Stage 2 metadata scan: 100% → 10-20% (80-90% reduction via hints)

Core Optimizations:

1. O(1) Metadata Lookup (superslab_types.h)
   - Added `shared_meta` pointer field to SuperSlab struct
   - Eliminates O(N) linear search through ss_metadata[] array
   - First access: O(N) scan + cache | Subsequent: O(1) direct return

2. sp_meta_find_or_create Fast Path (hakmem_shared_pool.c)
   - Check cached ss->shared_meta first before linear scan
   - Cache pointer after successful linear scan for future lookups
   - Reduces 7.8% CPU hotspot to near-zero for hot paths

3. Stage 2 Class Hints Fast Path (hakmem_shared_pool_acquire.c)
   - Try class_hints[class_idx] FIRST before full metadata scan
   - Uses O(1) ss->shared_meta lookup for hint validation
   - __builtin_expect() for branch prediction optimization
   - 80-90% of acquire calls now skip full metadata scan

4. Proper Initialization (ss_allocation_box.c)
   - Initialize shared_meta = NULL in superslab_allocate()
   - Ensures correct NULL-check semantics for new SuperSlabs

Additional Improvements:
- Updated ptr_trace and debug ring for release build efficiency
- Enhanced ENV variable documentation and analysis
- Added learner_env_box.h for configuration management
- Various Box optimizations for reduced overhead

Thread Safety:
- All atomic operations use correct memory ordering
- shared_meta cached under mutex protection
- Lock-free Stage 2 uses proper CAS with acquire/release semantics

Testing:
- Benchmark: 1M iterations, 3.8M ops/s stable
- Build: Clean compile RELEASE=0 and RELEASE=1
- No crashes, memory leaks, or correctness issues

Next Optimization Candidates:
- P1: Per-SuperSlab free slot bitmap for O(1) slot claiming
- P2: Reduce Stage 2 critical section size
- P3: Page pre-faulting (MAP_POPULATE)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 16:21:54 +09:00
+								    // P0 Optimization: Try class hint FIRST for fast path (same class locality)
 								    // This reduces metadata scan from 100% to ~10% when hints are effective
 								    {
 								        SuperSlab* hint_ss = g_shared_pool.class_hints[class_idx];
 								        if (__builtin_expect(hint_ss != NULL, 1)) {
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
+								            // P-Tier: Skip DRAINING tier SuperSlabs
 								            if (!ss_tier_is_hot(hint_ss)) {
 								                // Clear stale hint pointing to DRAINING SuperSlab
 								                g_shared_pool.class_hints[class_idx] = NULL;
 								                goto stage2_scan;
 								            }
-												P0 Optimization: Shared Pool fast path with O(1) metadata lookup

Performance Results:
- Throughput: 2.66M ops/s → 3.8M ops/s (+43% improvement)
- sp_meta_find_or_create: O(N) linear scan → O(1) direct pointer
- Stage 2 metadata scan: 100% → 10-20% (80-90% reduction via hints)

Core Optimizations:

1. O(1) Metadata Lookup (superslab_types.h)
   - Added `shared_meta` pointer field to SuperSlab struct
   - Eliminates O(N) linear search through ss_metadata[] array
   - First access: O(N) scan + cache | Subsequent: O(1) direct return

2. sp_meta_find_or_create Fast Path (hakmem_shared_pool.c)
   - Check cached ss->shared_meta first before linear scan
   - Cache pointer after successful linear scan for future lookups
   - Reduces 7.8% CPU hotspot to near-zero for hot paths

3. Stage 2 Class Hints Fast Path (hakmem_shared_pool_acquire.c)
   - Try class_hints[class_idx] FIRST before full metadata scan
   - Uses O(1) ss->shared_meta lookup for hint validation
   - __builtin_expect() for branch prediction optimization
   - 80-90% of acquire calls now skip full metadata scan

4. Proper Initialization (ss_allocation_box.c)
   - Initialize shared_meta = NULL in superslab_allocate()
   - Ensures correct NULL-check semantics for new SuperSlabs

Additional Improvements:
- Updated ptr_trace and debug ring for release build efficiency
- Enhanced ENV variable documentation and analysis
- Added learner_env_box.h for configuration management
- Various Box optimizations for reduced overhead

Thread Safety:
- All atomic operations use correct memory ordering
- shared_meta cached under mutex protection
- Lock-free Stage 2 uses proper CAS with acquire/release semantics

Testing:
- Benchmark: 1M iterations, 3.8M ops/s stable
- Build: Clean compile RELEASE=0 and RELEASE=1
- No crashes, memory leaks, or correctness issues

Next Optimization Candidates:
- P1: Per-SuperSlab free slot bitmap for O(1) slot claiming
- P2: Reduce Stage 2 critical section size
- P3: Page pre-faulting (MAP_POPULATE)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 16:21:54 +09:00
+								            // P0 Optimization: O(1) lookup via cached pointer (avoids metadata scan)
 								            SharedSSMeta* hint_meta = hint_ss->shared_meta;
 								            if (__builtin_expect(hint_meta != NULL, 1)) {
 								                // Try lock-free claiming on hint SuperSlab first
 								                int claimed_idx = sp_slot_claim_lockfree(hint_meta, class_idx);
 								                if (__builtin_expect(claimed_idx >= 0, 1)) {
 								                    // Fast path success! No need to scan all metadata
 								                    SuperSlab* ss = atomic_load_explicit(&hint_meta->ss, memory_order_acquire);
 								                    if (__builtin_expect(ss != NULL, 1)) {
 								                        #if !HAKMEM_BUILD_RELEASE
 								                        if (dbg_acquire == 1) {
 								                            fprintf(stderr, "[SP_ACQUIRE_STAGE2_HINT] class=%d claimed UNUSED slot from hint (ss=%p slab=%d)\n",
 								                                    class_idx, (void*)ss, claimed_idx);
 								                        }
 								                        #endif
 								                        // P0 instrumentation: count lock acquisitions
 								                        lock_stats_init();
 								                        if (g_lock_stats_enabled == 1) {
 								                            atomic_fetch_add(&g_lock_acquire_count, 1);
 								                            atomic_fetch_add(&g_lock_acquire_slab_count, 1);
 								                        }
 								                        pthread_mutex_lock(&g_shared_pool.alloc_lock);
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								                        // Performance measurement: count Stage 2 lock acquisitions
 								                        if (__builtin_expect(sp_measure_enabled(), 0)) {
-												Add Page Box layer for C7 class optimization

- Implement tiny_page_box.c/h: per-thread page cache between UC and Shared Pool
- Integrate Page Box into Unified Cache refill path
- Remove legacy SuperSlab implementation (merged into smallmid)
- Add HAKMEM_TINY_PAGE_BOX_CLASSES env var for selective class enabling
- Update bench_random_mixed.c with Page Box statistics

Current status: Implementation safe, no regressions.
Page Box ON/OFF shows minimal difference - pool strategy needs tuning.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 15:31:44 +09:00
+								                            atomic_fetch_add_explicit(&g_sp_stage2_lock_acquired_global,
 , memory_order_relaxed);
 								                            atomic_fetch_add_explicit(&g_sp_alloc_lock_contention_global,
 , memory_order_relaxed);
 								                            atomic_fetch_add_explicit(
 								                                &g_sp_stage2_lock_acquired_by_class[class_idx],
 , memory_order_relaxed);
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								                        }
-												P0 Optimization: Shared Pool fast path with O(1) metadata lookup

Performance Results:
- Throughput: 2.66M ops/s → 3.8M ops/s (+43% improvement)
- sp_meta_find_or_create: O(N) linear scan → O(1) direct pointer
- Stage 2 metadata scan: 100% → 10-20% (80-90% reduction via hints)

Core Optimizations:

1. O(1) Metadata Lookup (superslab_types.h)
   - Added `shared_meta` pointer field to SuperSlab struct
   - Eliminates O(N) linear search through ss_metadata[] array
   - First access: O(N) scan + cache | Subsequent: O(1) direct return

2. sp_meta_find_or_create Fast Path (hakmem_shared_pool.c)
   - Check cached ss->shared_meta first before linear scan
   - Cache pointer after successful linear scan for future lookups
   - Reduces 7.8% CPU hotspot to near-zero for hot paths

3. Stage 2 Class Hints Fast Path (hakmem_shared_pool_acquire.c)
   - Try class_hints[class_idx] FIRST before full metadata scan
   - Uses O(1) ss->shared_meta lookup for hint validation
   - __builtin_expect() for branch prediction optimization
   - 80-90% of acquire calls now skip full metadata scan

4. Proper Initialization (ss_allocation_box.c)
   - Initialize shared_meta = NULL in superslab_allocate()
   - Ensures correct NULL-check semantics for new SuperSlabs

Additional Improvements:
- Updated ptr_trace and debug ring for release build efficiency
- Enhanced ENV variable documentation and analysis
- Added learner_env_box.h for configuration management
- Various Box optimizations for reduced overhead

Thread Safety:
- All atomic operations use correct memory ordering
- shared_meta cached under mutex protection
- Lock-free Stage 2 uses proper CAS with acquire/release semantics

Testing:
- Benchmark: 1M iterations, 3.8M ops/s stable
- Build: Clean compile RELEASE=0 and RELEASE=1
- No crashes, memory leaks, or correctness issues

Next Optimization Candidates:
- P1: Per-SuperSlab free slot bitmap for O(1) slot claiming
- P2: Reduce Stage 2 critical section size
- P3: Page pre-faulting (MAP_POPULATE)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 16:21:54 +09:00
+								                        // Update SuperSlab metadata under mutex
 								                        ss->slab_bitmap |= (1u << claimed_idx);
 								                        ss_slab_meta_class_idx_set(ss, claimed_idx, (uint8_t)class_idx);
 								                        if (ss->active_slabs == 0) {
 								                            ss->active_slabs = 1;
 								                            g_shared_pool.active_count++;
 								                        }
 								                        if (class_idx < TINY_NUM_CLASSES_SS) {
 								                            g_shared_pool.class_active_slots[class_idx]++;
 								                        }
 								                        // Hint is still good, no need to update
 								                        *ss_out = ss;
 								                        *slab_idx_out = claimed_idx;
 								                        sp_fix_geometry_if_needed(ss, claimed_idx, class_idx);
 								                        if (g_lock_stats_enabled == 1) {
 								                            atomic_fetch_add(&g_lock_release_count, 1);
 								                        }
 								                        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								                        if (g_sp_stage_stats_enabled) {
 								                            atomic_fetch_add(&g_sp_stage2_hits[class_idx], 1);
 								                        }
 								                        return 0;  // ✅ Stage 2 (hint fast path) success
 								                    }
 								                }
 								            }
 								        }
 								    }
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
+								stage2_scan:
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								    // P0-5: Lock-free atomic CAS claiming (no mutex needed for slot state transition!)
 								    // RACE FIX: Read ss_meta_count atomically (now properly declared as _Atomic)
 								    // No cast needed! memory_order_acquire synchronizes with release in sp_meta_find_or_create
 								    uint32_t meta_count = atomic_load_explicit(
 								        &g_shared_pool.ss_meta_count,
 								        memory_order_acquire
 								    );
 								    for (uint32_t i = 0; i < meta_count; i++) {
 								        SharedSSMeta* meta = &g_shared_pool.ss_metadata[i];
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
+								        // RACE FIX: Load SuperSlab pointer atomically BEFORE claiming
 								        // Use memory_order_acquire to synchronize with release in sp_meta_find_or_create
 								        SuperSlab* ss_preflight = atomic_load_explicit(&meta->ss, memory_order_acquire);
 								        if (!ss_preflight) {
 								            // SuperSlab was freed - skip this entry
 								            continue;
 								        }
 								        // P-Tier: Skip DRAINING tier SuperSlabs
 								        if (!ss_tier_is_hot(ss_preflight)) {
 								            continue;
 								        }
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								        // Try lock-free claiming (UNUSED → ACTIVE via CAS)
 								        int claimed_idx = sp_slot_claim_lockfree(meta, class_idx);
 								        if (claimed_idx >= 0) {
-												P-Tier + Tiny Route Policy: Aggressive Superslab Management + Safe Routing

## Phase 1: Utilization-Aware Superslab Tiering (案B実装済)

- Add ss_tier_box.h: Classify SuperSlabs into HOT/DRAINING/FREE based on utilization
  - HOT (>25%): Accept new allocations
  - DRAINING (≤25%): Drain only, no new allocs
  - FREE (0%): Ready for eager munmap

- Enhanced shared_pool_release_slab():
  - Check tier transition after each slab release
  - If tier→FREE: Force remaining slots to EMPTY and call superslab_free() immediately
  - Bypasses LRU cache to prevent registry bloat from accumulating DRAINING SuperSlabs

- Test results (bench_random_mixed_hakmem):
  - 1M iterations: ✅ ~1.03M ops/s (previously passed)
  - 10M iterations: ✅ ~1.15M ops/s (previously: registry full error)
  - 50M iterations: ✅ ~1.08M ops/s (stress test)

## Phase 2: Tiny Front Routing Policy (新規Box)

- Add tiny_route_box.h/c: Single 8-byte table for class→routing decisions
  - ROUTE_TINY_ONLY: Tiny front exclusive (no fallback)
  - ROUTE_TINY_FIRST: Try Tiny, fallback to Pool if fails
  - ROUTE_POOL_ONLY: Skip Tiny entirely

- Profiles via HAKMEM_TINY_PROFILE ENV:
  - "hot": C0-C3=TINY_ONLY, C4-C6=TINY_FIRST, C7=POOL_ONLY
  - "conservative" (default): All TINY_FIRST
  - "off": All POOL_ONLY (disable Tiny)
  - "full": All TINY_ONLY (microbench mode)

- A/B test results (ws=256, 100k ops random_mixed):
  - Default (conservative): ~2.90M ops/s
  - hot: ~2.65M ops/s (more conservative)
  - off: ~2.86M ops/s
  - full: ~2.98M ops/s (slightly best)

## Design Rationale

### Registry Pressure Fix (案B)
- Problem: DRAINING tier SS occupied registry indefinitely
- Solution: When total_active_blocks→0, immediately free to clear registry slot
- Result: No more "registry full" errors under stress

### Routing Policy Box (新)
- Problem: Tiny front optimization scattered across ENV/branches
- Solution: Centralize routing in single table, select profiles via ENV
- Benefit: Safe A/B testing without touching hot path code
- Future: Integrate with RSS budget/learning layers for dynamic profile switching

## Next Steps (性能最適化)
- Profile Tiny front internals (TLS SLL, FastCache, Superslab backend latency)
- Identify bottleneck between current ~2.9M ops/s and mimalloc ~100M ops/s
- Consider:
  - Reduce shared pool lock contention
  - Optimize unified cache hit rate
  - Streamline Superslab carving logic

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:01:25 +09:00
+								            // RACE FIX: Load SuperSlab pointer atomically again after claiming
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								            // Use memory_order_acquire to synchronize with release in sp_meta_find_or_create
 								            SuperSlab* ss = atomic_load_explicit(&meta->ss, memory_order_acquire);
 								            if (!ss) {
 								                // SuperSlab was freed between claiming and loading - skip this entry
 								                continue;
 								            }
 								            #if !HAKMEM_BUILD_RELEASE
 								            if (dbg_acquire == 1) {
 								                fprintf(stderr, "[SP_ACQUIRE_STAGE2_LOCKFREE] class=%d claimed UNUSED slot (ss=%p slab=%d)\n",
 								                        class_idx, (void*)ss, claimed_idx);
 								            }
 								            #endif
 								            // P0 instrumentation: count lock acquisitions
 								            lock_stats_init();
 								            if (g_lock_stats_enabled == 1) {
 								                atomic_fetch_add(&g_lock_acquire_count, 1);
 								                atomic_fetch_add(&g_lock_acquire_slab_count, 1);
 								            }
 								            pthread_mutex_lock(&g_shared_pool.alloc_lock);
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								            // Performance measurement: count Stage 2 scan lock acquisitions
 								            if (__builtin_expect(sp_measure_enabled(), 0)) {
-												Add Page Box layer for C7 class optimization

- Implement tiny_page_box.c/h: per-thread page cache between UC and Shared Pool
- Integrate Page Box into Unified Cache refill path
- Remove legacy SuperSlab implementation (merged into smallmid)
- Add HAKMEM_TINY_PAGE_BOX_CLASSES env var for selective class enabling
- Update bench_random_mixed.c with Page Box statistics

Current status: Implementation safe, no regressions.
Page Box ON/OFF shows minimal difference - pool strategy needs tuning.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 15:31:44 +09:00
+								                atomic_fetch_add_explicit(&g_sp_stage2_lock_acquired_global,
 , memory_order_relaxed);
 								                atomic_fetch_add_explicit(&g_sp_alloc_lock_contention_global,
 , memory_order_relaxed);
 								                atomic_fetch_add_explicit(
 								                    &g_sp_stage2_lock_acquired_by_class[class_idx],
 , memory_order_relaxed);
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								            }
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								            // Update SuperSlab metadata under mutex
 								            ss->slab_bitmap |= (1u << claimed_idx);
 								            ss_slab_meta_class_idx_set(ss, claimed_idx, (uint8_t)class_idx);
 								            if (ss->active_slabs == 0) {
 								                ss->active_slabs = 1;
 								                g_shared_pool.active_count++;
 								            }
 								            if (class_idx < TINY_NUM_CLASSES_SS) {
 								                g_shared_pool.class_active_slots[class_idx]++;
 								            }
 								            // Update hint
 								            g_shared_pool.class_hints[class_idx] = ss;
 								            *ss_out = ss;
 								            *slab_idx_out = claimed_idx;
 								            sp_fix_geometry_if_needed(ss, claimed_idx, class_idx);
 								            if (g_lock_stats_enabled == 1) {
 								                atomic_fetch_add(&g_lock_release_count, 1);
 								            }
 								            pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 									            if (g_sp_stage_stats_enabled) {
 									                atomic_fetch_add(&g_sp_stage2_hits[class_idx], 1);
 									            }
 								            return 0;  // ✅ Stage 2 (lock-free) success
 								        }
 								        // Claim failed (no UNUSED slots in this meta) - continue to next SuperSlab
 								    }
 								    // ========== Tension-Based Drain: Try to create EMPTY slots before Stage 3 ==========
 								    // If TLS SLL has accumulated blocks, drain them to enable EMPTY slot detection
 								    // This can avoid allocating new SuperSlabs by reusing EMPTY slots in Stage 1
 								    // ENV: HAKMEM_TINY_TENSION_DRAIN_ENABLE=0 to disable (default=1)
 								    // ENV: HAKMEM_TINY_TENSION_DRAIN_THRESHOLD=N to set threshold (default=1024)
 								    {
-												Priority-2 ENV Cache: Shared Pool Acquire (5変数追加、5箇所置換)

【追加ENV変数】
- HAKMEM_SS_EMPTY_REUSE (default: 1)
- HAKMEM_SS_EMPTY_SCAN_LIMIT (default: 32)
- HAKMEM_SS_ACQUIRE_DEBUG (default: 0)
- HAKMEM_TINY_TENSION_DRAIN_ENABLE (default: 1)
- HAKMEM_TINY_TENSION_DRAIN_THRESHOLD (default: 1024)

【置換ファイル】
- core/hakmem_shared_pool_acquire.c (5箇所 → ENV Cache)

【変更詳細】
1. ENV Cache (hakmem_env_cache.h):
   - 構造体に5変数追加 (41→46変数)
   - hakmem_env_cache_init()に初期化追加
   - アクセサマクロ5個追加
   - カウント更新: 41→46

2. hakmem_shared_pool_acquire.c:
   - getenv("HAKMEM_SS_EMPTY_REUSE") → HAK_ENV_SS_EMPTY_REUSE()
   - getenv("HAKMEM_SS_EMPTY_SCAN_LIMIT") → HAK_ENV_SS_EMPTY_SCAN_LIMIT()
   - getenv("HAKMEM_SS_ACQUIRE_DEBUG") → HAK_ENV_SS_ACQUIRE_DEBUG()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_ENABLE") → HAK_ENV_TINY_TENSION_DRAIN_ENABLE()
   - getenv("HAKMEM_TINY_TENSION_DRAIN_THRESHOLD") → HAK_ENV_TINY_TENSION_DRAIN_THRESHOLD()
   - #include "hakmem_env_cache.h" 追加

【効果】
- Shared Pool Acquire warm pathからgetenv()呼び出しを完全排除
- Lock-free Stage2のgetenv()オーバーヘッド削減

【テスト】
✅ make shared → 成功
✅ /tmp/test_mixed3_final → PASSED

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-02 20:51:50 +09:00
+								        // Priority-2: Use cached ENV
 								        int tension_drain_enabled = HAK_ENV_TINY_TENSION_DRAIN_ENABLE();
 								        uint32_t tension_threshold = (uint32_t)HAK_ENV_TINY_TENSION_DRAIN_THRESHOLD();
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
 								        if (tension_drain_enabled) {
 								            extern __thread TinyTLSSLL g_tls_sll[TINY_NUM_CLASSES];
 								            extern uint32_t tiny_tls_sll_drain(int class_idx, uint32_t batch_size);
 								            uint32_t sll_count = (class_idx < TINY_NUM_CLASSES) ? g_tls_sll[class_idx].count : 0;
 								            if (sll_count >= tension_threshold) {
 								                // Drain all blocks to maximize EMPTY slot creation
 								                uint32_t drained = tiny_tls_sll_drain(class_idx, 0);  // 0 = drain all
 								                if (drained > 0) {
 								                    // Retry Stage 1 (EMPTY reuse) after drain
 								                    // Some slabs might have become EMPTY (meta->used == 0)
 								                    goto stage1_retry_after_tension_drain;
 								                }
 								            }
 								        }
 								    }
 								    // ========== Stage 3: Mutex-protected fallback (new SuperSlab allocation) ==========
 								    // All existing SuperSlabs have no UNUSED slots → need new SuperSlab
 								    // P0 instrumentation: count lock acquisitions
 								    lock_stats_init();
 								    if (g_lock_stats_enabled == 1) {
 								        atomic_fetch_add(&g_lock_acquire_count, 1);
 								        atomic_fetch_add(&g_lock_acquire_slab_count, 1);
 								    }
 								    pthread_mutex_lock(&g_shared_pool.alloc_lock);
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								    // Performance measurement: count Stage 3 lock acquisitions
 								    if (__builtin_expect(sp_measure_enabled(), 0)) {
-												Add Page Box layer for C7 class optimization

- Implement tiny_page_box.c/h: per-thread page cache between UC and Shared Pool
- Integrate Page Box into Unified Cache refill path
- Remove legacy SuperSlab implementation (merged into smallmid)
- Add HAKMEM_TINY_PAGE_BOX_CLASSES env var for selective class enabling
- Update bench_random_mixed.c with Page Box statistics

Current status: Implementation safe, no regressions.
Page Box ON/OFF shows minimal difference - pool strategy needs tuning.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 15:31:44 +09:00
+								        atomic_fetch_add_explicit(&g_sp_stage3_lock_acquired_global,
 , memory_order_relaxed);
 								        atomic_fetch_add_explicit(&g_sp_alloc_lock_contention_global,
 , memory_order_relaxed);
 								        atomic_fetch_add_explicit(&g_sp_stage3_lock_acquired_by_class[class_idx],
 , memory_order_relaxed);
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								    }
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
+								    // ========== Stage 3: Get new SuperSlab ==========
 								    // Try LRU cache first, then mmap
 								    SuperSlab* new_ss = NULL;
 								    // Stage 3a: Try LRU cache
 								    extern SuperSlab* hak_ss_lru_pop(uint8_t size_class);
 								    new_ss = hak_ss_lru_pop((uint8_t)class_idx);
 								    int from_lru = (new_ss != NULL);
 								    // Stage 3b: If LRU miss, allocate new SuperSlab
 								    if (!new_ss) {
 								        // Release the alloc_lock to avoid deadlock with registry during superslab_allocate
 								        if (g_lock_stats_enabled == 1) {
 								            atomic_fetch_add(&g_lock_release_count, 1);
 								        }
 								        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
-												WIP: Add TLS SLL validation and SuperSlab registry fallback

ChatGPT's diagnostic changes to address TLS_SLL_HDR_RESET issue.
Current status: Partial mitigation, but root cause remains.

Changes Applied:
1. SuperSlab Registry Fallback (hakmem_super_registry.h)
   - Added legacy table probe when hash map lookup misses
   - Prevents NULL returns for valid SuperSlabs during initialization
   - Status: ✅ Works but may hide underlying registration issues

2. TLS SLL Push Validation (tls_sll_box.h)
   - Reject push if SuperSlab lookup returns NULL
   - Reject push if class_idx mismatch detected
   - Added [TLS_SLL_PUSH_NO_SS] diagnostic message
   - Status: ✅ Prevents list corruption (defensive)

3. SuperSlab Allocation Class Fix (superslab_allocate.c)
   - Pass actual class_idx to sp_internal_allocate_superslab
   - Prevents dummy class=8 causing OOB access
   - Status: ✅ Root cause fix for allocation path

4. Debug Output Additions
   - First 256 push/pop operations traced
   - First 4 mismatches logged with details
   - SuperSlab registration state logged
   - Status: ✅ Diagnostic tool (not a fix)

5. TLS Hint Box Removed
   - Deleted ss_tls_hint_box.{c,h} (Phase 1 optimization)
   - Simplified to focus on stability first
   - Status: ⏳ Can be re-added after root cause fixed

Current Problem (REMAINS UNSOLVED):
- [TLS_SLL_HDR_RESET] still occurs after ~60 seconds of sh8bench
- Pointer is 16 bytes offset from expected (class 1 → class 2 boundary)
- hak_super_lookup returns NULL for that pointer
- Suggests: Use-After-Free, Double-Free, or pointer arithmetic error

Root Cause Analysis:
- Pattern: Pointer offset by +16 (one class 1 stride)
- Timing: Cumulative problem (appears after 60s, not immediately)
- Location: Header corruption detected during TLS SLL pop

Remaining Issues:
⚠️ Registry fallback is defensive (may hide registration bugs)
⚠️ Push validation prevents symptoms but not root cause
⚠️ 16-byte pointer offset source unidentified

Next Steps for Investigation:
1. Full pointer arithmetic audit (Magazine ⇔ TLS SLL paths)
2. Enhanced logging at HDR_RESET point:
   - Expected vs actual pointer value
   - Pointer provenance (where it came from)
   - Allocation trace for that block
3. Verify Headerless flag is OFF throughout build
4. Check for double-offset application in conversions

Technical Assessment:
- 60% root cause fixes (allocation class, validation)
- 40% defensive mitigation (registry fallback, push rejection)

Performance Impact:
- Registry fallback: +10-30 cycles on cold path (negligible)
- Push validation: +5-10 cycles per push (acceptable)
- Overall: < 2% performance impact estimated

Related Issues:
- Phase 1 TLS Hint Box removed temporarily
- Phase 2 Headerless blocked until stability achieved

🤖 Generated with Claude Code (https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-03 20:42:28 +09:00
+								        SuperSlab* allocated_ss = sp_internal_allocate_superslab(class_idx);
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
 								        // Re-acquire the alloc_lock
 								        if (g_lock_stats_enabled == 1) {
 								            atomic_fetch_add(&g_lock_acquire_count, 1);
 								            atomic_fetch_add(&g_lock_acquire_slab_count, 1); // This is part of acquisition path
 								        }
 								        pthread_mutex_lock(&g_shared_pool.alloc_lock);
 								        if (!allocated_ss) {
 								            // Allocation failed; return now.
 								            if (g_lock_stats_enabled == 1) {
 								                atomic_fetch_add(&g_lock_release_count, 1);
 								            }
 								            pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								            return -1; // Out of memory
 								        }
 								        new_ss = allocated_ss;
 								        // Add newly allocated SuperSlab to the shared pool's internal array
 								        if (g_shared_pool.total_count >= g_shared_pool.capacity) {
 								            shared_pool_ensure_capacity_unlocked(g_shared_pool.total_count + 1);
 								            if (g_shared_pool.total_count >= g_shared_pool.capacity) {
 								                // Pool table expansion failed; leave ss alive (registry-owned),
 								                // but do not treat it as part of shared_pool.
 								                // This is a critical error, return early.
 								                if (g_lock_stats_enabled == 1) {
 								                    atomic_fetch_add(&g_lock_release_count, 1);
 								                }
 								                pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								                return -1;
 								            }
 								        }
 								        g_shared_pool.slabs[g_shared_pool.total_count] = new_ss;
 								        g_shared_pool.total_count++;
 								    }
 								    #if !HAKMEM_BUILD_RELEASE
 								    if (dbg_acquire == 1 && new_ss) {
 								        fprintf(stderr, "[SP_ACQUIRE_STAGE3] class=%d new SuperSlab (ss=%p from_lru=%d)\n",
 								                class_idx, (void*)new_ss, from_lru);
 								    }
 								    #endif
 								    if (!new_ss) {
 								        if (g_lock_stats_enabled == 1) {
 								            atomic_fetch_add(&g_lock_release_count, 1);
 								        }
 								        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								        return -1;  // ❌ Out of memory
 								    }
 								    // Before creating a new SuperSlab, consult learning-layer soft cap.
-												Phase 9-2: Remove Legacy Backend & Unify to Shared Pool (50M ops/s)

- Removed Legacy Backend fallback; Shared Pool is now the sole backend.
- Removed Soft Cap limit in Shared Pool to allow full memory management.
- Implemented EMPTY slab recycling with batched meta->used decrement in remote drain.
- Updated tiny_free_local_box to return is_empty status for safe recycling.
- Fixed race condition in release path by removing from legacy list early.
- Achieved 50.3M ops/s in WS8192 benchmark (+200% vs baseline).

											
										
										
											2025-12-01 13:47:23 +09:00
+								    // Phase 9-2: Soft Cap removed to allow Shared Pool to fully replace Legacy Backend.
 								    // We now rely on LRU eviction and EMPTY recycling to manage memory pressure.
-												Refactor: Split monolithic hakmem_shared_pool.c into acquire/release modules

- Split core/hakmem_shared_pool.c into acquire/release modules for maintainability.
- Introduced core/hakmem_shared_pool_internal.h for shared internal API.
- Fixed incorrect function name usage (superslab_alloc -> superslab_allocate).
- Increased SUPER_REG_SIZE to 1M to support large working sets (Phase 9-2 fix).
- Updated Makefile.
- Verified with benchmarks.

											
										
										
											2025-11-30 18:11:08 +09:00
 								    // Create metadata for this new SuperSlab
 								    SharedSSMeta* new_meta = sp_meta_find_or_create(new_ss);
 								    if (!new_meta) {
 								        if (g_lock_stats_enabled == 1) {
 								            atomic_fetch_add(&g_lock_release_count, 1);
 								        }
 								        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								        return -1;  // ❌ Metadata allocation failed
 								    }
 								    // Assign first slot to this class
 								    int first_slot = 0;
 								    if (sp_slot_mark_active(new_meta, first_slot, class_idx) != 0) {
 								        if (g_lock_stats_enabled == 1) {
 								            atomic_fetch_add(&g_lock_release_count, 1);
 								        }
 								        pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 								        return -1;  // ❌ Should not happen
 								    }
 								    // Update SuperSlab metadata
 								    new_ss->slab_bitmap |= (1u << first_slot);
 								    ss_slab_meta_class_idx_set(new_ss, first_slot, (uint8_t)class_idx);
 								    new_ss->active_slabs = 1;
 								    g_shared_pool.active_count++;
 								    if (class_idx < TINY_NUM_CLASSES_SS) {
 								        g_shared_pool.class_active_slots[class_idx]++;
 								    }
 								    // Update hint
 								    g_shared_pool.class_hints[class_idx] = new_ss;
 								    *ss_out = new_ss;
 								    *slab_idx_out = first_slot;
 								    sp_fix_geometry_if_needed(new_ss, first_slot, class_idx);
 								    if (g_lock_stats_enabled == 1) {
 								        atomic_fetch_add(&g_lock_release_count, 1);
 								    }
 								    pthread_mutex_unlock(&g_shared_pool.alloc_lock);
 									    if (g_sp_stage_stats_enabled) {
 									        atomic_fetch_add(&g_sp_stage3_hits[class_idx], 1);
 									    }
 								    return 0;  // ✅ Stage 3 success
 								}
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
 								// ============================================================================
 								// Performance Measurement: Print Statistics
 								// ============================================================================
 								void shared_pool_print_measurements(void) {
 								    if (!sp_measure_enabled()) {
 								        return;  // Measurement disabled
 								    }
-												Add Page Box layer for C7 class optimization

- Implement tiny_page_box.c/h: per-thread page cache between UC and Shared Pool
- Integrate Page Box into Unified Cache refill path
- Remove legacy SuperSlab implementation (merged into smallmid)
- Add HAKMEM_TINY_PAGE_BOX_CLASSES env var for selective class enabling
- Update bench_random_mixed.c with Page Box statistics

Current status: Implementation safe, no regressions.
Page Box ON/OFF shows minimal difference - pool strategy needs tuning.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 15:31:44 +09:00
+								    uint64_t stage2 = atomic_load_explicit(&g_sp_stage2_lock_acquired_global,
 								                                           memory_order_relaxed);
 								    uint64_t stage3 = atomic_load_explicit(&g_sp_stage3_lock_acquired_global,
 								                                           memory_order_relaxed);
 								    uint64_t total_locks = atomic_load_explicit(&g_sp_alloc_lock_contention_global,
 								                                                memory_order_relaxed);
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
 								    if (total_locks == 0) {
 								        fprintf(stderr, "\n========================================\n");
 								        fprintf(stderr, "Shared Pool Contention Statistics\n");
 								        fprintf(stderr, "========================================\n");
 								        fprintf(stderr, "No lock acquisitions recorded\n");
 								        fprintf(stderr, "========================================\n\n");
 								        return;
 								    }
 								    double stage2_pct = (100.0 * stage2) / total_locks;
 								    double stage3_pct = (100.0 * stage3) / total_locks;
 								    fprintf(stderr, "\n========================================\n");
 								    fprintf(stderr, "Shared Pool Contention Statistics\n");
 								    fprintf(stderr, "========================================\n");
 								    fprintf(stderr, "Stage 2 Locks:    %llu (%.1f%%)\n",
 								            (unsigned long long)stage2, stage2_pct);
 								    fprintf(stderr, "Stage 3 Locks:    %llu (%.1f%%)\n",
 								            (unsigned long long)stage3, stage3_pct);
 								    fprintf(stderr, "Total Contention: %llu lock acquisitions\n",
 								            (unsigned long long)total_locks);
-												Add Page Box layer for C7 class optimization

- Implement tiny_page_box.c/h: per-thread page cache between UC and Shared Pool
- Integrate Page Box into Unified Cache refill path
- Remove legacy SuperSlab implementation (merged into smallmid)
- Add HAKMEM_TINY_PAGE_BOX_CLASSES env var for selective class enabling
- Update bench_random_mixed.c with Page Box statistics

Current status: Implementation safe, no regressions.
Page Box ON/OFF shows minimal difference - pool strategy needs tuning.

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-05 15:31:44 +09:00
 								    // Per-class breakdown（Tiny 用クラス 0-7、特に C5–C7 を観測）
 								    fprintf(stderr, "\nPer-class Shared Pool Locks (Stage2/Stage3):\n");
 								    for (int cls = 0; cls < TINY_NUM_CLASSES_SS; cls++) {
 								        uint64_t s2c = atomic_load_explicit(
 								            &g_sp_stage2_lock_acquired_by_class[cls],
 								            memory_order_relaxed);
 								        uint64_t s3c = atomic_load_explicit(
 								            &g_sp_stage3_lock_acquired_by_class[cls],
 								            memory_order_relaxed);
 								        uint64_t tc = s2c + s3c;
 								        if (tc == 0) {
 								            continue;  // ロック取得のないクラスは省略
 								        }
 								        fprintf(stderr,
 								                "  C%d: Stage2=%llu Stage3=%llu Total=%llu\n",
 								                cls,
 								                (unsigned long long)s2c,
 								                (unsigned long long)s3c,
 								                (unsigned long long)tc);
 								    }
-												Performance Measurement Framework: Unified Cache, TLS SLL, Shared Pool Analysis

## Summary

Implemented production-grade measurement infrastructure to quantify top 3 bottlenecks:
- Unified cache hit/miss rates + refill cost
- TLS SLL usage patterns
- Shared pool lock contention distribution

## Changes

### 1. Unified Cache Metrics (tiny_unified_cache.h/c)
- Added atomic counters:
  - g_unified_cache_hits_global: successful cache pops
  - g_unified_cache_misses_global: refill triggers
  - g_unified_cache_refill_cycles_global: refill cost in CPU cycles (rdtsc)
- Instrumented `unified_cache_pop_or_refill()` to count hits
- Instrumented `unified_cache_refill()` with cycle measurement
- ENV-gated: HAKMEM_MEASURE_UNIFIED_CACHE=1 (default: off)
- Added unified_cache_print_measurements() output function

### 2. TLS SLL Metrics (tls_sll_box.h)
- Added atomic counters:
  - g_tls_sll_push_count_global: total pushes
  - g_tls_sll_pop_count_global: successful pops
  - g_tls_sll_pop_empty_count_global: empty list conditions
- Instrumented push/pop paths
- Added tls_sll_print_measurements() output function

### 3. Shared Pool Contention (hakmem_shared_pool_acquire.c)
- Added atomic counters:
  - g_sp_stage2_lock_acquired_global: Stage 2 locks
  - g_sp_stage3_lock_acquired_global: Stage 3 allocations
  - g_sp_alloc_lock_contention_global: total lock acquisitions
- Instrumented all pthread_mutex_lock calls in hot paths
- Added shared_pool_print_measurements() output function

### 4. Benchmark Integration (bench_random_mixed.c)
- Called all 3 print functions after benchmark loop
- Functions active only when HAKMEM_MEASURE_UNIFIED_CACHE=1 set

## Design Principles

- **Zero overhead when disabled**: Inline checks with __builtin_expect hints
- **Atomic relaxed memory order**: Minimal synchronization overhead
- **ENV-gated**: Single flag controls all measurements
- **Production-safe**: Compiles in release builds, no functional changes

## Usage

```bash
HAKMEM_MEASURE_UNIFIED_CACHE=1 ./bench_allocators_hakmem bench_random_mixed_hakmem 1000000 256 42
```

Output (when enabled):
```
========================================
Unified Cache Statistics
========================================
Hits:        1234567
Misses:      56789
Hit Rate:    95.6%
Avg Refill Cycles: 1234

========================================
TLS SLL Statistics
========================================
Total Pushes:     1234567
Total Pops:       345678
Pop Empty Count:  12345
Hit Rate:         98.8%

========================================
Shared Pool Contention Statistics
========================================
Stage 2 Locks:    123456 (33%)
Stage 3 Locks:    234567 (67%)
Total Contention: 357 locks per 1M ops
```

## Next Steps

1. **Enable measurements** and run benchmarks to gather data
2. **Analyze miss rates**: Which bottleneck dominates?
3. **Profile hottest stage**: Focus optimization on top contributor
4. Possible targets:
   - Increase unified cache capacity if miss rate >5%
   - Profile if TLS SLL is unused (potential legacy code removal)
   - Analyze if Stage 2 lock can be replaced with CAS

## Makefile Updates

Added core/box/tiny_route_box.o to:
- OBJS_BASE (test build)
- SHARED_OBJS (shared library)
- BENCH_HAKMEM_OBJS_BASE (benchmark)
- TINY_BENCH_OBJS_BASE (tiny benchmark)

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude <noreply@anthropic.com>

											
										
										
											2025-12-04 18:26:39 +09:00
+								    fprintf(stderr, "========================================\n\n");
 								}