numa-programming
NUMA programming skill for multi-socket memory locality. Use when detecting NUMA topology,…
Custom allocator skill for memory allocation strategies. Use when implementing pool/slab/arena allocators, tuning jemalloc/mimalloc, writing Rust GlobalAlloc, or benchmarking allocator performance. Activates on queries about jemalloc, mimalloc, tcmalloc, arena allocator,
$ npx -y skills add mohitmishra786/low-level-dev-skills --skill custom-allocators --agent claude-codeHow it fires
How this skill gets triggered: by you, by Claude, or both.
/custom-allocatorsContext preview
The summary Claude sees to decide when to auto-load this skill.
Custom allocator skill for memory allocation strategies. Use when implementing pool/slab/arena allocators, tuning jemalloc/mimalloc, writing Rust GlobalAlloc, or benchmarking allocator performance. Activates on queries about jemalloc, mimalloc, tcmalloc, arena allocator,
name: custom-allocators description: Custom allocator skill for memory allocation strategies. Use when implementing pool/slab/arena allocators, tuning jemalloc/mimalloc, writing Rust GlobalAlloc, or benchmarking allocator performance. Activates on queries about jemalloc, mimalloc, tcmalloc, arena allocator, GlobalAlloc, or fragmentation.
Guide agents through memory allocator design and tuning: pool/slab/arena/buddy taxonomy, jemalloc internals and `MALLOC_CONF`, mimalloc design, tcmalloc thread-caching, writing a simple pool allocator in C, Rust `GlobalAlloc` trait, fragmentation metrics, and benchmarking.
Allocator types ├── Pool/fixed-size — O(1) alloc/free, fixed block sizes ├── Slab — kernel-style, cache-friendly size classes ├── Arena/bump — fast alloc, bulk free (reset arena) ├── Buddy — power-of-two blocks, low fragmentation for large allocs └── General (malloc) — jemalloc, mimalloc, tcmalloc
#include <stddef.h>
#include <stdint.h>
#include <stdlib.h>
typedef struct pool_block {
struct pool_block *next;
} pool_block_t;
typedef struct {
void *memory;
size_t block_size;
size_t num_blocks;
pool_block_t *free_list;
} pool_t;
int pool_init(pool_t *p, size_t block_size, size_t num_blocks) {
p->block_size = block_size < sizeof(pool_block_t) ?
sizeof(pool_block_t) : block_size;
p->num_blocks = num_blocks;
p->memory = aligned_alloc(64, p->block_size * num_blocks);
if (!p->memory) return -1;
p->free_list = NULL;
for (size_t i = 0; i < num_blocks; i++) {
pool_block_t *blk = (pool_block_t *)((char *)p->memory + i * p->block_size);
blk->next = p->free_list;
p->free_list = blk;
}
return 0;
}
void *pool_alloc(pool_t *p) {
if (!p->free_list) return NULL;
pool_block_t *blk = p->free_list;
p->free_list = blk->next;
return blk;
}
void pool_free(pool_t *p, void *ptr) {
pool_block_t *blk = (pool_block_t *)ptr;
blk->next = p->free_list;
p->free_list = blk;
}typedef struct {
char *base;
size_t capacity;
size_t offset;
} arena_t;
void *arena_alloc(arena_t *a, size_t size, size_t align) {
uintptr_t cur = (uintptr_t)(a->base + a->offset);
uintptr_t aligned = (cur + align - 1) & ~(align - 1);
size_t padding = aligned - cur;
if (a->offset + padding + size > a->capacity)
return NULL;
a->offset += padding + size;
return (void *)aligned;
}
void arena_reset(arena_t *a) { a->offset = 0; }Use per-request arenas in servers — reset after response, no per-object free.
# Use jemalloc as system allocator LD_PRELOAD=/usr/lib/x86_64-linux-gnu/libjemalloc.so.2 ./myapp # Runtime tuning export MALLOC_CONF="background_thread:true,dirty_decay_ms:1000,muzzy_decay_ms:1000" # Profiling build # ./configure --enable-prof export MALLOC_CONF="prof:true,prof_active:true,lg_prof_sample:19"
jemalloc concepts:
# Statistics
export MALLOC_CONF="stats_print:true"
# prints on exit, or:
mallctl("epoch", ...); mallctl("stats.allocated", ...);# Use mimalloc LD_PRELOAD=/usr/lib/libmimalloc.so ./myapp # Environment options export MIMALLOC_SHOW_STATS=1 export MIMALLOC_PAGE_RESET=1
mimalloc hierarchy: segment → page → block. Generally lower metadata overhead than jemalloc.
# Statistics at exit MIMALLOC_SHOW_STATS=1 ./myapp
LD_PRELOAD=/usr/lib/x86_64-linux-gnu/libtcmalloc.so.4 ./myapp # Environment export TCMALLOC_SAMPLE_PARAMETER=524288 # sampling profiler
Per-thread caches for small objects; central heap for larger allocations.
use std::alloc::{GlobalAlloc, Layout, System};
use std::sync::atomic::{AtomicUsize, Ordering};
struct TrackingAllocator;
static ALLOCATED: AtomicUsize = AtomicUsize::new(0);
unsafe impl GlobalAlloc for TrackingAllocator {
unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
let ptr = System.alloc(layout);
if !ptr.is_null() {
ALLOCATED.fetch_add(layout.size(), Ordering::Relaxed);
}
ptr
}
unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) {
System.dealloc(ptr, layout);
ALLOCATED.fetch_sub(layout.size(), Ordering::Relaxed);
}
}
#[global_allocator]
static GLOBAL: TrackingAllocator = TrackingAllocator;For `#![no_std]` with `alloc` crate, implement `GlobalAlloc` without `System`.
| Type | Definition | Detection | |------|------------|-----------| | Internal | Allocated block larger than requested | Allocator stats; size class rounding | | External | Free memory not usable for request | `mallinfo`/`malloc_info`; RSS vs heap |
# glibc malloc_info malloc_info(0, stdout); # jemalloc jeprof --show_bytes ./myapp jeprof.heap
# mimalloc bench (if built from source) cd mimalloc && mkdir build && cd build cmake .. && make ./mimalloc-bench # Custom microbenchmark pattern perf stat -e cycles,instructions ./allocator_bench --threads 8 --size 64
Compare: single-thread alloc/free, multi-thread contention, mixed size distribution.
| Symptom | Cause | Fix | |---------|-------|-----| | Pool alloc returns NULL | Free list exhausted | Incr
A curated suite of AI agent skills for systems and low-level programming — C/C++, Rust, Zig, GPU, bare-metal firmware, Linux kernel/driver development, computer architecture, compiler internals, HPC, and more.
Repo: mohitmishra786/low-level-dev-skills
NUMA programming skill for multi-socket memory locality. Use when detecting NUMA topology,…
AF_XDP skill for high-performance XDP sockets. Use when creating AF_XDP sockets, configuring…
DPDK skill for userspace packet I/O. Use when initializing EAL, configuring PMD drivers,…
io_uring skill for Linux async I/O. Use when building high-performance servers with liburing,…
Bare-metal ADC and DAC skill. Use when configuring analog sampling, DMA-driven ADC,…
Bare-metal startup skill for reset-to-main bring-up. Use when writing startup code, vector…