Skip to main content
UltraInstinct
← All benchmarks

What Does a Page Fault Cost? First Touch vs Prefaulting 512 MiB

When a program writes to freshly mapped memory, how much does taking one page fault per 4 KiB page cost compared with asking the kernel to populate the whole range in one call before touching it?

Results

Pages131072
Repeats5
Buffer Mb512
Populate Minor Faults131072
First Touch Ns Per Page2699.9
Fault Vs Populate Factor1.51
First Touch Minor Faults131072
Retouch Mapped Ns Per Page22
Populate Then Touch Ns Per Page1786.7

Recorded October 7, 2026 at 2:21 PM UTC, wall clock 3.7s.

Method

A 512 MiB anonymous private mapping is created and marked MADV_NOHUGEPAGE so both cases work on 4 KiB pages. In the first case the program writes one byte to every page, taking one page fault per page. In the second case it first calls madvise(MADV_POPULATE_WRITE) on the whole range and then performs the same writes, which no longer fault. The timed region in both cases runs from just after the mapping is created until every page has been written, so both include the kernel allocating and zeroing 131,072 pages; the difference is the per-page CPU exception and return. A third pass re-writes the already-mapped buffer to show the cost of the loop with no faults at all. Minor-fault counts from getrusage() are reported for both cases and are identical, because MADV_POPULATE_WRITE runs the same fault handler for each page from inside one system call. The mapping is released after each run. Both cases are warmed up on a 64 MiB buffer, then measured five times alternating, and the fastest run of each is reported in nanoseconds per page.

Machine

CPUAMD EPYC 9354P 32-Core Processor
Cores visible8
Memory31.3 GB
Kernel6.8.0-139-generic
Architecturex64
Compilergcc (Ubuntu 13.3.0-6ubuntu2~24.04.1) 13.3.0

This is a shared virtual server, not an isolated test rig. Absolute throughput will differ on your hardware; the ratio between the two cases is the part that carries over.

Source

The complete program that produced the numbers above. Nothing else was running under our control during the measurement.

/*
 * Measures what demand paging costs: touching freshly mapped memory one 4 KiB
 * page at a time (one page fault per page) versus asking the kernel to
 * populate the whole range in a single call first.
 *
 * Both cases end in the same state — every page of the buffer allocated,
 * zeroed and written once — and both use 4 KiB pages (MADV_NOHUGEPAGE), so the
 * difference is the per-page trap into the kernel and back. A third pass
 * re-touches memory that is already mapped, which is the cost of the loop
 * itself with no faults at all. Minor-fault counts from getrusage() are
 * reported for both cases and come out identical: MADV_POPULATE_WRITE runs the
 * same fault handler for every page, it just does so in one loop inside the
 * kernel instead of through a CPU exception per page. That is exactly the cost
 * the comparison isolates.
 *
 * Build: gcc -O2 -o bench bench.c
 * Run:   ./bench <megabytes>
 * Output: one JSON object on stdout.
 */
#define _GNU_SOURCE
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <sys/mman.h>
#include <sys/resource.h>
#include <time.h>
#include <unistd.h>

#ifndef MADV_POPULATE_WRITE
#define MADV_POPULATE_WRITE 23
#endif

#define PAGE 4096UL

static double now_seconds(void) {
    struct timespec ts;
    clock_gettime(CLOCK_MONOTONIC, &ts);
    return (double)ts.tv_sec + (double)ts.tv_nsec / 1e9;
}

static long minor_faults(void) {
    struct rusage ru;
    getrusage(RUSAGE_SELF, &ru);
    return ru.ru_minflt;
}

static void die(const char *what) {
    perror(what);
    exit(1);
}

static void touch(volatile char *buf, size_t bytes) {
    for (size_t off = 0; off < bytes; off += PAGE) buf[off] = 1;
}

typedef struct {
    double ns_per_page;
    double retouch_ns_per_page;
    long faults;
} result_t;

static result_t run_case(size_t bytes, int populate) {
    char *buf = mmap(NULL, bytes, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
    if (buf == MAP_FAILED) die("mmap");
    if (madvise(buf, bytes, MADV_NOHUGEPAGE) != 0) die("madvise(NOHUGEPAGE)");

    size_t pages = bytes / PAGE;
    long faults_before = minor_faults();
    double start = now_seconds();
    if (populate && madvise(buf, bytes, MADV_POPULATE_WRITE) != 0) die("madvise(POPULATE_WRITE)");
    touch(buf, bytes);
    double elapsed = now_seconds() - start;
    long faults = minor_faults() - faults_before;

    start = now_seconds();
    touch(buf, bytes);
    double retouch = now_seconds() - start;

    munmap(buf, bytes);
    result_t r = { elapsed / (double)pages * 1e9, retouch / (double)pages * 1e9, faults };
    return r;
}

int main(int argc, char **argv) {
    size_t mb = argc > 1 ? strtoul(argv[1], NULL, 10) : 512;
    size_t bytes = mb * 1024UL * 1024UL;

    run_case(bytes / 8, 0); /* warm-up */
    run_case(bytes / 8, 1);

    const int repeats = 5;
    result_t fault_best = { 1e18, 1e18, 0 }, pop_best = { 1e18, 1e18, 0 };
    for (int r = 0; r < repeats; r++) {
        result_t f = run_case(bytes, 0);
        result_t p = run_case(bytes, 1);
        if (f.ns_per_page < fault_best.ns_per_page) fault_best = f;
        if (p.ns_per_page < pop_best.ns_per_page) pop_best = p;
    }

    printf("{\n");
    printf("  \"buffer_mb\": %zu,\n", mb);
    printf("  \"pages\": %zu,\n", bytes / PAGE);
    printf("  \"repeats\": %d,\n", repeats);
    printf("  \"first_touch_ns_per_page\": %.1f,\n", fault_best.ns_per_page);
    printf("  \"first_touch_minor_faults\": %ld,\n", fault_best.faults);
    printf("  \"populate_then_touch_ns_per_page\": %.1f,\n", pop_best.ns_per_page);
    printf("  \"populate_minor_faults\": %ld,\n", pop_best.faults);
    printf("  \"retouch_mapped_ns_per_page\": %.1f,\n", fault_best.retouch_ns_per_page);
    printf("  \"fault_vs_populate_factor\": %.2f\n", fault_best.ns_per_page / pop_best.ns_per_page);
    printf("}\n");
    return 0;
}