poll() vs epoll: What One Wakeup Costs With 10 to 10,000 Idle Descriptors
When a process watches many file descriptors and only one becomes ready, how does the cost of noticing it grow with the number of idle descriptors under poll() and under epoll_wait()?
Results
| Repeats | 5 |
|---|---|
| Largest Fd Count | 10000 |
| Epoll Speedup At Largest | 240.95 |
| Poll Ns Per Wakeup 10 Fds | 3018 |
| Epoll Ns Per Wakeup 10 Fds | 2343 |
| Poll Ns Per Wakeup 100 Fds | 11240 |
| Epoll Ns Per Wakeup 100 Fds | 2329 |
| Poll Ns Per Wakeup 1000 Fds | 89811 |
| Epoll Ns Per Wakeup 1000 Fds | 2304 |
| Poll Ns Per Wakeup 10000 Fds | 631088 |
| Epoll Ns Per Wakeup 10000 Fds | 2619 |
Recorded October 7, 2026 at 2:20 PM UTC, wall clock 29.4s.
Method
10,000 non-blocking eventfds are created up front. For each watch-set size (10, 100, 1,000 and 10,000 descriptors) the program runs the same loop under poll() and under epoll_wait(): write to one eventfd, rotating through the set, wait for readiness with an infinite timeout, then read the eventfd back so it is idle again. The write and the read are identical in both cases, so the difference between the two columns is the wait call. epoll uses level-triggered registration made once before timing; poll() is handed the whole pollfd array on every call, as real poll() users must. Iterations per size are 4,000,000 divided by the set size, with a floor of 4,000, so every size runs long enough to swamp timer resolution. Both cases are warmed up at a tenth of the workload, then measured five times alternating, and the fastest run of each is reported in nanoseconds per wakeup round trip (write + wait + read).
Machine
| CPU | AMD EPYC 9354P 32-Core Processor |
|---|---|
| Cores visible | 8 |
| Memory | 31.3 GB |
| Kernel | 6.8.0-139-generic |
| Architecture | x64 |
| Compiler | gcc (Ubuntu 13.3.0-6ubuntu2~24.04.1) 13.3.0 |
This is a shared virtual server, not an isolated test rig. Absolute throughput will differ on your hardware; the ratio between the two cases is the part that carries over.
Source
The complete program that produced the numbers above. Nothing else was running under our control during the measurement.
/*
* Measures how the cost of one wakeup grows with the number of watched file
* descriptors, for poll() and for epoll_wait().
*
* N eventfds are watched; on every iteration exactly one of them (rotating) is
* made readable, the process waits for readiness, and the event is consumed.
* The write and the read are identical in both cases, so the gap between the
* two is the wait call itself: poll() hands the kernel the whole array and the
* kernel checks every entry, epoll_wait() returns from a ready list that the
* eventfd write already populated.
*
* Build: gcc -O2 -o bench bench.c
* Run: ./bench <max_fds>
* Output: one JSON object on stdout.
*/
#define _GNU_SOURCE
#include <poll.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <sys/epoll.h>
#include <sys/eventfd.h>
#include <sys/resource.h>
#include <time.h>
#include <unistd.h>
static double now_seconds(void) {
struct timespec ts;
clock_gettime(CLOCK_MONOTONIC, &ts);
return (double)ts.tv_sec + (double)ts.tv_nsec / 1e9;
}
static void die(const char *what) {
perror(what);
exit(1);
}
/* Enough iterations to dominate timer noise, few enough that poll() over
* 10,000 descriptors finishes in seconds. */
static long iterations_for(int n) {
long it = 4000000L / n;
return it < 4000 ? 4000 : it;
}
static double run_poll(int *fds, int n, long iterations) {
struct pollfd *pfds = calloc((size_t)n, sizeof(struct pollfd));
for (int i = 0; i < n; i++) {
pfds[i].fd = fds[i];
pfds[i].events = POLLIN;
}
uint64_t one = 1, sink;
double start = now_seconds();
for (long it = 0; it < iterations; it++) {
int target = (int)(it % n);
if (write(fds[target], &one, sizeof(one)) != sizeof(one)) die("write");
if (poll(pfds, (nfds_t)n, -1) != 1) die("poll");
if (read(fds[target], &sink, sizeof(sink)) != sizeof(sink)) die("read");
}
double elapsed = now_seconds() - start;
free(pfds);
return elapsed / (double)iterations * 1e9;
}
static double run_epoll(int *fds, int n, long iterations) {
int ep = epoll_create1(0);
if (ep < 0) die("epoll_create1");
for (int i = 0; i < n; i++) {
struct epoll_event ev = { .events = EPOLLIN, .data.fd = fds[i] };
if (epoll_ctl(ep, EPOLL_CTL_ADD, fds[i], &ev) != 0) die("epoll_ctl");
}
struct epoll_event out[8];
uint64_t one = 1, sink;
double start = now_seconds();
for (long it = 0; it < iterations; it++) {
int target = (int)(it % n);
if (write(fds[target], &one, sizeof(one)) != sizeof(one)) die("write");
if (epoll_wait(ep, out, 8, -1) != 1) die("epoll_wait");
if (read(out[0].data.fd, &sink, sizeof(sink)) != sizeof(sink)) die("read");
}
double elapsed = now_seconds() - start;
close(ep);
return elapsed / (double)iterations * 1e9;
}
int main(int argc, char **argv) {
int max_fds = argc > 1 ? atoi(argv[1]) : 10000;
/* The soft descriptor limit is often 1024; raise it to the hard limit. */
struct rlimit rl;
if (getrlimit(RLIMIT_NOFILE, &rl) == 0) {
rl.rlim_cur = rl.rlim_max;
setrlimit(RLIMIT_NOFILE, &rl);
}
int *fds = malloc(sizeof(int) * (size_t)max_fds);
for (int i = 0; i < max_fds; i++) {
fds[i] = eventfd(0, EFD_NONBLOCK);
if (fds[i] < 0) die("eventfd");
}
int sizes[] = { 10, 100, 1000, 10000, 0 };
const int repeats = 5;
double ratio_at_max = 0;
int largest = 0;
printf("{\n");
for (int s = 0; sizes[s] && sizes[s] <= max_fds; s++) {
int n = sizes[s];
long it = iterations_for(n);
run_poll(fds, n, it / 10); /* warm-up */
run_epoll(fds, n, it / 10);
double poll_best = 1e18, epoll_best = 1e18;
for (int r = 0; r < repeats; r++) {
double p = run_poll(fds, n, it);
double e = run_epoll(fds, n, it);
if (p < poll_best) poll_best = p;
if (e < epoll_best) epoll_best = e;
}
printf(" \"poll_ns_per_wakeup_%d_fds\": %.0f,\n", n, poll_best);
printf(" \"epoll_ns_per_wakeup_%d_fds\": %.0f,\n", n, epoll_best);
ratio_at_max = poll_best / epoll_best;
largest = n;
}
printf(" \"repeats\": %d,\n", repeats);
printf(" \"largest_fd_count\": %d,\n", largest);
printf(" \"epoll_speedup_at_largest\": %.2f\n", ratio_at_max);
printf("}\n");
return 0;
}