二、io_uring 版本:异步 send/recv 压测
io_uring 核心:把 send/recv 提交到SQ,内核异步执行,批量收割CQE。减少系统调用次数,高性能网络首选。
网络io_uring有两种模式:
IORING_OP_SEND/IORING_OP_RECV:原生异步send/recv(最简单,下面示例)- io_uring 多连接模式 + accept(
IORING_OP_ACCEPT)
⚠️ 编译需要链接 liburing:apt install liburing-dev/yum install liburing-devel
io_uring_server.c
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
#include <sys/socket.h>
#include <netinet/in.h>
#include <liburing.h>
#include <time.h>
#define PORT 8888
#define BUF_SIZE (1<<20)
#define QUEUE_DEPTH 16
static int set_nonblock(int fd) {
int flags = fcntl(fd, F_GETFL, 0);
return fcntl(fd, F_SETFL, flags | O_NONBLOCK);
}
int main() {
int listen_fd = socket(AF_INET, SOCK_STREAM, 0);
set_nonblock(listen_fd);
struct sockaddr_in addr = {0};
addr.sin_family = AF_INET;
addr.sin_addr.s_addr = INADDR_ANY;
addr.sin_port = htons(PORT);
bind(listen_fd, (struct sockaddr*)&addr, sizeof(addr));
listen(listen_fd, 128);
struct io_uring ring;
io_uring_queue_init(QUEUE_DEPTH, &ring, 0);
char *buf = malloc(BUF_SIZE);
struct io_uring_sqe *sqe;
struct io_uring_cqe *cqe;
int conn_fd = -1;
long long total_recv = 0;
clock_t start = 0;
// submit accept
sqe = io_uring_get_sqe(&ring);
io_uring_prep_accept(sqe, listen_fd, NULL, NULL, 0);
io_uring_submit(&ring);
while (1) {
io_uring_wait_cqe(&ring, &cqe);
if (cqe->res < 0) {
perror("cqe err");
io_uring_cqe_seen(&ring, cqe);
continue;
}
if (conn_fd < 0) {
// accept完成
conn_fd = cqe->res;
printf("io_uring client connected fd=%d\n", conn_fd);
start = clock();
io_uring_cqe_seen(&ring, cqe);
// 提交recv
sqe = io_uring_get_sqe(&ring);
io_uring_prep_recv(sqe, conn_fd, buf, BUF_SIZE, 0);
io_uring_submit(&ring);
} else {
// recv完成
int n = cqe->res;
io_uring_cqe_seen(&ring, cqe);
if (n <= 0) {
clock_t end = clock();
double sec = (double)(end - start)/CLOCKS_PER_SEC;
printf("io_uring server recv total: %lld bytes, bw: %.2f MB/s\n",
total_recv, total_recv / sec / (1<<20));
close(conn_fd);
free(buf);
io_uring_queue_exit(&ring);
close(listen_fd);
return 0;
}
total_recv += n;
// 继续提交下一次recv
sqe = io_uring_get_sqe(&ring);
io_uring_prep_recv(sqe, conn_fd, buf, BUF_SIZE, 0);
io_uring_submit(&ring);
}
}
}
io_uring_client.c
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
#include <sys/socket.h>
#include <netinet/in.h>
#include <arpa/inet.h>
#include <liburing.h>
#include <time.h>
#define PORT 8888
#define BUF_SIZE (1<<20)
#define SEND_TOTAL (10LL << 30)
#define QUEUE_DEPTH 16
int main() {
int sock = socket(AF_INET, SOCK_STREAM, 0);
struct sockaddr_in serv_addr = {0};
serv_addr.sin_family = AF_INET;
serv_addr.sin_port = htons(PORT);
inet_pton(AF_INET, "127.0.0.1", &serv_addr.sin_addr);
connect(sock, (struct sockaddr*)&serv_addr, sizeof(serv_addr));
struct io_uring ring;
io_uring_queue_init(QUEUE_DEPTH, &ring, 0);
char *buf = malloc(BUF_SIZE);
memset(buf, 'A', BUF_SIZE);
long long remain = SEND_TOTAL;
clock_t start = clock();
struct io_uring_sqe *sqe;
struct io_uring_cqe *cqe;
// 提交第一次send
sqe = io_uring_get_sqe(&ring);
int send_len = remain > BUF_SIZE ? BUF_SIZE : remain;
io_uring_prep_send(sqe, sock, buf, send_len, MSG_NOSIGNAL);
io_uring_submit(&ring);
while (remain > 0) {
io_uring_wait_cqe(&ring, &cqe);
if (cqe->res < 0) {
perror("send cqe err");
exit(1);
}
int n = cqe->res;
io_uring_cqe_seen(&ring, cqe);
remain -= n;
if (remain <= 0) break;
// 继续提交send
sqe = io_uring_get_sqe(&ring);
send_len = remain > BUF_SIZE ? BUF_SIZE : remain;
io_uring_prep_send(sqe, sock, buf, send_len, MSG_NOSIGNAL);
io_uring_submit(&ring);
}
clock_t end = clock();
double sec = (double)(end - start)/CLOCKS_PER_SEC;
printf("io_uring client send total: %lld bytes, bw: %.2f MB/s\n",
SEND_TOTAL, SEND_TOTAL / sec / (1<<20));
close(sock);
free(buf);
io_uring_queue_exit(&ring);
return 0;
}
编译&运行:
gcc io_uring_server.c -o io_uring_server -O2 -luring
gcc io_uring_client.c -o io_uring_client -O2 -luring
# 窗口1
./io_uring_server
# 窗口2
./io_uring_client
io_uring关键点:
QUEUE_DEPTH:SQ/CQ队列深度,对性能影响很大,小包场景调大可以批量提交;io_uring_submit():可以攒一批SQE一次性提交,减少syscall;- 高级:使用
IORING_OP_SEND_ZC零拷贝send(内核5.18+),进一步减少内存拷贝。
三、压测指标采集 & perf 分析(必做)
# 看CPU,sys占比是重点
perf stat ./io_uring_server
# 火焰图看调用栈
perf record -g ./io_uring_server
perf script | ./FlameGraph/stackcollapse-perf.pl | ./FlameGraph/flamegraph.pl > io_uring.svg
采集指标:
- 吞吐 MB/s(server recv字节为准)
- 系统CPU%:epoll小包会sys很高;io_uring因为批量提交sys开销更低
- 平均延迟:改成ping-pong小包测试
- 内存占用、socket buffer溢出(
ss -ti看tcp队列)
内核调优(网络压测)
sysctl -w net.core.rmem_max=33554432
sysctl -w net.core.wmem_max=33554432
sysctl -w net.ipv4.tcp_rmem="4096 87380 33554432"
sysctl -w net.ipv4.tcp_wmem="4096 65536 33554432"
四、epoll vs io_uring 对比总结
| 特性 | epoll | io_uring |
|---|---|---|
| 模型 | 等待fd就绪,就绪后调用send/recv | 异步提交send/recv,内核完成后返回CQE |
| syscall开销 | 每次收发都syscall,小包差 | 可批量提交SQE,syscall更少 |
| 内存拷贝 | 用户→内核拷贝 | 支持send_zc零拷贝 |
| 适用场景 | 大量长连接,连接数优先 | 高吞吐、低sys开销,高性能IO |
参考链接 :0voice · GitHub