☰
用io_uring进行io测试
2026/10/5 10:26:43 网站建设 项目流程

二、io_uring 版本:异步 send/recv 压测

io_uring 核心:把send/recv提交到SQ,内核异步执行,批量收割CQE。减少系统调用次数,高性能网络首选。

网络io_uring有两种模式:

  1. IORING_OP_SEND/IORING_OP_RECV:原生异步send/recv(最简单,下面示例)
  2. io_uring 多连接模式 + accept(IORING_OP_ACCEPT)

⚠️ 编译需要链接 liburing:apt install liburing-dev/yum install liburing-devel

io_uring_server.c

#include <stdio.h> #include <stdlib.h> #include <string.h> #include <unistd.h> #include <sys/socket.h> #include <netinet/in.h> #include <liburing.h> #include <time.h> #define PORT 8888 #define BUF_SIZE (1<<20) #define QUEUE_DEPTH 16 static int set_nonblock(int fd) { int flags = fcntl(fd, F_GETFL, 0); return fcntl(fd, F_SETFL, flags | O_NONBLOCK); } int main() { int listen_fd = socket(AF_INET, SOCK_STREAM, 0); set_nonblock(listen_fd); struct sockaddr_in addr = {0}; addr.sin_family = AF_INET; addr.sin_addr.s_addr = INADDR_ANY; addr.sin_port = htons(PORT); bind(listen_fd, (struct sockaddr*)&addr, sizeof(addr)); listen(listen_fd, 128); struct io_uring ring; io_uring_queue_init(QUEUE_DEPTH, &ring, 0); char *buf = malloc(BUF_SIZE); struct io_uring_sqe *sqe; struct io_uring_cqe *cqe; int conn_fd = -1; long long total_recv = 0; clock_t start = 0; // submit accept sqe = io_uring_get_sqe(&ring); io_uring_prep_accept(sqe, listen_fd, NULL, NULL, 0); io_uring_submit(&ring); while (1) { io_uring_wait_cqe(&ring, &cqe); if (cqe->res < 0) { perror("cqe err"); io_uring_cqe_seen(&ring, cqe); continue; } if (conn_fd < 0) { // accept完成 conn_fd = cqe->res; printf("io_uring client connected fd=%d\n", conn_fd); start = clock(); io_uring_cqe_seen(&ring, cqe); // 提交recv sqe = io_uring_get_sqe(&ring); io_uring_prep_recv(sqe, conn_fd, buf, BUF_SIZE, 0); io_uring_submit(&ring); } else { // recv完成 int n = cqe->res; io_uring_cqe_seen(&ring, cqe); if (n <= 0) { clock_t end = clock(); double sec = (double)(end - start)/CLOCKS_PER_SEC; printf("io_uring server recv total: %lld bytes, bw: %.2f MB/s\n", total_recv, total_recv / sec / (1<<20)); close(conn_fd); free(buf); io_uring_queue_exit(&ring); close(listen_fd); return 0; } total_recv += n; // 继续提交下一次recv sqe = io_uring_get_sqe(&ring); io_uring_prep_recv(sqe, conn_fd, buf, BUF_SIZE, 0); io_uring_submit(&ring); } } }

io_uring_client.c

#include <stdio.h> #include <stdlib.h> #include <string.h> #include <unistd.h> #include <sys/socket.h> #include <netinet/in.h> #include <arpa/inet.h> #include <liburing.h> #include <time.h> #define PORT 8888 #define BUF_SIZE (1<<20) #define SEND_TOTAL (10LL << 30) #define QUEUE_DEPTH 16 int main() { int sock = socket(AF_INET, SOCK_STREAM, 0); struct sockaddr_in serv_addr = {0}; serv_addr.sin_family = AF_INET; serv_addr.sin_port = htons(PORT); inet_pton(AF_INET, "127.0.0.1", &serv_addr.sin_addr); connect(sock, (struct sockaddr*)&serv_addr, sizeof(serv_addr)); struct io_uring ring; io_uring_queue_init(QUEUE_DEPTH, &ring, 0); char *buf = malloc(BUF_SIZE); memset(buf, 'A', BUF_SIZE); long long remain = SEND_TOTAL; clock_t start = clock(); struct io_uring_sqe *sqe; struct io_uring_cqe *cqe; // 提交第一次send sqe = io_uring_get_sqe(&ring); int send_len = remain > BUF_SIZE ? BUF_SIZE : remain; io_uring_prep_send(sqe, sock, buf, send_len, MSG_NOSIGNAL); io_uring_submit(&ring); while (remain > 0) { io_uring_wait_cqe(&ring, &cqe); if (cqe->res < 0) { perror("send cqe err"); exit(1); } int n = cqe->res; io_uring_cqe_seen(&ring, cqe); remain -= n; if (remain <= 0) break; // 继续提交send sqe = io_uring_get_sqe(&ring); send_len = remain > BUF_SIZE ? BUF_SIZE : remain; io_uring_prep_send(sqe, sock, buf, send_len, MSG_NOSIGNAL); io_uring_submit(&ring); } clock_t end = clock(); double sec = (double)(end - start)/CLOCKS_PER_SEC; printf("io_uring client send total: %lld bytes, bw: %.2f MB/s\n", SEND_TOTAL, SEND_TOTAL / sec / (1<<20)); close(sock); free(buf); io_uring_queue_exit(&ring); return 0; }

编译&运行:

gcc io_uring_server.c -o io_uring_server -O2 -luring gcc io_uring_client.c -o io_uring_client -O2 -luring # 窗口1 ./io_uring_server # 窗口2 ./io_uring_client

io_uring关键点:

  1. QUEUE_DEPTH:SQ/CQ队列深度,对性能影响很大,小包场景调大可以批量提交;
  2. io_uring_submit():可以攒一批SQE一次性提交,减少syscall;
  3. 高级:使用IORING_OP_SEND_ZC零拷贝send(内核5.18+),进一步减少内存拷贝。

三、压测指标采集 & perf 分析(必做)

# 看CPU,sys占比是重点 perf stat ./io_uring_server # 火焰图看调用栈 perf record -g ./io_uring_server perf script | ./FlameGraph/stackcollapse-perf.pl | ./FlameGraph/flamegraph.pl > io_uring.svg

采集指标:

  1. 吞吐 MB/s(server recv字节为准)
  2. 系统CPU%:epoll小包会sys很高;io_uring因为批量提交sys开销更低
  3. 平均延迟:改成ping-pong小包测试
  4. 内存占用、socket buffer溢出(ss -ti看tcp队列)

内核调优(网络压测)

sysctl -w net.core.rmem_max=33554432 sysctl -w net.core.wmem_max=33554432 sysctl -w net.ipv4.tcp_rmem="4096 87380 33554432" sysctl -w net.ipv4.tcp_wmem="4096 65536 33554432"

四、epoll vs io_uring 对比总结

特性epollio_uring
模型等待fd就绪,就绪后调用send/recv异步提交send/recv,内核完成后返回CQE
syscall开销每次收发都syscall,小包差可批量提交SQE,syscall更少
内存拷贝用户→内核拷贝支持send_zc零拷贝
适用场景大量长连接,连接数优先高吞吐、低sys开销,高性能IO

参考链接 :0voice · GitHub

需要专业的网站建设服务?

联系我们获取免费的网站建设咨询和方案报价,让我们帮助您实现业务目标

立即咨询