[PATCH 1/3] shared/net-util-linux: add support for retrieving egress netdev
Nilay Shroff <[email protected]>
| Newsgroups | org.infradead.lists.linux-nvme |
|---|---|
| Message-ID | <[email protected]> |
Add a new shared helper, shr_route_get_egress_iface(), which performs a route lookup and retrieves the egress netdev for a given destination IPv4/IPv6 address and, optionally, source IPv4/IPv6 address. Use netlink to perform the route lookup. This is a preparatory patch. A subsequent patch will use this helper to determine the egress netdev when calculating the default --nr-io-queues value for NVMe/TCP connections. Signed-off-by: Nilay Shroff <[email protected]> --- shared/net-util-linux.c | 228 ++++++++++++++++++++++++++++++++++++++++ shared/net-util.h | 16 +++ 2 files changed, 244 insertions(+) diff --git a/shared/net-util-linux.c b/shared/net-util-linux.c index 57c025b05..887755ce9 100644 --- a/shared/net-util-linux.c +++ b/shared/net-util-linux.c @@ -7,15 +7,23 @@ */ #include <arpa/inet.h> +#include <asm/types.h> #include <errno.h> +#include <linux/netlink.h> +#include <linux/rtnetlink.h> #include <net/if.h> #include <netinet/in.h> +#include <stdio.h> #include <stdlib.h> #include <string.h> #include <sys/socket.h> +#include <unistd.h> #include "net-util.h" +#define NETLINK_BUFFER_SIZE 4096 +#define NETLINK_SEQ_NUM 1 + /* * Parse @addr (IPv4, or IPv6 with an optional "%scope" suffix on a * link-local address) into @ss. @addr is never a hostname -- resolving one @@ -181,3 +189,223 @@ bool shr_iface_primary_addr_matches(const struct ifaddrs *iface_list, return match_found; } + +static int rtattr_append(struct nlmsghdr *nlh, char *buf, size_t buflen, + unsigned short type, void *attrval, int attrlen) +{ + struct rtattr *rta; + int nlen = NLMSG_ALIGN(nlh->nlmsg_len); + int rlen = RTA_LENGTH(attrlen); + + if (nlen + rlen >= buflen) + return -ENOSPC; + + rta = (struct rtattr *)(buf + nlen); + rta->rta_type = type; + rta->rta_len = rlen; + memcpy(RTA_DATA(rta), attrval, attrlen); + + nlh->nlmsg_len = nlen + rlen; + + return 0; +} + +int shr_route_get_egress_iface(const char *saddr, const char *daddr, + char *ifname, size_t iflen) +{ + struct sockaddr_nl nl = {.nl_family = AF_NETLINK}; + struct sockaddr_storage ss_src, ss_dst; + char buf[NETLINK_BUFFER_SIZE] = {}; + struct sockaddr_in6 *src6, *dst6; + struct sockaddr_in *src, *dst; + struct msghdr msg = {}; + struct nlmsghdr *nlh; + struct rtmsg *rtmsg; + struct rtattr *rta; + int nlen, attrlen; + struct iovec iov; + int fd, ret = 0; + + if (!daddr) + return -EINVAL; + + ret = parse_numeric_addr(daddr, &ss_dst); + if (ret < 0) { + fprintf(stderr, "Invalid dest address\n"); + return ret; + } + + if (saddr) { + ret = parse_numeric_addr(saddr, &ss_src); + if (ret < 0) { + fprintf(stderr, "Invalid src address\n"); + return ret; + } + + if (ss_src.ss_family != ss_dst.ss_family) { + fprintf(stderr, "src/dst address family doesn't match\n"); + return -EINVAL; + } + } + + if (iflen < IF_NAMESIZE) { + fprintf(stderr, "Invalid ifname length\n"); + return -EINVAL; + } + + fd = socket(AF_NETLINK, SOCK_DGRAM, NETLINK_ROUTE); + if (fd < 0) { + fprintf(stderr, "Opening netlink socket failed\n"); + return -errno; + } + + /* encode nlmsghdr */ + nlh = (struct nlmsghdr *)buf; + + nlh->nlmsg_type = RTM_GETROUTE; + nlh->nlmsg_flags = NLM_F_REQUEST; + nlh->nlmsg_pid = 0; /* to kernel */ + nlh->nlmsg_seq = NETLINK_SEQ_NUM; + nlh->nlmsg_len = sizeof(struct nlmsghdr); + + /* append rtmsg */ + nlh->nlmsg_len = NLMSG_ALIGN(nlh->nlmsg_len); + + rtmsg = (struct rtmsg *)(buf + nlh->nlmsg_len); + rtmsg->rtm_family = ss_dst.ss_family; + rtmsg->rtm_dst_len = (ss_dst.ss_family == AF_INET) ? 32 : 128; + + nlh->nlmsg_len += sizeof(struct rtmsg); + + /* append attribute RTA_DST */ + if (ss_dst.ss_family == AF_INET) { + dst = (struct sockaddr_in *)&ss_dst; + + ret = rtattr_append(nlh, buf, sizeof(buf), RTA_DST, + &dst->sin_addr, sizeof(struct in_addr)); + if (ret) + goto out; + } else { + dst6 = (struct sockaddr_in6 *)&ss_dst; + + ret = rtattr_append(nlh, buf, sizeof(buf), RTA_DST, + &dst6->sin6_addr, sizeof(struct in6_addr)); + if (ret) + goto out; + } + + if (saddr) { + /* append attribute RTA_SRC */ + if (ss_src.ss_family == AF_INET) { + src = (struct sockaddr_in *)&ss_src; + + ret = rtattr_append(nlh, buf, sizeof(buf), RTA_SRC, + &src->sin_addr, sizeof(struct in_addr)); + if (ret) + goto out; + } else { + src6 = (struct sockaddr_in6 *)&ss_src; + + ret = rtattr_append(nlh, buf, sizeof(buf), RTA_SRC, + &src6->sin6_addr, sizeof(struct in6_addr)); + if (ret) + goto out; + } + } + + /* construct sendmsg and send to kernel */ + iov.iov_base = buf; + iov.iov_len = nlh->nlmsg_len; + + msg.msg_iov = &iov; + msg.msg_iovlen = 1; + msg.msg_name = &nl; + msg.msg_namelen = sizeof(nl); + + if (sendmsg(fd, &msg, 0) < 0) { + fprintf(stderr, "Sendmsg to kernel failed\n"); + ret = -errno; + goto out; + } + + /* construct recvmsg buffer and wait for the response from kernel */ + memset(&msg, 0, sizeof(msg)); + memset(buf, 0, sizeof(buf)); + + iov.iov_base = buf; + iov.iov_len = sizeof(buf); + + msg.msg_iov = &iov; + msg.msg_iovlen = 1; + msg.msg_name = &nl; + msg.msg_namelen = sizeof(nl); + + nlen = recvmsg(fd, &msg, 0); + if (nlen < 0) { + fprintf(stderr, "Recvmsg from kernel failed\n"); + ret = -errno; + goto out; + } + close(fd); + + nlh = msg.msg_iov->iov_base; + + for (; NLMSG_OK(nlh, nlen); nlh = NLMSG_NEXT(nlh, nlen)) { + + if (nlh->nlmsg_type == NLMSG_ERROR) { + struct nlmsgerr *err = NLMSG_DATA(nlh); + + /* + * Netlink error message is sent as an ack when user + * explicitly requested for it and in which case + * err->error is set to zero. + */ + if (!err->error) + continue; + + fprintf(stderr, "Received nlmsg error: %s\n", + strerror(-err->error)); + break; + } + + if (nlh->nlmsg_type == NLMSG_DONE) + break; + + /* + * RTM_GETROUTE reply is expected to be of type RTM_NEWROUTE. + */ + if (nlh->nlmsg_type != RTM_NEWROUTE) + continue; + + if (nlh->nlmsg_seq != NETLINK_SEQ_NUM) + continue; + + /* parse and decode received netlink message */ + rtmsg = NLMSG_DATA(nlh); + rta = RTM_RTA(rtmsg); + attrlen = RTM_PAYLOAD(nlh); + + for (; RTA_OK(rta, attrlen); rta = RTA_NEXT(rta, attrlen)) { + + switch (rta->rta_type) { + case RTA_OIF: { + int ifindex; + + ifindex = *((int *)RTA_DATA(rta)); + if (!if_indextoname(ifindex, ifname)) { + fprintf(stderr, "if_indextoname failed\n"); + ret = -errno; + } + return ret; + } + default: + break; + } + } + } + + return -ENOENT; +out: + close(fd); + return ret; +} diff --git a/shared/net-util.h b/shared/net-util.h index 577e582f6..0208923e5 100644 --- a/shared/net-util.h +++ b/shared/net-util.h @@ -48,3 +48,19 @@ const char *shr_iface_matching_addr(const struct ifaddrs *iface_list, */ bool shr_iface_primary_addr_matches(const struct ifaddrs *iface_list, const char *iface, const char *addr); + +/* + * shr_route_get_egress_iface - Lookup route table and find egress inerface + * + * @saddr: optional source address + * @daddr: destintion address + * @ifname: buffer to hold the matching iface name + * @iflen: length of ifname buffer (must be greater or equal to IF_NAMESIZE) + * + * Lookup route table using specified saddr and daddr and find the egress + * interface. The saddr is optional however daddr is mandatory. + * + * Return: 0 on success and negative errno on failure. + */ +int shr_route_get_egress_iface(const char *saddr, const char *daddr, + char *ifname, size_t iflen); -- 2.53.0