Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
99 changes: 94 additions & 5 deletions src/if-linux.c
Original file line number Diff line number Diff line change
Expand Up @@ -413,15 +413,84 @@ if_vlanid(const struct interface *ifp)
return (unsigned short)v.u.VID;
}

#ifdef SO_ATTACH_FILTER
/*
* Reject route messages for tables we do not manage before the kernel queues
* them. if_copyrt() already discards everything outside RT_TABLE_MAIN, but by
* then the message has been allocated, queued and charged against SO_RCVBUF.
* A routing daemon churning a large foreign table -- a full BGP feed in its own
* "kernel table N" -- can therefore overflow the socket in milliseconds and
* cost us the link and address messages we actually need.
*
* A broadcast rejected here is never queued and never charged, so no amount of
* foreign-table churn can overflow us. The test mirrors if_copyrt() exactly,
* so nothing dhcpcd would have acted upon is affected.
*
* A table id wider than rtm_table's 8 bits is reported as RT_TABLE_COMPAT with
* the real id in RTA_TABLE, which cBPF cannot walk. That needs no special
* case: accepting nothing but RT_TABLE_MAIN is the same test if_copyrt() makes,
* so the widest tables are kept out of the queue along with the rest.
*/
/*
* htons() is not a constant expression, so swap the comparands at compile
* time; BPF loads big endian while netlink is host endian.
*/
#if BYTE_ORDER == BIG_ENDIAN
#define BPF_NLMSG_TYPE(t) (t)
#else
#define BPF_NLMSG_TYPE(t) ((((t) & 0xff) << 8) | (((t) >> 8) & 0xff))
#endif

static int
if_linkfilter(int fd)
{
/* A constant program; keep it out of the stack frame. */
static struct sock_filter filter[] = {
/* A = nlmsghdr.nlmsg_type. */
BPF_STMT(BPF_LD | BPF_H | BPF_ABS,
offsetof(struct nlmsghdr, nlmsg_type)),
BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K,
BPF_NLMSG_TYPE(RTM_NEWROUTE), 1, 0),
BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K,
BPF_NLMSG_TYPE(RTM_DELROUTE), 0, 2), /* not a route */

/* A = rtmsg.rtm_table. */
BPF_STMT(BPF_LD | BPF_B | BPF_ABS,
NLMSG_LENGTH(0) + offsetof(struct rtmsg, rtm_table)),
BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, RT_TABLE_MAIN, 0, 1),

BPF_STMT(BPF_RET | BPF_K, (uint32_t)-1), /* accept */
BPF_STMT(BPF_RET | BPF_K, 0), /* drop */
};
struct sock_fprog prog = {
.len = __arraycount(filter),
.filter = filter,
};

return setsockopt(fd, SOL_SOCKET, SO_ATTACH_FILTER, &prog,
sizeof(prog));
}
#endif

int
if_linksocket(struct sockaddr_nl *nl, int protocol, int flags)
if_linksocket(struct sockaddr_nl *nl, int protocol, int flags, bool filter)
{
int fd;

fd = xsocket(AF_NETLINK, SOCK_RAW | SOCK_CLOEXEC | flags, protocol);
if (fd == -1)
return -1;
nl->nl_family = AF_NETLINK;
#ifdef SO_ATTACH_FILTER
/*
* Attach before bind(2) so that not even the leading messages of a
* flood already in progress can be queued unfiltered.
*/
if (filter && if_linkfilter(fd) == -1)
logerr("%s: SO_ATTACH_FILTER", __func__);
#else
UNUSED(filter);
#endif
if (bind(fd, (struct sockaddr *)nl, sizeof(*nl)) == -1) {
close(fd);
return -1;
Expand Down Expand Up @@ -477,7 +546,7 @@ if_opensockets_os(struct dhcpcd_ctx *ctx)
struct priv *priv;
struct sockaddr_nl snl;
socklen_t len;
#ifdef NETLINK_BROADCAST_ERROR
#if defined(NETLINK_BROADCAST_ERROR) || defined(NETLINK_GET_STRICT_CHK)
int on = 1;
#endif

Expand All @@ -501,7 +570,7 @@ if_opensockets_os(struct dhcpcd_ctx *ctx)
snl.nl_groups |= RTMGRP_IPV6_ROUTE | RTMGRP_IPV6_IFADDR | RTMGRP_NEIGH;
#endif

ctx->link_fd = if_linksocket(&snl, NETLINK_ROUTE, SOCK_NONBLOCK);
ctx->link_fd = if_linksocket(&snl, NETLINK_ROUTE, SOCK_NONBLOCK, true);
if (ctx->link_fd == -1)
return -1;
#ifdef NETLINK_BROADCAST_ERROR
Expand All @@ -518,16 +587,36 @@ if_opensockets_os(struct dhcpcd_ctx *ctx)

ctx->priv = priv;
memset(&snl, 0, sizeof(snl));
priv->route_fd = if_linksocket(&snl, NETLINK_ROUTE, 0);
priv->route_fd = if_linksocket(&snl, NETLINK_ROUTE, 0, false);
if (priv->route_fd == -1)
return -1;
#ifdef NETLINK_GET_STRICT_CHK
/*
* Ask the kernel to honour the filters we put in dump requests.
* Without this it ignores them: if_initrt() asks for RT_TABLE_MAIN and
* gets every table, and if_addrflags6() asks for one interface and gets
* every address on the system. Both then filter in userland, so the
* result is unchanged -- but on a host whose routing daemon holds a
* full BGP feed in its own kernel table, that is a quarter of a million
* messages per dump to keep a handful, and rt_build() dumps on events
* as frequent as a Router Advertisement.
*
* This must be done here rather than around each dump because the
* privsep sandbox does not permit setsockopt(2) once it is in force.
* Kernels without the option simply keep the old behaviour.
*/
if (setsockopt(priv->route_fd, SOL_NETLINK, NETLINK_GET_STRICT_CHK, &on,
sizeof(on)) == -1 &&
errno != ENOPROTOOPT)
logerr("%s: NETLINK_GET_STRICT_CHK", __func__);
#endif
len = sizeof(snl);
if (getsockname(priv->route_fd, (struct sockaddr *)&snl, &len) == -1)
return -1;
priv->route_pid = snl.nl_pid;

memset(&snl, 0, sizeof(snl));
priv->generic_fd = if_linksocket(&snl, NETLINK_GENERIC, 0);
priv->generic_fd = if_linksocket(&snl, NETLINK_GENERIC, 0, false);
if (priv->generic_fd == -1)
return -1;

Expand Down
2 changes: 1 addition & 1 deletion src/if.h
Original file line number Diff line number Diff line change
Expand Up @@ -287,7 +287,7 @@ struct interface *if_findifpfromcmsg(struct dhcpcd_ctx *, struct msghdr *,
int *);

#ifdef __linux__
int if_linksocket(struct sockaddr_nl *, int, int);
int if_linksocket(struct sockaddr_nl *, int, int, bool);
int if_getnetlink(struct dhcpcd_ctx *, struct iovec *, int, int,
int (*)(struct dhcpcd_ctx *, void *, struct nlmsghdr *), void *);
#endif
Expand Down