Skip to content

RTM_NEWROUTE without RTA_OIF installs a gateway route with no NIC #14967

Description

@dany74q

localRoute takes a route's NIC from RTA_OIF alone (pkg/sentry/socket/netstack/stack.go:1074), so NewRoute installs a gateway route added without RTA_OIF with NIC 0 (stack.go:1122), and FindRoute skips routes whose NIC does not exist (pkg/tcpip/stack/stack.go:1684): the route is acked but no traffic can use it, where Linux takes the device from the route that reaches the gateway (net/ipv4/fib_semantics.c:1250, net/ipv6/route.c:3490).

ip route add default via <gateway> sends no RTA_OIF, so network setup in the sandbox that adds its default route that way gets ENETUNREACH on every connection through it.

reproducer
#define _GNU_SOURCE
#include <arpa/inet.h>
#include <errno.h>
#include <linux/if_addr.h>
#include <linux/rtnetlink.h>
#include <linux/veth.h>
#include <net/if.h>
#include <sched.h>
#include <stdio.h>
#include <string.h>
#include <sys/socket.h>
#include <unistd.h>

struct req {
  struct nlmsghdr hdr;
  char buf[256];
};

static int fd;
static unsigned seq;

static struct rtattr* attr(struct req* r, int type, const void* data, int len) {
  struct rtattr* rta = (void*)((char*)r + NLMSG_ALIGN(r->hdr.nlmsg_len));
  rta->rta_type = type;
  rta->rta_len = RTA_LENGTH(len);
  if (len) memcpy(RTA_DATA(rta), data, len);
  r->hdr.nlmsg_len = NLMSG_ALIGN(r->hdr.nlmsg_len) + RTA_ALIGN(rta->rta_len);
  return rta;
}

static int request(struct req* r, int type, int flags) {
  r->hdr.nlmsg_type = type;
  r->hdr.nlmsg_flags = NLM_F_REQUEST | NLM_F_ACK | flags;
  r->hdr.nlmsg_seq = ++seq;
  send(fd, r, r->hdr.nlmsg_len, 0);
  char buf[4096];
  int n = recv(fd, buf, sizeof(buf), 0);
  struct nlmsghdr* ack = (struct nlmsghdr*)buf;
  if (n < (int)NLMSG_LENGTH(sizeof(struct nlmsgerr)) ||
      ack->nlmsg_type != NLMSG_ERROR) {
    return -1;
  }
  return -((struct nlmsgerr*)NLMSG_DATA(ack))->error;
}

static void link_up(const char* name) {
  struct req r = {.hdr.nlmsg_len = NLMSG_LENGTH(sizeof(struct ifinfomsg))};
  struct ifinfomsg* ifi = NLMSG_DATA(&r.hdr);
  ifi->ifi_index = if_nametoindex(name);
  ifi->ifi_flags = IFF_UP;
  ifi->ifi_change = IFF_UP;
  request(&r, RTM_NEWLINK, 0);
}

static void add_veth(void) {
  struct req r = {.hdr.nlmsg_len = NLMSG_LENGTH(sizeof(struct ifinfomsg))};
  attr(&r, IFLA_IFNAME, "veth1", 6);
  struct rtattr* info = attr(&r, IFLA_LINKINFO, NULL, 0);
  attr(&r, IFLA_INFO_KIND, "veth", 4);
  struct rtattr* data = attr(&r, IFLA_INFO_DATA, NULL, 0);
  struct ifinfomsg peer = {};
  struct rtattr* p = attr(&r, VETH_INFO_PEER, &peer, sizeof(peer));
  attr(&r, IFLA_IFNAME, "veth2", 6);
  p->rta_len = (char*)&r + r.hdr.nlmsg_len - (char*)p;
  data->rta_len = (char*)&r + r.hdr.nlmsg_len - (char*)data;
  info->rta_len = (char*)&r + r.hdr.nlmsg_len - (char*)info;
  request(&r, RTM_NEWLINK, NLM_F_CREATE | NLM_F_EXCL);
}

static void add_addr(int family, const char* local, int prefixlen) {
  struct req r = {.hdr.nlmsg_len = NLMSG_LENGTH(sizeof(struct ifaddrmsg))};
  struct ifaddrmsg* ifa = NLMSG_DATA(&r.hdr);
  ifa->ifa_family = family;
  ifa->ifa_prefixlen = prefixlen;
  ifa->ifa_flags = IFA_F_NODAD;
  ifa->ifa_index = if_nametoindex("veth1");
  unsigned char addr[16];
  inet_pton(family, local, addr);
  attr(&r, IFA_LOCAL, addr, family == AF_INET ? 4 : 16);
  request(&r, RTM_NEWADDR, NLM_F_CREATE);
}

static void default_route_and_connect(int family, const char* gateway,
                                      const char* remote) {
  struct req r = {.hdr.nlmsg_len = NLMSG_LENGTH(sizeof(struct rtmsg))};
  struct rtmsg* rtm = NLMSG_DATA(&r.hdr);
  rtm->rtm_family = family;
  rtm->rtm_table = RT_TABLE_MAIN;
  rtm->rtm_protocol = RTPROT_BOOT;
  rtm->rtm_type = RTN_UNICAST;
  unsigned char addr[16];
  inet_pton(family, gateway, addr);
  attr(&r, RTA_GATEWAY, addr, family == AF_INET ? 4 : 16);
  int err = request(&r, RTM_NEWROUTE, NLM_F_CREATE | NLM_F_EXCL);
  printf("default via %s: %s\n", gateway, strerror(err));

  struct sockaddr_storage ss = {};
  struct sockaddr_in* sin = (struct sockaddr_in*)&ss;
  struct sockaddr_in6* sin6 = (struct sockaddr_in6*)&ss;
  if (family == AF_INET) {
    sin->sin_family = AF_INET;
    sin->sin_port = htons(9);
    inet_pton(AF_INET, remote, &sin->sin_addr);
  } else {
    sin6->sin6_family = AF_INET6;
    sin6->sin6_port = htons(9);
    inet_pton(AF_INET6, remote, &sin6->sin6_addr);
  }
  int s = socket(family, SOCK_DGRAM, 0);
  int rc = connect(s, (struct sockaddr*)&ss,
                   family == AF_INET ? sizeof(*sin) : sizeof(*sin6));
  printf("connect %s: %s\n", remote, rc == 0 ? "Success" : strerror(errno));
  close(s);
}

int main(void) {
  unshare(CLONE_NEWNET);
  fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_ROUTE);
  add_veth();
  link_up("veth1");
  link_up("veth2");
  add_addr(AF_INET, "192.0.2.1", 24);
  add_addr(AF_INET6, "2001:db8::1", 64);
  default_route_and_connect(AF_INET, "192.0.2.2", "203.0.113.1");
  default_route_and_connect(AF_INET6, "2001:db8::2", "2001:db8:1::1");
  return 0;
}

Proposed fix in #14969.

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Assignees

No one assigned

    Labels

    No labels
    No labels

    Type

    No type

    Projects

    No projects

      Milestone

      No milestone

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions