xref: /openbmc/qemu/net/tap-linux.c (revision 6e47f7cfcd78ed8e6f192cb0a4c61f209d0c2aaf)
1c28b1c10SMark McLoughlin /*
2c28b1c10SMark McLoughlin  * QEMU System Emulator
3c28b1c10SMark McLoughlin  *
4c28b1c10SMark McLoughlin  * Copyright (c) 2003-2008 Fabrice Bellard
5c28b1c10SMark McLoughlin  * Copyright (c) 2009 Red Hat, Inc.
6c28b1c10SMark McLoughlin  *
7c28b1c10SMark McLoughlin  * Permission is hereby granted, free of charge, to any person obtaining a copy
8c28b1c10SMark McLoughlin  * of this software and associated documentation files (the "Software"), to deal
9c28b1c10SMark McLoughlin  * in the Software without restriction, including without limitation the rights
10c28b1c10SMark McLoughlin  * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
11c28b1c10SMark McLoughlin  * copies of the Software, and to permit persons to whom the Software is
12c28b1c10SMark McLoughlin  * furnished to do so, subject to the following conditions:
13c28b1c10SMark McLoughlin  *
14c28b1c10SMark McLoughlin  * The above copyright notice and this permission notice shall be included in
15c28b1c10SMark McLoughlin  * all copies or substantial portions of the Software.
16c28b1c10SMark McLoughlin  *
17c28b1c10SMark McLoughlin  * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
18c28b1c10SMark McLoughlin  * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
19c28b1c10SMark McLoughlin  * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
20c28b1c10SMark McLoughlin  * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
21c28b1c10SMark McLoughlin  * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
22c28b1c10SMark McLoughlin  * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
23c28b1c10SMark McLoughlin  * THE SOFTWARE.
24c28b1c10SMark McLoughlin  */
25c28b1c10SMark McLoughlin 
262744d920SPeter Maydell #include "qemu/osdep.h"
271422e32dSPaolo Bonzini #include "tap_int.h"
281422e32dSPaolo Bonzini #include "tap-linux.h"
29c28b1c10SMark McLoughlin #include "net/tap.h"
30c28b1c10SMark McLoughlin 
31c28b1c10SMark McLoughlin #include <net/if.h>
32c28b1c10SMark McLoughlin #include <sys/ioctl.h>
33c28b1c10SMark McLoughlin 
34da34e65cSMarkus Armbruster #include "qapi/error.h"
351de7afc9SPaolo Bonzini #include "qemu/error-report.h"
36f348b6d1SVeronia Bahaa #include "qemu/cutils.h"
37c28b1c10SMark McLoughlin 
3891ca60e0SMichael Tokarev #define PATH_NET_TUN "/dev/net/tun"
3991ca60e0SMichael Tokarev 
tap_open(char * ifname,int ifname_size,int * vnet_hdr,int vnet_hdr_required,int mq_required,Error ** errp)40264986e2SJason Wang int tap_open(char *ifname, int ifname_size, int *vnet_hdr,
41468dd824SMarkus Armbruster              int vnet_hdr_required, int mq_required, Error **errp)
42c28b1c10SMark McLoughlin {
43c28b1c10SMark McLoughlin     struct ifreq ifr;
44c28b1c10SMark McLoughlin     int fd, ret;
4589e6d68eSMichael S. Tsirkin     int len = sizeof(struct virtio_net_hdr);
46d26e445cSPeter Lieven     unsigned int features;
47c28b1c10SMark McLoughlin 
488b6aa693SNikita Ivanov     fd = RETRY_ON_EINTR(open(PATH_NET_TUN, O_RDWR));
49c28b1c10SMark McLoughlin     if (fd < 0) {
5047896e2fSMarkus Armbruster         error_setg_errno(errp, errno, "could not open %s", PATH_NET_TUN);
51c28b1c10SMark McLoughlin         return -1;
52c28b1c10SMark McLoughlin     }
53c28b1c10SMark McLoughlin     memset(&ifr, 0, sizeof(ifr));
54c28b1c10SMark McLoughlin     ifr.ifr_flags = IFF_TAP | IFF_NO_PI;
55c28b1c10SMark McLoughlin 
561f149e72SKusanagi Kouichi     if (ioctl(fd, TUNGETFEATURES, &features) == -1) {
573dc6f869SAlistair Francis         warn_report("TUNGETFEATURES failed: %s", strerror(errno));
581f149e72SKusanagi Kouichi         features = 0;
591f149e72SKusanagi Kouichi     }
601f149e72SKusanagi Kouichi 
611f149e72SKusanagi Kouichi     if (features & IFF_ONE_QUEUE) {
62d26e445cSPeter Lieven         ifr.ifr_flags |= IFF_ONE_QUEUE;
63d26e445cSPeter Lieven     }
64c28b1c10SMark McLoughlin 
65d26e445cSPeter Lieven     if (*vnet_hdr) {
661f149e72SKusanagi Kouichi         if (features & IFF_VNET_HDR) {
67c28b1c10SMark McLoughlin             *vnet_hdr = 1;
68c28b1c10SMark McLoughlin             ifr.ifr_flags |= IFF_VNET_HDR;
696720b35bSPierre Riteau         } else {
706720b35bSPierre Riteau             *vnet_hdr = 0;
71c28b1c10SMark McLoughlin         }
72c28b1c10SMark McLoughlin 
73c28b1c10SMark McLoughlin         if (vnet_hdr_required && !*vnet_hdr) {
7447896e2fSMarkus Armbruster             error_setg(errp, "vnet_hdr=1 requested, but no kernel "
75c28b1c10SMark McLoughlin                        "support for IFF_VNET_HDR available");
76c28b1c10SMark McLoughlin             close(fd);
77c28b1c10SMark McLoughlin             return -1;
78c28b1c10SMark McLoughlin         }
7989e6d68eSMichael S. Tsirkin         /*
8089e6d68eSMichael S. Tsirkin          * Make sure vnet header size has the default value: for a persistent
8189e6d68eSMichael S. Tsirkin          * tap it might have been modified e.g. by another instance of qemu.
8289e6d68eSMichael S. Tsirkin          * Ignore errors since old kernels do not support this ioctl: in this
8389e6d68eSMichael S. Tsirkin          * case the header size implicitly has the correct value.
8489e6d68eSMichael S. Tsirkin          */
8589e6d68eSMichael S. Tsirkin         ioctl(fd, TUNSETVNETHDRSZ, &len);
86c28b1c10SMark McLoughlin     }
87c28b1c10SMark McLoughlin 
8894fdc6d0SJason Wang     if (mq_required) {
891f149e72SKusanagi Kouichi         if (!(features & IFF_MULTI_QUEUE)) {
9047896e2fSMarkus Armbruster             error_setg(errp, "multiqueue required, but no kernel "
9194fdc6d0SJason Wang                        "support for IFF_MULTI_QUEUE available");
9294fdc6d0SJason Wang             close(fd);
9394fdc6d0SJason Wang             return -1;
9494fdc6d0SJason Wang         } else {
9594fdc6d0SJason Wang             ifr.ifr_flags |= IFF_MULTI_QUEUE;
9694fdc6d0SJason Wang         }
9794fdc6d0SJason Wang     }
9894fdc6d0SJason Wang 
99c28b1c10SMark McLoughlin     if (ifname[0] != '\0')
100c28b1c10SMark McLoughlin         pstrcpy(ifr.ifr_name, IFNAMSIZ, ifname);
101c28b1c10SMark McLoughlin     else
102c28b1c10SMark McLoughlin         pstrcpy(ifr.ifr_name, IFNAMSIZ, "tap%d");
103c28b1c10SMark McLoughlin     ret = ioctl(fd, TUNSETIFF, (void *) &ifr);
104c28b1c10SMark McLoughlin     if (ret != 0) {
10593a7320eSLuiz Capitulino         if (ifname[0] != '\0') {
10647896e2fSMarkus Armbruster             error_setg_errno(errp, errno, "could not configure %s (%s)",
10747896e2fSMarkus Armbruster                              PATH_NET_TUN, ifr.ifr_name);
10893a7320eSLuiz Capitulino         } else {
10947896e2fSMarkus Armbruster             error_setg_errno(errp, errno, "could not configure %s",
11047896e2fSMarkus Armbruster                              PATH_NET_TUN);
11193a7320eSLuiz Capitulino         }
112c28b1c10SMark McLoughlin         close(fd);
113c28b1c10SMark McLoughlin         return -1;
114c28b1c10SMark McLoughlin     }
115c28b1c10SMark McLoughlin     pstrcpy(ifname, ifname_size, ifr.ifr_name);
11622e135fcSMarc-André Lureau     g_unix_set_fd_nonblocking(fd, true, NULL);
117c28b1c10SMark McLoughlin     return fd;
118c28b1c10SMark McLoughlin }
11915ac913bSMark McLoughlin 
120f157ed20SMichael S. Tsirkin /* sndbuf implements a kind of flow control for tap.
121f157ed20SMichael S. Tsirkin  * Unfortunately when it's enabled, and packets are sent
122f157ed20SMichael S. Tsirkin  * to other guests on the same host, the receiver
123f157ed20SMichael S. Tsirkin  * can lock up the transmitter indefinitely.
124f157ed20SMichael S. Tsirkin  *
125f157ed20SMichael S. Tsirkin  * To avoid packet loss, sndbuf should be set to a value lower than the tx
126f157ed20SMichael S. Tsirkin  * queue capacity of any destination network interface.
12715ac913bSMark McLoughlin  * Ethernet NICs generally have txqueuelen=1000, so 1Mb is
128f157ed20SMichael S. Tsirkin  * a good value, given a 1500 byte MTU.
12915ac913bSMark McLoughlin  */
130f157ed20SMichael S. Tsirkin #define TAP_DEFAULT_SNDBUF 0
13115ac913bSMark McLoughlin 
tap_set_sndbuf(int fd,const NetdevTapOptions * tap,Error ** errp)13280b832c3SMarkus Armbruster void tap_set_sndbuf(int fd, const NetdevTapOptions *tap, Error **errp)
13315ac913bSMark McLoughlin {
13415ac913bSMark McLoughlin     int sndbuf;
13515ac913bSMark McLoughlin 
13608c573a8SLaszlo Ersek     sndbuf = !tap->has_sndbuf       ? TAP_DEFAULT_SNDBUF :
13708c573a8SLaszlo Ersek              tap->sndbuf > INT_MAX  ? INT_MAX :
13808c573a8SLaszlo Ersek              tap->sndbuf;
13908c573a8SLaszlo Ersek 
14015ac913bSMark McLoughlin     if (!sndbuf) {
14115ac913bSMark McLoughlin         sndbuf = INT_MAX;
14215ac913bSMark McLoughlin     }
14315ac913bSMark McLoughlin 
14408c573a8SLaszlo Ersek     if (ioctl(fd, TUNSETSNDBUF, &sndbuf) == -1 && tap->has_sndbuf) {
14580b832c3SMarkus Armbruster         error_setg_errno(errp, errno, "TUNSETSNDBUF ioctl failed");
14615ac913bSMark McLoughlin     }
14715ac913bSMark McLoughlin }
148dc69004cSMark McLoughlin 
tap_probe_vnet_hdr(int fd,Error ** errp)149e7b347d0SDaniel P. Berrange int tap_probe_vnet_hdr(int fd, Error **errp)
150dc69004cSMark McLoughlin {
151dc69004cSMark McLoughlin     struct ifreq ifr;
152e29919c9SPeter Foley     memset(&ifr, 0, sizeof(ifr));
153dc69004cSMark McLoughlin 
154dc69004cSMark McLoughlin     if (ioctl(fd, TUNGETIFF, &ifr) != 0) {
155e7b347d0SDaniel P. Berrange         /* TUNGETIFF is available since kernel v2.6.27 */
156e7b347d0SDaniel P. Berrange         error_setg_errno(errp, errno,
157e7b347d0SDaniel P. Berrange                          "Unable to query TUNGETIFF on FD %d", fd);
158e7b347d0SDaniel P. Berrange         return -1;
159dc69004cSMark McLoughlin     }
160dc69004cSMark McLoughlin 
161dc69004cSMark McLoughlin     return ifr.ifr_flags & IFF_VNET_HDR;
162dc69004cSMark McLoughlin }
1631faac1f7SMark McLoughlin 
tap_probe_has_ufo(int fd)1649c282718SMark McLoughlin int tap_probe_has_ufo(int fd)
1659c282718SMark McLoughlin {
1669c282718SMark McLoughlin     unsigned offload;
1679c282718SMark McLoughlin 
1689c282718SMark McLoughlin     offload = TUN_F_CSUM | TUN_F_UFO;
1699c282718SMark McLoughlin 
1709c282718SMark McLoughlin     if (ioctl(fd, TUNSETOFFLOAD, offload) < 0)
1719c282718SMark McLoughlin         return 0;
1729c282718SMark McLoughlin 
1739c282718SMark McLoughlin     return 1;
1749c282718SMark McLoughlin }
1759c282718SMark McLoughlin 
tap_probe_has_uso(int fd)176*f03e0cf6SYuri Benditovich int tap_probe_has_uso(int fd)
177*f03e0cf6SYuri Benditovich {
178*f03e0cf6SYuri Benditovich     unsigned offload;
179*f03e0cf6SYuri Benditovich 
180*f03e0cf6SYuri Benditovich     offload = TUN_F_CSUM | TUN_F_USO4 | TUN_F_USO6;
181*f03e0cf6SYuri Benditovich 
182*f03e0cf6SYuri Benditovich     if (ioctl(fd, TUNSETOFFLOAD, offload) < 0) {
183*f03e0cf6SYuri Benditovich         return 0;
184*f03e0cf6SYuri Benditovich     }
185*f03e0cf6SYuri Benditovich     return 1;
186*f03e0cf6SYuri Benditovich }
187*f03e0cf6SYuri Benditovich 
tap_fd_set_vnet_hdr_len(int fd,int len)188445d892fSMichael S. Tsirkin void tap_fd_set_vnet_hdr_len(int fd, int len)
189445d892fSMichael S. Tsirkin {
190445d892fSMichael S. Tsirkin     if (ioctl(fd, TUNSETVNETHDRSZ, &len) == -1) {
191445d892fSMichael S. Tsirkin         fprintf(stderr, "TUNSETVNETHDRSZ ioctl() failed: %s. Exiting.\n",
192445d892fSMichael S. Tsirkin                 strerror(errno));
19328a65891SJason Wang         abort();
194445d892fSMichael S. Tsirkin     }
195445d892fSMichael S. Tsirkin }
196445d892fSMichael S. Tsirkin 
tap_fd_set_vnet_le(int fd,int is_le)197c80cd6bbSGreg Kurz int tap_fd_set_vnet_le(int fd, int is_le)
198c80cd6bbSGreg Kurz {
199c80cd6bbSGreg Kurz     int arg = is_le ? 1 : 0;
200c80cd6bbSGreg Kurz 
201c80cd6bbSGreg Kurz     if (!ioctl(fd, TUNSETVNETLE, &arg)) {
202c80cd6bbSGreg Kurz         return 0;
203c80cd6bbSGreg Kurz     }
204c80cd6bbSGreg Kurz 
205c80cd6bbSGreg Kurz     /* Check if our kernel supports TUNSETVNETLE */
206c80cd6bbSGreg Kurz     if (errno == EINVAL) {
207c80cd6bbSGreg Kurz         return -errno;
208c80cd6bbSGreg Kurz     }
209c80cd6bbSGreg Kurz 
210594fd211SJohn Snow     error_report("TUNSETVNETLE ioctl() failed: %s.", strerror(errno));
211c80cd6bbSGreg Kurz     abort();
212c80cd6bbSGreg Kurz }
213c80cd6bbSGreg Kurz 
tap_fd_set_vnet_be(int fd,int is_be)214c80cd6bbSGreg Kurz int tap_fd_set_vnet_be(int fd, int is_be)
215c80cd6bbSGreg Kurz {
216c80cd6bbSGreg Kurz     int arg = is_be ? 1 : 0;
217c80cd6bbSGreg Kurz 
218c80cd6bbSGreg Kurz     if (!ioctl(fd, TUNSETVNETBE, &arg)) {
219c80cd6bbSGreg Kurz         return 0;
220c80cd6bbSGreg Kurz     }
221c80cd6bbSGreg Kurz 
222c80cd6bbSGreg Kurz     /* Check if our kernel supports TUNSETVNETBE */
223c80cd6bbSGreg Kurz     if (errno == EINVAL) {
224c80cd6bbSGreg Kurz         return -errno;
225c80cd6bbSGreg Kurz     }
226c80cd6bbSGreg Kurz 
227594fd211SJohn Snow     error_report("TUNSETVNETBE ioctl() failed: %s.", strerror(errno));
228c80cd6bbSGreg Kurz     abort();
229c80cd6bbSGreg Kurz }
230c80cd6bbSGreg Kurz 
tap_fd_set_offload(int fd,int csum,int tso4,int tso6,int ecn,int ufo,int uso4,int uso6)2311faac1f7SMark McLoughlin void tap_fd_set_offload(int fd, int csum, int tso4,
2322ab0ec31SAndrew Melnychenko                         int tso6, int ecn, int ufo, int uso4, int uso6)
2331faac1f7SMark McLoughlin {
2341faac1f7SMark McLoughlin     unsigned int offload = 0;
2351faac1f7SMark McLoughlin 
2362e50326cSPierre Riteau     /* Check if our kernel supports TUNSETOFFLOAD */
2372e50326cSPierre Riteau     if (ioctl(fd, TUNSETOFFLOAD, 0) != 0 && errno == EINVAL) {
2382e50326cSPierre Riteau         return;
2392e50326cSPierre Riteau     }
2402e50326cSPierre Riteau 
2411faac1f7SMark McLoughlin     if (csum) {
2421faac1f7SMark McLoughlin         offload |= TUN_F_CSUM;
2431faac1f7SMark McLoughlin         if (tso4)
2441faac1f7SMark McLoughlin             offload |= TUN_F_TSO4;
2451faac1f7SMark McLoughlin         if (tso6)
2461faac1f7SMark McLoughlin             offload |= TUN_F_TSO6;
2471faac1f7SMark McLoughlin         if ((tso4 || tso6) && ecn)
2481faac1f7SMark McLoughlin             offload |= TUN_F_TSO_ECN;
2491faac1f7SMark McLoughlin         if (ufo)
2501faac1f7SMark McLoughlin             offload |= TUN_F_UFO;
2512ab0ec31SAndrew Melnychenko         if (uso4) {
2522ab0ec31SAndrew Melnychenko             offload |= TUN_F_USO4;
2532ab0ec31SAndrew Melnychenko         }
2542ab0ec31SAndrew Melnychenko         if (uso6) {
2552ab0ec31SAndrew Melnychenko             offload |= TUN_F_USO6;
2562ab0ec31SAndrew Melnychenko         }
2571faac1f7SMark McLoughlin     }
2581faac1f7SMark McLoughlin 
2591faac1f7SMark McLoughlin     if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
2602ab0ec31SAndrew Melnychenko         offload &= ~(TUN_F_USO4 | TUN_F_USO6);
2612ab0ec31SAndrew Melnychenko         if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
2621faac1f7SMark McLoughlin             offload &= ~TUN_F_UFO;
2631faac1f7SMark McLoughlin             if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
2641faac1f7SMark McLoughlin                 fprintf(stderr, "TUNSETOFFLOAD ioctl() failed: %s\n",
2651faac1f7SMark McLoughlin                     strerror(errno));
2661faac1f7SMark McLoughlin             }
2671faac1f7SMark McLoughlin         }
2681faac1f7SMark McLoughlin     }
2692ab0ec31SAndrew Melnychenko }
27094fdc6d0SJason Wang 
27194fdc6d0SJason Wang /* Enable a specific queue of tap. */
tap_fd_enable(int fd)27294fdc6d0SJason Wang int tap_fd_enable(int fd)
27394fdc6d0SJason Wang {
27494fdc6d0SJason Wang     struct ifreq ifr;
27594fdc6d0SJason Wang     int ret;
27694fdc6d0SJason Wang 
27794fdc6d0SJason Wang     memset(&ifr, 0, sizeof(ifr));
27894fdc6d0SJason Wang 
27994fdc6d0SJason Wang     ifr.ifr_flags = IFF_ATTACH_QUEUE;
28094fdc6d0SJason Wang     ret = ioctl(fd, TUNSETQUEUE, (void *) &ifr);
28194fdc6d0SJason Wang 
28294fdc6d0SJason Wang     if (ret != 0) {
28394fdc6d0SJason Wang         error_report("could not enable queue");
28494fdc6d0SJason Wang     }
28594fdc6d0SJason Wang 
28694fdc6d0SJason Wang     return ret;
28794fdc6d0SJason Wang }
28894fdc6d0SJason Wang 
28994fdc6d0SJason Wang /* Disable a specific queue of tap/ */
tap_fd_disable(int fd)29094fdc6d0SJason Wang int tap_fd_disable(int fd)
29194fdc6d0SJason Wang {
29294fdc6d0SJason Wang     struct ifreq ifr;
29394fdc6d0SJason Wang     int ret;
29494fdc6d0SJason Wang 
29594fdc6d0SJason Wang     memset(&ifr, 0, sizeof(ifr));
29694fdc6d0SJason Wang 
29794fdc6d0SJason Wang     ifr.ifr_flags = IFF_DETACH_QUEUE;
29894fdc6d0SJason Wang     ret = ioctl(fd, TUNSETQUEUE, (void *) &ifr);
29994fdc6d0SJason Wang 
30094fdc6d0SJason Wang     if (ret != 0) {
30194fdc6d0SJason Wang         error_report("could not disable queue");
30294fdc6d0SJason Wang     }
30394fdc6d0SJason Wang 
30494fdc6d0SJason Wang     return ret;
30594fdc6d0SJason Wang }
30694fdc6d0SJason Wang 
tap_fd_get_ifname(int fd,char * ifname)307e5dc0b40SJason Wang int tap_fd_get_ifname(int fd, char *ifname)
308e5dc0b40SJason Wang {
309e5dc0b40SJason Wang     struct ifreq ifr;
310e5dc0b40SJason Wang 
311e5dc0b40SJason Wang     if (ioctl(fd, TUNGETIFF, &ifr) != 0) {
312e5dc0b40SJason Wang         error_report("TUNGETIFF ioctl() failed: %s",
313e5dc0b40SJason Wang                      strerror(errno));
314e5dc0b40SJason Wang         return -1;
315e5dc0b40SJason Wang     }
316e5dc0b40SJason Wang 
317e5dc0b40SJason Wang     pstrcpy(ifname, sizeof(ifr.ifr_name), ifr.ifr_name);
318e5dc0b40SJason Wang     return 0;
319e5dc0b40SJason Wang }
3208f364e34SAndrew Melnychenko 
tap_fd_set_steering_ebpf(int fd,int prog_fd)3218f364e34SAndrew Melnychenko int tap_fd_set_steering_ebpf(int fd, int prog_fd)
3228f364e34SAndrew Melnychenko {
3238f364e34SAndrew Melnychenko     if (ioctl(fd, TUNSETSTEERINGEBPF, (void *) &prog_fd) != 0) {
3248f364e34SAndrew Melnychenko         error_report("Issue while setting TUNSETSTEERINGEBPF:"
3258f364e34SAndrew Melnychenko                     " %s with fd: %d, prog_fd: %d",
3268f364e34SAndrew Melnychenko                     strerror(errno), fd, prog_fd);
3278f364e34SAndrew Melnychenko 
3288f364e34SAndrew Melnychenko        return -1;
3298f364e34SAndrew Melnychenko     }
3308f364e34SAndrew Melnychenko 
3318f364e34SAndrew Melnychenko     return 0;
3328f364e34SAndrew Melnychenko }
333