blob: 36c09e24d8a697841563aba1ad6e72111d9d3388 [file] [log] [blame]
Mark McLoughlinc28b1c12009-10-22 17:49:12 +01001/*
2 * QEMU System Emulator
3 *
4 * Copyright (c) 2003-2008 Fabrice Bellard
5 * Copyright (c) 2009 Red Hat, Inc.
6 *
7 * Permission is hereby granted, free of charge, to any person obtaining a copy
8 * of this software and associated documentation files (the "Software"), to deal
9 * in the Software without restriction, including without limitation the rights
10 * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
11 * copies of the Software, and to permit persons to whom the Software is
12 * furnished to do so, subject to the following conditions:
13 *
14 * The above copyright notice and this permission notice shall be included in
15 * all copies or substantial portions of the Software.
16 *
17 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
18 * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
19 * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
20 * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
21 * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
22 * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
23 * THE SOFTWARE.
24 */
25
Paolo Bonzini1422e322012-10-24 08:43:34 +020026#include "tap_int.h"
27#include "tap-linux.h"
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010028#include "net/tap.h"
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010029
30#include <net/if.h>
31#include <sys/ioctl.h>
32
Paolo Bonzini9c17d612012-12-17 18:20:04 +010033#include "sysemu/sysemu.h"
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010034#include "qemu-common.h"
Paolo Bonzini1de7afc2012-12-17 18:20:00 +010035#include "qemu/error-report.h"
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010036
Michael Tokarev91ca60e2010-06-02 14:33:01 -030037#define PATH_NET_TUN "/dev/net/tun"
38
Jason Wang264986e2013-01-30 19:12:34 +080039int tap_open(char *ifname, int ifname_size, int *vnet_hdr,
40 int vnet_hdr_required, int mq_required)
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010041{
42 struct ifreq ifr;
43 int fd, ret;
Michael S. Tsirkin89e6d682012-11-12 09:13:04 +020044 int len = sizeof(struct virtio_net_hdr);
Peter Lievend26e4452013-02-25 10:17:08 +010045 unsigned int features;
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010046
Michael Tokarev91ca60e2010-06-02 14:33:01 -030047 TFR(fd = open(PATH_NET_TUN, O_RDWR));
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010048 if (fd < 0) {
Michael Tokarev91ca60e2010-06-02 14:33:01 -030049 error_report("could not open %s: %m", PATH_NET_TUN);
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010050 return -1;
51 }
52 memset(&ifr, 0, sizeof(ifr));
53 ifr.ifr_flags = IFF_TAP | IFF_NO_PI;
54
Peter Lievend26e4452013-02-25 10:17:08 +010055 if (ioctl(fd, TUNGETFEATURES, &features) == 0 &&
56 features & IFF_ONE_QUEUE) {
57 ifr.ifr_flags |= IFF_ONE_QUEUE;
58 }
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010059
Peter Lievend26e4452013-02-25 10:17:08 +010060 if (*vnet_hdr) {
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010061 if (ioctl(fd, TUNGETFEATURES, &features) == 0 &&
62 features & IFF_VNET_HDR) {
63 *vnet_hdr = 1;
64 ifr.ifr_flags |= IFF_VNET_HDR;
Pierre Riteau6720b352009-11-25 18:49:34 +000065 } else {
66 *vnet_hdr = 0;
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010067 }
68
69 if (vnet_hdr_required && !*vnet_hdr) {
Markus Armbruster1ecda022010-02-18 17:25:24 +010070 error_report("vnet_hdr=1 requested, but no kernel "
71 "support for IFF_VNET_HDR available");
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010072 close(fd);
73 return -1;
74 }
Michael S. Tsirkin89e6d682012-11-12 09:13:04 +020075 /*
76 * Make sure vnet header size has the default value: for a persistent
77 * tap it might have been modified e.g. by another instance of qemu.
78 * Ignore errors since old kernels do not support this ioctl: in this
79 * case the header size implicitly has the correct value.
80 */
81 ioctl(fd, TUNSETVNETHDRSZ, &len);
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010082 }
83
Jason Wang94fdc6d2013-01-30 19:12:31 +080084 if (mq_required) {
Jason Wang94fdc6d2013-01-30 19:12:31 +080085 if ((ioctl(fd, TUNGETFEATURES, &features) != 0) ||
86 !(features & IFF_MULTI_QUEUE)) {
87 error_report("multiqueue required, but no kernel "
88 "support for IFF_MULTI_QUEUE available");
89 close(fd);
90 return -1;
91 } else {
92 ifr.ifr_flags |= IFF_MULTI_QUEUE;
93 }
94 }
95
Mark McLoughlinc28b1c12009-10-22 17:49:12 +010096 if (ifname[0] != '\0')
97 pstrcpy(ifr.ifr_name, IFNAMSIZ, ifname);
98 else
99 pstrcpy(ifr.ifr_name, IFNAMSIZ, "tap%d");
100 ret = ioctl(fd, TUNSETIFF, (void *) &ifr);
101 if (ret != 0) {
Luiz Capitulino93a73202011-10-14 15:05:10 -0300102 if (ifname[0] != '\0') {
103 error_report("could not configure %s (%s): %m", PATH_NET_TUN, ifr.ifr_name);
104 } else {
105 error_report("could not configure %s: %m", PATH_NET_TUN);
106 }
Mark McLoughlinc28b1c12009-10-22 17:49:12 +0100107 close(fd);
108 return -1;
109 }
110 pstrcpy(ifname, ifname_size, ifr.ifr_name);
111 fcntl(fd, F_SETFL, O_NONBLOCK);
112 return fd;
113}
Mark McLoughlin15ac9132009-10-22 17:49:13 +0100114
Michael S. Tsirkinf157ed22011-02-01 14:25:40 +0200115/* sndbuf implements a kind of flow control for tap.
116 * Unfortunately when it's enabled, and packets are sent
117 * to other guests on the same host, the receiver
118 * can lock up the transmitter indefinitely.
119 *
120 * To avoid packet loss, sndbuf should be set to a value lower than the tx
121 * queue capacity of any destination network interface.
Mark McLoughlin15ac9132009-10-22 17:49:13 +0100122 * Ethernet NICs generally have txqueuelen=1000, so 1Mb is
Michael S. Tsirkinf157ed22011-02-01 14:25:40 +0200123 * a good value, given a 1500 byte MTU.
Mark McLoughlin15ac9132009-10-22 17:49:13 +0100124 */
Michael S. Tsirkinf157ed22011-02-01 14:25:40 +0200125#define TAP_DEFAULT_SNDBUF 0
Mark McLoughlin15ac9132009-10-22 17:49:13 +0100126
Laszlo Ersek08c573a2012-07-17 16:17:19 +0200127int tap_set_sndbuf(int fd, const NetdevTapOptions *tap)
Mark McLoughlin15ac9132009-10-22 17:49:13 +0100128{
129 int sndbuf;
130
Laszlo Ersek08c573a2012-07-17 16:17:19 +0200131 sndbuf = !tap->has_sndbuf ? TAP_DEFAULT_SNDBUF :
132 tap->sndbuf > INT_MAX ? INT_MAX :
133 tap->sndbuf;
134
Mark McLoughlin15ac9132009-10-22 17:49:13 +0100135 if (!sndbuf) {
136 sndbuf = INT_MAX;
137 }
138
Laszlo Ersek08c573a2012-07-17 16:17:19 +0200139 if (ioctl(fd, TUNSETSNDBUF, &sndbuf) == -1 && tap->has_sndbuf) {
Markus Armbruster1ecda022010-02-18 17:25:24 +0100140 error_report("TUNSETSNDBUF ioctl failed: %s", strerror(errno));
Mark McLoughlin15ac9132009-10-22 17:49:13 +0100141 return -1;
142 }
143 return 0;
144}
Mark McLoughlindc690042009-10-22 17:49:14 +0100145
146int tap_probe_vnet_hdr(int fd)
147{
148 struct ifreq ifr;
149
150 if (ioctl(fd, TUNGETIFF, &ifr) != 0) {
Markus Armbruster1ecda022010-02-18 17:25:24 +0100151 error_report("TUNGETIFF ioctl() failed: %s", strerror(errno));
Mark McLoughlindc690042009-10-22 17:49:14 +0100152 return 0;
153 }
154
155 return ifr.ifr_flags & IFF_VNET_HDR;
156}
Mark McLoughlin1faac1f2009-10-22 17:49:15 +0100157
Mark McLoughlin9c282712009-10-22 17:49:16 +0100158int tap_probe_has_ufo(int fd)
159{
160 unsigned offload;
161
162 offload = TUN_F_CSUM | TUN_F_UFO;
163
164 if (ioctl(fd, TUNSETOFFLOAD, offload) < 0)
165 return 0;
166
167 return 1;
168}
169
Michael S. Tsirkin445d8922010-07-16 11:16:06 +0300170/* Verify that we can assign given length */
171int tap_probe_vnet_hdr_len(int fd, int len)
172{
173 int orig;
174 if (ioctl(fd, TUNGETVNETHDRSZ, &orig) == -1) {
175 return 0;
176 }
177 if (ioctl(fd, TUNSETVNETHDRSZ, &len) == -1) {
178 return 0;
179 }
180 /* Restore original length: we can't handle failure. */
181 if (ioctl(fd, TUNSETVNETHDRSZ, &orig) == -1) {
182 fprintf(stderr, "TUNGETVNETHDRSZ ioctl() failed: %s. Exiting.\n",
183 strerror(errno));
Jason Wang28a65892013-01-30 19:12:21 +0800184 abort();
Michael S. Tsirkin445d8922010-07-16 11:16:06 +0300185 return -errno;
186 }
187 return 1;
188}
189
190void tap_fd_set_vnet_hdr_len(int fd, int len)
191{
192 if (ioctl(fd, TUNSETVNETHDRSZ, &len) == -1) {
193 fprintf(stderr, "TUNSETVNETHDRSZ ioctl() failed: %s. Exiting.\n",
194 strerror(errno));
Jason Wang28a65892013-01-30 19:12:21 +0800195 abort();
Michael S. Tsirkin445d8922010-07-16 11:16:06 +0300196 }
197}
198
Mark McLoughlin1faac1f2009-10-22 17:49:15 +0100199void tap_fd_set_offload(int fd, int csum, int tso4,
200 int tso6, int ecn, int ufo)
201{
202 unsigned int offload = 0;
203
Pierre Riteau2e503262009-11-25 18:49:35 +0000204 /* Check if our kernel supports TUNSETOFFLOAD */
205 if (ioctl(fd, TUNSETOFFLOAD, 0) != 0 && errno == EINVAL) {
206 return;
207 }
208
Mark McLoughlin1faac1f2009-10-22 17:49:15 +0100209 if (csum) {
210 offload |= TUN_F_CSUM;
211 if (tso4)
212 offload |= TUN_F_TSO4;
213 if (tso6)
214 offload |= TUN_F_TSO6;
215 if ((tso4 || tso6) && ecn)
216 offload |= TUN_F_TSO_ECN;
217 if (ufo)
218 offload |= TUN_F_UFO;
219 }
220
221 if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
222 offload &= ~TUN_F_UFO;
223 if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
224 fprintf(stderr, "TUNSETOFFLOAD ioctl() failed: %s\n",
225 strerror(errno));
226 }
227 }
228}
Jason Wang94fdc6d2013-01-30 19:12:31 +0800229
230/* Enable a specific queue of tap. */
231int tap_fd_enable(int fd)
232{
233 struct ifreq ifr;
234 int ret;
235
236 memset(&ifr, 0, sizeof(ifr));
237
238 ifr.ifr_flags = IFF_ATTACH_QUEUE;
239 ret = ioctl(fd, TUNSETQUEUE, (void *) &ifr);
240
241 if (ret != 0) {
242 error_report("could not enable queue");
243 }
244
245 return ret;
246}
247
248/* Disable a specific queue of tap/ */
249int tap_fd_disable(int fd)
250{
251 struct ifreq ifr;
252 int ret;
253
254 memset(&ifr, 0, sizeof(ifr));
255
256 ifr.ifr_flags = IFF_DETACH_QUEUE;
257 ret = ioctl(fd, TUNSETQUEUE, (void *) &ifr);
258
259 if (ret != 0) {
260 error_report("could not disable queue");
261 }
262
263 return ret;
264}
265
Jason Wange5dc0b42013-01-30 19:12:33 +0800266int tap_fd_get_ifname(int fd, char *ifname)
267{
268 struct ifreq ifr;
269
270 if (ioctl(fd, TUNGETIFF, &ifr) != 0) {
271 error_report("TUNGETIFF ioctl() failed: %s",
272 strerror(errno));
273 return -1;
274 }
275
276 pstrcpy(ifname, sizeof(ifr.ifr_name), ifr.ifr_name);
277 return 0;
278}