tap-linux.c 8.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343
  1. /*
  2. * QEMU System Emulator
  3. *
  4. * Copyright (c) 2003-2008 Fabrice Bellard
  5. * Copyright (c) 2009 Red Hat, Inc.
  6. *
  7. * Permission is hereby granted, free of charge, to any person obtaining a copy
  8. * of this software and associated documentation files (the "Software"), to deal
  9. * in the Software without restriction, including without limitation the rights
  10. * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
  11. * copies of the Software, and to permit persons to whom the Software is
  12. * furnished to do so, subject to the following conditions:
  13. *
  14. * The above copyright notice and this permission notice shall be included in
  15. * all copies or substantial portions of the Software.
  16. *
  17. * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
  18. * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
  19. * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
  20. * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
  21. * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
  22. * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
  23. * THE SOFTWARE.
  24. */
  25. #include "qemu/osdep.h"
  26. #include "tap_int.h"
  27. #include "tap-linux.h"
  28. #include "net/tap.h"
  29. #include <net/if.h>
  30. #include <sys/ioctl.h>
  31. #include "qapi/error.h"
  32. #include "qemu/error-report.h"
  33. #include "qemu/cutils.h"
  34. #define PATH_NET_TUN "/dev/net/tun"
  35. int tap_open(char *ifname, int ifname_size, int *vnet_hdr,
  36. int vnet_hdr_required, int mq_required, Error **errp)
  37. {
  38. struct ifreq ifr;
  39. int fd, ret;
  40. int len = sizeof(struct virtio_net_hdr);
  41. unsigned int features;
  42. ret = if_nametoindex(ifname);
  43. if (ret) {
  44. g_autofree char *file = g_strdup_printf("/dev/tap%d", ret);
  45. fd = open(file, O_RDWR);
  46. } else {
  47. fd = -1;
  48. }
  49. if (fd < 0) {
  50. fd = RETRY_ON_EINTR(open(PATH_NET_TUN, O_RDWR));
  51. if (fd < 0) {
  52. error_setg_errno(errp, errno, "could not open %s", PATH_NET_TUN);
  53. return -1;
  54. }
  55. }
  56. memset(&ifr, 0, sizeof(ifr));
  57. ifr.ifr_flags = IFF_TAP | IFF_NO_PI;
  58. if (ioctl(fd, TUNGETFEATURES, &features) == -1) {
  59. warn_report("TUNGETFEATURES failed: %s", strerror(errno));
  60. features = 0;
  61. }
  62. if (features & IFF_ONE_QUEUE) {
  63. ifr.ifr_flags |= IFF_ONE_QUEUE;
  64. }
  65. if (*vnet_hdr) {
  66. if (features & IFF_VNET_HDR) {
  67. *vnet_hdr = 1;
  68. ifr.ifr_flags |= IFF_VNET_HDR;
  69. } else {
  70. *vnet_hdr = 0;
  71. }
  72. if (vnet_hdr_required && !*vnet_hdr) {
  73. error_setg(errp, "vnet_hdr=1 requested, but no kernel "
  74. "support for IFF_VNET_HDR available");
  75. close(fd);
  76. return -1;
  77. }
  78. /*
  79. * Make sure vnet header size has the default value: for a persistent
  80. * tap it might have been modified e.g. by another instance of qemu.
  81. * Ignore errors since old kernels do not support this ioctl: in this
  82. * case the header size implicitly has the correct value.
  83. */
  84. ioctl(fd, TUNSETVNETHDRSZ, &len);
  85. }
  86. if (mq_required) {
  87. if (!(features & IFF_MULTI_QUEUE)) {
  88. error_setg(errp, "multiqueue required, but no kernel "
  89. "support for IFF_MULTI_QUEUE available");
  90. close(fd);
  91. return -1;
  92. } else {
  93. ifr.ifr_flags |= IFF_MULTI_QUEUE;
  94. }
  95. }
  96. if (ifname[0] != '\0')
  97. pstrcpy(ifr.ifr_name, IFNAMSIZ, ifname);
  98. else
  99. pstrcpy(ifr.ifr_name, IFNAMSIZ, "tap%d");
  100. ret = ioctl(fd, TUNSETIFF, (void *) &ifr);
  101. if (ret != 0) {
  102. if (ifname[0] != '\0') {
  103. error_setg_errno(errp, errno, "could not configure %s (%s)",
  104. PATH_NET_TUN, ifr.ifr_name);
  105. } else {
  106. error_setg_errno(errp, errno, "could not configure %s",
  107. PATH_NET_TUN);
  108. }
  109. close(fd);
  110. return -1;
  111. }
  112. pstrcpy(ifname, ifname_size, ifr.ifr_name);
  113. g_unix_set_fd_nonblocking(fd, true, NULL);
  114. return fd;
  115. }
  116. /* sndbuf implements a kind of flow control for tap.
  117. * Unfortunately when it's enabled, and packets are sent
  118. * to other guests on the same host, the receiver
  119. * can lock up the transmitter indefinitely.
  120. *
  121. * To avoid packet loss, sndbuf should be set to a value lower than the tx
  122. * queue capacity of any destination network interface.
  123. * Ethernet NICs generally have txqueuelen=1000, so 1Mb is
  124. * a good value, given a 1500 byte MTU.
  125. */
  126. #define TAP_DEFAULT_SNDBUF 0
  127. void tap_set_sndbuf(int fd, const NetdevTapOptions *tap, Error **errp)
  128. {
  129. int sndbuf;
  130. sndbuf = !tap->has_sndbuf ? TAP_DEFAULT_SNDBUF :
  131. tap->sndbuf > INT_MAX ? INT_MAX :
  132. tap->sndbuf;
  133. if (!sndbuf) {
  134. sndbuf = INT_MAX;
  135. }
  136. if (ioctl(fd, TUNSETSNDBUF, &sndbuf) == -1 && tap->has_sndbuf) {
  137. error_setg_errno(errp, errno, "TUNSETSNDBUF ioctl failed");
  138. }
  139. }
  140. int tap_probe_vnet_hdr(int fd, Error **errp)
  141. {
  142. struct ifreq ifr;
  143. memset(&ifr, 0, sizeof(ifr));
  144. if (ioctl(fd, TUNGETIFF, &ifr) != 0) {
  145. /* TUNGETIFF is available since kernel v2.6.27 */
  146. error_setg_errno(errp, errno,
  147. "Unable to query TUNGETIFF on FD %d", fd);
  148. return -1;
  149. }
  150. return ifr.ifr_flags & IFF_VNET_HDR;
  151. }
  152. int tap_probe_has_ufo(int fd)
  153. {
  154. unsigned offload;
  155. offload = TUN_F_CSUM | TUN_F_UFO;
  156. if (ioctl(fd, TUNSETOFFLOAD, offload) < 0)
  157. return 0;
  158. return 1;
  159. }
  160. int tap_probe_has_uso(int fd)
  161. {
  162. unsigned offload;
  163. offload = TUN_F_CSUM | TUN_F_USO4 | TUN_F_USO6;
  164. if (ioctl(fd, TUNSETOFFLOAD, offload) < 0) {
  165. return 0;
  166. }
  167. return 1;
  168. }
  169. void tap_fd_set_vnet_hdr_len(int fd, int len)
  170. {
  171. if (ioctl(fd, TUNSETVNETHDRSZ, &len) == -1) {
  172. fprintf(stderr, "TUNSETVNETHDRSZ ioctl() failed: %s. Exiting.\n",
  173. strerror(errno));
  174. abort();
  175. }
  176. }
  177. int tap_fd_set_vnet_le(int fd, int is_le)
  178. {
  179. int arg = is_le ? 1 : 0;
  180. if (!ioctl(fd, TUNSETVNETLE, &arg)) {
  181. return 0;
  182. }
  183. /* Check if our kernel supports TUNSETVNETLE */
  184. if (errno == EINVAL) {
  185. return -errno;
  186. }
  187. error_report("TUNSETVNETLE ioctl() failed: %s.", strerror(errno));
  188. abort();
  189. }
  190. int tap_fd_set_vnet_be(int fd, int is_be)
  191. {
  192. int arg = is_be ? 1 : 0;
  193. if (!ioctl(fd, TUNSETVNETBE, &arg)) {
  194. return 0;
  195. }
  196. /* Check if our kernel supports TUNSETVNETBE */
  197. if (errno == EINVAL) {
  198. return -errno;
  199. }
  200. error_report("TUNSETVNETBE ioctl() failed: %s.", strerror(errno));
  201. abort();
  202. }
  203. void tap_fd_set_offload(int fd, int csum, int tso4,
  204. int tso6, int ecn, int ufo, int uso4, int uso6)
  205. {
  206. unsigned int offload = 0;
  207. /* Check if our kernel supports TUNSETOFFLOAD */
  208. if (ioctl(fd, TUNSETOFFLOAD, 0) != 0 && errno == EINVAL) {
  209. return;
  210. }
  211. if (csum) {
  212. offload |= TUN_F_CSUM;
  213. if (tso4)
  214. offload |= TUN_F_TSO4;
  215. if (tso6)
  216. offload |= TUN_F_TSO6;
  217. if ((tso4 || tso6) && ecn)
  218. offload |= TUN_F_TSO_ECN;
  219. if (ufo)
  220. offload |= TUN_F_UFO;
  221. if (uso4) {
  222. offload |= TUN_F_USO4;
  223. }
  224. if (uso6) {
  225. offload |= TUN_F_USO6;
  226. }
  227. }
  228. if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
  229. offload &= ~(TUN_F_USO4 | TUN_F_USO6);
  230. if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
  231. offload &= ~TUN_F_UFO;
  232. if (ioctl(fd, TUNSETOFFLOAD, offload) != 0) {
  233. fprintf(stderr, "TUNSETOFFLOAD ioctl() failed: %s\n",
  234. strerror(errno));
  235. }
  236. }
  237. }
  238. }
  239. /* Enable a specific queue of tap. */
  240. int tap_fd_enable(int fd)
  241. {
  242. struct ifreq ifr;
  243. int ret;
  244. memset(&ifr, 0, sizeof(ifr));
  245. ifr.ifr_flags = IFF_ATTACH_QUEUE;
  246. ret = ioctl(fd, TUNSETQUEUE, (void *) &ifr);
  247. if (ret != 0) {
  248. error_report("could not enable queue");
  249. }
  250. return ret;
  251. }
  252. /* Disable a specific queue of tap/ */
  253. int tap_fd_disable(int fd)
  254. {
  255. struct ifreq ifr;
  256. int ret;
  257. memset(&ifr, 0, sizeof(ifr));
  258. ifr.ifr_flags = IFF_DETACH_QUEUE;
  259. ret = ioctl(fd, TUNSETQUEUE, (void *) &ifr);
  260. if (ret != 0) {
  261. error_report("could not disable queue");
  262. }
  263. return ret;
  264. }
  265. int tap_fd_get_ifname(int fd, char *ifname)
  266. {
  267. struct ifreq ifr;
  268. if (ioctl(fd, TUNGETIFF, &ifr) != 0) {
  269. error_report("TUNGETIFF ioctl() failed: %s",
  270. strerror(errno));
  271. return -1;
  272. }
  273. pstrcpy(ifname, sizeof(ifr.ifr_name), ifr.ifr_name);
  274. return 0;
  275. }
  276. int tap_fd_set_steering_ebpf(int fd, int prog_fd)
  277. {
  278. if (ioctl(fd, TUNSETSTEERINGEBPF, (void *) &prog_fd) != 0) {
  279. error_report("Issue while setting TUNSETSTEERINGEBPF:"
  280. " %s with fd: %d, prog_fd: %d",
  281. strerror(errno), fd, prog_fd);
  282. return -1;
  283. }
  284. return 0;
  285. }