1 /* SPDX-License-Identifier: GPL-2.0-only */ 2 /* 3 * VMware vSockets Driver 4 * 5 * Copyright (C) 2007-2013 VMware, Inc. All rights reserved. 6 */ 7 8 #ifndef __AF_VSOCK_H__ 9 #define __AF_VSOCK_H__ 10 11 #include <linux/kernel.h> 12 #include <linux/workqueue.h> 13 #include <net/netns/vsock.h> 14 #include <net/sock.h> 15 #include <uapi/linux/vm_sockets.h> 16 17 #include "vsock_addr.h" 18 19 #define LAST_RESERVED_PORT 1023 20 21 #define VSOCK_HASH_SIZE 251 22 extern struct list_head vsock_bind_table[VSOCK_HASH_SIZE + 1]; 23 extern struct list_head vsock_connected_table[VSOCK_HASH_SIZE]; 24 extern spinlock_t vsock_table_lock; 25 26 #define vsock_sk(__sk) ((struct vsock_sock *)__sk) 27 #define sk_vsock(__vsk) (&(__vsk)->sk) 28 29 struct vsock_sock { 30 /* sk must be the first member. */ 31 struct sock sk; 32 const struct vsock_transport *transport; 33 struct sockaddr_vm local_addr; 34 struct sockaddr_vm remote_addr; 35 /* Links for the global tables of bound and connected sockets. */ 36 struct list_head bound_table; 37 struct list_head connected_table; 38 /* Accessed without the socket lock held. This means it can never be 39 * modified outsided of socket create or destruct. 40 */ 41 bool trusted; 42 bool cached_peer_allow_dgram; /* Dgram communication allowed to 43 * cached peer? 44 */ 45 u32 cached_peer; /* Context ID of last dgram destination check. */ 46 const struct cred *owner; 47 /* Rest are SOCK_STREAM only. */ 48 long connect_timeout; 49 /* Listening socket that this came from. */ 50 struct sock *listener; 51 /* Used for pending list and accept queue during connection handshake. 52 * The listening socket is the head for both lists. Sockets created 53 * for connection requests are placed in the pending list until they 54 * are connected, at which point they are put in the accept queue list 55 * so they can be accepted in accept(). 56 */ 57 struct list_head pending_links; 58 struct list_head accept_queue; 59 struct delayed_work connect_work; 60 struct delayed_work pending_work; 61 struct delayed_work close_work; 62 bool close_work_scheduled; 63 u32 peer_shutdown; 64 bool sent_request; 65 bool ignore_connecting_rst; 66 67 /* Protected by lock_sock(sk) */ 68 u64 buffer_size; 69 u64 buffer_min_size; 70 u64 buffer_max_size; 71 72 /* Private to transport. */ 73 void *trans; 74 }; 75 76 s64 vsock_connectible_has_data(struct vsock_sock *vsk); 77 s64 vsock_stream_has_data(struct vsock_sock *vsk); 78 s64 vsock_stream_has_space(struct vsock_sock *vsk); 79 struct sock *vsock_create_connected(struct sock *parent); 80 void vsock_data_ready(struct sock *sk); 81 82 /**** TRANSPORT ****/ 83 84 struct vsock_transport_recv_notify_data { 85 u64 data1; /* Transport-defined. */ 86 u64 data2; /* Transport-defined. */ 87 bool notify_on_block; 88 }; 89 90 struct vsock_transport_send_notify_data { 91 u64 data1; /* Transport-defined. */ 92 u64 data2; /* Transport-defined. */ 93 }; 94 95 /* Transport features flags */ 96 /* Transport provides host->guest communication */ 97 #define VSOCK_TRANSPORT_F_H2G 0x00000001 98 /* Transport provides guest->host communication */ 99 #define VSOCK_TRANSPORT_F_G2H 0x00000002 100 /* Transport provides DGRAM communication */ 101 #define VSOCK_TRANSPORT_F_DGRAM 0x00000004 102 /* Transport provides local (loopback) communication */ 103 #define VSOCK_TRANSPORT_F_LOCAL 0x00000008 104 105 struct vsock_transport { 106 struct module *module; 107 108 /* Initialize/tear-down socket. */ 109 int (*init)(struct vsock_sock *, struct vsock_sock *); 110 void (*destruct)(struct vsock_sock *); 111 void (*release)(struct vsock_sock *); 112 113 /* Cancel all pending packets sent on vsock. */ 114 int (*cancel_pkt)(struct vsock_sock *vsk); 115 116 /* Connections. */ 117 int (*connect)(struct vsock_sock *); 118 119 /* DGRAM. */ 120 int (*dgram_bind)(struct vsock_sock *, struct sockaddr_vm *); 121 int (*dgram_dequeue)(struct vsock_sock *vsk, struct msghdr *msg, 122 size_t len, int flags); 123 int (*dgram_enqueue)(struct vsock_sock *, struct sockaddr_vm *, 124 struct msghdr *, size_t len); 125 bool (*dgram_allow)(struct vsock_sock *vsk, u32 cid, u32 port); 126 127 /* STREAM. */ 128 /* TODO: stream_bind() */ 129 ssize_t (*stream_dequeue)(struct vsock_sock *, struct msghdr *, 130 size_t len, int flags); 131 ssize_t (*stream_enqueue)(struct vsock_sock *, struct msghdr *, 132 size_t len); 133 s64 (*stream_has_data)(struct vsock_sock *); 134 s64 (*stream_has_space)(struct vsock_sock *); 135 u64 (*stream_rcvhiwat)(struct vsock_sock *); 136 bool (*stream_is_active)(struct vsock_sock *); 137 bool (*stream_allow)(struct vsock_sock *vsk, u32 cid, u32 port); 138 139 /* SEQ_PACKET. */ 140 ssize_t (*seqpacket_dequeue)(struct vsock_sock *vsk, struct msghdr *msg, 141 int flags); 142 int (*seqpacket_enqueue)(struct vsock_sock *vsk, struct msghdr *msg, 143 size_t len); 144 bool (*seqpacket_allow)(struct vsock_sock *vsk, u32 remote_cid); 145 u32 (*seqpacket_has_data)(struct vsock_sock *vsk); 146 147 /* Notification. */ 148 int (*notify_poll_in)(struct vsock_sock *, size_t, bool *); 149 int (*notify_poll_out)(struct vsock_sock *, size_t, bool *); 150 int (*notify_recv_init)(struct vsock_sock *, size_t, 151 struct vsock_transport_recv_notify_data *); 152 int (*notify_recv_pre_block)(struct vsock_sock *, size_t, 153 struct vsock_transport_recv_notify_data *); 154 int (*notify_recv_pre_dequeue)(struct vsock_sock *, size_t, 155 struct vsock_transport_recv_notify_data *); 156 int (*notify_recv_post_dequeue)(struct vsock_sock *, size_t, 157 ssize_t, bool, struct vsock_transport_recv_notify_data *); 158 int (*notify_send_init)(struct vsock_sock *, 159 struct vsock_transport_send_notify_data *); 160 int (*notify_send_pre_block)(struct vsock_sock *, 161 struct vsock_transport_send_notify_data *); 162 int (*notify_send_pre_enqueue)(struct vsock_sock *, 163 struct vsock_transport_send_notify_data *); 164 int (*notify_send_post_enqueue)(struct vsock_sock *, ssize_t, 165 struct vsock_transport_send_notify_data *); 166 /* sk_lock held by the caller */ 167 void (*notify_buffer_size)(struct vsock_sock *, u64 *); 168 int (*notify_set_rcvlowat)(struct vsock_sock *vsk, int val); 169 170 /* SIOCOUTQ ioctl */ 171 ssize_t (*unsent_bytes)(struct vsock_sock *vsk); 172 173 /* Shutdown. */ 174 int (*shutdown)(struct vsock_sock *, int); 175 176 /* Addressing. */ 177 u32 (*get_local_cid)(void); 178 179 /* Check if this transport serves a specific remote CID. 180 * For H2G transports: return true if the CID belongs to a registered 181 * guest. If not implemented, all CIDs > VMADDR_CID_HOST go to H2G. 182 * For G2H transports: return true if the transport can reach arbitrary 183 * CIDs via the hypervisor (i.e. supports the fallback overlay). VMCI 184 * does not implement this as it only serves CIDs 0 and 2. 185 */ 186 bool (*has_remote_cid)(struct vsock_sock *vsk, u32 remote_cid); 187 188 /* Read a single skb */ 189 int (*read_skb)(struct vsock_sock *, skb_read_actor_t); 190 191 /* Zero-copy. */ 192 bool (*msgzerocopy_allow)(void); 193 }; 194 195 /**** CORE ****/ 196 197 int vsock_core_register(const struct vsock_transport *t, int features); 198 void vsock_core_unregister(const struct vsock_transport *t); 199 200 /* The transport may downcast this to access transport-specific functions */ 201 const struct vsock_transport *vsock_core_get_transport(struct vsock_sock *vsk); 202 203 /**** UTILS ****/ 204 205 /* vsock_table_lock must be held */ 206 static inline bool __vsock_in_bound_table(struct vsock_sock *vsk) 207 { 208 return !list_empty(&vsk->bound_table); 209 } 210 211 /* vsock_table_lock must be held */ 212 static inline bool __vsock_in_connected_table(struct vsock_sock *vsk) 213 { 214 return !list_empty(&vsk->connected_table); 215 } 216 217 void vsock_add_pending(struct sock *listener, struct sock *pending); 218 void vsock_remove_pending(struct sock *listener, struct sock *pending); 219 void vsock_enqueue_accept(struct sock *listener, struct sock *connected); 220 void vsock_pending_to_accept(struct sock *listener, struct sock *pending); 221 void vsock_insert_connected(struct vsock_sock *vsk); 222 void vsock_remove_bound(struct vsock_sock *vsk); 223 void vsock_remove_connected(struct vsock_sock *vsk); 224 struct sock *vsock_find_bound_socket(struct sockaddr_vm *addr); 225 struct sock *vsock_find_connected_socket(struct sockaddr_vm *src, 226 struct sockaddr_vm *dst); 227 struct sock *vsock_find_bound_socket_net(struct sockaddr_vm *addr, 228 struct net *net); 229 struct sock *vsock_find_connected_socket_net(struct sockaddr_vm *src, 230 struct sockaddr_vm *dst, 231 struct net *net); 232 void vsock_remove_sock(struct vsock_sock *vsk); 233 void vsock_for_each_connected_socket(struct vsock_transport *transport, 234 void (*fn)(struct sock *sk)); 235 int vsock_assign_transport(struct vsock_sock *vsk, struct vsock_sock *psk); 236 bool vsock_find_cid(unsigned int cid); 237 void vsock_linger(struct sock *sk); 238 239 /**** TAP ****/ 240 241 struct vsock_tap { 242 struct net_device *dev; 243 struct module *module; 244 struct list_head list; 245 }; 246 247 int vsock_add_tap(struct vsock_tap *vt); 248 int vsock_remove_tap(struct vsock_tap *vt); 249 void vsock_deliver_tap(struct sk_buff *build_skb(void *opaque), void *opaque); 250 int __vsock_connectible_recvmsg(struct socket *sock, struct msghdr *msg, size_t len, 251 int flags); 252 int vsock_connectible_recvmsg(struct socket *sock, struct msghdr *msg, size_t len, 253 int flags); 254 int __vsock_dgram_recvmsg(struct socket *sock, struct msghdr *msg, 255 size_t len, int flags); 256 int vsock_dgram_recvmsg(struct socket *sock, struct msghdr *msg, 257 size_t len, int flags); 258 259 extern struct proto vsock_proto; 260 #ifdef CONFIG_BPF_SYSCALL 261 int vsock_bpf_update_proto(struct sock *sk, struct sk_psock *psock, bool restore); 262 void __init vsock_bpf_build_proto(void); 263 #else 264 static inline void __init vsock_bpf_build_proto(void) 265 {} 266 #endif 267 268 static inline bool vsock_msgzerocopy_allow(const struct vsock_transport *t) 269 { 270 return t->msgzerocopy_allow && t->msgzerocopy_allow(); 271 } 272 273 static inline enum vsock_net_mode vsock_net_mode(struct net *net) 274 { 275 if (!net) 276 return VSOCK_NET_MODE_GLOBAL; 277 278 return READ_ONCE(net->vsock.mode); 279 } 280 281 static inline bool vsock_net_mode_global(struct vsock_sock *vsk) 282 { 283 return vsock_net_mode(sock_net(sk_vsock(vsk))) == VSOCK_NET_MODE_GLOBAL; 284 } 285 286 static inline bool vsock_net_set_child_mode(struct net *net, 287 enum vsock_net_mode mode) 288 { 289 int new_locked = mode + 1; 290 int old_locked = 0; /* unlocked */ 291 292 if (try_cmpxchg(&net->vsock.child_ns_mode_locked, 293 &old_locked, new_locked)) { 294 WRITE_ONCE(net->vsock.child_ns_mode, mode); 295 return true; 296 } 297 298 return old_locked == new_locked; 299 } 300 301 static inline enum vsock_net_mode vsock_net_child_mode(struct net *net) 302 { 303 return READ_ONCE(net->vsock.child_ns_mode); 304 } 305 306 /* Return true if two namespaces pass the mode rules. Otherwise, return false. 307 * 308 * A NULL namespace is treated as VSOCK_NET_MODE_GLOBAL. 309 * 310 * Read more about modes in the comment header of net/vmw_vsock/af_vsock.c. 311 */ 312 static inline bool vsock_net_check_mode(struct net *ns0, struct net *ns1) 313 { 314 enum vsock_net_mode mode0, mode1; 315 316 /* Any vsocks within the same network namespace are always reachable, 317 * regardless of the mode. 318 */ 319 if (net_eq(ns0, ns1)) 320 return true; 321 322 mode0 = vsock_net_mode(ns0); 323 mode1 = vsock_net_mode(ns1); 324 325 /* Different namespaces are only reachable if they are both 326 * global mode. 327 */ 328 return mode0 == VSOCK_NET_MODE_GLOBAL && mode0 == mode1; 329 } 330 #endif /* __AF_VSOCK_H__ */ 331