1 /*-
2 * Copyright (c) 2015-2016
3 * Alexander V. Chernikov <melifaro@FreeBSD.org>
4 *
5 * Redistribution and use in source and binary forms, with or without
6 * modification, are permitted provided that the following conditions
7 * are met:
8 * 1. Redistributions of source code must retain the above copyright
9 * notice, this list of conditions and the following disclaimer.
10 * 2. Redistributions in binary form must reproduce the above copyright
11 * notice, this list of conditions and the following disclaimer in the
12 * documentation and/or other materials provided with the distribution.
13 * 3. Neither the name of the University nor the names of its contributors
14 * may be used to endorse or promote products derived from this software
15 * without specific prior written permission.
16 *
17 * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
18 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
19 * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
20 * ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
21 * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
22 * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
23 * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
24 * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
25 * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
26 * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
27 * SUCH DAMAGE.
28 */
29
30 #ifndef _NET_ROUTE_VAR_H_
31 #define _NET_ROUTE_VAR_H_
32
33 #ifndef RNF_NORMAL
34 #include <net/radix.h>
35 #endif
36 #include <sys/ck.h>
37 #include <sys/epoch.h>
38 #include <netinet/in.h> /* struct sockaddr_in */
39 #include <sys/counter.h>
40 #include <net/route/nhop.h>
41
42 struct nh_control;
43 /* Sets prefix-specific nexthop flags (NHF_DEFAULT, RTF/NHF_HOST, RTF_BROADCAST,..) */
44 typedef int rnh_set_nh_pfxflags_f_t(u_int fibnum, const struct sockaddr *addr,
45 const struct sockaddr *mask, struct nhop_object *nh);
46 /* Fills in family-specific details that are not yet set up (mtu, nhop type, ..) */
47 typedef int rnh_augment_nh_f_t(u_int fibnum, struct nhop_object *nh);
48
49 struct rib_head {
50 struct radix_head head;
51 rn_matchaddr_f_t *rnh_matchaddr; /* longest match for sockaddr */
52 rn_addaddr_f_t *rnh_addaddr; /* add based on sockaddr*/
53 rn_deladdr_f_t *rnh_deladdr; /* remove based on sockaddr */
54 rn_lookup_f_t *rnh_lookup; /* exact match for sockaddr */
55 rn_walktree_t *rnh_walktree; /* traverse tree */
56 rn_walktree_from_t *rnh_walktree_from; /* traverse tree below a */
57 rnh_set_nh_pfxflags_f_t *rnh_set_nh_pfxflags; /* hook to alter record prior to insertion */
58 rt_gen_t rnh_gen; /* datapath generation counter */
59 struct radix_node rnh_nodes[3]; /* empty tree for common case */
60 struct rmlock rib_lock; /* config/data path lock */
61 struct radix_mask_head rmhead; /* masks radix head */
62 struct vnet *rib_vnet; /* vnet pointer */
63 int rib_family; /* AF of the rtable */
64 u_int rib_fibnum; /* fib number */
65 struct callout expire_callout; /* Callout for expiring dynamic routes */
66 time_t next_expire; /* Next expire run ts */
67 uint32_t rnh_prefixes; /* Number of prefixes */
68 rt_gen_t rnh_gen_rib; /* fib algo: rib generation counter */
69 bool rib_dying:1, /* rib is detaching */
70 rib_algo_init:1;/* algo init done */
71 struct nh_control *nh_control; /* nexthop subsystem data */
72 rnh_augment_nh_f_t *rnh_augment_nh;/* hook to alter nexthop prior to insertion */
73 CK_STAILQ_HEAD(, rib_subscription) rnh_subscribers;/* notification subscribers */
74 };
75
76 #define RIB_RLOCK_TRACKER struct rm_priotracker _rib_tracker
77 #define RIB_LOCK_INIT(rh) rm_init_flags(&(rh)->rib_lock, "rib head lock", RM_DUPOK)
78 #define RIB_LOCK_DESTROY(rh) rm_destroy(&(rh)->rib_lock)
79 #define RIB_RLOCK(rh) rm_rlock(&(rh)->rib_lock, &_rib_tracker)
80 #define RIB_RUNLOCK(rh) rm_runlock(&(rh)->rib_lock, &_rib_tracker)
81 #define RIB_WLOCK(rh) rm_wlock(&(rh)->rib_lock)
82 #define RIB_WUNLOCK(rh) rm_wunlock(&(rh)->rib_lock)
83 #define RIB_LOCK_ASSERT(rh) rm_assert(&(rh)->rib_lock, RA_LOCKED)
84 #define RIB_WLOCK_ASSERT(rh) rm_assert(&(rh)->rib_lock, RA_WLOCKED)
85
86 /* Constants */
87 #define RIB_MAX_RETRIES 3
88 #define RT_MAXFIBS UINT16_MAX
89 #define RIB_MAX_MPATH_WIDTH 64
90
91 /* Macro for verifying fields in af-specific 'struct route' structures */
92 #define CHK_STRUCT_FIELD_GENERIC(_s1, _f1, _s2, _f2) \
93 _Static_assert(sizeof(((_s1 *)0)->_f1) == sizeof(((_s2 *)0)->_f2), \
94 "Fields " #_f1 " and " #_f2 " size differs"); \
95 _Static_assert(__offsetof(_s1, _f1) == __offsetof(_s2, _f2), \
96 "Fields " #_f1 " and " #_f2 " offset differs");
97
98 #define _CHK_ROUTE_FIELD(_route_new, _field) \
99 CHK_STRUCT_FIELD_GENERIC(struct route, _field, _route_new, _field)
100
101 #define CHK_STRUCT_ROUTE_FIELDS(_route_new) \
102 _CHK_ROUTE_FIELD(_route_new, ro_nh) \
103 _CHK_ROUTE_FIELD(_route_new, ro_lle) \
104 _CHK_ROUTE_FIELD(_route_new, ro_prepend)\
105 _CHK_ROUTE_FIELD(_route_new, ro_plen) \
106 _CHK_ROUTE_FIELD(_route_new, ro_flags) \
107 _CHK_ROUTE_FIELD(_route_new, ro_mtu) \
108 _CHK_ROUTE_FIELD(_route_new, spare)
109
110 #define CHK_STRUCT_ROUTE_COMPAT(_ro_new, _dst_new) \
111 CHK_STRUCT_ROUTE_FIELDS(_ro_new); \
112 _Static_assert(__offsetof(struct route, ro_dst) == __offsetof(_ro_new, _dst_new),\
113 "ro_dst and " #_dst_new " are at different offset")
114
115 static inline void
rib_bump_gen(struct rib_head * rnh)116 rib_bump_gen(struct rib_head *rnh)
117 {
118 #ifdef FIB_ALGO
119 rnh->rnh_gen_rib++;
120 #else
121 rnh->rnh_gen++;
122 #endif
123 }
124
125 struct rib_head *rt_tables_get_rnh(uint32_t table, sa_family_t family);
126 int rt_getifa_fib(struct rt_addrinfo *info, u_int fibnum);
127 struct rib_cmd_info;
128
129 VNET_PCPUSTAT_DECLARE(struct rtstat, rtstat);
130 #define RTSTAT_ADD(name, val) \
131 VNET_PCPUSTAT_ADD(struct rtstat, rtstat, name, (val))
132 #define RTSTAT_INC(name) RTSTAT_ADD(name, 1)
133
134 /*
135 * Convert a 'struct radix_node *' to a 'struct rtentry *'.
136 * The operation can be done safely (in this code) because a
137 * 'struct rtentry' starts with two 'struct radix_node''s, the first
138 * one representing leaf nodes in the routing tree, which is
139 * what the code in radix.c passes us as a 'struct radix_node'.
140 *
141 * But because there are a lot of assumptions in this conversion,
142 * do not cast explicitly, but always use the macro below.
143 */
144 #define RNTORT(p) ((struct rtentry *)(p))
145
146 struct rtentry {
147 struct radix_node rt_nodes[2]; /* tree glue, and other values */
148 /*
149 * XXX struct rtentry must begin with a struct radix_node (or two!)
150 * because the code does some casts of a 'struct radix_node *'
151 * to a 'struct rtentry *'
152 */
153 #define rt_key(r) (*((struct sockaddr **)(&(r)->rt_nodes->rn_key)))
154 #define rt_mask(r) (*((struct sockaddr **)(&(r)->rt_nodes->rn_mask)))
155 #define rt_key_const(r) (*((const struct sockaddr * const *)(&(r)->rt_nodes->rn_key)))
156 #define rt_mask_const(r) (*((const struct sockaddr * const *)(&(r)->rt_nodes->rn_mask)))
157
158 /*
159 * 2 radix_node structurs above consists of 2x6 pointers, leaving
160 * 4 pointers (32 bytes) of the second cache line on amd64.
161 *
162 */
163 struct nhop_object *rt_nhop; /* nexthop data */
164 union {
165 /*
166 * Destination address storage.
167 * sizeof(struct sockaddr_in6) == 28, however
168 * the dataplane-relevant part (e.g. address) lies
169 * at offset 8..24, making the address not crossing
170 * cacheline boundary.
171 */
172 struct sockaddr_in rt_dst4;
173 struct sockaddr_in6 rt_dst6;
174 struct sockaddr rt_dst;
175 char rt_dstb[28];
176 };
177
178 int rte_flags; /* up/down?, host/net */
179 u_long rt_weight; /* absolute weight */
180 struct rtentry *rt_chain; /* pointer to next rtentry to delete */
181 struct epoch_context rt_epoch_ctx; /* net epoch tracker */
182 };
183
184 /*
185 * With the split between the routing entry and the nexthop,
186 * rt_flags has to be split between these 2 entries. As rtentry
187 * mostly contains prefix data and is thought to be generic enough
188 * so one can transparently change the nexthop pointer w/o requiring
189 * any other rtentry changes, most of rt_flags shifts to the particular nexthop.
190 * /
191 *
192 * RTF_UP: rtentry, as an indication that it is linked.
193 * RTF_HOST: rtentry, nhop. The latter indication is needed for the datapath
194 * RTF_DYNAMIC: nhop, to make rtentry generic.
195 * RTF_MODIFIED: nhop, to make rtentry generic. (legacy)
196 * -- "native" path (nhop) properties:
197 * RTF_GATEWAY, RTF_STATIC, RTF_PROTO1, RTF_PROTO2, RTF_PROTO3, RTF_FIXEDMTU,
198 * RTF_PINNED, RTF_REJECT, RTF_BLACKHOLE, RTF_BROADCAST
199 */
200
201 /* rtentry rt flag mask */
202 #define RTE_RT_FLAG_MASK (RTF_UP | RTF_HOST)
203
204 /* route_temporal.c */
205 void tmproutes_update(struct rib_head *rnh, struct rtentry *rt, struct nhop_object *nh);
206 void tmproutes_init(struct rib_head *rh);
207 void tmproutes_destroy(struct rib_head *rh);
208
209 /* route_ctl.c */
210 struct route_nhop_data;
211 int change_route(struct rib_head *rnh, struct rtentry *rt,
212 struct route_nhop_data *rnd, struct rib_cmd_info *rc);
213 int change_route_conditional(struct rib_head *rnh, struct rtentry *rt,
214 struct route_nhop_data *nhd_orig, struct route_nhop_data *nhd_new,
215 struct rib_cmd_info *rc);
216 struct rtentry *lookup_prefix(struct rib_head *rnh,
217 const struct rt_addrinfo *info, struct route_nhop_data *rnd);
218 struct rtentry *lookup_prefix_rt(struct rib_head *rnh, const struct rtentry *rt,
219 struct route_nhop_data *rnd);
220 int rib_copy_route(struct rtentry *rt, const struct route_nhop_data *rnd_src,
221 struct rib_head *rh_dst, struct rib_cmd_info *rc);
222
223 bool nhop_can_multipath(const struct nhop_object *nh);
224 bool match_nhop_gw(const struct nhop_object *nh, const struct sockaddr *gw);
225 int check_info_match_nhop(const struct rt_addrinfo *info,
226 const struct rtentry *rt, const struct nhop_object *nh);
227 bool rib_can_4o6_nhop(void);
228
229 /* route_rtentry.c */
230 void vnet_rtzone_init(void);
231 void vnet_rtzone_destroy(void);
232 void rt_free(struct rtentry *rt);
233 void rt_free_immediate(struct rtentry *rt);
234 struct rtentry *rt_alloc(struct rib_head *rnh, const struct sockaddr *dst,
235 struct sockaddr *netmask);
236
237 /* subscriptions */
238 void rib_init_subscriptions(struct rib_head *rnh);
239 void rib_destroy_subscriptions(struct rib_head *rnh);
240
241 /* route_ifaddrs.c */
242 void rib_copy_kernel_routes(struct rib_head *rh_src, struct rib_head *rh_dst);
243
244 /* Nexhops */
245 void nhops_init(void);
246 int nhops_init_rib(struct rib_head *rh);
247 void nhops_destroy_rib(struct rib_head *rh);
248 void nhop_ref_object(struct nhop_object *nh);
249 int nhop_try_ref_object(struct nhop_object *nh);
250 void nhop_ref_any(struct nhop_object *nh);
251 void nhop_free_any(struct nhop_object *nh);
252 struct nhop_object *nhop_get_nhop_internal(struct rib_head *rnh,
253 struct nhop_object *nh, int *perror);
254
255 bool nhop_check_gateway(int upper_family, int neigh_family);
256
257 int nhop_create_from_info(struct rib_head *rnh, struct rt_addrinfo *info,
258 struct nhop_object **nh_ret);
259 int nhop_create_from_nhop(struct rib_head *rnh, const struct nhop_object *nh_orig,
260 struct rt_addrinfo *info, struct nhop_object **pnh_priv);
261
262 void nhops_update_ifmtu(struct rib_head *rh, struct ifnet *ifp, uint32_t mtu);
263 int nhops_dump_sysctl(struct rib_head *rh, struct sysctl_req *w);
264
265 /* MULTIPATH */
266 #define MPF_MULTIPATH 0x08 /* need to be consistent with NHF_MULTIPATH */
267
268 struct nhgrp_object {
269 uint16_t nhg_flags; /* nexthop group flags */
270 uint8_t nhg_size; /* dataplain group size */
271 uint8_t spare;
272 struct nhop_object *nhops[0]; /* nhops */
273 };
274
275 static inline struct nhop_object *
nhop_select(struct nhop_object * nh,uint32_t flowid)276 nhop_select(struct nhop_object *nh, uint32_t flowid)
277 {
278 struct nhgrp_object *nhg;
279
280 if (NH_IS_NHGRP(nh)) {
281 nhg = (struct nhgrp_object *)nh;
282 nh = nhg->nhops[flowid % nhg->nhg_size];
283 }
284 return (nh);
285 }
286
287
288 struct weightened_nhop;
289
290 /* mpath_ctl.c */
291 int add_route_mpath(struct rib_head *rnh, struct rt_addrinfo *info,
292 struct rtentry *rt, struct route_nhop_data *rnd_add,
293 struct route_nhop_data *rnd_orig, struct rib_cmd_info *rc);
294
295 /* nhgrp.c */
296 int nhgrp_ctl_init(struct nh_control *ctl);
297 void nhgrp_ctl_free(struct nh_control *ctl);
298 void nhgrp_ctl_unlink_all(struct nh_control *ctl);
299
300
301 /* nhgrp_ctl.c */
302 int nhgrp_dump_sysctl(struct rib_head *rh, struct sysctl_req *w);
303
304 int nhgrp_get_filtered_group(struct rib_head *rh, const struct rtentry *rt,
305 const struct nhgrp_object *src, rib_filter_f_t flt_func, void *flt_data,
306 struct route_nhop_data *rnd);
307 int nhgrp_get_addition_group(struct rib_head *rnh,
308 struct route_nhop_data *rnd_orig, struct route_nhop_data *rnd_add,
309 struct route_nhop_data *rnd_new);
310 int nhgrp_get_merge_group(struct rib_head *rnh,
311 struct route_nhop_data *rnd_orig, struct route_nhop_data *rnd_add,
312 struct route_nhop_data *rnd_new);
313 void nhgrp_recompile(struct rib_head *rh);
314
315 void nhgrp_ref_object(struct nhgrp_object *nhg);
316 uint32_t nhgrp_get_idx(const struct nhgrp_object *nhg);
317 void nhgrp_free(struct nhgrp_object *nhg);
318
319 /* rtsock */
320 int rtsock_routemsg(int cmd, struct rtentry *rt, struct nhop_object *nh,
321 int fibnum);
322 int rtsock_routemsg_info(int cmd, struct rt_addrinfo *info, int fibnum);
323 int rtsock_addrmsg(int cmd, struct ifaddr *ifa, int fibnum);
324
325
326 /* lookup_framework.c */
327 void fib_grow_rtables(uint32_t new_num_tables);
328 void fib_setup_family(int family, uint32_t num_tables);
329 void fib_destroy_rib(struct rib_head *rh);
330 void vnet_fib_init(void);
331 void vnet_fib_destroy(void);
332
333 /* Entropy data used for outbound hashing */
334 #define MPATH_ENTROPY_KEY_LEN 40
335 extern uint8_t mpath_entropy_key[MPATH_ENTROPY_KEY_LEN];
336
337 #endif
338