xref: /linux/net/netfilter/nf_flow_table_core.c (revision 7c549fb7eecd01fdba7c0d12c01da876ef8a3214)
1 // SPDX-License-Identifier: GPL-2.0-only
2 #include <linux/kernel.h>
3 #include <linux/init.h>
4 #include <linux/module.h>
5 #include <linux/netfilter.h>
6 #include <linux/rhashtable.h>
7 #include <linux/netdevice.h>
8 #include <net/ip.h>
9 #include <net/ip6_route.h>
10 #include <net/netfilter/nf_tables.h>
11 #include <net/netfilter/nf_flow_table.h>
12 #include <net/netfilter/nf_conntrack.h>
13 #include <net/netfilter/nf_conntrack_core.h>
14 #include <net/netfilter/nf_conntrack_l4proto.h>
15 #include <net/netfilter/nf_conntrack_tuple.h>
16 
17 static DEFINE_MUTEX(flowtable_lock);
18 static LIST_HEAD(flowtables);
19 static __read_mostly struct kmem_cache *flow_offload_cachep;
20 
21 static void
22 flow_offload_fill_dir(struct flow_offload *flow,
23 		      enum flow_offload_tuple_dir dir)
24 {
25 	struct flow_offload_tuple *ft = &flow->tuplehash[dir].tuple;
26 	struct nf_conntrack_tuple *ctt = &flow->ct->tuplehash[dir].tuple;
27 
28 	ft->dir = dir;
29 
30 	switch (ctt->src.l3num) {
31 	case NFPROTO_IPV4:
32 		ft->src_v4 = ctt->src.u3.in;
33 		ft->dst_v4 = ctt->dst.u3.in;
34 		break;
35 	case NFPROTO_IPV6:
36 		ft->src_v6 = ctt->src.u3.in6;
37 		ft->dst_v6 = ctt->dst.u3.in6;
38 		break;
39 	}
40 
41 	ft->l3proto = ctt->src.l3num;
42 	ft->l4proto = ctt->dst.protonum;
43 
44 	switch (ctt->dst.protonum) {
45 	case IPPROTO_TCP:
46 	case IPPROTO_UDP:
47 		ft->src_port = ctt->src.u.tcp.port;
48 		ft->dst_port = ctt->dst.u.tcp.port;
49 		break;
50 	}
51 }
52 
53 struct flow_offload *flow_offload_alloc(struct nf_conn *ct)
54 {
55 	struct flow_offload *flow;
56 
57 	if (unlikely(nf_ct_is_dying(ct)))
58 		return NULL;
59 
60 	flow = kmem_cache_zalloc(flow_offload_cachep, GFP_ATOMIC);
61 	if (!flow)
62 		return NULL;
63 
64 	refcount_inc(&ct->ct_general.use);
65 	flow->ct = ct;
66 
67 	flow_offload_fill_dir(flow, FLOW_OFFLOAD_DIR_ORIGINAL);
68 	flow_offload_fill_dir(flow, FLOW_OFFLOAD_DIR_REPLY);
69 
70 	if (ct->status & IPS_SRC_NAT)
71 		__set_bit(NF_FLOW_SNAT, &flow->flags);
72 	if (ct->status & IPS_DST_NAT)
73 		__set_bit(NF_FLOW_DNAT, &flow->flags);
74 
75 	return flow;
76 }
77 EXPORT_SYMBOL_GPL(flow_offload_alloc);
78 
79 static u32 flow_offload_dst_cookie(struct flow_offload_tuple *flow_tuple)
80 {
81 	if (flow_tuple->l3proto == NFPROTO_IPV6)
82 		return rt6_get_cookie(dst_rt6_info(flow_tuple->dst_cache));
83 
84 	return 0;
85 }
86 
87 static struct dst_entry *nft_route_dst_fetch(struct nf_flow_route *route,
88 					     enum flow_offload_tuple_dir dir)
89 {
90 	struct dst_entry *dst = route->tuple[dir].dst;
91 
92 	route->tuple[dir].dst = NULL;
93 
94 	return dst;
95 }
96 
97 static int flow_offload_fill_route(struct flow_offload *flow,
98 				   struct nf_flow_route *route,
99 				   enum flow_offload_tuple_dir dir)
100 {
101 	struct flow_offload_tuple *flow_tuple = &flow->tuplehash[dir].tuple;
102 	struct dst_entry *dst = nft_route_dst_fetch(route, dir);
103 	int i, j = 0;
104 
105 	switch (flow_tuple->l3proto) {
106 	case NFPROTO_IPV4:
107 		flow_tuple->mtu = ip_dst_mtu_maybe_forward(dst, true);
108 		break;
109 	case NFPROTO_IPV6:
110 		flow_tuple->mtu = ip6_dst_mtu_maybe_forward(dst, true);
111 		break;
112 	}
113 
114 	flow_tuple->iifidx = route->tuple[dir].in.ifindex;
115 	for (i = route->tuple[dir].in.num_encaps - 1; i >= 0; i--) {
116 		flow_tuple->encap[j].id = route->tuple[dir].in.encap[i].id;
117 		flow_tuple->encap[j].proto = route->tuple[dir].in.encap[i].proto;
118 		if (route->tuple[dir].in.ingress_vlans & BIT(i))
119 			flow_tuple->in_vlan_ingress |= BIT(j);
120 		j++;
121 	}
122 
123 	flow_tuple->tun = route->tuple[dir].in.tun;
124 	flow_tuple->encap_num = route->tuple[dir].in.num_encaps;
125 	flow_tuple->needs_gso_segment = route->tuple[dir].out.needs_gso_segment;
126 	flow_tuple->tun_num = route->tuple[dir].in.num_tuns;
127 
128 	switch (route->tuple[dir].xmit_type) {
129 	case FLOW_OFFLOAD_XMIT_DIRECT:
130 		if (route->tuple[!dir].in.num_tuns) {
131 			flow_tuple->dst_cache = dst;
132 			flow_tuple->dst_cookie =
133 				flow_offload_dst_cookie(flow_tuple);
134 		} else {
135 			dst_release(dst);
136 		}
137 		memcpy(flow_tuple->out.h_dest, route->tuple[dir].out.h_dest,
138 		       ETH_ALEN);
139 		memcpy(flow_tuple->out.h_source, route->tuple[dir].out.h_source,
140 		       ETH_ALEN);
141 		flow_tuple->out.ifidx = route->tuple[dir].out.ifindex;
142 		break;
143 	case FLOW_OFFLOAD_XMIT_XFRM:
144 	case FLOW_OFFLOAD_XMIT_NEIGH:
145 		flow_tuple->ifidx = route->tuple[dir].out.ifindex;
146 		flow_tuple->dst_cache = dst;
147 		flow_tuple->dst_cookie = flow_offload_dst_cookie(flow_tuple);
148 		break;
149 	default:
150 		WARN_ON_ONCE(1);
151 		break;
152 	}
153 	flow_tuple->xmit_type = route->tuple[dir].xmit_type;
154 
155 	return 0;
156 }
157 
158 static void nft_flow_dst_release(struct flow_offload *flow,
159 				 enum flow_offload_tuple_dir dir)
160 {
161 	dst_release(flow->tuplehash[dir].tuple.dst_cache);
162 }
163 
164 void flow_offload_route_init(struct flow_offload *flow,
165 			     struct nf_flow_route *route)
166 {
167 	flow_offload_fill_route(flow, route, FLOW_OFFLOAD_DIR_ORIGINAL);
168 	flow_offload_fill_route(flow, route, FLOW_OFFLOAD_DIR_REPLY);
169 	flow->type = NF_FLOW_OFFLOAD_ROUTE;
170 }
171 EXPORT_SYMBOL_GPL(flow_offload_route_init);
172 
173 static inline bool nf_flow_has_expired(const struct flow_offload *flow)
174 {
175 	return nf_flow_timeout_delta(flow->timeout) <= 0;
176 }
177 
178 static void flow_offload_fixup_tcp(struct nf_conn *ct, u8 tcp_state)
179 {
180 	struct ip_ct_tcp *tcp = &ct->proto.tcp;
181 
182 	spin_lock_bh(&ct->lock);
183 	if (tcp->state != tcp_state)
184 		tcp->state = tcp_state;
185 
186 	/* syn packet triggers the TCP reopen case from conntrack. */
187 	if (tcp->state == TCP_CONNTRACK_CLOSE)
188 		ct->proto.tcp.seen[0].flags |= IP_CT_TCP_FLAG_CLOSE_INIT;
189 
190 	/* Conntrack state is outdated due to offload bypass.
191 	 * Clear IP_CT_TCP_FLAG_MAXACK_SET, otherwise conntracks
192 	 * TCP reset validation will fail.
193 	 */
194 	tcp->seen[0].td_maxwin = 0;
195 	tcp->seen[0].flags &= ~IP_CT_TCP_FLAG_MAXACK_SET;
196 	tcp->seen[1].td_maxwin = 0;
197 	tcp->seen[1].flags &= ~IP_CT_TCP_FLAG_MAXACK_SET;
198 	spin_unlock_bh(&ct->lock);
199 }
200 
201 static void flow_offload_fixup_ct(struct flow_offload *flow)
202 {
203 	struct nf_conn *ct = flow->ct;
204 	struct net *net = nf_ct_net(ct);
205 	int l4num = nf_ct_protonum(ct);
206 	bool expired, closing = false;
207 	u32 offload_timeout = 0;
208 	s32 timeout;
209 
210 	if (l4num == IPPROTO_TCP) {
211 		const struct nf_tcp_net *tn = nf_tcp_pernet(net);
212 		u8 tcp_state;
213 
214 		/* Enter CLOSE state if fin/rst packet has been seen, this
215 		 * allows TCP reopen from conntrack. Otherwise, pick up from
216 		 * the last seen TCP state.
217 		 */
218 		closing = test_bit(NF_FLOW_CLOSING, &flow->flags);
219 		if (closing) {
220 			flow_offload_fixup_tcp(ct, TCP_CONNTRACK_CLOSE);
221 			timeout = READ_ONCE(tn->timeouts[TCP_CONNTRACK_CLOSE]);
222 			expired = false;
223 		} else {
224 			tcp_state = READ_ONCE(ct->proto.tcp.state);
225 			flow_offload_fixup_tcp(ct, tcp_state);
226 			timeout = READ_ONCE(tn->timeouts[tcp_state]);
227 			expired = nf_flow_has_expired(flow);
228 		}
229 		offload_timeout = READ_ONCE(tn->offload_timeout);
230 
231 	} else if (l4num == IPPROTO_UDP) {
232 		const struct nf_udp_net *tn = nf_udp_pernet(net);
233 		enum udp_conntrack state =
234 			test_bit(IPS_SEEN_REPLY_BIT, &ct->status) ?
235 			UDP_CT_REPLIED : UDP_CT_UNREPLIED;
236 
237 		timeout = READ_ONCE(tn->timeouts[state]);
238 		expired = nf_flow_has_expired(flow);
239 		offload_timeout = READ_ONCE(tn->offload_timeout);
240 	} else {
241 		return;
242 	}
243 
244 	if (expired)
245 		timeout -= offload_timeout;
246 
247 	if (timeout < 0)
248 		timeout = 0;
249 
250 	if (closing ||
251 	    nf_flow_timeout_delta(READ_ONCE(ct->timeout)) > (__s32)timeout)
252 		nf_ct_refresh(ct, timeout);
253 }
254 
255 static void flow_offload_route_release(struct flow_offload *flow)
256 {
257 	nft_flow_dst_release(flow, FLOW_OFFLOAD_DIR_ORIGINAL);
258 	nft_flow_dst_release(flow, FLOW_OFFLOAD_DIR_REPLY);
259 }
260 
261 static void flow_offload_free_rcu(struct rcu_head *rcu_head)
262 {
263 	struct flow_offload *flow = container_of(rcu_head, struct flow_offload, rcu_head);
264 
265 	nf_ct_put(flow->ct);
266 	kfree(flow);
267 }
268 
269 void flow_offload_free(struct flow_offload *flow)
270 {
271 	switch (flow->type) {
272 	case NF_FLOW_OFFLOAD_ROUTE:
273 		flow_offload_route_release(flow);
274 		break;
275 	default:
276 		break;
277 	}
278 	call_rcu(&flow->rcu_head, flow_offload_free_rcu);
279 }
280 EXPORT_SYMBOL_GPL(flow_offload_free);
281 
282 static u32 flow_offload_hash(const void *data, u32 len, u32 seed)
283 {
284 	const struct flow_offload_tuple *tuple = data;
285 
286 	return jhash(tuple, offsetof(struct flow_offload_tuple, __hash), seed);
287 }
288 
289 static u32 flow_offload_hash_obj(const void *data, u32 len, u32 seed)
290 {
291 	const struct flow_offload_tuple_rhash *tuplehash = data;
292 
293 	return jhash(&tuplehash->tuple, offsetof(struct flow_offload_tuple, __hash), seed);
294 }
295 
296 static int flow_offload_hash_cmp(struct rhashtable_compare_arg *arg,
297 					const void *ptr)
298 {
299 	const struct flow_offload_tuple *tuple = arg->key;
300 	const struct flow_offload_tuple_rhash *x = ptr;
301 
302 	if (memcmp(&x->tuple, tuple, offsetof(struct flow_offload_tuple, __hash)))
303 		return 1;
304 
305 	return 0;
306 }
307 
308 static const struct rhashtable_params nf_flow_offload_rhash_params = {
309 	.head_offset		= offsetof(struct flow_offload_tuple_rhash, node),
310 	.hashfn			= flow_offload_hash,
311 	.obj_hashfn		= flow_offload_hash_obj,
312 	.obj_cmpfn		= flow_offload_hash_cmp,
313 	.automatic_shrinking	= true,
314 };
315 
316 unsigned long flow_offload_get_timeout(struct flow_offload *flow)
317 {
318 	unsigned long timeout = NF_FLOW_TIMEOUT;
319 	struct net *net = nf_ct_net(flow->ct);
320 	int l4num = nf_ct_protonum(flow->ct);
321 
322 	if (l4num == IPPROTO_TCP) {
323 		struct nf_tcp_net *tn = nf_tcp_pernet(net);
324 
325 		timeout = tn->offload_timeout;
326 	} else if (l4num == IPPROTO_UDP) {
327 		struct nf_udp_net *tn = nf_udp_pernet(net);
328 
329 		timeout = tn->offload_timeout;
330 	}
331 
332 	return timeout;
333 }
334 
335 int flow_offload_add(struct nf_flowtable *flow_table, struct flow_offload *flow)
336 {
337 	int err;
338 
339 	flow->timeout = nf_flowtable_time_stamp + flow_offload_get_timeout(flow);
340 
341 	err = rhashtable_insert_fast(&flow_table->rhashtable,
342 				     &flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].node,
343 				     nf_flow_offload_rhash_params);
344 	if (err < 0)
345 		return err;
346 
347 	/* GC only iterates original-direction entries; publish original last. */
348 	err = rhashtable_insert_fast(&flow_table->rhashtable,
349 				     &flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].node,
350 				     nf_flow_offload_rhash_params);
351 	if (err < 0) {
352 		rhashtable_remove_fast(&flow_table->rhashtable,
353 				       &flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].node,
354 				       nf_flow_offload_rhash_params);
355 		return err;
356 	}
357 
358 	nf_ct_refresh(flow->ct, NF_CT_DAY);
359 
360 	if (nf_flowtable_hw_offload(flow_table))
361 		nf_flow_offload_add(flow_table, flow);
362 
363 	return 0;
364 }
365 EXPORT_SYMBOL_GPL(flow_offload_add);
366 
367 void flow_offload_refresh(struct nf_flowtable *flow_table,
368 			  struct flow_offload *flow, bool force)
369 {
370 	u32 timeout;
371 
372 	timeout = nf_flowtable_time_stamp + flow_offload_get_timeout(flow);
373 	if (force || timeout - READ_ONCE(flow->timeout) > HZ)
374 		WRITE_ONCE(flow->timeout, timeout);
375 	else
376 		return;
377 
378 	if (likely(!nf_flowtable_hw_offload(flow_table)) ||
379 	    test_bit(NF_FLOW_CLOSING, &flow->flags))
380 		return;
381 
382 	if (test_bit(NF_FLOW_HW, &flow->flags))
383 		nf_flow_offload_refresh(flow_table, flow);
384 }
385 EXPORT_SYMBOL_GPL(flow_offload_refresh);
386 
387 static void flow_offload_del(struct nf_flowtable *flow_table,
388 			     struct flow_offload *flow)
389 {
390 	rhashtable_remove_fast(&flow_table->rhashtable,
391 			       &flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].node,
392 			       nf_flow_offload_rhash_params);
393 	rhashtable_remove_fast(&flow_table->rhashtable,
394 			       &flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].node,
395 			       nf_flow_offload_rhash_params);
396 	flow_offload_free(flow);
397 }
398 
399 void flow_offload_teardown(struct flow_offload *flow)
400 {
401 	clear_bit(IPS_OFFLOAD_BIT, &flow->ct->status);
402 	if (!test_and_set_bit(NF_FLOW_TEARDOWN, &flow->flags))
403 		flow_offload_fixup_ct(flow);
404 }
405 EXPORT_SYMBOL_GPL(flow_offload_teardown);
406 
407 struct flow_offload_tuple_rhash *
408 flow_offload_lookup(struct nf_flowtable *flow_table,
409 		    struct flow_offload_tuple *tuple)
410 {
411 	struct flow_offload_tuple_rhash *tuplehash;
412 	struct flow_offload *flow;
413 	int dir;
414 
415 	tuplehash = rhashtable_lookup(&flow_table->rhashtable, tuple,
416 				      nf_flow_offload_rhash_params);
417 	if (!tuplehash)
418 		return NULL;
419 
420 	dir = tuplehash->tuple.dir;
421 	flow = container_of(tuplehash, struct flow_offload, tuplehash[dir]);
422 	if (test_bit(NF_FLOW_TEARDOWN, &flow->flags))
423 		return NULL;
424 
425 	if (unlikely(nf_ct_is_dying(flow->ct)))
426 		return NULL;
427 
428 	return tuplehash;
429 }
430 EXPORT_SYMBOL_GPL(flow_offload_lookup);
431 
432 static int
433 nf_flow_table_iterate(struct nf_flowtable *flow_table,
434 		      void (*iter)(struct nf_flowtable *flowtable,
435 				   struct flow_offload *flow, void *data),
436 		      void *data)
437 {
438 	struct flow_offload_tuple_rhash *tuplehash;
439 	struct rhashtable_iter hti;
440 	struct flow_offload *flow;
441 	int err = 0;
442 
443 	rhashtable_walk_enter(&flow_table->rhashtable, &hti);
444 	rhashtable_walk_start(&hti);
445 
446 	while ((tuplehash = rhashtable_walk_next(&hti))) {
447 		if (IS_ERR(tuplehash)) {
448 			if (PTR_ERR(tuplehash) != -EAGAIN) {
449 				err = PTR_ERR(tuplehash);
450 				break;
451 			}
452 			continue;
453 		}
454 		if (tuplehash->tuple.dir)
455 			continue;
456 
457 		flow = container_of(tuplehash, struct flow_offload, tuplehash[0]);
458 
459 		iter(flow_table, flow, data);
460 	}
461 	rhashtable_walk_stop(&hti);
462 	rhashtable_walk_exit(&hti);
463 
464 	return err;
465 }
466 
467 static bool nf_flow_custom_gc(struct nf_flowtable *flow_table,
468 			      const struct flow_offload *flow)
469 {
470 	return flow_table->type->gc && flow_table->type->gc(flow);
471 }
472 
473 /**
474  * nf_flow_table_tcp_timeout() - new timeout of offloaded tcp entry
475  * @ct:		Flowtable offloaded tcp ct
476  *
477  * Return: number of seconds when ct entry should expire.
478  */
479 static u32 nf_flow_table_tcp_timeout(const struct nf_conn *ct)
480 {
481 	u8 state = READ_ONCE(ct->proto.tcp.state);
482 
483 	switch (state) {
484 	case TCP_CONNTRACK_SYN_SENT:
485 	case TCP_CONNTRACK_SYN_RECV:
486 		return 0;
487 	case TCP_CONNTRACK_ESTABLISHED:
488 		return NF_CT_DAY;
489 	case TCP_CONNTRACK_FIN_WAIT:
490 	case TCP_CONNTRACK_CLOSE_WAIT:
491 	case TCP_CONNTRACK_LAST_ACK:
492 	case TCP_CONNTRACK_TIME_WAIT:
493 		return 5 * 60 * HZ;
494 	case TCP_CONNTRACK_CLOSE:
495 		return 0;
496 	}
497 
498 	return 0;
499 }
500 
501 /**
502  * nf_flow_table_extend_ct_timeout() - Extend ct timeout of offloaded conntrack entry
503  * @ct:		Flowtable offloaded ct
504  *
505  * Datapath lookups in the conntrack table will evict nf_conn entries
506  * if they have expired.
507  *
508  * Once nf_conn entries have been offloaded, nf_conntrack might not see any
509  * packets anymore.  Thus ct->timeout is no longer refreshed and ct can
510  * be evicted.
511  *
512  * To avoid the need for an additional check on the offload bit for every
513  * packet processed via nf_conntrack_in(), set an arbitrary timeout large
514  * enough not to ever expire, this save us a check for the IPS_OFFLOAD_BIT
515  * from the packet path via nf_ct_is_expired().
516  */
517 static void nf_flow_table_extend_ct_timeout(struct nf_conn *ct)
518 {
519 	static const s32 min_timeout = 5 * 60 * HZ;
520 	u32 ct_timeout = READ_ONCE(ct->timeout);
521 	s32 expires;
522 
523 	expires = ct_timeout - nfct_time_stamp;
524 	if (expires <= 0) /* already expired */
525 		return;
526 
527 	/* normal case: large enough timeout, nothing to do. */
528 	if (likely(expires >= min_timeout))
529 		return;
530 
531 	/* must check offload bit after this, we do not hold any locks.
532 	 * flowtable and ct entries could have been removed on another CPU.
533 	 */
534 	if (!refcount_inc_not_zero(&ct->ct_general.use))
535 		return;
536 
537 	/* load ct->status after refcount increase */
538 	smp_acquire__after_ctrl_dep();
539 
540 	if (nf_ct_is_confirmed(ct) &&
541 	    test_bit(IPS_OFFLOAD_BIT, &ct->status)) {
542 		u8 l4proto = nf_ct_protonum(ct);
543 		u32 new_timeout = 1;
544 
545 		switch (l4proto) {
546 		case IPPROTO_UDP:
547 			new_timeout = NF_CT_DAY;
548 			break;
549 		case IPPROTO_TCP:
550 			new_timeout = nf_flow_table_tcp_timeout(ct);
551 			break;
552 		default:
553 			WARN_ON_ONCE(1);
554 			break;
555 		}
556 
557 		/* Update to ct->timeout from nf_conntrack happens
558 		 * without holding ct->lock.
559 		 *
560 		 * Use cmpxchg to ensure timeout extension doesn't
561 		 * happen when we race with conntrack datapath.
562 		 *
563 		 * The inverse -- datapath updating ->timeout right
564 		 * after this -- is fine, datapath is authoritative.
565 		 */
566 		if (new_timeout) {
567 			new_timeout += nfct_time_stamp;
568 			cmpxchg(&ct->timeout, ct_timeout, new_timeout);
569 		}
570 	}
571 
572 	nf_ct_put(ct);
573 }
574 
575 static void nf_flow_offload_gc_step(struct nf_flowtable *flow_table,
576 				    struct flow_offload *flow, void *data)
577 {
578 	bool teardown;
579 
580 	if (test_bit(NF_FLOW_PENDING, &flow->flags))
581 		return;
582 
583 	teardown = test_bit(NF_FLOW_TEARDOWN, &flow->flags);
584 
585 	if (nf_flow_has_expired(flow) ||
586 	    nf_ct_is_dying(flow->ct) ||
587 	    !nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple) ||
588 	    !nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple) ||
589 	    nf_flow_custom_gc(flow_table, flow)) {
590 		flow_offload_teardown(flow);
591 		teardown = true;
592 	} else if (!teardown) {
593 		nf_flow_table_extend_ct_timeout(flow->ct);
594 	}
595 
596 	if (teardown) {
597 		if (test_bit(NF_FLOW_HW, &flow->flags)) {
598 			if (!test_bit(NF_FLOW_HW_DYING, &flow->flags))
599 				nf_flow_offload_del(flow_table, flow);
600 			else if (test_bit(NF_FLOW_HW_DEAD, &flow->flags))
601 				flow_offload_del(flow_table, flow);
602 		} else {
603 			flow_offload_del(flow_table, flow);
604 		}
605 	} else if (test_bit(NF_FLOW_CLOSING, &flow->flags) &&
606 		   test_bit(NF_FLOW_HW, &flow->flags) &&
607 		   !test_bit(NF_FLOW_HW_DYING, &flow->flags)) {
608 		nf_flow_offload_del(flow_table, flow);
609 	} else if (test_bit(NF_FLOW_HW, &flow->flags)) {
610 		nf_flow_offload_stats(flow_table, flow);
611 	}
612 }
613 
614 void nf_flow_table_gc_run(struct nf_flowtable *flow_table)
615 {
616 	nf_flow_table_iterate(flow_table, nf_flow_offload_gc_step, NULL);
617 }
618 
619 static void nf_flow_offload_work_gc(struct work_struct *work)
620 {
621 	struct nf_flowtable *flow_table;
622 
623 	flow_table = container_of(work, struct nf_flowtable, gc_work.work);
624 	nf_flow_table_gc_run(flow_table);
625 	queue_delayed_work(system_power_efficient_wq, &flow_table->gc_work, HZ);
626 }
627 
628 static void nf_flow_nat_port_tcp(struct sk_buff *skb, unsigned int thoff,
629 				 __be16 port, __be16 new_port)
630 {
631 	struct tcphdr *tcph;
632 
633 	tcph = (void *)(skb_network_header(skb) + thoff);
634 	inet_proto_csum_replace2(&tcph->check, skb, port, new_port, false);
635 }
636 
637 static void nf_flow_nat_port_udp(struct sk_buff *skb, unsigned int thoff,
638 				 __be16 port, __be16 new_port)
639 {
640 	struct udphdr *udph;
641 
642 	udph = (void *)(skb_network_header(skb) + thoff);
643 	if (udph->check || skb->ip_summed == CHECKSUM_PARTIAL) {
644 		inet_proto_csum_replace2(&udph->check, skb, port,
645 					 new_port, false);
646 		if (!udph->check)
647 			udph->check = CSUM_MANGLED_0;
648 	}
649 }
650 
651 static void nf_flow_nat_port(struct sk_buff *skb, unsigned int thoff,
652 			     u8 protocol, __be16 port, __be16 new_port)
653 {
654 	switch (protocol) {
655 	case IPPROTO_TCP:
656 		nf_flow_nat_port_tcp(skb, thoff, port, new_port);
657 		break;
658 	case IPPROTO_UDP:
659 		nf_flow_nat_port_udp(skb, thoff, port, new_port);
660 		break;
661 	}
662 }
663 
664 void nf_flow_snat_port(const struct flow_offload *flow,
665 		       struct sk_buff *skb, unsigned int thoff,
666 		       u8 protocol, enum flow_offload_tuple_dir dir)
667 {
668 	struct flow_ports *hdr;
669 	__be16 port, new_port;
670 
671 	hdr = (void *)(skb_network_header(skb) + thoff);
672 
673 	switch (dir) {
674 	case FLOW_OFFLOAD_DIR_ORIGINAL:
675 		port = hdr->source;
676 		new_port = flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple.dst_port;
677 		hdr->source = new_port;
678 		break;
679 	case FLOW_OFFLOAD_DIR_REPLY:
680 		port = hdr->dest;
681 		new_port = flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple.src_port;
682 		hdr->dest = new_port;
683 		break;
684 	}
685 
686 	nf_flow_nat_port(skb, thoff, protocol, port, new_port);
687 }
688 EXPORT_SYMBOL_GPL(nf_flow_snat_port);
689 
690 void nf_flow_dnat_port(const struct flow_offload *flow, struct sk_buff *skb,
691 		       unsigned int thoff, u8 protocol,
692 		       enum flow_offload_tuple_dir dir)
693 {
694 	struct flow_ports *hdr;
695 	__be16 port, new_port;
696 
697 	hdr = (void *)(skb_network_header(skb) + thoff);
698 
699 	switch (dir) {
700 	case FLOW_OFFLOAD_DIR_ORIGINAL:
701 		port = hdr->dest;
702 		new_port = flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple.src_port;
703 		hdr->dest = new_port;
704 		break;
705 	case FLOW_OFFLOAD_DIR_REPLY:
706 		port = hdr->source;
707 		new_port = flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple.dst_port;
708 		hdr->source = new_port;
709 		break;
710 	}
711 
712 	nf_flow_nat_port(skb, thoff, protocol, port, new_port);
713 }
714 EXPORT_SYMBOL_GPL(nf_flow_dnat_port);
715 
716 int nf_flow_table_init(struct nf_flowtable *flowtable)
717 {
718 	int err;
719 
720 	INIT_DELAYED_WORK(&flowtable->gc_work, nf_flow_offload_work_gc);
721 	flow_block_init(&flowtable->flow_block);
722 	init_rwsem(&flowtable->flow_block_lock);
723 
724 	err = rhashtable_init(&flowtable->rhashtable,
725 			      &nf_flow_offload_rhash_params);
726 	if (err < 0)
727 		return err;
728 
729 	queue_delayed_work(system_power_efficient_wq,
730 			   &flowtable->gc_work, HZ);
731 
732 	mutex_lock(&flowtable_lock);
733 	list_add(&flowtable->list, &flowtables);
734 	mutex_unlock(&flowtable_lock);
735 
736 	return 0;
737 }
738 EXPORT_SYMBOL_GPL(nf_flow_table_init);
739 
740 static void nf_flow_table_do_cleanup(struct nf_flowtable *flow_table,
741 				     struct flow_offload *flow, void *data)
742 {
743 	struct net_device *dev = data;
744 
745 	if (!dev) {
746 		flow_offload_teardown(flow);
747 		return;
748 	}
749 
750 	if (net_eq(nf_ct_net(flow->ct), dev_net(dev)) &&
751 	    (flow->tuplehash[0].tuple.iifidx == dev->ifindex ||
752 	     flow->tuplehash[1].tuple.iifidx == dev->ifindex))
753 		flow_offload_teardown(flow);
754 }
755 
756 void nf_flow_table_gc_cleanup(struct nf_flowtable *flowtable,
757 			      struct net_device *dev)
758 {
759 	nf_flow_table_iterate(flowtable, nf_flow_table_do_cleanup, dev);
760 	flush_delayed_work(&flowtable->gc_work);
761 	nf_flow_table_offload_flush(flowtable);
762 }
763 
764 void nf_flow_table_cleanup(struct net_device *dev)
765 {
766 	struct nf_flowtable *flowtable;
767 
768 	mutex_lock(&flowtable_lock);
769 	list_for_each_entry(flowtable, &flowtables, list)
770 		nf_flow_table_gc_cleanup(flowtable, dev);
771 	mutex_unlock(&flowtable_lock);
772 }
773 EXPORT_SYMBOL_GPL(nf_flow_table_cleanup);
774 
775 void nf_flow_table_free(struct nf_flowtable *flow_table)
776 {
777 	mutex_lock(&flowtable_lock);
778 	list_del(&flow_table->list);
779 	mutex_unlock(&flowtable_lock);
780 
781 	cancel_delayed_work_sync(&flow_table->gc_work);
782 	nf_flow_table_offload_flush(flow_table);
783 	/* ... no more pending work after this stage ... */
784 	nf_flow_table_iterate(flow_table, nf_flow_table_do_cleanup, NULL);
785 	nf_flow_table_gc_run(flow_table);
786 	nf_flow_table_offload_flush_cleanup(flow_table);
787 	rhashtable_destroy(&flow_table->rhashtable);
788 }
789 EXPORT_SYMBOL_GPL(nf_flow_table_free);
790 
791 static int nf_flow_table_init_net(struct net *net)
792 {
793 	net->ft.stat = alloc_percpu(struct nf_flow_table_stat);
794 	return net->ft.stat ? 0 : -ENOMEM;
795 }
796 
797 static void nf_flow_table_fini_net(struct net *net)
798 {
799 	free_percpu(net->ft.stat);
800 }
801 
802 static int nf_flow_table_pernet_init(struct net *net)
803 {
804 	int ret;
805 
806 	ret = nf_flow_table_init_net(net);
807 	if (ret < 0)
808 		return ret;
809 
810 	ret = nf_flow_table_init_proc(net);
811 	if (ret < 0)
812 		goto out_proc;
813 
814 	return 0;
815 
816 out_proc:
817 	nf_flow_table_fini_net(net);
818 	return ret;
819 }
820 
821 static void nf_flow_table_pernet_exit(struct list_head *net_exit_list)
822 {
823 	struct net *net;
824 
825 	list_for_each_entry(net, net_exit_list, exit_list) {
826 		nf_flow_table_fini_proc(net);
827 		nf_flow_table_fini_net(net);
828 	}
829 }
830 
831 static struct pernet_operations nf_flow_table_net_ops = {
832 	.init = nf_flow_table_pernet_init,
833 	.exit_batch = nf_flow_table_pernet_exit,
834 };
835 
836 static int __init nf_flow_table_module_init(void)
837 {
838 	int ret;
839 
840 	flow_offload_cachep = KMEM_CACHE(flow_offload, SLAB_HWCACHE_ALIGN);
841 	if (!flow_offload_cachep)
842 		return -ENOMEM;
843 
844 	ret = register_pernet_subsys(&nf_flow_table_net_ops);
845 	if (ret < 0)
846 		goto out_pernet;
847 
848 	ret = nf_flow_table_offload_init();
849 	if (ret)
850 		goto out_offload;
851 
852 	ret = nf_flow_register_bpf();
853 	if (ret)
854 		goto out_bpf;
855 
856 	return 0;
857 
858 out_bpf:
859 	nf_flow_table_offload_exit();
860 out_offload:
861 	unregister_pernet_subsys(&nf_flow_table_net_ops);
862 out_pernet:
863 	kmem_cache_destroy(flow_offload_cachep);
864 	return ret;
865 }
866 
867 static void __exit nf_flow_table_module_exit(void)
868 {
869 	rcu_barrier();
870 	nf_flow_table_offload_exit();
871 	unregister_pernet_subsys(&nf_flow_table_net_ops);
872 	kmem_cache_destroy(flow_offload_cachep);
873 }
874 
875 module_init(nf_flow_table_module_init);
876 module_exit(nf_flow_table_module_exit);
877 
878 MODULE_LICENSE("GPL");
879 MODULE_AUTHOR("Pablo Neira Ayuso <pablo@netfilter.org>");
880 MODULE_DESCRIPTION("Netfilter flow table module");
881