xref: /illumos-gate/usr/src/uts/common/io/mac/mac_util.c (revision c43a1d0aaff11327f09707167cca88394cf8d0a8)
1 /*
2  * CDDL HEADER START
3  *
4  * The contents of this file are subject to the terms of the
5  * Common Development and Distribution License (the "License").
6  * You may not use this file except in compliance with the License.
7  *
8  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
9  * or http://www.opensolaris.org/os/licensing.
10  * See the License for the specific language governing permissions
11  * and limitations under the License.
12  *
13  * When distributing Covered Code, include this CDDL HEADER in each
14  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
15  * If applicable, add the following below this CDDL HEADER, with the
16  * fields enclosed by brackets "[]" replaced with your own identifying
17  * information: Portions Copyright [yyyy] [name of copyright owner]
18  *
19  * CDDL HEADER END
20  */
21 /*
22  * Copyright (c) 2008, 2010, Oracle and/or its affiliates. All rights reserved.
23  * Copyright 2019 Joyent, Inc.
24  * Copyright 2026 Oxide Computer Company
25  */
26 
27 /*
28  * MAC Services Module - misc utilities
29  */
30 
31 #include <sys/types.h>
32 #include <sys/mac.h>
33 #include <sys/mac_impl.h>
34 #include <sys/mac_client_priv.h>
35 #include <sys/mac_client_impl.h>
36 #include <sys/mac_soft_ring.h>
37 #include <sys/strsubr.h>
38 #include <sys/strsun.h>
39 #include <sys/vlan.h>
40 #include <sys/pattr.h>
41 #include <sys/pci_tools.h>
42 #include <inet/ip.h>
43 #include <inet/ip_impl.h>
44 #include <inet/ip6.h>
45 #include <sys/vtrace.h>
46 #include <sys/dlpi.h>
47 #include <sys/sunndi.h>
48 #include <inet/ipsec_impl.h>
49 #include <inet/sadb.h>
50 #include <inet/ipsecesp.h>
51 #include <inet/ipsecah.h>
52 #include <inet/tcp.h>
53 #include <inet/sctp_ip.h>
54 
55 /*
56  * The next two functions are used for dropping packets or chains of
57  * packets, respectively. We could use one function for both but
58  * separating the use cases allows us to specify intent and prevent
59  * dropping more data than intended.
60  *
61  * The purpose of these functions is to aid the debugging effort,
62  * especially in production. Rather than use freemsg()/freemsgchain(),
63  * it's preferable to use these functions when dropping a packet in
64  * the MAC layer. These functions should only be used during
65  * unexpected conditions. That is, any time a packet is dropped
66  * outside of the regular, successful datapath. Consolidating all
67  * drops on these functions allows the user to trace one location and
68  * determine why the packet was dropped based on the msg. It also
69  * allows the user to inspect the packet before it is freed. Finally,
70  * it allows the user to avoid tracing freemsg()/freemsgchain() thus
71  * keeping the hot path running as efficiently as possible.
72  *
73  * NOTE: At this time not all MAC drops are aggregated on these
74  * functions; but that is the plan. This comment should be erased once
75  * completed.
76  */
77 
78 /*PRINTFLIKE2*/
79 void
mac_drop_pkt(mblk_t * mp,const char * fmt,...)80 mac_drop_pkt(mblk_t *mp, const char *fmt, ...)
81 {
82 	va_list adx;
83 	char msg[128];
84 	char *msgp = msg;
85 
86 	ASSERT3P(mp->b_next, ==, NULL);
87 
88 	va_start(adx, fmt);
89 	(void) vsnprintf(msgp, sizeof (msg), fmt, adx);
90 	va_end(adx);
91 
92 	DTRACE_PROBE2(mac__drop, mblk_t *, mp, char *, msgp);
93 	freemsg(mp);
94 }
95 
96 /*PRINTFLIKE2*/
97 void
mac_drop_chain(mblk_t * chain,const char * fmt,...)98 mac_drop_chain(mblk_t *chain, const char *fmt, ...)
99 {
100 	va_list adx;
101 	char msg[128];
102 	char *msgp = msg;
103 
104 	va_start(adx, fmt);
105 	(void) vsnprintf(msgp, sizeof (msg), fmt, adx);
106 	va_end(adx);
107 
108 	/*
109 	 * We could use freemsgchain() for the actual freeing but
110 	 * since we are already walking the chain to fire the dtrace
111 	 * probe we might as well free the msg here too.
112 	 */
113 	for (mblk_t *mp = chain, *next; mp != NULL; ) {
114 		next = mp->b_next;
115 		DTRACE_PROBE2(mac__drop, mblk_t *, mp, char *, msgp);
116 		mp->b_next = NULL;
117 		freemsg(mp);
118 		mp = next;
119 	}
120 }
121 
122 /*
123  * Perform software checksum on a single message, if needed. The emulation
124  * performed is determined by an intersection of the mblk's flags and the emul
125  * flags requested. The emul flags are documented in mac.h.
126  */
127 static mblk_t *
mac_sw_cksum(mblk_t * mp,mac_emul_t emul)128 mac_sw_cksum(mblk_t *mp, mac_emul_t emul)
129 {
130 	mac_ether_offload_info_t meoi = { 0 };
131 	const char *err = "";
132 
133 	/*
134 	 * The only current caller is mac_hw_emul(), which handles any chaining
135 	 * of mblks prior to now.
136 	 */
137 	VERIFY3P(mp->b_next, ==, NULL);
138 
139 	uint32_t flags = DB_CKSUMFLAGS(mp);
140 
141 	/* Why call this if checksum emulation isn't needed? */
142 	ASSERT3U(flags & (HCK_FLAGS), !=, 0);
143 	/* But also, requesting both ULP cksum types is improper */
144 	if ((flags & HCK_FULLCKSUM) != 0 && (flags & HCK_PARTIALCKSUM) != 0) {
145 		err = "full and partial ULP cksum requested";
146 		goto bail;
147 	}
148 
149 	const boolean_t do_v4_cksum = (emul & MAC_IPCKSUM_EMUL) != 0 &&
150 	    (flags & HCK_IPV4_HDRCKSUM) != 0;
151 	const boolean_t do_ulp_cksum = (emul & MAC_HWCKSUM_EMUL) != 0 &&
152 	    (flags & (HCK_FULLCKSUM | HCK_PARTIALCKSUM)) != 0;
153 	const boolean_t ulp_prefer_partial = (flags & HCK_PARTIALCKSUM) != 0;
154 
155 	mac_ether_offload_info(mp, &meoi);
156 	if ((meoi.meoi_flags & MEOI_L2INFO_SET) == 0 ||
157 	    (meoi.meoi_l3proto != ETHERTYPE_IP &&
158 	    meoi.meoi_l3proto != ETHERTYPE_IPV6)) {
159 		/* Non-IP traffic (like ARP) is left alone */
160 		return (mp);
161 	}
162 
163 	/*
164 	 * Ensure that requested checksum type(s) are supported by the
165 	 * protocols encoded in the packet headers.
166 	 */
167 	if (do_v4_cksum) {
168 		if (meoi.meoi_l3proto != ETHERTYPE_IP) {
169 			err = "IPv4 csum requested on non-IPv4 packet";
170 			goto bail;
171 		}
172 	}
173 	if (do_ulp_cksum) {
174 		if ((meoi.meoi_flags & MEOI_L4INFO_SET) == 0) {
175 			err = "missing ULP header";
176 			goto bail;
177 		}
178 		switch (meoi.meoi_l4proto) {
179 		case IPPROTO_TCP:
180 		case IPPROTO_UDP:
181 		case IPPROTO_ICMP:
182 		case IPPROTO_ICMPV6:
183 		case IPPROTO_SCTP:
184 			break;
185 		default:
186 			err = "unexpected ULP";
187 			goto bail;
188 		}
189 	}
190 
191 	/*
192 	 * If the first mblk of this packet contains only the Ethernet header,
193 	 * skip past it for now. Packets with their data contained in only a
194 	 * single mblk can then use the fastpaths tuned to that possibility.
195 	 */
196 	mblk_t *skipped_hdr = NULL;
197 	if (MBLKL(mp) == meoi.meoi_l2hlen) {
198 		meoi.meoi_len -= meoi.meoi_l2hlen;
199 		meoi.meoi_l2hlen = 0;
200 		skipped_hdr = mp;
201 		mp = mp->b_cont;
202 
203 		ASSERT(mp != NULL);
204 	}
205 
206 	/*
207 	 * Ensure that all of the headers we need to access are:
208 	 * 1. Collected in the first mblk
209 	 * 2. Held in a data-block which is safe for us to modify
210 	 *    (It must have a refcount of 1)
211 	 * 3. IP headers are 4-byte aligned. IP header size is always a multiple
212 	 *    of 4 bytes, thus L4 headers will also be safe to access.
213 	 */
214 	const size_t hdr_len_reqd = (meoi.meoi_l2hlen + meoi.meoi_l3hlen) +
215 	    (do_ulp_cksum ? meoi.meoi_l4hlen : 0);
216 	if (MBLKL(mp) < hdr_len_reqd || DB_REF(mp) > 1 ||
217 	    !OK_32PTR(mp->b_rptr + meoi.meoi_l2hlen)) {
218 		const size_t pad_by = (4 - (meoi.meoi_l2hlen % 4)) % 4;
219 		mblk_t *hdrmp = msgpullup_pad(mp, hdr_len_reqd, pad_by);
220 
221 		if (hdrmp == NULL) {
222 			err = "could not pullup msg headers";
223 			goto bail;
224 		}
225 
226 		mac_hcksum_clone(mp, hdrmp);
227 		if (skipped_hdr != NULL) {
228 			ASSERT3P(skipped_hdr->b_cont, ==, mp);
229 			skipped_hdr->b_cont = hdrmp;
230 		}
231 		freemsg(mp);
232 		mp = hdrmp;
233 	}
234 
235 	/* Calculate IPv4 header checksum, if requested */
236 	if (do_v4_cksum) {
237 		/*
238 		 * While unlikely, it's possible to write code that might end up
239 		 * calling mac_sw_cksum() twice on the same mblk (performing
240 		 * both LSO and checksum emulation in a single mblk chain loop
241 		 * -- the LSO emulation inserts a new chain into the existing
242 		 * chain and then the loop iterates back over the new segments
243 		 * and emulates the checksum a second time).  Normally this
244 		 * wouldn't be a problem, because the HCK_*_OK flags are
245 		 * supposed to indicate that we don't need to do peform the
246 		 * work. But HCK_IPV4_HDRCKSUM and HCK_IPV4_HDRCKSUM_OK have the
247 		 * same value; so we cannot use these flags to determine if the
248 		 * IP header checksum has already been calculated or not. For
249 		 * this reason, we zero out the the checksum first. In the
250 		 * future, we should fix the HCK_* flags.
251 		 */
252 		ipha_t *ipha = (ipha_t *)(mp->b_rptr + meoi.meoi_l2hlen);
253 		ipha->ipha_hdr_checksum = 0;
254 		ipha->ipha_hdr_checksum = (uint16_t)ip_csum_hdr(ipha);
255 		flags &= ~HCK_IPV4_HDRCKSUM;
256 		flags |= HCK_IPV4_HDRCKSUM_OK;
257 	}
258 
259 	/*
260 	 * The SCTP is different from all the other protocols in that it uses
261 	 * CRC32 for its checksum, rather than ones' complement.
262 	 */
263 	if (do_ulp_cksum && meoi.meoi_l4proto == IPPROTO_SCTP) {
264 		if (ulp_prefer_partial) {
265 			err = "SCTP does not support partial checksum";
266 			goto bail;
267 		}
268 
269 		const uint_t ulp_off = meoi.meoi_l2hlen + meoi.meoi_l3hlen;
270 		sctp_hdr_t *sctph = (sctp_hdr_t *)(mp->b_rptr + ulp_off);
271 
272 		sctph->sh_chksum = 0;
273 		sctph->sh_chksum = sctp_cksum(mp, ulp_off);
274 
275 		flags &= ~HCK_FULLCKSUM;
276 		flags |= HCK_FULLCKSUM_OK;
277 		goto success;
278 	}
279 
280 	/* Calculate full ULP checksum, if requested */
281 	if (do_ulp_cksum && !ulp_prefer_partial) {
282 		/*
283 		 * Calculate address and length portions of pseudo-header csum
284 		 */
285 		uint32_t cksum = 0;
286 		if (meoi.meoi_l3proto == ETHERTYPE_IP) {
287 			const ipha_t *ipha =
288 			    (const ipha_t *)(mp->b_rptr + meoi.meoi_l2hlen);
289 			const uint16_t *ipp =
290 			    (const uint16_t *)(&ipha->ipha_src);
291 
292 			cksum += ipp[0] + ipp[1] + ipp[2] + ipp[3];
293 
294 			/*
295 			 * While it is tempting to calculate the payload length
296 			 * solely from `meoi`, like as done below for IPv6,
297 			 * doing so is a trap.  Packets shorter than 60 bytes
298 			 * will get padded out to that length in order to meet
299 			 * the minimums for Ethernet.  Instead, we pull the
300 			 * length from the IP header.
301 			 */
302 			const uint16_t payload_len =
303 			    ntohs(ipha->ipha_length) - meoi.meoi_l3hlen;
304 			cksum += htons(payload_len);
305 		} else if (meoi.meoi_l3proto == ETHERTYPE_IPV6) {
306 			const ip6_t *ip6h =
307 			    (const ip6_t *)(mp->b_rptr + meoi.meoi_l2hlen);
308 			const uint16_t *ipp =
309 			    (const uint16_t *)(&ip6h->ip6_src);
310 
311 			cksum += ipp[0] + ipp[1] + ipp[2] + ipp[3] +
312 			    ipp[4] + ipp[5] + ipp[6] + ipp[7];
313 			cksum += ipp[8] + ipp[9] + ipp[10] + ipp[11] +
314 			    ipp[12] + ipp[13] + ipp[14] + ipp[15];
315 
316 			const uint16_t payload_len = meoi.meoi_len -
317 			    ((uint16_t)meoi.meoi_l2hlen + meoi.meoi_l3hlen);
318 			cksum += htons(payload_len);
319 		} else {
320 			/*
321 			 * Since we already checked for recognized L3 protocols
322 			 * earlier, this should not be reachable.
323 			 */
324 			panic("L3 protocol unexpectedly changed");
325 		}
326 
327 		/* protocol portion of pseudo-header */
328 		uint_t cksum_off;
329 		switch (meoi.meoi_l4proto) {
330 		case IPPROTO_TCP:
331 			cksum += IP_TCP_CSUM_COMP;
332 			cksum_off = TCP_CHECKSUM_OFFSET;
333 			break;
334 		case IPPROTO_UDP:
335 			cksum += IP_UDP_CSUM_COMP;
336 			cksum_off = UDP_CHECKSUM_OFFSET;
337 			break;
338 		case IPPROTO_ICMP:
339 			/* ICMP cksum does not include pseudo-header contents */
340 			cksum = 0;
341 			cksum_off = ICMP_CHECKSUM_OFFSET;
342 			break;
343 		case IPPROTO_ICMPV6:
344 			cksum += IP_ICMPV6_CSUM_COMP;
345 			cksum_off = ICMPV6_CHECKSUM_OFFSET;
346 			break;
347 		default:
348 			err = "unrecognized L4 protocol";
349 			goto bail;
350 		}
351 
352 		/*
353 		 * With IP_CSUM() taking into account the pseudo-header
354 		 * checksum, make sure the ULP checksum field is zeroed before
355 		 * computing the rest;
356 		 */
357 		const uint_t l4_off = meoi.meoi_l3hlen + meoi.meoi_l2hlen;
358 		uint16_t *up = (uint16_t *)(mp->b_rptr + l4_off + cksum_off);
359 		*up = 0;
360 		cksum = IP_CSUM(mp, l4_off, cksum);
361 
362 		if (meoi.meoi_l4proto == IPPROTO_UDP && cksum == 0) {
363 			/*
364 			 * A zero checksum is not allowed on UDPv6, and on UDPv4
365 			 * implies no checksum.  In either case, invert to a
366 			 * values of all-1s.
367 			 */
368 			*up = 0xffff;
369 		} else {
370 			*up = cksum;
371 		}
372 
373 		flags &= ~HCK_FULLCKSUM;
374 		flags |= HCK_FULLCKSUM_OK;
375 		goto success;
376 	}
377 
378 	/* Calculate partial ULP checksum, if requested */
379 	if (do_ulp_cksum && ulp_prefer_partial) {
380 		uint32_t start, stuff, end, value;
381 		mac_hcksum_get(mp, &start, &stuff, &end, &value, NULL);
382 
383 		ASSERT3S(end, >, start);
384 
385 		/*
386 		 * The prior size checks against the header length data ensure
387 		 * that the mblk contains everything through at least the ULP
388 		 * header, but if the partial checksum (unexpectedly) requests
389 		 * its result be stored past that, we cannot continue.
390 		 */
391 		if (stuff + sizeof (uint16_t) > MBLKL(mp)) {
392 			err = "partial csum request is out of bounds";
393 			goto bail;
394 		}
395 
396 		uchar_t *ipp = (uchar_t *)(mp->b_rptr + meoi.meoi_l2hlen);
397 		uint16_t *up = (uint16_t *)(ipp + stuff);
398 
399 		const uint16_t partial = *up;
400 		*up = 0;
401 		const uint16_t cksum =
402 		    ~IP_CSUM_PARTIAL(mp, start + meoi.meoi_l2hlen, partial);
403 		*up = cksum != 0 ? cksum : ~cksum;
404 
405 		flags &= ~HCK_PARTIALCKSUM;
406 		flags |= HCK_FULLCKSUM_OK;
407 	}
408 
409 success:
410 	/*
411 	 * With the checksum(s) calculated, store the updated flags to reflect
412 	 * the current status, and zero out any of the partial-checksum fields
413 	 * which would be irrelevant now.
414 	 */
415 	mac_hcksum_set(mp, 0, 0, 0, 0, flags);
416 
417 	/* Don't forget to reattach the header. */
418 	if (skipped_hdr != NULL) {
419 		ASSERT3P(skipped_hdr->b_cont, ==, mp);
420 
421 		/*
422 		 * Duplicate the HCKSUM data into the header mblk.
423 		 *
424 		 * This mimics mac_add_vlan_tag() which ensures that both the
425 		 * first mblk _and_ the first data bearing mblk possess the
426 		 * HCKSUM information. Consumers like IP will end up discarding
427 		 * the ether_header mblk, so for now, it is important that the
428 		 * data be available in both places.
429 		 */
430 		mac_hcksum_clone(mp, skipped_hdr);
431 		mp = skipped_hdr;
432 	}
433 	return (mp);
434 
435 bail:
436 	if (skipped_hdr != NULL) {
437 		ASSERT3P(skipped_hdr->b_cont, ==, mp);
438 		mp = skipped_hdr;
439 	}
440 
441 	mac_drop_pkt(mp, err);
442 	return (NULL);
443 }
444 
445 /*
446  * Build a single data segment from an LSO packet. The mblk chain
447  * returned, seg_head, represents the data segment and is always
448  * exactly seg_len bytes long. The lso_mp and offset input/output
449  * parameters track our position in the LSO packet. This function
450  * exists solely as a helper to mac_sw_lso().
451  *
452  * Case A
453  *
454  *     The current lso_mp is larger than the requested seg_len. The
455  *     beginning of seg_head may start at the beginning of lso_mp or
456  *     offset into it. In either case, a single mblk is returned, and
457  *     *offset is updated to reflect our new position in the current
458  *     lso_mp.
459  *
460  *          +----------------------------+
461  *          |  in *lso_mp / out *lso_mp  |
462  *          +----------------------------+
463  *          ^                        ^
464  *          |                        |
465  *          |                        |
466  *          |                        |
467  *          +------------------------+
468  *          |        seg_head        |
469  *          +------------------------+
470  *          ^                        ^
471  *          |                        |
472  *   in *offset = 0        out *offset = seg_len
473  *
474  *          |------   seg_len    ----|
475  *
476  *
477  *       +------------------------------+
478  *       |   in *lso_mp / out *lso_mp   |
479  *       +------------------------------+
480  *          ^                        ^
481  *          |                        |
482  *          |                        |
483  *          |                        |
484  *          +------------------------+
485  *          |        seg_head        |
486  *          +------------------------+
487  *          ^                        ^
488  *          |                        |
489  *   in *offset = N        out *offset = N + seg_len
490  *
491  *          |------   seg_len    ----|
492  *
493  *
494  *
495  * Case B
496  *
497  *    The requested seg_len consumes exactly the rest of the lso_mp.
498  *    I.e., the seg_head's b_wptr is equivalent to lso_mp's b_wptr.
499  *    The seg_head may start at the beginning of the lso_mp or at some
500  *    offset into it. In either case we return a single mblk, reset
501  *    *offset to zero, and walk to the next lso_mp.
502  *
503  *          +------------------------+           +------------------------+
504  *          |       in *lso_mp       |---------->|      out *lso_mp       |
505  *          +------------------------+           +------------------------+
506  *          ^                        ^           ^
507  *          |                        |           |
508  *          |                        |    out *offset = 0
509  *          |                        |
510  *          +------------------------+
511  *          |        seg_head        |
512  *          +------------------------+
513  *          ^
514  *          |
515  *   in *offset = 0
516  *
517  *          |------   seg_len    ----|
518  *
519  *
520  *
521  *      +----------------------------+           +------------------------+
522  *      |         in *lso_mp         |---------->|      out *lso_mp       |
523  *      +----------------------------+           +------------------------+
524  *          ^                        ^           ^
525  *          |                        |           |
526  *          |                        |    out *offset = 0
527  *          |                        |
528  *          +------------------------+
529  *          |        seg_head        |
530  *          +------------------------+
531  *          ^
532  *          |
533  *   in *offset = N
534  *
535  *          |------   seg_len    ----|
536  *
537  *
538  * Case C
539  *
540  *    The requested seg_len is greater than the current lso_mp. In
541  *    this case we must consume LSO mblks until we have enough data to
542  *    satisfy either case (A) or (B) above. We will return multiple
543  *    mblks linked via b_cont, offset will be set based on the cases
544  *    above, and lso_mp will walk forward at least one mblk, but maybe
545  *    more.
546  *
547  *    N.B. This digram is not exhaustive. The seg_head may start on
548  *    the beginning of an lso_mp. The seg_tail may end exactly on the
549  *    boundary of an lso_mp. And there may be two (in this case the
550  *    middle block wouldn't exist), three, or more mblks in the
551  *    seg_head chain. This is meant as one example of what might
552  *    happen. The main thing to remember is that the seg_tail mblk
553  *    must be one of case (A) or (B) above.
554  *
555  *  +------------------+    +----------------+    +------------------+
556  *  |    in *lso_mp    |--->|    *lso_mp     |--->|   out *lso_mp    |
557  *  +------------------+    +----------------+    +------------------+
558  *        ^            ^    ^                ^    ^            ^
559  *        |            |    |                |    |            |
560  *        |            |    |                |    |            |
561  *        |            |    |                |    |            |
562  *        |            |    |                |    |            |
563  *        +------------+    +----------------+    +------------+
564  *        |  seg_head  |--->|                |--->|  seg_tail  |
565  *        +------------+    +----------------+    +------------+
566  *        ^                                                    ^
567  *        |                                                    |
568  *  in *offset = N                          out *offset = MBLKL(seg_tail)
569  *
570  *        |-------------------   seg_len    -------------------|
571  *
572  */
573 static mblk_t *
build_data_seg(mblk_t ** lso_mp,uint32_t * offset,uint32_t seg_len)574 build_data_seg(mblk_t **lso_mp, uint32_t *offset, uint32_t seg_len)
575 {
576 	mblk_t *seg_head, *seg_tail, *seg_mp;
577 
578 	ASSERT3P(*lso_mp, !=, NULL);
579 	ASSERT3U((*lso_mp)->b_rptr + *offset, <, (*lso_mp)->b_wptr);
580 
581 	seg_mp = dupb(*lso_mp);
582 	if (seg_mp == NULL)
583 		return (NULL);
584 
585 	seg_head = seg_mp;
586 	seg_tail = seg_mp;
587 
588 	/* Continue where we left off from in the lso_mp. */
589 	seg_mp->b_rptr += *offset;
590 
591 last_mblk:
592 	/* Case (A) */
593 	if ((seg_mp->b_rptr + seg_len) < seg_mp->b_wptr) {
594 		*offset += seg_len;
595 		seg_mp->b_wptr = seg_mp->b_rptr + seg_len;
596 		return (seg_head);
597 	}
598 
599 	/* Case (B) */
600 	if ((seg_mp->b_rptr + seg_len) == seg_mp->b_wptr) {
601 		*offset = 0;
602 		*lso_mp = (*lso_mp)->b_cont;
603 		return (seg_head);
604 	}
605 
606 	/* Case (C) */
607 	ASSERT3U(seg_mp->b_rptr + seg_len, >, seg_mp->b_wptr);
608 
609 	/*
610 	 * The current LSO mblk doesn't have enough data to satisfy
611 	 * seg_len -- continue peeling off LSO mblks to build the new
612 	 * segment message. If allocation fails we free the previously
613 	 * allocated segment mblks and return NULL.
614 	 */
615 	while ((seg_mp->b_rptr + seg_len) > seg_mp->b_wptr) {
616 		ASSERT3U(MBLKL(seg_mp), <=, seg_len);
617 		seg_len -= MBLKL(seg_mp);
618 		*offset = 0;
619 		*lso_mp = (*lso_mp)->b_cont;
620 		seg_mp = dupb(*lso_mp);
621 
622 		if (seg_mp == NULL) {
623 			freemsgchain(seg_head);
624 			return (NULL);
625 		}
626 
627 		seg_tail->b_cont = seg_mp;
628 		seg_tail = seg_mp;
629 	}
630 
631 	/*
632 	 * We've walked enough LSO mblks that we can now satisfy the
633 	 * remaining seg_len. At this point we need to jump back to
634 	 * determine if we have arrived at case (A) or (B).
635 	 */
636 
637 	/* Just to be paranoid that we didn't underflow. */
638 	ASSERT3U(seg_len, <, IP_MAXPACKET);
639 	ASSERT3U(seg_len, >, 0);
640 	goto last_mblk;
641 }
642 
643 /*
644  * Perform software segmentation of a single LSO message. Take an LSO
645  * message as input and return head/tail pointers as output. This
646  * function should not be invoked directly but instead through
647  * mac_hw_emul().
648  *
649  * The resulting chain is comprised of multiple (nsegs) MSS sized
650  * segments. Each segment will consist of two or more mblks joined by
651  * b_cont: a header and one or more data mblks. The header mblk is
652  * allocated anew for each message. The first segment's header is used
653  * as a template for the rest with adjustments made for things such as
654  * ID, sequence, length, TCP flags, etc. The data mblks reference into
655  * the existing LSO mblk (passed in as omp) by way of dupb(). Their
656  * b_rptr/b_wptr values are adjusted to reference only the fraction of
657  * the LSO message they are responsible for. At the successful
658  * completion of this function the original mblk (omp) is freed,
659  * leaving the newely created segment chain as the only remaining
660  * reference to the data.
661  */
662 static void
mac_sw_lso(mblk_t * omp,mac_emul_t emul,mblk_t ** head,mblk_t ** tail,uint_t * count)663 mac_sw_lso(mblk_t *omp, mac_emul_t emul, mblk_t **head, mblk_t **tail,
664     uint_t *count)
665 {
666 	uint32_t ocsum_flags, ocsum_start, ocsum_stuff;
667 	uint32_t mss;
668 	uint32_t oehlen, oiphlen, otcphlen, ohdrslen, opktlen;
669 	uint32_t odatalen, oleft;
670 	uint_t nsegs, seg;
671 	int len;
672 
673 	const void *oiph;
674 	const tcph_t *otcph;
675 	ipha_t *niph;
676 	tcph_t *ntcph;
677 	uint16_t ip_id;
678 	uint32_t tcp_seq, tcp_sum, otcp_sum;
679 
680 	boolean_t is_v6 = B_FALSE;
681 	ip6_t *niph6;
682 
683 	uint32_t offset = 0;
684 	mblk_t *odatamp;
685 	mblk_t *seg_chain, *prev_nhdrmp, *next_nhdrmp, *nhdrmp, *ndatamp;
686 	mblk_t *tmptail;
687 
688 	mac_ether_offload_info_t meoi = { 0 };
689 
690 	ASSERT3P(head, !=, NULL);
691 	ASSERT3P(tail, !=, NULL);
692 	ASSERT3P(count, !=, NULL);
693 	ASSERT3U((DB_CKSUMFLAGS(omp) & HW_LSO), !=, 0);
694 
695 	/* Assume we are dealing with a single LSO message. */
696 	ASSERT3P(omp->b_next, ==, NULL);
697 
698 	mac_ether_offload_info(omp, &meoi);
699 	opktlen = meoi.meoi_len;
700 	oehlen = meoi.meoi_l2hlen;
701 	oiphlen = meoi.meoi_l3hlen;
702 	otcphlen = meoi.meoi_l4hlen;
703 	ohdrslen = oehlen + oiphlen + otcphlen;
704 
705 	/* Performing LSO requires that we successfully read fully up to L4 */
706 	if ((MEOI_L4INFO_SET & meoi.meoi_flags) == 0) {
707 		mac_drop_pkt(omp, "unable to fully parse packet to L4");
708 		goto fail;
709 	}
710 
711 	if (meoi.meoi_l3proto != ETHERTYPE_IP &&
712 	    meoi.meoi_l3proto != ETHERTYPE_IPV6) {
713 		mac_drop_pkt(omp, "LSO'd packet has non-IP L3 header: %x",
714 		    meoi.meoi_l3proto);
715 		goto fail;
716 	}
717 
718 	if (meoi.meoi_l4proto != IPPROTO_TCP) {
719 		mac_drop_pkt(omp, "LSO unsupported protocol: %x",
720 		    meoi.meoi_l4proto);
721 		goto fail;
722 	}
723 
724 	is_v6 = meoi.meoi_l3proto == ETHERTYPE_IPV6;
725 
726 	mss = DB_LSOMSS(omp);
727 	if (mss == 0) {
728 		mac_drop_pkt(omp, "packet misconfigured for LSO (MSS == 0)");
729 		goto fail;
730 	}
731 	ASSERT3U(opktlen, <=, IP_MAXPACKET + oehlen);
732 
733 	/*
734 	 * Ensure the headers are contiguous and that L3 and L4 headers are 4B
735 	 * aligned. The IP header is used only for the benefit of DTrace SDTs,
736 	 * whereas the TCP header is actively read. This small pullup should
737 	 * only practically happen when mac_add_vlan_tag is in play, which
738 	 * prepends a new mblk in front containing the amended Ethernet header.
739 	 */
740 	const size_t pad_by = (4 - (meoi.meoi_l2hlen % 4)) % 4;
741 	if (MBLKL(omp) < ohdrslen || !OK_32PTR(omp->b_rptr + oehlen)) {
742 		mblk_t *tmp = msgpullup_pad(omp, ohdrslen, pad_by);
743 
744 		if (tmp == NULL) {
745 			mac_drop_pkt(omp, "failed to pull up");
746 			goto fail;
747 		}
748 
749 		mac_hcksum_clone(omp, tmp);
750 		freemsg(omp);
751 		omp = tmp;
752 	}
753 
754 	oiph = (void *)(omp->b_rptr + oehlen);
755 	otcph = (tcph_t *)(omp->b_rptr + oehlen + oiphlen);
756 
757 	if (otcph->th_flags[0] & (TH_SYN | TH_RST | TH_URG)) {
758 		mac_drop_pkt(omp, "LSO packet has SYN|RST|URG set");
759 		goto fail;
760 	}
761 
762 	len = MBLKL(omp);
763 
764 	/*
765 	 * Either we have data in the first mblk or it's just the
766 	 * header. In either case, we need to set rptr to the start of
767 	 * the TCP data.
768 	 */
769 	if (len > ohdrslen) {
770 		odatamp = omp;
771 		offset = ohdrslen;
772 	} else {
773 		ASSERT3U(len, ==, ohdrslen);
774 		odatamp = omp->b_cont;
775 		offset = 0;
776 	}
777 
778 	/* Make sure we still have enough data. */
779 	odatalen = opktlen - ohdrslen;
780 	ASSERT3U(msgsize(odatamp), >=, odatalen);
781 
782 	/*
783 	 * If a MAC negotiated LSO then it must negotiate both
784 	 * HCKSUM_IPHDRCKSUM and either HCKSUM_INET_FULL_V4 or
785 	 * HCKSUM_INET_PARTIAL; because both the IP and TCP headers
786 	 * change during LSO segmentation (only the 3 fields of the
787 	 * pseudo header checksum don't change: src, dst, proto). Thus
788 	 * we would expect these flags (HCK_IPV4_HDRCKSUM |
789 	 * HCK_PARTIALCKSUM | HCK_FULLCKSUM) to be set and for this
790 	 * function to emulate those checksums in software. However,
791 	 * that assumes a world where we only expose LSO if the
792 	 * underlying hardware exposes LSO. Moving forward the plan is
793 	 * to assume LSO in the upper layers and have MAC perform
794 	 * software LSO when the underlying provider doesn't support
795 	 * it. In such a world, if the provider doesn't support LSO
796 	 * but does support hardware checksum offload, then we could
797 	 * simply perform the segmentation and allow the hardware to
798 	 * calculate the checksums. To the hardware it's just another
799 	 * chain of non-LSO packets.
800 	 */
801 	ASSERT3S(DB_TYPE(omp), ==, M_DATA);
802 	ocsum_flags = DB_CKSUMFLAGS(omp);
803 	ASSERT3U(ocsum_flags & (HCK_PARTIALCKSUM | HCK_FULLCKSUM), !=, 0);
804 
805 	/*
806 	 * If hardware only provides partial checksum then software
807 	 * must supply the pseudo-header checksum. In the case of LSO
808 	 * we leave the TCP length at zero to be filled in by
809 	 * hardware. This function must handle two scenarios.
810 	 *
811 	 * 1. Being called by a MAC client on the Rx path to segment
812 	 *    an LSO packet and calculate the checksum.
813 	 *
814 	 * 2. Being called by a MAC provider to segment an LSO packet.
815 	 *    In this case the LSO segmentation is performed in
816 	 *    software (by this routine) but the MAC provider should
817 	 *    still calculate the TCP/IP checksums in hardware.
818 	 *
819 	 *  To elaborate on the second case: we cannot have the
820 	 *  scenario where IP sends LSO packets but the underlying HW
821 	 *  doesn't support checksum offload -- because in that case
822 	 *  TCP/IP would calculate the checksum in software (for the
823 	 *  LSO packet) but then MAC would segment the packet and have
824 	 *  to redo all the checksum work. So IP should never do LSO
825 	 *  if HW doesn't support both IP and TCP checksum.
826 	 */
827 	if (ocsum_flags & HCK_PARTIALCKSUM) {
828 		ocsum_start = (uint32_t)DB_CKSUMSTART(omp);
829 		ocsum_stuff = (uint32_t)DB_CKSUMSTUFF(omp);
830 	}
831 
832 	/*
833 	 * Subtract one to account for the case where the data length
834 	 * is evenly divisble by the MSS. Add one to account for the
835 	 * fact that the division will always result in one less
836 	 * segment than needed.
837 	 */
838 	nsegs = ((odatalen - 1) / mss) + 1;
839 	if (nsegs < 2) {
840 		mac_drop_pkt(omp, "LSO not enough segs: %u", nsegs);
841 		goto fail;
842 	}
843 
844 	DTRACE_PROBE6(sw__lso__start, mblk_t *, omp, void_ip_t *, oiph,
845 	    __dtrace_tcp_tcph_t *, otcph, uint_t, odatalen, uint_t, mss,
846 	    uint_t, nsegs);
847 
848 	seg_chain = NULL;
849 	tmptail = seg_chain;
850 	oleft = odatalen;
851 
852 	for (uint_t i = 0; i < nsegs; i++) {
853 		boolean_t last_seg = ((i + 1) == nsegs);
854 		uint32_t seg_len;
855 
856 		/*
857 		 * Ensure that we have 4B L3/L4 alignment for any output frames.
858 		 * If we fail to allocate, then drop the partially
859 		 * allocated chain as well as the LSO packet. Let the
860 		 * sender deal with the fallout.
861 		 */
862 		if ((nhdrmp = allocb(pad_by + ohdrslen, 0)) == NULL) {
863 			freemsgchain(seg_chain);
864 			mac_drop_pkt(omp, "failed to alloc segment header");
865 			goto fail;
866 		}
867 		ASSERT3P(nhdrmp->b_cont, ==, NULL);
868 
869 		/* Copy over the header stack. */
870 		nhdrmp->b_rptr += pad_by;
871 		nhdrmp->b_wptr = nhdrmp->b_rptr + ohdrslen;
872 		bcopy(omp->b_rptr, nhdrmp->b_rptr, ohdrslen);
873 
874 		if (seg_chain == NULL) {
875 			seg_chain = nhdrmp;
876 		} else {
877 			ASSERT3P(tmptail, !=, NULL);
878 			tmptail->b_next = nhdrmp;
879 		}
880 
881 		tmptail = nhdrmp;
882 
883 		/*
884 		 * Calculate this segment's length. It's either the MSS
885 		 * or whatever remains for the last segment.
886 		 */
887 		seg_len = last_seg ? oleft : mss;
888 		ASSERT3U(seg_len, <=, mss);
889 		ndatamp = build_data_seg(&odatamp, &offset, seg_len);
890 
891 		if (ndatamp == NULL) {
892 			freemsgchain(seg_chain);
893 			mac_drop_pkt(omp, "LSO failed to segment data");
894 			goto fail;
895 		}
896 
897 		/* Attach data mblk to header mblk. */
898 		nhdrmp->b_cont = ndatamp;
899 		DB_CKSUMFLAGS(ndatamp) &= ~HW_LSO;
900 		ASSERT3U(seg_len, <=, oleft);
901 		oleft -= seg_len;
902 
903 		/* Setup partial checksum offsets. */
904 		if (ocsum_flags & HCK_PARTIALCKSUM) {
905 			DB_CKSUMSTART(nhdrmp) = ocsum_start;
906 			DB_CKSUMEND(nhdrmp) = oiphlen + otcphlen + seg_len;
907 			DB_CKSUMSTUFF(nhdrmp) = ocsum_stuff;
908 		}
909 	}
910 
911 	/* We should have consumed entire LSO msg. */
912 	ASSERT3S(oleft, ==, 0);
913 	ASSERT3P(odatamp, ==, NULL);
914 
915 	/*
916 	 * All seg data mblks are referenced by the header mblks, null
917 	 * out this pointer to catch any bad derefs.
918 	 */
919 	ndatamp = NULL;
920 
921 	/*
922 	 * Set headers and checksum for first segment.
923 	 */
924 	nhdrmp = seg_chain;
925 	ASSERT3U(msgsize(nhdrmp->b_cont), ==, mss);
926 
927 	if (is_v6) {
928 		niph6 = (ip6_t *)(nhdrmp->b_rptr + oehlen);
929 		niph6->ip6_plen = htons(
930 		    (oiphlen - IPV6_HDR_LEN) + otcphlen + mss);
931 	} else {
932 		niph = (ipha_t *)(nhdrmp->b_rptr + oehlen);
933 		niph->ipha_length = htons(oiphlen + otcphlen + mss);
934 		/*
935 		 * If the v4 checksum was filled, we won't have a v4 offload
936 		 * flag. We can't write zero checksums without inserting said
937 		 * flag, but our output frames won't necessarily be rechecked by
938 		 * the caller! As a compromise, we need to force emulation to
939 		 * uphold the same contracts the packet already agreed to.
940 		 */
941 		if (niph->ipha_hdr_checksum != 0) {
942 			emul |= MAC_IPCKSUM_EMUL;
943 			ocsum_flags |= HCK_IPV4_HDRCKSUM;
944 		}
945 		niph->ipha_hdr_checksum = 0;
946 		ip_id = ntohs(niph->ipha_ident);
947 	}
948 
949 	ntcph = (tcph_t *)(nhdrmp->b_rptr + oehlen + oiphlen);
950 	tcp_seq = BE32_TO_U32(ntcph->th_seq);
951 	tcp_seq += mss;
952 
953 	/*
954 	 * The first segment shouldn't:
955 	 *
956 	 *	o indicate end of data transmission (FIN),
957 	 *	o indicate immediate handling of the data (PUSH).
958 	 */
959 	ntcph->th_flags[0] &= ~(TH_FIN | TH_PUSH);
960 	DB_CKSUMFLAGS(nhdrmp) = (uint16_t)(ocsum_flags & ~HW_LSO);
961 
962 	/*
963 	 * If the underlying HW provides partial checksum, then make
964 	 * sure to correct the pseudo header checksum before calling
965 	 * mac_sw_cksum(). The native TCP stack doesn't include the
966 	 * length field in the pseudo header when LSO is in play -- so
967 	 * we need to calculate it here.
968 	 */
969 	if (ocsum_flags & HCK_PARTIALCKSUM) {
970 		tcp_sum = BE16_TO_U16(ntcph->th_sum);
971 		otcp_sum = tcp_sum;
972 		tcp_sum += mss + otcphlen;
973 		tcp_sum = (tcp_sum >> 16) + (tcp_sum & 0xFFFF);
974 		U16_TO_BE16(tcp_sum, ntcph->th_sum);
975 	}
976 
977 	if ((ocsum_flags & HCK_TX_FLAGS) && (emul & MAC_HWCKSUM_EMULS)) {
978 		next_nhdrmp = nhdrmp->b_next;
979 		nhdrmp->b_next = NULL;
980 		nhdrmp = mac_sw_cksum(nhdrmp, emul);
981 		/*
982 		 * The mblk could be replaced (via pull-up) or freed (due to
983 		 * failure) during mac_sw_cksum(), so we must take care with the
984 		 * result here.
985 		 */
986 		if (nhdrmp != NULL) {
987 			nhdrmp->b_next = next_nhdrmp;
988 			next_nhdrmp = NULL;
989 			seg_chain = nhdrmp;
990 		} else {
991 			freemsgchain(next_nhdrmp);
992 			/*
993 			 * nhdrmp referenced the head of seg_chain when it was
994 			 * freed, so further clean-up there is unnecessary
995 			 */
996 			seg_chain = NULL;
997 			mac_drop_pkt(omp, "LSO cksum emulation failed");
998 			goto fail;
999 		}
1000 	}
1001 
1002 	ASSERT3P(nhdrmp, !=, NULL);
1003 
1004 	seg = 1;
1005 	DTRACE_PROBE5(sw__lso__seg, mblk_t *, nhdrmp, void_ip_t *,
1006 	    (is_v6 ? (void *)niph6 : (void *)niph),
1007 	    __dtrace_tcp_tcph_t *, ntcph, uint_t, mss, int_t, seg);
1008 	seg++;
1009 
1010 	/* There better be at least 2 segs. */
1011 	ASSERT3P(nhdrmp->b_next, !=, NULL);
1012 	prev_nhdrmp = nhdrmp;
1013 	nhdrmp = nhdrmp->b_next;
1014 
1015 	/*
1016 	 * Now adjust the headers of the middle segments. For each
1017 	 * header we need to adjust the following.
1018 	 *
1019 	 *	o IP ID
1020 	 *	o IP length
1021 	 *	o TCP sequence
1022 	 *	o TCP flags
1023 	 *	o cksum flags
1024 	 *	o cksum values (if MAC_HWCKSUM_EMUL is set)
1025 	 */
1026 	for (; seg < nsegs; seg++) {
1027 		/*
1028 		 * We use seg_chain as a reference to the first seg
1029 		 * header mblk -- this first header is a template for
1030 		 * the rest of the segments. This copy will include
1031 		 * the now updated checksum values from the first
1032 		 * header. We must reset these checksum values to
1033 		 * their original to make sure we produce the correct
1034 		 * value.
1035 		 */
1036 		ASSERT3P(msgsize(nhdrmp->b_cont), ==, mss);
1037 		if (is_v6) {
1038 			niph6 = (ip6_t *)(nhdrmp->b_rptr + oehlen);
1039 			niph6->ip6_plen = htons(
1040 			    (oiphlen - IPV6_HDR_LEN) + otcphlen + mss);
1041 		} else {
1042 			niph = (ipha_t *)(nhdrmp->b_rptr + oehlen);
1043 			niph->ipha_ident = htons(++ip_id);
1044 			niph->ipha_length = htons(oiphlen + otcphlen + mss);
1045 			niph->ipha_hdr_checksum = 0;
1046 		}
1047 		ntcph = (tcph_t *)(nhdrmp->b_rptr + oehlen + oiphlen);
1048 		U32_TO_BE32(tcp_seq, ntcph->th_seq);
1049 		tcp_seq += mss;
1050 		/*
1051 		 * Just like the first segment, the middle segments
1052 		 * shouldn't have these flags set.
1053 		 */
1054 		ntcph->th_flags[0] &= ~(TH_FIN | TH_PUSH);
1055 		DB_CKSUMFLAGS(nhdrmp) = (uint16_t)(ocsum_flags & ~HW_LSO);
1056 
1057 		/*
1058 		 * First and middle segs have same
1059 		 * pseudo-header checksum.
1060 		 */
1061 		if (ocsum_flags & HCK_PARTIALCKSUM)
1062 			U16_TO_BE16(tcp_sum, ntcph->th_sum);
1063 
1064 		if ((ocsum_flags & HCK_TX_FLAGS) &&
1065 		    (emul & MAC_HWCKSUM_EMULS)) {
1066 			next_nhdrmp = nhdrmp->b_next;
1067 			nhdrmp->b_next = NULL;
1068 			nhdrmp = mac_sw_cksum(nhdrmp, emul);
1069 			/*
1070 			 * Like above, handle cases where mac_sw_cksum() does a
1071 			 * pull-up or drop of the mblk.
1072 			 */
1073 			if (nhdrmp != NULL) {
1074 				nhdrmp->b_next = next_nhdrmp;
1075 				next_nhdrmp = NULL;
1076 				prev_nhdrmp->b_next = nhdrmp;
1077 			} else {
1078 				freemsgchain(next_nhdrmp);
1079 				/*
1080 				 * Critical to de-link the now-freed nhdrmp
1081 				 * before freeing the rest of the preceding
1082 				 * chain.
1083 				 */
1084 				prev_nhdrmp->b_next = NULL;
1085 				freemsgchain(seg_chain);
1086 				seg_chain = NULL;
1087 				mac_drop_pkt(omp, "LSO cksum emulation failed");
1088 				goto fail;
1089 			}
1090 		}
1091 
1092 		DTRACE_PROBE5(sw__lso__seg, mblk_t *, nhdrmp, void_ip_t *,
1093 		    (is_v6 ? (void *)niph6 : (void *)niph),
1094 		    __dtrace_tcp_tcph_t *, ntcph, uint_t, mss, uint_t, seg);
1095 
1096 		ASSERT3P(nhdrmp->b_next, !=, NULL);
1097 		prev_nhdrmp = nhdrmp;
1098 		nhdrmp = nhdrmp->b_next;
1099 	}
1100 
1101 	/* Make sure we are on the last segment. */
1102 	ASSERT3U(seg, ==, nsegs);
1103 	ASSERT3P(nhdrmp->b_next, ==, NULL);
1104 
1105 	/*
1106 	 * Now we set the last segment header. The difference being
1107 	 * that FIN/PSH/RST flags are allowed.
1108 	 */
1109 	len = msgsize(nhdrmp->b_cont);
1110 	ASSERT3S(len, >, 0);
1111 	if (is_v6) {
1112 		niph6 = (ip6_t *)(nhdrmp->b_rptr + oehlen);
1113 		niph6->ip6_plen = htons(
1114 		    (oiphlen - IPV6_HDR_LEN) + otcphlen + len);
1115 	} else {
1116 		niph = (ipha_t *)(nhdrmp->b_rptr + oehlen);
1117 		niph->ipha_ident = htons(++ip_id);
1118 		niph->ipha_length = htons(oiphlen + otcphlen + len);
1119 		niph->ipha_hdr_checksum = 0;
1120 	}
1121 	ntcph = (tcph_t *)(nhdrmp->b_rptr + oehlen + oiphlen);
1122 	U32_TO_BE32(tcp_seq, ntcph->th_seq);
1123 
1124 	DB_CKSUMFLAGS(nhdrmp) = (uint16_t)(ocsum_flags & ~HW_LSO);
1125 	if (ocsum_flags & HCK_PARTIALCKSUM) {
1126 		tcp_sum = otcp_sum;
1127 		tcp_sum += len + otcphlen;
1128 		tcp_sum = (tcp_sum >> 16) + (tcp_sum & 0xFFFF);
1129 		U16_TO_BE16(tcp_sum, ntcph->th_sum);
1130 	}
1131 
1132 	if ((ocsum_flags & HCK_TX_FLAGS) && (emul & MAC_HWCKSUM_EMULS)) {
1133 		/* This should be the last mblk. */
1134 		ASSERT3P(nhdrmp->b_next, ==, NULL);
1135 		nhdrmp = mac_sw_cksum(nhdrmp, emul);
1136 		/*
1137 		 * If the final mblk happens to be dropped as part of
1138 		 * mac_sw_cksum(), that is unfortunate, but it need not be a
1139 		 * show-stopper at this point.  We can just pretend that final
1140 		 * packet was dropped in transit.
1141 		 */
1142 		prev_nhdrmp->b_next = nhdrmp;
1143 	}
1144 
1145 	DTRACE_PROBE5(sw__lso__seg, mblk_t *, nhdrmp, void_ip_t *,
1146 	    (is_v6 ? (void *)niph6 : (void *)niph),
1147 	    __dtrace_tcp_tcph_t *, ntcph, uint_t, len, uint_t, seg);
1148 
1149 	/*
1150 	 * Free the reference to the original LSO message as it is
1151 	 * being replaced by seg_cahin.
1152 	 */
1153 	freemsg(omp);
1154 	*head = seg_chain;
1155 	*tail = nhdrmp;
1156 	*count = nsegs;
1157 	return;
1158 
1159 fail:
1160 	*head = NULL;
1161 	*tail = NULL;
1162 	*count = 0;
1163 }
1164 
1165 #define	HCK_NEEDED	(HCK_IPV4_HDRCKSUM | HCK_PARTIALCKSUM | HCK_FULLCKSUM)
1166 
1167 /*
1168  * Emulate various hardware offload features in software. Take a chain
1169  * of packets as input and emulate the hardware features specified in
1170  * 'emul'. The resulting chain's head pointer replaces the 'mp_chain'
1171  * pointer given as input, and its tail pointer is written to
1172  * '*otail'. The number of packets in the new chain is written to
1173  * '*ocount'. The 'otail' and 'ocount' arguments are optional and thus
1174  * may be NULL. The 'mp_chain' argument may point to a NULL chain; in
1175  * which case 'mp_chain' will simply stay a NULL chain.
1176  *
1177  * While unlikely, it is technically possible that this function could
1178  * receive a non-NULL chain as input and return a NULL chain as output
1179  * ('*mp_chain' and '*otail' would be NULL and '*ocount' would be
1180  * zero). This could happen if all the packets in the chain are
1181  * dropped or if we fail to allocate new mblks. In this case, there is
1182  * nothing for the caller to free. In any event, the caller shouldn't
1183  * assume that '*mp_chain' is non-NULL on return.
1184  *
1185  * This function was written with three main use cases in mind.
1186  *
1187  * 1. To emulate hardware offloads when traveling mac-loopback (two
1188  *    clients on the same mac). This is wired up in mac_tx_send().
1189  *
1190  * 2. To provide hardware offloads to the client when the underlying
1191  *    provider cannot. This is currently wired up in mac_tx() but we
1192  *    still only negotiate offloads when the underlying provider
1193  *    supports them.
1194  *
1195  * 3. To emulate real hardware in simnet.
1196  */
1197 void
mac_hw_emul(mblk_t ** mp_chain,mblk_t ** otail,uint_t * ocount,mac_emul_t emul)1198 mac_hw_emul(mblk_t **mp_chain, mblk_t **otail, uint_t *ocount, mac_emul_t emul)
1199 {
1200 	mblk_t *head = NULL, *tail = NULL;
1201 	uint_t count = 0;
1202 
1203 	ASSERT3S(~(MAC_HWCKSUM_EMULS | MAC_LSO_EMUL) & emul, ==, 0);
1204 	ASSERT3P(mp_chain, !=, NULL);
1205 
1206 	for (mblk_t *mp = *mp_chain; mp != NULL; ) {
1207 		mblk_t *tmp, *next, *tmphead, *tmptail;
1208 		struct ether_header *ehp;
1209 		uint32_t flags;
1210 		uint_t len = MBLKL(mp), l2len;
1211 
1212 		/* Perform LSO/cksum one message at a time. */
1213 		next = mp->b_next;
1214 		mp->b_next = NULL;
1215 
1216 		/*
1217 		 * For our sanity the first mblk should contain at
1218 		 * least the full L2 header.
1219 		 */
1220 		if (len < sizeof (struct ether_header)) {
1221 			mac_drop_pkt(mp, "packet too short (A): %u", len);
1222 			mp = next;
1223 			continue;
1224 		}
1225 
1226 		ehp = (struct ether_header *)mp->b_rptr;
1227 		if (ntohs(ehp->ether_type) == VLAN_TPID)
1228 			l2len = sizeof (struct ether_vlan_header);
1229 		else
1230 			l2len = sizeof (struct ether_header);
1231 
1232 		/*
1233 		 * If the first mblk is solely the L2 header, then
1234 		 * there better be more data.
1235 		 */
1236 		if (len < l2len || (len == l2len && mp->b_cont == NULL)) {
1237 			mac_drop_pkt(mp, "packet too short (C): %u", len);
1238 			mp = next;
1239 			continue;
1240 		}
1241 
1242 		DTRACE_PROBE2(mac__emul, mblk_t *, mp, mac_emul_t, emul);
1243 
1244 		/*
1245 		 * We use DB_CKSUMFLAGS (instead of mac_hcksum_get())
1246 		 * because we don't want to mask-out the LSO flag.
1247 		 */
1248 		flags = DB_CKSUMFLAGS(mp);
1249 
1250 		if ((flags & HW_LSO) && (emul & MAC_LSO_EMUL)) {
1251 			uint_t tmpcount = 0;
1252 
1253 			/*
1254 			 * LSO fix-up handles checksum emulation
1255 			 * inline (if requested). It also frees mp.
1256 			 */
1257 			mac_sw_lso(mp, emul, &tmphead, &tmptail,
1258 			    &tmpcount);
1259 			if (tmphead == NULL) {
1260 				/* mac_sw_lso() freed the mp. */
1261 				mp = next;
1262 				continue;
1263 			}
1264 			count += tmpcount;
1265 		} else if ((flags & HCK_NEEDED) && (emul & MAC_HWCKSUM_EMULS)) {
1266 			tmp = mac_sw_cksum(mp, emul);
1267 			if (tmp == NULL) {
1268 				/* mac_sw_cksum() freed the mp. */
1269 				mp = next;
1270 				continue;
1271 			}
1272 			tmphead = tmp;
1273 			tmptail = tmp;
1274 			count++;
1275 		} else {
1276 			/* There is nothing to emulate. */
1277 			tmp = mp;
1278 			tmphead = tmp;
1279 			tmptail = tmp;
1280 			count++;
1281 		}
1282 
1283 		/*
1284 		 * The tmp mblk chain is either the start of the new
1285 		 * chain or added to the tail of the new chain.
1286 		 */
1287 		if (head == NULL) {
1288 			head = tmphead;
1289 			tail = tmptail;
1290 		} else {
1291 			/* Attach the new mblk to the end of the new chain. */
1292 			tail->b_next = tmphead;
1293 			tail = tmptail;
1294 		}
1295 
1296 		mp = next;
1297 	}
1298 
1299 	*mp_chain = head;
1300 
1301 	if (otail != NULL)
1302 		*otail = tail;
1303 
1304 	if (ocount != NULL)
1305 		*ocount = count;
1306 }
1307 
1308 /*
1309  * Add VLAN tag to the specified mblk.
1310  */
1311 mblk_t *
mac_add_vlan_tag(mblk_t * mp,uint_t pri,uint16_t vid)1312 mac_add_vlan_tag(mblk_t *mp, uint_t pri, uint16_t vid)
1313 {
1314 	mblk_t *hmp;
1315 	struct ether_vlan_header *evhp;
1316 	struct ether_header *ehp;
1317 
1318 	ASSERT(pri != 0 || vid != 0);
1319 
1320 	/*
1321 	 * Allocate an mblk for the new tagged ethernet header,
1322 	 * and copy the MAC addresses and ethertype from the
1323 	 * original header.
1324 	 */
1325 
1326 	hmp = allocb(sizeof (struct ether_vlan_header), BPRI_MED);
1327 	if (hmp == NULL) {
1328 		freemsg(mp);
1329 		return (NULL);
1330 	}
1331 
1332 	evhp = (struct ether_vlan_header *)hmp->b_rptr;
1333 	ehp = (struct ether_header *)mp->b_rptr;
1334 
1335 	bcopy(ehp, evhp, (ETHERADDRL * 2));
1336 	evhp->ether_type = ehp->ether_type;
1337 	evhp->ether_tpid = htons(ETHERTYPE_VLAN);
1338 
1339 	hmp->b_wptr += sizeof (struct ether_vlan_header);
1340 	mp->b_rptr += sizeof (struct ether_header);
1341 
1342 	/*
1343 	 * Free the original message if it's now empty. Link the
1344 	 * rest of messages to the header message.
1345 	 */
1346 	mac_hcksum_clone(mp, hmp);
1347 	if (MBLKL(mp) == 0) {
1348 		hmp->b_cont = mp->b_cont;
1349 		freeb(mp);
1350 	} else {
1351 		hmp->b_cont = mp;
1352 	}
1353 	ASSERT(MBLKL(hmp) >= sizeof (struct ether_vlan_header));
1354 
1355 	/*
1356 	 * Initialize the new TCI (Tag Control Information).
1357 	 */
1358 	evhp->ether_tci = htons(VLAN_TCI(pri, 0, vid));
1359 
1360 	return (hmp);
1361 }
1362 
1363 /*
1364  * Adds a VLAN tag with the specified VID and priority to each mblk of
1365  * the specified chain.
1366  */
1367 mblk_t *
mac_add_vlan_tag_chain(mblk_t * mp_chain,uint_t pri,uint16_t vid)1368 mac_add_vlan_tag_chain(mblk_t *mp_chain, uint_t pri, uint16_t vid)
1369 {
1370 	mblk_t *next_mp, **prev, *mp;
1371 
1372 	mp = mp_chain;
1373 	prev = &mp_chain;
1374 
1375 	while (mp != NULL) {
1376 		next_mp = mp->b_next;
1377 		mp->b_next = NULL;
1378 		if ((mp = mac_add_vlan_tag(mp, pri, vid)) == NULL) {
1379 			freemsgchain(next_mp);
1380 			break;
1381 		}
1382 		*prev = mp;
1383 		prev = &mp->b_next;
1384 		mp = mp->b_next = next_mp;
1385 	}
1386 
1387 	return (mp_chain);
1388 }
1389 
1390 /*
1391  * Strip VLAN tag
1392  */
1393 mblk_t *
mac_strip_vlan_tag(mblk_t * mp)1394 mac_strip_vlan_tag(mblk_t *mp)
1395 {
1396 	mblk_t *newmp;
1397 	struct ether_vlan_header *evhp;
1398 
1399 	evhp = (struct ether_vlan_header *)mp->b_rptr;
1400 	if (ntohs(evhp->ether_tpid) == ETHERTYPE_VLAN) {
1401 		ASSERT(MBLKL(mp) >= sizeof (struct ether_vlan_header));
1402 
1403 		if (DB_REF(mp) > 1) {
1404 			newmp = copymsg(mp);
1405 			if (newmp == NULL)
1406 				return (NULL);
1407 			freemsg(mp);
1408 			mp = newmp;
1409 		}
1410 
1411 		evhp = (struct ether_vlan_header *)mp->b_rptr;
1412 
1413 		ovbcopy(mp->b_rptr, mp->b_rptr + VLAN_TAGSZ, 2 * ETHERADDRL);
1414 		mp->b_rptr += VLAN_TAGSZ;
1415 	}
1416 	return (mp);
1417 }
1418 
1419 /*
1420  * Strip VLAN tag from each mblk of the chain.
1421  */
1422 mblk_t *
mac_strip_vlan_tag_chain(mblk_t * mp_chain)1423 mac_strip_vlan_tag_chain(mblk_t *mp_chain)
1424 {
1425 	mblk_t *mp, *next_mp, **prev;
1426 
1427 	mp = mp_chain;
1428 	prev = &mp_chain;
1429 
1430 	while (mp != NULL) {
1431 		next_mp = mp->b_next;
1432 		mp->b_next = NULL;
1433 		if ((mp = mac_strip_vlan_tag(mp)) == NULL) {
1434 			freemsgchain(next_mp);
1435 			break;
1436 		}
1437 		*prev = mp;
1438 		prev = &mp->b_next;
1439 		mp = mp->b_next = next_mp;
1440 	}
1441 
1442 	return (mp_chain);
1443 }
1444 
1445 /*
1446  * Default callback function. Used when the datapath is not yet initialized.
1447  */
1448 /* ARGSUSED */
1449 void
mac_rx_def(void * arg,mac_resource_handle_t resource,mblk_t * mp_chain,boolean_t loopback)1450 mac_rx_def(void *arg, mac_resource_handle_t resource, mblk_t *mp_chain,
1451     boolean_t loopback)
1452 {
1453 	freemsgchain(mp_chain);
1454 }
1455 
1456 /*
1457  * Determines the IPv6 header length accounting for all the optional IPv6
1458  * headers (hop-by-hop, destination, routing and fragment). The header length
1459  * and next header value (a transport header) is captured.
1460  *
1461  * Returns B_FALSE if all the IP headers are not in the same mblk otherwise
1462  * returns B_TRUE.
1463  */
1464 boolean_t
mac_ip_hdr_length_v6(ip6_t * ip6h,uint8_t * endptr,uint16_t * hdr_length,uint8_t * next_hdr,ip6_frag_t ** fragp)1465 mac_ip_hdr_length_v6(ip6_t *ip6h, uint8_t *endptr, uint16_t *hdr_length,
1466     uint8_t *next_hdr, ip6_frag_t **fragp)
1467 {
1468 	uint16_t length;
1469 	uint_t	ehdrlen;
1470 	uint8_t *whereptr;
1471 	uint8_t *nexthdrp;
1472 	ip6_dest_t *desthdr;
1473 	ip6_rthdr_t *rthdr;
1474 	ip6_frag_t *fraghdr;
1475 
1476 	if (((uchar_t *)ip6h + IPV6_HDR_LEN) > endptr)
1477 		return (B_FALSE);
1478 	ASSERT(IPH_HDR_VERSION(ip6h) == IPV6_VERSION);
1479 	length = IPV6_HDR_LEN;
1480 	whereptr = ((uint8_t *)&ip6h[1]); /* point to next hdr */
1481 
1482 	if (fragp != NULL)
1483 		*fragp = NULL;
1484 
1485 	nexthdrp = &ip6h->ip6_nxt;
1486 	while (whereptr < endptr) {
1487 		/* Is there enough left for len + nexthdr? */
1488 		if (whereptr + MIN_EHDR_LEN > endptr)
1489 			break;
1490 
1491 		switch (*nexthdrp) {
1492 		case IPPROTO_HOPOPTS:
1493 		case IPPROTO_DSTOPTS:
1494 			/* Assumes the headers are identical for hbh and dst */
1495 			desthdr = (ip6_dest_t *)whereptr;
1496 			ehdrlen = 8 * (desthdr->ip6d_len + 1);
1497 			if ((uchar_t *)desthdr +  ehdrlen > endptr)
1498 				return (B_FALSE);
1499 			nexthdrp = &desthdr->ip6d_nxt;
1500 			break;
1501 		case IPPROTO_ROUTING:
1502 			rthdr = (ip6_rthdr_t *)whereptr;
1503 			ehdrlen =  8 * (rthdr->ip6r_len + 1);
1504 			if ((uchar_t *)rthdr +  ehdrlen > endptr)
1505 				return (B_FALSE);
1506 			nexthdrp = &rthdr->ip6r_nxt;
1507 			break;
1508 		case IPPROTO_FRAGMENT:
1509 			fraghdr = (ip6_frag_t *)whereptr;
1510 			ehdrlen = sizeof (ip6_frag_t);
1511 			if ((uchar_t *)&fraghdr[1] > endptr)
1512 				return (B_FALSE);
1513 			nexthdrp = &fraghdr->ip6f_nxt;
1514 			if (fragp != NULL)
1515 				*fragp = fraghdr;
1516 			break;
1517 		case IPPROTO_NONE:
1518 			/* No next header means we're finished */
1519 		default:
1520 			*hdr_length = length;
1521 			*next_hdr = *nexthdrp;
1522 			return (B_TRUE);
1523 		}
1524 		length += ehdrlen;
1525 		whereptr += ehdrlen;
1526 		*hdr_length = length;
1527 		*next_hdr = *nexthdrp;
1528 	}
1529 	switch (*nexthdrp) {
1530 	case IPPROTO_HOPOPTS:
1531 	case IPPROTO_DSTOPTS:
1532 	case IPPROTO_ROUTING:
1533 	case IPPROTO_FRAGMENT:
1534 		/*
1535 		 * If any know extension headers are still to be processed,
1536 		 * the packet's malformed (or at least all the IP header(s) are
1537 		 * not in the same mblk - and that should never happen.
1538 		 */
1539 		return (B_FALSE);
1540 
1541 	default:
1542 		/*
1543 		 * If we get here, we know that all of the IP headers were in
1544 		 * the same mblk, even if the ULP header is in the next mblk.
1545 		 */
1546 		*hdr_length = length;
1547 		*next_hdr = *nexthdrp;
1548 		return (B_TRUE);
1549 	}
1550 }
1551 
1552 /*
1553  * The following set of routines are there to take care of interrupt
1554  * re-targeting for legacy (fixed) interrupts. Some older versions
1555  * of the popular NICs like e1000g do not support MSI-X interrupts
1556  * and they reserve fixed interrupts for RX/TX rings. To re-target
1557  * these interrupts, PCITOOL ioctls need to be used.
1558  */
1559 typedef struct mac_dladm_intr {
1560 	int	ino;
1561 	int	cpu_id;
1562 	char	driver_path[MAXPATHLEN];
1563 	char	nexus_path[MAXPATHLEN];
1564 } mac_dladm_intr_t;
1565 
1566 /* Bind the interrupt to cpu_num */
1567 static int
mac_set_intr(ldi_handle_t lh,processorid_t cpu_num,int oldcpuid,int ino)1568 mac_set_intr(ldi_handle_t lh, processorid_t cpu_num, int oldcpuid, int ino)
1569 {
1570 	pcitool_intr_set_t	iset;
1571 	int			err;
1572 
1573 	iset.old_cpu = oldcpuid;
1574 	iset.ino = ino;
1575 	iset.cpu_id = cpu_num;
1576 	iset.user_version = PCITOOL_VERSION;
1577 	err = ldi_ioctl(lh, PCITOOL_DEVICE_SET_INTR, (intptr_t)&iset, FKIOCTL,
1578 	    kcred, NULL);
1579 
1580 	return (err);
1581 }
1582 
1583 /*
1584  * Search interrupt information. iget is filled in with the info to search
1585  */
1586 static boolean_t
mac_search_intrinfo(pcitool_intr_get_t * iget_p,mac_dladm_intr_t * dln)1587 mac_search_intrinfo(pcitool_intr_get_t *iget_p, mac_dladm_intr_t *dln)
1588 {
1589 	int	i;
1590 	char	driver_path[2 * MAXPATHLEN];
1591 
1592 	for (i = 0; i < iget_p->num_devs; i++) {
1593 		(void) strlcpy(driver_path, iget_p->dev[i].path, MAXPATHLEN);
1594 		(void) snprintf(&driver_path[strlen(driver_path)], MAXPATHLEN,
1595 		    ":%s%d", iget_p->dev[i].driver_name,
1596 		    iget_p->dev[i].dev_inst);
1597 		/* Match the device path for the device path */
1598 		if (strcmp(driver_path, dln->driver_path) == 0) {
1599 			dln->ino = iget_p->ino;
1600 			dln->cpu_id = iget_p->cpu_id;
1601 			return (B_TRUE);
1602 		}
1603 	}
1604 	return (B_FALSE);
1605 }
1606 
1607 /*
1608  * Get information about ino, i.e. if this is the interrupt for our
1609  * device and where it is bound etc.
1610  */
1611 static boolean_t
mac_get_single_intr(ldi_handle_t lh,int oldcpuid,int ino,mac_dladm_intr_t * dln)1612 mac_get_single_intr(ldi_handle_t lh, int oldcpuid, int ino,
1613     mac_dladm_intr_t *dln)
1614 {
1615 	pcitool_intr_get_t	*iget_p;
1616 	int			ipsz;
1617 	int			nipsz;
1618 	int			err;
1619 	uint8_t			inum;
1620 
1621 	/*
1622 	 * Check if SLEEP is OK, i.e if could come here in response to
1623 	 * changing the fanout due to some callback from the driver, say
1624 	 * link speed changes.
1625 	 */
1626 	ipsz = PCITOOL_IGET_SIZE(0);
1627 	iget_p = kmem_zalloc(ipsz, KM_SLEEP);
1628 
1629 	iget_p->num_devs_ret = 0;
1630 	iget_p->user_version = PCITOOL_VERSION;
1631 	iget_p->cpu_id = oldcpuid;
1632 	iget_p->ino = ino;
1633 
1634 	err = ldi_ioctl(lh, PCITOOL_DEVICE_GET_INTR, (intptr_t)iget_p,
1635 	    FKIOCTL, kcred, NULL);
1636 	if (err != 0) {
1637 		kmem_free(iget_p, ipsz);
1638 		return (B_FALSE);
1639 	}
1640 	if (iget_p->num_devs == 0) {
1641 		kmem_free(iget_p, ipsz);
1642 		return (B_FALSE);
1643 	}
1644 	inum = iget_p->num_devs;
1645 	if (iget_p->num_devs_ret < iget_p->num_devs) {
1646 		/* Reallocate */
1647 		nipsz = PCITOOL_IGET_SIZE(iget_p->num_devs);
1648 
1649 		kmem_free(iget_p, ipsz);
1650 		ipsz = nipsz;
1651 		iget_p = kmem_zalloc(ipsz, KM_SLEEP);
1652 
1653 		iget_p->num_devs_ret = inum;
1654 		iget_p->cpu_id = oldcpuid;
1655 		iget_p->ino = ino;
1656 		iget_p->user_version = PCITOOL_VERSION;
1657 		err = ldi_ioctl(lh, PCITOOL_DEVICE_GET_INTR, (intptr_t)iget_p,
1658 		    FKIOCTL, kcred, NULL);
1659 		if (err != 0) {
1660 			kmem_free(iget_p, ipsz);
1661 			return (B_FALSE);
1662 		}
1663 		/* defensive */
1664 		if (iget_p->num_devs != iget_p->num_devs_ret) {
1665 			kmem_free(iget_p, ipsz);
1666 			return (B_FALSE);
1667 		}
1668 	}
1669 
1670 	if (mac_search_intrinfo(iget_p, dln)) {
1671 		kmem_free(iget_p, ipsz);
1672 		return (B_TRUE);
1673 	}
1674 	kmem_free(iget_p, ipsz);
1675 	return (B_FALSE);
1676 }
1677 
1678 /*
1679  * Get the interrupts and check each one to see if it is for our device.
1680  */
1681 static int
mac_validate_intr(ldi_handle_t lh,mac_dladm_intr_t * dln,processorid_t cpuid)1682 mac_validate_intr(ldi_handle_t lh, mac_dladm_intr_t *dln, processorid_t cpuid)
1683 {
1684 	pcitool_intr_info_t	intr_info;
1685 	int			err;
1686 	int			ino;
1687 	int			oldcpuid;
1688 
1689 	err = ldi_ioctl(lh, PCITOOL_SYSTEM_INTR_INFO, (intptr_t)&intr_info,
1690 	    FKIOCTL, kcred, NULL);
1691 	if (err != 0)
1692 		return (-1);
1693 
1694 	for (oldcpuid = 0; oldcpuid < intr_info.num_cpu; oldcpuid++) {
1695 		for (ino = 0; ino < intr_info.num_intr; ino++) {
1696 			if (mac_get_single_intr(lh, oldcpuid, ino, dln)) {
1697 				if (dln->cpu_id == cpuid)
1698 					return (0);
1699 				return (1);
1700 			}
1701 		}
1702 	}
1703 	return (-1);
1704 }
1705 
1706 /*
1707  * Obtain the nexus parent node info. for mdip.
1708  */
1709 static dev_info_t *
mac_get_nexus_node(dev_info_t * mdip,mac_dladm_intr_t * dln)1710 mac_get_nexus_node(dev_info_t *mdip, mac_dladm_intr_t *dln)
1711 {
1712 	struct dev_info		*tdip = (struct dev_info *)mdip;
1713 	struct ddi_minor_data	*minordata;
1714 	dev_info_t		*pdip;
1715 	char			pathname[MAXPATHLEN];
1716 
1717 	while (tdip != NULL) {
1718 		/*
1719 		 * The netboot code could call this function while walking the
1720 		 * device tree so we need to use ndi_devi_tryenter() here to
1721 		 * avoid deadlock.
1722 		 */
1723 		if (ndi_devi_tryenter((dev_info_t *)tdip) == 0)
1724 			break;
1725 
1726 		for (minordata = tdip->devi_minor; minordata != NULL;
1727 		    minordata = minordata->next) {
1728 			if (strncmp(minordata->ddm_node_type, DDI_NT_INTRCTL,
1729 			    strlen(DDI_NT_INTRCTL)) == 0) {
1730 				pdip = minordata->dip;
1731 				(void) ddi_pathname(pdip, pathname);
1732 				(void) snprintf(dln->nexus_path, MAXPATHLEN,
1733 				    "/devices%s:intr", pathname);
1734 				(void) ddi_pathname_minor(minordata, pathname);
1735 				ndi_devi_exit((dev_info_t *)tdip);
1736 				return (pdip);
1737 			}
1738 		}
1739 		ndi_devi_exit((dev_info_t *)tdip);
1740 		tdip = tdip->devi_parent;
1741 	}
1742 	return (NULL);
1743 }
1744 
1745 /*
1746  * For a primary MAC client, if the user has set a list or CPUs or
1747  * we have obtained it implicitly, we try to retarget the interrupt
1748  * for that device on one of the CPUs in the list.
1749  * We assign the interrupt to the same CPU as the poll thread.
1750  */
1751 static boolean_t
mac_check_interrupt_binding(dev_info_t * mdip,int32_t cpuid)1752 mac_check_interrupt_binding(dev_info_t *mdip, int32_t cpuid)
1753 {
1754 	ldi_handle_t		lh = NULL;
1755 	ldi_ident_t		li = NULL;
1756 	int			err;
1757 	int			ret;
1758 	mac_dladm_intr_t	dln;
1759 	dev_info_t		*dip;
1760 	struct ddi_minor_data	*minordata;
1761 
1762 	dln.nexus_path[0] = '\0';
1763 	dln.driver_path[0] = '\0';
1764 
1765 	minordata = ((struct dev_info *)mdip)->devi_minor;
1766 	while (minordata != NULL) {
1767 		if (minordata->type == DDM_MINOR)
1768 			break;
1769 		minordata = minordata->next;
1770 	}
1771 	if (minordata == NULL)
1772 		return (B_FALSE);
1773 
1774 	(void) ddi_pathname_minor(minordata, dln.driver_path);
1775 
1776 	dip = mac_get_nexus_node(mdip, &dln);
1777 	/* defensive */
1778 	if (dip == NULL)
1779 		return (B_FALSE);
1780 
1781 	err = ldi_ident_from_major(ddi_driver_major(dip), &li);
1782 	if (err != 0)
1783 		return (B_FALSE);
1784 
1785 	err = ldi_open_by_name(dln.nexus_path, FREAD|FWRITE, kcred, &lh, li);
1786 	if (err != 0)
1787 		return (B_FALSE);
1788 
1789 	ret = mac_validate_intr(lh, &dln, cpuid);
1790 	if (ret < 0) {
1791 		(void) ldi_close(lh, FREAD|FWRITE, kcred);
1792 		return (B_FALSE);
1793 	}
1794 	/* cmn_note? */
1795 	if (ret != 0)
1796 		if ((err = (mac_set_intr(lh, cpuid, dln.cpu_id, dln.ino)))
1797 		    != 0) {
1798 			(void) ldi_close(lh, FREAD|FWRITE, kcred);
1799 			return (B_FALSE);
1800 		}
1801 	(void) ldi_close(lh, FREAD|FWRITE, kcred);
1802 	return (B_TRUE);
1803 }
1804 
1805 void
mac_client_set_intr_cpu(void * arg,mac_client_handle_t mch,int32_t cpuid)1806 mac_client_set_intr_cpu(void *arg, mac_client_handle_t mch, int32_t cpuid)
1807 {
1808 	dev_info_t		*mdip = (dev_info_t *)arg;
1809 	mac_client_impl_t	*mcip = (mac_client_impl_t *)mch;
1810 	mac_resource_props_t	*mrp;
1811 	mac_perim_handle_t	mph;
1812 	flow_entry_t		*flent = mcip->mci_flent;
1813 	mac_soft_ring_set_t	*rx_srs;
1814 	mac_cpus_t		*srs_cpu;
1815 
1816 	if (!mac_check_interrupt_binding(mdip, cpuid))
1817 		cpuid = -1;
1818 	mac_perim_enter_by_mh((mac_handle_t)mcip->mci_mip, &mph);
1819 	mrp = MCIP_RESOURCE_PROPS(mcip);
1820 	mrp->mrp_rx_intr_cpu = cpuid;
1821 	if (flent != NULL && flent->fe_rx_srs_cnt == 2) {
1822 		rx_srs = flent->fe_rx_srs[1];
1823 		srs_cpu = &rx_srs->srs_cpu;
1824 		srs_cpu->mc_rx_intr_cpu = cpuid;
1825 	}
1826 	mac_perim_exit(mph);
1827 }
1828 
1829 int32_t
mac_client_intr_cpu(mac_client_handle_t mch)1830 mac_client_intr_cpu(mac_client_handle_t mch)
1831 {
1832 	mac_client_impl_t	*mcip = (mac_client_impl_t *)mch;
1833 	mac_cpus_t		*srs_cpu;
1834 	mac_soft_ring_set_t	*rx_srs;
1835 	flow_entry_t		*flent = mcip->mci_flent;
1836 	mac_resource_props_t	*mrp = MCIP_RESOURCE_PROPS(mcip);
1837 	mac_ring_t		*ring;
1838 	mac_intr_t		*mintr;
1839 
1840 	/*
1841 	 * Check if we need to retarget the interrupt. We do this only
1842 	 * for the primary MAC client. We do this if we have the only
1843 	 * exclusive ring in the group.
1844 	 */
1845 	if (mac_is_primary_client(mcip) && flent->fe_rx_srs_cnt == 2) {
1846 		rx_srs = flent->fe_rx_srs[1];
1847 		srs_cpu = &rx_srs->srs_cpu;
1848 		ring = rx_srs->srs_ring;
1849 		mintr = &ring->mr_info.mri_intr;
1850 		/*
1851 		 * If ddi_handle is present or the poll CPU is
1852 		 * already bound to the interrupt CPU, return -1.
1853 		 */
1854 		if (mintr->mi_ddi_handle != NULL ||
1855 		    ((mrp->mrp_ncpus != 0) &&
1856 		    (mrp->mrp_rx_intr_cpu == srs_cpu->mc_rx_pollid))) {
1857 			return (-1);
1858 		}
1859 		return (srs_cpu->mc_rx_pollid);
1860 	}
1861 	return (-1);
1862 }
1863 
1864 void *
mac_get_devinfo(mac_handle_t mh)1865 mac_get_devinfo(mac_handle_t mh)
1866 {
1867 	mac_impl_t	*mip = (mac_impl_t *)mh;
1868 
1869 	return ((void *)mip->mi_dip);
1870 }
1871 
1872 #define	PKT_HASH_2BYTES(x) ((x)[0] ^ (x)[1])
1873 #define	PKT_HASH_4BYTES(x) ((x)[0] ^ (x)[1] ^ (x)[2] ^ (x)[3])
1874 #define	PKT_HASH_MAC(x) ((x)[0] ^ (x)[1] ^ (x)[2] ^ (x)[3] ^ (x)[4] ^ (x)[5])
1875 
1876 uint64_t
mac_pkt_hash(uint_t media,mblk_t * mp,uint8_t policy,boolean_t is_outbound)1877 mac_pkt_hash(uint_t media, mblk_t *mp, uint8_t policy, boolean_t is_outbound)
1878 {
1879 	struct ether_header *ehp;
1880 	uint64_t hash = 0;
1881 	uint16_t sap;
1882 	uint_t skip_len;
1883 	uint8_t proto;
1884 	boolean_t ip_fragmented;
1885 
1886 	/*
1887 	 * We may want to have one of these per MAC type plugin in the
1888 	 * future. For now supports only ethernet.
1889 	 */
1890 	if (media != DL_ETHER)
1891 		return (0L);
1892 
1893 	/* for now we support only outbound packets */
1894 	ASSERT(is_outbound);
1895 	ASSERT(IS_P2ALIGNED(mp->b_rptr, sizeof (uint16_t)));
1896 	ASSERT(MBLKL(mp) >= sizeof (struct ether_header));
1897 
1898 	/* compute L2 hash */
1899 
1900 	ehp = (struct ether_header *)mp->b_rptr;
1901 
1902 	if ((policy & MAC_PKT_HASH_L2) != 0) {
1903 		uchar_t *mac_src = ehp->ether_shost.ether_addr_octet;
1904 		uchar_t *mac_dst = ehp->ether_dhost.ether_addr_octet;
1905 		hash = PKT_HASH_MAC(mac_src) ^ PKT_HASH_MAC(mac_dst);
1906 		policy &= ~MAC_PKT_HASH_L2;
1907 	}
1908 
1909 	if (policy == 0)
1910 		goto done;
1911 
1912 	/* skip ethernet header */
1913 
1914 	sap = ntohs(ehp->ether_type);
1915 	if (sap == ETHERTYPE_VLAN) {
1916 		struct ether_vlan_header *evhp;
1917 		mblk_t *newmp = NULL;
1918 
1919 		skip_len = sizeof (struct ether_vlan_header);
1920 		if (MBLKL(mp) < skip_len) {
1921 			/* the vlan tag is the payload, pull up first */
1922 			newmp = msgpullup(mp, -1);
1923 			if ((newmp == NULL) || (MBLKL(newmp) < skip_len)) {
1924 				goto done;
1925 			}
1926 			evhp = (struct ether_vlan_header *)newmp->b_rptr;
1927 		} else {
1928 			evhp = (struct ether_vlan_header *)mp->b_rptr;
1929 		}
1930 
1931 		sap = ntohs(evhp->ether_type);
1932 		freemsg(newmp);
1933 	} else {
1934 		skip_len = sizeof (struct ether_header);
1935 	}
1936 
1937 	/* if ethernet header is in its own mblk, skip it */
1938 	if (MBLKL(mp) <= skip_len) {
1939 		skip_len -= MBLKL(mp);
1940 		mp = mp->b_cont;
1941 		if (mp == NULL)
1942 			goto done;
1943 	}
1944 
1945 	sap = (sap < ETHERTYPE_802_MIN) ? 0 : sap;
1946 
1947 	/* compute IP src/dst addresses hash and skip IPv{4,6} header */
1948 
1949 	switch (sap) {
1950 	case ETHERTYPE_IP: {
1951 		ipha_t *iphp;
1952 
1953 		/*
1954 		 * If the header is not aligned or the header doesn't fit
1955 		 * in the mblk, bail now. Note that this may cause packets
1956 		 * reordering.
1957 		 */
1958 		iphp = (ipha_t *)(mp->b_rptr + skip_len);
1959 		if (((unsigned char *)iphp + sizeof (ipha_t) > mp->b_wptr) ||
1960 		    !OK_32PTR((char *)iphp))
1961 			goto done;
1962 
1963 		proto = iphp->ipha_protocol;
1964 		skip_len += IPH_HDR_LENGTH(iphp);
1965 
1966 		/* Check if the packet is fragmented. */
1967 		ip_fragmented = ntohs(iphp->ipha_fragment_offset_and_flags) &
1968 		    IPH_OFFSET;
1969 
1970 		/*
1971 		 * For fragmented packets, use addresses in addition to
1972 		 * the frag_id to generate the hash inorder to get
1973 		 * better distribution.
1974 		 */
1975 		if (ip_fragmented || (policy & MAC_PKT_HASH_L3) != 0) {
1976 			uint8_t *ip_src = (uint8_t *)&(iphp->ipha_src);
1977 			uint8_t *ip_dst = (uint8_t *)&(iphp->ipha_dst);
1978 
1979 			hash ^= (PKT_HASH_4BYTES(ip_src) ^
1980 			    PKT_HASH_4BYTES(ip_dst));
1981 			policy &= ~MAC_PKT_HASH_L3;
1982 		}
1983 
1984 		if (ip_fragmented) {
1985 			uint8_t *identp = (uint8_t *)&iphp->ipha_ident;
1986 			hash ^= PKT_HASH_2BYTES(identp);
1987 			goto done;
1988 		}
1989 		break;
1990 	}
1991 	case ETHERTYPE_IPV6: {
1992 		ip6_t *ip6hp;
1993 		ip6_frag_t *frag = NULL;
1994 		uint16_t hdr_length;
1995 
1996 		/*
1997 		 * If the header is not aligned or the header doesn't fit
1998 		 * in the mblk, bail now. Note that this may cause packets
1999 		 * reordering.
2000 		 */
2001 
2002 		ip6hp = (ip6_t *)(mp->b_rptr + skip_len);
2003 		if (((unsigned char *)ip6hp + IPV6_HDR_LEN > mp->b_wptr) ||
2004 		    !OK_32PTR((char *)ip6hp))
2005 			goto done;
2006 
2007 		if (!mac_ip_hdr_length_v6(ip6hp, mp->b_wptr, &hdr_length,
2008 		    &proto, &frag))
2009 			goto done;
2010 		skip_len += hdr_length;
2011 
2012 		/*
2013 		 * For fragmented packets, use addresses in addition to
2014 		 * the frag_id to generate the hash inorder to get
2015 		 * better distribution.
2016 		 */
2017 		if (frag != NULL || (policy & MAC_PKT_HASH_L3) != 0) {
2018 			uint8_t *ip_src = &(ip6hp->ip6_src.s6_addr8[12]);
2019 			uint8_t *ip_dst = &(ip6hp->ip6_dst.s6_addr8[12]);
2020 
2021 			hash ^= (PKT_HASH_4BYTES(ip_src) ^
2022 			    PKT_HASH_4BYTES(ip_dst));
2023 			policy &= ~MAC_PKT_HASH_L3;
2024 		}
2025 
2026 		if (frag != NULL) {
2027 			uint8_t *identp = (uint8_t *)&frag->ip6f_ident;
2028 			hash ^= PKT_HASH_4BYTES(identp);
2029 			goto done;
2030 		}
2031 		break;
2032 	}
2033 	default:
2034 		goto done;
2035 	}
2036 
2037 	if (policy == 0)
2038 		goto done;
2039 
2040 	/* if ip header is in its own mblk, skip it */
2041 	if (MBLKL(mp) <= skip_len) {
2042 		skip_len -= MBLKL(mp);
2043 		mp = mp->b_cont;
2044 		if (mp == NULL)
2045 			goto done;
2046 	}
2047 
2048 	/* parse ULP header */
2049 again:
2050 	switch (proto) {
2051 	case IPPROTO_TCP:
2052 	case IPPROTO_UDP:
2053 	case IPPROTO_ESP:
2054 	case IPPROTO_SCTP:
2055 		/*
2056 		 * These Internet Protocols are intentionally designed
2057 		 * for hashing from the git-go.  Port numbers are in the first
2058 		 * word for transports, SPI is first for ESP.
2059 		 */
2060 		if (mp->b_rptr + skip_len + 4 > mp->b_wptr)
2061 			goto done;
2062 		hash ^= PKT_HASH_4BYTES((mp->b_rptr + skip_len));
2063 		break;
2064 
2065 	case IPPROTO_AH: {
2066 		ah_t *ah = (ah_t *)(mp->b_rptr + skip_len);
2067 		uint_t ah_length = AH_TOTAL_LEN(ah);
2068 
2069 		if ((unsigned char *)ah + sizeof (ah_t) > mp->b_wptr)
2070 			goto done;
2071 
2072 		proto = ah->ah_nexthdr;
2073 		skip_len += ah_length;
2074 
2075 		/* if AH header is in its own mblk, skip it */
2076 		if (MBLKL(mp) <= skip_len) {
2077 			skip_len -= MBLKL(mp);
2078 			mp = mp->b_cont;
2079 			if (mp == NULL)
2080 				goto done;
2081 		}
2082 
2083 		goto again;
2084 	}
2085 	}
2086 
2087 done:
2088 	return (hash);
2089 }
2090