1 // SPDX-License-Identifier: GPL-2.0-or-later 2 /* 3 * NET3: Implementation of the ICMP protocol layer. 4 * 5 * Alan Cox, <alan@lxorguk.ukuu.org.uk> 6 * 7 * Some of the function names and the icmp unreach table for this 8 * module were derived from [icmp.c 1.0.11 06/02/93] by 9 * Ross Biro, Fred N. van Kempen, Mark Evans, Alan Cox, Gerhard Koerting. 10 * Other than that this module is a complete rewrite. 11 * 12 * Fixes: 13 * Clemens Fruhwirth : introduce global icmp rate limiting 14 * with icmp type masking ability instead 15 * of broken per type icmp timeouts. 16 * Mike Shaver : RFC1122 checks. 17 * Alan Cox : Multicast ping reply as self. 18 * Alan Cox : Fix atomicity lockup in ip_build_xmit 19 * call. 20 * Alan Cox : Added 216,128 byte paths to the MTU 21 * code. 22 * Martin Mares : RFC1812 checks. 23 * Martin Mares : Can be configured to follow redirects 24 * if acting as a router _without_ a 25 * routing protocol (RFC 1812). 26 * Martin Mares : Echo requests may be configured to 27 * be ignored (RFC 1812). 28 * Martin Mares : Limitation of ICMP error message 29 * transmit rate (RFC 1812). 30 * Martin Mares : TOS and Precedence set correctly 31 * (RFC 1812). 32 * Martin Mares : Now copying as much data from the 33 * original packet as we can without 34 * exceeding 576 bytes (RFC 1812). 35 * Willy Konynenberg : Transparent proxying support. 36 * Keith Owens : RFC1191 correction for 4.2BSD based 37 * path MTU bug. 38 * Thomas Quinot : ICMP Dest Unreach codes up to 15 are 39 * valid (RFC 1812). 40 * Andi Kleen : Check all packet lengths properly 41 * and moved all kfree_skb() up to 42 * icmp_rcv. 43 * Andi Kleen : Move the rate limit bookkeeping 44 * into the dest entry and use a token 45 * bucket filter (thanks to ANK). Make 46 * the rates sysctl configurable. 47 * Yu Tianli : Fixed two ugly bugs in icmp_send 48 * - IP option length was accounted wrongly 49 * - ICMP header length was not accounted 50 * at all. 51 * Tristan Greaves : Added sysctl option to ignore bogus 52 * broadcast responses from broken routers. 53 * 54 * To Fix: 55 * 56 * - Should use skb_pull() instead of all the manual checking. 57 * This would also greatly simply some upper layer error handlers. --AK 58 */ 59 60 #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt 61 62 #include <linux/module.h> 63 #include <linux/types.h> 64 #include <linux/jiffies.h> 65 #include <linux/kernel.h> 66 #include <linux/fcntl.h> 67 #include <linux/nospec.h> 68 #include <linux/socket.h> 69 #include <linux/in.h> 70 #include <linux/inet.h> 71 #include <linux/inetdevice.h> 72 #include <linux/netdevice.h> 73 #include <linux/string.h> 74 #include <linux/netfilter_ipv4.h> 75 #include <linux/slab.h> 76 #include <net/flow.h> 77 #include <net/snmp.h> 78 #include <net/ip.h> 79 #include <net/route.h> 80 #include <net/protocol.h> 81 #include <net/icmp.h> 82 #include <net/tcp.h> 83 #include <net/udp.h> 84 #include <net/raw.h> 85 #include <net/ping.h> 86 #include <linux/skbuff.h> 87 #include <net/sock.h> 88 #include <linux/errno.h> 89 #include <linux/timer.h> 90 #include <linux/init.h> 91 #include <linux/uaccess.h> 92 #include <net/checksum.h> 93 #include <net/xfrm.h> 94 #include <net/inet_common.h> 95 #include <net/ip_fib.h> 96 #include <net/l3mdev.h> 97 #include <net/addrconf.h> 98 #include <net/inet_dscp.h> 99 #define CREATE_TRACE_POINTS 100 #include <trace/events/icmp.h> 101 102 /* 103 * Build xmit assembly blocks 104 */ 105 106 struct icmp_bxm { 107 struct sk_buff *skb; 108 int offset; 109 int data_len; 110 111 struct { 112 struct icmphdr icmph; 113 __be32 times[3]; 114 } data; 115 int head_len; 116 117 /* Must be last as it ends in a flexible-array member. */ 118 struct ip_options_rcu replyopts; 119 }; 120 121 /* An array of errno for error messages from dest unreach. */ 122 /* RFC 1122: 3.2.2.1 States that NET_UNREACH, HOST_UNREACH and SR_FAILED MUST be considered 'transient errs'. */ 123 124 const struct icmp_err icmp_err_convert[] = { 125 { 126 .errno = ENETUNREACH, /* ICMP_NET_UNREACH */ 127 .fatal = 0, 128 }, 129 { 130 .errno = EHOSTUNREACH, /* ICMP_HOST_UNREACH */ 131 .fatal = 0, 132 }, 133 { 134 .errno = ENOPROTOOPT /* ICMP_PROT_UNREACH */, 135 .fatal = 1, 136 }, 137 { 138 .errno = ECONNREFUSED, /* ICMP_PORT_UNREACH */ 139 .fatal = 1, 140 }, 141 { 142 .errno = EMSGSIZE, /* ICMP_FRAG_NEEDED */ 143 .fatal = 0, 144 }, 145 { 146 .errno = EOPNOTSUPP, /* ICMP_SR_FAILED */ 147 .fatal = 0, 148 }, 149 { 150 .errno = ENETUNREACH, /* ICMP_NET_UNKNOWN */ 151 .fatal = 1, 152 }, 153 { 154 .errno = EHOSTDOWN, /* ICMP_HOST_UNKNOWN */ 155 .fatal = 1, 156 }, 157 { 158 .errno = ENONET, /* ICMP_HOST_ISOLATED */ 159 .fatal = 1, 160 }, 161 { 162 .errno = ENETUNREACH, /* ICMP_NET_ANO */ 163 .fatal = 1, 164 }, 165 { 166 .errno = EHOSTUNREACH, /* ICMP_HOST_ANO */ 167 .fatal = 1, 168 }, 169 { 170 .errno = ENETUNREACH, /* ICMP_NET_UNR_TOS */ 171 .fatal = 0, 172 }, 173 { 174 .errno = EHOSTUNREACH, /* ICMP_HOST_UNR_TOS */ 175 .fatal = 0, 176 }, 177 { 178 .errno = EHOSTUNREACH, /* ICMP_PKT_FILTERED */ 179 .fatal = 1, 180 }, 181 { 182 .errno = EHOSTUNREACH, /* ICMP_PREC_VIOLATION */ 183 .fatal = 1, 184 }, 185 { 186 .errno = EHOSTUNREACH, /* ICMP_PREC_CUTOFF */ 187 .fatal = 1, 188 }, 189 }; 190 EXPORT_SYMBOL(icmp_err_convert); 191 192 /* 193 * ICMP control array. This specifies what to do with each ICMP. 194 */ 195 196 struct icmp_control { 197 enum skb_drop_reason (*handler)(struct sk_buff *skb); 198 short error; /* This ICMP is classed as an error message */ 199 }; 200 201 static const struct icmp_control icmp_pointers[NR_ICMP_TYPES+1]; 202 203 static DEFINE_PER_CPU(struct sock *, ipv4_icmp_sk); 204 205 /* Called with BH disabled */ 206 static inline struct sock *icmp_xmit_lock(struct net *net) 207 { 208 struct sock *sk; 209 210 sk = this_cpu_read(ipv4_icmp_sk); 211 212 if (unlikely(!spin_trylock(&sk->sk_lock.slock))) { 213 /* This can happen if the output path signals a 214 * dst_link_failure() for an outgoing ICMP packet. 215 */ 216 return NULL; 217 } 218 sock_net_set(sk, net); 219 return sk; 220 } 221 222 static inline void icmp_xmit_unlock(struct sock *sk) 223 { 224 sock_net_set(sk, &init_net); 225 spin_unlock(&sk->sk_lock.slock); 226 } 227 228 /** 229 * icmp_global_allow - Are we allowed to send one more ICMP message ? 230 * @net: network namespace 231 * 232 * Uses a token bucket to limit our ICMP messages to ~sysctl_icmp_msgs_per_sec. 233 * Returns false if we reached the limit and can not send another packet. 234 * Works in tandem with icmp_global_consume(). 235 */ 236 bool icmp_global_allow(struct net *net) 237 { 238 u32 delta, now, oldstamp; 239 int incr, new, old; 240 241 /* Note: many cpus could find this condition true. 242 * Then later icmp_global_consume() could consume more credits, 243 * this is an acceptable race. 244 */ 245 if (atomic_read(&net->ipv4.icmp_global_credit) > 0) 246 return true; 247 248 now = jiffies; 249 oldstamp = READ_ONCE(net->ipv4.icmp_global_stamp); 250 delta = min_t(u32, now - oldstamp, HZ); 251 if (delta < HZ / 50) 252 return false; 253 254 incr = READ_ONCE(net->ipv4.sysctl_icmp_msgs_per_sec); 255 incr = div_u64((u64)incr * delta, HZ); 256 if (!incr) 257 return false; 258 259 if (cmpxchg(&net->ipv4.icmp_global_stamp, oldstamp, now) == oldstamp) { 260 old = atomic_read(&net->ipv4.icmp_global_credit); 261 do { 262 new = min(old + incr, READ_ONCE(net->ipv4.sysctl_icmp_msgs_burst)); 263 } while (!atomic_try_cmpxchg(&net->ipv4.icmp_global_credit, &old, new)); 264 } 265 return true; 266 } 267 268 void icmp_global_consume(struct net *net) 269 { 270 int credits = get_random_u32_below(3); 271 272 /* Note: this might make icmp_global.credit negative. */ 273 if (credits) 274 atomic_sub(credits, &net->ipv4.icmp_global_credit); 275 } 276 277 static bool icmpv4_mask_allow(struct net *net, int type, int code) 278 { 279 if (type > NR_ICMP_TYPES) 280 return true; 281 282 /* Don't limit PMTU discovery. */ 283 if (type == ICMP_DEST_UNREACH && code == ICMP_FRAG_NEEDED) 284 return true; 285 286 /* Limit if icmp type is enabled in ratemask. */ 287 if (!((1 << type) & READ_ONCE(net->ipv4.sysctl_icmp_ratemask))) 288 return true; 289 290 return false; 291 } 292 293 static bool icmpv4_global_allow(struct net *net, int type, int code, 294 bool *apply_ratelimit) 295 { 296 if (icmpv4_mask_allow(net, type, code)) 297 return true; 298 299 if (icmp_global_allow(net)) { 300 *apply_ratelimit = true; 301 return true; 302 } 303 __ICMP_INC_STATS(net, ICMP_MIB_RATELIMITGLOBAL); 304 return false; 305 } 306 307 /* 308 * Send an ICMP frame. 309 */ 310 311 static bool icmpv4_xrlim_allow(struct net *net, struct rtable *rt, 312 struct flowi4 *fl4, int type, int code, 313 bool apply_ratelimit) 314 { 315 struct dst_entry *dst = &rt->dst; 316 struct inet_peer *peer; 317 struct net_device *dev; 318 int peer_timeout; 319 bool rc = true; 320 321 if (!apply_ratelimit) 322 return true; 323 324 peer_timeout = READ_ONCE(net->ipv4.sysctl_icmp_ratelimit); 325 if (!peer_timeout) 326 goto out; 327 328 /* No rate limit on loopback */ 329 rcu_read_lock(); 330 dev = dst_dev_rcu(dst); 331 if (dev && (dev->flags & IFF_LOOPBACK)) 332 goto out_unlock; 333 334 peer = inet_getpeer_v4(net->ipv4.peers, fl4->daddr, 335 l3mdev_master_ifindex_rcu(dev)); 336 rc = inet_peer_xrlim_allow(peer, peer_timeout); 337 338 out_unlock: 339 rcu_read_unlock(); 340 out: 341 if (!rc) 342 __ICMP_INC_STATS(net, ICMP_MIB_RATELIMITHOST); 343 else 344 icmp_global_consume(net); 345 return rc; 346 } 347 348 /* 349 * Maintain the counters used in the SNMP statistics for outgoing ICMP 350 */ 351 void icmp_out_count(struct net *net, unsigned char type) 352 { 353 ICMPMSGOUT_INC_STATS(net, type); 354 ICMP_INC_STATS(net, ICMP_MIB_OUTMSGS); 355 } 356 357 /* 358 * Checksum each fragment, and on the first include the headers and final 359 * checksum. 360 */ 361 static int icmp_glue_bits(void *from, char *to, int offset, int len, int odd, 362 struct sk_buff *skb) 363 { 364 DEFINE_RAW_FLEX(struct icmp_bxm, icmp_param, replyopts.opt.__data, 365 IP_OPTIONS_DATA_FIXED_SIZE); 366 __wsum csum; 367 368 icmp_param = from; 369 370 csum = skb_copy_and_csum_bits(icmp_param->skb, 371 icmp_param->offset + offset, 372 to, len); 373 374 skb->csum = csum_block_add(skb->csum, csum, odd); 375 if (icmp_param->data.icmph.type <= NR_ICMP_TYPES && 376 icmp_pointers[array_index_nospec(icmp_param->data.icmph.type, 377 NR_ICMP_TYPES + 1)].error) 378 nf_ct_attach(skb, icmp_param->skb); 379 return 0; 380 } 381 382 static void icmp_push_reply(struct sock *sk, 383 struct icmp_bxm *icmp_param, 384 struct flowi4 *fl4, 385 struct ipcm_cookie *ipc, struct rtable **rt) 386 { 387 struct sk_buff *skb; 388 389 if (ip_append_data(sk, fl4, icmp_glue_bits, icmp_param, 390 icmp_param->data_len+icmp_param->head_len, 391 icmp_param->head_len, 392 ipc, rt, MSG_DONTWAIT) < 0) { 393 __ICMP_INC_STATS(sock_net(sk), ICMP_MIB_OUTERRORS); 394 ip_flush_pending_frames(sk); 395 } else if ((skb = skb_peek(&sk->sk_write_queue)) != NULL) { 396 struct icmphdr *icmph = icmp_hdr(skb); 397 __wsum csum; 398 struct sk_buff *skb1; 399 400 csum = csum_partial_copy_nocheck((void *)&icmp_param->data, 401 (char *)icmph, 402 icmp_param->head_len); 403 skb_queue_walk(&sk->sk_write_queue, skb1) { 404 csum = csum_add(csum, skb1->csum); 405 } 406 icmph->checksum = csum_fold(csum); 407 skb->ip_summed = CHECKSUM_NONE; 408 ip_push_pending_frames(sk, fl4); 409 } 410 } 411 412 /* 413 * Driving logic for building and sending ICMP messages. 414 */ 415 416 static void icmp_reply(struct icmp_bxm *icmp_param, struct sk_buff *skb) 417 { 418 struct rtable *rt = skb_rtable(skb); 419 struct net *net = dev_net_rcu(rt->dst.dev); 420 bool apply_ratelimit = false; 421 struct ipcm_cookie ipc; 422 struct flowi4 fl4; 423 struct sock *sk; 424 __be32 daddr, saddr; 425 u32 mark = IP4_REPLY_MARK(net, skb->mark); 426 int type = icmp_param->data.icmph.type; 427 int code = icmp_param->data.icmph.code; 428 429 if (ip_options_echo(net, &icmp_param->replyopts.opt, skb)) 430 return; 431 432 /* Needed by both icmpv4_global_allow and icmp_xmit_lock */ 433 local_bh_disable(); 434 435 /* is global icmp_msgs_per_sec exhausted ? */ 436 if (!icmpv4_global_allow(net, type, code, &apply_ratelimit)) 437 goto out_bh_enable; 438 439 sk = icmp_xmit_lock(net); 440 if (!sk) 441 goto out_bh_enable; 442 443 icmp_param->data.icmph.checksum = 0; 444 445 ipcm_init(&ipc); 446 ipc.tos = ip_hdr(skb)->tos; 447 ipc.sockc.mark = mark; 448 daddr = ipc.addr = ip_hdr(skb)->saddr; 449 saddr = fib_compute_spec_dst(skb); 450 451 if (icmp_param->replyopts.opt.optlen) { 452 ipc.opt = &icmp_param->replyopts; 453 if (ipc.opt->opt.srr) 454 daddr = icmp_param->replyopts.opt.faddr; 455 } 456 memset(&fl4, 0, sizeof(fl4)); 457 fl4.daddr = daddr; 458 fl4.saddr = saddr; 459 fl4.flowi4_mark = mark; 460 fl4.flowi4_uid = sock_net_uid(net, NULL); 461 fl4.flowi4_dscp = ip4h_dscp(ip_hdr(skb)); 462 fl4.flowi4_proto = IPPROTO_ICMP; 463 fl4.flowi4_oif = l3mdev_master_ifindex(skb->dev); 464 security_skb_classify_flow(skb, flowi4_to_flowi_common(&fl4)); 465 rt = ip_route_output_key(net, &fl4); 466 if (IS_ERR(rt)) 467 goto out_unlock; 468 if (icmpv4_xrlim_allow(net, rt, &fl4, type, code, apply_ratelimit)) 469 icmp_push_reply(sk, icmp_param, &fl4, &ipc, &rt); 470 ip_rt_put(rt); 471 out_unlock: 472 icmp_xmit_unlock(sk); 473 out_bh_enable: 474 local_bh_enable(); 475 } 476 477 /* 478 * The device used for looking up which routing table to use for sending an ICMP 479 * error is preferably the source whenever it is set, which should ensure the 480 * icmp error can be sent to the source host, else lookup using the routing 481 * table of the destination device, else use the main routing table (index 0). 482 */ 483 static struct net_device *icmp_get_route_lookup_dev(struct sk_buff *skb) 484 { 485 struct net_device *dev = skb->dev; 486 const struct dst_entry *dst; 487 488 if (dev) 489 return dev; 490 dst = skb_dst(skb); 491 return dst ? dst_dev(dst) : NULL; 492 } 493 494 static struct rtable *icmp_route_lookup(struct net *net, struct flowi4 *fl4, 495 struct sk_buff *skb_in, 496 const struct iphdr *iph, __be32 saddr, 497 dscp_t dscp, u32 mark, int type, 498 int code, struct icmp_bxm *param) 499 { 500 struct net_device *route_lookup_dev; 501 struct dst_entry *dst, *dst2; 502 struct rtable *rt, *rt2; 503 struct flowi4 fl4_dec; 504 int err; 505 506 memset(fl4, 0, sizeof(*fl4)); 507 fl4->daddr = (param->replyopts.opt.srr ? 508 param->replyopts.opt.faddr : iph->saddr); 509 fl4->saddr = saddr; 510 fl4->flowi4_mark = mark; 511 fl4->flowi4_uid = sock_net_uid(net, NULL); 512 fl4->flowi4_dscp = dscp; 513 fl4->flowi4_proto = IPPROTO_ICMP; 514 fl4->fl4_icmp_type = type; 515 fl4->fl4_icmp_code = code; 516 route_lookup_dev = icmp_get_route_lookup_dev(skb_in); 517 fl4->flowi4_oif = l3mdev_master_ifindex(route_lookup_dev); 518 519 security_skb_classify_flow(skb_in, flowi4_to_flowi_common(fl4)); 520 rt = ip_route_output_key_hash(net, fl4, skb_in); 521 if (IS_ERR(rt)) 522 return rt; 523 524 /* No need to clone since we're just using its address. */ 525 rt2 = rt; 526 527 dst = xfrm_lookup(net, &rt->dst, 528 flowi4_to_flowi(fl4), NULL, 0); 529 rt = dst_rtable(dst); 530 if (!IS_ERR(dst)) { 531 if (rt != rt2) 532 return rt; 533 if (inet_addr_type_dev_table(net, route_lookup_dev, 534 fl4->daddr) == RTN_LOCAL) 535 return rt; 536 } else if (PTR_ERR(dst) == -EPERM) { 537 rt = NULL; 538 } else { 539 return rt; 540 } 541 err = xfrm_decode_session_reverse(net, skb_in, flowi4_to_flowi(&fl4_dec), AF_INET); 542 if (err) 543 goto relookup_failed; 544 545 if (inet_addr_type_dev_table(net, route_lookup_dev, 546 fl4_dec.saddr) == RTN_LOCAL) { 547 rt2 = __ip_route_output_key(net, &fl4_dec); 548 if (IS_ERR(rt2)) 549 err = PTR_ERR(rt2); 550 } else { 551 struct flowi4 fl4_2 = fl4_dec; 552 unsigned long orefdst; 553 554 swap(fl4_2.daddr, fl4_2.saddr); 555 switch (fl4_2.flowi4_proto) { 556 case IPPROTO_TCP: 557 case IPPROTO_UDP: 558 case IPPROTO_SCTP: 559 case IPPROTO_DCCP: 560 swap(fl4_2.fl4_sport, fl4_2.fl4_dport); 561 break; 562 } 563 564 fl4_2.flowi4_oif = l3mdev_master_ifindex(route_lookup_dev); 565 fl4_2.flowi4_flags |= FLOWI_FLAG_ANYSRC; 566 567 rt2 = __ip_route_output_key(net, &fl4_2); 568 if (IS_ERR(rt2)) { 569 err = PTR_ERR(rt2); 570 goto relookup_failed; 571 } 572 /* Ugh! */ 573 orefdst = skb_dstref_steal(skb_in); 574 err = ip_route_input(skb_in, fl4_dec.daddr, fl4_dec.saddr, 575 dscp, rt2->dst.dev) ? -EINVAL : 0; 576 577 dst_release(&rt2->dst); 578 rt2 = skb_rtable(skb_in); 579 /* steal dst entry from skb_in, don't drop refcnt */ 580 skb_dstref_steal(skb_in); 581 skb_dstref_restore(skb_in, orefdst); 582 583 /* 584 * At this point, fl4_dec.daddr should NOT be local (we 585 * checked fl4_dec.saddr above). However, a race condition 586 * may occur if the address is added to the interface 587 * concurrently. In that case, ip_route_input() returns a 588 * LOCAL route with dst.output=ip_rt_bug, which must not 589 * be used for output. 590 */ 591 if (!err && rt2 && rt2->rt_type == RTN_LOCAL) { 592 net_warn_ratelimited("detected local route for %pI4 during ICMP sending, src %pI4\n", 593 &fl4_dec.daddr, &fl4_dec.saddr); 594 dst_release(&rt2->dst); 595 err = -EINVAL; 596 } 597 } 598 599 if (err) 600 goto relookup_failed; 601 602 dst2 = xfrm_lookup(net, &rt2->dst, flowi4_to_flowi(&fl4_dec), NULL, 603 XFRM_LOOKUP_ICMP); 604 rt2 = dst_rtable(dst2); 605 if (!IS_ERR(dst2)) { 606 dst_release(&rt->dst); 607 rt = rt2; 608 } else if (PTR_ERR(dst2) == -EPERM) { 609 if (rt) 610 dst_release(&rt->dst); 611 return rt2; 612 } else { 613 err = PTR_ERR(dst2); 614 goto relookup_failed; 615 } 616 return rt; 617 618 relookup_failed: 619 if (rt) 620 return rt; 621 return ERR_PTR(err); 622 } 623 624 struct icmp_ext_iio_addr4_subobj { 625 __be16 afi; 626 __be16 reserved; 627 __be32 addr4; 628 }; 629 630 static unsigned int icmp_ext_iio_len(void) 631 { 632 return sizeof(struct icmp_extobj_hdr) + 633 /* ifIndex */ 634 sizeof(__be32) + 635 /* Interface Address Sub-Object */ 636 sizeof(struct icmp_ext_iio_addr4_subobj) + 637 /* Interface Name Sub-Object. Length must be a multiple of 4 638 * bytes. 639 */ 640 ALIGN(sizeof(struct icmp_ext_iio_name_subobj), 4) + 641 /* MTU */ 642 sizeof(__be32); 643 } 644 645 static unsigned int icmp_ext_max_len(u8 ext_objs) 646 { 647 unsigned int ext_max_len; 648 649 ext_max_len = sizeof(struct icmp_ext_hdr); 650 651 if (ext_objs & BIT(ICMP_ERR_EXT_IIO_IIF)) 652 ext_max_len += icmp_ext_iio_len(); 653 654 return ext_max_len; 655 } 656 657 static __be32 icmp_ext_iio_addr4_find(const struct net_device *dev) 658 { 659 struct in_device *in_dev; 660 struct in_ifaddr *ifa; 661 662 in_dev = __in_dev_get_rcu(dev); 663 if (!in_dev) 664 return 0; 665 666 /* It is unclear from RFC 5837 which IP address should be chosen, but 667 * it makes sense to choose a global unicast address. 668 */ 669 in_dev_for_each_ifa_rcu(ifa, in_dev) { 670 if (READ_ONCE(ifa->ifa_flags) & IFA_F_SECONDARY) 671 continue; 672 if (ifa->ifa_scope != RT_SCOPE_UNIVERSE || 673 ipv4_is_multicast(ifa->ifa_address)) 674 continue; 675 return ifa->ifa_address; 676 } 677 678 return 0; 679 } 680 681 static void icmp_ext_iio_iif_append(struct net *net, struct sk_buff *skb, 682 int iif) 683 { 684 struct icmp_ext_iio_name_subobj *name_subobj; 685 struct icmp_extobj_hdr *objh; 686 struct net_device *dev; 687 __be32 data; 688 689 if (!iif) 690 return; 691 692 /* Add the fields in the order specified by RFC 5837. */ 693 objh = skb_put(skb, sizeof(*objh)); 694 objh->class_num = ICMP_EXT_OBJ_CLASS_IIO; 695 objh->class_type = ICMP_EXT_CTYPE_IIO_ROLE(ICMP_EXT_CTYPE_IIO_ROLE_IIF); 696 697 data = htonl(iif); 698 skb_put_data(skb, &data, sizeof(__be32)); 699 objh->class_type |= ICMP_EXT_CTYPE_IIO_IFINDEX; 700 701 rcu_read_lock(); 702 703 dev = dev_get_by_index_rcu(net, iif); 704 if (!dev) 705 goto out; 706 707 data = icmp_ext_iio_addr4_find(dev); 708 if (data) { 709 struct icmp_ext_iio_addr4_subobj *addr4_subobj; 710 711 addr4_subobj = skb_put_zero(skb, sizeof(*addr4_subobj)); 712 addr4_subobj->afi = htons(ICMP_AFI_IP); 713 addr4_subobj->addr4 = data; 714 objh->class_type |= ICMP_EXT_CTYPE_IIO_IPADDR; 715 } 716 717 name_subobj = skb_put_zero(skb, ALIGN(sizeof(*name_subobj), 4)); 718 name_subobj->len = ALIGN(sizeof(*name_subobj), 4); 719 netdev_copy_name(dev, name_subobj->name); 720 objh->class_type |= ICMP_EXT_CTYPE_IIO_NAME; 721 722 data = htonl(READ_ONCE(dev->mtu)); 723 skb_put_data(skb, &data, sizeof(__be32)); 724 objh->class_type |= ICMP_EXT_CTYPE_IIO_MTU; 725 726 out: 727 rcu_read_unlock(); 728 objh->length = htons(skb_tail_pointer(skb) - (unsigned char *)objh); 729 } 730 731 static void icmp_ext_objs_append(struct net *net, struct sk_buff *skb, 732 u8 ext_objs, int iif) 733 { 734 if (ext_objs & BIT(ICMP_ERR_EXT_IIO_IIF)) 735 icmp_ext_iio_iif_append(net, skb, iif); 736 } 737 738 static struct sk_buff * 739 icmp_ext_append(struct net *net, struct sk_buff *skb_in, struct icmphdr *icmph, 740 unsigned int room, int iif) 741 { 742 unsigned int payload_len, ext_max_len, ext_len; 743 struct icmp_ext_hdr *ext_hdr; 744 struct sk_buff *skb; 745 u8 ext_objs; 746 int nhoff; 747 748 switch (icmph->type) { 749 case ICMP_DEST_UNREACH: 750 case ICMP_TIME_EXCEEDED: 751 case ICMP_PARAMETERPROB: 752 break; 753 default: 754 return NULL; 755 } 756 757 ext_objs = READ_ONCE(net->ipv4.sysctl_icmp_errors_extension_mask); 758 if (!ext_objs) 759 return NULL; 760 761 ext_max_len = icmp_ext_max_len(ext_objs); 762 if (ICMP_EXT_ORIG_DGRAM_MIN_LEN + ext_max_len > room) 763 return NULL; 764 765 skb = skb_clone(skb_in, GFP_ATOMIC); 766 if (!skb) 767 return NULL; 768 769 nhoff = skb_network_offset(skb); 770 payload_len = min(skb->len - nhoff, ICMP_EXT_ORIG_DGRAM_MIN_LEN); 771 772 if (!pskb_network_may_pull(skb, payload_len)) 773 goto free_skb; 774 775 if (pskb_trim(skb, nhoff + ICMP_EXT_ORIG_DGRAM_MIN_LEN) || 776 __skb_put_padto(skb, nhoff + ICMP_EXT_ORIG_DGRAM_MIN_LEN, false)) 777 goto free_skb; 778 779 if (pskb_expand_head(skb, 0, ext_max_len, GFP_ATOMIC)) 780 goto free_skb; 781 782 ext_hdr = skb_put_zero(skb, sizeof(*ext_hdr)); 783 ext_hdr->version = ICMP_EXT_VERSION_2; 784 785 icmp_ext_objs_append(net, skb, ext_objs, iif); 786 787 /* Do not send an empty extension structure. */ 788 ext_len = skb_tail_pointer(skb) - (unsigned char *)ext_hdr; 789 if (ext_len == sizeof(*ext_hdr)) 790 goto free_skb; 791 792 ext_hdr->checksum = ip_compute_csum(ext_hdr, ext_len); 793 /* The length of the original datagram in 32-bit words (RFC 4884). */ 794 icmph->un.reserved[1] = ICMP_EXT_ORIG_DGRAM_MIN_LEN / sizeof(u32); 795 796 return skb; 797 798 free_skb: 799 consume_skb(skb); 800 return NULL; 801 } 802 803 /* 804 * Send an ICMP message in response to a situation 805 * 806 * RFC 1122: 3.2.2 MUST send at least the IP header and 8 bytes of header. 807 * MAY send more (we do). 808 * MUST NOT change this header information. 809 * MUST NOT reply to a multicast/broadcast IP address. 810 * MUST NOT reply to a multicast/broadcast MAC address. 811 * MUST reply to only the first fragment. 812 */ 813 814 void __icmp_send(struct sk_buff *skb_in, int type, int code, __be32 info, 815 const struct inet_skb_parm *parm) 816 { 817 DEFINE_RAW_FLEX(struct icmp_bxm, icmp_param, replyopts.opt.__data, 818 IP_OPTIONS_DATA_FIXED_SIZE); 819 struct iphdr *iph; 820 int room; 821 struct rtable *rt = skb_rtable(skb_in); 822 bool apply_ratelimit = false; 823 struct sk_buff *ext_skb; 824 struct ipcm_cookie ipc; 825 struct flowi4 fl4; 826 __be32 saddr; 827 u8 tos; 828 u32 mark; 829 struct net *net; 830 struct sock *sk; 831 832 if (!rt) 833 return; 834 835 rcu_read_lock(); 836 837 if (rt->dst.dev) 838 net = dev_net_rcu(rt->dst.dev); 839 else if (skb_in->dev) 840 net = dev_net_rcu(skb_in->dev); 841 else 842 goto out; 843 844 /* 845 * Find the original header. It is expected to be valid, of course. 846 * Check this, icmp_send is called from the most obscure devices 847 * sometimes. 848 */ 849 iph = ip_hdr(skb_in); 850 851 if ((u8 *)iph < skb_in->head || 852 (skb_network_header(skb_in) + sizeof(*iph)) > 853 skb_tail_pointer(skb_in)) 854 goto out; 855 856 /* 857 * No replies to physical multicast/broadcast 858 */ 859 if (skb_in->pkt_type != PACKET_HOST) 860 goto out; 861 862 /* 863 * Now check at the protocol level 864 */ 865 if (rt->rt_flags & (RTCF_BROADCAST | RTCF_MULTICAST)) 866 goto out; 867 868 /* 869 * Only reply to fragment 0. We byte re-order the constant 870 * mask for efficiency. 871 */ 872 if (iph->frag_off & htons(IP_OFFSET)) 873 goto out; 874 875 /* 876 * If we send an ICMP error to an ICMP error a mess would result.. 877 */ 878 if (icmp_pointers[type].error) { 879 /* 880 * We are an error, check if we are replying to an 881 * ICMP error 882 */ 883 if (iph->protocol == IPPROTO_ICMP) { 884 u8 _inner_type, *itp; 885 886 itp = skb_header_pointer(skb_in, 887 skb_network_header(skb_in) + 888 (iph->ihl << 2) + 889 offsetof(struct icmphdr, 890 type) - 891 skb_in->data, 892 sizeof(_inner_type), 893 &_inner_type); 894 if (!itp) 895 goto out; 896 897 /* 898 * Assume any unknown ICMP type is an error. This 899 * isn't specified by the RFC, but think about it.. 900 */ 901 if (*itp > NR_ICMP_TYPES || 902 icmp_pointers[*itp].error) 903 goto out; 904 } 905 } 906 907 /* Needed by both icmpv4_global_allow and icmp_xmit_lock */ 908 local_bh_disable(); 909 910 /* Check global sysctl_icmp_msgs_per_sec ratelimit, unless 911 * incoming dev is loopback. If outgoing dev change to not be 912 * loopback, then peer ratelimit still work (in icmpv4_xrlim_allow) 913 */ 914 if (!(skb_in->dev && (skb_in->dev->flags&IFF_LOOPBACK)) && 915 !icmpv4_global_allow(net, type, code, &apply_ratelimit)) 916 goto out_bh_enable; 917 918 sk = icmp_xmit_lock(net); 919 if (!sk) 920 goto out_bh_enable; 921 922 /* 923 * Construct source address and options. 924 */ 925 926 saddr = iph->daddr; 927 if (!(rt->rt_flags & RTCF_LOCAL)) { 928 struct net_device *dev = NULL; 929 930 rcu_read_lock(); 931 if (rt_is_input_route(rt) && 932 READ_ONCE(net->ipv4.sysctl_icmp_errors_use_inbound_ifaddr)) 933 dev = dev_get_by_index_rcu(net, parm->iif ? parm->iif : 934 inet_iif(skb_in)); 935 936 if (dev) 937 saddr = inet_select_addr(dev, iph->saddr, 938 RT_SCOPE_LINK); 939 else 940 saddr = 0; 941 rcu_read_unlock(); 942 } 943 944 tos = icmp_pointers[type].error ? (RT_TOS(iph->tos) | 945 IPTOS_PREC_INTERNETCONTROL) : 946 iph->tos; 947 mark = IP4_REPLY_MARK(net, skb_in->mark); 948 949 if (__ip_options_echo(net, &icmp_param->replyopts.opt, skb_in, 950 &parm->opt)) 951 goto out_unlock; 952 953 954 /* 955 * Prepare data for ICMP header. 956 */ 957 958 icmp_param->data.icmph.type = type; 959 icmp_param->data.icmph.code = code; 960 icmp_param->data.icmph.un.gateway = info; 961 icmp_param->data.icmph.checksum = 0; 962 icmp_param->skb = skb_in; 963 icmp_param->offset = skb_network_offset(skb_in); 964 ipcm_init(&ipc); 965 ipc.tos = tos; 966 ipc.addr = iph->saddr; 967 ipc.opt = &icmp_param->replyopts; 968 ipc.sockc.mark = mark; 969 970 rt = icmp_route_lookup(net, &fl4, skb_in, iph, saddr, 971 inet_dsfield_to_dscp(tos), mark, type, code, 972 icmp_param); 973 if (IS_ERR(rt)) 974 goto out_unlock; 975 976 if (rt->rt_flags & (RTCF_BROADCAST | RTCF_MULTICAST)) 977 goto ende; 978 979 /* peer icmp_ratelimit */ 980 if (!icmpv4_xrlim_allow(net, rt, &fl4, type, code, apply_ratelimit)) 981 goto ende; 982 983 /* RFC says return as much as we can without exceeding 576 bytes. */ 984 985 room = dst4_mtu(&rt->dst); 986 if (room > 576) 987 room = 576; 988 room -= sizeof(struct iphdr) + icmp_param->replyopts.opt.optlen; 989 room -= sizeof(struct icmphdr); 990 /* Guard against tiny mtu. We need to include at least one 991 * IP network header for this message to make any sense. 992 */ 993 if (room <= (int)sizeof(struct iphdr)) 994 goto ende; 995 996 ext_skb = icmp_ext_append(net, skb_in, &icmp_param->data.icmph, room, 997 parm->iif); 998 if (ext_skb) 999 icmp_param->skb = ext_skb; 1000 1001 icmp_param->data_len = icmp_param->skb->len - icmp_param->offset; 1002 if (icmp_param->data_len > room) 1003 icmp_param->data_len = room; 1004 icmp_param->head_len = sizeof(struct icmphdr); 1005 1006 /* if we don't have a source address at this point, fall back to the 1007 * dummy address instead of sending out a packet with a source address 1008 * of 0.0.0.0 1009 */ 1010 if (!fl4.saddr) 1011 fl4.saddr = htonl(INADDR_DUMMY); 1012 1013 trace_icmp_send(skb_in, type, code); 1014 1015 icmp_push_reply(sk, icmp_param, &fl4, &ipc, &rt); 1016 1017 if (ext_skb) 1018 consume_skb(ext_skb); 1019 ende: 1020 ip_rt_put(rt); 1021 out_unlock: 1022 icmp_xmit_unlock(sk); 1023 out_bh_enable: 1024 local_bh_enable(); 1025 out: 1026 rcu_read_unlock(); 1027 } 1028 EXPORT_SYMBOL(__icmp_send); 1029 1030 #if IS_ENABLED(CONFIG_NF_NAT) 1031 #include <net/netfilter/nf_conntrack.h> 1032 void icmp_ndo_send(struct sk_buff *skb_in, int type, int code, __be32 info) 1033 { 1034 struct sk_buff *cloned_skb = NULL; 1035 enum ip_conntrack_info ctinfo; 1036 enum ip_conntrack_dir dir; 1037 struct inet_skb_parm parm; 1038 struct nf_conn *ct; 1039 __be32 orig_ip; 1040 1041 memset(&parm, 0, sizeof(parm)); 1042 ct = nf_ct_get(skb_in, &ctinfo); 1043 if (!ct || !(READ_ONCE(ct->status) & IPS_NAT_MASK)) { 1044 __icmp_send(skb_in, type, code, info, &parm); 1045 return; 1046 } 1047 1048 if (skb_shared(skb_in)) 1049 skb_in = cloned_skb = skb_clone(skb_in, GFP_ATOMIC); 1050 1051 if (unlikely(!skb_in || skb_network_header(skb_in) < skb_in->head || 1052 (skb_network_header(skb_in) + sizeof(struct iphdr)) > 1053 skb_tail_pointer(skb_in) || skb_ensure_writable(skb_in, 1054 skb_network_offset(skb_in) + sizeof(struct iphdr)))) 1055 goto out; 1056 1057 orig_ip = ip_hdr(skb_in)->saddr; 1058 dir = CTINFO2DIR(ctinfo); 1059 ip_hdr(skb_in)->saddr = ct->tuplehash[dir].tuple.src.u3.ip; 1060 __icmp_send(skb_in, type, code, info, &parm); 1061 ip_hdr(skb_in)->saddr = orig_ip; 1062 out: 1063 consume_skb(cloned_skb); 1064 } 1065 EXPORT_SYMBOL(icmp_ndo_send); 1066 #endif 1067 1068 static void icmp_socket_deliver(struct sk_buff *skb, u32 info) 1069 { 1070 const struct iphdr *iph = (const struct iphdr *)skb->data; 1071 const struct net_protocol *ipprot; 1072 int protocol = iph->protocol; 1073 1074 /* Checkin full IP header plus 8 bytes of protocol to 1075 * avoid additional coding at protocol handlers. 1076 */ 1077 if (!pskb_may_pull(skb, iph->ihl * 4 + 8)) 1078 goto out; 1079 1080 /* IPPROTO_RAW sockets are not supposed to receive anything. */ 1081 if (protocol == IPPROTO_RAW) 1082 goto out; 1083 1084 raw_icmp_error(skb, protocol, info); 1085 1086 ipprot = rcu_dereference(inet_protos[protocol]); 1087 if (ipprot && ipprot->err_handler) 1088 ipprot->err_handler(skb, info); 1089 return; 1090 1091 out: 1092 __ICMP_INC_STATS(dev_net_rcu(skb->dev), ICMP_MIB_INERRORS); 1093 } 1094 1095 static bool icmp_tag_validation(int proto) 1096 { 1097 const struct net_protocol *ipprot; 1098 bool ok; 1099 1100 rcu_read_lock(); 1101 ipprot = rcu_dereference(inet_protos[proto]); 1102 ok = ipprot ? ipprot->icmp_strict_tag_validation : false; 1103 rcu_read_unlock(); 1104 return ok; 1105 } 1106 1107 /* 1108 * Handle ICMP_DEST_UNREACH, ICMP_TIME_EXCEEDED, ICMP_QUENCH, and 1109 * ICMP_PARAMETERPROB. 1110 */ 1111 1112 static enum skb_drop_reason icmp_unreach(struct sk_buff *skb) 1113 { 1114 enum skb_drop_reason reason = SKB_NOT_DROPPED_YET; 1115 const struct iphdr *iph; 1116 struct icmphdr *icmph; 1117 struct net *net; 1118 u32 info = 0; 1119 1120 net = skb_dst_dev_net_rcu(skb); 1121 1122 /* 1123 * Incomplete header ? 1124 * Only checks for the IP header, there should be an 1125 * additional check for longer headers in upper levels. 1126 */ 1127 1128 if (!pskb_may_pull(skb, sizeof(struct iphdr))) 1129 goto out_err; 1130 1131 icmph = icmp_hdr(skb); 1132 iph = (const struct iphdr *)skb->data; 1133 1134 if (iph->ihl < 5) { /* Mangled header, drop. */ 1135 reason = SKB_DROP_REASON_IP_INHDR; 1136 goto out_err; 1137 } 1138 1139 switch (icmph->type) { 1140 case ICMP_DEST_UNREACH: 1141 switch (icmph->code & 15) { 1142 case ICMP_NET_UNREACH: 1143 case ICMP_HOST_UNREACH: 1144 case ICMP_PROT_UNREACH: 1145 case ICMP_PORT_UNREACH: 1146 break; 1147 case ICMP_FRAG_NEEDED: 1148 /* for documentation of the ip_no_pmtu_disc 1149 * values please see 1150 * Documentation/networking/ip-sysctl.rst 1151 */ 1152 switch (READ_ONCE(net->ipv4.sysctl_ip_no_pmtu_disc)) { 1153 default: 1154 net_dbg_ratelimited("%pI4: fragmentation needed and DF set\n", 1155 &iph->daddr); 1156 break; 1157 case 2: 1158 goto out; 1159 case 3: 1160 if (!icmp_tag_validation(iph->protocol)) 1161 goto out; 1162 fallthrough; 1163 case 0: 1164 info = ntohs(icmph->un.frag.mtu); 1165 } 1166 break; 1167 case ICMP_SR_FAILED: 1168 net_dbg_ratelimited("%pI4: Source Route Failed\n", 1169 &iph->daddr); 1170 break; 1171 default: 1172 break; 1173 } 1174 if (icmph->code > NR_ICMP_UNREACH) 1175 goto out; 1176 break; 1177 case ICMP_PARAMETERPROB: 1178 info = ntohl(icmph->un.gateway) >> 24; 1179 break; 1180 case ICMP_TIME_EXCEEDED: 1181 __ICMP_INC_STATS(net, ICMP_MIB_INTIMEEXCDS); 1182 if (icmph->code == ICMP_EXC_FRAGTIME) 1183 goto out; 1184 break; 1185 } 1186 1187 /* 1188 * Throw it at our lower layers 1189 * 1190 * RFC 1122: 3.2.2 MUST extract the protocol ID from the passed 1191 * header. 1192 * RFC 1122: 3.2.2.1 MUST pass ICMP unreach messages to the 1193 * transport layer. 1194 * RFC 1122: 3.2.2.2 MUST pass ICMP time expired messages to 1195 * transport layer. 1196 */ 1197 1198 /* 1199 * Check the other end isn't violating RFC 1122. Some routers send 1200 * bogus responses to broadcast frames. If you see this message 1201 * first check your netmask matches at both ends, if it does then 1202 * get the other vendor to fix their kit. 1203 */ 1204 1205 if (!READ_ONCE(net->ipv4.sysctl_icmp_ignore_bogus_error_responses) && 1206 inet_addr_type_dev_table(net, skb->dev, iph->daddr) == RTN_BROADCAST) { 1207 net_warn_ratelimited("%pI4 sent an invalid ICMP type %u, code %u error to a broadcast: %pI4 on %s\n", 1208 &ip_hdr(skb)->saddr, 1209 icmph->type, icmph->code, 1210 &iph->daddr, skb->dev->name); 1211 goto out; 1212 } 1213 1214 icmp_socket_deliver(skb, info); 1215 1216 out: 1217 return reason; 1218 out_err: 1219 __ICMP_INC_STATS(net, ICMP_MIB_INERRORS); 1220 return reason ?: SKB_DROP_REASON_NOT_SPECIFIED; 1221 } 1222 1223 1224 /* 1225 * Handle ICMP_REDIRECT. 1226 */ 1227 1228 static enum skb_drop_reason icmp_redirect(struct sk_buff *skb) 1229 { 1230 if (skb->len < sizeof(struct iphdr)) { 1231 __ICMP_INC_STATS(dev_net_rcu(skb->dev), ICMP_MIB_INERRORS); 1232 return SKB_DROP_REASON_PKT_TOO_SMALL; 1233 } 1234 1235 if (!pskb_may_pull(skb, sizeof(struct iphdr))) { 1236 /* there aught to be a stat */ 1237 return SKB_DROP_REASON_NOMEM; 1238 } 1239 1240 icmp_socket_deliver(skb, ntohl(icmp_hdr(skb)->un.gateway)); 1241 return SKB_NOT_DROPPED_YET; 1242 } 1243 1244 /* 1245 * Handle ICMP_ECHO ("ping") and ICMP_EXT_ECHO ("PROBE") requests. 1246 * 1247 * RFC 1122: 3.2.2.6 MUST have an echo server that answers ICMP echo 1248 * requests. 1249 * RFC 1122: 3.2.2.6 Data received in the ICMP_ECHO request MUST be 1250 * included in the reply. 1251 * RFC 1812: 4.3.3.6 SHOULD have a config option for silently ignoring 1252 * echo requests, MUST have default=NOT. 1253 * RFC 8335: 8 MUST have a config option to enable/disable ICMP 1254 * Extended Echo Functionality, MUST be disabled by default 1255 * See also WRT handling of options once they are done and working. 1256 */ 1257 1258 static enum skb_drop_reason icmp_echo(struct sk_buff *skb) 1259 { 1260 DEFINE_RAW_FLEX(struct icmp_bxm, icmp_param, replyopts.opt.__data, 1261 IP_OPTIONS_DATA_FIXED_SIZE); 1262 struct net *net; 1263 1264 net = skb_dst_dev_net_rcu(skb); 1265 /* should there be an ICMP stat for ignored echos? */ 1266 if (READ_ONCE(net->ipv4.sysctl_icmp_echo_ignore_all)) 1267 return SKB_NOT_DROPPED_YET; 1268 1269 icmp_param->data.icmph = *icmp_hdr(skb); 1270 icmp_param->skb = skb; 1271 icmp_param->offset = 0; 1272 icmp_param->data_len = skb->len; 1273 icmp_param->head_len = sizeof(struct icmphdr); 1274 1275 if (icmp_param->data.icmph.type == ICMP_ECHO) 1276 icmp_param->data.icmph.type = ICMP_ECHOREPLY; 1277 else if (!icmp_build_probe(skb, &icmp_param->data.icmph)) 1278 return SKB_NOT_DROPPED_YET; 1279 1280 icmp_reply(icmp_param, skb); 1281 return SKB_NOT_DROPPED_YET; 1282 } 1283 1284 /* Helper for icmp_echo and icmpv6_echo_reply. 1285 * Searches for net_device that matches PROBE interface identifier 1286 * and builds PROBE reply message in icmphdr. 1287 * 1288 * Returns false if PROBE responses are disabled via sysctl 1289 */ 1290 1291 bool icmp_build_probe(struct sk_buff *skb, struct icmphdr *icmphdr) 1292 { 1293 struct net *net = dev_net_rcu(skb->dev); 1294 struct icmp_ext_hdr *ext_hdr, _ext_hdr; 1295 struct icmp_ext_echo_iio *iio, _iio; 1296 struct inet6_dev *in6_dev; 1297 struct in_device *in_dev; 1298 struct net_device *dev; 1299 char buff[IFNAMSIZ]; 1300 u16 ident_len; 1301 u8 status; 1302 1303 if (!READ_ONCE(net->ipv4.sysctl_icmp_echo_enable_probe)) 1304 return false; 1305 1306 /* We currently only support probing interfaces on the proxy node 1307 * Check to ensure L-bit is set 1308 */ 1309 if (!(ntohs(icmphdr->un.echo.sequence) & 1)) 1310 return false; 1311 /* Clear status bits in reply message */ 1312 icmphdr->un.echo.sequence &= htons(0xFF00); 1313 if (icmphdr->type == ICMP_EXT_ECHO) 1314 icmphdr->type = ICMP_EXT_ECHOREPLY; 1315 else 1316 icmphdr->type = ICMPV6_EXT_ECHO_REPLY; 1317 ext_hdr = skb_header_pointer(skb, 0, sizeof(_ext_hdr), &_ext_hdr); 1318 /* Size of iio is class_type dependent. 1319 * Only check header here and assign length based on ctype in the switch statement 1320 */ 1321 iio = skb_header_pointer(skb, sizeof(_ext_hdr), sizeof(iio->extobj_hdr), &_iio); 1322 if (!ext_hdr || !iio) 1323 goto send_mal_query; 1324 if (ntohs(iio->extobj_hdr.length) <= sizeof(iio->extobj_hdr) || 1325 ntohs(iio->extobj_hdr.length) > sizeof(_iio)) 1326 goto send_mal_query; 1327 ident_len = ntohs(iio->extobj_hdr.length) - sizeof(iio->extobj_hdr); 1328 iio = skb_header_pointer(skb, sizeof(_ext_hdr), 1329 sizeof(iio->extobj_hdr) + ident_len, &_iio); 1330 if (!iio) 1331 goto send_mal_query; 1332 1333 status = 0; 1334 dev = NULL; 1335 switch (iio->extobj_hdr.class_type) { 1336 case ICMP_EXT_ECHO_CTYPE_NAME: 1337 if (ident_len >= IFNAMSIZ) 1338 goto send_mal_query; 1339 memset(buff, 0, sizeof(buff)); 1340 memcpy(buff, &iio->ident.name, ident_len); 1341 dev = dev_get_by_name(net, buff); 1342 break; 1343 case ICMP_EXT_ECHO_CTYPE_INDEX: 1344 if (ident_len != sizeof(iio->ident.ifindex)) 1345 goto send_mal_query; 1346 dev = dev_get_by_index(net, ntohl(iio->ident.ifindex)); 1347 break; 1348 case ICMP_EXT_ECHO_CTYPE_ADDR: 1349 if (ident_len < sizeof(iio->ident.addr.ctype3_hdr) || 1350 ident_len != sizeof(iio->ident.addr.ctype3_hdr) + 1351 iio->ident.addr.ctype3_hdr.addrlen) 1352 goto send_mal_query; 1353 switch (ntohs(iio->ident.addr.ctype3_hdr.afi)) { 1354 case ICMP_AFI_IP: 1355 if (iio->ident.addr.ctype3_hdr.addrlen != sizeof(struct in_addr)) 1356 goto send_mal_query; 1357 dev = ip_dev_find(net, iio->ident.addr.ip_addr.ipv4_addr); 1358 break; 1359 #if IS_ENABLED(CONFIG_IPV6) 1360 case ICMP_AFI_IP6: 1361 if (iio->ident.addr.ctype3_hdr.addrlen != sizeof(struct in6_addr)) 1362 goto send_mal_query; 1363 dev = ipv6_dev_find(net, &iio->ident.addr.ip_addr.ipv6_addr, dev); 1364 dev_hold(dev); 1365 break; 1366 #endif 1367 default: 1368 goto send_mal_query; 1369 } 1370 break; 1371 default: 1372 goto send_mal_query; 1373 } 1374 if (!dev) { 1375 icmphdr->code = ICMP_EXT_CODE_NO_IF; 1376 return true; 1377 } 1378 /* Fill bits in reply message */ 1379 if (dev->flags & IFF_UP) 1380 status |= ICMP_EXT_ECHOREPLY_ACTIVE; 1381 1382 in_dev = __in_dev_get_rcu(dev); 1383 if (in_dev && rcu_access_pointer(in_dev->ifa_list)) 1384 status |= ICMP_EXT_ECHOREPLY_IPV4; 1385 1386 in6_dev = __in6_dev_get(dev); 1387 if (in6_dev && !list_empty(&in6_dev->addr_list)) 1388 status |= ICMP_EXT_ECHOREPLY_IPV6; 1389 1390 dev_put(dev); 1391 icmphdr->un.echo.sequence |= htons(status); 1392 return true; 1393 send_mal_query: 1394 icmphdr->code = ICMP_EXT_CODE_MAL_QUERY; 1395 return true; 1396 } 1397 1398 /* 1399 * Handle ICMP Timestamp requests. 1400 * RFC 1122: 3.2.2.8 MAY implement ICMP timestamp requests. 1401 * SHOULD be in the kernel for minimum random latency. 1402 * MUST be accurate to a few minutes. 1403 * MUST be updated at least at 15Hz. 1404 */ 1405 static enum skb_drop_reason icmp_timestamp(struct sk_buff *skb) 1406 { 1407 DEFINE_RAW_FLEX(struct icmp_bxm, icmp_param, replyopts.opt.__data, 1408 IP_OPTIONS_DATA_FIXED_SIZE); 1409 /* 1410 * Too short. 1411 */ 1412 if (skb->len < 4) 1413 goto out_err; 1414 1415 /* 1416 * Fill in the current time as ms since midnight UT: 1417 */ 1418 icmp_param->data.times[1] = inet_current_timestamp(); 1419 icmp_param->data.times[2] = icmp_param->data.times[1]; 1420 1421 BUG_ON(skb_copy_bits(skb, 0, &icmp_param->data.times[0], 4)); 1422 1423 icmp_param->data.icmph = *icmp_hdr(skb); 1424 icmp_param->data.icmph.type = ICMP_TIMESTAMPREPLY; 1425 icmp_param->data.icmph.code = 0; 1426 icmp_param->skb = skb; 1427 icmp_param->offset = 0; 1428 icmp_param->data_len = 0; 1429 icmp_param->head_len = sizeof(struct icmphdr) + 12; 1430 icmp_reply(icmp_param, skb); 1431 return SKB_NOT_DROPPED_YET; 1432 1433 out_err: 1434 __ICMP_INC_STATS(skb_dst_dev_net_rcu(skb), ICMP_MIB_INERRORS); 1435 return SKB_DROP_REASON_PKT_TOO_SMALL; 1436 } 1437 1438 static enum skb_drop_reason icmp_discard(struct sk_buff *skb) 1439 { 1440 /* pretend it was a success */ 1441 return SKB_NOT_DROPPED_YET; 1442 } 1443 1444 /* 1445 * Deal with incoming ICMP packets. 1446 */ 1447 int icmp_rcv(struct sk_buff *skb) 1448 { 1449 enum skb_drop_reason reason = SKB_DROP_REASON_NOT_SPECIFIED; 1450 struct rtable *rt = skb_rtable(skb); 1451 struct net *net = dev_net_rcu(rt->dst.dev); 1452 struct icmphdr *icmph; 1453 1454 if (!xfrm4_policy_check(NULL, XFRM_POLICY_IN, skb)) { 1455 struct sec_path *sp = skb_sec_path(skb); 1456 int nh; 1457 1458 if (!(sp && sp->xvec[sp->len - 1]->props.flags & 1459 XFRM_STATE_ICMP)) { 1460 reason = SKB_DROP_REASON_XFRM_POLICY; 1461 goto drop; 1462 } 1463 1464 if (!pskb_may_pull(skb, sizeof(*icmph) + sizeof(struct iphdr))) 1465 goto drop; 1466 1467 nh = skb_network_offset(skb); 1468 skb_set_network_header(skb, sizeof(*icmph)); 1469 1470 if (!xfrm4_policy_check_reverse(NULL, XFRM_POLICY_IN, 1471 skb)) { 1472 reason = SKB_DROP_REASON_XFRM_POLICY; 1473 goto drop; 1474 } 1475 1476 skb_set_network_header(skb, nh); 1477 } 1478 1479 __ICMP_INC_STATS(net, ICMP_MIB_INMSGS); 1480 1481 if (skb_checksum_simple_validate(skb)) 1482 goto csum_error; 1483 1484 if (!pskb_pull(skb, sizeof(*icmph))) 1485 goto error; 1486 1487 icmph = icmp_hdr(skb); 1488 1489 ICMPMSGIN_INC_STATS(net, icmph->type); 1490 1491 /* Check for ICMP Extended Echo (PROBE) messages */ 1492 if (icmph->type == ICMP_EXT_ECHO) { 1493 /* We can't use icmp_pointers[].handler() because it is an array of 1494 * size NR_ICMP_TYPES + 1 (19 elements) and PROBE has code 42. 1495 */ 1496 reason = icmp_echo(skb); 1497 goto reason_check; 1498 } 1499 1500 /* 1501 * Parse the ICMP message 1502 */ 1503 1504 if (rt->rt_flags & (RTCF_BROADCAST | RTCF_MULTICAST)) { 1505 /* 1506 * RFC 1122: 3.2.2.6 An ICMP_ECHO to broadcast MAY be 1507 * silently ignored (we let user decide with a sysctl). 1508 * RFC 1122: 3.2.2.8 An ICMP_TIMESTAMP MAY be silently 1509 * discarded if to broadcast/multicast. 1510 */ 1511 if ((icmph->type == ICMP_ECHO || 1512 icmph->type == ICMP_TIMESTAMP) && 1513 READ_ONCE(net->ipv4.sysctl_icmp_echo_ignore_broadcasts)) { 1514 reason = SKB_DROP_REASON_INVALID_PROTO; 1515 goto error; 1516 } 1517 if (icmph->type != ICMP_ECHO && 1518 icmph->type != ICMP_TIMESTAMP && 1519 icmph->type != ICMP_ADDRESS && 1520 icmph->type != ICMP_ADDRESSREPLY) { 1521 reason = SKB_DROP_REASON_INVALID_PROTO; 1522 goto error; 1523 } 1524 } 1525 1526 if (icmph->type == ICMP_EXT_ECHOREPLY || 1527 icmph->type == ICMP_ECHOREPLY) { 1528 reason = ping_rcv(skb); 1529 return reason ? NET_RX_DROP : NET_RX_SUCCESS; 1530 } 1531 1532 /* 1533 * 18 is the highest 'known' ICMP type. Anything else is a mystery 1534 * 1535 * RFC 1122: 3.2.2 Unknown ICMP messages types MUST be silently 1536 * discarded. 1537 */ 1538 if (icmph->type > NR_ICMP_TYPES) { 1539 reason = SKB_DROP_REASON_UNHANDLED_PROTO; 1540 goto error; 1541 } 1542 1543 reason = icmp_pointers[icmph->type].handler(skb); 1544 reason_check: 1545 if (!reason) { 1546 consume_skb(skb); 1547 return NET_RX_SUCCESS; 1548 } 1549 1550 drop: 1551 kfree_skb_reason(skb, reason); 1552 return NET_RX_DROP; 1553 csum_error: 1554 reason = SKB_DROP_REASON_ICMP_CSUM; 1555 __ICMP_INC_STATS(net, ICMP_MIB_CSUMERRORS); 1556 error: 1557 __ICMP_INC_STATS(net, ICMP_MIB_INERRORS); 1558 goto drop; 1559 } 1560 1561 static bool ip_icmp_error_rfc4884_validate(const struct sk_buff *skb, int off) 1562 { 1563 struct icmp_extobj_hdr *objh, _objh; 1564 struct icmp_ext_hdr *exth, _exth; 1565 u16 olen; 1566 1567 exth = skb_header_pointer(skb, off, sizeof(_exth), &_exth); 1568 if (!exth) 1569 return false; 1570 if (exth->version != 2) 1571 return true; 1572 1573 if (exth->checksum && 1574 csum_fold(skb_checksum(skb, off, skb->len - off, 0))) 1575 return false; 1576 1577 off += sizeof(_exth); 1578 while (off < skb->len) { 1579 objh = skb_header_pointer(skb, off, sizeof(_objh), &_objh); 1580 if (!objh) 1581 return false; 1582 1583 olen = ntohs(objh->length); 1584 if (olen < sizeof(_objh)) 1585 return false; 1586 1587 off += olen; 1588 if (off > skb->len) 1589 return false; 1590 } 1591 1592 return true; 1593 } 1594 1595 void ip_icmp_error_rfc4884(const struct sk_buff *skb, 1596 struct sock_ee_data_rfc4884 *out, 1597 int thlen, int off) 1598 { 1599 int hlen; 1600 1601 /* original datagram headers: end of icmph to payload (skb->data) */ 1602 hlen = -skb_transport_offset(skb) - thlen; 1603 1604 /* per rfc 4884: minimal datagram length of 128 bytes */ 1605 if (off < 128 || off < hlen) 1606 return; 1607 1608 /* kernel has stripped headers: return payload offset in bytes */ 1609 off -= hlen; 1610 if (off + sizeof(struct icmp_ext_hdr) > skb->len) 1611 return; 1612 1613 out->len = off; 1614 1615 if (!ip_icmp_error_rfc4884_validate(skb, off)) 1616 out->flags |= SO_EE_RFC4884_FLAG_INVALID; 1617 } 1618 1619 int icmp_err(struct sk_buff *skb, u32 info) 1620 { 1621 struct iphdr *iph = (struct iphdr *)skb->data; 1622 int offset = iph->ihl<<2; 1623 struct icmphdr *icmph = (struct icmphdr *)(skb->data + offset); 1624 struct net *net = dev_net_rcu(skb->dev); 1625 int type = icmp_hdr(skb)->type; 1626 int code = icmp_hdr(skb)->code; 1627 1628 /* 1629 * Use ping_err to handle all icmp errors except those 1630 * triggered by ICMP_ECHOREPLY which sent from kernel. 1631 */ 1632 if (icmph->type != ICMP_ECHOREPLY) { 1633 ping_err(skb, offset, info); 1634 return 0; 1635 } 1636 1637 if (type == ICMP_DEST_UNREACH && code == ICMP_FRAG_NEEDED) 1638 ipv4_update_pmtu(skb, net, info, 0, IPPROTO_ICMP); 1639 else if (type == ICMP_REDIRECT) 1640 ipv4_redirect(skb, net, 0, IPPROTO_ICMP); 1641 1642 return 0; 1643 } 1644 1645 /* 1646 * This table is the definition of how we handle ICMP. 1647 */ 1648 static const struct icmp_control icmp_pointers[NR_ICMP_TYPES + 1] = { 1649 [ICMP_ECHOREPLY] = { 1650 .handler = ping_rcv, 1651 }, 1652 [1] = { 1653 .handler = icmp_discard, 1654 .error = 1, 1655 }, 1656 [2] = { 1657 .handler = icmp_discard, 1658 .error = 1, 1659 }, 1660 [ICMP_DEST_UNREACH] = { 1661 .handler = icmp_unreach, 1662 .error = 1, 1663 }, 1664 [ICMP_SOURCE_QUENCH] = { 1665 .handler = icmp_unreach, 1666 .error = 1, 1667 }, 1668 [ICMP_REDIRECT] = { 1669 .handler = icmp_redirect, 1670 .error = 1, 1671 }, 1672 [6] = { 1673 .handler = icmp_discard, 1674 .error = 1, 1675 }, 1676 [7] = { 1677 .handler = icmp_discard, 1678 .error = 1, 1679 }, 1680 [ICMP_ECHO] = { 1681 .handler = icmp_echo, 1682 }, 1683 [9] = { 1684 .handler = icmp_discard, 1685 .error = 1, 1686 }, 1687 [10] = { 1688 .handler = icmp_discard, 1689 .error = 1, 1690 }, 1691 [ICMP_TIME_EXCEEDED] = { 1692 .handler = icmp_unreach, 1693 .error = 1, 1694 }, 1695 [ICMP_PARAMETERPROB] = { 1696 .handler = icmp_unreach, 1697 .error = 1, 1698 }, 1699 [ICMP_TIMESTAMP] = { 1700 .handler = icmp_timestamp, 1701 }, 1702 [ICMP_TIMESTAMPREPLY] = { 1703 .handler = icmp_discard, 1704 }, 1705 [ICMP_INFO_REQUEST] = { 1706 .handler = icmp_discard, 1707 }, 1708 [ICMP_INFO_REPLY] = { 1709 .handler = icmp_discard, 1710 }, 1711 [ICMP_ADDRESS] = { 1712 .handler = icmp_discard, 1713 }, 1714 [ICMP_ADDRESSREPLY] = { 1715 .handler = icmp_discard, 1716 }, 1717 }; 1718 1719 static int __net_init icmp_sk_init(struct net *net) 1720 { 1721 /* Control parameters for ECHO replies. */ 1722 net->ipv4.sysctl_icmp_echo_ignore_all = 0; 1723 net->ipv4.sysctl_icmp_echo_enable_probe = 0; 1724 net->ipv4.sysctl_icmp_echo_ignore_broadcasts = 1; 1725 1726 /* Control parameter - ignore bogus broadcast responses? */ 1727 net->ipv4.sysctl_icmp_ignore_bogus_error_responses = 1; 1728 1729 /* 1730 * Configurable global rate limit. 1731 * 1732 * ratelimit defines tokens/packet consumed for dst->rate_token 1733 * bucket ratemask defines which icmp types are ratelimited by 1734 * setting it's bit position. 1735 * 1736 * default: 1737 * dest unreachable (3), source quench (4), 1738 * time exceeded (11), parameter problem (12) 1739 */ 1740 1741 net->ipv4.sysctl_icmp_ratelimit = 1 * HZ; 1742 net->ipv4.sysctl_icmp_ratemask = 0x1818; 1743 net->ipv4.sysctl_icmp_errors_use_inbound_ifaddr = 0; 1744 net->ipv4.sysctl_icmp_errors_extension_mask = 0; 1745 net->ipv4.sysctl_icmp_msgs_per_sec = 10000; 1746 net->ipv4.sysctl_icmp_msgs_burst = 10000; 1747 1748 return 0; 1749 } 1750 1751 static struct pernet_operations __net_initdata icmp_sk_ops = { 1752 .init = icmp_sk_init, 1753 }; 1754 1755 int __init icmp_init(void) 1756 { 1757 int err, i; 1758 1759 for_each_possible_cpu(i) { 1760 struct sock *sk; 1761 1762 err = inet_ctl_sock_create(&sk, PF_INET, 1763 SOCK_RAW, IPPROTO_ICMP, &init_net); 1764 if (err < 0) 1765 return err; 1766 1767 per_cpu(ipv4_icmp_sk, i) = sk; 1768 1769 /* Enough space for 2 64K ICMP packets, including 1770 * sk_buff/skb_shared_info struct overhead. 1771 */ 1772 sk->sk_sndbuf = 2 * SKB_TRUESIZE(64 * 1024); 1773 1774 /* 1775 * Speedup sock_wfree() 1776 */ 1777 sock_set_flag(sk, SOCK_USE_WRITE_QUEUE); 1778 inet_sk(sk)->pmtudisc = IP_PMTUDISC_DONT; 1779 } 1780 return register_pernet_subsys(&icmp_sk_ops); 1781 } 1782