1 /* 2 * Copyright (c) 2016, Mellanox Technologies. All rights reserved. 3 * 4 * This software is available to you under a choice of one of two 5 * licenses. You may choose to be licensed under the terms of the GNU 6 * General Public License (GPL) Version 2, available from the file 7 * COPYING in the main directory of this source tree, or the 8 * OpenIB.org BSD license below: 9 * 10 * Redistribution and use in source and binary forms, with or 11 * without modification, are permitted provided that the following 12 * conditions are met: 13 * 14 * - Redistributions of source code must retain the above 15 * copyright notice, this list of conditions and the following 16 * disclaimer. 17 * 18 * - Redistributions in binary form must reproduce the above 19 * copyright notice, this list of conditions and the following 20 * disclaimer in the documentation and/or other materials 21 * provided with the distribution. 22 * 23 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 24 * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF 25 * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 26 * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS 27 * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN 28 * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN 29 * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE 30 * SOFTWARE. 31 */ 32 33 #include <linux/netdevice.h> 34 #include <net/bonding.h> 35 #include <linux/mlx5/driver.h> 36 #include <linux/mlx5/eswitch.h> 37 #include <linux/mlx5/vport.h> 38 #include <linux/mlx5/lag.h> 39 #include "lib/mlx5.h" 40 #include "lib/devcom.h" 41 #include "mlx5_core.h" 42 #include "eswitch.h" 43 #include "esw/acl/ofld.h" 44 #include "lag.h" 45 #include "mp.h" 46 #include "mpesw.h" 47 48 49 /* General purpose, use for short periods of time. 50 * Beware of lock dependencies (preferably, no locks should be acquired 51 * under it). 52 */ 53 static DEFINE_SPINLOCK(lag_lock); 54 55 static int get_port_sel_mode(enum mlx5_lag_mode mode, unsigned long flags) 56 { 57 if (test_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, &flags)) 58 return MLX5_LAG_PORT_SELECT_MODE_PORT_SELECT_FT; 59 60 if (mode == MLX5_LAG_MODE_MPESW) 61 return MLX5_LAG_PORT_SELECT_MODE_PORT_SELECT_MPESW; 62 63 return MLX5_LAG_PORT_SELECT_MODE_QUEUE_AFFINITY; 64 } 65 66 static u8 lag_active_port_bits(struct mlx5_lag *ldev, 67 struct lag_tracker *tracker) 68 { 69 u8 enabled_ports[MLX5_MAX_PORTS] = {}; 70 u8 active_port = 0; 71 int num_enabled; 72 int idx; 73 74 mlx5_infer_tx_enabled(tracker, ldev, enabled_ports, 75 &num_enabled); 76 for (idx = 0; idx < num_enabled; idx++) 77 active_port |= BIT_MASK(enabled_ports[idx]); 78 79 return active_port; 80 } 81 82 static int mlx5_cmd_create_lag(struct mlx5_core_dev *dev, struct mlx5_lag *ldev, 83 struct lag_tracker *tracker, int mode, 84 unsigned long flags) 85 { 86 bool fdb_sel_mode = test_bit(MLX5_LAG_MODE_FLAG_FDB_SEL_MODE_NATIVE, 87 &flags); 88 int port_sel_mode = get_port_sel_mode(mode, flags); 89 u32 in[MLX5_ST_SZ_DW(create_lag_in)] = {}; 90 u8 *ports = ldev->v2p_map; 91 int idx0, idx1; 92 void *lag_ctx; 93 94 lag_ctx = MLX5_ADDR_OF(create_lag_in, in, ctx); 95 MLX5_SET(create_lag_in, in, opcode, MLX5_CMD_OP_CREATE_LAG); 96 MLX5_SET(lagc, lag_ctx, fdb_selection_mode, fdb_sel_mode); 97 idx0 = mlx5_lag_get_dev_index_by_seq(ldev, 0); 98 idx1 = mlx5_lag_get_dev_index_by_seq(ldev, 1); 99 100 if (idx0 < 0 || idx1 < 0) 101 return -EINVAL; 102 103 switch (port_sel_mode) { 104 case MLX5_LAG_PORT_SELECT_MODE_QUEUE_AFFINITY: 105 MLX5_SET(lagc, lag_ctx, tx_remap_affinity_1, ports[idx0]); 106 MLX5_SET(lagc, lag_ctx, tx_remap_affinity_2, ports[idx1]); 107 break; 108 case MLX5_LAG_PORT_SELECT_MODE_PORT_SELECT_FT: 109 if (!MLX5_CAP_PORT_SELECTION(dev, port_select_flow_table_bypass)) 110 break; 111 112 MLX5_SET(lagc, lag_ctx, active_port, 113 lag_active_port_bits(ldev, tracker)); 114 break; 115 default: 116 break; 117 } 118 MLX5_SET(lagc, lag_ctx, port_select_mode, port_sel_mode); 119 120 return mlx5_cmd_exec_in(dev, create_lag, in); 121 } 122 123 static int mlx5_cmd_modify_lag(struct mlx5_core_dev *dev, struct mlx5_lag *ldev, 124 u8 *ports) 125 { 126 u32 in[MLX5_ST_SZ_DW(modify_lag_in)] = {}; 127 void *lag_ctx = MLX5_ADDR_OF(modify_lag_in, in, ctx); 128 int idx0, idx1; 129 130 idx0 = mlx5_lag_get_dev_index_by_seq(ldev, 0); 131 idx1 = mlx5_lag_get_dev_index_by_seq(ldev, 1); 132 if (idx0 < 0 || idx1 < 0) 133 return -EINVAL; 134 135 MLX5_SET(modify_lag_in, in, opcode, MLX5_CMD_OP_MODIFY_LAG); 136 MLX5_SET(modify_lag_in, in, field_select, 0x1); 137 138 MLX5_SET(lagc, lag_ctx, tx_remap_affinity_1, ports[idx0]); 139 MLX5_SET(lagc, lag_ctx, tx_remap_affinity_2, ports[idx1]); 140 141 return mlx5_cmd_exec_in(dev, modify_lag, in); 142 } 143 144 static u32 mlx5_lag_dev_group_id(struct mlx5_core_dev *dev) 145 { 146 struct mlx5_lag *ldev = mlx5_lag_dev(dev); 147 struct lag_func *pf; 148 int i; 149 150 if (!ldev) 151 return 0; 152 153 mlx5_lag_for_each(i, 0, ldev, MLX5_LAG_FILTER_ALL) { 154 pf = mlx5_lag_pf(ldev, i); 155 if (pf->dev == dev) 156 return pf->sd_fdb_active ? pf->group_id : 0; 157 } 158 return 0; 159 } 160 161 static int mlx5_lag_is_sw_lag(struct mlx5_core_dev *dev) 162 { 163 return mlx5_lag_is_sd(dev); 164 } 165 166 int mlx5_cmd_create_vport_lag(struct mlx5_core_dev *dev) 167 { 168 u32 in[MLX5_ST_SZ_DW(create_vport_lag_in)] = {}; 169 struct mlx5_lag *ldev = mlx5_lag_dev(dev); 170 int ret; 171 172 if (mlx5_lag_is_sw_lag(dev)) { 173 if (!ldev) 174 return -ENODEV; 175 176 mutex_lock(&ldev->lock); 177 ret = mlx5_lag_create_vport_lag(mlx5_lag_dev(dev), 178 mlx5_lag_dev_group_id(dev)); 179 mutex_unlock(&ldev->lock); 180 return ret; 181 } 182 183 MLX5_SET(create_vport_lag_in, in, opcode, MLX5_CMD_OP_CREATE_VPORT_LAG); 184 185 return mlx5_cmd_exec_in(dev, create_vport_lag, in); 186 } 187 EXPORT_SYMBOL(mlx5_cmd_create_vport_lag); 188 189 int mlx5_cmd_destroy_vport_lag(struct mlx5_core_dev *dev) 190 { 191 u32 in[MLX5_ST_SZ_DW(destroy_vport_lag_in)] = {}; 192 struct mlx5_lag *ldev = mlx5_lag_dev(dev); 193 194 if (mlx5_lag_is_sw_lag(dev)) { 195 if (!ldev) 196 return 0; 197 198 mutex_lock(&ldev->lock); 199 mlx5_lag_destroy_vport_lag(mlx5_lag_dev(dev), 200 mlx5_lag_dev_group_id(dev)); 201 mutex_unlock(&ldev->lock); 202 return 0; 203 } 204 205 MLX5_SET(destroy_vport_lag_in, in, opcode, MLX5_CMD_OP_DESTROY_VPORT_LAG); 206 207 return mlx5_cmd_exec_in(dev, destroy_vport_lag, in); 208 } 209 EXPORT_SYMBOL(mlx5_cmd_destroy_vport_lag); 210 211 static void mlx5_infer_tx_disabled(struct lag_tracker *tracker, struct mlx5_lag *ldev, 212 u8 *ports, int *num_disabled) 213 { 214 int i; 215 216 *num_disabled = 0; 217 mlx5_ldev_for_each(i, 0, ldev) 218 if (!tracker->netdev_state[i].tx_enabled || 219 !tracker->netdev_state[i].link_up) 220 ports[(*num_disabled)++] = i; 221 } 222 223 void mlx5_infer_tx_enabled(struct lag_tracker *tracker, struct mlx5_lag *ldev, 224 u8 *ports, int *num_enabled) 225 { 226 int i; 227 228 *num_enabled = 0; 229 mlx5_ldev_for_each(i, 0, ldev) 230 if (tracker->netdev_state[i].tx_enabled && 231 tracker->netdev_state[i].link_up) 232 ports[(*num_enabled)++] = i; 233 234 if (*num_enabled == 0) 235 mlx5_infer_tx_disabled(tracker, ldev, ports, num_enabled); 236 } 237 238 static void mlx5_lag_print_mapping(struct mlx5_core_dev *dev, 239 struct mlx5_lag *ldev, 240 struct lag_tracker *tracker, 241 unsigned long flags) 242 { 243 char buf[MLX5_MAX_PORTS * 10 + 1] = {}; 244 u8 enabled_ports[MLX5_MAX_PORTS] = {}; 245 int written = 0; 246 int num_enabled; 247 int idx; 248 int err; 249 int i; 250 int j; 251 252 if (test_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, &flags)) { 253 mlx5_infer_tx_enabled(tracker, ldev, enabled_ports, 254 &num_enabled); 255 for (i = 0; i < num_enabled; i++) { 256 err = scnprintf(buf + written, 4, "%d, ", enabled_ports[i] + 1); 257 if (err != 3) 258 return; 259 written += err; 260 } 261 buf[written - 2] = 0; 262 mlx5_core_info(dev, "lag map active ports: %s\n", buf); 263 } else { 264 mlx5_ldev_for_each(i, 0, ldev) { 265 for (j = 0; j < ldev->buckets; j++) { 266 idx = i * ldev->buckets + j; 267 err = scnprintf(buf + written, 10, 268 " port %d:%d", i + 1, ldev->v2p_map[idx]); 269 if (err != 9) 270 return; 271 written += err; 272 } 273 } 274 mlx5_core_info(dev, "lag map:%s\n", buf); 275 } 276 } 277 278 static int mlx5_lag_netdev_event(struct notifier_block *this, 279 unsigned long event, void *ptr); 280 static void mlx5_do_bond_work(struct work_struct *work); 281 282 static void mlx5_ldev_free(struct kref *ref) 283 { 284 struct mlx5_lag *ldev = container_of(ref, struct mlx5_lag, ref); 285 struct lag_func *pf; 286 struct net *net; 287 int i; 288 289 if (ldev->nb.notifier_call) { 290 net = read_pnet(&ldev->net); 291 unregister_netdevice_notifier_net(net, &ldev->nb); 292 } 293 294 mlx5_lag_for_each(i, 0, ldev, MLX5_LAG_FILTER_ALL) { 295 pf = mlx5_lag_pf(ldev, i); 296 if (pf->port_change_nb.nb.notifier_call) { 297 struct mlx5_nb *nb = &pf->port_change_nb; 298 299 mlx5_eq_notifier_unregister(pf->dev, nb); 300 } 301 xa_erase(&ldev->pfs, i); 302 kfree(pf); 303 } 304 xa_destroy(&ldev->pfs); 305 306 mlx5_lag_mp_cleanup(ldev); 307 cancel_delayed_work_sync(&ldev->bond_work); 308 cancel_work_sync(&ldev->speed_update_work); 309 destroy_workqueue(ldev->wq); 310 mutex_destroy(&ldev->lock); 311 kfree(ldev); 312 } 313 314 static void mlx5_ldev_put(struct mlx5_lag *ldev) 315 { 316 kref_put(&ldev->ref, mlx5_ldev_free); 317 } 318 319 static void mlx5_ldev_get(struct mlx5_lag *ldev) 320 { 321 kref_get(&ldev->ref); 322 } 323 324 static struct mlx5_lag *mlx5_lag_dev_alloc(struct mlx5_core_dev *dev) 325 { 326 struct mlx5_lag *ldev; 327 int err; 328 329 ldev = kzalloc_obj(*ldev); 330 if (!ldev) 331 return NULL; 332 333 ldev->wq = create_singlethread_workqueue("mlx5_lag"); 334 if (!ldev->wq) { 335 kfree(ldev); 336 return NULL; 337 } 338 339 kref_init(&ldev->ref); 340 mutex_init(&ldev->lock); 341 xa_init_flags(&ldev->pfs, XA_FLAGS_ALLOC); 342 INIT_DELAYED_WORK(&ldev->bond_work, mlx5_do_bond_work); 343 INIT_WORK(&ldev->speed_update_work, mlx5_mpesw_speed_update_work); 344 345 if (!mlx5_sd_is_supported(dev)) { 346 ldev->nb.notifier_call = mlx5_lag_netdev_event; 347 write_pnet(&ldev->net, mlx5_core_net(dev)); 348 if (register_netdevice_notifier_net(read_pnet(&ldev->net), 349 &ldev->nb)) { 350 ldev->nb.notifier_call = NULL; 351 mlx5_core_err(dev, "Failed to register LAG netdev notifier\n"); 352 } 353 } 354 ldev->mode = MLX5_LAG_MODE_NONE; 355 356 err = mlx5_lag_mp_init(ldev); 357 if (err) 358 mlx5_core_err(dev, "Failed to init multipath lag err=%d\n", 359 err); 360 361 ldev->ports = MLX5_CAP_GEN(dev, num_lag_ports); 362 ldev->buckets = 1; 363 364 return ldev; 365 } 366 367 int mlx5_lag_dev_get_netdev_idx(struct mlx5_lag *ldev, 368 struct net_device *ndev) 369 { 370 struct lag_func *pf; 371 int i; 372 373 mlx5_ldev_for_each(i, 0, ldev) { 374 pf = mlx5_lag_pf(ldev, i); 375 if (pf->netdev == ndev) 376 return i; 377 } 378 379 return -ENOENT; 380 } 381 382 static int mlx5_lag_get_master_idx(struct mlx5_lag *ldev) 383 { 384 unsigned long idx = 0; 385 void *entry; 386 387 if (!ldev) 388 return -ENOENT; 389 390 entry = xa_find(&ldev->pfs, &idx, U8_MAX, MLX5_LAG_XA_MARK_MASTER); 391 if (!entry) 392 return -ENOENT; 393 394 return (int)idx; 395 } 396 397 int mlx5_lag_get_dev_index_by_seq(struct mlx5_lag *ldev, int seq) 398 { 399 int master_idx, i, num = 0; 400 401 if (!ldev) 402 return -ENOENT; 403 404 master_idx = mlx5_lag_get_master_idx(ldev); 405 406 /* If seq 0 is requested and there's a primary PF, return it */ 407 if (master_idx >= 0) { 408 if (seq == 0) 409 return master_idx; 410 num++; 411 } 412 413 mlx5_ldev_for_each(i, 0, ldev) { 414 /* Skip the primary PF in the loop */ 415 if (i == master_idx) 416 continue; 417 418 if (num == seq) 419 return i; 420 num++; 421 } 422 return -ENOENT; 423 } 424 425 /* Return the appropriate iterator filter for a device in LAG: 426 * - SD shared FDB active: iterate only the device's SD group 427 * - SD group exists but shared FDB not active: iterate all devices 428 * - No SD: iterate ports only 429 */ 430 static u32 mlx5_lag_get_filter(struct mlx5_lag *ldev, struct mlx5_core_dev *dev) 431 { 432 struct lag_func *pf = mlx5_lag_pf_by_dev(ldev, dev); 433 434 if (pf && pf->sd_fdb_active) 435 return pf->group_id; 436 if (pf && pf->group_id) 437 return MLX5_LAG_FILTER_ALL; 438 return MLX5_LAG_FILTER_PORTS; 439 } 440 441 /* Reverse of mlx5_lag_get_dev_index_by_seq: given a device, return its 442 * sequence number in the LAG. Master is always 0, others numbered 443 * sequentially starting from 1. 444 */ 445 int mlx5_lag_get_dev_seq(struct mlx5_core_dev *dev) 446 { 447 struct mlx5_lag *ldev = mlx5_lag_dev(dev); 448 int master_idx, i, num = 1; 449 struct lag_func *pf; 450 u32 filter; 451 452 if (!ldev) 453 return -ENOENT; 454 455 filter = mlx5_lag_get_filter(ldev, dev); 456 master_idx = mlx5_lag_get_dev_index_by_seq_filter(ldev, 0, filter); 457 if (master_idx < 0) 458 return -ENOENT; 459 460 pf = mlx5_lag_pf(ldev, master_idx); 461 if (pf && pf->dev == dev) 462 return 0; 463 464 mlx5_lag_for_each(i, 0, ldev, filter) { 465 if (i == master_idx) 466 continue; 467 pf = mlx5_lag_pf(ldev, i); 468 if (pf->dev == dev) 469 return num; 470 num++; 471 } 472 return -ENOENT; 473 } 474 EXPORT_SYMBOL(mlx5_lag_get_dev_seq); 475 476 /* seq 0 = master, then all remaining devices */ 477 static int mlx5_lag_get_dev_index_by_seq_all(struct mlx5_lag *ldev, int seq) 478 { 479 int master_idx, i, num = 0; 480 481 master_idx = mlx5_lag_get_master_idx(ldev); 482 483 if (master_idx >= 0) { 484 if (seq == 0) 485 return master_idx; 486 num++; 487 } 488 489 mlx5_lag_for_each(i, 0, ldev, MLX5_LAG_FILTER_ALL) { 490 if (i == master_idx) 491 continue; 492 if (num == seq) 493 return i; 494 num++; 495 } 496 return -ENOENT; 497 } 498 499 /* From group POV, port-marked entry is the lag master */ 500 static int mlx5_lag_get_dev_index_by_seq_group(struct mlx5_lag *ldev, int seq, 501 u32 group_id) 502 { 503 int i, num = 0; 504 505 mlx5_lag_for_each(i, 0, ldev, group_id) { 506 if (xa_get_mark(&ldev->pfs, i, MLX5_LAG_XA_MARK_PORT)) { 507 if (seq == 0) 508 return i; 509 num++; 510 break; 511 } 512 } 513 514 mlx5_lag_for_each(i, 0, ldev, group_id) { 515 if (xa_get_mark(&ldev->pfs, i, MLX5_LAG_XA_MARK_PORT)) 516 continue; 517 if (num == seq) 518 return i; 519 num++; 520 } 521 return -ENOENT; 522 } 523 524 int mlx5_lag_get_dev_index_by_seq_filter(struct mlx5_lag *ldev, int seq, 525 u32 filter) 526 { 527 if (!ldev) 528 return -ENOENT; 529 530 if (!filter || filter == MLX5_LAG_FILTER_PORTS) 531 return mlx5_lag_get_dev_index_by_seq(ldev, seq); 532 533 if (filter == MLX5_LAG_FILTER_ALL) 534 return mlx5_lag_get_dev_index_by_seq_all(ldev, seq); 535 536 return mlx5_lag_get_dev_index_by_seq_group(ldev, seq, filter); 537 } 538 539 /* Devcom events for LAG master marking */ 540 #define LAG_DEVCOM_PAIR (0) 541 #define LAG_DEVCOM_UNPAIR (1) 542 543 static void mlx5_lag_mark_master(struct mlx5_lag *ldev) 544 { 545 int lowest_dev_idx = INT_MAX; 546 struct lag_func *pf; 547 int master_xa_idx = -1; 548 int dev_idx; 549 int i; 550 551 mlx5_ldev_for_each(i, 0, ldev) { 552 pf = mlx5_lag_pf(ldev, i); 553 dev_idx = mlx5_get_dev_index(pf->dev); 554 if (dev_idx < lowest_dev_idx) { 555 lowest_dev_idx = dev_idx; 556 master_xa_idx = i; 557 } 558 } 559 560 if (master_xa_idx >= 0) 561 xa_set_mark(&ldev->pfs, master_xa_idx, MLX5_LAG_XA_MARK_MASTER); 562 } 563 564 static void mlx5_lag_clear_master(struct mlx5_lag *ldev) 565 { 566 unsigned long idx = 0; 567 void *entry; 568 569 entry = xa_find(&ldev->pfs, &idx, U8_MAX, MLX5_LAG_XA_MARK_MASTER); 570 if (!entry) 571 return; 572 573 xa_clear_mark(&ldev->pfs, idx, MLX5_LAG_XA_MARK_MASTER); 574 } 575 576 /* Devcom event handler to manage LAG master marking */ 577 static int mlx5_lag_devcom_event(int event, void *my_data, void *event_data) 578 { 579 struct mlx5_core_dev *dev = my_data; 580 struct mlx5_lag *ldev; 581 int idx; 582 583 ldev = mlx5_lag_dev(dev); 584 if (!ldev) 585 return 0; 586 587 mutex_lock(&ldev->lock); 588 switch (event) { 589 case LAG_DEVCOM_PAIR: 590 /* No need to mark more than once */ 591 idx = mlx5_lag_get_master_idx(ldev); 592 if (idx >= 0) 593 break; 594 /* Check if all LAG ports are now registered */ 595 if (mlx5_lag_num_devs(ldev) == ldev->ports) 596 mlx5_lag_mark_master(ldev); 597 break; 598 599 case LAG_DEVCOM_UNPAIR: 600 /* Clear master mark when a device is removed */ 601 mlx5_lag_clear_master(ldev); 602 break; 603 } 604 mutex_unlock(&ldev->lock); 605 return 0; 606 } 607 608 int mlx5_lag_num_devs(struct mlx5_lag *ldev) 609 { 610 int i, num = 0; 611 612 if (!ldev) 613 return 0; 614 615 mlx5_ldev_for_each(i, 0, ldev) { 616 (void)i; 617 num++; 618 } 619 return num; 620 } 621 622 int mlx5_lag_num_netdevs(struct mlx5_lag *ldev) 623 { 624 struct lag_func *pf; 625 int i, num = 0; 626 627 if (!ldev) 628 return 0; 629 630 mlx5_ldev_for_each(i, 0, ldev) { 631 pf = mlx5_lag_pf(ldev, i); 632 if (pf->netdev) 633 num++; 634 } 635 return num; 636 } 637 638 static bool __mlx5_lag_is_roce(struct mlx5_lag *ldev) 639 { 640 return ldev->mode == MLX5_LAG_MODE_ROCE; 641 } 642 643 static bool __mlx5_lag_is_sriov(struct mlx5_lag *ldev) 644 { 645 return ldev->mode == MLX5_LAG_MODE_SRIOV; 646 } 647 648 static bool __mlx5_lag_is_sd_active(struct mlx5_lag *ldev, 649 struct mlx5_core_dev *dev) 650 { 651 struct lag_func *pf = mlx5_lag_pf_by_dev(ldev, dev); 652 653 return pf && pf->sd_fdb_active; 654 } 655 656 /* Create a mapping between steering slots and active ports. 657 * As we have ldev->buckets slots per port first assume the native 658 * mapping should be used. 659 * If there are ports that are disabled fill the relevant slots 660 * with mapping that points to active ports. 661 */ 662 static void mlx5_infer_tx_affinity_mapping(struct lag_tracker *tracker, 663 struct mlx5_lag *ldev, 664 u8 buckets, 665 u8 *ports) 666 { 667 int disabled[MLX5_MAX_PORTS] = {}; 668 int enabled[MLX5_MAX_PORTS] = {}; 669 int disabled_ports_num = 0; 670 int enabled_ports_num = 0; 671 int idx; 672 u32 rand; 673 int i; 674 int j; 675 676 mlx5_ldev_for_each(i, 0, ldev) { 677 if (tracker->netdev_state[i].tx_enabled && 678 tracker->netdev_state[i].link_up) 679 enabled[enabled_ports_num++] = i; 680 else 681 disabled[disabled_ports_num++] = i; 682 } 683 684 /* Use native mapping by default where each port's buckets 685 * point the native port: 1 1 1 .. 1 2 2 2 ... 2 3 3 3 ... 3 etc 686 * ports[] values are 1-indexed device indices for FW. 687 */ 688 mlx5_ldev_for_each(i, 0, ldev) { 689 for (j = 0; j < buckets; j++) { 690 idx = i * buckets + j; 691 ports[idx] = mlx5_lag_xa_to_dev_idx(ldev, i) + 1; 692 } 693 } 694 695 /* If all ports are disabled/enabled keep native mapping */ 696 if (enabled_ports_num == ldev->ports || 697 disabled_ports_num == ldev->ports) 698 return; 699 700 /* Go over the disabled ports and for each assign a random active port */ 701 for (i = 0; i < disabled_ports_num; i++) { 702 for (j = 0; j < buckets; j++) { 703 int rand_xa_idx; 704 705 get_random_bytes(&rand, 4); 706 rand_xa_idx = enabled[rand % enabled_ports_num]; 707 ports[disabled[i] * buckets + j] = 708 mlx5_lag_xa_to_dev_idx(ldev, rand_xa_idx) + 1; 709 } 710 } 711 } 712 713 static bool mlx5_lag_has_drop_rule(struct mlx5_lag *ldev) 714 { 715 struct lag_func *pf; 716 int i; 717 718 mlx5_ldev_for_each(i, 0, ldev) { 719 pf = mlx5_lag_pf(ldev, i); 720 if (pf->has_drop) 721 return true; 722 } 723 return false; 724 } 725 726 static void mlx5_lag_drop_rule_cleanup(struct mlx5_lag *ldev) 727 { 728 struct lag_func *pf; 729 int i; 730 731 mlx5_ldev_for_each(i, 0, ldev) { 732 pf = mlx5_lag_pf(ldev, i); 733 if (!pf->has_drop) 734 continue; 735 736 mlx5_esw_acl_ingress_vport_drop_rule_destroy(pf->dev->priv.eswitch, 737 MLX5_VPORT_UPLINK); 738 pf->has_drop = false; 739 } 740 } 741 742 static void mlx5_lag_drop_rule_setup(struct mlx5_lag *ldev, 743 struct lag_tracker *tracker) 744 { 745 u8 disabled_ports[MLX5_MAX_PORTS] = {}; 746 struct mlx5_core_dev *dev; 747 struct lag_func *pf; 748 int disabled_index; 749 int num_disabled; 750 int err; 751 int i; 752 753 /* First delete the current drop rule so there won't be any dropped 754 * packets 755 */ 756 mlx5_lag_drop_rule_cleanup(ldev); 757 758 if (!ldev->tracker.has_inactive) 759 return; 760 761 mlx5_infer_tx_disabled(tracker, ldev, disabled_ports, &num_disabled); 762 763 for (i = 0; i < num_disabled; i++) { 764 disabled_index = disabled_ports[i]; 765 pf = mlx5_lag_pf(ldev, disabled_index); 766 dev = pf->dev; 767 err = mlx5_esw_acl_ingress_vport_drop_rule_create(dev->priv.eswitch, 768 MLX5_VPORT_UPLINK); 769 if (!err) 770 pf->has_drop = true; 771 else 772 mlx5_core_err(dev, 773 "Failed to create lag drop rule, error: %d", err); 774 } 775 } 776 777 static int mlx5_cmd_modify_active_port(struct mlx5_core_dev *dev, u8 ports) 778 { 779 u32 in[MLX5_ST_SZ_DW(modify_lag_in)] = {}; 780 void *lag_ctx; 781 782 lag_ctx = MLX5_ADDR_OF(modify_lag_in, in, ctx); 783 784 MLX5_SET(modify_lag_in, in, opcode, MLX5_CMD_OP_MODIFY_LAG); 785 MLX5_SET(modify_lag_in, in, field_select, 0x2); 786 787 MLX5_SET(lagc, lag_ctx, active_port, ports); 788 789 return mlx5_cmd_exec_in(dev, modify_lag, in); 790 } 791 792 static int _mlx5_modify_lag(struct mlx5_lag *ldev, 793 struct lag_tracker *tracker, u8 *ports) 794 { 795 int idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 796 struct mlx5_core_dev *dev0; 797 u8 active_ports; 798 int ret; 799 800 if (idx < 0) 801 return -EINVAL; 802 803 dev0 = mlx5_lag_pf(ldev, idx)->dev; 804 if (test_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, &ldev->mode_flags)) { 805 ret = mlx5_lag_port_sel_modify(ldev, ports); 806 if (ret || 807 !MLX5_CAP_PORT_SELECTION(dev0, port_select_flow_table_bypass)) 808 return ret; 809 810 active_ports = lag_active_port_bits(ldev, tracker); 811 812 return mlx5_cmd_modify_active_port(dev0, active_ports); 813 } 814 return mlx5_cmd_modify_lag(dev0, ldev, ports); 815 } 816 817 static struct net_device *mlx5_lag_active_backup_get_netdev(struct mlx5_core_dev *dev) 818 { 819 struct net_device *ndev = NULL; 820 struct lag_func *pf; 821 struct mlx5_lag *ldev; 822 unsigned long flags; 823 int i, last_idx; 824 825 spin_lock_irqsave(&lag_lock, flags); 826 ldev = mlx5_lag_dev(dev); 827 828 if (!ldev) 829 goto unlock; 830 831 mlx5_ldev_for_each(i, 0, ldev) { 832 pf = mlx5_lag_pf(ldev, i); 833 if (ldev->tracker.netdev_state[i].tx_enabled) 834 ndev = pf->netdev; 835 } 836 if (!ndev) { 837 last_idx = mlx5_lag_get_dev_index_by_seq(ldev, ldev->ports - 1); 838 if (last_idx < 0) 839 goto unlock; 840 pf = mlx5_lag_pf(ldev, last_idx); 841 ndev = pf->netdev; 842 } 843 844 dev_hold(ndev); 845 846 unlock: 847 spin_unlock_irqrestore(&lag_lock, flags); 848 849 return ndev; 850 } 851 852 void mlx5_modify_lag(struct mlx5_lag *ldev, 853 struct lag_tracker *tracker) 854 { 855 int first_idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 856 u8 ports[MLX5_MAX_PORTS * MLX5_LAG_MAX_HASH_BUCKETS] = {}; 857 struct mlx5_core_dev *dev0; 858 int idx; 859 int err; 860 int i; 861 int j; 862 863 if (first_idx < 0) 864 return; 865 866 dev0 = mlx5_lag_pf(ldev, first_idx)->dev; 867 mlx5_infer_tx_affinity_mapping(tracker, ldev, ldev->buckets, ports); 868 869 mlx5_ldev_for_each(i, 0, ldev) { 870 for (j = 0; j < ldev->buckets; j++) { 871 idx = i * ldev->buckets + j; 872 if (ports[idx] == ldev->v2p_map[idx]) 873 continue; 874 err = _mlx5_modify_lag(ldev, tracker, ports); 875 if (err) { 876 mlx5_core_err(dev0, 877 "Failed to modify LAG (%d)\n", 878 err); 879 return; 880 } 881 memcpy(ldev->v2p_map, ports, sizeof(ports)); 882 883 mlx5_lag_print_mapping(dev0, ldev, tracker, 884 ldev->mode_flags); 885 break; 886 } 887 } 888 889 if (tracker->tx_type == NETDEV_LAG_TX_TYPE_ACTIVEBACKUP) { 890 struct net_device *ndev = mlx5_lag_active_backup_get_netdev(dev0); 891 892 if(!(ldev->mode == MLX5_LAG_MODE_ROCE)) 893 mlx5_lag_drop_rule_setup(ldev, tracker); 894 /** Only sriov and roce lag should have tracker->tx_type set so 895 * no need to check the mode 896 */ 897 blocking_notifier_call_chain(&dev0->priv.lag_nh, 898 MLX5_DRIVER_EVENT_ACTIVE_BACKUP_LAG_CHANGE_LOWERSTATE, 899 ndev); 900 dev_put(ndev); 901 } 902 } 903 904 static int mlx5_lag_set_port_sel_mode(struct mlx5_lag *ldev, 905 enum mlx5_lag_mode mode, 906 unsigned long *flags) 907 { 908 int first_idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 909 struct mlx5_core_dev *dev0; 910 911 if (first_idx < 0) 912 return -EINVAL; 913 914 if (mode == MLX5_LAG_MODE_MPESW || 915 mode == MLX5_LAG_MODE_MULTIPATH) 916 return 0; 917 918 dev0 = mlx5_lag_pf(ldev, first_idx)->dev; 919 920 if (!MLX5_CAP_PORT_SELECTION(dev0, port_select_flow_table)) { 921 if (ldev->ports > 2) 922 return -EINVAL; 923 return 0; 924 } 925 926 if (ldev->ports > 2) 927 ldev->buckets = MLX5_LAG_MAX_HASH_BUCKETS; 928 929 set_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, flags); 930 931 return 0; 932 } 933 934 static int mlx5_lag_set_flags(struct mlx5_lag *ldev, enum mlx5_lag_mode mode, 935 struct lag_tracker *tracker, bool shared_fdb, 936 unsigned long *flags) 937 { 938 *flags = 0; 939 if (shared_fdb) { 940 set_bit(MLX5_LAG_MODE_FLAG_SHARED_FDB, flags); 941 set_bit(MLX5_LAG_MODE_FLAG_FDB_SEL_MODE_NATIVE, flags); 942 } 943 944 if (mode == MLX5_LAG_MODE_MPESW) 945 set_bit(MLX5_LAG_MODE_FLAG_FDB_SEL_MODE_NATIVE, flags); 946 947 return mlx5_lag_set_port_sel_mode(ldev, mode, flags); 948 } 949 950 char *mlx5_get_str_port_sel_mode(enum mlx5_lag_mode mode, unsigned long flags) 951 { 952 int port_sel_mode = get_port_sel_mode(mode, flags); 953 954 switch (port_sel_mode) { 955 case MLX5_LAG_PORT_SELECT_MODE_QUEUE_AFFINITY: return "queue_affinity"; 956 case MLX5_LAG_PORT_SELECT_MODE_PORT_SELECT_FT: return "hash"; 957 case MLX5_LAG_PORT_SELECT_MODE_PORT_SELECT_MPESW: return "mpesw"; 958 default: return "invalid"; 959 } 960 } 961 962 static int mlx5_create_lag(struct mlx5_lag *ldev, 963 struct lag_tracker *tracker, 964 enum mlx5_lag_mode mode, 965 unsigned long flags) 966 { 967 int first_idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 968 bool shared_fdb = test_bit(MLX5_LAG_MODE_FLAG_SHARED_FDB, &flags); 969 u32 in[MLX5_ST_SZ_DW(destroy_lag_in)] = {}; 970 struct mlx5_core_dev *dev0; 971 int err; 972 973 if (first_idx < 0) 974 return -EINVAL; 975 976 dev0 = mlx5_lag_pf(ldev, first_idx)->dev; 977 if (tracker) 978 mlx5_lag_print_mapping(dev0, ldev, tracker, flags); 979 mlx5_core_info(dev0, "shared_fdb:%d mode:%s\n", 980 shared_fdb, mlx5_get_str_port_sel_mode(mode, flags)); 981 982 err = mlx5_cmd_create_lag(dev0, ldev, tracker, mode, flags); 983 if (err) { 984 mlx5_core_err(dev0, 985 "Failed to create LAG (%d)\n", 986 err); 987 return err; 988 } 989 990 if (shared_fdb) { 991 err = mlx5_lag_create_single_fdb(ldev); 992 if (err) 993 mlx5_core_err(dev0, "Can't enable single FDB mode\n"); 994 else 995 mlx5_core_info(dev0, "Operation mode is single FDB\n"); 996 } 997 998 if (err) { 999 MLX5_SET(destroy_lag_in, in, opcode, MLX5_CMD_OP_DESTROY_LAG); 1000 if (mlx5_cmd_exec_in(dev0, destroy_lag, in)) 1001 mlx5_core_err(dev0, 1002 "Failed to deactivate RoCE LAG; driver restart required\n"); 1003 } 1004 BLOCKING_INIT_NOTIFIER_HEAD(&dev0->priv.lag_nh); 1005 1006 return err; 1007 } 1008 1009 int mlx5_activate_lag(struct mlx5_lag *ldev, 1010 struct lag_tracker *tracker, 1011 enum mlx5_lag_mode mode, 1012 bool shared_fdb) 1013 { 1014 bool roce_lag = mode == MLX5_LAG_MODE_ROCE; 1015 struct mlx5_core_dev *dev0; 1016 unsigned long flags = 0; 1017 int master_idx; 1018 int err; 1019 1020 master_idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 1021 if (master_idx < 0) 1022 return -EINVAL; 1023 1024 dev0 = mlx5_lag_pf(ldev, master_idx)->dev; 1025 err = mlx5_lag_set_flags(ldev, mode, tracker, shared_fdb, &flags); 1026 if (err) 1027 return err; 1028 1029 if (mode != MLX5_LAG_MODE_MPESW) { 1030 mlx5_infer_tx_affinity_mapping(tracker, ldev, ldev->buckets, ldev->v2p_map); 1031 if (test_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, &flags)) { 1032 err = mlx5_lag_port_sel_create(ldev, tracker->hash_type, 1033 ldev->v2p_map); 1034 if (err) { 1035 mlx5_core_err(dev0, 1036 "Failed to create LAG port selection(%d)\n", 1037 err); 1038 return err; 1039 } 1040 } 1041 } 1042 1043 err = mlx5_create_lag(ldev, tracker, mode, flags); 1044 if (err) { 1045 if (test_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, &flags)) 1046 mlx5_lag_port_sel_destroy(ldev); 1047 if (roce_lag) 1048 mlx5_core_err(dev0, 1049 "Failed to activate RoCE LAG\n"); 1050 else 1051 mlx5_core_err(dev0, 1052 "Failed to activate VF LAG\n" 1053 "Make sure all VFs are unbound prior to VF LAG activation or deactivation\n"); 1054 return err; 1055 } 1056 1057 if (tracker && tracker->tx_type == NETDEV_LAG_TX_TYPE_ACTIVEBACKUP && 1058 !roce_lag) 1059 mlx5_lag_drop_rule_setup(ldev, tracker); 1060 1061 ldev->mode = mode; 1062 ldev->mode_flags = flags; 1063 return 0; 1064 } 1065 1066 int mlx5_deactivate_lag(struct mlx5_lag *ldev) 1067 { 1068 int master_idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 1069 u32 in[MLX5_ST_SZ_DW(destroy_lag_in)] = {}; 1070 bool roce_lag = __mlx5_lag_is_roce(ldev); 1071 unsigned long flags = ldev->mode_flags; 1072 struct mlx5_core_dev *dev0; 1073 int err; 1074 1075 if (master_idx < 0) 1076 return -EINVAL; 1077 1078 dev0 = mlx5_lag_pf(ldev, master_idx)->dev; 1079 ldev->mode = MLX5_LAG_MODE_NONE; 1080 ldev->mode_flags = 0; 1081 mlx5_lag_mp_reset(ldev); 1082 1083 if (test_bit(MLX5_LAG_MODE_FLAG_SHARED_FDB, &flags)) { 1084 mlx5_lag_destroy_single_fdb(ldev); 1085 clear_bit(MLX5_LAG_MODE_FLAG_SHARED_FDB, &flags); 1086 } 1087 1088 MLX5_SET(destroy_lag_in, in, opcode, MLX5_CMD_OP_DESTROY_LAG); 1089 err = mlx5_cmd_exec_in(dev0, destroy_lag, in); 1090 if (err) { 1091 if (roce_lag) { 1092 mlx5_core_err(dev0, 1093 "Failed to deactivate RoCE LAG; driver restart required\n"); 1094 } else { 1095 mlx5_core_err(dev0, 1096 "Failed to deactivate VF LAG; driver restart required\n" 1097 "Make sure all VFs are unbound prior to VF LAG activation or deactivation\n"); 1098 } 1099 return err; 1100 } 1101 1102 if (test_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, &flags)) { 1103 mlx5_lag_port_sel_destroy(ldev); 1104 ldev->buckets = 1; 1105 } 1106 if (mlx5_lag_has_drop_rule(ldev)) 1107 mlx5_lag_drop_rule_cleanup(ldev); 1108 1109 return 0; 1110 } 1111 1112 bool mlx5_lag_check_prereq(struct mlx5_lag *ldev) 1113 { 1114 int master_idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 1115 #ifdef CONFIG_MLX5_ESWITCH 1116 struct mlx5_core_dev *dev; 1117 u8 mode; 1118 #endif 1119 struct lag_func *pf; 1120 bool roce_support; 1121 int i; 1122 1123 if (master_idx < 0 || mlx5_lag_num_devs(ldev) != ldev->ports) 1124 return false; 1125 1126 #ifdef CONFIG_MLX5_ESWITCH 1127 mlx5_ldev_for_each(i, 0, ldev) { 1128 pf = mlx5_lag_pf(ldev, i); 1129 dev = pf->dev; 1130 if (mlx5_eswitch_num_vfs(dev->priv.eswitch) && !is_mdev_switchdev_mode(dev)) 1131 return false; 1132 } 1133 1134 pf = mlx5_lag_pf(ldev, master_idx); 1135 dev = pf->dev; 1136 mode = mlx5_eswitch_mode(dev); 1137 mlx5_ldev_for_each(i, 0, ldev) { 1138 pf = mlx5_lag_pf(ldev, i); 1139 if (mlx5_eswitch_mode(pf->dev) != mode) 1140 return false; 1141 } 1142 1143 #else 1144 mlx5_ldev_for_each(i, 0, ldev) { 1145 pf = mlx5_lag_pf(ldev, i); 1146 if (mlx5_sriov_is_enabled(pf->dev)) 1147 return false; 1148 } 1149 #endif 1150 pf = mlx5_lag_pf(ldev, master_idx); 1151 roce_support = mlx5_get_roce_state(pf->dev); 1152 mlx5_ldev_for_each(i, 0, ldev) { 1153 if (i == master_idx) 1154 continue; 1155 pf = mlx5_lag_pf(ldev, i); 1156 if (mlx5_get_roce_state(pf->dev) != roce_support) 1157 return false; 1158 } 1159 1160 return true; 1161 } 1162 1163 static void mlx5_lag_assert_locked_transition(struct mlx5_lag *ldev, u32 filter) 1164 { 1165 struct mlx5_devcom_comp_dev *devcom = NULL; 1166 struct lag_func *pf; 1167 int i; 1168 1169 lockdep_assert_held(&ldev->lock); 1170 1171 i = mlx5_get_next_lag_func(ldev, 0, filter); 1172 if (i < MLX5_MAX_PORTS) { 1173 pf = mlx5_lag_pf(ldev, i); 1174 if (filter == MLX5_LAG_FILTER_PORTS || 1175 filter == MLX5_LAG_FILTER_ALL) 1176 devcom = pf->dev->priv.hca_devcom_comp; 1177 else 1178 devcom = mlx5_sd_get_devcom(pf->dev); 1179 } 1180 mlx5_devcom_comp_assert_locked(devcom); 1181 } 1182 1183 static void mlx5_lag_drop_lock_for_reps(struct mlx5_lag *ldev, u32 filter) 1184 { 1185 mlx5_lag_assert_locked_transition(ldev, filter); 1186 1187 /* Keep PF membership stable while ldev->lock is dropped. Device add 1188 * and remove paths observe mode_changes_in_progress and retry. 1189 */ 1190 ldev->mode_changes_in_progress++; 1191 mutex_unlock(&ldev->lock); 1192 } 1193 1194 static void mlx5_lag_retake_lock_after_reps(struct mlx5_lag *ldev) 1195 { 1196 mutex_lock(&ldev->lock); 1197 ldev->mode_changes_in_progress--; 1198 } 1199 1200 void mlx5_lag_rescan_dev_locked(struct mlx5_lag *ldev, 1201 struct mlx5_core_dev *dev, 1202 bool enable) 1203 { 1204 if (dev->priv.flags & MLX5_PRIV_FLAGS_DISABLE_ALL_ADEV) 1205 return; 1206 1207 if (enable) 1208 dev->priv.flags &= ~MLX5_PRIV_FLAGS_DISABLE_IB_ADEV; 1209 else 1210 dev->priv.flags |= MLX5_PRIV_FLAGS_DISABLE_IB_ADEV; 1211 1212 /* Auxiliary bus probe/remove can register or unregister representor 1213 * callbacks and take reps_lock. Drop ldev->lock so the only ordering 1214 * remains reps_lock -> ldev->lock from representor callbacks. 1215 */ 1216 mlx5_lag_drop_lock_for_reps(ldev, mlx5_lag_get_filter(ldev, dev)); 1217 mlx5_rescan_drivers_locked(dev); 1218 mlx5_lag_retake_lock_after_reps(ldev); 1219 } 1220 1221 static void mlx5_lag_rescan_devices_locked_filter(struct mlx5_lag *ldev, 1222 bool enable, u32 filter) 1223 { 1224 struct mlx5_core_dev *devs[MLX5_MAX_PORTS]; 1225 struct lag_func *pf; 1226 int num_devs = 0; 1227 int i; 1228 1229 mlx5_lag_assert_locked_transition(ldev, filter); 1230 1231 mlx5_lag_for_each(i, 0, ldev, filter) { 1232 pf = mlx5_lag_pf(ldev, i); 1233 if (pf->dev->priv.flags & MLX5_PRIV_FLAGS_DISABLE_ALL_ADEV) 1234 continue; 1235 1236 if (enable) 1237 pf->dev->priv.flags &= ~MLX5_PRIV_FLAGS_DISABLE_IB_ADEV; 1238 else 1239 pf->dev->priv.flags |= MLX5_PRIV_FLAGS_DISABLE_IB_ADEV; 1240 devs[num_devs++] = pf->dev; 1241 } 1242 1243 mlx5_lag_drop_lock_for_reps(ldev, filter); 1244 for (i = 0; i < num_devs; i++) 1245 mlx5_rescan_drivers_locked(devs[i]); 1246 mlx5_lag_retake_lock_after_reps(ldev); 1247 } 1248 1249 void mlx5_lag_add_devices_filter(struct mlx5_lag *ldev, u32 filter) 1250 { 1251 mlx5_lag_rescan_devices_locked_filter(ldev, true, filter); 1252 } 1253 1254 void mlx5_lag_add_devices(struct mlx5_lag *ldev) 1255 { 1256 mlx5_lag_add_devices_filter(ldev, MLX5_LAG_FILTER_PORTS); 1257 } 1258 1259 void mlx5_lag_remove_devices_filter(struct mlx5_lag *ldev, u32 filter) 1260 { 1261 mlx5_lag_rescan_devices_locked_filter(ldev, false, filter); 1262 } 1263 1264 void mlx5_lag_remove_devices(struct mlx5_lag *ldev) 1265 { 1266 mlx5_lag_remove_devices_filter(ldev, MLX5_LAG_FILTER_PORTS); 1267 } 1268 1269 static int mlx5_lag_reload_ib_reps_idx(struct mlx5_lag *ldev, int idx, 1270 u32 flags) 1271 { 1272 struct lag_func *pf = mlx5_lag_pf(ldev, idx); 1273 struct mlx5_eswitch *esw; 1274 int ret; 1275 1276 if (pf->dev->priv.flags & flags) 1277 return 0; 1278 1279 esw = pf->dev->priv.eswitch; 1280 mlx5_esw_reps_block(esw); 1281 ret = mlx5_eswitch_reload_ib_reps(esw); 1282 mlx5_esw_reps_unblock(esw); 1283 1284 return ret; 1285 } 1286 1287 static int mlx5_lag_reload_ib_reps_unlocked(struct mlx5_lag *ldev, u32 flags, 1288 u32 filter, bool cont_on_fail) 1289 { 1290 int master_idx = mlx5_lag_get_dev_index_by_seq_filter(ldev, MLX5_LAG_P1, 1291 filter); 1292 int ret; 1293 int i; 1294 1295 if (master_idx < 0) 1296 return -EINVAL; 1297 1298 ret = mlx5_lag_reload_ib_reps_idx(ldev, master_idx, flags); 1299 if (ret && !cont_on_fail) 1300 return ret; 1301 1302 mlx5_lag_for_each(i, 0, ldev, filter) { 1303 if (i == master_idx) 1304 continue; 1305 ret = mlx5_lag_reload_ib_reps_idx(ldev, i, flags); 1306 if (ret && !cont_on_fail) 1307 return ret; 1308 } 1309 1310 return 0; 1311 } 1312 1313 static int mlx5_lag_reload_ib_reps(struct mlx5_lag *ldev, u32 flags, 1314 u32 filter, bool cont_on_fail) 1315 { 1316 int ret; 1317 1318 /* The HCA devcom component lock serializes LAG mode transitions while 1319 * ldev->lock is dropped here. Dropping ldev->lock is required because 1320 * the reload takes the per-E-Switch reps_lock, and representor 1321 * load/unload callbacks can re-enter LAG netdev add/remove and take 1322 * ldev->lock. Keep the ordering reps_lock -> ldev->lock. 1323 */ 1324 mlx5_lag_drop_lock_for_reps(ldev, filter); 1325 ret = mlx5_lag_reload_ib_reps_unlocked(ldev, flags, filter, 1326 cont_on_fail); 1327 mlx5_lag_retake_lock_after_reps(ldev); 1328 1329 return ret; 1330 } 1331 1332 int mlx5_lag_reload_ib_reps_from_locked(struct mlx5_lag *ldev, u32 flags, 1333 u32 filter, bool cont_on_fail) 1334 { 1335 return mlx5_lag_reload_ib_reps(ldev, flags, filter, cont_on_fail); 1336 } 1337 1338 static void mlx5_lag_unload_reps_unlocked(struct mlx5_lag *ldev, u32 filter) 1339 { 1340 struct lag_func *pf; 1341 int i; 1342 1343 mlx5_lag_for_each(i, 0, ldev, filter) { 1344 struct mlx5_eswitch *esw; 1345 1346 pf = mlx5_lag_pf(ldev, i); 1347 esw = pf->dev->priv.eswitch; 1348 mlx5_esw_reps_block(esw); 1349 mlx5_eswitch_unload_reps(esw); 1350 mlx5_esw_reps_unblock(esw); 1351 } 1352 } 1353 1354 void mlx5_lag_unload_reps_from_locked(struct mlx5_lag *ldev, u32 filter) 1355 { 1356 /* Same lock dance as mlx5_lag_reload_ib_reps: drop ldev->lock around 1357 * the per-eswitch reps_lock to keep the reps_lock -> ldev->lock order. 1358 */ 1359 mlx5_lag_drop_lock_for_reps(ldev, filter); 1360 mlx5_lag_unload_reps_unlocked(ldev, filter); 1361 mlx5_lag_retake_lock_after_reps(ldev); 1362 } 1363 1364 void mlx5_disable_lag(struct mlx5_lag *ldev) 1365 { 1366 bool shared_fdb = test_bit(MLX5_LAG_MODE_FLAG_SHARED_FDB, &ldev->mode_flags); 1367 int idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 1368 struct mlx5_core_dev *dev0; 1369 bool roce_lag; 1370 int err; 1371 int i; 1372 1373 if (idx < 0) 1374 return; 1375 1376 if (shared_fdb) { 1377 mlx5_lag_shared_fdb_destroy(ldev, 0); 1378 return; 1379 } 1380 1381 dev0 = mlx5_lag_pf(ldev, idx)->dev; 1382 roce_lag = __mlx5_lag_is_roce(ldev); 1383 1384 if (roce_lag) { 1385 mlx5_lag_rescan_dev_locked(ldev, dev0, false); 1386 mlx5_ldev_for_each(i, 0, ldev) { 1387 if (i == idx) 1388 continue; 1389 mlx5_nic_vport_disable_roce(mlx5_lag_pf(ldev, i)->dev); 1390 } 1391 } 1392 1393 err = mlx5_deactivate_lag(ldev); 1394 if (err) 1395 return; 1396 1397 if (roce_lag) 1398 mlx5_lag_add_devices(ldev); 1399 } 1400 1401 static bool mlx5_lag_is_roce_lag(struct mlx5_lag *ldev) 1402 { 1403 bool roce_lag = true; 1404 struct lag_func *pf; 1405 int i; 1406 1407 mlx5_ldev_for_each(i, 0, ldev) { 1408 pf = mlx5_lag_pf(ldev, i); 1409 roce_lag = roce_lag && !mlx5_sriov_is_enabled(pf->dev); 1410 } 1411 1412 #ifdef CONFIG_MLX5_ESWITCH 1413 mlx5_ldev_for_each(i, 0, ldev) { 1414 pf = mlx5_lag_pf(ldev, i); 1415 roce_lag = roce_lag && is_mdev_legacy_mode(pf->dev); 1416 } 1417 #endif 1418 1419 return roce_lag; 1420 } 1421 1422 static bool mlx5_lag_should_modify_lag(struct mlx5_lag *ldev, bool do_bond) 1423 { 1424 return do_bond && __mlx5_lag_is_active(ldev) && 1425 ldev->mode != MLX5_LAG_MODE_MPESW; 1426 } 1427 1428 static bool mlx5_lag_should_disable_lag(struct mlx5_lag *ldev, bool do_bond) 1429 { 1430 return !do_bond && __mlx5_lag_is_active(ldev) && 1431 ldev->mode != MLX5_LAG_MODE_MPESW; 1432 } 1433 1434 #ifdef CONFIG_MLX5_ESWITCH 1435 static int 1436 mlx5_lag_sum_devices_speed(struct mlx5_lag *ldev, u32 *sum_speed, 1437 int (*get_speed)(struct mlx5_core_dev *, u32 *)) 1438 { 1439 struct mlx5_core_dev *pf_mdev; 1440 struct lag_func *pf; 1441 int pf_idx; 1442 u32 speed; 1443 int ret; 1444 1445 *sum_speed = 0; 1446 mlx5_ldev_for_each(pf_idx, 0, ldev) { 1447 pf = mlx5_lag_pf(ldev, pf_idx); 1448 if (!pf) 1449 continue; 1450 pf_mdev = pf->dev; 1451 if (!pf_mdev) 1452 continue; 1453 1454 ret = get_speed(pf_mdev, &speed); 1455 if (ret) { 1456 mlx5_core_dbg(pf_mdev, 1457 "Failed to get device speed using %ps. Device %s speed is not available (err=%d)\n", 1458 get_speed, dev_name(pf_mdev->device), 1459 ret); 1460 return ret; 1461 } 1462 1463 *sum_speed += speed; 1464 } 1465 1466 return 0; 1467 } 1468 1469 static int mlx5_lag_sum_devices_max_speed(struct mlx5_lag *ldev, u32 *max_speed) 1470 { 1471 return mlx5_lag_sum_devices_speed(ldev, max_speed, 1472 mlx5_port_max_linkspeed); 1473 } 1474 1475 static int mlx5_lag_sum_devices_oper_speed(struct mlx5_lag *ldev, 1476 u32 *oper_speed) 1477 { 1478 return mlx5_lag_sum_devices_speed(ldev, oper_speed, 1479 mlx5_port_oper_linkspeed); 1480 } 1481 1482 static void mlx5_lag_modify_device_vports_speed(struct mlx5_core_dev *mdev, 1483 u32 speed) 1484 { 1485 u16 op_mod = MLX5_VPORT_STATE_OP_MOD_ESW_VPORT; 1486 struct mlx5_eswitch *esw = mdev->priv.eswitch; 1487 struct mlx5_vport *vport; 1488 unsigned long i; 1489 int ret; 1490 1491 if (!esw) 1492 return; 1493 1494 if (!MLX5_CAP_ESW(mdev, esw_vport_state_max_tx_speed)) 1495 return; 1496 1497 mutex_lock(&esw->state_lock); 1498 mlx5_esw_for_each_vport(esw, i, vport) { 1499 if (!vport) 1500 continue; 1501 1502 if (vport->vport == MLX5_VPORT_UPLINK) 1503 continue; 1504 1505 vport->agg_max_tx_speed = speed; 1506 1507 if (!vport->enabled) 1508 continue; 1509 1510 ret = mlx5_modify_vport_max_tx_speed(mdev, op_mod, 1511 vport->vport, true, speed); 1512 if (ret) 1513 mlx5_core_dbg(mdev, 1514 "Failed to set vport %d speed %d, err=%d\n", 1515 vport->vport, speed, ret); 1516 } 1517 mutex_unlock(&esw->state_lock); 1518 } 1519 1520 void mlx5_lag_set_vports_agg_speed(struct mlx5_lag *ldev) 1521 { 1522 struct mlx5_core_dev *mdev; 1523 struct lag_func *pf; 1524 u32 speed; 1525 int pf_idx; 1526 1527 if (ldev->mode == MLX5_LAG_MODE_MPESW) { 1528 if (mlx5_lag_sum_devices_oper_speed(ldev, &speed)) 1529 return; 1530 } else { 1531 speed = ldev->tracker.bond_speed_mbps; 1532 if (speed == SPEED_UNKNOWN) 1533 return; 1534 } 1535 1536 /* If speed is not set, use the sum of max speeds of all PFs */ 1537 if (!speed && mlx5_lag_sum_devices_max_speed(ldev, &speed)) 1538 return; 1539 1540 speed = speed / MLX5_MAX_TX_SPEED_UNIT; 1541 1542 mlx5_ldev_for_each(pf_idx, 0, ldev) { 1543 pf = mlx5_lag_pf(ldev, pf_idx); 1544 if (!pf) 1545 continue; 1546 mdev = pf->dev; 1547 if (!mdev) 1548 continue; 1549 1550 mlx5_lag_modify_device_vports_speed(mdev, speed); 1551 } 1552 } 1553 1554 void mlx5_lag_reset_vports_speed(struct mlx5_lag *ldev) 1555 { 1556 struct mlx5_core_dev *mdev; 1557 struct lag_func *pf; 1558 u32 speed; 1559 int pf_idx; 1560 int ret; 1561 1562 mlx5_ldev_for_each(pf_idx, 0, ldev) { 1563 pf = mlx5_lag_pf(ldev, pf_idx); 1564 if (!pf) 1565 continue; 1566 mdev = pf->dev; 1567 if (!mdev) 1568 continue; 1569 1570 ret = mlx5_port_oper_linkspeed(mdev, &speed); 1571 if (ret) { 1572 mlx5_core_dbg(mdev, 1573 "Failed to reset vports speed for device %s. Oper speed is not available (err=%d)\n", 1574 dev_name(mdev->device), ret); 1575 continue; 1576 } 1577 1578 speed = speed / MLX5_MAX_TX_SPEED_UNIT; 1579 mlx5_lag_modify_device_vports_speed(mdev, speed); 1580 } 1581 } 1582 #endif 1583 1584 static void mlx5_do_bond(struct mlx5_lag *ldev) 1585 { 1586 int idx = mlx5_lag_get_dev_index_by_seq(ldev, MLX5_LAG_P1); 1587 struct lag_tracker tracker = { }; 1588 struct mlx5_core_dev *dev0; 1589 struct net_device *ndev; 1590 bool do_bond, roce_lag; 1591 int err; 1592 int i; 1593 1594 if (idx < 0) 1595 return; 1596 1597 dev0 = mlx5_lag_pf(ldev, idx)->dev; 1598 if (!mlx5_lag_is_ready(ldev)) { 1599 do_bond = false; 1600 } else { 1601 /* VF LAG is in multipath mode, ignore bond change requests */ 1602 if (mlx5_lag_is_multipath(dev0)) 1603 return; 1604 1605 tracker = ldev->tracker; 1606 1607 do_bond = tracker.is_bonded && mlx5_lag_check_prereq(ldev); 1608 } 1609 1610 if (do_bond && !__mlx5_lag_is_active(ldev)) { 1611 bool shared_fdb = mlx5_lag_shared_fdb_supported(ldev); 1612 1613 roce_lag = mlx5_lag_is_roce_lag(ldev); 1614 1615 if (shared_fdb) { 1616 err = mlx5_lag_shared_fdb_create(ldev, &tracker, 1617 MLX5_LAG_MODE_SRIOV, 1618 0); 1619 if (err) 1620 return; 1621 } else { 1622 if (roce_lag) 1623 mlx5_lag_remove_devices(ldev); 1624 1625 err = mlx5_activate_lag(ldev, &tracker, 1626 roce_lag ? MLX5_LAG_MODE_ROCE : 1627 MLX5_LAG_MODE_SRIOV, 1628 false); 1629 if (err) { 1630 if (roce_lag) 1631 mlx5_lag_add_devices(ldev); 1632 return; 1633 } 1634 1635 if (roce_lag) { 1636 struct mlx5_core_dev *dev; 1637 1638 mlx5_lag_rescan_dev_locked(ldev, dev0, true); 1639 mlx5_ldev_for_each(i, 0, ldev) { 1640 if (i == idx) 1641 continue; 1642 dev = mlx5_lag_pf(ldev, i)->dev; 1643 if (mlx5_get_roce_state(dev)) 1644 mlx5_nic_vport_enable_roce(dev); 1645 } 1646 } 1647 } 1648 if (tracker.tx_type == NETDEV_LAG_TX_TYPE_ACTIVEBACKUP) { 1649 ndev = mlx5_lag_active_backup_get_netdev(dev0); 1650 /** Only sriov and roce lag should have tracker->TX_type 1651 * set so no need to check the mode 1652 */ 1653 blocking_notifier_call_chain(&dev0->priv.lag_nh, 1654 MLX5_DRIVER_EVENT_ACTIVE_BACKUP_LAG_CHANGE_LOWERSTATE, 1655 ndev); 1656 dev_put(ndev); 1657 } 1658 if (!shared_fdb) 1659 mlx5_lag_set_vports_agg_speed(ldev); 1660 } else if (mlx5_lag_should_modify_lag(ldev, do_bond)) { 1661 mlx5_modify_lag(ldev, &tracker); 1662 mlx5_lag_set_vports_agg_speed(ldev); 1663 } else if (mlx5_lag_should_disable_lag(ldev, do_bond)) { 1664 mlx5_lag_reset_vports_speed(ldev); 1665 mlx5_disable_lag(ldev); 1666 } 1667 } 1668 1669 /* The last mdev to unregister will destroy the workqueue before removing the 1670 * devcom component, and as all the mdevs use the same devcom component we are 1671 * guaranteed that the devcom is valid while the calling work is running. 1672 */ 1673 struct mlx5_devcom_comp_dev *mlx5_lag_get_devcom_comp(struct mlx5_lag *ldev) 1674 { 1675 struct mlx5_devcom_comp_dev *devcom = NULL; 1676 struct lag_func *pf; 1677 int i; 1678 1679 mutex_lock(&ldev->lock); 1680 i = mlx5_get_next_lag_func(ldev, 0, MLX5_LAG_FILTER_PORTS); 1681 if (i < MLX5_MAX_PORTS) { 1682 pf = mlx5_lag_pf(ldev, i); 1683 devcom = pf->dev->priv.hca_devcom_comp; 1684 } 1685 mutex_unlock(&ldev->lock); 1686 return devcom; 1687 } 1688 1689 static int mlx5_lag_demux_ft_fg_init(struct mlx5_core_dev *dev, 1690 struct mlx5_flow_table_attr *ft_attr, 1691 struct lag_func *pf) 1692 { 1693 #ifdef CONFIG_MLX5_ESWITCH 1694 struct mlx5_flow_namespace *ns; 1695 struct mlx5_flow_group *fg; 1696 int err; 1697 1698 ns = mlx5_get_flow_namespace(dev, MLX5_FLOW_NAMESPACE_LAG); 1699 if (!ns) 1700 return 0; 1701 1702 pf->lag_demux_ft = mlx5_create_flow_table(ns, ft_attr); 1703 if (IS_ERR(pf->lag_demux_ft)) 1704 return PTR_ERR(pf->lag_demux_ft); 1705 1706 fg = mlx5_esw_lag_demux_fg_create(dev->priv.eswitch, 1707 pf->lag_demux_ft); 1708 if (IS_ERR(fg)) { 1709 err = PTR_ERR(fg); 1710 mlx5_destroy_flow_table(pf->lag_demux_ft); 1711 pf->lag_demux_ft = NULL; 1712 return err; 1713 } 1714 1715 pf->lag_demux_fg = fg; 1716 return 0; 1717 #else 1718 return -EOPNOTSUPP; 1719 #endif 1720 } 1721 1722 static int mlx5_lag_demux_fw_init(struct mlx5_core_dev *dev, 1723 struct mlx5_flow_table_attr *ft_attr, 1724 struct lag_func *pf) 1725 { 1726 struct mlx5_flow_namespace *ns; 1727 int err; 1728 1729 ns = mlx5_get_flow_namespace(dev, MLX5_FLOW_NAMESPACE_LAG); 1730 if (!ns) 1731 return 0; 1732 1733 pf->lag_demux_fg = NULL; 1734 ft_attr->max_fte = 1; 1735 pf->lag_demux_ft = mlx5_create_lag_demux_flow_table(ns, ft_attr); 1736 if (IS_ERR(pf->lag_demux_ft)) { 1737 err = PTR_ERR(pf->lag_demux_ft); 1738 pf->lag_demux_ft = NULL; 1739 return err; 1740 } 1741 1742 return 0; 1743 } 1744 1745 int mlx5_lag_demux_init(struct mlx5_core_dev *dev, 1746 struct mlx5_flow_table_attr *ft_attr) 1747 { 1748 struct mlx5_lag *ldev; 1749 struct lag_func *pf; 1750 1751 if (!ft_attr) 1752 return -EINVAL; 1753 1754 ldev = mlx5_lag_dev(dev); 1755 if (!ldev) 1756 return -ENODEV; 1757 1758 pf = mlx5_lag_pf_by_dev(ldev, dev); 1759 if (!pf) 1760 return -ENODEV; 1761 1762 xa_init(&pf->lag_demux_rules); 1763 1764 if (mlx5_lag_is_sw_lag(dev)) 1765 return mlx5_lag_demux_ft_fg_init(dev, ft_attr, pf); 1766 1767 return mlx5_lag_demux_fw_init(dev, ft_attr, pf); 1768 } 1769 EXPORT_SYMBOL(mlx5_lag_demux_init); 1770 1771 void mlx5_lag_demux_cleanup(struct mlx5_core_dev *dev) 1772 { 1773 struct mlx5_flow_handle *rule; 1774 struct mlx5_lag *ldev; 1775 unsigned long vport_num; 1776 struct lag_func *pf; 1777 1778 ldev = mlx5_lag_dev(dev); 1779 if (!ldev) 1780 return; 1781 1782 pf = mlx5_lag_pf_by_dev(ldev, dev); 1783 if (!pf) 1784 return; 1785 1786 xa_for_each(&pf->lag_demux_rules, vport_num, rule) 1787 mlx5_del_flow_rules(rule); 1788 xa_destroy(&pf->lag_demux_rules); 1789 1790 if (pf->lag_demux_fg) 1791 mlx5_destroy_flow_group(pf->lag_demux_fg); 1792 if (pf->lag_demux_ft) 1793 mlx5_destroy_flow_table(pf->lag_demux_ft); 1794 pf->lag_demux_fg = NULL; 1795 pf->lag_demux_ft = NULL; 1796 } 1797 EXPORT_SYMBOL(mlx5_lag_demux_cleanup); 1798 1799 static struct lag_func *mlx5_lag_dev_get_master_pf(struct mlx5_lag *ldev, 1800 struct mlx5_core_dev *dev) 1801 { 1802 u32 filter = mlx5_lag_get_filter(ldev, dev); 1803 int idx; 1804 1805 idx = mlx5_lag_get_dev_index_by_seq_filter(ldev, MLX5_LAG_P1, filter); 1806 if (idx < 0) 1807 return NULL; 1808 1809 return mlx5_lag_pf(ldev, idx); 1810 } 1811 1812 int mlx5_lag_demux_rule_add(struct mlx5_core_dev *vport_dev, u16 vport_num, 1813 int index) 1814 { 1815 struct mlx5_flow_handle *rule; 1816 struct lag_func *master; 1817 struct mlx5_lag *ldev; 1818 int err; 1819 1820 ldev = mlx5_lag_dev(vport_dev); 1821 if (!ldev) 1822 return 0; 1823 1824 master = mlx5_lag_dev_get_master_pf(ldev, vport_dev); 1825 if (!master || !master->lag_demux_fg) 1826 return 0; 1827 1828 if (xa_load(&master->lag_demux_rules, index)) 1829 return 0; 1830 1831 rule = mlx5_esw_lag_demux_rule_create(vport_dev->priv.eswitch, 1832 vport_num, master->lag_demux_ft); 1833 if (IS_ERR(rule)) { 1834 err = PTR_ERR(rule); 1835 mlx5_core_warn(vport_dev, 1836 "Failed to create LAG demux rule for vport %u, err %d\n", 1837 vport_num, err); 1838 return err; 1839 } 1840 1841 err = xa_err(xa_store(&master->lag_demux_rules, index, rule, 1842 GFP_KERNEL)); 1843 if (err) { 1844 mlx5_del_flow_rules(rule); 1845 mlx5_core_warn(vport_dev, 1846 "Failed to store LAG demux rule for vport %u, err %d\n", 1847 vport_num, err); 1848 } 1849 1850 return err; 1851 } 1852 EXPORT_SYMBOL(mlx5_lag_demux_rule_add); 1853 1854 void mlx5_lag_demux_rule_del(struct mlx5_core_dev *dev, int index) 1855 { 1856 struct mlx5_flow_handle *rule; 1857 struct lag_func *master_pf; 1858 struct mlx5_lag *ldev; 1859 1860 ldev = mlx5_lag_dev(dev); 1861 if (!ldev) 1862 return; 1863 1864 master_pf = mlx5_lag_dev_get_master_pf(ldev, dev); 1865 if (!master_pf || !master_pf->lag_demux_fg) 1866 return; 1867 1868 rule = xa_erase(&master_pf->lag_demux_rules, index); 1869 if (rule) 1870 mlx5_del_flow_rules(rule); 1871 } 1872 EXPORT_SYMBOL(mlx5_lag_demux_rule_del); 1873 1874 static void mlx5_queue_bond_work(struct mlx5_lag *ldev, unsigned long delay) 1875 { 1876 queue_delayed_work(ldev->wq, &ldev->bond_work, delay); 1877 } 1878 1879 static void mlx5_do_bond_work(struct work_struct *work) 1880 { 1881 struct delayed_work *delayed_work = to_delayed_work(work); 1882 struct mlx5_lag *ldev = container_of(delayed_work, struct mlx5_lag, 1883 bond_work); 1884 struct mlx5_devcom_comp_dev *devcom; 1885 int status; 1886 1887 devcom = mlx5_lag_get_devcom_comp(ldev); 1888 if (!devcom) 1889 return; 1890 1891 status = mlx5_devcom_comp_trylock(devcom); 1892 if (!status) { 1893 mlx5_queue_bond_work(ldev, HZ); 1894 return; 1895 } 1896 1897 mutex_lock(&ldev->lock); 1898 if (ldev->mode_changes_in_progress) { 1899 mutex_unlock(&ldev->lock); 1900 mlx5_devcom_comp_unlock(devcom); 1901 mlx5_queue_bond_work(ldev, HZ); 1902 return; 1903 } 1904 1905 mlx5_do_bond(ldev); 1906 mutex_unlock(&ldev->lock); 1907 mlx5_devcom_comp_unlock(devcom); 1908 } 1909 1910 static int mlx5_handle_changeupper_event(struct mlx5_lag *ldev, 1911 struct lag_tracker *tracker, 1912 struct netdev_notifier_changeupper_info *info) 1913 { 1914 struct net_device *upper = info->upper_dev, *ndev_tmp; 1915 struct netdev_lag_upper_info *lag_upper_info = NULL; 1916 bool is_bonded, is_in_lag, mode_supported; 1917 bool has_inactive = 0; 1918 struct lag_func *pf; 1919 struct slave *slave; 1920 u8 bond_status = 0; 1921 int num_slaves = 0; 1922 int changed = 0; 1923 int i, idx = -1; 1924 1925 if (!netif_is_lag_master(upper)) 1926 return 0; 1927 1928 if (info->linking) 1929 lag_upper_info = info->upper_info; 1930 1931 /* The event may still be of interest if the slave does not belong to 1932 * us, but is enslaved to a master which has one or more of our netdevs 1933 * as slaves (e.g., if a new slave is added to a master that bonds two 1934 * of our netdevs, we should unbond). 1935 */ 1936 rcu_read_lock(); 1937 for_each_netdev_in_bond_rcu(upper, ndev_tmp) { 1938 mlx5_ldev_for_each(i, 0, ldev) { 1939 pf = mlx5_lag_pf(ldev, i); 1940 if (pf->netdev == ndev_tmp) { 1941 idx++; 1942 break; 1943 } 1944 } 1945 if (i < MLX5_MAX_PORTS) { 1946 if (netif_is_bond_master(upper)) { 1947 slave = bond_slave_get_rcu(ndev_tmp); 1948 if (slave) 1949 has_inactive |= bond_is_slave_inactive(slave); 1950 } 1951 bond_status |= (1 << idx); 1952 } 1953 1954 num_slaves++; 1955 } 1956 rcu_read_unlock(); 1957 1958 /* None of this lagdev's netdevs are slaves of this master. */ 1959 if (!(bond_status & GENMASK(ldev->ports - 1, 0))) 1960 return 0; 1961 1962 if (lag_upper_info) { 1963 tracker->tx_type = lag_upper_info->tx_type; 1964 tracker->hash_type = lag_upper_info->hash_type; 1965 } 1966 1967 tracker->has_inactive = has_inactive; 1968 /* Determine bonding status: 1969 * A device is considered bonded if both its physical ports are slaves 1970 * of the same lag master, and only them. 1971 */ 1972 is_in_lag = num_slaves == ldev->ports && 1973 bond_status == GENMASK(ldev->ports - 1, 0); 1974 1975 /* Lag mode must be activebackup or hash. */ 1976 mode_supported = tracker->tx_type == NETDEV_LAG_TX_TYPE_ACTIVEBACKUP || 1977 tracker->tx_type == NETDEV_LAG_TX_TYPE_HASH; 1978 1979 is_bonded = is_in_lag && mode_supported; 1980 if (tracker->is_bonded != is_bonded) { 1981 tracker->is_bonded = is_bonded; 1982 changed = 1; 1983 } 1984 1985 if (!is_in_lag) 1986 return changed; 1987 1988 if (!mlx5_lag_is_ready(ldev)) 1989 NL_SET_ERR_MSG_MOD(info->info.extack, 1990 "Can't activate LAG offload, PF is configured with more than 64 VFs"); 1991 else if (!mode_supported) 1992 NL_SET_ERR_MSG_MOD(info->info.extack, 1993 "Can't activate LAG offload, TX type isn't supported"); 1994 1995 return changed; 1996 } 1997 1998 static int mlx5_handle_changelowerstate_event(struct mlx5_lag *ldev, 1999 struct lag_tracker *tracker, 2000 struct net_device *ndev, 2001 struct netdev_notifier_changelowerstate_info *info) 2002 { 2003 struct netdev_lag_lower_state_info *lag_lower_info; 2004 int idx; 2005 2006 if (!netif_is_lag_port(ndev)) 2007 return 0; 2008 2009 idx = mlx5_lag_dev_get_netdev_idx(ldev, ndev); 2010 if (idx < 0) 2011 return 0; 2012 2013 /* This information is used to determine virtual to physical 2014 * port mapping. 2015 */ 2016 lag_lower_info = info->lower_state_info; 2017 if (!lag_lower_info) 2018 return 0; 2019 2020 tracker->netdev_state[idx] = *lag_lower_info; 2021 2022 return 1; 2023 } 2024 2025 static int mlx5_handle_changeinfodata_event(struct mlx5_lag *ldev, 2026 struct lag_tracker *tracker, 2027 struct net_device *ndev) 2028 { 2029 struct net_device *ndev_tmp; 2030 struct slave *slave; 2031 bool has_inactive = 0; 2032 int idx; 2033 2034 if (!netif_is_bond_master(ndev)) 2035 return 0; 2036 2037 rcu_read_lock(); 2038 for_each_netdev_in_bond_rcu(ndev, ndev_tmp) { 2039 idx = mlx5_lag_dev_get_netdev_idx(ldev, ndev_tmp); 2040 if (idx < 0) 2041 continue; 2042 2043 slave = bond_slave_get_rcu(ndev_tmp); 2044 if (slave) 2045 has_inactive |= bond_is_slave_inactive(slave); 2046 } 2047 rcu_read_unlock(); 2048 2049 if (tracker->has_inactive == has_inactive) 2050 return 0; 2051 2052 tracker->has_inactive = has_inactive; 2053 2054 return 1; 2055 } 2056 2057 static void mlx5_lag_update_tracker_speed(struct lag_tracker *tracker, 2058 struct net_device *ndev) 2059 { 2060 struct ethtool_link_ksettings lksettings; 2061 struct net_device *bond_dev; 2062 int err; 2063 2064 if (netif_is_lag_master(ndev)) 2065 bond_dev = ndev; 2066 else 2067 bond_dev = netdev_master_upper_dev_get(ndev); 2068 2069 if (!bond_dev) { 2070 tracker->bond_speed_mbps = SPEED_UNKNOWN; 2071 return; 2072 } 2073 2074 err = __ethtool_get_link_ksettings(bond_dev, &lksettings); 2075 if (err) { 2076 netdev_dbg(bond_dev, 2077 "Failed to get speed for bond dev %s, err=%d\n", 2078 bond_dev->name, err); 2079 tracker->bond_speed_mbps = SPEED_UNKNOWN; 2080 return; 2081 } 2082 2083 if (lksettings.base.speed == SPEED_UNKNOWN) 2084 tracker->bond_speed_mbps = 0; 2085 else 2086 tracker->bond_speed_mbps = lksettings.base.speed; 2087 } 2088 2089 /* Returns speed in Mbps. */ 2090 int mlx5_lag_query_bond_speed(struct mlx5_core_dev *mdev, u32 *speed) 2091 { 2092 struct mlx5_lag *ldev; 2093 unsigned long flags; 2094 int ret = 0; 2095 2096 spin_lock_irqsave(&lag_lock, flags); 2097 ldev = mlx5_lag_dev(mdev); 2098 if (!ldev) { 2099 ret = -ENODEV; 2100 goto unlock; 2101 } 2102 2103 *speed = ldev->tracker.bond_speed_mbps; 2104 2105 if (*speed == SPEED_UNKNOWN) { 2106 mlx5_core_dbg(mdev, "Bond speed is unknown\n"); 2107 ret = -EINVAL; 2108 } 2109 2110 unlock: 2111 spin_unlock_irqrestore(&lag_lock, flags); 2112 return ret; 2113 } 2114 EXPORT_SYMBOL_GPL(mlx5_lag_query_bond_speed); 2115 2116 /* this handler is always registered to netdev events */ 2117 static int mlx5_lag_netdev_event(struct notifier_block *this, 2118 unsigned long event, void *ptr) 2119 { 2120 struct net_device *ndev = netdev_notifier_info_to_dev(ptr); 2121 struct lag_tracker tracker; 2122 struct mlx5_lag *ldev; 2123 int changed = 0; 2124 2125 if (event != NETDEV_CHANGEUPPER && 2126 event != NETDEV_CHANGELOWERSTATE && 2127 event != NETDEV_CHANGEINFODATA) 2128 return NOTIFY_DONE; 2129 2130 ldev = container_of(this, struct mlx5_lag, nb); 2131 2132 tracker = ldev->tracker; 2133 2134 switch (event) { 2135 case NETDEV_CHANGEUPPER: 2136 changed = mlx5_handle_changeupper_event(ldev, &tracker, ptr); 2137 break; 2138 case NETDEV_CHANGELOWERSTATE: 2139 changed = mlx5_handle_changelowerstate_event(ldev, &tracker, 2140 ndev, ptr); 2141 break; 2142 case NETDEV_CHANGEINFODATA: 2143 changed = mlx5_handle_changeinfodata_event(ldev, &tracker, ndev); 2144 break; 2145 } 2146 2147 if (changed) 2148 mlx5_lag_update_tracker_speed(&tracker, ndev); 2149 2150 ldev->tracker = tracker; 2151 2152 if (changed) 2153 mlx5_queue_bond_work(ldev, 0); 2154 2155 return NOTIFY_DONE; 2156 } 2157 2158 static void mlx5_ldev_add_netdev(struct mlx5_lag *ldev, 2159 struct mlx5_core_dev *dev, 2160 struct net_device *netdev) 2161 { 2162 struct lag_func *pf; 2163 unsigned long flags; 2164 int i; 2165 2166 spin_lock_irqsave(&lag_lock, flags); 2167 /* Find pf entry by matching dev pointer */ 2168 mlx5_ldev_for_each(i, 0, ldev) { 2169 pf = mlx5_lag_pf(ldev, i); 2170 if (pf->dev == dev) { 2171 pf->netdev = netdev; 2172 ldev->tracker.netdev_state[i].link_up = 0; 2173 ldev->tracker.netdev_state[i].tx_enabled = 0; 2174 break; 2175 } 2176 } 2177 spin_unlock_irqrestore(&lag_lock, flags); 2178 } 2179 2180 static void mlx5_ldev_remove_netdev(struct mlx5_lag *ldev, 2181 struct net_device *netdev) 2182 { 2183 struct lag_func *pf; 2184 unsigned long flags; 2185 int i; 2186 2187 spin_lock_irqsave(&lag_lock, flags); 2188 mlx5_ldev_for_each(i, 0, ldev) { 2189 pf = mlx5_lag_pf(ldev, i); 2190 if (pf->netdev == netdev) { 2191 pf->netdev = NULL; 2192 break; 2193 } 2194 } 2195 spin_unlock_irqrestore(&lag_lock, flags); 2196 } 2197 2198 int mlx5_ldev_add_mdev(struct mlx5_lag *ldev, 2199 struct mlx5_core_dev *dev, 2200 u32 group_id) 2201 { 2202 struct lag_func *pf; 2203 u32 idx; 2204 int err; 2205 2206 pf = kzalloc_obj(*pf); 2207 if (!pf) 2208 return -ENOMEM; 2209 2210 err = xa_alloc(&ldev->pfs, &idx, pf, XA_LIMIT(0, MLX5_MAX_PORTS - 1), 2211 GFP_KERNEL); 2212 if (err) { 2213 kfree(pf); 2214 return err; 2215 } 2216 2217 pf->idx = idx; 2218 pf->dev = dev; 2219 pf->group_id = group_id; 2220 dev->priv.lag = ldev; 2221 2222 if (group_id) 2223 return 0; 2224 2225 xa_set_mark(&ldev->pfs, idx, MLX5_LAG_XA_MARK_PORT); 2226 2227 MLX5_NB_INIT(&pf->port_change_nb, 2228 mlx5_lag_mpesw_port_change_event, PORT_CHANGE); 2229 mlx5_eq_notifier_register(dev, &pf->port_change_nb); 2230 2231 return 0; 2232 } 2233 2234 void mlx5_ldev_remove_mdev(struct mlx5_lag *ldev, 2235 struct mlx5_core_dev *dev) 2236 { 2237 struct lag_func *pf; 2238 int i; 2239 2240 mlx5_lag_for_each(i, 0, ldev, MLX5_LAG_FILTER_ALL) { 2241 pf = mlx5_lag_pf(ldev, i); 2242 if (pf->dev == dev) 2243 break; 2244 } 2245 if (i >= MLX5_MAX_PORTS) 2246 return; 2247 2248 if (pf->port_change_nb.nb.notifier_call) 2249 mlx5_eq_notifier_unregister(dev, &pf->port_change_nb); 2250 2251 pf->dev = NULL; 2252 dev->priv.lag = NULL; 2253 xa_erase(&ldev->pfs, pf->idx); 2254 kfree(pf); 2255 } 2256 2257 /* Must be called with HCA devcom component lock held */ 2258 static int __mlx5_lag_dev_add_mdev(struct mlx5_core_dev *dev) 2259 { 2260 struct mlx5_devcom_comp_dev *pos = NULL; 2261 struct mlx5_lag *ldev = NULL; 2262 struct mlx5_core_dev *tmp_dev; 2263 int err; 2264 2265 tmp_dev = mlx5_devcom_get_next_peer_data(dev->priv.hca_devcom_comp, &pos); 2266 if (tmp_dev) 2267 ldev = mlx5_lag_dev(tmp_dev); 2268 2269 if (!ldev) { 2270 ldev = mlx5_lag_dev_alloc(dev); 2271 if (!ldev) { 2272 mlx5_core_err(dev, "Failed to alloc lag dev\n"); 2273 return 0; 2274 } 2275 err = mlx5_ldev_add_mdev(ldev, dev, 0); 2276 if (err) { 2277 mlx5_core_err(dev, "Failed to add mdev to lag dev\n"); 2278 mlx5_ldev_put(ldev); 2279 return 0; 2280 } 2281 return 0; 2282 } 2283 2284 mutex_lock(&ldev->lock); 2285 if (ldev->mode_changes_in_progress) { 2286 mutex_unlock(&ldev->lock); 2287 return -EAGAIN; 2288 } 2289 mlx5_ldev_get(ldev); 2290 err = mlx5_ldev_add_mdev(ldev, dev, 0); 2291 if (err) { 2292 mlx5_ldev_put(ldev); 2293 mutex_unlock(&ldev->lock); 2294 return err; 2295 } 2296 mutex_unlock(&ldev->lock); 2297 2298 return 0; 2299 } 2300 2301 static void mlx5_lag_unregister_hca_devcom_comp(struct mlx5_core_dev *dev) 2302 { 2303 mlx5_devcom_unregister_component(dev->priv.hca_devcom_comp); 2304 dev->priv.hca_devcom_comp = NULL; 2305 } 2306 2307 static int mlx5_lag_register_hca_devcom_comp(struct mlx5_core_dev *dev) 2308 { 2309 struct mlx5_devcom_match_attr attr = { 2310 .flags = MLX5_DEVCOM_MATCH_FLAGS_NS, 2311 .net = mlx5_core_net(dev), 2312 }; 2313 u8 len __always_unused; 2314 2315 mlx5_query_nic_sw_system_image_guid(dev, attr.key.buf, &len); 2316 2317 /* This component is use to sync adding core_dev to lag_dev and to sync 2318 * changes of mlx5_adev_devices between LAG layer and other layers. 2319 */ 2320 dev->priv.hca_devcom_comp = 2321 mlx5_devcom_register_component(dev->priv.devc, 2322 MLX5_DEVCOM_HCA_PORTS, 2323 &attr, mlx5_lag_devcom_event, 2324 dev); 2325 if (!dev->priv.hca_devcom_comp) { 2326 mlx5_core_err(dev, 2327 "Failed to register devcom HCA component."); 2328 return -EINVAL; 2329 } 2330 2331 return 0; 2332 } 2333 2334 void mlx5_lag_remove_mdev(struct mlx5_core_dev *dev) 2335 { 2336 struct mlx5_lag *ldev; 2337 2338 ldev = mlx5_lag_dev(dev); 2339 if (!ldev) 2340 return; 2341 2342 /* mdev is being removed, might as well remove debugfs 2343 * as early as possible. 2344 */ 2345 mlx5_ldev_remove_debugfs(dev->priv.dbg.lag_debugfs); 2346 recheck: 2347 mutex_lock(&ldev->lock); 2348 if (ldev->mode_changes_in_progress) { 2349 mutex_unlock(&ldev->lock); 2350 msleep(100); 2351 goto recheck; 2352 } 2353 mlx5_ldev_remove_mdev(ldev, dev); 2354 mutex_unlock(&ldev->lock); 2355 /* Send devcom event to notify peers that a device is being removed */ 2356 mlx5_devcom_send_event(dev->priv.hca_devcom_comp, 2357 LAG_DEVCOM_UNPAIR, LAG_DEVCOM_UNPAIR, dev); 2358 mlx5_lag_unregister_hca_devcom_comp(dev); 2359 mlx5_ldev_put(ldev); 2360 } 2361 2362 void mlx5_lag_add_mdev(struct mlx5_core_dev *dev) 2363 { 2364 int err; 2365 2366 if (!mlx5_lag_is_supported(dev)) 2367 return; 2368 2369 if (mlx5_lag_register_hca_devcom_comp(dev)) 2370 return; 2371 2372 recheck: 2373 mlx5_devcom_comp_lock(dev->priv.hca_devcom_comp); 2374 err = __mlx5_lag_dev_add_mdev(dev); 2375 mlx5_devcom_comp_unlock(dev->priv.hca_devcom_comp); 2376 2377 if (err) { 2378 msleep(100); 2379 goto recheck; 2380 } 2381 /* Send devcom event to notify peers that a device was added */ 2382 mlx5_devcom_send_event(dev->priv.hca_devcom_comp, 2383 LAG_DEVCOM_PAIR, LAG_DEVCOM_UNPAIR, dev); 2384 mlx5_ldev_add_debugfs(dev); 2385 } 2386 2387 void mlx5_lag_remove_netdev(struct mlx5_core_dev *dev, 2388 struct net_device *netdev) 2389 { 2390 struct mlx5_lag *ldev; 2391 bool lag_is_active; 2392 2393 ldev = mlx5_lag_dev(dev); 2394 if (!ldev) 2395 return; 2396 2397 mutex_lock(&ldev->lock); 2398 mlx5_ldev_remove_netdev(ldev, netdev); 2399 clear_bit(MLX5_LAG_FLAG_NDEVS_READY, &ldev->state_flags); 2400 2401 lag_is_active = __mlx5_lag_is_active(ldev); 2402 mutex_unlock(&ldev->lock); 2403 2404 if (lag_is_active) 2405 mlx5_queue_bond_work(ldev, 0); 2406 } 2407 2408 void mlx5_lag_add_netdev(struct mlx5_core_dev *dev, 2409 struct net_device *netdev) 2410 { 2411 struct mlx5_lag *ldev; 2412 int num = 0; 2413 2414 ldev = mlx5_lag_dev(dev); 2415 if (!ldev) 2416 return; 2417 2418 mutex_lock(&ldev->lock); 2419 mlx5_ldev_add_netdev(ldev, dev, netdev); 2420 num = mlx5_lag_num_netdevs(ldev); 2421 if (num >= ldev->ports) 2422 set_bit(MLX5_LAG_FLAG_NDEVS_READY, &ldev->state_flags); 2423 mutex_unlock(&ldev->lock); 2424 mlx5_queue_bond_work(ldev, 0); 2425 } 2426 2427 int mlx5_get_pre_lag_func(struct mlx5_lag *ldev, int start_idx, int end_idx, 2428 u32 filter) 2429 { 2430 struct lag_func *pf; 2431 int i; 2432 2433 for (i = start_idx; i >= end_idx; i--) { 2434 pf = xa_load(&ldev->pfs, i); 2435 if (!pf || !pf->dev) 2436 continue; 2437 if (filter == MLX5_LAG_FILTER_PORTS) { 2438 if (xa_get_mark(&ldev->pfs, i, MLX5_LAG_XA_MARK_PORT)) 2439 return i; 2440 } else if (filter == MLX5_LAG_FILTER_ALL || 2441 filter == pf->group_id) { 2442 return i; 2443 } 2444 } 2445 return -1; 2446 } 2447 2448 int mlx5_get_next_lag_func(struct mlx5_lag *ldev, int start_idx, u32 filter) 2449 { 2450 struct lag_func *pf; 2451 unsigned long idx; 2452 2453 if (filter == MLX5_LAG_FILTER_PORTS) { 2454 xa_for_each_marked_start(&ldev->pfs, idx, pf, 2455 MLX5_LAG_XA_MARK_PORT, start_idx) 2456 if (pf->dev) 2457 return idx; 2458 return MLX5_MAX_PORTS; 2459 } 2460 2461 xa_for_each_start(&ldev->pfs, idx, pf, start_idx) { 2462 if (!pf->dev) 2463 continue; 2464 if (filter == MLX5_LAG_FILTER_ALL || 2465 filter == pf->group_id) 2466 return idx; 2467 } 2468 return MLX5_MAX_PORTS; 2469 } 2470 2471 bool mlx5_lag_is_roce(struct mlx5_core_dev *dev) 2472 { 2473 struct mlx5_lag *ldev; 2474 unsigned long flags; 2475 bool res; 2476 2477 spin_lock_irqsave(&lag_lock, flags); 2478 ldev = mlx5_lag_dev(dev); 2479 res = ldev && __mlx5_lag_is_roce(ldev); 2480 spin_unlock_irqrestore(&lag_lock, flags); 2481 2482 return res; 2483 } 2484 EXPORT_SYMBOL(mlx5_lag_is_roce); 2485 2486 bool mlx5_lag_is_active(struct mlx5_core_dev *dev) 2487 { 2488 struct mlx5_lag *ldev; 2489 unsigned long flags; 2490 bool res; 2491 2492 spin_lock_irqsave(&lag_lock, flags); 2493 ldev = mlx5_lag_dev(dev); 2494 res = ldev && (__mlx5_lag_is_active(ldev) || 2495 __mlx5_lag_is_sd_active(ldev, dev)); 2496 spin_unlock_irqrestore(&lag_lock, flags); 2497 2498 return res; 2499 } 2500 EXPORT_SYMBOL(mlx5_lag_is_active); 2501 2502 bool mlx5_lag_mode_is_hash(struct mlx5_core_dev *dev) 2503 { 2504 struct mlx5_lag *ldev; 2505 unsigned long flags; 2506 bool res = 0; 2507 2508 spin_lock_irqsave(&lag_lock, flags); 2509 ldev = mlx5_lag_dev(dev); 2510 if (ldev) 2511 res = test_bit(MLX5_LAG_MODE_FLAG_HASH_BASED, &ldev->mode_flags); 2512 spin_unlock_irqrestore(&lag_lock, flags); 2513 2514 return res; 2515 } 2516 EXPORT_SYMBOL(mlx5_lag_mode_is_hash); 2517 2518 bool mlx5_lag_is_master(struct mlx5_core_dev *dev) 2519 { 2520 struct mlx5_lag *ldev; 2521 unsigned long flags; 2522 struct lag_func *pf; 2523 bool res = false; 2524 int idx; 2525 2526 spin_lock_irqsave(&lag_lock, flags); 2527 ldev = mlx5_lag_dev(dev); 2528 if (ldev) { 2529 u32 filter; 2530 2531 filter = mlx5_lag_get_filter(ldev, dev); 2532 idx = mlx5_lag_get_dev_index_by_seq_filter(ldev, MLX5_LAG_P1, 2533 filter); 2534 if ((__mlx5_lag_is_active(ldev) || 2535 __mlx5_lag_is_sd_active(ldev, dev)) && idx >= 0) { 2536 pf = mlx5_lag_pf(ldev, idx); 2537 res = pf && dev == pf->dev; 2538 } 2539 } 2540 spin_unlock_irqrestore(&lag_lock, flags); 2541 2542 return res; 2543 } 2544 EXPORT_SYMBOL(mlx5_lag_is_master); 2545 2546 bool mlx5_lag_is_sriov(struct mlx5_core_dev *dev) 2547 { 2548 struct mlx5_lag *ldev; 2549 unsigned long flags; 2550 bool res; 2551 2552 spin_lock_irqsave(&lag_lock, flags); 2553 ldev = mlx5_lag_dev(dev); 2554 res = ldev && __mlx5_lag_is_sriov(ldev); 2555 spin_unlock_irqrestore(&lag_lock, flags); 2556 2557 return res; 2558 } 2559 EXPORT_SYMBOL(mlx5_lag_is_sriov); 2560 2561 bool mlx5_lag_is_sd(struct mlx5_core_dev *dev) 2562 { 2563 struct mlx5_lag *ldev; 2564 unsigned long flags; 2565 bool res; 2566 2567 spin_lock_irqsave(&lag_lock, flags); 2568 ldev = mlx5_lag_dev(dev); 2569 res = ldev && __mlx5_lag_is_sd(ldev, dev); 2570 spin_unlock_irqrestore(&lag_lock, flags); 2571 2572 return res; 2573 } 2574 2575 bool mlx5_lag_is_shared_fdb(struct mlx5_core_dev *dev) 2576 { 2577 struct mlx5_lag *ldev; 2578 unsigned long flags; 2579 bool res = false; 2580 2581 spin_lock_irqsave(&lag_lock, flags); 2582 ldev = mlx5_lag_dev(dev); 2583 if (ldev) { 2584 res = test_bit(MLX5_LAG_MODE_FLAG_SHARED_FDB, 2585 &ldev->mode_flags); 2586 if (__mlx5_lag_is_sd(ldev, dev) && !__mlx5_lag_is_active(ldev)) 2587 res = __mlx5_lag_is_sd_active(ldev, dev); 2588 } 2589 spin_unlock_irqrestore(&lag_lock, flags); 2590 2591 return res; 2592 } 2593 EXPORT_SYMBOL(mlx5_lag_is_shared_fdb); 2594 2595 void mlx5_lag_disable_change(struct mlx5_core_dev *dev) 2596 { 2597 struct mlx5_devcom_comp_dev *sd_devcom = mlx5_sd_get_devcom(dev); 2598 struct mlx5_core_dev *primary = dev; 2599 struct mlx5_lag *ldev; 2600 struct lag_func *pf; 2601 bool mpesw; 2602 int i; 2603 2604 ldev = mlx5_lag_dev(dev); 2605 if (!ldev) 2606 return; 2607 2608 if (sd_devcom) { 2609 mlx5_devcom_comp_lock(sd_devcom); 2610 primary = mlx5_sd_get_primary(dev) ?: dev; 2611 mlx5_devcom_comp_unlock(sd_devcom); 2612 } 2613 mlx5_devcom_comp_lock(primary->priv.hca_devcom_comp); 2614 mpesw = ldev->mode == MLX5_LAG_MODE_MPESW; 2615 if (mpesw) 2616 mlx5_mpesw_sd_devcoms_lock(ldev); 2617 mutex_lock(&ldev->lock); 2618 2619 ldev->mode_changes_in_progress++; 2620 if (__mlx5_lag_is_active(ldev)) { 2621 if (ldev->mode == MLX5_LAG_MODE_MPESW) 2622 mlx5_lag_disable_mpesw(ldev); 2623 else 2624 mlx5_disable_lag(ldev); 2625 } 2626 2627 mutex_unlock(&ldev->lock); 2628 if (mpesw) 2629 mlx5_mpesw_sd_devcoms_unlock(ldev); 2630 mlx5_devcom_comp_unlock(primary->priv.hca_devcom_comp); 2631 2632 if (!sd_devcom) 2633 return; 2634 2635 /* Teardown SD shared FDB for this device's group if active */ 2636 mlx5_devcom_comp_lock(sd_devcom); 2637 mutex_lock(&ldev->lock); 2638 mlx5_lag_for_each(i, 0, ldev, MLX5_LAG_FILTER_ALL) { 2639 pf = mlx5_lag_pf(ldev, i); 2640 if (pf->dev == dev && pf->sd_fdb_active) { 2641 mlx5_lag_shared_fdb_destroy(ldev, pf->group_id); 2642 break; 2643 } 2644 } 2645 mutex_unlock(&ldev->lock); 2646 mlx5_devcom_comp_unlock(sd_devcom); 2647 } 2648 2649 void mlx5_lag_enable_change(struct mlx5_core_dev *dev) 2650 { 2651 struct mlx5_lag *ldev; 2652 2653 ldev = mlx5_lag_dev(dev); 2654 if (!ldev) 2655 return; 2656 2657 mutex_lock(&ldev->lock); 2658 ldev->mode_changes_in_progress--; 2659 mutex_unlock(&ldev->lock); 2660 mlx5_queue_bond_work(ldev, 0); 2661 } 2662 2663 u8 mlx5_lag_get_slave_port(struct mlx5_core_dev *dev, 2664 struct net_device *slave) 2665 { 2666 struct mlx5_lag *ldev; 2667 unsigned long flags; 2668 struct lag_func *pf; 2669 u8 port = 0; 2670 int i; 2671 2672 spin_lock_irqsave(&lag_lock, flags); 2673 ldev = mlx5_lag_dev(dev); 2674 if (!(ldev && __mlx5_lag_is_roce(ldev))) 2675 goto unlock; 2676 2677 mlx5_ldev_for_each(i, 0, ldev) { 2678 pf = mlx5_lag_pf(ldev, i); 2679 if (pf->netdev == slave) { 2680 port = i; 2681 break; 2682 } 2683 } 2684 2685 port = ldev->v2p_map[port * ldev->buckets]; 2686 2687 unlock: 2688 spin_unlock_irqrestore(&lag_lock, flags); 2689 return port; 2690 } 2691 EXPORT_SYMBOL(mlx5_lag_get_slave_port); 2692 2693 u8 mlx5_lag_get_num_ports(struct mlx5_core_dev *dev) 2694 { 2695 struct mlx5_lag *ldev; 2696 2697 ldev = mlx5_lag_dev(dev); 2698 if (!ldev) 2699 return 0; 2700 2701 return ldev->ports; 2702 } 2703 EXPORT_SYMBOL(mlx5_lag_get_num_ports); 2704 2705 struct mlx5_core_dev *mlx5_lag_get_next_peer_mdev(struct mlx5_core_dev *dev, int *i) 2706 { 2707 struct mlx5_core_dev *peer_dev = NULL; 2708 struct mlx5_lag *ldev; 2709 unsigned long flags; 2710 struct lag_func *pf; 2711 int idx; 2712 2713 spin_lock_irqsave(&lag_lock, flags); 2714 ldev = mlx5_lag_dev(dev); 2715 if (!ldev) 2716 goto unlock; 2717 2718 if (*i == MLX5_MAX_PORTS) 2719 goto unlock; 2720 mlx5_lag_for_each(idx, *i, ldev, mlx5_lag_get_filter(ldev, dev)) { 2721 pf = mlx5_lag_pf(ldev, idx); 2722 if (pf->dev != dev) 2723 break; 2724 } 2725 2726 if (idx == MLX5_MAX_PORTS) { 2727 *i = idx; 2728 goto unlock; 2729 } 2730 *i = idx + 1; 2731 2732 pf = mlx5_lag_pf(ldev, idx); 2733 peer_dev = pf->dev; 2734 2735 unlock: 2736 spin_unlock_irqrestore(&lag_lock, flags); 2737 return peer_dev; 2738 } 2739 EXPORT_SYMBOL(mlx5_lag_get_next_peer_mdev); 2740 2741 int mlx5_lag_query_cong_counters(struct mlx5_core_dev *dev, 2742 u64 *values, 2743 int num_counters, 2744 size_t *offsets) 2745 { 2746 int outlen = MLX5_ST_SZ_BYTES(query_cong_statistics_out); 2747 struct mlx5_core_dev **mdev; 2748 int ret = 0, i, j, idx = 0; 2749 struct mlx5_lag *ldev; 2750 unsigned long flags; 2751 struct lag_func *pf; 2752 int num_ports; 2753 void *out; 2754 2755 out = kvzalloc(outlen, GFP_KERNEL); 2756 if (!out) 2757 return -ENOMEM; 2758 2759 mdev = kvzalloc(sizeof(mdev[0]) * MLX5_MAX_PORTS, GFP_KERNEL); 2760 if (!mdev) { 2761 ret = -ENOMEM; 2762 goto free_out; 2763 } 2764 2765 memset(values, 0, sizeof(*values) * num_counters); 2766 2767 spin_lock_irqsave(&lag_lock, flags); 2768 ldev = mlx5_lag_dev(dev); 2769 if (ldev && __mlx5_lag_is_active(ldev)) { 2770 num_ports = ldev->ports; 2771 mlx5_ldev_for_each(i, 0, ldev) { 2772 pf = mlx5_lag_pf(ldev, i); 2773 mdev[idx++] = pf->dev; 2774 } 2775 } else { 2776 num_ports = 1; 2777 mdev[MLX5_LAG_P1] = dev; 2778 } 2779 spin_unlock_irqrestore(&lag_lock, flags); 2780 2781 for (i = 0; i < num_ports; ++i) { 2782 u32 in[MLX5_ST_SZ_DW(query_cong_statistics_in)] = {}; 2783 2784 MLX5_SET(query_cong_statistics_in, in, opcode, 2785 MLX5_CMD_OP_QUERY_CONG_STATISTICS); 2786 ret = mlx5_cmd_exec_inout(mdev[i], query_cong_statistics, in, 2787 out); 2788 if (ret) 2789 goto free_mdev; 2790 2791 for (j = 0; j < num_counters; ++j) 2792 values[j] += be64_to_cpup((__be64 *)(out + offsets[j])); 2793 } 2794 2795 free_mdev: 2796 kvfree(mdev); 2797 free_out: 2798 kvfree(out); 2799 return ret; 2800 } 2801 EXPORT_SYMBOL(mlx5_lag_query_cong_counters); 2802