1 // SPDX-License-Identifier: GPL-2.0-only 2 /* 3 * Resource Director Technology(RDT) 4 * - Monitoring code 5 * 6 * Copyright (C) 2017 Intel Corporation 7 * 8 * Author: 9 * Vikas Shivappa <vikas.shivappa@intel.com> 10 * 11 * This replaces the cqm.c based on perf but we reuse a lot of 12 * code and datastructures originally from Peter Zijlstra and Matt Fleming. 13 * 14 * More information about RDT be found in the Intel (R) x86 Architecture 15 * Software Developer Manual June 2016, volume 3, section 17.17. 16 */ 17 18 #define pr_fmt(fmt) "resctrl: " fmt 19 20 #include <linux/cpu.h> 21 #include <linux/resctrl.h> 22 23 #include <asm/cpu_device_id.h> 24 #include <asm/cpuid/api.h> 25 #include <asm/msr.h> 26 27 #include "internal.h" 28 29 /* 30 * Global boolean for rdt_monitor which is true if any 31 * resource monitoring is enabled. 32 */ 33 bool rdt_mon_capable; 34 35 #define CF(cf) ((unsigned long)(1048576 * (cf) + 0.5)) 36 37 static int snc_nodes_per_l3_cache = 1; 38 39 /* 40 * The correction factor table is documented in Documentation/filesystems/resctrl.rst. 41 * If rmid > rmid threshold, MBM total and local values should be multiplied 42 * by the correction factor. 43 * 44 * The original table is modified for better code: 45 * 46 * 1. The threshold 0 is changed to rmid count - 1 so don't do correction 47 * for the case. 48 * 2. MBM total and local correction table indexed by core counter which is 49 * equal to (x86_cache_max_rmid + 1) / 8 - 1 and is from 0 up to 27. 50 * 3. The correction factor is normalized to 2^20 (1048576) so it's faster 51 * to calculate corrected value by shifting: 52 * corrected_value = (original_value * correction_factor) >> 20 53 */ 54 static const struct mbm_correction_factor_table { 55 u32 rmidthreshold; 56 u64 cf; 57 } mbm_cf_table[] __initconst = { 58 {7, CF(1.000000)}, 59 {15, CF(1.000000)}, 60 {15, CF(0.969650)}, 61 {31, CF(1.000000)}, 62 {31, CF(1.066667)}, 63 {31, CF(0.969650)}, 64 {47, CF(1.142857)}, 65 {63, CF(1.000000)}, 66 {63, CF(1.185115)}, 67 {63, CF(1.066553)}, 68 {79, CF(1.454545)}, 69 {95, CF(1.000000)}, 70 {95, CF(1.230769)}, 71 {95, CF(1.142857)}, 72 {95, CF(1.066667)}, 73 {127, CF(1.000000)}, 74 {127, CF(1.254863)}, 75 {127, CF(1.185255)}, 76 {151, CF(1.000000)}, 77 {127, CF(1.066667)}, 78 {167, CF(1.000000)}, 79 {159, CF(1.454334)}, 80 {183, CF(1.000000)}, 81 {127, CF(0.969744)}, 82 {191, CF(1.280246)}, 83 {191, CF(1.230921)}, 84 {215, CF(1.000000)}, 85 {191, CF(1.143118)}, 86 }; 87 88 static u32 mbm_cf_rmidthreshold __read_mostly = UINT_MAX; 89 90 static u64 mbm_cf __read_mostly; 91 92 static inline u64 get_corrected_mbm_count(u32 rmid, unsigned long val) 93 { 94 /* Correct MBM value. */ 95 if (rmid > mbm_cf_rmidthreshold) 96 val = (val * mbm_cf) >> 20; 97 98 return val; 99 } 100 101 /* 102 * When Sub-NUMA Cluster (SNC) mode is not enabled (as indicated by 103 * "snc_nodes_per_l3_cache == 1") no translation of the RMID value is 104 * needed. The physical RMID is the same as the logical RMID. 105 * 106 * On a platform with SNC mode enabled, Linux enables RMID sharing mode 107 * via MSR 0xCA0 (see the "RMID Sharing Mode" section in the "Intel 108 * Resource Director Technology Architecture Specification" for a full 109 * description of RMID sharing mode). 110 * 111 * In RMID sharing mode there are fewer "logical RMID" values available 112 * to accumulate data ("physical RMIDs" are divided evenly between SNC 113 * nodes that share an L3 cache). Linux creates an rdt_l3_mon_domain for 114 * each SNC node. 115 * 116 * The value loaded into IA32_PQR_ASSOC is the "logical RMID". 117 * 118 * Data is collected independently on each SNC node and can be retrieved 119 * using the "physical RMID" value computed by this function and loaded 120 * into IA32_QM_EVTSEL. @cpu can be any CPU in the SNC node. 121 * 122 * The scope of the IA32_QM_EVTSEL and IA32_QM_CTR MSRs is at the L3 123 * cache. So a "physical RMID" may be read from any CPU that shares 124 * the L3 cache with the desired SNC node, not just from a CPU in 125 * the specific SNC node. 126 */ 127 static int logical_rmid_to_physical_rmid(int cpu, int lrmid) 128 { 129 struct rdt_resource *r = &rdt_resources_all[RDT_RESOURCE_L3].r_resctrl; 130 131 if (snc_nodes_per_l3_cache == 1) 132 return lrmid; 133 134 return lrmid + (cpu_to_node(cpu) % snc_nodes_per_l3_cache) * r->mon.num_rmid; 135 } 136 137 static int __rmid_read_phys(u32 prmid, enum resctrl_event_id eventid, u64 *val) 138 { 139 u64 msr_val; 140 141 /* 142 * As per the SDM, when IA32_QM_EVTSEL.EvtID (bits 7:0) is configured 143 * with a valid event code for supported resource type and the bits 144 * IA32_QM_EVTSEL.RMID (bits 41:32) are configured with valid RMID, 145 * IA32_QM_CTR.data (bits 61:0) reports the monitored data. 146 * IA32_QM_CTR.Error (bit 63) and IA32_QM_CTR.Unavailable (bit 62) 147 * are error bits. 148 */ 149 wrmsr(MSR_IA32_QM_EVTSEL, eventid, prmid); 150 rdmsrq(MSR_IA32_QM_CTR, msr_val); 151 152 if (msr_val & RMID_VAL_ERROR) 153 return -EIO; 154 if (msr_val & RMID_VAL_UNAVAIL) 155 return -EINVAL; 156 157 *val = msr_val; 158 return 0; 159 } 160 161 static struct arch_mbm_state *get_arch_mbm_state(struct rdt_hw_l3_mon_domain *hw_dom, 162 u32 rmid, 163 enum resctrl_event_id eventid) 164 { 165 struct arch_mbm_state *state; 166 167 if (!resctrl_is_mbm_event(eventid)) 168 return NULL; 169 170 state = hw_dom->arch_mbm_states[MBM_STATE_IDX(eventid)]; 171 172 return state ? &state[rmid] : NULL; 173 } 174 175 void resctrl_arch_reset_rmid(struct rdt_resource *r, struct rdt_l3_mon_domain *d, 176 u32 unused, u32 rmid, 177 enum resctrl_event_id eventid) 178 { 179 struct rdt_hw_l3_mon_domain *hw_dom = resctrl_to_arch_mon_dom(d); 180 int cpu = cpumask_any(&d->hdr.cpu_mask); 181 struct arch_mbm_state *am; 182 u32 prmid; 183 184 am = get_arch_mbm_state(hw_dom, rmid, eventid); 185 if (am) { 186 memset(am, 0, sizeof(*am)); 187 188 prmid = logical_rmid_to_physical_rmid(cpu, rmid); 189 /* Record any initial, non-zero count value. */ 190 __rmid_read_phys(prmid, eventid, &am->prev_msr); 191 } 192 } 193 194 /* 195 * Assumes that hardware counters are also reset and thus that there is 196 * no need to record initial non-zero counts. 197 */ 198 void resctrl_arch_reset_rmid_all(struct rdt_resource *r, struct rdt_l3_mon_domain *d) 199 { 200 struct rdt_hw_l3_mon_domain *hw_dom = resctrl_to_arch_mon_dom(d); 201 enum resctrl_event_id eventid; 202 int idx; 203 204 for_each_mbm_event_id(eventid) { 205 if (!resctrl_is_mon_event_enabled(eventid)) 206 continue; 207 idx = MBM_STATE_IDX(eventid); 208 memset(hw_dom->arch_mbm_states[idx], 0, 209 sizeof(*hw_dom->arch_mbm_states[0]) * r->mon.num_rmid); 210 } 211 } 212 213 static u64 mbm_overflow_count(u64 prev_msr, u64 cur_msr, unsigned int width) 214 { 215 u64 shift = 64 - width, chunks; 216 217 chunks = (cur_msr << shift) - (prev_msr << shift); 218 return chunks >> shift; 219 } 220 221 static u64 get_corrected_val(struct rdt_resource *r, struct rdt_l3_mon_domain *d, 222 u32 rmid, enum resctrl_event_id eventid, u64 msr_val) 223 { 224 struct rdt_hw_l3_mon_domain *hw_dom = resctrl_to_arch_mon_dom(d); 225 struct rdt_hw_resource *hw_res = resctrl_to_arch_res(r); 226 struct arch_mbm_state *am; 227 u64 chunks; 228 229 am = get_arch_mbm_state(hw_dom, rmid, eventid); 230 if (am) { 231 am->chunks += mbm_overflow_count(am->prev_msr, msr_val, 232 hw_res->mbm_width); 233 chunks = get_corrected_mbm_count(rmid, am->chunks); 234 am->prev_msr = msr_val; 235 } else { 236 chunks = msr_val; 237 } 238 239 return chunks * hw_res->mon_scale; 240 } 241 242 int resctrl_arch_rmid_read(struct rdt_resource *r, struct rdt_domain_hdr *hdr, 243 u32 unused, u32 rmid, enum resctrl_event_id eventid, 244 void *arch_priv, u64 *val, void *ignored) 245 { 246 struct rdt_hw_l3_mon_domain *hw_dom; 247 struct rdt_l3_mon_domain *d; 248 struct arch_mbm_state *am; 249 u64 msr_val; 250 u32 prmid; 251 int cpu; 252 int ret; 253 254 resctrl_arch_rmid_read_context_check(); 255 256 if (r->rid == RDT_RESOURCE_PERF_PKG) 257 return intel_aet_read_event(hdr->id, rmid, arch_priv, val); 258 259 if (!domain_header_is_valid(hdr, RESCTRL_MON_DOMAIN, RDT_RESOURCE_L3)) 260 return -EINVAL; 261 262 if (cpumask_empty(&hdr->cpu_mask)) { 263 pr_warn_once("Domain %d has no CPUs\n", hdr->id); 264 return -EINVAL; 265 } 266 267 d = container_of(hdr, struct rdt_l3_mon_domain, hdr); 268 hw_dom = resctrl_to_arch_mon_dom(d); 269 cpu = cpumask_any(&hdr->cpu_mask); 270 prmid = logical_rmid_to_physical_rmid(cpu, rmid); 271 ret = __rmid_read_phys(prmid, eventid, &msr_val); 272 273 if (!ret) { 274 *val = get_corrected_val(r, d, rmid, eventid, msr_val); 275 } else if (ret == -EINVAL) { 276 am = get_arch_mbm_state(hw_dom, rmid, eventid); 277 if (am) 278 am->prev_msr = 0; 279 } 280 281 return ret; 282 } 283 284 static int __cntr_id_read(u32 cntr_id, u64 *val) 285 { 286 u64 msr_val; 287 288 /* 289 * QM_EVTSEL Register definition: 290 * ======================================================= 291 * Bits Mnemonic Description 292 * ======================================================= 293 * 63:44 -- Reserved 294 * 43:32 RMID RMID or counter ID in ABMC mode 295 * when reading an MBM event 296 * 31 ExtendedEvtID Extended Event Identifier 297 * 30:8 -- Reserved 298 * 7:0 EvtID Event Identifier 299 * ======================================================= 300 * The contents of a specific counter can be read by setting the 301 * following fields in QM_EVTSEL.ExtendedEvtID(=1) and 302 * QM_EVTSEL.EvtID = L3CacheABMC (=1) and setting QM_EVTSEL.RMID 303 * to the desired counter ID. Reading the QM_CTR then returns the 304 * contents of the specified counter. The RMID_VAL_ERROR bit is set 305 * if the counter configuration is invalid, or if an invalid counter 306 * ID is set in the QM_EVTSEL.RMID field. The RMID_VAL_UNAVAIL bit 307 * is set if the counter data is unavailable. 308 */ 309 wrmsr(MSR_IA32_QM_EVTSEL, ABMC_EXTENDED_EVT_ID | ABMC_EVT_ID, cntr_id); 310 rdmsrq(MSR_IA32_QM_CTR, msr_val); 311 312 if (msr_val & RMID_VAL_ERROR) 313 return -EIO; 314 if (msr_val & RMID_VAL_UNAVAIL) 315 return -EINVAL; 316 317 *val = msr_val; 318 return 0; 319 } 320 321 void resctrl_arch_reset_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d, 322 u32 unused, u32 rmid, int cntr_id, 323 enum resctrl_event_id eventid) 324 { 325 struct rdt_hw_l3_mon_domain *hw_dom = resctrl_to_arch_mon_dom(d); 326 struct arch_mbm_state *am; 327 328 am = get_arch_mbm_state(hw_dom, rmid, eventid); 329 if (am) { 330 memset(am, 0, sizeof(*am)); 331 332 /* Record any initial, non-zero count value. */ 333 __cntr_id_read(cntr_id, &am->prev_msr); 334 } 335 } 336 337 int resctrl_arch_cntr_read(struct rdt_resource *r, struct rdt_l3_mon_domain *d, 338 u32 unused, u32 rmid, int cntr_id, 339 enum resctrl_event_id eventid, u64 *val) 340 { 341 u64 msr_val; 342 int ret; 343 344 ret = __cntr_id_read(cntr_id, &msr_val); 345 if (ret) 346 return ret; 347 348 *val = get_corrected_val(r, d, rmid, eventid, msr_val); 349 350 return 0; 351 } 352 353 /* 354 * The power-on reset value of MSR_RMID_SNC_CONFIG is 0x1 355 * which indicates that RMIDs are configured in legacy mode. 356 * This mode is incompatible with Linux resctrl semantics 357 * as RMIDs are partitioned between SNC nodes, which requires 358 * a user to know which RMID is allocated to a task. 359 * Clearing bit 0 reconfigures the RMID counters for use 360 * in RMID sharing mode. This mode is better for Linux. 361 * The RMID space is divided between all SNC nodes with the 362 * RMIDs renumbered to start from zero in each node when 363 * counting operations from tasks. Code to read the counters 364 * must adjust RMID counter numbers based on SNC node. See 365 * logical_rmid_to_physical_rmid() for code that does this. 366 */ 367 void arch_mon_domain_online(struct rdt_resource *r, struct rdt_l3_mon_domain *d) 368 { 369 if (snc_nodes_per_l3_cache > 1) 370 msr_clear_bit(MSR_RMID_SNC_CONFIG, 0); 371 } 372 373 /* CPU models that support SNC and MSR_RMID_SNC_CONFIG */ 374 static const struct x86_cpu_id snc_cpu_ids[] __initconst = { 375 X86_MATCH_VFM(INTEL_ICELAKE_X, 0), 376 X86_MATCH_VFM(INTEL_SAPPHIRERAPIDS_X, 0), 377 X86_MATCH_VFM(INTEL_EMERALDRAPIDS_X, 0), 378 X86_MATCH_VFM(INTEL_GRANITERAPIDS_X, 0), 379 X86_MATCH_VFM(INTEL_ATOM_CRESTMONT_X, 0), 380 X86_MATCH_VFM(INTEL_ATOM_DARKMONT_X, 0), 381 {} 382 }; 383 384 static __init int snc_get_config(void) 385 { 386 int ret; 387 388 if (boot_cpu_data.x86_vendor != X86_VENDOR_INTEL) 389 return 1; 390 391 ret = topology_num_nodes_per_package(); 392 393 if (ret > 1 && !x86_match_cpu(snc_cpu_ids)) { 394 pr_warn("CoD enabled system? Resctrl not supported\n"); 395 return 1; 396 } 397 398 /* sanity check: Only valid results are 1, 2, 3, 4, 6 */ 399 switch (ret) { 400 case 1: 401 break; 402 case 2 ... 4: 403 case 6: 404 pr_info("Sub-NUMA Cluster mode detected with %d nodes per L3 cache\n", ret); 405 rdt_resources_all[RDT_RESOURCE_L3].r_resctrl.mon_scope = RESCTRL_L3_NODE; 406 break; 407 default: 408 pr_warn("Ignore improbable SNC node count %d\n", ret); 409 ret = 1; 410 break; 411 } 412 413 return ret; 414 } 415 416 int __init rdt_get_l3_mon_config(struct rdt_resource *r) 417 { 418 unsigned int mbm_offset = boot_cpu_data.x86_cache_mbm_width_offset; 419 struct rdt_hw_resource *hw_res = resctrl_to_arch_res(r); 420 unsigned int threshold; 421 u32 eax, ebx, ecx, edx; 422 423 snc_nodes_per_l3_cache = snc_get_config(); 424 425 resctrl_rmid_realloc_limit = boot_cpu_data.x86_cache_size * 1024; 426 hw_res->mon_scale = boot_cpu_data.x86_cache_occ_scale / snc_nodes_per_l3_cache; 427 r->mon.num_rmid = (boot_cpu_data.x86_cache_max_rmid + 1) / snc_nodes_per_l3_cache; 428 hw_res->mbm_width = MBM_CNTR_WIDTH_BASE; 429 430 if (mbm_offset > 0 && mbm_offset <= MBM_CNTR_WIDTH_OFFSET_MAX) 431 hw_res->mbm_width += mbm_offset; 432 else if (mbm_offset > MBM_CNTR_WIDTH_OFFSET_MAX) 433 pr_warn("Ignoring impossible MBM counter offset\n"); 434 435 /* 436 * A reasonable upper limit on the max threshold is the number 437 * of lines tagged per RMID if all RMIDs have the same number of 438 * lines tagged in the LLC. 439 * 440 * For a 35MB LLC and 56 RMIDs, this is ~1.8% of the LLC. 441 */ 442 threshold = resctrl_rmid_realloc_limit / r->mon.num_rmid; 443 444 /* 445 * Because num_rmid may not be a power of two, round the value 446 * to the nearest multiple of hw_res->mon_scale so it matches a 447 * value the hardware will measure. mon_scale may not be a power of 2. 448 */ 449 resctrl_rmid_realloc_threshold = resctrl_arch_round_mon_val(threshold); 450 451 if (rdt_cpu_has(X86_FEATURE_BMEC) || rdt_cpu_has(X86_FEATURE_ABMC)) { 452 /* Detect list of bandwidth sources that can be tracked */ 453 cpuid_count(0x80000020, 3, &eax, &ebx, &ecx, &edx); 454 r->mon.mbm_cfg_mask = ecx & MAX_EVT_CONFIG_BITS; 455 } 456 457 /* 458 * resctrl assumes a system that supports assignable counters can 459 * switch to "default" mode. Ensure that there is a "default" mode 460 * to switch to. This enforces a dependency between the independent 461 * X86_FEATURE_ABMC and X86_FEATURE_CQM_MBM_TOTAL/X86_FEATURE_CQM_MBM_LOCAL 462 * hardware features. 463 */ 464 if (rdt_cpu_has(X86_FEATURE_ABMC) && 465 (rdt_cpu_has(X86_FEATURE_CQM_MBM_TOTAL) || 466 rdt_cpu_has(X86_FEATURE_CQM_MBM_LOCAL))) { 467 r->mon.mbm_cntr_assignable = true; 468 r->mon.mbm_cntr_configurable = true; 469 cpuid_count(0x80000020, 5, &eax, &ebx, &ecx, &edx); 470 r->mon.num_mbm_cntrs = (ebx & GENMASK(15, 0)) + 1; 471 hw_res->mbm_cntr_assign_enabled = true; 472 } 473 474 r->mon_capable = true; 475 476 return 0; 477 } 478 479 void __init intel_rdt_mbm_apply_quirk(void) 480 { 481 int cf_index; 482 483 cf_index = (boot_cpu_data.x86_cache_max_rmid + 1) / 8 - 1; 484 if (cf_index >= ARRAY_SIZE(mbm_cf_table)) { 485 pr_info("No MBM correction factor available\n"); 486 return; 487 } 488 489 mbm_cf_rmidthreshold = mbm_cf_table[cf_index].rmidthreshold; 490 mbm_cf = mbm_cf_table[cf_index].cf; 491 } 492 493 static void resctrl_abmc_set_one_amd(void *arg) 494 { 495 bool *enable = arg; 496 497 if (*enable) 498 msr_set_bit(MSR_IA32_L3_QOS_EXT_CFG, ABMC_ENABLE_BIT); 499 else 500 msr_clear_bit(MSR_IA32_L3_QOS_EXT_CFG, ABMC_ENABLE_BIT); 501 } 502 503 /* 504 * ABMC enable/disable requires update of L3_QOS_EXT_CFG MSR on all the CPUs 505 * associated with all monitor domains. 506 */ 507 static void _resctrl_abmc_enable(struct rdt_resource *r, bool enable) 508 { 509 struct rdt_l3_mon_domain *d; 510 511 lockdep_assert_cpus_held(); 512 513 list_for_each_entry(d, &r->mon_domains, hdr.list) { 514 on_each_cpu_mask(&d->hdr.cpu_mask, resctrl_abmc_set_one_amd, 515 &enable, 1); 516 resctrl_arch_reset_rmid_all(r, d); 517 } 518 } 519 520 int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable) 521 { 522 struct rdt_hw_resource *hw_res = resctrl_to_arch_res(r); 523 524 if (r->mon.mbm_cntr_assignable && 525 hw_res->mbm_cntr_assign_enabled != enable) { 526 _resctrl_abmc_enable(r, enable); 527 hw_res->mbm_cntr_assign_enabled = enable; 528 } 529 530 return 0; 531 } 532 533 bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r) 534 { 535 return resctrl_to_arch_res(r)->mbm_cntr_assign_enabled; 536 } 537 538 static void resctrl_abmc_config_one_amd(void *info) 539 { 540 union l3_qos_abmc_cfg *abmc_cfg = info; 541 542 wrmsrq(MSR_IA32_L3_QOS_ABMC_CFG, abmc_cfg->full); 543 } 544 545 /* 546 * Send an IPI to the domain to assign the counter to RMID, event pair. 547 */ 548 void resctrl_arch_config_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d, 549 enum resctrl_event_id evtid, u32 rmid, u32 closid, 550 u32 cntr_id, bool assign) 551 { 552 struct rdt_hw_l3_mon_domain *hw_dom = resctrl_to_arch_mon_dom(d); 553 union l3_qos_abmc_cfg abmc_cfg = { 0 }; 554 struct arch_mbm_state *am; 555 556 abmc_cfg.split.cfg_en = 1; 557 abmc_cfg.split.cntr_en = assign ? 1 : 0; 558 abmc_cfg.split.cntr_id = cntr_id; 559 abmc_cfg.split.bw_src = rmid; 560 if (assign) 561 abmc_cfg.split.bw_type = resctrl_get_mon_evt_cfg(evtid); 562 563 smp_call_function_any(&d->hdr.cpu_mask, resctrl_abmc_config_one_amd, &abmc_cfg, 1); 564 565 /* 566 * The hardware counter is reset (because cfg_en == 1) so there is no 567 * need to record initial non-zero counts. 568 */ 569 am = get_arch_mbm_state(hw_dom, rmid, evtid); 570 if (am) 571 memset(am, 0, sizeof(*am)); 572 } 573 574 void resctrl_arch_mbm_cntr_assign_set_one(struct rdt_resource *r) 575 { 576 struct rdt_hw_resource *hw_res = resctrl_to_arch_res(r); 577 578 resctrl_abmc_set_one_amd(&hw_res->mbm_cntr_assign_enabled); 579 } 580