xref: /linux/arch/x86/events/intel/ds.c (revision 0ac5d6ca2c3b7d047963b67dccef17902dc6c017)
1 // SPDX-License-Identifier: GPL-2.0
2 #include <linux/bitops.h>
3 #include <linux/types.h>
4 #include <linux/slab.h>
5 #include <linux/sched/clock.h>
6 
7 #include <asm/cpu_entry_area.h>
8 #include <asm/debugreg.h>
9 #include <asm/perf_event.h>
10 #include <asm/tlbflush.h>
11 #include <asm/insn.h>
12 #include <asm/io.h>
13 #include <asm/msr.h>
14 #include <asm/timer.h>
15 
16 #include "../perf_event.h"
17 
18 /* Waste a full page so it can be mapped into the cpu_entry_area */
19 DEFINE_PER_CPU_PAGE_ALIGNED(struct debug_store, cpu_debug_store);
20 
21 /* The size of a BTS record in bytes: */
22 #define BTS_RECORD_SIZE		24
23 
24 #define PEBS_FIXUP_SIZE		PAGE_SIZE
25 
26 /*
27  * pebs_record_32 for p4 and core not supported
28 
29 struct pebs_record_32 {
30 	u32 flags, ip;
31 	u32 ax, bc, cx, dx;
32 	u32 si, di, bp, sp;
33 };
34 
35  */
36 
37 union omr_encoding {
38 	struct {
39 		u8 omr_source : 4;
40 		u8 omr_remote : 1;
41 		u8 omr_hitm : 1;
42 		u8 omr_snoop : 1;
43 		u8 omr_promoted : 1;
44 	};
45 	u8 omr_full;
46 };
47 
48 union intel_x86_pebs_dse {
49 	u64 val;
50 	struct {
51 		unsigned int ld_dse:4;
52 		unsigned int ld_stlb_miss:1;
53 		unsigned int ld_locked:1;
54 		unsigned int ld_data_blk:1;
55 		unsigned int ld_addr_blk:1;
56 		unsigned int ld_reserved:24;
57 	};
58 	struct {
59 		unsigned int st_l1d_hit:1;
60 		unsigned int st_reserved1:3;
61 		unsigned int st_stlb_miss:1;
62 		unsigned int st_locked:1;
63 		unsigned int st_reserved2:26;
64 	};
65 	struct {
66 		unsigned int st_lat_dse:4;
67 		unsigned int st_lat_stlb_miss:1;
68 		unsigned int st_lat_locked:1;
69 		unsigned int ld_reserved3:26;
70 	};
71 	struct {
72 		unsigned int mtl_dse:5;
73 		unsigned int mtl_locked:1;
74 		unsigned int mtl_stlb_miss:1;
75 		unsigned int mtl_fwd_blk:1;
76 		unsigned int ld_reserved4:24;
77 	};
78 	struct {
79 		unsigned int lnc_dse:8;
80 		unsigned int ld_reserved5:2;
81 		unsigned int lnc_stlb_miss:1;
82 		unsigned int lnc_locked:1;
83 		unsigned int lnc_data_blk:1;
84 		unsigned int lnc_addr_blk:1;
85 		unsigned int ld_reserved6:18;
86 	};
87 	struct {
88 		unsigned int pnc_dse: 8;
89 		unsigned int pnc_l2_miss:1;
90 		unsigned int pnc_stlb_clean_hit:1;
91 		unsigned int pnc_stlb_any_hit:1;
92 		unsigned int pnc_stlb_miss:1;
93 		unsigned int pnc_locked:1;
94 		unsigned int pnc_data_blk:1;
95 		unsigned int pnc_addr_blk:1;
96 		unsigned int pnc_fb_full:1;
97 		unsigned int ld_reserved8:16;
98 	};
99 	struct {
100 		unsigned int arw_dse:8;
101 		unsigned int arw_l2_miss:1;
102 		unsigned int arw_xq_promotion:1;
103 		unsigned int arw_reissue:1;
104 		unsigned int arw_stlb_miss:1;
105 		unsigned int arw_locked:1;
106 		unsigned int arw_data_blk:1;
107 		unsigned int arw_addr_blk:1;
108 		unsigned int arw_fb_full:1;
109 		unsigned int ld_reserved9:16;
110 	};
111 };
112 
113 
114 /*
115  * Map PEBS Load Latency Data Source encodings to generic
116  * memory data source information
117  */
118 #define P(a, b) PERF_MEM_S(a, b)
119 #define OP_LH (P(OP, LOAD) | P(LVL, HIT))
120 #define LEVEL(x) P(LVLNUM, x)
121 #define REM P(REMOTE, REMOTE)
122 #define SNOOP_NONE_MISS (P(SNOOP, NONE) | P(SNOOP, MISS))
123 
124 /* Version for Sandy Bridge and later */
125 static u64 pebs_data_source[PERF_PEBS_DATA_SOURCE_MAX] = {
126 	P(OP, LOAD) | P(LVL, MISS) | LEVEL(L3) | P(SNOOP, NA),/* 0x00:ukn L3 */
127 	OP_LH | P(LVL, L1)  | LEVEL(L1) | P(SNOOP, NONE),  /* 0x01: L1 local */
128 	OP_LH | P(LVL, LFB) | LEVEL(LFB) | P(SNOOP, NONE), /* 0x02: LFB hit */
129 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, NONE),  /* 0x03: L2 hit */
130 	OP_LH | P(LVL, L3)  | LEVEL(L3) | P(SNOOP, NONE),  /* 0x04: L3 hit */
131 	OP_LH | P(LVL, L3)  | LEVEL(L3) | P(SNOOP, MISS),  /* 0x05: L3 hit, snoop miss */
132 	OP_LH | P(LVL, L3)  | LEVEL(L3) | P(SNOOP, HIT),   /* 0x06: L3 hit, snoop hit */
133 	OP_LH | P(LVL, L3)  | LEVEL(L3) | P(SNOOP, HITM),  /* 0x07: L3 hit, snoop hitm */
134 	OP_LH | P(LVL, REM_CCE1) | REM | LEVEL(L3) | P(SNOOP, HIT),  /* 0x08: L3 miss snoop hit */
135 	OP_LH | P(LVL, REM_CCE1) | REM | LEVEL(L3) | P(SNOOP, HITM), /* 0x09: L3 miss snoop hitm*/
136 	OP_LH | P(LVL, LOC_RAM)  | LEVEL(RAM) | P(SNOOP, HIT),       /* 0x0a: L3 miss, shared */
137 	OP_LH | P(LVL, REM_RAM1) | REM | LEVEL(L3) | P(SNOOP, HIT),  /* 0x0b: L3 miss, shared */
138 	OP_LH | P(LVL, LOC_RAM)  | LEVEL(RAM) | SNOOP_NONE_MISS,     /* 0x0c: L3 miss, excl */
139 	OP_LH | P(LVL, REM_RAM1) | LEVEL(RAM) | REM | SNOOP_NONE_MISS, /* 0x0d: L3 miss, excl */
140 	OP_LH | P(LVL, IO)  | LEVEL(NA) | P(SNOOP, NONE), /* 0x0e: I/O */
141 	OP_LH | P(LVL, UNC) | LEVEL(NA) | P(SNOOP, NONE), /* 0x0f: uncached */
142 };
143 
144 /* Patch up minor differences in the bits */
145 void __init intel_pmu_pebs_data_source_nhm(void)
146 {
147 	pebs_data_source[0x05] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOP, HIT);
148 	pebs_data_source[0x06] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOP, HITM);
149 	pebs_data_source[0x07] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOP, HITM);
150 }
151 
152 static void __init __intel_pmu_pebs_data_source_skl(bool pmem, u64 *data_source)
153 {
154 	u64 pmem_or_l4 = pmem ? LEVEL(PMEM) : LEVEL(L4);
155 
156 	data_source[0x08] = OP_LH | pmem_or_l4 | P(SNOOP, HIT);
157 	data_source[0x09] = OP_LH | pmem_or_l4 | REM | P(SNOOP, HIT);
158 	data_source[0x0b] = OP_LH | LEVEL(RAM) | REM | P(SNOOP, NONE);
159 	data_source[0x0c] = OP_LH | LEVEL(ANY_CACHE) | REM | P(SNOOPX, FWD);
160 	data_source[0x0d] = OP_LH | LEVEL(ANY_CACHE) | REM | P(SNOOP, HITM);
161 }
162 
163 void __init intel_pmu_pebs_data_source_skl(bool pmem)
164 {
165 	__intel_pmu_pebs_data_source_skl(pmem, pebs_data_source);
166 }
167 
168 static void __init __intel_pmu_pebs_data_source_grt(u64 *data_source)
169 {
170 	data_source[0x05] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOP, HIT);
171 	data_source[0x06] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOP, HITM);
172 	data_source[0x08] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOPX, FWD);
173 }
174 
175 void __init intel_pmu_pebs_data_source_grt(void)
176 {
177 	__intel_pmu_pebs_data_source_grt(pebs_data_source);
178 }
179 
180 void __init intel_pmu_pebs_data_source_adl(void)
181 {
182 	u64 *data_source;
183 
184 	data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_CORE_IDX].pebs_data_source;
185 	memcpy(data_source, pebs_data_source, sizeof(pebs_data_source));
186 	__intel_pmu_pebs_data_source_skl(false, data_source);
187 
188 	data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_ATOM_IDX].pebs_data_source;
189 	memcpy(data_source, pebs_data_source, sizeof(pebs_data_source));
190 	__intel_pmu_pebs_data_source_grt(data_source);
191 }
192 
193 static void __init __intel_pmu_pebs_data_source_cmt(u64 *data_source)
194 {
195 	data_source[0x07] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOPX, FWD);
196 	data_source[0x08] = OP_LH | P(LVL, L3) | LEVEL(L3) | P(SNOOP, HITM);
197 	data_source[0x0a] = OP_LH | P(LVL, LOC_RAM)  | LEVEL(RAM) | P(SNOOP, NONE);
198 	data_source[0x0b] = OP_LH | LEVEL(RAM) | REM | P(SNOOP, NONE);
199 	data_source[0x0c] = OP_LH | LEVEL(RAM) | REM | P(SNOOPX, FWD);
200 	data_source[0x0d] = OP_LH | LEVEL(RAM) | REM | P(SNOOP, HITM);
201 }
202 
203 void __init intel_pmu_pebs_data_source_mtl(void)
204 {
205 	u64 *data_source;
206 
207 	data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_CORE_IDX].pebs_data_source;
208 	memcpy(data_source, pebs_data_source, sizeof(pebs_data_source));
209 	__intel_pmu_pebs_data_source_skl(false, data_source);
210 
211 	data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_ATOM_IDX].pebs_data_source;
212 	memcpy(data_source, pebs_data_source, sizeof(pebs_data_source));
213 	__intel_pmu_pebs_data_source_cmt(data_source);
214 }
215 
216 void __init intel_pmu_pebs_data_source_arl_h(void)
217 {
218 	u64 *data_source;
219 
220 	intel_pmu_pebs_data_source_lnl();
221 
222 	data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_TINY_IDX].pebs_data_source;
223 	memcpy(data_source, pebs_data_source, sizeof(pebs_data_source));
224 	__intel_pmu_pebs_data_source_cmt(data_source);
225 }
226 
227 void __init intel_pmu_pebs_data_source_cmt(void)
228 {
229 	__intel_pmu_pebs_data_source_cmt(pebs_data_source);
230 }
231 
232 /* Version for Lion Cove and later */
233 static u64 lnc_pebs_data_source[PERF_PEBS_DATA_SOURCE_MAX] = {
234 	P(OP, LOAD) | P(LVL, MISS) | LEVEL(L3) | P(SNOOP, NA),	/* 0x00: ukn L3 */
235 	OP_LH | P(LVL, L1)  | LEVEL(L1) | P(SNOOP, NONE),	/* 0x01: L1 hit */
236 	OP_LH | P(LVL, L1)  | LEVEL(L1) | P(SNOOP, NONE),	/* 0x02: L1 hit */
237 	OP_LH | P(LVL, LFB) | LEVEL(LFB) | P(SNOOP, NONE),	/* 0x03: LFB/L1 Miss Handling Buffer hit */
238 	0,							/* 0x04: Reserved */
239 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, NONE),	/* 0x05: L2 Hit */
240 	OP_LH | LEVEL(L2_MHB) | P(SNOOP, NONE),			/* 0x06: L2 Miss Handling Buffer Hit */
241 	0,							/* 0x07: Reserved */
242 	OP_LH | P(LVL, L3)  | LEVEL(L3) | P(SNOOP, NONE),	/* 0x08: L3 Hit */
243 	0,							/* 0x09: Reserved */
244 	0,							/* 0x0a: Reserved */
245 	0,							/* 0x0b: Reserved */
246 	OP_LH | P(LVL, L3)  | LEVEL(L3) | P(SNOOPX, FWD),	/* 0x0c: L3 Hit Snoop Fwd */
247 	OP_LH | P(LVL, L3)  | LEVEL(L3) | P(SNOOP, HITM),	/* 0x0d: L3 Hit Snoop HitM */
248 	0,							/* 0x0e: Reserved */
249 	P(OP, LOAD) | P(LVL, MISS) | P(LVL, L3)  | LEVEL(L3) | P(SNOOP, HITM),	/* 0x0f: L3 Miss Snoop HitM */
250 	OP_LH | LEVEL(MSC) | P(SNOOP, NONE),			/* 0x10: Memory-side Cache Hit */
251 	OP_LH | P(LVL, LOC_RAM)  | LEVEL(RAM) | P(SNOOP, NONE), /* 0x11: Local Memory Hit */
252 };
253 
254 void __init intel_pmu_pebs_data_source_lnl(void)
255 {
256 	u64 *data_source;
257 
258 	data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_CORE_IDX].pebs_data_source;
259 	memcpy(data_source, lnc_pebs_data_source, sizeof(lnc_pebs_data_source));
260 
261 	data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_ATOM_IDX].pebs_data_source;
262 	memcpy(data_source, pebs_data_source, sizeof(pebs_data_source));
263 	__intel_pmu_pebs_data_source_cmt(data_source);
264 }
265 
266 /* Version for Panthercove and later */
267 
268 /* L2 hit */
269 #define PNC_PEBS_DATA_SOURCE_MAX	16
270 static u64 pnc_pebs_l2_hit_data_source[PNC_PEBS_DATA_SOURCE_MAX] = {
271 	P(OP, LOAD) | P(LVL, NA) | LEVEL(NA) | P(SNOOP, NA),	/* 0x00: non-cache access */
272 	OP_LH               | LEVEL(L0) | P(SNOOP, NONE),	/* 0x01: L0 hit */
273 	OP_LH | P(LVL, L1)  | LEVEL(L1) | P(SNOOP, NONE),	/* 0x02: L1 hit */
274 	OP_LH | P(LVL, LFB) | LEVEL(LFB) | P(SNOOP, NONE),	/* 0x03: L1 Miss Handling Buffer hit */
275 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, NONE),	/* 0x04: L2 Hit Clean */
276 	0,							/* 0x05: Reserved */
277 	0,							/* 0x06: Reserved */
278 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, HIT),	/* 0x07: L2 Hit Snoop HIT */
279 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, HITM),	/* 0x08: L2 Hit Snoop Hit Modified */
280 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, NONE),	/* 0x09: Prefetch Promotion */
281 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, NONE),	/* 0x0a: Cross Core Prefetch Promotion */
282 	0,							/* 0x0b: Reserved */
283 	0,							/* 0x0c: Reserved */
284 	0,							/* 0x0d: Reserved */
285 	0,							/* 0x0e: Reserved */
286 	OP_LH | P(LVL, UNC) | LEVEL(NA) | P(SNOOP, NONE),	/* 0x0f: uncached */
287 };
288 
289 /* Version for Arctic Wolf and later */
290 
291 /* L2 hit */
292 #define ARW_PEBS_DATA_SOURCE_MAX	16
293 static u64 arw_pebs_l2_hit_data_source[ARW_PEBS_DATA_SOURCE_MAX] = {
294 	P(OP, LOAD) | P(LVL, NA) | LEVEL(NA) | P(SNOOP, NA),	/* 0x00: non-cache access */
295 	OP_LH | P(LVL, L1)  | LEVEL(L1) | P(SNOOP, NONE),	/* 0x01: L1 hit */
296 	OP_LH | P(LVL, LFB) | LEVEL(LFB) | P(SNOOP, NONE),	/* 0x02: WCB Hit */
297 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, NONE),	/* 0x03: L2 Hit Clean */
298 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, HIT),	/* 0x04: L2 Hit Snoop HIT */
299 	OP_LH | P(LVL, L2)  | LEVEL(L2) | P(SNOOP, HITM),	/* 0x05: L2 Hit Snoop Hit Modified */
300 	OP_LH | P(LVL, UNC) | LEVEL(NA) | P(SNOOP, NONE),	/* 0x06: uncached */
301 	0,							/* 0x07: Reserved */
302 	0,							/* 0x08: Reserved */
303 	0,							/* 0x09: Reserved */
304 	0,							/* 0x0a: Reserved */
305 	0,							/* 0x0b: Reserved */
306 	0,							/* 0x0c: Reserved */
307 	0,							/* 0x0d: Reserved */
308 	0,							/* 0x0e: Reserved */
309 	0,							/* 0x0f: Reserved */
310 };
311 
312 /* L2 miss */
313 #define OMR_DATA_SOURCE_MAX		16
314 static u64 omr_data_source[OMR_DATA_SOURCE_MAX] = {
315 	P(OP, LOAD) | P(LVL, NA) | LEVEL(NA) | P(SNOOP, NA),	/* 0x00: invalid */
316 	0,							/* 0x01: Reserved */
317 	OP_LH | P(LVL, L3) | LEVEL(L3) | P(REGION, L_SHARE),	/* 0x02: local CA shared cache */
318 	OP_LH | P(LVL, L3) | LEVEL(L3) | P(REGION, L_NON_SHARE),/* 0x03: local CA non-shared cache */
319 	OP_LH | P(LVL, L3) | LEVEL(L3) | P(REGION, O_IO),	/* 0x04: other CA IO agent */
320 	OP_LH | P(LVL, L3) | LEVEL(L3) | P(REGION, O_SHARE),	/* 0x05: other CA shared cache */
321 	OP_LH | P(LVL, L3) | LEVEL(L3) | P(REGION, O_NON_SHARE),/* 0x06: other CA non-shared cache */
322 	OP_LH | LEVEL(RAM) | P(REGION, MMIO),			/* 0x07: MMIO */
323 	OP_LH | LEVEL(RAM) | P(REGION, MEM0),			/* 0x08: Memory region 0 */
324 	OP_LH | LEVEL(RAM) | P(REGION, MEM1),			/* 0x09: Memory region 1 */
325 	OP_LH | LEVEL(RAM) | P(REGION, MEM2),			/* 0x0a: Memory region 2 */
326 	OP_LH | LEVEL(RAM) | P(REGION, MEM3),			/* 0x0b: Memory region 3 */
327 	OP_LH | LEVEL(RAM) | P(REGION, MEM4),			/* 0x0c: Memory region 4 */
328 	OP_LH | LEVEL(RAM) | P(REGION, MEM5),			/* 0x0d: Memory region 5 */
329 	OP_LH | LEVEL(RAM) | P(REGION, MEM6),			/* 0x0e: Memory region 6 */
330 	OP_LH | LEVEL(RAM) | P(REGION, MEM7),			/* 0x0f: Memory region 7 */
331 };
332 
333 static u64 parse_omr_data_source(u8 dse)
334 {
335 	union omr_encoding omr;
336 	u64 val = 0;
337 
338 	omr.omr_full = dse;
339 	val = omr_data_source[omr.omr_source];
340 	if (omr.omr_source > 0x1 && omr.omr_source < 0x7)
341 		val |= omr.omr_remote ? P(LVL, REM_CCE1) : 0;
342 	else if (omr.omr_source > 0x7)
343 		val |= omr.omr_remote ? P(LVL, REM_RAM1) : P(LVL, LOC_RAM);
344 
345 	if (omr.omr_remote)
346 		val |= REM;
347 
348 	if (omr.omr_source == 0x2) {
349 		u8 snoop = omr.omr_snoop | (omr.omr_promoted << 1);
350 
351 		if (omr.omr_hitm)
352 			val |= P(SNOOP, HITM);
353 		else if (snoop == 0x0)
354 			val |= P(SNOOP, NA);
355 		else if (snoop == 0x1)
356 			val |= P(SNOOP, MISS);
357 		else if (snoop == 0x2)
358 			val |= P(SNOOP, HIT);
359 		else if (snoop == 0x3)
360 			val |= P(SNOOP, NONE);
361 	} else if (omr.omr_source > 0x2 && omr.omr_source < 0x7) {
362 		val |= omr.omr_hitm ? P(SNOOP, HITM) : P(SNOOP, HIT);
363 		val |= omr.omr_snoop ? P(SNOOPX, FWD) : 0;
364 	} else {
365 		val |= P(SNOOP, NONE);
366 	}
367 
368 	return val;
369 }
370 
371 static u64 precise_store_data(u64 status)
372 {
373 	union intel_x86_pebs_dse dse;
374 	u64 val = P(OP, STORE) | P(SNOOP, NA) | P(LVL, L1) | P(TLB, L2);
375 
376 	dse.val = status;
377 
378 	/*
379 	 * bit 4: TLB access
380 	 * 1 = stored missed 2nd level TLB
381 	 *
382 	 * so it either hit the walker or the OS
383 	 * otherwise hit 2nd level TLB
384 	 */
385 	if (dse.st_stlb_miss)
386 		val |= P(TLB, MISS);
387 	else
388 		val |= P(TLB, HIT);
389 
390 	/*
391 	 * bit 0: hit L1 data cache
392 	 * if not set, then all we know is that
393 	 * it missed L1D
394 	 */
395 	if (dse.st_l1d_hit)
396 		val |= P(LVL, HIT);
397 	else
398 		val |= P(LVL, MISS);
399 
400 	/*
401 	 * bit 5: Locked prefix
402 	 */
403 	if (dse.st_locked)
404 		val |= P(LOCK, LOCKED);
405 
406 	return val;
407 }
408 
409 static u64 precise_datala_hsw(struct perf_event *event, u64 status)
410 {
411 	union perf_mem_data_src dse;
412 
413 	dse.val = PERF_MEM_NA;
414 
415 	if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW)
416 		dse.mem_op = PERF_MEM_OP_STORE;
417 	else if (event->hw.flags & PERF_X86_EVENT_PEBS_LD_HSW)
418 		dse.mem_op = PERF_MEM_OP_LOAD;
419 
420 	/*
421 	 * L1 info only valid for following events:
422 	 *
423 	 * MEM_UOPS_RETIRED.STLB_MISS_STORES
424 	 * MEM_UOPS_RETIRED.LOCK_STORES
425 	 * MEM_UOPS_RETIRED.SPLIT_STORES
426 	 * MEM_UOPS_RETIRED.ALL_STORES
427 	 */
428 	if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW) {
429 		if (status & 1)
430 			dse.mem_lvl = PERF_MEM_LVL_L1 | PERF_MEM_LVL_HIT;
431 		else
432 			dse.mem_lvl = PERF_MEM_LVL_L1 | PERF_MEM_LVL_MISS;
433 	}
434 	return dse.val;
435 }
436 
437 static inline void pebs_set_tlb_lock(u64 *val, bool tlb, bool lock)
438 {
439 	/*
440 	 * TLB access
441 	 * 0 = did not miss 2nd level TLB
442 	 * 1 = missed 2nd level TLB
443 	 */
444 	if (tlb)
445 		*val |= P(TLB, MISS) | P(TLB, L2);
446 	else
447 		*val |= P(TLB, HIT) | P(TLB, L1) | P(TLB, L2);
448 
449 	/* locked prefix */
450 	if (lock)
451 		*val |= P(LOCK, LOCKED);
452 }
453 
454 /* Retrieve the latency data for e-core of ADL */
455 static u64 __grt_latency_data(struct perf_event *event, u64 status,
456 			       u8 dse, bool tlb, bool lock, bool blk)
457 {
458 	union perf_mem_data_src src;
459 	u64 val;
460 
461 	WARN_ON_ONCE(is_hybrid() &&
462 		     hybrid_pmu(event->pmu)->pmu_type == hybrid_big);
463 
464 	dse &= PERF_PEBS_DATA_SOURCE_GRT_MASK;
465 	val = hybrid_var(event->pmu, pebs_data_source)[dse];
466 
467 	pebs_set_tlb_lock(&val, tlb, lock);
468 
469 	if (blk)
470 		val |= P(BLK, DATA);
471 	else
472 		val |= P(BLK, NA);
473 
474 	src.val = val;
475 
476 	if (event->hw.flags &
477 	    (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
478 		src.mem_op = P(OP, LOAD);
479 	if (event->hw.flags &
480 	    (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
481 		src.mem_op = P(OP, STORE);
482 
483 	return src.val;
484 }
485 
486 u64 grt_latency_data(struct perf_event *event, u64 status)
487 {
488 	union intel_x86_pebs_dse dse;
489 
490 	dse.val = status;
491 
492 	return __grt_latency_data(event, status, dse.ld_dse,
493 				  dse.ld_locked, dse.ld_stlb_miss,
494 				  dse.ld_data_blk);
495 }
496 
497 /* Retrieve the latency data for e-core of MTL */
498 u64 cmt_latency_data(struct perf_event *event, u64 status)
499 {
500 	union intel_x86_pebs_dse dse;
501 
502 	dse.val = status;
503 
504 	return __grt_latency_data(event, status, dse.mtl_dse,
505 				  dse.mtl_stlb_miss, dse.mtl_locked,
506 				  dse.mtl_fwd_blk);
507 }
508 
509 static u64 arw_latency_data(struct perf_event *event, u64 status)
510 {
511 	union intel_x86_pebs_dse dse;
512 	union perf_mem_data_src src;
513 	u64 val;
514 
515 	dse.val = status;
516 
517 	if (!dse.arw_l2_miss)
518 		val = arw_pebs_l2_hit_data_source[dse.arw_dse & 0xf];
519 	else
520 		val = parse_omr_data_source(dse.arw_dse);
521 
522 	if (!val)
523 		val = P(OP, LOAD) | LEVEL(NA) | P(SNOOP, NA);
524 
525 	if (dse.arw_stlb_miss)
526 		val |= P(TLB, MISS) | P(TLB, L2);
527 	else
528 		val |= P(TLB, HIT) | P(TLB, L1) | P(TLB, L2);
529 
530 	if (dse.arw_locked)
531 		val |= P(LOCK, LOCKED);
532 
533 	if (dse.arw_data_blk)
534 		val |= P(BLK, DATA);
535 	if (dse.arw_addr_blk)
536 		val |= P(BLK, ADDR);
537 	if (!dse.arw_data_blk && !dse.arw_addr_blk)
538 		val |= P(BLK, NA);
539 
540 	src.val = val;
541 	if (event->hw.flags &
542 	    (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
543 		src.mem_op = P(OP, LOAD);
544 	if (event->hw.flags &
545 	    (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
546 		src.mem_op = P(OP, STORE);
547 
548 	return src.val;
549 }
550 
551 static u64 lnc_latency_data(struct perf_event *event, u64 status)
552 {
553 	union intel_x86_pebs_dse dse;
554 	union perf_mem_data_src src;
555 	u64 val;
556 
557 	dse.val = status;
558 
559 	/* LNC core latency data */
560 	val = hybrid_var(event->pmu, pebs_data_source)[status & PERF_PEBS_DATA_SOURCE_MASK];
561 	if (!val)
562 		val = P(OP, LOAD) | LEVEL(NA) | P(SNOOP, NA);
563 
564 	if (dse.lnc_stlb_miss)
565 		val |= P(TLB, MISS) | P(TLB, L2);
566 	else
567 		val |= P(TLB, HIT) | P(TLB, L1) | P(TLB, L2);
568 
569 	if (dse.lnc_locked)
570 		val |= P(LOCK, LOCKED);
571 
572 	if (dse.lnc_data_blk)
573 		val |= P(BLK, DATA);
574 	if (dse.lnc_addr_blk)
575 		val |= P(BLK, ADDR);
576 	if (!dse.lnc_data_blk && !dse.lnc_addr_blk)
577 		val |= P(BLK, NA);
578 
579 	src.val = val;
580 	if (event->hw.flags &
581 	    (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
582 		src.mem_op = P(OP, LOAD);
583 	if (event->hw.flags &
584 	    (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
585 		src.mem_op = P(OP, STORE);
586 
587 	return src.val;
588 }
589 
590 u64 lnl_latency_data(struct perf_event *event, u64 status)
591 {
592 	struct x86_hybrid_pmu *pmu = hybrid_pmu(event->pmu);
593 
594 	if (pmu->pmu_type == hybrid_small)
595 		return cmt_latency_data(event, status);
596 
597 	return lnc_latency_data(event, status);
598 }
599 
600 u64 arl_h_latency_data(struct perf_event *event, u64 status)
601 {
602 	struct x86_hybrid_pmu *pmu = hybrid_pmu(event->pmu);
603 
604 	if (pmu->pmu_type == hybrid_tiny)
605 		return cmt_latency_data(event, status);
606 
607 	return lnl_latency_data(event, status);
608 }
609 
610 u64 pnc_latency_data(struct perf_event *event, u64 status)
611 {
612 	union intel_x86_pebs_dse dse;
613 	union perf_mem_data_src src;
614 	u64 val;
615 
616 	dse.val = status;
617 
618 	if (!dse.pnc_l2_miss)
619 		val = pnc_pebs_l2_hit_data_source[dse.pnc_dse & 0xf];
620 	else
621 		val = parse_omr_data_source(dse.pnc_dse);
622 
623 	if (!val)
624 		val = P(OP, LOAD) | LEVEL(NA) | P(SNOOP, NA);
625 
626 	if (dse.pnc_stlb_miss)
627 		val |= P(TLB, MISS) | P(TLB, L2);
628 	else
629 		val |= P(TLB, HIT) | P(TLB, L1) | P(TLB, L2);
630 
631 	if (dse.pnc_locked)
632 		val |= P(LOCK, LOCKED);
633 
634 	if (dse.pnc_data_blk)
635 		val |= P(BLK, DATA);
636 	if (dse.pnc_addr_blk)
637 		val |= P(BLK, ADDR);
638 	if (!dse.pnc_data_blk && !dse.pnc_addr_blk)
639 		val |= P(BLK, NA);
640 
641 	src.val = val;
642 	if (event->hw.flags &
643 	    (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
644 		src.mem_op = P(OP, LOAD);
645 	if (event->hw.flags &
646 	    (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
647 		src.mem_op = P(OP, STORE);
648 
649 	return src.val;
650 }
651 
652 u64 nvl_latency_data(struct perf_event *event, u64 status)
653 {
654 	struct x86_hybrid_pmu *pmu = hybrid_pmu(event->pmu);
655 
656 	if (pmu->pmu_type == hybrid_small)
657 		return arw_latency_data(event, status);
658 
659 	return pnc_latency_data(event, status);
660 }
661 
662 static u64 load_latency_data(struct perf_event *event, u64 status)
663 {
664 	union intel_x86_pebs_dse dse;
665 	u64 val;
666 
667 	dse.val = status;
668 
669 	/*
670 	 * use the mapping table for bit 0-3
671 	 */
672 	val = hybrid_var(event->pmu, pebs_data_source)[dse.ld_dse];
673 
674 	/*
675 	 * Nehalem models do not support TLB, Lock infos
676 	 */
677 	if (x86_pmu.pebs_no_tlb) {
678 		val |= P(TLB, NA) | P(LOCK, NA);
679 		return val;
680 	}
681 
682 	pebs_set_tlb_lock(&val, dse.ld_stlb_miss, dse.ld_locked);
683 
684 	/*
685 	 * Ice Lake and earlier models do not support block infos.
686 	 */
687 	if (!x86_pmu.pebs_block) {
688 		val |= P(BLK, NA);
689 		return val;
690 	}
691 	/*
692 	 * bit 6: load was blocked since its data could not be forwarded
693 	 *        from a preceding store
694 	 */
695 	if (dse.ld_data_blk)
696 		val |= P(BLK, DATA);
697 
698 	/*
699 	 * bit 7: load was blocked due to potential address conflict with
700 	 *        a preceding store
701 	 */
702 	if (dse.ld_addr_blk)
703 		val |= P(BLK, ADDR);
704 
705 	if (!dse.ld_data_blk && !dse.ld_addr_blk)
706 		val |= P(BLK, NA);
707 
708 	return val;
709 }
710 
711 static u64 store_latency_data(struct perf_event *event, u64 status)
712 {
713 	union intel_x86_pebs_dse dse;
714 	union perf_mem_data_src src;
715 	u64 val;
716 
717 	dse.val = status;
718 
719 	/*
720 	 * use the mapping table for bit 0-3
721 	 */
722 	val = hybrid_var(event->pmu, pebs_data_source)[dse.st_lat_dse];
723 
724 	pebs_set_tlb_lock(&val, dse.st_lat_stlb_miss, dse.st_lat_locked);
725 
726 	val |= P(BLK, NA);
727 
728 	/*
729 	 * the pebs_data_source table is only for loads
730 	 * so override the mem_op to say STORE instead
731 	 */
732 	src.val = val;
733 	src.mem_op = P(OP,STORE);
734 
735 	return src.val;
736 }
737 
738 struct pebs_record_core {
739 	u64 flags, ip;
740 	u64 ax, bx, cx, dx;
741 	u64 si, di, bp, sp;
742 	u64 r8,  r9,  r10, r11;
743 	u64 r12, r13, r14, r15;
744 };
745 
746 struct pebs_record_nhm {
747 	u64 flags, ip;
748 	u64 ax, bx, cx, dx;
749 	u64 si, di, bp, sp;
750 	u64 r8,  r9,  r10, r11;
751 	u64 r12, r13, r14, r15;
752 	u64 status, dla, dse, lat;
753 };
754 
755 /*
756  * Same as pebs_record_nhm, with two additional fields.
757  */
758 struct pebs_record_hsw {
759 	u64 flags, ip;
760 	u64 ax, bx, cx, dx;
761 	u64 si, di, bp, sp;
762 	u64 r8,  r9,  r10, r11;
763 	u64 r12, r13, r14, r15;
764 	u64 status, dla, dse, lat;
765 	u64 real_ip, tsx_tuning;
766 };
767 
768 union hsw_tsx_tuning {
769 	struct {
770 		u32 cycles_last_block     : 32,
771 		    hle_abort		  : 1,
772 		    rtm_abort		  : 1,
773 		    instruction_abort     : 1,
774 		    non_instruction_abort : 1,
775 		    retry		  : 1,
776 		    data_conflict	  : 1,
777 		    capacity_writes	  : 1,
778 		    capacity_reads	  : 1;
779 	};
780 	u64	    value;
781 };
782 
783 #define PEBS_HSW_TSX_FLAGS	0xff00000000ULL
784 
785 /* Same as HSW, plus TSC */
786 
787 struct pebs_record_skl {
788 	u64 flags, ip;
789 	u64 ax, bx, cx, dx;
790 	u64 si, di, bp, sp;
791 	u64 r8,  r9,  r10, r11;
792 	u64 r12, r13, r14, r15;
793 	u64 status, dla, dse, lat;
794 	u64 real_ip, tsx_tuning;
795 	u64 tsc;
796 };
797 
798 void init_debug_store_on_cpu(int cpu)
799 {
800 	struct debug_store *ds = per_cpu(cpu_hw_events, cpu).ds;
801 
802 	if (!ds)
803 		return;
804 
805 	wrmsrq_on_cpu(cpu, MSR_IA32_DS_AREA, (u64)(unsigned long)ds);
806 }
807 
808 void fini_debug_store_on_cpu(int cpu)
809 {
810 	if (!per_cpu(cpu_hw_events, cpu).ds)
811 		return;
812 
813 	wrmsrq_on_cpu(cpu, MSR_IA32_DS_AREA, 0);
814 }
815 
816 static DEFINE_PER_CPU(void *, insn_buffer);
817 
818 static void ds_update_cea(void *cea, void *addr, size_t size, pgprot_t prot)
819 {
820 	unsigned long start = (unsigned long)cea;
821 	phys_addr_t pa;
822 	size_t msz = 0;
823 
824 	pa = virt_to_phys(addr);
825 
826 	preempt_disable();
827 	for (; msz < size; msz += PAGE_SIZE, pa += PAGE_SIZE, cea += PAGE_SIZE)
828 		cea_set_pte(cea, pa, prot);
829 
830 	/*
831 	 * This is a cross-CPU update of the cpu_entry_area, we must shoot down
832 	 * all TLB entries for it.
833 	 */
834 	flush_tlb_kernel_range(start, start + size);
835 	preempt_enable();
836 }
837 
838 static void ds_clear_cea(void *cea, size_t size)
839 {
840 	unsigned long start = (unsigned long)cea;
841 	size_t msz = 0;
842 
843 	preempt_disable();
844 	for (; msz < size; msz += PAGE_SIZE, cea += PAGE_SIZE)
845 		cea_set_pte(cea, 0, PAGE_NONE);
846 
847 	flush_tlb_kernel_range(start, start + size);
848 	preempt_enable();
849 }
850 
851 static void *dsalloc_pages(size_t size, gfp_t flags, int cpu)
852 {
853 	unsigned int order = get_order(size);
854 	int node = cpu_to_node(cpu);
855 	struct page *page;
856 
857 	page = alloc_pages_node(node, flags | __GFP_ZERO, order);
858 	return page ? page_address(page) : NULL;
859 }
860 
861 static void dsfree_pages(const void *buffer, size_t size)
862 {
863 	if (buffer)
864 		free_pages((unsigned long)buffer, get_order(size));
865 }
866 
867 static int alloc_pebs_buffer(int cpu)
868 {
869 	struct cpu_hw_events *hwev = per_cpu_ptr(&cpu_hw_events, cpu);
870 	struct debug_store *ds = hwev->ds;
871 	size_t bsiz = x86_pmu.pebs_buffer_size;
872 	int max, node = cpu_to_node(cpu);
873 	void *buffer, *insn_buff, *cea;
874 
875 	if (!intel_pmu_has_pebs())
876 		return 0;
877 
878 	buffer = dsalloc_pages(bsiz, GFP_KERNEL, cpu);
879 	if (unlikely(!buffer))
880 		return -ENOMEM;
881 
882 	if (x86_pmu.arch_pebs) {
883 		hwev->pebs_vaddr = buffer;
884 		return 0;
885 	}
886 
887 	/*
888 	 * HSW+ already provides us the eventing ip; no need to allocate this
889 	 * buffer then.
890 	 */
891 	if (x86_pmu.intel_cap.pebs_format < 2) {
892 		insn_buff = kzalloc_node(PEBS_FIXUP_SIZE, GFP_KERNEL, node);
893 		if (!insn_buff) {
894 			dsfree_pages(buffer, bsiz);
895 			return -ENOMEM;
896 		}
897 		per_cpu(insn_buffer, cpu) = insn_buff;
898 	}
899 	hwev->pebs_vaddr = buffer;
900 	/* Update the cpu entry area mapping */
901 	cea = &get_cpu_entry_area(cpu)->cpu_debug_buffers.pebs_buffer;
902 	ds->pebs_buffer_base = (unsigned long) cea;
903 	ds_update_cea(cea, buffer, bsiz, PAGE_KERNEL);
904 	ds->pebs_index = ds->pebs_buffer_base;
905 	max = x86_pmu.pebs_record_size * (bsiz / x86_pmu.pebs_record_size);
906 	ds->pebs_absolute_maximum = ds->pebs_buffer_base + max;
907 	return 0;
908 }
909 
910 static void release_pebs_buffer(int cpu)
911 {
912 	struct cpu_hw_events *hwev = per_cpu_ptr(&cpu_hw_events, cpu);
913 	void *cea;
914 
915 	if (!intel_pmu_has_pebs())
916 		return;
917 
918 	if (x86_pmu.ds_pebs) {
919 		kfree(per_cpu(insn_buffer, cpu));
920 		per_cpu(insn_buffer, cpu) = NULL;
921 
922 		/* Clear the fixmap */
923 		cea = &get_cpu_entry_area(cpu)->cpu_debug_buffers.pebs_buffer;
924 		ds_clear_cea(cea, x86_pmu.pebs_buffer_size);
925 	}
926 
927 	dsfree_pages(hwev->pebs_vaddr, x86_pmu.pebs_buffer_size);
928 	hwev->pebs_vaddr = NULL;
929 }
930 
931 static int alloc_bts_buffer(int cpu)
932 {
933 	struct cpu_hw_events *hwev = per_cpu_ptr(&cpu_hw_events, cpu);
934 	struct debug_store *ds = hwev->ds;
935 	void *buffer, *cea;
936 	int max;
937 
938 	if (!x86_pmu.bts)
939 		return 0;
940 
941 	buffer = dsalloc_pages(BTS_BUFFER_SIZE, GFP_KERNEL | __GFP_NOWARN, cpu);
942 	if (unlikely(!buffer)) {
943 		WARN_ONCE(1, "%s: BTS buffer allocation failure\n", __func__);
944 		return -ENOMEM;
945 	}
946 	hwev->ds_bts_vaddr = buffer;
947 	/* Update the fixmap */
948 	cea = &get_cpu_entry_area(cpu)->cpu_debug_buffers.bts_buffer;
949 	ds->bts_buffer_base = (unsigned long) cea;
950 	ds_update_cea(cea, buffer, BTS_BUFFER_SIZE, PAGE_KERNEL);
951 	ds->bts_index = ds->bts_buffer_base;
952 	max = BTS_BUFFER_SIZE / BTS_RECORD_SIZE;
953 	ds->bts_absolute_maximum = ds->bts_buffer_base +
954 					max * BTS_RECORD_SIZE;
955 	ds->bts_interrupt_threshold = ds->bts_absolute_maximum -
956 					(max / 16) * BTS_RECORD_SIZE;
957 	return 0;
958 }
959 
960 static void release_bts_buffer(int cpu)
961 {
962 	struct cpu_hw_events *hwev = per_cpu_ptr(&cpu_hw_events, cpu);
963 	void *cea;
964 
965 	if (!x86_pmu.bts)
966 		return;
967 
968 	/* Clear the fixmap */
969 	cea = &get_cpu_entry_area(cpu)->cpu_debug_buffers.bts_buffer;
970 	ds_clear_cea(cea, BTS_BUFFER_SIZE);
971 	dsfree_pages(hwev->ds_bts_vaddr, BTS_BUFFER_SIZE);
972 	hwev->ds_bts_vaddr = NULL;
973 }
974 
975 static int alloc_ds_buffer(int cpu)
976 {
977 	struct debug_store *ds = &get_cpu_entry_area(cpu)->cpu_debug_store;
978 
979 	memset(ds, 0, sizeof(*ds));
980 	per_cpu(cpu_hw_events, cpu).ds = ds;
981 	return 0;
982 }
983 
984 static void release_ds_buffer(int cpu)
985 {
986 	per_cpu(cpu_hw_events, cpu).ds = NULL;
987 }
988 
989 void release_ds_buffers(void)
990 {
991 	int cpu;
992 
993 	if (!x86_pmu.bts && !x86_pmu.ds_pebs)
994 		return;
995 
996 	for_each_possible_cpu(cpu)
997 		release_ds_buffer(cpu);
998 
999 	for_each_possible_cpu(cpu) {
1000 		/*
1001 		 * Again, ignore errors from offline CPUs, they will no longer
1002 		 * observe cpu_hw_events.ds and not program the DS_AREA when
1003 		 * they come up.
1004 		 */
1005 		fini_debug_store_on_cpu(cpu);
1006 	}
1007 
1008 	for_each_possible_cpu(cpu) {
1009 		if (x86_pmu.ds_pebs)
1010 			release_pebs_buffer(cpu);
1011 		release_bts_buffer(cpu);
1012 	}
1013 }
1014 
1015 void reserve_ds_buffers(void)
1016 {
1017 	int bts_err = 0, pebs_err = 0;
1018 	int cpu;
1019 
1020 	x86_pmu.bts_active = 0;
1021 
1022 	if (x86_pmu.ds_pebs)
1023 		x86_pmu.pebs_active = 0;
1024 
1025 	if (!x86_pmu.bts && !x86_pmu.ds_pebs)
1026 		return;
1027 
1028 	if (!x86_pmu.bts)
1029 		bts_err = 1;
1030 
1031 	if (!x86_pmu.ds_pebs)
1032 		pebs_err = 1;
1033 
1034 	for_each_possible_cpu(cpu) {
1035 		if (alloc_ds_buffer(cpu)) {
1036 			bts_err = 1;
1037 			pebs_err = 1;
1038 		}
1039 
1040 		if (!bts_err && alloc_bts_buffer(cpu))
1041 			bts_err = 1;
1042 
1043 		if (x86_pmu.ds_pebs && !pebs_err &&
1044 		    alloc_pebs_buffer(cpu))
1045 			pebs_err = 1;
1046 
1047 		if (bts_err && pebs_err)
1048 			break;
1049 	}
1050 
1051 	if (bts_err) {
1052 		for_each_possible_cpu(cpu)
1053 			release_bts_buffer(cpu);
1054 	}
1055 
1056 	if (x86_pmu.ds_pebs && pebs_err) {
1057 		for_each_possible_cpu(cpu)
1058 			release_pebs_buffer(cpu);
1059 	}
1060 
1061 	if (bts_err && pebs_err) {
1062 		for_each_possible_cpu(cpu)
1063 			release_ds_buffer(cpu);
1064 	} else {
1065 		if (x86_pmu.bts && !bts_err)
1066 			x86_pmu.bts_active = 1;
1067 
1068 		if (x86_pmu.ds_pebs && !pebs_err)
1069 			x86_pmu.pebs_active = 1;
1070 
1071 		for_each_possible_cpu(cpu) {
1072 			/*
1073 			 * Ignores wrmsr_on_cpu() errors for offline CPUs they
1074 			 * will get this call through intel_pmu_cpu_starting().
1075 			 */
1076 			init_debug_store_on_cpu(cpu);
1077 		}
1078 	}
1079 }
1080 
1081 inline int alloc_arch_pebs_buf_on_cpu(int cpu)
1082 {
1083 	if (!x86_pmu.arch_pebs)
1084 		return 0;
1085 
1086 	return alloc_pebs_buffer(cpu);
1087 }
1088 
1089 inline void release_arch_pebs_buf_on_cpu(int cpu)
1090 {
1091 	if (!x86_pmu.arch_pebs)
1092 		return;
1093 
1094 	release_pebs_buffer(cpu);
1095 }
1096 
1097 void init_arch_pebs_on_cpu(int cpu)
1098 {
1099 	struct cpu_hw_events *cpuc = per_cpu_ptr(&cpu_hw_events, cpu);
1100 	u64 arch_pebs_base;
1101 
1102 	if (!x86_pmu.arch_pebs)
1103 		return;
1104 
1105 	if (!cpuc->pebs_vaddr) {
1106 		WARN(1, "Fail to allocate PEBS buffer on CPU %d\n", cpu);
1107 		x86_pmu.pebs_active = 0;
1108 		return;
1109 	}
1110 
1111 	/*
1112 	 * 4KB-aligned pointer of the output buffer
1113 	 * (alloc_pages_node() returns page aligned address)
1114 	 * Buffer Size = 4KB * 2^SIZE
1115 	 * contiguous physical buffer (alloc_pages_node() with order)
1116 	 */
1117 	arch_pebs_base = virt_to_phys(cpuc->pebs_vaddr) | PEBS_BUFFER_SHIFT;
1118 	wrmsrq_on_cpu(cpu, MSR_IA32_PEBS_BASE, arch_pebs_base);
1119 	x86_pmu.pebs_active = 1;
1120 }
1121 
1122 inline void fini_arch_pebs_on_cpu(int cpu)
1123 {
1124 	if (!x86_pmu.arch_pebs)
1125 		return;
1126 
1127 	wrmsrq_on_cpu(cpu, MSR_IA32_PEBS_BASE, 0);
1128 }
1129 
1130 /*
1131  * BTS
1132  */
1133 
1134 struct event_constraint bts_constraint =
1135 	EVENT_CONSTRAINT(0, 1ULL << INTEL_PMC_IDX_FIXED_BTS, 0);
1136 
1137 void intel_pmu_enable_bts(u64 config)
1138 {
1139 	unsigned long debugctlmsr;
1140 
1141 	debugctlmsr = get_debugctlmsr();
1142 
1143 	debugctlmsr |= DEBUGCTLMSR_TR;
1144 	debugctlmsr |= DEBUGCTLMSR_BTS;
1145 	if (config & ARCH_PERFMON_EVENTSEL_INT)
1146 		debugctlmsr |= DEBUGCTLMSR_BTINT;
1147 
1148 	if (!(config & ARCH_PERFMON_EVENTSEL_OS))
1149 		debugctlmsr |= DEBUGCTLMSR_BTS_OFF_OS;
1150 
1151 	if (!(config & ARCH_PERFMON_EVENTSEL_USR))
1152 		debugctlmsr |= DEBUGCTLMSR_BTS_OFF_USR;
1153 
1154 	update_debugctlmsr(debugctlmsr);
1155 }
1156 
1157 void intel_pmu_disable_bts(void)
1158 {
1159 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1160 	unsigned long debugctlmsr;
1161 
1162 	if (!cpuc->ds)
1163 		return;
1164 
1165 	debugctlmsr = get_debugctlmsr();
1166 
1167 	debugctlmsr &=
1168 		~(DEBUGCTLMSR_TR | DEBUGCTLMSR_BTS | DEBUGCTLMSR_BTINT |
1169 		  DEBUGCTLMSR_BTS_OFF_OS | DEBUGCTLMSR_BTS_OFF_USR);
1170 
1171 	update_debugctlmsr(debugctlmsr);
1172 }
1173 
1174 int intel_pmu_drain_bts_buffer(void)
1175 {
1176 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1177 	struct debug_store *ds = cpuc->ds;
1178 	struct bts_record {
1179 		u64	from;
1180 		u64	to;
1181 		u64	flags;
1182 	};
1183 	struct perf_event *event = cpuc->events[INTEL_PMC_IDX_FIXED_BTS];
1184 	struct bts_record *at, *base, *top;
1185 	struct perf_output_handle handle;
1186 	struct perf_event_header header;
1187 	struct perf_sample_data data;
1188 	unsigned long skip = 0;
1189 	struct pt_regs regs;
1190 
1191 	if (!event)
1192 		return 0;
1193 
1194 	if (!x86_pmu.bts_active)
1195 		return 0;
1196 
1197 	base = (struct bts_record *)(unsigned long)ds->bts_buffer_base;
1198 	top  = (struct bts_record *)(unsigned long)ds->bts_index;
1199 
1200 	if (top <= base)
1201 		return 0;
1202 
1203 	memset(&regs, 0, sizeof(regs));
1204 
1205 	ds->bts_index = ds->bts_buffer_base;
1206 
1207 	perf_sample_data_init(&data, 0, event->hw.last_period);
1208 
1209 	/*
1210 	 * BTS leaks kernel addresses in branches across the cpl boundary,
1211 	 * such as traps or system calls, so unless the user is asking for
1212 	 * kernel tracing (and right now it's not possible), we'd need to
1213 	 * filter them out. But first we need to count how many of those we
1214 	 * have in the current batch. This is an extra O(n) pass, however,
1215 	 * it's much faster than the other one especially considering that
1216 	 * n <= 2560 (BTS_BUFFER_SIZE / BTS_RECORD_SIZE * 15/16; see the
1217 	 * alloc_bts_buffer()).
1218 	 */
1219 	for (at = base; at < top; at++) {
1220 		/*
1221 		 * Note that right now *this* BTS code only works if
1222 		 * attr::exclude_kernel is set, but let's keep this extra
1223 		 * check here in case that changes.
1224 		 */
1225 		if (event->attr.exclude_kernel &&
1226 		    (kernel_ip(at->from) || kernel_ip(at->to)))
1227 			skip++;
1228 	}
1229 
1230 	/*
1231 	 * Prepare a generic sample, i.e. fill in the invariant fields.
1232 	 * We will overwrite the from and to address before we output
1233 	 * the sample.
1234 	 */
1235 	rcu_read_lock();
1236 	perf_prepare_sample(&data, event, &regs);
1237 	perf_prepare_header(&header, &data, event, &regs);
1238 
1239 	if (perf_output_begin(&handle, &data, event,
1240 			      header.size * (top - base - skip)))
1241 		goto unlock;
1242 
1243 	for (at = base; at < top; at++) {
1244 		/* Filter out any records that contain kernel addresses. */
1245 		if (event->attr.exclude_kernel &&
1246 		    (kernel_ip(at->from) || kernel_ip(at->to)))
1247 			continue;
1248 
1249 		data.ip		= at->from;
1250 		data.addr	= at->to;
1251 
1252 		perf_output_sample(&handle, &header, &data, event);
1253 	}
1254 
1255 	perf_output_end(&handle);
1256 
1257 	/* There's new data available. */
1258 	event->hw.interrupts++;
1259 	event->pending_kill = POLL_IN;
1260 unlock:
1261 	rcu_read_unlock();
1262 	return 1;
1263 }
1264 
1265 void intel_pmu_drain_pebs_buffer(void)
1266 {
1267 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1268 	struct perf_sample_data data;
1269 
1270 	WARN_ON_ONCE(cpuc->enabled);
1271 
1272 	static_call(x86_pmu_drain_pebs)(NULL, &data);
1273 }
1274 
1275 /*
1276  * PEBS
1277  */
1278 struct event_constraint intel_core2_pebs_event_constraints[] = {
1279 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x00c0, 0x1), /* INST_RETIRED.ANY */
1280 	INTEL_FLAGS_UEVENT_CONSTRAINT(0xfec1, 0x1), /* X87_OPS_RETIRED.ANY */
1281 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x00c5, 0x1), /* BR_INST_RETIRED.MISPRED */
1282 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x1fc7, 0x1), /* SIMD_INST_RETURED.ANY */
1283 	INTEL_FLAGS_EVENT_CONSTRAINT(0xcb, 0x1),    /* MEM_LOAD_RETIRED.* */
1284 	/* INST_RETIRED.ANY_P, inv=1, cmask=16 (cycles:p). */
1285 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108000c0, 0x01),
1286 	EVENT_CONSTRAINT_END
1287 };
1288 
1289 struct event_constraint intel_atom_pebs_event_constraints[] = {
1290 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x00c0, 0x1), /* INST_RETIRED.ANY */
1291 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x00c5, 0x1), /* MISPREDICTED_BRANCH_RETIRED */
1292 	INTEL_FLAGS_EVENT_CONSTRAINT(0xcb, 0x1),    /* MEM_LOAD_RETIRED.* */
1293 	/* INST_RETIRED.ANY_P, inv=1, cmask=16 (cycles:p). */
1294 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108000c0, 0x01),
1295 	/* Allow all events as PEBS with no flags */
1296 	INTEL_ALL_EVENT_CONSTRAINT(0, 0x1),
1297 	EVENT_CONSTRAINT_END
1298 };
1299 
1300 struct event_constraint intel_slm_pebs_event_constraints[] = {
1301 	/* INST_RETIRED.ANY_P, inv=1, cmask=16 (cycles:p). */
1302 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108000c0, 0x1),
1303 	/* Allow all events as PEBS with no flags */
1304 	INTEL_ALL_EVENT_CONSTRAINT(0, 0x1),
1305 	EVENT_CONSTRAINT_END
1306 };
1307 
1308 struct event_constraint intel_glm_pebs_event_constraints[] = {
1309 	/* Allow all events as PEBS with no flags */
1310 	INTEL_ALL_EVENT_CONSTRAINT(0, 0x1),
1311 	EVENT_CONSTRAINT_END
1312 };
1313 
1314 struct event_constraint intel_grt_pebs_event_constraints[] = {
1315 	/* Allow all events as PEBS with no flags */
1316 	INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0x3),
1317 	INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0x3f),
1318 	EVENT_CONSTRAINT_END
1319 };
1320 
1321 struct event_constraint intel_cmt_pebs_event_constraints[] = {
1322 	/* Allow all events as PEBS with no flags */
1323 	INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0x3),
1324 	INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0xff),
1325 	EVENT_CONSTRAINT_END
1326 };
1327 
1328 struct event_constraint intel_dkt_pebs_event_constraints[] = {
1329 	/* Allow all events as PEBS with no flags */
1330 	INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0xff),
1331 	INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0xff),
1332 	EVENT_CONSTRAINT_END
1333 };
1334 
1335 struct event_constraint intel_nehalem_pebs_event_constraints[] = {
1336 	INTEL_PLD_CONSTRAINT(0x100b, 0xf),      /* MEM_INST_RETIRED.* */
1337 	INTEL_FLAGS_EVENT_CONSTRAINT(0x0f, 0xf),    /* MEM_UNCORE_RETIRED.* */
1338 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x010c, 0xf), /* MEM_STORE_RETIRED.DTLB_MISS */
1339 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc0, 0xf),    /* INST_RETIRED.ANY */
1340 	INTEL_EVENT_CONSTRAINT(0xc2, 0xf),    /* UOPS_RETIRED.* */
1341 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc4, 0xf),    /* BR_INST_RETIRED.* */
1342 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x02c5, 0xf), /* BR_MISP_RETIRED.NEAR_CALL */
1343 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc7, 0xf),    /* SSEX_UOPS_RETIRED.* */
1344 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x20c8, 0xf), /* ITLB_MISS_RETIRED */
1345 	INTEL_FLAGS_EVENT_CONSTRAINT(0xcb, 0xf),    /* MEM_LOAD_RETIRED.* */
1346 	INTEL_FLAGS_EVENT_CONSTRAINT(0xf7, 0xf),    /* FP_ASSIST.* */
1347 	/* INST_RETIRED.ANY_P, inv=1, cmask=16 (cycles:p). */
1348 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108000c0, 0x0f),
1349 	EVENT_CONSTRAINT_END
1350 };
1351 
1352 struct event_constraint intel_westmere_pebs_event_constraints[] = {
1353 	INTEL_PLD_CONSTRAINT(0x100b, 0xf),      /* MEM_INST_RETIRED.* */
1354 	INTEL_FLAGS_EVENT_CONSTRAINT(0x0f, 0xf),    /* MEM_UNCORE_RETIRED.* */
1355 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x010c, 0xf), /* MEM_STORE_RETIRED.DTLB_MISS */
1356 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc0, 0xf),    /* INSTR_RETIRED.* */
1357 	INTEL_EVENT_CONSTRAINT(0xc2, 0xf),    /* UOPS_RETIRED.* */
1358 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc4, 0xf),    /* BR_INST_RETIRED.* */
1359 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc5, 0xf),    /* BR_MISP_RETIRED.* */
1360 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc7, 0xf),    /* SSEX_UOPS_RETIRED.* */
1361 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x20c8, 0xf), /* ITLB_MISS_RETIRED */
1362 	INTEL_FLAGS_EVENT_CONSTRAINT(0xcb, 0xf),    /* MEM_LOAD_RETIRED.* */
1363 	INTEL_FLAGS_EVENT_CONSTRAINT(0xf7, 0xf),    /* FP_ASSIST.* */
1364 	/* INST_RETIRED.ANY_P, inv=1, cmask=16 (cycles:p). */
1365 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108000c0, 0x0f),
1366 	EVENT_CONSTRAINT_END
1367 };
1368 
1369 struct event_constraint intel_snb_pebs_event_constraints[] = {
1370 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x01c0, 0x2), /* INST_RETIRED.PRECDIST */
1371 	INTEL_PLD_CONSTRAINT(0x01cd, 0x8),    /* MEM_TRANS_RETIRED.LAT_ABOVE_THR */
1372 	INTEL_PST_CONSTRAINT(0x02cd, 0x8),    /* MEM_TRANS_RETIRED.PRECISE_STORES */
1373 	/* UOPS_RETIRED.ALL, inv=1, cmask=16 (cycles:p). */
1374 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c2, 0xf),
1375         INTEL_EXCLEVT_CONSTRAINT(0xd0, 0xf),    /* MEM_UOP_RETIRED.* */
1376         INTEL_EXCLEVT_CONSTRAINT(0xd1, 0xf),    /* MEM_LOAD_UOPS_RETIRED.* */
1377         INTEL_EXCLEVT_CONSTRAINT(0xd2, 0xf),    /* MEM_LOAD_UOPS_LLC_HIT_RETIRED.* */
1378         INTEL_EXCLEVT_CONSTRAINT(0xd3, 0xf),    /* MEM_LOAD_UOPS_LLC_MISS_RETIRED.* */
1379 	/* Allow all events as PEBS with no flags */
1380 	INTEL_ALL_EVENT_CONSTRAINT(0, 0xf),
1381 	EVENT_CONSTRAINT_END
1382 };
1383 
1384 struct event_constraint intel_ivb_pebs_event_constraints[] = {
1385         INTEL_FLAGS_UEVENT_CONSTRAINT(0x01c0, 0x2), /* INST_RETIRED.PRECDIST */
1386         INTEL_PLD_CONSTRAINT(0x01cd, 0x8),    /* MEM_TRANS_RETIRED.LAT_ABOVE_THR */
1387 	INTEL_PST_CONSTRAINT(0x02cd, 0x8),    /* MEM_TRANS_RETIRED.PRECISE_STORES */
1388 	/* UOPS_RETIRED.ALL, inv=1, cmask=16 (cycles:p). */
1389 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c2, 0xf),
1390 	/* INST_RETIRED.PREC_DIST, inv=1, cmask=16 (cycles:ppp). */
1391 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c0, 0x2),
1392 	INTEL_EXCLEVT_CONSTRAINT(0xd0, 0xf),    /* MEM_UOP_RETIRED.* */
1393 	INTEL_EXCLEVT_CONSTRAINT(0xd1, 0xf),    /* MEM_LOAD_UOPS_RETIRED.* */
1394 	INTEL_EXCLEVT_CONSTRAINT(0xd2, 0xf),    /* MEM_LOAD_UOPS_LLC_HIT_RETIRED.* */
1395 	INTEL_EXCLEVT_CONSTRAINT(0xd3, 0xf),    /* MEM_LOAD_UOPS_LLC_MISS_RETIRED.* */
1396 	/* Allow all events as PEBS with no flags */
1397 	INTEL_ALL_EVENT_CONSTRAINT(0, 0xf),
1398         EVENT_CONSTRAINT_END
1399 };
1400 
1401 struct event_constraint intel_hsw_pebs_event_constraints[] = {
1402 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x01c0, 0x2), /* INST_RETIRED.PRECDIST */
1403 	INTEL_PLD_CONSTRAINT(0x01cd, 0xf),    /* MEM_TRANS_RETIRED.* */
1404 	/* UOPS_RETIRED.ALL, inv=1, cmask=16 (cycles:p). */
1405 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c2, 0xf),
1406 	/* INST_RETIRED.PREC_DIST, inv=1, cmask=16 (cycles:ppp). */
1407 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c0, 0x2),
1408 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_NA(0x01c2, 0xf), /* UOPS_RETIRED.ALL */
1409 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_XLD(0x11d0, 0xf), /* MEM_UOPS_RETIRED.STLB_MISS_LOADS */
1410 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_XLD(0x21d0, 0xf), /* MEM_UOPS_RETIRED.LOCK_LOADS */
1411 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_XLD(0x41d0, 0xf), /* MEM_UOPS_RETIRED.SPLIT_LOADS */
1412 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_XLD(0x81d0, 0xf), /* MEM_UOPS_RETIRED.ALL_LOADS */
1413 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_XST(0x12d0, 0xf), /* MEM_UOPS_RETIRED.STLB_MISS_STORES */
1414 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_XST(0x42d0, 0xf), /* MEM_UOPS_RETIRED.SPLIT_STORES */
1415 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_XST(0x82d0, 0xf), /* MEM_UOPS_RETIRED.ALL_STORES */
1416 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_XLD(0xd1, 0xf),    /* MEM_LOAD_UOPS_RETIRED.* */
1417 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_XLD(0xd2, 0xf),    /* MEM_LOAD_UOPS_L3_HIT_RETIRED.* */
1418 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_XLD(0xd3, 0xf),    /* MEM_LOAD_UOPS_L3_MISS_RETIRED.* */
1419 	/* Allow all events as PEBS with no flags */
1420 	INTEL_ALL_EVENT_CONSTRAINT(0, 0xf),
1421 	EVENT_CONSTRAINT_END
1422 };
1423 
1424 struct event_constraint intel_bdw_pebs_event_constraints[] = {
1425 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x01c0, 0x2), /* INST_RETIRED.PRECDIST */
1426 	INTEL_PLD_CONSTRAINT(0x01cd, 0xf),    /* MEM_TRANS_RETIRED.* */
1427 	/* UOPS_RETIRED.ALL, inv=1, cmask=16 (cycles:p). */
1428 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c2, 0xf),
1429 	/* INST_RETIRED.PREC_DIST, inv=1, cmask=16 (cycles:ppp). */
1430 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c0, 0x2),
1431 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_NA(0x01c2, 0xf), /* UOPS_RETIRED.ALL */
1432 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf), /* MEM_UOPS_RETIRED.STLB_MISS_LOADS */
1433 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf), /* MEM_UOPS_RETIRED.LOCK_LOADS */
1434 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf), /* MEM_UOPS_RETIRED.SPLIT_LOADS */
1435 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf), /* MEM_UOPS_RETIRED.ALL_LOADS */
1436 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf), /* MEM_UOPS_RETIRED.STLB_MISS_STORES */
1437 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf), /* MEM_UOPS_RETIRED.SPLIT_STORES */
1438 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf), /* MEM_UOPS_RETIRED.ALL_STORES */
1439 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD(0xd1, 0xf),    /* MEM_LOAD_UOPS_RETIRED.* */
1440 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD(0xd2, 0xf),    /* MEM_LOAD_UOPS_L3_HIT_RETIRED.* */
1441 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD(0xd3, 0xf),    /* MEM_LOAD_UOPS_L3_MISS_RETIRED.* */
1442 	/* Allow all events as PEBS with no flags */
1443 	INTEL_ALL_EVENT_CONSTRAINT(0, 0xf),
1444 	EVENT_CONSTRAINT_END
1445 };
1446 
1447 
1448 struct event_constraint intel_skl_pebs_event_constraints[] = {
1449 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x1c0, 0x2),	/* INST_RETIRED.PREC_DIST */
1450 	/* INST_RETIRED.PREC_DIST, inv=1, cmask=16 (cycles:ppp). */
1451 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108001c0, 0x2),
1452 	/* INST_RETIRED.TOTAL_CYCLES_PS (inv=1, cmask=16) (cycles:p). */
1453 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x108000c0, 0x0f),
1454 	INTEL_PLD_CONSTRAINT(0x1cd, 0xf),		      /* MEM_TRANS_RETIRED.* */
1455 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_LOADS */
1456 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_STORES */
1457 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf), /* MEM_INST_RETIRED.LOCK_LOADS */
1458 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x22d0, 0xf), /* MEM_INST_RETIRED.LOCK_STORES */
1459 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf), /* MEM_INST_RETIRED.SPLIT_LOADS */
1460 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf), /* MEM_INST_RETIRED.SPLIT_STORES */
1461 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf), /* MEM_INST_RETIRED.ALL_LOADS */
1462 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf), /* MEM_INST_RETIRED.ALL_STORES */
1463 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD(0xd1, 0xf),    /* MEM_LOAD_RETIRED.* */
1464 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD(0xd2, 0xf),    /* MEM_LOAD_L3_HIT_RETIRED.* */
1465 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD(0xd3, 0xf),    /* MEM_LOAD_L3_MISS_RETIRED.* */
1466 	/* Allow all events as PEBS with no flags */
1467 	INTEL_ALL_EVENT_CONSTRAINT(0, 0xf),
1468 	EVENT_CONSTRAINT_END
1469 };
1470 
1471 struct event_constraint intel_icl_pebs_event_constraints[] = {
1472 	INTEL_PLD_CONSTRAINT(0x1cd, 0xff),			/* MEM_TRANS_RETIRED.LOAD_LATENCY */
1473 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf),	/* MEM_INST_RETIRED.STLB_MISS_LOADS */
1474 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf),	/* MEM_INST_RETIRED.STLB_MISS_STORES */
1475 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf),	/* MEM_INST_RETIRED.LOCK_LOADS */
1476 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf),	/* MEM_INST_RETIRED.SPLIT_LOADS */
1477 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf),	/* MEM_INST_RETIRED.SPLIT_STORES */
1478 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf),	/* MEM_INST_RETIRED.ALL_LOADS */
1479 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf),	/* MEM_INST_RETIRED.ALL_STORES */
1480 
1481 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD_RANGE(0xd1, 0xd4, 0xf), /* MEM_LOAD_*_RETIRED.* */
1482 
1483 	INTEL_FLAGS_EVENT_CONSTRAINT(0xd0, 0xf),		/* MEM_INST_RETIRED.* */
1484 
1485 	/*
1486 	 * Everything else is handled by PMU_FL_PEBS_ALL, because we
1487 	 * need the full constraints from the main table.
1488 	 */
1489 
1490 	EVENT_CONSTRAINT_END
1491 };
1492 
1493 struct event_constraint intel_glc_pebs_event_constraints[] = {
1494 	INTEL_FLAGS_EVENT_CONSTRAINT(0xc0, 0xfe),
1495 	INTEL_PLD_CONSTRAINT(0x1cd, 0xfe),
1496 	INTEL_PSD_CONSTRAINT(0x2cd, 0x1),
1497 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf),	/* MEM_INST_RETIRED.STLB_MISS_LOADS */
1498 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf),	/* MEM_INST_RETIRED.STLB_MISS_STORES */
1499 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf),	/* MEM_INST_RETIRED.LOCK_LOADS */
1500 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf),	/* MEM_INST_RETIRED.SPLIT_LOADS */
1501 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf),	/* MEM_INST_RETIRED.SPLIT_STORES */
1502 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf),	/* MEM_INST_RETIRED.ALL_LOADS */
1503 	INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf),	/* MEM_INST_RETIRED.ALL_STORES */
1504 
1505 	INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD_RANGE(0xd1, 0xd4, 0xf),
1506 
1507 	INTEL_FLAGS_EVENT_CONSTRAINT(0xd0, 0xf),
1508 
1509 	/*
1510 	 * Everything else is handled by PMU_FL_PEBS_ALL, because we
1511 	 * need the full constraints from the main table.
1512 	 */
1513 
1514 	EVENT_CONSTRAINT_END
1515 };
1516 
1517 struct event_constraint intel_lnc_pebs_event_constraints[] = {
1518 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x012a, 0x1),		/* OCR.* events */
1519 	INTEL_FLAGS_UEVENT_CONSTRAINT(0x012b, 0x1),		/* OCR.* events */
1520 
1521 	INTEL_HYBRID_LDLAT_CONSTRAINT(0x1cd, 0x3fc),
1522 	INTEL_HYBRID_STLAT_CONSTRAINT(0x2cd, 0x3),
1523 
1524 	/*
1525 	 * Everything else is handled by PMU_FL_PEBS_ALL, because we
1526 	 * need the full constraints from the main table.
1527 	 */
1528 
1529 	EVENT_CONSTRAINT_END
1530 };
1531 
1532 struct event_constraint intel_pnc_pebs_event_constraints[] = {
1533 	INTEL_HYBRID_LDLAT_CONSTRAINT(0x1cd, 0xfc),
1534 	INTEL_HYBRID_STLAT_CONSTRAINT(0x2cd, 0x3),
1535 
1536 	/*
1537 	 * Everything else is handled by PMU_FL_PEBS_ALL, because we
1538 	 * need the full constraints from the main table.
1539 	 */
1540 
1541 	EVENT_CONSTRAINT_END
1542 };
1543 
1544 struct event_constraint *intel_pebs_constraints(struct perf_event *event)
1545 {
1546 	struct event_constraint *pebs_constraints = hybrid(event->pmu, pebs_constraints);
1547 	struct event_constraint *c;
1548 
1549 	if (!event->attr.precise_ip)
1550 		return NULL;
1551 
1552 	if (pebs_constraints) {
1553 		for_each_event_constraint(c, pebs_constraints) {
1554 			if (constraint_match(c, event->hw.config)) {
1555 				event->hw.flags |= c->flags;
1556 				return c;
1557 			}
1558 		}
1559 	}
1560 
1561 	/*
1562 	 * Extended PEBS support
1563 	 * Makes the PEBS code search the normal constraints.
1564 	 */
1565 	if (x86_pmu.flags & PMU_FL_PEBS_ALL)
1566 		return NULL;
1567 
1568 	return &emptyconstraint;
1569 }
1570 
1571 /*
1572  * We need the sched_task callback even for per-cpu events when we use
1573  * the large interrupt threshold, such that we can provide PID and TID
1574  * to PEBS samples.
1575  */
1576 static inline bool pebs_needs_sched_cb(struct cpu_hw_events *cpuc)
1577 {
1578 	if (cpuc->n_pebs == cpuc->n_pebs_via_pt)
1579 		return false;
1580 
1581 	return cpuc->n_pebs && (cpuc->n_pebs == cpuc->n_large_pebs);
1582 }
1583 
1584 void intel_pmu_pebs_sched_task(struct perf_event_pmu_context *pmu_ctx, bool sched_in)
1585 {
1586 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1587 
1588 	if (!sched_in && pebs_needs_sched_cb(cpuc))
1589 		intel_pmu_drain_pebs_buffer();
1590 }
1591 
1592 static inline void pebs_update_threshold(struct cpu_hw_events *cpuc)
1593 {
1594 	struct debug_store *ds = cpuc->ds;
1595 	int max_pebs_events = intel_pmu_max_num_pebs(cpuc->pmu);
1596 	u64 threshold;
1597 	int reserved;
1598 
1599 	if (cpuc->n_pebs_via_pt)
1600 		return;
1601 
1602 	if (x86_pmu.flags & PMU_FL_PEBS_ALL)
1603 		reserved = max_pebs_events + x86_pmu_max_num_counters_fixed(cpuc->pmu);
1604 	else
1605 		reserved = max_pebs_events;
1606 
1607 	if (cpuc->n_pebs == cpuc->n_large_pebs) {
1608 		threshold = ds->pebs_absolute_maximum -
1609 			reserved * cpuc->pebs_record_size;
1610 	} else {
1611 		threshold = ds->pebs_buffer_base + cpuc->pebs_record_size;
1612 	}
1613 
1614 	ds->pebs_interrupt_threshold = threshold;
1615 }
1616 
1617 #define PEBS_DATACFG_CNTRS(x)						\
1618 	((x >> PEBS_DATACFG_CNTR_SHIFT) & PEBS_DATACFG_CNTR_MASK)
1619 
1620 #define PEBS_DATACFG_CNTR_BIT(x)					\
1621 	(((1ULL << x) & PEBS_DATACFG_CNTR_MASK) << PEBS_DATACFG_CNTR_SHIFT)
1622 
1623 #define PEBS_DATACFG_FIX(x)						\
1624 	((x >> PEBS_DATACFG_FIX_SHIFT) & PEBS_DATACFG_FIX_MASK)
1625 
1626 #define PEBS_DATACFG_FIX_BIT(x)						\
1627 	(((1ULL << (x)) & PEBS_DATACFG_FIX_MASK)			\
1628 	 << PEBS_DATACFG_FIX_SHIFT)
1629 
1630 static void adaptive_pebs_record_size_update(void)
1631 {
1632 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1633 	u64 pebs_data_cfg = cpuc->pebs_data_cfg;
1634 	int sz = sizeof(struct pebs_basic);
1635 
1636 	if (pebs_data_cfg & PEBS_DATACFG_MEMINFO)
1637 		sz += sizeof(struct pebs_meminfo);
1638 	if (pebs_data_cfg & PEBS_DATACFG_GP)
1639 		sz += sizeof(struct pebs_gprs);
1640 	if (pebs_data_cfg & PEBS_DATACFG_XMMS)
1641 		sz += sizeof(struct pebs_xmm);
1642 	if (pebs_data_cfg & PEBS_DATACFG_LBRS)
1643 		sz += x86_pmu.lbr_nr * sizeof(struct lbr_entry);
1644 	if (pebs_data_cfg & (PEBS_DATACFG_METRICS | PEBS_DATACFG_CNTR)) {
1645 		sz += sizeof(struct pebs_cntr_header);
1646 
1647 		/* Metrics base and Metrics Data */
1648 		if (pebs_data_cfg & PEBS_DATACFG_METRICS)
1649 			sz += 2 * sizeof(u64);
1650 
1651 		if (pebs_data_cfg & PEBS_DATACFG_CNTR) {
1652 			sz += (hweight64(PEBS_DATACFG_CNTRS(pebs_data_cfg)) +
1653 			       hweight64(PEBS_DATACFG_FIX(pebs_data_cfg))) *
1654 			      sizeof(u64);
1655 		}
1656 	}
1657 
1658 	cpuc->pebs_record_size = sz;
1659 }
1660 
1661 static void __intel_pmu_pebs_update_cfg(struct perf_event *event,
1662 					int idx, u64 *pebs_data_cfg)
1663 {
1664 	if (is_metric_event(event)) {
1665 		*pebs_data_cfg |= PEBS_DATACFG_METRICS;
1666 		return;
1667 	}
1668 
1669 	*pebs_data_cfg |= PEBS_DATACFG_CNTR;
1670 
1671 	if (idx >= INTEL_PMC_IDX_FIXED)
1672 		*pebs_data_cfg |= PEBS_DATACFG_FIX_BIT(idx - INTEL_PMC_IDX_FIXED);
1673 	else
1674 		*pebs_data_cfg |= PEBS_DATACFG_CNTR_BIT(idx);
1675 }
1676 
1677 
1678 void intel_pmu_pebs_late_setup(struct cpu_hw_events *cpuc)
1679 {
1680 	struct perf_event *event;
1681 	u64 pebs_data_cfg = 0;
1682 	int i;
1683 
1684 	for (i = 0; i < cpuc->n_events; i++) {
1685 		event = cpuc->event_list[i];
1686 		if (!is_pebs_counter_event_group(event))
1687 			continue;
1688 		__intel_pmu_pebs_update_cfg(event, cpuc->assign[i], &pebs_data_cfg);
1689 	}
1690 
1691 	if (pebs_data_cfg & ~cpuc->pebs_data_cfg)
1692 		cpuc->pebs_data_cfg |= pebs_data_cfg | PEBS_UPDATE_DS_SW;
1693 }
1694 
1695 #define PERF_PEBS_MEMINFO_TYPE	(PERF_SAMPLE_ADDR | PERF_SAMPLE_DATA_SRC |   \
1696 				PERF_SAMPLE_PHYS_ADDR |			     \
1697 				PERF_SAMPLE_WEIGHT_TYPE |		     \
1698 				PERF_SAMPLE_TRANSACTION |		     \
1699 				PERF_SAMPLE_DATA_PAGE_SIZE)
1700 
1701 static u64 pebs_update_adaptive_cfg(struct perf_event *event)
1702 {
1703 	struct perf_event_attr *attr = &event->attr;
1704 	u64 sample_type = attr->sample_type;
1705 	u64 pebs_data_cfg = 0;
1706 	bool gprs, tsx_weight;
1707 
1708 	if (!(sample_type & ~(PERF_SAMPLE_IP|PERF_SAMPLE_TIME)) &&
1709 	    attr->precise_ip > 1)
1710 		return pebs_data_cfg;
1711 
1712 	if (sample_type & PERF_PEBS_MEMINFO_TYPE)
1713 		pebs_data_cfg |= PEBS_DATACFG_MEMINFO;
1714 
1715 	/*
1716 	 * We need GPRs when:
1717 	 * + user requested them
1718 	 * + precise_ip < 2 for the non event IP
1719 	 * + For RTM TSX weight we need GPRs for the abort code.
1720 	 */
1721 	gprs = ((sample_type & PERF_SAMPLE_REGS_INTR) &&
1722 		(attr->sample_regs_intr & PEBS_GP_REGS)) ||
1723 	       ((sample_type & PERF_SAMPLE_REGS_USER) &&
1724 		(attr->sample_regs_user & PEBS_GP_REGS));
1725 
1726 	tsx_weight = (sample_type & PERF_SAMPLE_WEIGHT_TYPE) &&
1727 		     ((attr->config & INTEL_ARCH_EVENT_MASK) ==
1728 		      x86_pmu.rtm_abort_event);
1729 
1730 	if (gprs || (attr->precise_ip < 2) || tsx_weight)
1731 		pebs_data_cfg |= PEBS_DATACFG_GP;
1732 
1733 	if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
1734 	    (attr->sample_regs_intr & PERF_REG_EXTENDED_MASK))
1735 		pebs_data_cfg |= PEBS_DATACFG_XMMS;
1736 
1737 	if (sample_type & PERF_SAMPLE_BRANCH_STACK) {
1738 		/*
1739 		 * For now always log all LBRs. Could configure this
1740 		 * later.
1741 		 */
1742 		pebs_data_cfg |= PEBS_DATACFG_LBRS |
1743 			((x86_pmu.lbr_nr-1) << PEBS_DATACFG_LBR_SHIFT);
1744 	}
1745 
1746 	return pebs_data_cfg;
1747 }
1748 
1749 static void
1750 pebs_update_state(bool needed_cb, struct cpu_hw_events *cpuc,
1751 		  struct perf_event *event, bool add)
1752 {
1753 	struct pmu *pmu = event->pmu;
1754 
1755 	/*
1756 	 * Make sure we get updated with the first PEBS event.
1757 	 * During removal, ->pebs_data_cfg is still valid for
1758 	 * the last PEBS event. Don't clear it.
1759 	 */
1760 	if ((cpuc->n_pebs == 1) && add)
1761 		cpuc->pebs_data_cfg = PEBS_UPDATE_DS_SW;
1762 
1763 	if (needed_cb != pebs_needs_sched_cb(cpuc)) {
1764 		if (!needed_cb)
1765 			perf_sched_cb_inc(pmu);
1766 		else
1767 			perf_sched_cb_dec(pmu);
1768 
1769 		cpuc->pebs_data_cfg |= PEBS_UPDATE_DS_SW;
1770 	}
1771 
1772 	/*
1773 	 * The PEBS record doesn't shrink on pmu::del(). Doing so would require
1774 	 * iterating all remaining PEBS events to reconstruct the config.
1775 	 */
1776 	if (x86_pmu.intel_cap.pebs_baseline && add) {
1777 		u64 pebs_data_cfg;
1778 
1779 		pebs_data_cfg = pebs_update_adaptive_cfg(event);
1780 		/*
1781 		 * Be sure to update the thresholds when we change the record.
1782 		 */
1783 		if (pebs_data_cfg & ~cpuc->pebs_data_cfg)
1784 			cpuc->pebs_data_cfg |= pebs_data_cfg | PEBS_UPDATE_DS_SW;
1785 	}
1786 }
1787 
1788 u64 intel_get_arch_pebs_data_config(struct perf_event *event)
1789 {
1790 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1791 	u64 pebs_data_cfg = 0;
1792 	u64 cntr_mask;
1793 
1794 	if (WARN_ON(event->hw.idx < 0 || event->hw.idx >= X86_PMC_IDX_MAX))
1795 		return 0;
1796 
1797 	pebs_data_cfg |= pebs_update_adaptive_cfg(event);
1798 
1799 	cntr_mask = (PEBS_DATACFG_CNTR_MASK << PEBS_DATACFG_CNTR_SHIFT) |
1800 		    (PEBS_DATACFG_FIX_MASK << PEBS_DATACFG_FIX_SHIFT) |
1801 		    PEBS_DATACFG_CNTR | PEBS_DATACFG_METRICS;
1802 	pebs_data_cfg |= cpuc->pebs_data_cfg & cntr_mask;
1803 
1804 	return pebs_data_cfg;
1805 }
1806 
1807 void intel_pmu_pebs_add(struct perf_event *event)
1808 {
1809 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1810 	struct hw_perf_event *hwc = &event->hw;
1811 	bool needed_cb = pebs_needs_sched_cb(cpuc);
1812 
1813 	cpuc->n_pebs++;
1814 	if (hwc->flags & PERF_X86_EVENT_LARGE_PEBS)
1815 		cpuc->n_large_pebs++;
1816 	if (hwc->flags & PERF_X86_EVENT_PEBS_VIA_PT)
1817 		cpuc->n_pebs_via_pt++;
1818 
1819 	pebs_update_state(needed_cb, cpuc, event, true);
1820 }
1821 
1822 static void intel_pmu_pebs_via_pt_disable(struct perf_event *event)
1823 {
1824 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1825 
1826 	if (!is_pebs_pt(event))
1827 		return;
1828 
1829 	if (!(cpuc->pebs_enabled & ~PEBS_VIA_PT_MASK))
1830 		cpuc->pebs_enabled &= ~PEBS_VIA_PT_MASK;
1831 }
1832 
1833 static void intel_pmu_pebs_via_pt_enable(struct perf_event *event)
1834 {
1835 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1836 	struct hw_perf_event *hwc = &event->hw;
1837 	struct debug_store *ds = cpuc->ds;
1838 	u64 value = ds->pebs_event_reset[hwc->idx];
1839 	u32 base = MSR_RELOAD_PMC0;
1840 	unsigned int idx = hwc->idx;
1841 
1842 	if (!is_pebs_pt(event))
1843 		return;
1844 
1845 	if (!(event->hw.flags & PERF_X86_EVENT_LARGE_PEBS))
1846 		cpuc->pebs_enabled |= PEBS_PMI_AFTER_EACH_RECORD;
1847 
1848 	cpuc->pebs_enabled |= PEBS_OUTPUT_PT;
1849 
1850 	if (hwc->idx >= INTEL_PMC_IDX_FIXED) {
1851 		base = MSR_RELOAD_FIXED_CTR0;
1852 		idx = hwc->idx - INTEL_PMC_IDX_FIXED;
1853 		if (x86_pmu.intel_cap.pebs_format < 5)
1854 			value = ds->pebs_event_reset[MAX_PEBS_EVENTS_FMT4 + idx];
1855 		else
1856 			value = ds->pebs_event_reset[MAX_PEBS_EVENTS + idx];
1857 	}
1858 	wrmsrq(base + idx, value);
1859 }
1860 
1861 static inline void intel_pmu_drain_large_pebs(struct cpu_hw_events *cpuc)
1862 {
1863 	if (cpuc->n_pebs == cpuc->n_large_pebs &&
1864 	    cpuc->n_pebs != cpuc->n_pebs_via_pt) {
1865 		int enabled = __intel_pmu_quiesce();
1866 		intel_pmu_drain_pebs_buffer();
1867 		__intel_pmu_resume(enabled);
1868 	}
1869 }
1870 
1871 static void __intel_pmu_pebs_enable(struct perf_event *event)
1872 {
1873 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1874 	struct hw_perf_event *hwc = &event->hw;
1875 
1876 	hwc->config &= ~ARCH_PERFMON_EVENTSEL_INT;
1877 	cpuc->pebs_enabled |= 1ULL << hwc->idx;
1878 }
1879 
1880 void intel_pmu_pebs_enable(struct perf_event *event)
1881 {
1882 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1883 	u64 pebs_data_cfg = cpuc->pebs_data_cfg & ~PEBS_UPDATE_DS_SW;
1884 	struct hw_perf_event *hwc = &event->hw;
1885 	struct debug_store *ds = cpuc->ds;
1886 	unsigned int idx = hwc->idx;
1887 
1888 	__intel_pmu_pebs_enable(event);
1889 
1890 	if ((event->hw.flags & PERF_X86_EVENT_PEBS_LDLAT) && (x86_pmu.version < 5))
1891 		cpuc->pebs_enabled |= 1ULL << (hwc->idx + 32);
1892 	else if (event->hw.flags & PERF_X86_EVENT_PEBS_ST)
1893 		cpuc->pebs_enabled |= 1ULL << 63;
1894 
1895 	if (x86_pmu.intel_cap.pebs_baseline) {
1896 		hwc->config |= ICL_EVENTSEL_ADAPTIVE;
1897 		if (pebs_data_cfg != cpuc->active_pebs_data_cfg) {
1898 			/*
1899 			 * drain_pebs() assumes uniform record size;
1900 			 * hence we need to drain when changing said
1901 			 * size.
1902 			 */
1903 			intel_pmu_drain_pebs_buffer();
1904 			adaptive_pebs_record_size_update();
1905 			wrmsrq(MSR_PEBS_DATA_CFG, pebs_data_cfg);
1906 			cpuc->active_pebs_data_cfg = pebs_data_cfg;
1907 		}
1908 	}
1909 	if (cpuc->pebs_data_cfg & PEBS_UPDATE_DS_SW) {
1910 		cpuc->pebs_data_cfg = pebs_data_cfg;
1911 		pebs_update_threshold(cpuc);
1912 	}
1913 
1914 	if (idx >= INTEL_PMC_IDX_FIXED) {
1915 		if (x86_pmu.intel_cap.pebs_format < 5)
1916 			idx = MAX_PEBS_EVENTS_FMT4 + (idx - INTEL_PMC_IDX_FIXED);
1917 		else
1918 			idx = MAX_PEBS_EVENTS + (idx - INTEL_PMC_IDX_FIXED);
1919 	}
1920 
1921 	/*
1922 	 * Use auto-reload if possible to save a MSR write in the PMI.
1923 	 * This must be done in pmu::start(), because PERF_EVENT_IOC_PERIOD.
1924 	 */
1925 	if (hwc->flags & PERF_X86_EVENT_AUTO_RELOAD) {
1926 		ds->pebs_event_reset[idx] =
1927 			(u64)(-hwc->sample_period) & x86_pmu.cntval_mask;
1928 	} else {
1929 		ds->pebs_event_reset[idx] = 0;
1930 	}
1931 
1932 	intel_pmu_pebs_via_pt_enable(event);
1933 }
1934 
1935 void intel_pmu_pebs_del(struct perf_event *event)
1936 {
1937 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1938 	struct hw_perf_event *hwc = &event->hw;
1939 	bool needed_cb = pebs_needs_sched_cb(cpuc);
1940 
1941 	cpuc->n_pebs--;
1942 	if (hwc->flags & PERF_X86_EVENT_LARGE_PEBS)
1943 		cpuc->n_large_pebs--;
1944 	if (hwc->flags & PERF_X86_EVENT_PEBS_VIA_PT)
1945 		cpuc->n_pebs_via_pt--;
1946 
1947 	pebs_update_state(needed_cb, cpuc, event, false);
1948 }
1949 
1950 static void __intel_pmu_pebs_disable(struct perf_event *event)
1951 {
1952 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1953 	struct hw_perf_event *hwc = &event->hw;
1954 
1955 	intel_pmu_drain_large_pebs(cpuc);
1956 	cpuc->pebs_enabled &= ~(1ULL << hwc->idx);
1957 	hwc->config |= ARCH_PERFMON_EVENTSEL_INT;
1958 }
1959 
1960 void intel_pmu_pebs_disable(struct perf_event *event)
1961 {
1962 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1963 	struct hw_perf_event *hwc = &event->hw;
1964 
1965 	__intel_pmu_pebs_disable(event);
1966 
1967 	if ((event->hw.flags & PERF_X86_EVENT_PEBS_LDLAT) &&
1968 	    (x86_pmu.version < 5))
1969 		cpuc->pebs_enabled &= ~(1ULL << (hwc->idx + 32));
1970 	else if (event->hw.flags & PERF_X86_EVENT_PEBS_ST)
1971 		cpuc->pebs_enabled &= ~(1ULL << 63);
1972 
1973 	intel_pmu_pebs_via_pt_disable(event);
1974 
1975 	if (cpuc->enabled)
1976 		wrmsrq(MSR_IA32_PEBS_ENABLE, cpuc->pebs_enabled);
1977 }
1978 
1979 void intel_pmu_pebs_enable_all(void)
1980 {
1981 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1982 
1983 	if (cpuc->pebs_enabled)
1984 		wrmsrq(MSR_IA32_PEBS_ENABLE, cpuc->pebs_enabled);
1985 }
1986 
1987 void intel_pmu_pebs_disable_all(void)
1988 {
1989 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1990 
1991 	if (cpuc->pebs_enabled)
1992 		__intel_pmu_pebs_disable_all();
1993 }
1994 
1995 static int intel_pmu_pebs_fixup_ip(struct pt_regs *regs)
1996 {
1997 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
1998 	unsigned long from = cpuc->lbr_entries[0].from;
1999 	unsigned long old_to, to = cpuc->lbr_entries[0].to;
2000 	unsigned long ip = regs->ip;
2001 	int is_64bit = 0;
2002 	void *kaddr;
2003 	int size;
2004 
2005 	/*
2006 	 * We don't need to fixup if the PEBS assist is fault like
2007 	 */
2008 	if (!x86_pmu.intel_cap.pebs_trap)
2009 		return 1;
2010 
2011 	/*
2012 	 * No LBR entry, no basic block, no rewinding
2013 	 */
2014 	if (!cpuc->lbr_stack.nr || !from || !to)
2015 		return 0;
2016 
2017 	/*
2018 	 * Basic blocks should never cross user/kernel boundaries
2019 	 */
2020 	if (kernel_ip(ip) != kernel_ip(to))
2021 		return 0;
2022 
2023 	/*
2024 	 * unsigned math, either ip is before the start (impossible) or
2025 	 * the basic block is larger than 1 page (sanity)
2026 	 */
2027 	if ((ip - to) > PEBS_FIXUP_SIZE)
2028 		return 0;
2029 
2030 	/*
2031 	 * We sampled a branch insn, rewind using the LBR stack
2032 	 */
2033 	if (ip == to) {
2034 		set_linear_ip(regs, from);
2035 		return 1;
2036 	}
2037 
2038 	size = ip - to;
2039 	if (!kernel_ip(ip)) {
2040 		int bytes;
2041 		u8 *buf = this_cpu_read(insn_buffer);
2042 
2043 		/* 'size' must fit our buffer, see above */
2044 		bytes = copy_from_user_nmi(buf, (void __user *)to, size);
2045 		if (bytes != 0)
2046 			return 0;
2047 
2048 		kaddr = buf;
2049 	} else {
2050 		kaddr = (void *)to;
2051 	}
2052 
2053 	do {
2054 		struct insn insn;
2055 
2056 		old_to = to;
2057 
2058 #ifdef CONFIG_X86_64
2059 		is_64bit = kernel_ip(to) || any_64bit_mode(regs);
2060 #endif
2061 		insn_init(&insn, kaddr, size, is_64bit);
2062 
2063 		/*
2064 		 * Make sure there was not a problem decoding the instruction.
2065 		 * This is doubly important because we have an infinite loop if
2066 		 * insn.length=0.
2067 		 */
2068 		if (insn_get_length(&insn))
2069 			break;
2070 
2071 		to += insn.length;
2072 		kaddr += insn.length;
2073 		size -= insn.length;
2074 	} while (to < ip);
2075 
2076 	if (to == ip) {
2077 		set_linear_ip(regs, old_to);
2078 		return 1;
2079 	}
2080 
2081 	/*
2082 	 * Even though we decoded the basic block, the instruction stream
2083 	 * never matched the given IP, either the TO or the IP got corrupted.
2084 	 */
2085 	return 0;
2086 }
2087 
2088 static inline u64 intel_get_tsx_weight(u64 tsx_tuning)
2089 {
2090 	if (tsx_tuning) {
2091 		union hsw_tsx_tuning tsx = { .value = tsx_tuning };
2092 		return tsx.cycles_last_block;
2093 	}
2094 	return 0;
2095 }
2096 
2097 static inline u64 intel_get_tsx_transaction(u64 tsx_tuning, u64 ax)
2098 {
2099 	u64 txn = (tsx_tuning & PEBS_HSW_TSX_FLAGS) >> 32;
2100 
2101 	/* For RTM XABORTs also log the abort code from AX */
2102 	if ((txn & PERF_TXN_TRANSACTION) && (ax & 1))
2103 		txn |= ((ax >> 24) & 0xff) << PERF_TXN_ABORT_SHIFT;
2104 	return txn;
2105 }
2106 
2107 static inline u64 get_pebs_status(void *n)
2108 {
2109 	if (x86_pmu.intel_cap.pebs_format < 4)
2110 		return ((struct pebs_record_nhm *)n)->status;
2111 	return ((struct pebs_basic *)n)->applicable_counters;
2112 }
2113 
2114 #define PERF_X86_EVENT_PEBS_HSW_PREC \
2115 		(PERF_X86_EVENT_PEBS_ST_HSW | \
2116 		 PERF_X86_EVENT_PEBS_LD_HSW | \
2117 		 PERF_X86_EVENT_PEBS_NA_HSW)
2118 
2119 static u64 get_data_src(struct perf_event *event, u64 aux)
2120 {
2121 	u64 val = PERF_MEM_NA;
2122 	int fl = event->hw.flags;
2123 	bool fst = fl & (PERF_X86_EVENT_PEBS_ST | PERF_X86_EVENT_PEBS_HSW_PREC);
2124 
2125 	if (fl & PERF_X86_EVENT_PEBS_LDLAT)
2126 		val = load_latency_data(event, aux);
2127 	else if (fl & PERF_X86_EVENT_PEBS_STLAT)
2128 		val = store_latency_data(event, aux);
2129 	else if (fl & PERF_X86_EVENT_PEBS_LAT_HYBRID)
2130 		val = x86_pmu.pebs_latency_data(event, aux);
2131 	else if (fst && (fl & PERF_X86_EVENT_PEBS_HSW_PREC))
2132 		val = precise_datala_hsw(event, aux);
2133 	else if (fst)
2134 		val = precise_store_data(aux);
2135 	return val;
2136 }
2137 
2138 static void setup_pebs_time(struct perf_event *event,
2139 			    struct perf_sample_data *data,
2140 			    u64 tsc)
2141 {
2142 	/* Converting to a user-defined clock is not supported yet. */
2143 	if (event->attr.use_clockid != 0)
2144 		return;
2145 
2146 	/*
2147 	 * Doesn't support the conversion when the TSC is unstable.
2148 	 * The TSC unstable case is a corner case and very unlikely to
2149 	 * happen. If it happens, the TSC in a PEBS record will be
2150 	 * dropped and fall back to perf_event_clock().
2151 	 */
2152 	if (!using_native_sched_clock() || !sched_clock_stable())
2153 		return;
2154 
2155 	data->time = native_sched_clock_from_tsc(tsc) + __sched_clock_offset;
2156 	data->sample_flags |= PERF_SAMPLE_TIME;
2157 }
2158 
2159 #define PERF_SAMPLE_ADDR_TYPE	(PERF_SAMPLE_ADDR |		\
2160 				 PERF_SAMPLE_PHYS_ADDR |	\
2161 				 PERF_SAMPLE_DATA_PAGE_SIZE)
2162 
2163 static void setup_pebs_fixed_sample_data(struct perf_event *event,
2164 				   struct pt_regs *iregs, void *__pebs,
2165 				   struct perf_sample_data *data,
2166 				   struct pt_regs *regs)
2167 {
2168 	/*
2169 	 * We cast to the biggest pebs_record but are careful not to
2170 	 * unconditionally access the 'extra' entries.
2171 	 */
2172 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
2173 	struct pebs_record_skl *pebs = __pebs;
2174 	u64 sample_type;
2175 	int fll;
2176 
2177 	if (pebs == NULL)
2178 		return;
2179 
2180 	sample_type = event->attr.sample_type;
2181 	fll = event->hw.flags & PERF_X86_EVENT_PEBS_LDLAT;
2182 
2183 	perf_sample_data_init(data, 0, event->hw.last_period);
2184 
2185 	/*
2186 	 * Use latency for weight (only avail with PEBS-LL)
2187 	 */
2188 	if (fll && (sample_type & PERF_SAMPLE_WEIGHT_TYPE)) {
2189 		data->weight.full = pebs->lat;
2190 		data->sample_flags |= PERF_SAMPLE_WEIGHT_TYPE;
2191 	}
2192 
2193 	/*
2194 	 * data.data_src encodes the data source
2195 	 */
2196 	if (sample_type & PERF_SAMPLE_DATA_SRC) {
2197 		data->data_src.val = get_data_src(event, pebs->dse);
2198 		data->sample_flags |= PERF_SAMPLE_DATA_SRC;
2199 	}
2200 
2201 	/*
2202 	 * We must however always use iregs for the unwinder to stay sane; the
2203 	 * record BP,SP,IP can point into thin air when the record is from a
2204 	 * previous PMI context or an (I)RET happened between the record and
2205 	 * PMI.
2206 	 */
2207 	perf_sample_save_callchain(data, event, iregs);
2208 
2209 	/*
2210 	 * We use the interrupt regs as a base because the PEBS record does not
2211 	 * contain a full regs set, specifically it seems to lack segment
2212 	 * descriptors, which get used by things like user_mode().
2213 	 *
2214 	 * In the simple case fix up only the IP for PERF_SAMPLE_IP.
2215 	 */
2216 	*regs = *iregs;
2217 
2218 	/*
2219 	 * Initialize regs_>flags from PEBS,
2220 	 * Clear exact bit (which uses x86 EFLAGS Reserved bit 3),
2221 	 * i.e., do not rely on it being zero:
2222 	 */
2223 	regs->flags = pebs->flags & ~PERF_EFLAGS_EXACT;
2224 
2225 	if (sample_type & PERF_SAMPLE_REGS_INTR) {
2226 		regs->ax = pebs->ax;
2227 		regs->bx = pebs->bx;
2228 		regs->cx = pebs->cx;
2229 		regs->dx = pebs->dx;
2230 		regs->si = pebs->si;
2231 		regs->di = pebs->di;
2232 
2233 		regs->bp = pebs->bp;
2234 		regs->sp = pebs->sp;
2235 
2236 #ifndef CONFIG_X86_32
2237 		regs->r8 = pebs->r8;
2238 		regs->r9 = pebs->r9;
2239 		regs->r10 = pebs->r10;
2240 		regs->r11 = pebs->r11;
2241 		regs->r12 = pebs->r12;
2242 		regs->r13 = pebs->r13;
2243 		regs->r14 = pebs->r14;
2244 		regs->r15 = pebs->r15;
2245 #endif
2246 	}
2247 
2248 	if (event->attr.precise_ip > 1) {
2249 		/*
2250 		 * Haswell and later processors have an 'eventing IP'
2251 		 * (real IP) which fixes the off-by-1 skid in hardware.
2252 		 * Use it when precise_ip >= 2 :
2253 		 */
2254 		if (x86_pmu.intel_cap.pebs_format >= 2) {
2255 			set_linear_ip(regs, pebs->real_ip);
2256 			regs->flags |= PERF_EFLAGS_EXACT;
2257 		} else {
2258 			/* Otherwise, use PEBS off-by-1 IP: */
2259 			set_linear_ip(regs, pebs->ip);
2260 
2261 			/*
2262 			 * With precise_ip >= 2, try to fix up the off-by-1 IP
2263 			 * using the LBR. If successful, the fixup function
2264 			 * corrects regs->ip and calls set_linear_ip() on regs:
2265 			 */
2266 			if (intel_pmu_pebs_fixup_ip(regs))
2267 				regs->flags |= PERF_EFLAGS_EXACT;
2268 		}
2269 	} else {
2270 		/*
2271 		 * When precise_ip == 1, return the PEBS off-by-1 IP,
2272 		 * no fixup attempted:
2273 		 */
2274 		set_linear_ip(regs, pebs->ip);
2275 	}
2276 
2277 
2278 	if ((sample_type & PERF_SAMPLE_ADDR_TYPE) &&
2279 	    x86_pmu.intel_cap.pebs_format >= 1) {
2280 		data->addr = pebs->dla;
2281 		data->sample_flags |= PERF_SAMPLE_ADDR;
2282 	}
2283 
2284 	if (x86_pmu.intel_cap.pebs_format >= 2) {
2285 		/* Only set the TSX weight when no memory weight. */
2286 		if ((sample_type & PERF_SAMPLE_WEIGHT_TYPE) && !fll) {
2287 			data->weight.full = intel_get_tsx_weight(pebs->tsx_tuning);
2288 			data->sample_flags |= PERF_SAMPLE_WEIGHT_TYPE;
2289 		}
2290 		if (sample_type & PERF_SAMPLE_TRANSACTION) {
2291 			data->txn = intel_get_tsx_transaction(pebs->tsx_tuning,
2292 							      pebs->ax);
2293 			data->sample_flags |= PERF_SAMPLE_TRANSACTION;
2294 		}
2295 	}
2296 
2297 	/*
2298 	 * v3 supplies an accurate time stamp, so we use that
2299 	 * for the time stamp.
2300 	 *
2301 	 * We can only do this for the default trace clock.
2302 	 */
2303 	if (x86_pmu.intel_cap.pebs_format >= 3)
2304 		setup_pebs_time(event, data, pebs->tsc);
2305 
2306 	perf_sample_save_brstack(data, event, &cpuc->lbr_stack, NULL);
2307 }
2308 
2309 static void adaptive_pebs_save_regs(struct pt_regs *regs,
2310 				    struct pebs_gprs *gprs)
2311 {
2312 	regs->ax = gprs->ax;
2313 	regs->bx = gprs->bx;
2314 	regs->cx = gprs->cx;
2315 	regs->dx = gprs->dx;
2316 	regs->si = gprs->si;
2317 	regs->di = gprs->di;
2318 	regs->bp = gprs->bp;
2319 	regs->sp = gprs->sp;
2320 #ifndef CONFIG_X86_32
2321 	regs->r8 = gprs->r8;
2322 	regs->r9 = gprs->r9;
2323 	regs->r10 = gprs->r10;
2324 	regs->r11 = gprs->r11;
2325 	regs->r12 = gprs->r12;
2326 	regs->r13 = gprs->r13;
2327 	regs->r14 = gprs->r14;
2328 	regs->r15 = gprs->r15;
2329 #endif
2330 }
2331 
2332 static void intel_perf_event_update_pmc(struct perf_event *event, u64 pmc)
2333 {
2334 	int shift = 64 - x86_pmu.cntval_bits;
2335 	struct hw_perf_event *hwc;
2336 	u64 delta, prev_pmc;
2337 
2338 	/*
2339 	 * A recorded counter may not have an assigned event in the
2340 	 * following cases. The value should be dropped.
2341 	 * - An event is deleted. There is still an active PEBS event.
2342 	 *   The PEBS record doesn't shrink on pmu::del().
2343 	 *   If the counter of the deleted event once occurred in a PEBS
2344 	 *   record, PEBS still records the counter until the counter is
2345 	 *   reassigned.
2346 	 * - An event is stopped for some reason, e.g., throttled.
2347 	 *   During this period, another event is added and takes the
2348 	 *   counter of the stopped event. The stopped event is assigned
2349 	 *   to another new and uninitialized counter, since the
2350 	 *   x86_pmu_start(RELOAD) is not invoked for a stopped event.
2351 	 *   The PEBS__DATA_CFG is updated regardless of the event state.
2352 	 *   The uninitialized counter can be recorded in a PEBS record.
2353 	 *   But the cpuc->events[uninitialized_counter] is always NULL,
2354 	 *   because the event is stopped. The uninitialized value is
2355 	 *   safely dropped.
2356 	 */
2357 	if (!event)
2358 		return;
2359 
2360 	hwc = &event->hw;
2361 	prev_pmc = local64_read(&hwc->prev_count);
2362 
2363 	/* Only update the count when the PMU is disabled */
2364 	WARN_ON(this_cpu_read(cpu_hw_events.enabled));
2365 	local64_set(&hwc->prev_count, pmc);
2366 
2367 	delta = (pmc << shift) - (prev_pmc << shift);
2368 	delta >>= shift;
2369 
2370 	local64_add(delta, &event->count);
2371 	local64_sub(delta, &hwc->period_left);
2372 }
2373 
2374 static inline void __setup_pebs_counter_group(struct cpu_hw_events *cpuc,
2375 					      struct perf_event *event,
2376 					      struct pebs_cntr_header *cntr,
2377 					      void *next_record)
2378 {
2379 	int bit;
2380 
2381 	for_each_set_bit(bit, (unsigned long *)&cntr->cntr, INTEL_PMC_MAX_GENERIC) {
2382 		intel_perf_event_update_pmc(cpuc->events[bit], *(u64 *)next_record);
2383 		next_record += sizeof(u64);
2384 	}
2385 
2386 	for_each_set_bit(bit, (unsigned long *)&cntr->fixed, INTEL_PMC_MAX_FIXED) {
2387 		/* The slots event will be handled with perf_metric later */
2388 		if ((cntr->metrics == INTEL_CNTR_METRICS) &&
2389 		    (bit + INTEL_PMC_IDX_FIXED == INTEL_PMC_IDX_FIXED_SLOTS)) {
2390 			next_record += sizeof(u64);
2391 			continue;
2392 		}
2393 		intel_perf_event_update_pmc(cpuc->events[bit + INTEL_PMC_IDX_FIXED],
2394 					    *(u64 *)next_record);
2395 		next_record += sizeof(u64);
2396 	}
2397 
2398 	/* HW will reload the value right after the overflow. */
2399 	if (event->hw.flags & PERF_X86_EVENT_AUTO_RELOAD)
2400 		local64_set(&event->hw.prev_count, (u64)-event->hw.sample_period);
2401 
2402 	if (cntr->metrics == INTEL_CNTR_METRICS) {
2403 		static_call(intel_pmu_update_topdown_event)
2404 			   (cpuc->events[INTEL_PMC_IDX_FIXED_SLOTS],
2405 			    (u64 *)next_record);
2406 		next_record += 2 * sizeof(u64);
2407 	}
2408 }
2409 
2410 #define PEBS_LATENCY_MASK			0xffff
2411 
2412 static inline void __setup_perf_sample_data(struct perf_event *event,
2413 					    struct pt_regs *iregs,
2414 					    struct perf_sample_data *data)
2415 {
2416 	perf_sample_data_init(data, 0, event->hw.last_period);
2417 
2418 	/*
2419 	 * We must however always use iregs for the unwinder to stay sane; the
2420 	 * record BP,SP,IP can point into thin air when the record is from a
2421 	 * previous PMI context or an (I)RET happened between the record and
2422 	 * PMI.
2423 	 */
2424 	perf_sample_save_callchain(data, event, iregs);
2425 }
2426 
2427 static inline void __setup_pebs_basic_group(struct perf_event *event,
2428 					    struct pt_regs *regs,
2429 					    struct perf_sample_data *data,
2430 					    u64 sample_type, u64 ip,
2431 					    u64 tsc, u16 retire)
2432 {
2433 	/* The ip in basic is EventingIP */
2434 	set_linear_ip(regs, ip);
2435 	regs->flags |= PERF_EFLAGS_EXACT;
2436 	setup_pebs_time(event, data, tsc);
2437 
2438 	if (sample_type & PERF_SAMPLE_WEIGHT_STRUCT)
2439 		data->weight.var3_w = retire;
2440 }
2441 
2442 static inline void __setup_pebs_gpr_group(struct perf_event *event,
2443 					  struct pt_regs *regs,
2444 					  struct pebs_gprs *gprs,
2445 					  u64 sample_type)
2446 {
2447 	/*
2448 	 * Update flags with PEBS data. PERF_EFLAGS_EXACT must be set
2449 	 * in previous basic group handling.
2450 	 */
2451 	regs->flags = gprs->flags | PERF_EFLAGS_EXACT;
2452 
2453 	if (event->attr.precise_ip < 2) {
2454 		set_linear_ip(regs, gprs->ip);
2455 		regs->flags &= ~PERF_EFLAGS_EXACT;
2456 	} else if (regs->flags & X86_VM_MASK) {
2457 		regs->flags ^= (PERF_EFLAGS_VM | X86_VM_MASK);
2458 	}
2459 
2460 	if (sample_type & (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER))
2461 		adaptive_pebs_save_regs(regs, gprs);
2462 }
2463 
2464 static inline void __setup_pebs_meminfo_group(struct perf_event *event,
2465 					      struct perf_sample_data *data,
2466 					      u64 sample_type, u64 latency,
2467 					      u16 instr_latency, u64 address,
2468 					      u64 aux, u64 tsx_tuning, u64 ax)
2469 {
2470 	if (sample_type & PERF_SAMPLE_WEIGHT_TYPE) {
2471 		u64 tsx_latency = intel_get_tsx_weight(tsx_tuning);
2472 
2473 		data->weight.var2_w = instr_latency;
2474 
2475 		/*
2476 		 * Although meminfo::latency is defined as a u64,
2477 		 * only the lower 32 bits include the valid data
2478 		 * in practice on Ice Lake and earlier platforms.
2479 		 */
2480 		if (sample_type & PERF_SAMPLE_WEIGHT)
2481 			data->weight.full = latency ?: tsx_latency;
2482 		else
2483 			data->weight.var1_dw = (u32)latency ?: tsx_latency;
2484 
2485 		data->sample_flags |= PERF_SAMPLE_WEIGHT_TYPE;
2486 	}
2487 
2488 	if (sample_type & PERF_SAMPLE_DATA_SRC) {
2489 		data->data_src.val = get_data_src(event, aux);
2490 		data->sample_flags |= PERF_SAMPLE_DATA_SRC;
2491 	}
2492 
2493 	if (sample_type & PERF_SAMPLE_ADDR_TYPE) {
2494 		data->addr = address;
2495 		data->sample_flags |= PERF_SAMPLE_ADDR;
2496 	}
2497 
2498 	if (sample_type & PERF_SAMPLE_TRANSACTION) {
2499 		data->txn = intel_get_tsx_transaction(tsx_tuning, ax);
2500 		data->sample_flags |= PERF_SAMPLE_TRANSACTION;
2501 	}
2502 }
2503 
2504 /*
2505  * With adaptive PEBS the layout depends on what fields are configured.
2506  */
2507 static void setup_pebs_adaptive_sample_data(struct perf_event *event,
2508 					    struct pt_regs *iregs, void *__pebs,
2509 					    struct perf_sample_data *data,
2510 					    struct pt_regs *regs)
2511 {
2512 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
2513 	u64 sample_type = event->attr.sample_type;
2514 	struct pebs_basic *basic = __pebs;
2515 	void *next_record = basic + 1;
2516 	struct pebs_meminfo *meminfo = NULL;
2517 	struct pebs_gprs *gprs = NULL;
2518 	struct x86_perf_regs *perf_regs;
2519 	u64 format_group;
2520 	u16 retire;
2521 
2522 	if (basic == NULL)
2523 		return;
2524 
2525 	perf_regs = container_of(regs, struct x86_perf_regs, regs);
2526 	perf_regs->xmm_regs = NULL;
2527 
2528 	format_group = basic->format_group;
2529 
2530 	__setup_perf_sample_data(event, iregs, data);
2531 
2532 	*regs = *iregs;
2533 
2534 	/* basic group */
2535 	retire = x86_pmu.flags & PMU_FL_RETIRE_LATENCY ?
2536 			basic->retire_latency : 0;
2537 	__setup_pebs_basic_group(event, regs, data, sample_type,
2538 				 basic->ip, basic->tsc, retire);
2539 
2540 	/*
2541 	 * The record for MEMINFO is in front of GP
2542 	 * But PERF_SAMPLE_TRANSACTION needs gprs->ax.
2543 	 * Save the pointer here but process later.
2544 	 */
2545 	if (format_group & PEBS_DATACFG_MEMINFO) {
2546 		meminfo = next_record;
2547 		next_record = meminfo + 1;
2548 	}
2549 
2550 	if (format_group & PEBS_DATACFG_GP) {
2551 		gprs = next_record;
2552 		next_record = gprs + 1;
2553 
2554 		__setup_pebs_gpr_group(event, regs, gprs, sample_type);
2555 	}
2556 
2557 	if (format_group & PEBS_DATACFG_MEMINFO) {
2558 		u64 latency = x86_pmu.flags & PMU_FL_INSTR_LATENCY ?
2559 				meminfo->cache_latency : meminfo->mem_latency;
2560 		u64 instr_latency = x86_pmu.flags & PMU_FL_INSTR_LATENCY ?
2561 				meminfo->instr_latency : 0;
2562 		u64 ax = gprs ? gprs->ax : 0;
2563 
2564 		__setup_pebs_meminfo_group(event, data, sample_type, latency,
2565 					   instr_latency, meminfo->address,
2566 					   meminfo->aux, meminfo->tsx_tuning,
2567 					   ax);
2568 	}
2569 
2570 	if (format_group & PEBS_DATACFG_XMMS) {
2571 		struct pebs_xmm *xmm = next_record;
2572 
2573 		next_record = xmm + 1;
2574 		perf_regs->xmm_regs = xmm->xmm;
2575 	}
2576 
2577 	if (format_group & PEBS_DATACFG_LBRS) {
2578 		struct lbr_entry *lbr = next_record;
2579 		int num_lbr = ((format_group >> PEBS_DATACFG_LBR_SHIFT)
2580 					& 0xff) + 1;
2581 		next_record = next_record + num_lbr * sizeof(struct lbr_entry);
2582 
2583 		if (has_branch_stack(event)) {
2584 			intel_pmu_store_pebs_lbrs(lbr);
2585 			intel_pmu_lbr_save_brstack(data, cpuc, event);
2586 		}
2587 	}
2588 
2589 	if (format_group & (PEBS_DATACFG_CNTR | PEBS_DATACFG_METRICS)) {
2590 		struct pebs_cntr_header *cntr = next_record;
2591 		unsigned int nr;
2592 
2593 		next_record += sizeof(struct pebs_cntr_header);
2594 		/*
2595 		 * The PEBS_DATA_CFG is a global register, which is the
2596 		 * superset configuration for all PEBS events.
2597 		 * For the PEBS record of non-sample-read group, ignore
2598 		 * the counter snapshot fields.
2599 		 */
2600 		if (is_pebs_counter_event_group(event)) {
2601 			__setup_pebs_counter_group(cpuc, event, cntr, next_record);
2602 			data->sample_flags |= PERF_SAMPLE_READ;
2603 		}
2604 
2605 		nr = hweight32(cntr->cntr) + hweight32(cntr->fixed);
2606 		if (cntr->metrics == INTEL_CNTR_METRICS)
2607 			nr += 2;
2608 		next_record += nr * sizeof(u64);
2609 	}
2610 
2611 	WARN_ONCE(next_record != __pebs + basic->format_size,
2612 			"PEBS record size %u, expected %llu, config %llx\n",
2613 			basic->format_size,
2614 			(u64)(next_record - __pebs),
2615 			format_group);
2616 }
2617 
2618 static inline bool arch_pebs_record_continued(struct arch_pebs_header *header)
2619 {
2620 	/* Continue bit or null PEBS record indicates fragment follows. */
2621 	return header->cont || !(header->format & GENMASK_ULL(63, 16));
2622 }
2623 
2624 static void setup_arch_pebs_sample_data(struct perf_event *event,
2625 					struct pt_regs *iregs,
2626 					void *__pebs,
2627 					struct perf_sample_data *data,
2628 					struct pt_regs *regs)
2629 {
2630 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
2631 	u64 sample_type = event->attr.sample_type;
2632 	struct arch_pebs_header *header = NULL;
2633 	struct arch_pebs_aux *meminfo = NULL;
2634 	struct arch_pebs_gprs *gprs = NULL;
2635 	struct x86_perf_regs *perf_regs;
2636 	void *next_record;
2637 	void *at = __pebs;
2638 
2639 	if (at == NULL)
2640 		return;
2641 
2642 	perf_regs = container_of(regs, struct x86_perf_regs, regs);
2643 	perf_regs->xmm_regs = NULL;
2644 
2645 	__setup_perf_sample_data(event, iregs, data);
2646 
2647 	*regs = *iregs;
2648 
2649 again:
2650 	header = at;
2651 	next_record = at + sizeof(struct arch_pebs_header);
2652 	if (header->basic) {
2653 		struct arch_pebs_basic *basic = next_record;
2654 		u16 retire = 0;
2655 
2656 		next_record = basic + 1;
2657 
2658 		if (sample_type & PERF_SAMPLE_WEIGHT_STRUCT)
2659 			retire = basic->valid ? basic->retire : 0;
2660 		__setup_pebs_basic_group(event, regs, data, sample_type,
2661 				 basic->ip, basic->tsc, retire);
2662 	}
2663 
2664 	/*
2665 	 * The record for MEMINFO is in front of GP
2666 	 * But PERF_SAMPLE_TRANSACTION needs gprs->ax.
2667 	 * Save the pointer here but process later.
2668 	 */
2669 	if (header->aux) {
2670 		meminfo = next_record;
2671 		next_record = meminfo + 1;
2672 	}
2673 
2674 	if (header->gpr) {
2675 		gprs = next_record;
2676 		next_record = gprs + 1;
2677 
2678 		__setup_pebs_gpr_group(event, regs,
2679 				       (struct pebs_gprs *)gprs,
2680 				       sample_type);
2681 	}
2682 
2683 	if (header->aux) {
2684 		u64 ax = gprs ? gprs->ax : 0;
2685 
2686 		__setup_pebs_meminfo_group(event, data, sample_type,
2687 					   meminfo->cache_latency,
2688 					   meminfo->instr_latency,
2689 					   meminfo->address, meminfo->aux,
2690 					   meminfo->tsx_tuning, ax);
2691 	}
2692 
2693 	if (header->xmm) {
2694 		struct pebs_xmm *xmm;
2695 
2696 		next_record += sizeof(struct arch_pebs_xer_header);
2697 
2698 		xmm = next_record;
2699 		perf_regs->xmm_regs = xmm->xmm;
2700 		next_record = xmm + 1;
2701 	}
2702 
2703 	if (header->lbr) {
2704 		struct arch_pebs_lbr_header *lbr_header = next_record;
2705 		struct lbr_entry *lbr;
2706 		int num_lbr;
2707 
2708 		next_record = lbr_header + 1;
2709 		lbr = next_record;
2710 
2711 		num_lbr = header->lbr == ARCH_PEBS_LBR_NUM_VAR ?
2712 				lbr_header->depth :
2713 				header->lbr * ARCH_PEBS_BASE_LBR_ENTRIES;
2714 		next_record += num_lbr * sizeof(struct lbr_entry);
2715 
2716 		if (has_branch_stack(event)) {
2717 			intel_pmu_store_pebs_lbrs(lbr);
2718 			intel_pmu_lbr_save_brstack(data, cpuc, event);
2719 		}
2720 	}
2721 
2722 	if (header->cntr) {
2723 		struct arch_pebs_cntr_header *cntr = next_record;
2724 		unsigned int nr;
2725 
2726 		next_record += sizeof(struct arch_pebs_cntr_header);
2727 
2728 		if (is_pebs_counter_event_group(event)) {
2729 			__setup_pebs_counter_group(cpuc, event,
2730 				(struct pebs_cntr_header *)cntr, next_record);
2731 			data->sample_flags |= PERF_SAMPLE_READ;
2732 		}
2733 
2734 		nr = hweight32(cntr->cntr) + hweight32(cntr->fixed);
2735 		if (cntr->metrics == INTEL_CNTR_METRICS)
2736 			nr += 2;
2737 		next_record += nr * sizeof(u64);
2738 	}
2739 
2740 	/* Parse followed fragments if there are. */
2741 	if (arch_pebs_record_continued(header)) {
2742 		at = at + header->size;
2743 		goto again;
2744 	}
2745 }
2746 
2747 static inline void *
2748 get_next_pebs_record_by_bit(void *base, void *top, int bit)
2749 {
2750 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
2751 	void *at;
2752 	u64 pebs_status;
2753 
2754 	/*
2755 	 * fmt0 does not have a status bitfield (does not use
2756 	 * perf_record_nhm format)
2757 	 */
2758 	if (x86_pmu.intel_cap.pebs_format < 1)
2759 		return base;
2760 
2761 	if (base == NULL)
2762 		return NULL;
2763 
2764 	for (at = base; at < top; at += cpuc->pebs_record_size) {
2765 		unsigned long status = get_pebs_status(at);
2766 
2767 		if (test_bit(bit, (unsigned long *)&status)) {
2768 			/* PEBS v3 has accurate status bits */
2769 			if (x86_pmu.intel_cap.pebs_format >= 3)
2770 				return at;
2771 
2772 			if (status == (1 << bit))
2773 				return at;
2774 
2775 			/* clear non-PEBS bit and re-check */
2776 			pebs_status = status & cpuc->pebs_enabled;
2777 			pebs_status &= PEBS_COUNTER_MASK;
2778 			if (pebs_status == (1 << bit))
2779 				return at;
2780 		}
2781 	}
2782 	return NULL;
2783 }
2784 
2785 /*
2786  * Special variant of intel_pmu_save_and_restart() for auto-reload.
2787  */
2788 static int
2789 intel_pmu_save_and_restart_reload(struct perf_event *event, int count)
2790 {
2791 	struct hw_perf_event *hwc = &event->hw;
2792 	int shift = 64 - x86_pmu.cntval_bits;
2793 	u64 period = hwc->sample_period;
2794 	u64 prev_raw_count, new_raw_count;
2795 	s64 new, old;
2796 
2797 	WARN_ON(!period);
2798 
2799 	/*
2800 	 * drain_pebs() only happens when the PMU is disabled.
2801 	 */
2802 	WARN_ON(this_cpu_read(cpu_hw_events.enabled));
2803 
2804 	prev_raw_count = local64_read(&hwc->prev_count);
2805 	new_raw_count = rdpmc(hwc->event_base_rdpmc);
2806 	local64_set(&hwc->prev_count, new_raw_count);
2807 
2808 	/*
2809 	 * Since the counter increments a negative counter value and
2810 	 * overflows on the sign switch, giving the interval:
2811 	 *
2812 	 *   [-period, 0]
2813 	 *
2814 	 * the difference between two consecutive reads is:
2815 	 *
2816 	 *   A) value2 - value1;
2817 	 *      when no overflows have happened in between,
2818 	 *
2819 	 *   B) (0 - value1) + (value2 - (-period));
2820 	 *      when one overflow happened in between,
2821 	 *
2822 	 *   C) (0 - value1) + (n - 1) * (period) + (value2 - (-period));
2823 	 *      when @n overflows happened in between.
2824 	 *
2825 	 * Here A) is the obvious difference, B) is the extension to the
2826 	 * discrete interval, where the first term is to the top of the
2827 	 * interval and the second term is from the bottom of the next
2828 	 * interval and C) the extension to multiple intervals, where the
2829 	 * middle term is the whole intervals covered.
2830 	 *
2831 	 * An equivalent of C, by reduction, is:
2832 	 *
2833 	 *   value2 - value1 + n * period
2834 	 */
2835 	new = ((s64)(new_raw_count << shift) >> shift);
2836 	old = ((s64)(prev_raw_count << shift) >> shift);
2837 	local64_add(new - old + count * period, &event->count);
2838 
2839 	local64_set(&hwc->period_left, -new);
2840 
2841 	perf_event_update_userpage(event);
2842 
2843 	return 0;
2844 }
2845 
2846 typedef void (*setup_fn)(struct perf_event *, struct pt_regs *, void *,
2847 			 struct perf_sample_data *, struct pt_regs *);
2848 
2849 static struct pt_regs dummy_iregs;
2850 
2851 static __always_inline void
2852 __intel_pmu_pebs_event(struct perf_event *event,
2853 		       struct pt_regs *iregs,
2854 		       struct pt_regs *regs,
2855 		       struct perf_sample_data *data,
2856 		       void *at,
2857 		       setup_fn setup_sample)
2858 {
2859 	setup_sample(event, iregs, at, data, regs);
2860 	perf_event_output(event, data, regs);
2861 }
2862 
2863 static __always_inline void
2864 __intel_pmu_pebs_last_event(struct perf_event *event,
2865 			    struct pt_regs *iregs,
2866 			    struct pt_regs *regs,
2867 			    struct perf_sample_data *data,
2868 			    void *at,
2869 			    int count,
2870 			    setup_fn setup_sample)
2871 {
2872 	struct hw_perf_event *hwc = &event->hw;
2873 
2874 	setup_sample(event, iregs, at, data, regs);
2875 	if (iregs == &dummy_iregs) {
2876 		/*
2877 		 * The PEBS records may be drained in the non-overflow context,
2878 		 * e.g., large PEBS + context switch. Perf should treat the
2879 		 * last record the same as other PEBS records, and doesn't
2880 		 * invoke the generic overflow handler.
2881 		 */
2882 		perf_event_output(event, data, regs);
2883 	} else {
2884 		/*
2885 		 * All but the last records are processed.
2886 		 * The last one is left to be able to call the overflow handler.
2887 		 */
2888 		perf_event_overflow(event, data, regs);
2889 	}
2890 
2891 	if (hwc->flags & PERF_X86_EVENT_AUTO_RELOAD) {
2892 		if ((is_pebs_counter_event_group(event))) {
2893 			/*
2894 			 * The value of each sample has been updated when setup
2895 			 * the corresponding sample data.
2896 			 */
2897 			perf_event_update_userpage(event);
2898 		} else {
2899 			/*
2900 			 * Now, auto-reload is only enabled in fixed period mode.
2901 			 * The reload value is always hwc->sample_period.
2902 			 * May need to change it, if auto-reload is enabled in
2903 			 * freq mode later.
2904 			 */
2905 			intel_pmu_save_and_restart_reload(event, count);
2906 		}
2907 	} else {
2908 		/*
2909 		 * For a non-precise event, it's possible the
2910 		 * counters-snapshotting records a positive value for the
2911 		 * overflowed event. Then the HW auto-reload mechanism
2912 		 * reset the counter to 0 immediately, because the
2913 		 * pebs_event_reset is cleared if the PERF_X86_EVENT_AUTO_RELOAD
2914 		 * is not set. The counter backwards may be observed in a
2915 		 * PMI handler.
2916 		 *
2917 		 * Since the event value has been updated when processing the
2918 		 * counters-snapshotting record, only needs to set the new
2919 		 * period for the counter.
2920 		 */
2921 		if (is_pebs_counter_event_group(event))
2922 			static_call(x86_pmu_set_period)(event);
2923 		else
2924 			intel_pmu_save_and_restart(event);
2925 	}
2926 }
2927 
2928 static __always_inline void
2929 __intel_pmu_pebs_events(struct perf_event *event,
2930 			struct pt_regs *iregs,
2931 			struct perf_sample_data *data,
2932 			void *base, void *top,
2933 			int bit, int count,
2934 			setup_fn setup_sample)
2935 {
2936 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
2937 	struct x86_perf_regs perf_regs;
2938 	struct pt_regs *regs = &perf_regs.regs;
2939 	void *at = get_next_pebs_record_by_bit(base, top, bit);
2940 	int cnt = count;
2941 
2942 	if (!iregs)
2943 		iregs = &dummy_iregs;
2944 
2945 	while (cnt > 1) {
2946 		__intel_pmu_pebs_event(event, iregs, regs, data, at, setup_sample);
2947 		at += cpuc->pebs_record_size;
2948 		at = get_next_pebs_record_by_bit(at, top, bit);
2949 		cnt--;
2950 	}
2951 
2952 	__intel_pmu_pebs_last_event(event, iregs, regs, data, at, count, setup_sample);
2953 }
2954 
2955 static void intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_data *data)
2956 {
2957 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
2958 	struct debug_store *ds = cpuc->ds;
2959 	struct perf_event *event = cpuc->events[0]; /* PMC0 only */
2960 	struct pebs_record_core *at, *top;
2961 	int n;
2962 
2963 	if (!x86_pmu.pebs_active)
2964 		return;
2965 
2966 	at  = (struct pebs_record_core *)(unsigned long)ds->pebs_buffer_base;
2967 	top = (struct pebs_record_core *)(unsigned long)ds->pebs_index;
2968 
2969 	/*
2970 	 * Whatever else happens, drain the thing
2971 	 */
2972 	ds->pebs_index = ds->pebs_buffer_base;
2973 
2974 	if (!test_bit(0, cpuc->active_mask))
2975 		return;
2976 
2977 	WARN_ON_ONCE(!event);
2978 
2979 	if (!event->attr.precise_ip)
2980 		return;
2981 
2982 	n = top - at;
2983 	if (n <= 0) {
2984 		if (event->hw.flags & PERF_X86_EVENT_AUTO_RELOAD)
2985 			intel_pmu_save_and_restart_reload(event, 0);
2986 		return;
2987 	}
2988 
2989 	__intel_pmu_pebs_events(event, iregs, data, at, top, 0, n,
2990 				setup_pebs_fixed_sample_data);
2991 }
2992 
2993 static void intel_pmu_pebs_event_update_no_drain(struct cpu_hw_events *cpuc, u64 mask)
2994 {
2995 	u64 pebs_enabled = cpuc->pebs_enabled & mask;
2996 	struct perf_event *event;
2997 	int bit;
2998 
2999 	/*
3000 	 * The drain_pebs() could be called twice in a short period
3001 	 * for auto-reload event in pmu::read(). There are no
3002 	 * overflows have happened in between.
3003 	 * It needs to call intel_pmu_save_and_restart_reload() to
3004 	 * update the event->count for this case.
3005 	 */
3006 	for_each_set_bit(bit, (unsigned long *)&pebs_enabled, X86_PMC_IDX_MAX) {
3007 		event = cpuc->events[bit];
3008 		if (event->hw.flags & PERF_X86_EVENT_AUTO_RELOAD)
3009 			intel_pmu_save_and_restart_reload(event, 0);
3010 	}
3011 }
3012 
3013 static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_data *data)
3014 {
3015 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
3016 	struct debug_store *ds = cpuc->ds;
3017 	struct perf_event *event;
3018 	void *base, *at, *top;
3019 	short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
3020 	short error[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
3021 	int max_pebs_events = intel_pmu_max_num_pebs(NULL);
3022 	int bit, i, size;
3023 	u64 mask;
3024 
3025 	if (!x86_pmu.pebs_active)
3026 		return;
3027 
3028 	base = (struct pebs_record_nhm *)(unsigned long)ds->pebs_buffer_base;
3029 	top = (struct pebs_record_nhm *)(unsigned long)ds->pebs_index;
3030 
3031 	ds->pebs_index = ds->pebs_buffer_base;
3032 
3033 	mask = x86_pmu.pebs_events_mask;
3034 	size = max_pebs_events;
3035 	if (x86_pmu.flags & PMU_FL_PEBS_ALL) {
3036 		mask |= x86_pmu.fixed_cntr_mask64 << INTEL_PMC_IDX_FIXED;
3037 		size = INTEL_PMC_IDX_FIXED + x86_pmu_max_num_counters_fixed(NULL);
3038 	}
3039 
3040 	if (unlikely(base >= top)) {
3041 		intel_pmu_pebs_event_update_no_drain(cpuc, mask);
3042 		return;
3043 	}
3044 
3045 	for (at = base; at < top; at += x86_pmu.pebs_record_size) {
3046 		struct pebs_record_nhm *p = at;
3047 		u64 pebs_status;
3048 
3049 		pebs_status = p->status & cpuc->pebs_enabled;
3050 		pebs_status &= mask;
3051 
3052 		/* PEBS v3 has more accurate status bits */
3053 		if (x86_pmu.intel_cap.pebs_format >= 3) {
3054 			for_each_set_bit(bit, (unsigned long *)&pebs_status, size)
3055 				counts[bit]++;
3056 
3057 			continue;
3058 		}
3059 
3060 		/*
3061 		 * On some CPUs the PEBS status can be zero when PEBS is
3062 		 * racing with clearing of GLOBAL_STATUS.
3063 		 *
3064 		 * Normally we would drop that record, but in the
3065 		 * case when there is only a single active PEBS event
3066 		 * we can assume it's for that event.
3067 		 */
3068 		if (!pebs_status && cpuc->pebs_enabled &&
3069 			!(cpuc->pebs_enabled & (cpuc->pebs_enabled-1)))
3070 			pebs_status = p->status = cpuc->pebs_enabled;
3071 
3072 		bit = find_first_bit((unsigned long *)&pebs_status,
3073 				     max_pebs_events);
3074 
3075 		if (!(x86_pmu.pebs_events_mask & (1 << bit)))
3076 			continue;
3077 
3078 		/*
3079 		 * The PEBS hardware does not deal well with the situation
3080 		 * when events happen near to each other and multiple bits
3081 		 * are set. But it should happen rarely.
3082 		 *
3083 		 * If these events include one PEBS and multiple non-PEBS
3084 		 * events, it doesn't impact PEBS record. The record will
3085 		 * be handled normally. (slow path)
3086 		 *
3087 		 * If these events include two or more PEBS events, the
3088 		 * records for the events can be collapsed into a single
3089 		 * one, and it's not possible to reconstruct all events
3090 		 * that caused the PEBS record. It's called collision.
3091 		 * If collision happened, the record will be dropped.
3092 		 */
3093 		if (pebs_status != (1ULL << bit)) {
3094 			for_each_set_bit(i, (unsigned long *)&pebs_status, size)
3095 				error[i]++;
3096 			continue;
3097 		}
3098 
3099 		counts[bit]++;
3100 	}
3101 
3102 	for_each_set_bit(bit, (unsigned long *)&mask, size) {
3103 		if ((counts[bit] == 0) && (error[bit] == 0))
3104 			continue;
3105 
3106 		event = cpuc->events[bit];
3107 		if (WARN_ON_ONCE(!event))
3108 			continue;
3109 
3110 		if (WARN_ON_ONCE(!event->attr.precise_ip))
3111 			continue;
3112 
3113 		/* log dropped samples number */
3114 		if (error[bit]) {
3115 			perf_log_lost_samples(event, error[bit]);
3116 
3117 			if (iregs)
3118 				perf_event_account_interrupt(event);
3119 		}
3120 
3121 		if (counts[bit]) {
3122 			__intel_pmu_pebs_events(event, iregs, data, base,
3123 						top, bit, counts[bit],
3124 						setup_pebs_fixed_sample_data);
3125 		}
3126 	}
3127 }
3128 
3129 static __always_inline void
3130 __intel_pmu_handle_pebs_record(struct pt_regs *iregs,
3131 			       struct pt_regs *regs,
3132 			       struct perf_sample_data *data,
3133 			       void *at, u64 pebs_status,
3134 			       short *counts, void **last,
3135 			       setup_fn setup_sample)
3136 {
3137 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
3138 	struct perf_event *event;
3139 	int bit;
3140 
3141 	for_each_set_bit(bit, (unsigned long *)&pebs_status, X86_PMC_IDX_MAX) {
3142 		event = cpuc->events[bit];
3143 
3144 		if (WARN_ON_ONCE(!event) ||
3145 		    WARN_ON_ONCE(!event->attr.precise_ip))
3146 			continue;
3147 
3148 		if (counts[bit]++) {
3149 			__intel_pmu_pebs_event(event, iregs, regs, data,
3150 					       last[bit], setup_sample);
3151 		}
3152 
3153 		last[bit] = at;
3154 	}
3155 }
3156 
3157 static __always_inline void
3158 __intel_pmu_handle_last_pebs_record(struct pt_regs *iregs,
3159 				    struct pt_regs *regs,
3160 				    struct perf_sample_data *data,
3161 				    u64 mask, short *counts, void **last,
3162 				    setup_fn setup_sample)
3163 {
3164 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
3165 	struct perf_event *event;
3166 	int bit;
3167 
3168 	for_each_set_bit(bit, (unsigned long *)&mask, X86_PMC_IDX_MAX) {
3169 		if (!counts[bit])
3170 			continue;
3171 
3172 		event = cpuc->events[bit];
3173 
3174 		__intel_pmu_pebs_last_event(event, iregs, regs, data, last[bit],
3175 					    counts[bit], setup_sample);
3176 	}
3177 
3178 }
3179 
3180 static void intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_data *data)
3181 {
3182 	short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
3183 	void *last[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS];
3184 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
3185 	struct debug_store *ds = cpuc->ds;
3186 	struct x86_perf_regs perf_regs;
3187 	struct pt_regs *regs = &perf_regs.regs;
3188 	struct pebs_basic *basic;
3189 	void *base, *at, *top;
3190 	u64 mask;
3191 
3192 	if (!x86_pmu.pebs_active)
3193 		return;
3194 
3195 	base = (struct pebs_basic *)(unsigned long)ds->pebs_buffer_base;
3196 	top = (struct pebs_basic *)(unsigned long)ds->pebs_index;
3197 
3198 	ds->pebs_index = ds->pebs_buffer_base;
3199 
3200 	mask = hybrid(cpuc->pmu, pebs_events_mask) |
3201 	       (hybrid(cpuc->pmu, fixed_cntr_mask64) << INTEL_PMC_IDX_FIXED);
3202 	mask &= cpuc->pebs_enabled;
3203 
3204 	if (unlikely(base >= top)) {
3205 		intel_pmu_pebs_event_update_no_drain(cpuc, mask);
3206 		return;
3207 	}
3208 
3209 	if (!iregs)
3210 		iregs = &dummy_iregs;
3211 
3212 	/* Process all but the last event for each counter. */
3213 	for (at = base; at < top; at += basic->format_size) {
3214 		u64 pebs_status;
3215 
3216 		basic = at;
3217 		if (basic->format_size != cpuc->pebs_record_size)
3218 			continue;
3219 
3220 		pebs_status = mask & basic->applicable_counters;
3221 		__intel_pmu_handle_pebs_record(iregs, regs, data, at,
3222 					       pebs_status, counts, last,
3223 					       setup_pebs_adaptive_sample_data);
3224 	}
3225 
3226 	__intel_pmu_handle_last_pebs_record(iregs, regs, data, mask, counts, last,
3227 					    setup_pebs_adaptive_sample_data);
3228 }
3229 
3230 static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs,
3231 				      struct perf_sample_data *data)
3232 {
3233 	short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
3234 	void *last[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS];
3235 	struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
3236 	union arch_pebs_index index;
3237 	struct x86_perf_regs perf_regs;
3238 	struct pt_regs *regs = &perf_regs.regs;
3239 	void *base, *at, *top;
3240 	u64 mask;
3241 
3242 	rdmsrq(MSR_IA32_PEBS_INDEX, index.whole);
3243 
3244 	if (unlikely(!index.wr)) {
3245 		intel_pmu_pebs_event_update_no_drain(cpuc, X86_PMC_IDX_MAX);
3246 		return;
3247 	}
3248 
3249 	base = cpuc->pebs_vaddr;
3250 	top = cpuc->pebs_vaddr + (index.wr << ARCH_PEBS_INDEX_WR_SHIFT);
3251 
3252 	index.wr = 0;
3253 	index.full = 0;
3254 	index.en = 1;
3255 	if (cpuc->n_pebs == cpuc->n_large_pebs)
3256 		index.thresh = ARCH_PEBS_THRESH_MULTI;
3257 	else
3258 		index.thresh = ARCH_PEBS_THRESH_SINGLE;
3259 	wrmsrq(MSR_IA32_PEBS_INDEX, index.whole);
3260 
3261 	mask = hybrid(cpuc->pmu, arch_pebs_cap).counters & cpuc->pebs_enabled;
3262 
3263 	if (!iregs)
3264 		iregs = &dummy_iregs;
3265 
3266 	/* Process all but the last event for each counter. */
3267 	for (at = base; at < top;) {
3268 		struct arch_pebs_header *header;
3269 		struct arch_pebs_basic *basic;
3270 		u64 pebs_status;
3271 
3272 		header = at;
3273 
3274 		if (WARN_ON_ONCE(!header->size))
3275 			break;
3276 
3277 		/* 1st fragment or single record must have basic group */
3278 		if (!header->basic) {
3279 			at += header->size;
3280 			continue;
3281 		}
3282 
3283 		basic = at + sizeof(struct arch_pebs_header);
3284 		pebs_status = mask & basic->applicable_counters;
3285 		__intel_pmu_handle_pebs_record(iregs, regs, data, at,
3286 					       pebs_status, counts, last,
3287 					       setup_arch_pebs_sample_data);
3288 
3289 		/* Skip non-last fragments */
3290 		while (arch_pebs_record_continued(header)) {
3291 			if (!header->size)
3292 				break;
3293 			at += header->size;
3294 			header = at;
3295 		}
3296 
3297 		/* Skip last fragment or the single record */
3298 		at += header->size;
3299 	}
3300 
3301 	__intel_pmu_handle_last_pebs_record(iregs, regs, data, mask,
3302 					    counts, last,
3303 					    setup_arch_pebs_sample_data);
3304 }
3305 
3306 static void __init intel_arch_pebs_init(void)
3307 {
3308 	/*
3309 	 * Current hybrid platforms always both support arch-PEBS or not
3310 	 * on all kinds of cores. So directly set x86_pmu.arch_pebs flag
3311 	 * if boot cpu supports arch-PEBS.
3312 	 */
3313 	x86_pmu.arch_pebs = 1;
3314 	x86_pmu.pebs_buffer_size = PEBS_BUFFER_SIZE;
3315 	x86_pmu.drain_pebs = intel_pmu_drain_arch_pebs;
3316 	x86_pmu.pebs_capable = ~0ULL;
3317 	x86_pmu.flags |= PMU_FL_PEBS_ALL;
3318 
3319 	x86_pmu.pebs_enable = __intel_pmu_pebs_enable;
3320 	x86_pmu.pebs_disable = __intel_pmu_pebs_disable;
3321 }
3322 
3323 /*
3324  * PEBS probe and setup
3325  */
3326 
3327 static void __init intel_ds_pebs_init(void)
3328 {
3329 	/*
3330 	 * No support for 32bit formats
3331 	 */
3332 	if (!boot_cpu_has(X86_FEATURE_DTES64))
3333 		return;
3334 
3335 	x86_pmu.ds_pebs = boot_cpu_has(X86_FEATURE_PEBS);
3336 	x86_pmu.pebs_buffer_size = PEBS_BUFFER_SIZE;
3337 	if (x86_pmu.version <= 4)
3338 		x86_pmu.pebs_no_isolation = 1;
3339 
3340 	if (x86_pmu.ds_pebs) {
3341 		char pebs_type = x86_pmu.intel_cap.pebs_trap ?  '+' : '-';
3342 		char *pebs_qual = "";
3343 		int format = x86_pmu.intel_cap.pebs_format;
3344 
3345 		if (format < 4)
3346 			x86_pmu.intel_cap.pebs_baseline = 0;
3347 
3348 		x86_pmu.pebs_enable = intel_pmu_pebs_enable;
3349 		x86_pmu.pebs_disable = intel_pmu_pebs_disable;
3350 		x86_pmu.pebs_enable_all = intel_pmu_pebs_enable_all;
3351 		x86_pmu.pebs_disable_all = intel_pmu_pebs_disable_all;
3352 
3353 		switch (format) {
3354 		case 0:
3355 			pr_cont("PEBS fmt0%c, ", pebs_type);
3356 			x86_pmu.pebs_record_size = sizeof(struct pebs_record_core);
3357 			/*
3358 			 * Using >PAGE_SIZE buffers makes the WRMSR to
3359 			 * PERF_GLOBAL_CTRL in intel_pmu_enable_all()
3360 			 * mysteriously hang on Core2.
3361 			 *
3362 			 * As a workaround, we don't do this.
3363 			 */
3364 			x86_pmu.pebs_buffer_size = PAGE_SIZE;
3365 			x86_pmu.drain_pebs = intel_pmu_drain_pebs_core;
3366 			break;
3367 
3368 		case 1:
3369 			pr_cont("PEBS fmt1%c, ", pebs_type);
3370 			x86_pmu.pebs_record_size = sizeof(struct pebs_record_nhm);
3371 			x86_pmu.drain_pebs = intel_pmu_drain_pebs_nhm;
3372 			break;
3373 
3374 		case 2:
3375 			pr_cont("PEBS fmt2%c, ", pebs_type);
3376 			x86_pmu.pebs_record_size = sizeof(struct pebs_record_hsw);
3377 			x86_pmu.drain_pebs = intel_pmu_drain_pebs_nhm;
3378 			break;
3379 
3380 		case 3:
3381 			pr_cont("PEBS fmt3%c, ", pebs_type);
3382 			x86_pmu.pebs_record_size =
3383 						sizeof(struct pebs_record_skl);
3384 			x86_pmu.drain_pebs = intel_pmu_drain_pebs_nhm;
3385 			x86_pmu.large_pebs_flags |= PERF_SAMPLE_TIME;
3386 			break;
3387 
3388 		case 6:
3389 			if (x86_pmu.intel_cap.pebs_baseline)
3390 				x86_pmu.large_pebs_flags |= PERF_SAMPLE_READ;
3391 			fallthrough;
3392 		case 5:
3393 			x86_pmu.pebs_ept = 1;
3394 			fallthrough;
3395 		case 4:
3396 			x86_pmu.drain_pebs = intel_pmu_drain_pebs_icl;
3397 			x86_pmu.pebs_record_size = sizeof(struct pebs_basic);
3398 			if (x86_pmu.intel_cap.pebs_baseline) {
3399 				x86_pmu.large_pebs_flags |=
3400 					PERF_SAMPLE_BRANCH_STACK |
3401 					PERF_SAMPLE_TIME;
3402 				x86_pmu.flags |= PMU_FL_PEBS_ALL;
3403 				x86_pmu.pebs_capable = ~0ULL;
3404 				pebs_qual = "-baseline";
3405 				x86_get_pmu(smp_processor_id())->capabilities |= PERF_PMU_CAP_EXTENDED_REGS;
3406 			} else {
3407 				/* Only basic record supported */
3408 				x86_pmu.large_pebs_flags &=
3409 					~(PERF_SAMPLE_ADDR |
3410 					  PERF_SAMPLE_TIME |
3411 					  PERF_SAMPLE_DATA_SRC |
3412 					  PERF_SAMPLE_TRANSACTION |
3413 					  PERF_SAMPLE_REGS_USER |
3414 					  PERF_SAMPLE_REGS_INTR);
3415 			}
3416 			pr_cont("PEBS fmt%d%c%s, ", format, pebs_type, pebs_qual);
3417 
3418 			/*
3419 			 * The PEBS-via-PT is not supported on hybrid platforms,
3420 			 * because not all CPUs of a hybrid machine support it.
3421 			 * The global x86_pmu.intel_cap, which only contains the
3422 			 * common capabilities, is used to check the availability
3423 			 * of the feature. The per-PMU pebs_output_pt_available
3424 			 * in a hybrid machine should be ignored.
3425 			 */
3426 			if (x86_pmu.intel_cap.pebs_output_pt_available) {
3427 				pr_cont("PEBS-via-PT, ");
3428 				x86_get_pmu(smp_processor_id())->capabilities |= PERF_PMU_CAP_AUX_OUTPUT;
3429 			}
3430 
3431 			break;
3432 
3433 		default:
3434 			pr_cont("no PEBS fmt%d%c, ", format, pebs_type);
3435 			x86_pmu.ds_pebs = 0;
3436 		}
3437 	}
3438 }
3439 
3440 void __init intel_pebs_init(void)
3441 {
3442 	if (x86_pmu.intel_cap.pebs_format == 0xf)
3443 		intel_arch_pebs_init();
3444 	else
3445 		intel_ds_pebs_init();
3446 }
3447 
3448 void perf_restore_debug_store(void)
3449 {
3450 	struct debug_store *ds = __this_cpu_read(cpu_hw_events.ds);
3451 
3452 	if (!x86_pmu.bts && !x86_pmu.ds_pebs)
3453 		return;
3454 
3455 	wrmsrq(MSR_IA32_DS_AREA, (unsigned long)ds);
3456 }
3457