1 // SPDX-License-Identifier: GPL-2.0
2 /* Copyright (c) 2018-2019 The Linux Foundation. All rights reserved. */
3
4 #include <linux/ascii85.h>
5 #include "msm_gem.h"
6 #include "a6xx_gpu.h"
7 #include "a6xx_gmu.h"
8 #include "a6xx_gpu_state.h"
9 #include "a6xx_gmu.xml.h"
10
11 static const unsigned int *gen7_0_0_external_core_regs[] __always_unused;
12 static const unsigned int *gen7_2_0_external_core_regs[] __always_unused;
13 static const unsigned int *gen7_9_0_external_core_regs[] __always_unused;
14 static const unsigned int *gen7_17_0_external_core_regs[] __always_unused;
15 static const struct gen7_sptp_cluster_registers gen7_9_0_sptp_clusters[] __always_unused;
16 static const u32 gen7_9_0_cx_debugbus_blocks[] __always_unused;
17
18 #include "adreno_gen7_0_0_snapshot.h"
19 #include "adreno_gen7_2_0_snapshot.h"
20 #include "adreno_gen7_9_0_snapshot.h"
21 #include "adreno_gen7_17_0_snapshot.h"
22
23 struct a6xx_gpu_state_obj {
24 const void *handle;
25 u32 *data;
26 u32 count; /* optional, used when count potentially read from hw */
27 };
28
29 struct a6xx_gpu_state {
30 struct msm_gpu_state base;
31
32 struct a6xx_gpu_state_obj *gmu_registers;
33 int nr_gmu_registers;
34
35 struct a6xx_gpu_state_obj *registers;
36 int nr_registers;
37
38 struct a6xx_gpu_state_obj *shaders;
39 int nr_shaders;
40
41 struct a6xx_gpu_state_obj *clusters;
42 int nr_clusters;
43
44 struct a6xx_gpu_state_obj *dbgahb_clusters;
45 int nr_dbgahb_clusters;
46
47 struct a6xx_gpu_state_obj *indexed_regs;
48 int nr_indexed_regs;
49
50 struct a6xx_gpu_state_obj *debugbus;
51 int nr_debugbus;
52
53 struct a6xx_gpu_state_obj *vbif_debugbus;
54
55 struct a6xx_gpu_state_obj *cx_debugbus;
56 int nr_cx_debugbus;
57
58 struct msm_gpu_state_bo *gmu_log;
59 struct msm_gpu_state_bo *gmu_hfi;
60 struct msm_gpu_state_bo *gmu_debug;
61
62 s32 hfi_queue_history[HFI_MAX_QUEUES][HFI_HISTORY_SZ];
63
64 struct list_head objs;
65
66 bool gpu_initialized;
67 };
68
CRASHDUMP_WRITE(u64 * in,u32 reg,u32 val)69 static inline int CRASHDUMP_WRITE(u64 *in, u32 reg, u32 val)
70 {
71 in[0] = val;
72 in[1] = (((u64) reg) << 44 | (1 << 21) | 1);
73
74 return 2;
75 }
76
CRASHDUMP_READ(u64 * in,u32 reg,u32 dwords,u64 target)77 static inline int CRASHDUMP_READ(u64 *in, u32 reg, u32 dwords, u64 target)
78 {
79 in[0] = target;
80 in[1] = (((u64) reg) << 44 | dwords);
81
82 return 2;
83 }
84
CRASHDUMP_FINI(u64 * in)85 static inline int CRASHDUMP_FINI(u64 *in)
86 {
87 in[0] = 0;
88 in[1] = 0;
89
90 return 2;
91 }
92
93 struct a6xx_crashdumper {
94 void *ptr;
95 struct drm_gem_object *bo;
96 u64 iova;
97 };
98
99 struct a6xx_state_memobj {
100 struct list_head node;
101 unsigned long long data[];
102 };
103
state_kcalloc(struct a6xx_gpu_state * a6xx_state,int nr,size_t objsize)104 static void *state_kcalloc(struct a6xx_gpu_state *a6xx_state, int nr, size_t objsize)
105 {
106 struct a6xx_state_memobj *obj =
107 kvzalloc((nr * objsize) + sizeof(*obj), GFP_KERNEL);
108
109 if (!obj)
110 return NULL;
111
112 list_add_tail(&obj->node, &a6xx_state->objs);
113 return &obj->data;
114 }
115
state_kmemdup(struct a6xx_gpu_state * a6xx_state,void * src,size_t size)116 static void *state_kmemdup(struct a6xx_gpu_state *a6xx_state, void *src,
117 size_t size)
118 {
119 void *dst = state_kcalloc(a6xx_state, 1, size);
120
121 if (dst)
122 memcpy(dst, src, size);
123 return dst;
124 }
125
126 /*
127 * Allocate 1MB for the crashdumper scratch region - 8k for the script and
128 * the rest for the data
129 */
130 #define A6XX_CD_DATA_OFFSET 8192
131 #define A6XX_CD_DATA_SIZE (SZ_1M - 8192)
132
a6xx_crashdumper_init(struct msm_gpu * gpu,struct a6xx_crashdumper * dumper)133 static int a6xx_crashdumper_init(struct msm_gpu *gpu,
134 struct a6xx_crashdumper *dumper)
135 {
136 dumper->ptr = msm_gem_kernel_new(gpu->dev,
137 SZ_1M, MSM_BO_WC, gpu->vm,
138 &dumper->bo, &dumper->iova);
139
140 if (!IS_ERR(dumper->ptr))
141 msm_gem_object_set_name(dumper->bo, "crashdump");
142
143 return PTR_ERR_OR_ZERO(dumper->ptr);
144 }
145
a6xx_crashdumper_run(struct msm_gpu * gpu,struct a6xx_crashdumper * dumper)146 static int a6xx_crashdumper_run(struct msm_gpu *gpu,
147 struct a6xx_crashdumper *dumper)
148 {
149 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
150 struct a6xx_gpu *a6xx_gpu = to_a6xx_gpu(adreno_gpu);
151 u32 val;
152 int ret;
153
154 if (IS_ERR_OR_NULL(dumper->ptr))
155 return -EINVAL;
156
157 if (!a6xx_gmu_sptprac_is_on(&a6xx_gpu->gmu))
158 return -EINVAL;
159
160 /* Make sure all pending memory writes are posted */
161 wmb();
162
163 gpu_write64(gpu, REG_A6XX_CP_CRASH_DUMP_SCRIPT_BASE, dumper->iova);
164
165 gpu_write(gpu, REG_A6XX_CP_CRASH_DUMP_CNTL, 1);
166
167 ret = gpu_poll_timeout(gpu, REG_A6XX_CP_CRASH_DUMP_STATUS, val,
168 val & 0x02, 100, 10000);
169
170 gpu_write(gpu, REG_A6XX_CP_CRASH_DUMP_CNTL, 0);
171
172 return ret;
173 }
174
175 /* read a value from the GX debug bus */
debugbus_read(struct msm_gpu * gpu,u32 block,u32 offset,u32 * data)176 static int debugbus_read(struct msm_gpu *gpu, u32 block, u32 offset,
177 u32 *data)
178 {
179 u32 reg;
180
181 if (to_adreno_gpu(gpu)->info->family >= ADRENO_7XX_GEN1) {
182 reg = A7XX_DBGC_CFG_DBGBUS_SEL_D_PING_INDEX(offset) |
183 A7XX_DBGC_CFG_DBGBUS_SEL_D_PING_BLK_SEL(block);
184 } else {
185 reg = A6XX_DBGC_CFG_DBGBUS_SEL_D_PING_INDEX(offset) |
186 A6XX_DBGC_CFG_DBGBUS_SEL_D_PING_BLK_SEL(block);
187 }
188
189 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_SEL_A, reg);
190 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_SEL_B, reg);
191 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_SEL_C, reg);
192 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_SEL_D, reg);
193
194 /* Wait 1 us to make sure the data is flowing */
195 udelay(1);
196
197 data[0] = gpu_read(gpu, REG_A6XX_DBGC_CFG_DBGBUS_TRACE_BUF2);
198 data[1] = gpu_read(gpu, REG_A6XX_DBGC_CFG_DBGBUS_TRACE_BUF1);
199
200 return 2;
201 }
202
203 #define cxdbg_write(ptr, offset, val) \
204 writel((val), (ptr) + ((offset) << 2))
205
206 #define cxdbg_read(ptr, offset) \
207 readl((ptr) + ((offset) << 2))
208
209 /* read a value from the CX debug bus */
cx_debugbus_read(struct msm_gpu * gpu,void __iomem * cxdbg,u32 block,u32 offset,u32 * data)210 static int cx_debugbus_read(struct msm_gpu *gpu, void __iomem *cxdbg, u32 block, u32 offset,
211 u32 *data)
212 {
213 u32 reg;
214
215 if (to_adreno_gpu(gpu)->info->family >= ADRENO_7XX_GEN1) {
216 reg = A7XX_CX_DBGC_CFG_DBGBUS_SEL_A_PING_INDEX(offset) |
217 A7XX_CX_DBGC_CFG_DBGBUS_SEL_A_PING_BLK_SEL(block);
218 } else {
219 reg = A6XX_CX_DBGC_CFG_DBGBUS_SEL_A_PING_INDEX(offset) |
220 A6XX_CX_DBGC_CFG_DBGBUS_SEL_A_PING_BLK_SEL(block);
221 }
222
223 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_SEL_A, reg);
224 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_SEL_B, reg);
225 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_SEL_C, reg);
226 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_SEL_D, reg);
227
228 /* Wait 1 us to make sure the data is flowing */
229 udelay(1);
230
231 data[0] = cxdbg_read(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_TRACE_BUF2);
232 data[1] = cxdbg_read(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_TRACE_BUF1);
233
234 return 2;
235 }
236
237 /* Read a chunk of data from the VBIF debug bus */
vbif_debugbus_read(struct msm_gpu * gpu,u32 ctrl0,u32 ctrl1,u32 reg,int count,u32 * data)238 static int vbif_debugbus_read(struct msm_gpu *gpu, u32 ctrl0, u32 ctrl1,
239 u32 reg, int count, u32 *data)
240 {
241 int i;
242
243 gpu_write(gpu, ctrl0, reg);
244
245 for (i = 0; i < count; i++) {
246 gpu_write(gpu, ctrl1, i);
247 data[i] = gpu_read(gpu, REG_A6XX_VBIF_TEST_BUS_OUT);
248 }
249
250 return count;
251 }
252
253 #define AXI_ARB_BLOCKS 2
254 #define XIN_AXI_BLOCKS 5
255 #define XIN_CORE_BLOCKS 4
256
257 #define VBIF_DEBUGBUS_BLOCK_SIZE \
258 ((16 * AXI_ARB_BLOCKS) + \
259 (18 * XIN_AXI_BLOCKS) + \
260 (12 * XIN_CORE_BLOCKS))
261
a6xx_get_vbif_debugbus_block(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_gpu_state_obj * obj)262 static void a6xx_get_vbif_debugbus_block(struct msm_gpu *gpu,
263 struct a6xx_gpu_state *a6xx_state,
264 struct a6xx_gpu_state_obj *obj)
265 {
266 u32 clk, *ptr;
267 int i;
268
269 obj->data = state_kcalloc(a6xx_state, VBIF_DEBUGBUS_BLOCK_SIZE,
270 sizeof(u32));
271 if (!obj->data)
272 return;
273
274 obj->handle = NULL;
275
276 /* Get the current clock setting */
277 clk = gpu_read(gpu, REG_A6XX_VBIF_CLKON);
278
279 /* Force on the bus so we can read it */
280 gpu_write(gpu, REG_A6XX_VBIF_CLKON,
281 clk | A6XX_VBIF_CLKON_FORCE_ON_TESTBUS);
282
283 /* We will read from BUS2 first, so disable BUS1 */
284 gpu_write(gpu, REG_A6XX_VBIF_TEST_BUS1_CTRL0, 0);
285
286 /* Enable the VBIF bus for reading */
287 gpu_write(gpu, REG_A6XX_VBIF_TEST_BUS_OUT_CTRL, 1);
288
289 ptr = obj->data;
290
291 for (i = 0; i < AXI_ARB_BLOCKS; i++)
292 ptr += vbif_debugbus_read(gpu,
293 REG_A6XX_VBIF_TEST_BUS2_CTRL0,
294 REG_A6XX_VBIF_TEST_BUS2_CTRL1,
295 1 << (i + 16), 16, ptr);
296
297 for (i = 0; i < XIN_AXI_BLOCKS; i++)
298 ptr += vbif_debugbus_read(gpu,
299 REG_A6XX_VBIF_TEST_BUS2_CTRL0,
300 REG_A6XX_VBIF_TEST_BUS2_CTRL1,
301 1 << i, 18, ptr);
302
303 /* Stop BUS2 so we can turn on BUS1 */
304 gpu_write(gpu, REG_A6XX_VBIF_TEST_BUS2_CTRL0, 0);
305
306 for (i = 0; i < XIN_CORE_BLOCKS; i++)
307 ptr += vbif_debugbus_read(gpu,
308 REG_A6XX_VBIF_TEST_BUS1_CTRL0,
309 REG_A6XX_VBIF_TEST_BUS1_CTRL1,
310 1 << i, 12, ptr);
311
312 /* Restore the VBIF clock setting */
313 gpu_write(gpu, REG_A6XX_VBIF_CLKON, clk);
314 }
315
a6xx_get_debugbus_block(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_debugbus_block * block,struct a6xx_gpu_state_obj * obj)316 static void a6xx_get_debugbus_block(struct msm_gpu *gpu,
317 struct a6xx_gpu_state *a6xx_state,
318 const struct a6xx_debugbus_block *block,
319 struct a6xx_gpu_state_obj *obj)
320 {
321 int i;
322 u32 *ptr;
323
324 obj->data = state_kcalloc(a6xx_state, block->count, sizeof(u64));
325 if (!obj->data)
326 return;
327
328 obj->handle = block;
329
330 for (ptr = obj->data, i = 0; i < block->count; i++)
331 ptr += debugbus_read(gpu, block->id, i, ptr);
332 }
333
a6xx_get_cx_debugbus_block(struct msm_gpu * gpu,void __iomem * cxdbg,struct a6xx_gpu_state * a6xx_state,const struct a6xx_debugbus_block * block,struct a6xx_gpu_state_obj * obj)334 static void a6xx_get_cx_debugbus_block(struct msm_gpu *gpu,
335 void __iomem *cxdbg,
336 struct a6xx_gpu_state *a6xx_state,
337 const struct a6xx_debugbus_block *block,
338 struct a6xx_gpu_state_obj *obj)
339 {
340 int i;
341 u32 *ptr;
342
343 obj->data = state_kcalloc(a6xx_state, block->count, sizeof(u64));
344 if (!obj->data)
345 return;
346
347 obj->handle = block;
348
349 for (ptr = obj->data, i = 0; i < block->count; i++)
350 ptr += cx_debugbus_read(gpu, cxdbg, block->id, i, ptr);
351 }
352
a6xx_get_debugbus_blocks(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)353 static void a6xx_get_debugbus_blocks(struct msm_gpu *gpu,
354 struct a6xx_gpu_state *a6xx_state)
355 {
356 int nr_debugbus_blocks = ARRAY_SIZE(a6xx_debugbus_blocks) +
357 (a6xx_has_gbif(to_adreno_gpu(gpu)) ? 1 : 0);
358
359 if (adreno_is_a650_family(to_adreno_gpu(gpu)))
360 nr_debugbus_blocks += ARRAY_SIZE(a650_debugbus_blocks);
361
362 a6xx_state->debugbus = state_kcalloc(a6xx_state, nr_debugbus_blocks,
363 sizeof(*a6xx_state->debugbus));
364
365 if (a6xx_state->debugbus) {
366 int i, j;
367
368 for (i = 0; i < ARRAY_SIZE(a6xx_debugbus_blocks); i++)
369 a6xx_get_debugbus_block(gpu,
370 a6xx_state,
371 &a6xx_debugbus_blocks[i],
372 &a6xx_state->debugbus[i]);
373
374 /*
375 * GBIF has same debugbus as of other GPU blocks, fall back to
376 * default path if GPU uses GBIF, also GBIF uses exactly same
377 * ID as of VBIF.
378 */
379 if (a6xx_has_gbif(to_adreno_gpu(gpu))) {
380 a6xx_get_debugbus_block(gpu, a6xx_state,
381 &a6xx_gbif_debugbus_block,
382 &a6xx_state->debugbus[i]);
383
384 i++;
385 }
386
387
388 if (adreno_is_a650_family(to_adreno_gpu(gpu))) {
389 for (j = 0; j < ARRAY_SIZE(a650_debugbus_blocks); i++, j++)
390 a6xx_get_debugbus_block(gpu,
391 a6xx_state,
392 &a650_debugbus_blocks[j],
393 &a6xx_state->debugbus[i]);
394 }
395
396 a6xx_state->nr_debugbus = i;
397 }
398 }
399
a7xx_get_debugbus_blocks(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)400 static void a7xx_get_debugbus_blocks(struct msm_gpu *gpu,
401 struct a6xx_gpu_state *a6xx_state)
402 {
403 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
404 int debugbus_blocks_count, gbif_debugbus_blocks_count, total_debugbus_blocks;
405 const u32 *debugbus_blocks, *gbif_debugbus_blocks;
406 int i;
407
408 if (adreno_gpu->info->family == ADRENO_7XX_GEN1) {
409 if (adreno_is_a722(adreno_gpu)) {
410 debugbus_blocks = gen7_17_0_debugbus_blocks;
411 debugbus_blocks_count = ARRAY_SIZE(gen7_17_0_debugbus_blocks);
412 } else {
413 debugbus_blocks = gen7_0_0_debugbus_blocks;
414 debugbus_blocks_count = ARRAY_SIZE(gen7_0_0_debugbus_blocks);
415 }
416 gbif_debugbus_blocks = a7xx_gbif_debugbus_blocks;
417 gbif_debugbus_blocks_count = ARRAY_SIZE(a7xx_gbif_debugbus_blocks);
418 } else if (adreno_gpu->info->family == ADRENO_7XX_GEN2) {
419 debugbus_blocks = gen7_2_0_debugbus_blocks;
420 debugbus_blocks_count = ARRAY_SIZE(gen7_2_0_debugbus_blocks);
421 gbif_debugbus_blocks = a7xx_gbif_debugbus_blocks;
422 gbif_debugbus_blocks_count = ARRAY_SIZE(a7xx_gbif_debugbus_blocks);
423 } else {
424 BUG_ON(adreno_gpu->info->family != ADRENO_7XX_GEN3);
425 debugbus_blocks = gen7_9_0_debugbus_blocks;
426 debugbus_blocks_count = ARRAY_SIZE(gen7_9_0_debugbus_blocks);
427 gbif_debugbus_blocks = gen7_9_0_gbif_debugbus_blocks;
428 gbif_debugbus_blocks_count = ARRAY_SIZE(gen7_9_0_gbif_debugbus_blocks);
429 }
430
431 total_debugbus_blocks = debugbus_blocks_count + gbif_debugbus_blocks_count;
432
433 a6xx_state->debugbus = state_kcalloc(a6xx_state, total_debugbus_blocks,
434 sizeof(*a6xx_state->debugbus));
435
436 if (a6xx_state->debugbus) {
437 for (i = 0; i < debugbus_blocks_count; i++) {
438 a6xx_get_debugbus_block(gpu,
439 a6xx_state, &a7xx_debugbus_blocks[debugbus_blocks[i]],
440 &a6xx_state->debugbus[i]);
441 }
442
443 for (i = 0; i < gbif_debugbus_blocks_count; i++) {
444 a6xx_get_debugbus_block(gpu,
445 a6xx_state, &a7xx_debugbus_blocks[gbif_debugbus_blocks[i]],
446 &a6xx_state->debugbus[i + debugbus_blocks_count]);
447 }
448
449 a6xx_state->nr_debugbus = total_debugbus_blocks;
450 }
451 }
452
a6xx_get_debugbus(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)453 static void a6xx_get_debugbus(struct msm_gpu *gpu,
454 struct a6xx_gpu_state *a6xx_state)
455 {
456 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
457 struct resource *res;
458 void __iomem *cxdbg = NULL;
459
460 /* Set up the GX debug bus */
461
462 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_CNTLT,
463 A6XX_DBGC_CFG_DBGBUS_CNTLT_SEGT(0xf));
464
465 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_CNTLM,
466 A6XX_DBGC_CFG_DBGBUS_CNTLM_ENABLE(0xf));
467
468 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_IVTL_0, 0);
469 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_IVTL_1, 0);
470 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_IVTL_2, 0);
471 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_IVTL_3, 0);
472
473 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_BYTEL_0, 0x76543210);
474 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_BYTEL_1, 0xFEDCBA98);
475
476 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_MASKL_0, 0);
477 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_MASKL_1, 0);
478 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_MASKL_2, 0);
479 gpu_write(gpu, REG_A6XX_DBGC_CFG_DBGBUS_MASKL_3, 0);
480
481 /* Set up the CX debug bus - it lives elsewhere in the system so do a
482 * temporary ioremap for the registers
483 */
484 res = platform_get_resource_byname(gpu->pdev, IORESOURCE_MEM,
485 "cx_dbgc");
486
487 if (res)
488 cxdbg = ioremap(res->start, resource_size(res));
489
490 if (cxdbg) {
491 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_CNTLT,
492 A6XX_DBGC_CFG_DBGBUS_CNTLT_SEGT(0xf));
493
494 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_CNTLM,
495 A6XX_DBGC_CFG_DBGBUS_CNTLM_ENABLE(0xf));
496
497 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_IVTL_0, 0);
498 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_IVTL_1, 0);
499 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_IVTL_2, 0);
500 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_IVTL_3, 0);
501
502 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_BYTEL_0,
503 0x76543210);
504 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_BYTEL_1,
505 0xFEDCBA98);
506
507 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_MASKL_0, 0);
508 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_MASKL_1, 0);
509 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_MASKL_2, 0);
510 cxdbg_write(cxdbg, REG_A6XX_CX_DBGC_CFG_DBGBUS_MASKL_3, 0);
511 }
512
513 if (adreno_is_a7xx(adreno_gpu)) {
514 a7xx_get_debugbus_blocks(gpu, a6xx_state);
515 } else {
516 a6xx_get_debugbus_blocks(gpu, a6xx_state);
517 }
518
519 /* Dump the VBIF debugbus on applicable targets */
520 if (!a6xx_has_gbif(adreno_gpu)) {
521 a6xx_state->vbif_debugbus =
522 state_kcalloc(a6xx_state, 1,
523 sizeof(*a6xx_state->vbif_debugbus));
524
525 if (a6xx_state->vbif_debugbus)
526 a6xx_get_vbif_debugbus_block(gpu, a6xx_state,
527 a6xx_state->vbif_debugbus);
528 }
529
530 if (cxdbg) {
531 unsigned nr_cx_debugbus_blocks;
532 const struct a6xx_debugbus_block *cx_debugbus_blocks;
533
534 if (adreno_is_a7xx(adreno_gpu)) {
535 BUG_ON(adreno_gpu->info->family > ADRENO_7XX_GEN3);
536 cx_debugbus_blocks = a7xx_cx_debugbus_blocks;
537 nr_cx_debugbus_blocks = ARRAY_SIZE(a7xx_cx_debugbus_blocks);
538 } else {
539 cx_debugbus_blocks = a6xx_cx_debugbus_blocks;
540 nr_cx_debugbus_blocks = ARRAY_SIZE(a6xx_cx_debugbus_blocks);
541 }
542
543 a6xx_state->cx_debugbus =
544 state_kcalloc(a6xx_state,
545 nr_cx_debugbus_blocks,
546 sizeof(*a6xx_state->cx_debugbus));
547
548 if (a6xx_state->cx_debugbus) {
549 int i;
550
551 for (i = 0; i < nr_cx_debugbus_blocks; i++)
552 a6xx_get_cx_debugbus_block(gpu,
553 cxdbg,
554 a6xx_state,
555 &cx_debugbus_blocks[i],
556 &a6xx_state->cx_debugbus[i]);
557
558 a6xx_state->nr_cx_debugbus =
559 nr_cx_debugbus_blocks;
560 }
561
562 iounmap(cxdbg);
563 }
564 }
565
566 #define RANGE(reg, a) ((reg)[(a) + 1] - (reg)[(a)] + 1)
567
568 /* Read a data cluster from behind the AHB aperture */
a6xx_get_dbgahb_cluster(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_dbgahb_cluster * dbgahb,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)569 static void a6xx_get_dbgahb_cluster(struct msm_gpu *gpu,
570 struct a6xx_gpu_state *a6xx_state,
571 const struct a6xx_dbgahb_cluster *dbgahb,
572 struct a6xx_gpu_state_obj *obj,
573 struct a6xx_crashdumper *dumper)
574 {
575 u64 *in = dumper->ptr;
576 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
577 size_t datasize;
578 int i, regcount = 0;
579
580 for (i = 0; i < A6XX_NUM_CONTEXTS; i++) {
581 int j;
582
583 in += CRASHDUMP_WRITE(in, REG_A6XX_HLSQ_DBG_READ_SEL,
584 (dbgahb->statetype + i * 2) << 8);
585
586 for (j = 0; j < dbgahb->count; j += 2) {
587 int count = RANGE(dbgahb->registers, j);
588 u32 offset = REG_A6XX_HLSQ_DBG_AHB_READ_APERTURE +
589 dbgahb->registers[j] - (dbgahb->base >> 2);
590
591 in += CRASHDUMP_READ(in, offset, count, out);
592
593 out += count * sizeof(u32);
594
595 if (i == 0)
596 regcount += count;
597 }
598 }
599
600 CRASHDUMP_FINI(in);
601
602 datasize = regcount * A6XX_NUM_CONTEXTS * sizeof(u32);
603
604 if (WARN_ON(datasize > A6XX_CD_DATA_SIZE))
605 return;
606
607 if (a6xx_crashdumper_run(gpu, dumper))
608 return;
609
610 obj->handle = dbgahb;
611 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
612 datasize);
613 }
614
a7xx_get_dbgahb_cluster(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct gen7_sptp_cluster_registers * dbgahb,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)615 static void a7xx_get_dbgahb_cluster(struct msm_gpu *gpu,
616 struct a6xx_gpu_state *a6xx_state,
617 const struct gen7_sptp_cluster_registers *dbgahb,
618 struct a6xx_gpu_state_obj *obj,
619 struct a6xx_crashdumper *dumper)
620 {
621 u64 *in = dumper->ptr;
622 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
623 size_t datasize;
624 int i, regcount = 0;
625
626 in += CRASHDUMP_WRITE(in, REG_A7XX_SP_READ_SEL,
627 A7XX_SP_READ_SEL_LOCATION(dbgahb->location_id) |
628 A7XX_SP_READ_SEL_PIPE(dbgahb->pipe_id) |
629 A7XX_SP_READ_SEL_STATETYPE(dbgahb->statetype));
630
631 for (i = 0; dbgahb->regs[i] != UINT_MAX; i += 2) {
632 int count = RANGE(dbgahb->regs, i);
633 u32 offset = REG_A7XX_SP_AHB_READ_APERTURE +
634 dbgahb->regs[i] - dbgahb->regbase;
635
636 in += CRASHDUMP_READ(in, offset, count, out);
637
638 out += count * sizeof(u32);
639 regcount += count;
640 }
641
642 CRASHDUMP_FINI(in);
643
644 datasize = regcount * sizeof(u32);
645
646 if (WARN_ON(datasize > A6XX_CD_DATA_SIZE))
647 return;
648
649 if (a6xx_crashdumper_run(gpu, dumper))
650 return;
651
652 obj->handle = dbgahb;
653 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
654 datasize);
655 }
656
a6xx_get_dbgahb_clusters(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)657 static void a6xx_get_dbgahb_clusters(struct msm_gpu *gpu,
658 struct a6xx_gpu_state *a6xx_state,
659 struct a6xx_crashdumper *dumper)
660 {
661 int i;
662
663 a6xx_state->dbgahb_clusters = state_kcalloc(a6xx_state,
664 ARRAY_SIZE(a6xx_dbgahb_clusters),
665 sizeof(*a6xx_state->dbgahb_clusters));
666
667 if (!a6xx_state->dbgahb_clusters)
668 return;
669
670 a6xx_state->nr_dbgahb_clusters = ARRAY_SIZE(a6xx_dbgahb_clusters);
671
672 for (i = 0; i < ARRAY_SIZE(a6xx_dbgahb_clusters); i++)
673 a6xx_get_dbgahb_cluster(gpu, a6xx_state,
674 &a6xx_dbgahb_clusters[i],
675 &a6xx_state->dbgahb_clusters[i], dumper);
676 }
677
a7xx_get_dbgahb_clusters(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)678 static void a7xx_get_dbgahb_clusters(struct msm_gpu *gpu,
679 struct a6xx_gpu_state *a6xx_state,
680 struct a6xx_crashdumper *dumper)
681 {
682 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
683 int i;
684 const struct gen7_sptp_cluster_registers *dbgahb_clusters;
685 unsigned dbgahb_clusters_size;
686
687 if (adreno_gpu->info->family == ADRENO_7XX_GEN1) {
688 if (adreno_is_a722(adreno_gpu)) {
689 dbgahb_clusters = gen7_17_0_sptp_clusters;
690 dbgahb_clusters_size = ARRAY_SIZE(gen7_17_0_sptp_clusters);
691 } else {
692 dbgahb_clusters = gen7_0_0_sptp_clusters;
693 dbgahb_clusters_size = ARRAY_SIZE(gen7_0_0_sptp_clusters);
694 }
695 } else if (adreno_gpu->info->family == ADRENO_7XX_GEN2) {
696 dbgahb_clusters = gen7_2_0_sptp_clusters;
697 dbgahb_clusters_size = ARRAY_SIZE(gen7_2_0_sptp_clusters);
698 } else {
699 BUG_ON(adreno_gpu->info->family != ADRENO_7XX_GEN3);
700 dbgahb_clusters = gen7_9_0_sptp_clusters;
701 dbgahb_clusters_size = ARRAY_SIZE(gen7_9_0_sptp_clusters);
702 }
703
704 a6xx_state->dbgahb_clusters = state_kcalloc(a6xx_state,
705 dbgahb_clusters_size,
706 sizeof(*a6xx_state->dbgahb_clusters));
707
708 if (!a6xx_state->dbgahb_clusters)
709 return;
710
711 a6xx_state->nr_dbgahb_clusters = dbgahb_clusters_size;
712
713 for (i = 0; i < dbgahb_clusters_size; i++)
714 a7xx_get_dbgahb_cluster(gpu, a6xx_state,
715 &dbgahb_clusters[i],
716 &a6xx_state->dbgahb_clusters[i], dumper);
717 }
718
719 /* Read a data cluster from the CP aperture with the crashdumper */
a6xx_get_cluster(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_cluster * cluster,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)720 static void a6xx_get_cluster(struct msm_gpu *gpu,
721 struct a6xx_gpu_state *a6xx_state,
722 const struct a6xx_cluster *cluster,
723 struct a6xx_gpu_state_obj *obj,
724 struct a6xx_crashdumper *dumper)
725 {
726 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
727 u64 *in = dumper->ptr;
728 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
729 size_t datasize;
730 int i, regcount = 0;
731 u32 id = cluster->id;
732
733 /* Skip registers that are not present on older generation */
734 if (!adreno_is_a660_family(adreno_gpu) &&
735 cluster->registers == a660_fe_cluster)
736 return;
737
738 if (adreno_is_a650_family(adreno_gpu) &&
739 cluster->registers == a6xx_ps_cluster)
740 id = CLUSTER_VPC_PS;
741
742 /* Some clusters need a selector register to be programmed too */
743 if (cluster->sel_reg)
744 in += CRASHDUMP_WRITE(in, cluster->sel_reg, cluster->sel_val);
745
746 for (i = 0; i < A6XX_NUM_CONTEXTS; i++) {
747 int j;
748
749 in += CRASHDUMP_WRITE(in, REG_A6XX_CP_APERTURE_CNTL_CD,
750 (id << 8) | (i << 4) | i);
751
752 for (j = 0; j < cluster->count; j += 2) {
753 int count = RANGE(cluster->registers, j);
754
755 in += CRASHDUMP_READ(in, cluster->registers[j],
756 count, out);
757
758 out += count * sizeof(u32);
759
760 if (i == 0)
761 regcount += count;
762 }
763 }
764
765 CRASHDUMP_FINI(in);
766
767 datasize = regcount * A6XX_NUM_CONTEXTS * sizeof(u32);
768
769 if (WARN_ON(datasize > A6XX_CD_DATA_SIZE))
770 return;
771
772 if (a6xx_crashdumper_run(gpu, dumper))
773 return;
774
775 obj->handle = cluster;
776 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
777 datasize);
778 }
779
a7xx_get_cluster(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct gen7_cluster_registers * cluster,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)780 static void a7xx_get_cluster(struct msm_gpu *gpu,
781 struct a6xx_gpu_state *a6xx_state,
782 const struct gen7_cluster_registers *cluster,
783 struct a6xx_gpu_state_obj *obj,
784 struct a6xx_crashdumper *dumper)
785 {
786 u64 *in = dumper->ptr;
787 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
788 size_t datasize;
789 int i, regcount = 0;
790
791 in += CRASHDUMP_WRITE(in, REG_A7XX_CP_APERTURE_CNTL_CD,
792 A7XX_CP_APERTURE_CNTL_CD_PIPE(cluster->pipe_id) |
793 A7XX_CP_APERTURE_CNTL_CD_CLUSTER(cluster->cluster_id) |
794 A7XX_CP_APERTURE_CNTL_CD_CONTEXT(cluster->context_id));
795
796 /* Some clusters need a selector register to be programmed too */
797 if (cluster->sel)
798 in += CRASHDUMP_WRITE(in, cluster->sel->cd_reg, cluster->sel->val);
799
800 for (i = 0; cluster->regs[i] != UINT_MAX; i += 2) {
801 int count = RANGE(cluster->regs, i);
802
803 in += CRASHDUMP_READ(in, cluster->regs[i],
804 count, out);
805
806 out += count * sizeof(u32);
807 regcount += count;
808 }
809
810 CRASHDUMP_FINI(in);
811
812 datasize = regcount * sizeof(u32);
813
814 if (WARN_ON(datasize > A6XX_CD_DATA_SIZE))
815 return;
816
817 if (a6xx_crashdumper_run(gpu, dumper))
818 return;
819
820 obj->handle = cluster;
821 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
822 datasize);
823 }
824
a6xx_get_clusters(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)825 static void a6xx_get_clusters(struct msm_gpu *gpu,
826 struct a6xx_gpu_state *a6xx_state,
827 struct a6xx_crashdumper *dumper)
828 {
829 int i;
830
831 a6xx_state->clusters = state_kcalloc(a6xx_state,
832 ARRAY_SIZE(a6xx_clusters), sizeof(*a6xx_state->clusters));
833
834 if (!a6xx_state->clusters)
835 return;
836
837 a6xx_state->nr_clusters = ARRAY_SIZE(a6xx_clusters);
838
839 for (i = 0; i < ARRAY_SIZE(a6xx_clusters); i++)
840 a6xx_get_cluster(gpu, a6xx_state, &a6xx_clusters[i],
841 &a6xx_state->clusters[i], dumper);
842 }
843
a7xx_get_clusters(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)844 static void a7xx_get_clusters(struct msm_gpu *gpu,
845 struct a6xx_gpu_state *a6xx_state,
846 struct a6xx_crashdumper *dumper)
847 {
848 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
849 int i;
850 const struct gen7_cluster_registers *clusters;
851 unsigned clusters_size;
852
853 if (adreno_gpu->info->family == ADRENO_7XX_GEN1) {
854 if (adreno_is_a722(adreno_gpu)) {
855 clusters = gen7_17_0_clusters;
856 clusters_size = ARRAY_SIZE(gen7_17_0_clusters);
857 } else {
858 clusters = gen7_0_0_clusters;
859 clusters_size = ARRAY_SIZE(gen7_0_0_clusters);
860 }
861 } else if (adreno_gpu->info->family == ADRENO_7XX_GEN2) {
862 clusters = gen7_2_0_clusters;
863 clusters_size = ARRAY_SIZE(gen7_2_0_clusters);
864 } else {
865 BUG_ON(adreno_gpu->info->family != ADRENO_7XX_GEN3);
866 clusters = gen7_9_0_clusters;
867 clusters_size = ARRAY_SIZE(gen7_9_0_clusters);
868 }
869
870 a6xx_state->clusters = state_kcalloc(a6xx_state,
871 clusters_size, sizeof(*a6xx_state->clusters));
872
873 if (!a6xx_state->clusters)
874 return;
875
876 a6xx_state->nr_clusters = clusters_size;
877
878 for (i = 0; i < clusters_size; i++)
879 a7xx_get_cluster(gpu, a6xx_state, &clusters[i],
880 &a6xx_state->clusters[i], dumper);
881 }
882
883 /* Read a shader / debug block from the HLSQ aperture with the crashdumper */
a6xx_get_shader_block(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_shader_block * block,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)884 static void a6xx_get_shader_block(struct msm_gpu *gpu,
885 struct a6xx_gpu_state *a6xx_state,
886 const struct a6xx_shader_block *block,
887 struct a6xx_gpu_state_obj *obj,
888 struct a6xx_crashdumper *dumper)
889 {
890 u64 *in = dumper->ptr;
891 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
892 size_t datasize = block->size * A6XX_NUM_SHADER_BANKS * sizeof(u32);
893 int i;
894
895 if (WARN_ON(datasize > A6XX_CD_DATA_SIZE))
896 return;
897
898 for (i = 0; i < A6XX_NUM_SHADER_BANKS; i++) {
899 in += CRASHDUMP_WRITE(in, REG_A6XX_HLSQ_DBG_READ_SEL,
900 (block->type << 8) | i);
901
902 in += CRASHDUMP_READ(in, REG_A6XX_HLSQ_DBG_AHB_READ_APERTURE,
903 block->size, out);
904
905 out += block->size * sizeof(u32);
906 }
907
908 CRASHDUMP_FINI(in);
909
910 if (a6xx_crashdumper_run(gpu, dumper))
911 return;
912
913 obj->handle = block;
914 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
915 datasize);
916 }
917
a7xx_get_shader_block(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct gen7_shader_block * block,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)918 static void a7xx_get_shader_block(struct msm_gpu *gpu,
919 struct a6xx_gpu_state *a6xx_state,
920 const struct gen7_shader_block *block,
921 struct a6xx_gpu_state_obj *obj,
922 struct a6xx_crashdumper *dumper)
923 {
924 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
925 u64 *in = dumper->ptr;
926 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
927 size_t datasize = block->size * block->num_sps * block->num_usptps * sizeof(u32);
928 int i, j;
929
930 if (WARN_ON(datasize > A6XX_CD_DATA_SIZE))
931 return;
932
933 if (adreno_gpu->info->family == ADRENO_7XX_GEN1) {
934 gpu_rmw(gpu, REG_A7XX_SP_DBG_CNTL, GENMASK(1, 0), 3);
935 }
936
937 for (i = 0; i < block->num_sps; i++) {
938 for (j = 0; j < block->num_usptps; j++) {
939 in += CRASHDUMP_WRITE(in, REG_A7XX_SP_READ_SEL,
940 A7XX_SP_READ_SEL_LOCATION(block->location) |
941 A7XX_SP_READ_SEL_PIPE(block->pipeid) |
942 A7XX_SP_READ_SEL_STATETYPE(block->statetype) |
943 A7XX_SP_READ_SEL_USPTP(j) |
944 A7XX_SP_READ_SEL_SPTP(i));
945
946 in += CRASHDUMP_READ(in, REG_A7XX_SP_AHB_READ_APERTURE,
947 block->size, out);
948
949 out += block->size * sizeof(u32);
950 }
951 }
952
953 CRASHDUMP_FINI(in);
954
955 if (a6xx_crashdumper_run(gpu, dumper))
956 goto out;
957
958 obj->handle = block;
959 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
960 datasize);
961
962 out:
963 if (adreno_gpu->info->family == ADRENO_7XX_GEN1) {
964 gpu_rmw(gpu, REG_A7XX_SP_DBG_CNTL, GENMASK(1, 0), 0);
965 }
966 }
967
a6xx_get_shaders(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)968 static void a6xx_get_shaders(struct msm_gpu *gpu,
969 struct a6xx_gpu_state *a6xx_state,
970 struct a6xx_crashdumper *dumper)
971 {
972 int i;
973
974 a6xx_state->shaders = state_kcalloc(a6xx_state,
975 ARRAY_SIZE(a6xx_shader_blocks), sizeof(*a6xx_state->shaders));
976
977 if (!a6xx_state->shaders)
978 return;
979
980 a6xx_state->nr_shaders = ARRAY_SIZE(a6xx_shader_blocks);
981
982 for (i = 0; i < ARRAY_SIZE(a6xx_shader_blocks); i++)
983 a6xx_get_shader_block(gpu, a6xx_state, &a6xx_shader_blocks[i],
984 &a6xx_state->shaders[i], dumper);
985 }
986
a7xx_get_shaders(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)987 static void a7xx_get_shaders(struct msm_gpu *gpu,
988 struct a6xx_gpu_state *a6xx_state,
989 struct a6xx_crashdumper *dumper)
990 {
991 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
992 const struct gen7_shader_block *shader_blocks;
993 unsigned num_shader_blocks;
994 int i;
995
996 if (adreno_gpu->info->family == ADRENO_7XX_GEN1) {
997 if (adreno_is_a722(adreno_gpu)) {
998 shader_blocks = gen7_17_0_shader_blocks;
999 num_shader_blocks = ARRAY_SIZE(gen7_17_0_shader_blocks);
1000 } else {
1001 shader_blocks = gen7_0_0_shader_blocks;
1002 num_shader_blocks = ARRAY_SIZE(gen7_0_0_shader_blocks);
1003 }
1004 } else if (adreno_gpu->info->family == ADRENO_7XX_GEN2) {
1005 shader_blocks = gen7_2_0_shader_blocks;
1006 num_shader_blocks = ARRAY_SIZE(gen7_2_0_shader_blocks);
1007 } else {
1008 BUG_ON(adreno_gpu->info->family != ADRENO_7XX_GEN3);
1009 shader_blocks = gen7_9_0_shader_blocks;
1010 num_shader_blocks = ARRAY_SIZE(gen7_9_0_shader_blocks);
1011 }
1012
1013 a6xx_state->shaders = state_kcalloc(a6xx_state,
1014 num_shader_blocks, sizeof(*a6xx_state->shaders));
1015
1016 if (!a6xx_state->shaders)
1017 return;
1018
1019 a6xx_state->nr_shaders = num_shader_blocks;
1020
1021 for (i = 0; i < num_shader_blocks; i++)
1022 a7xx_get_shader_block(gpu, a6xx_state, &shader_blocks[i],
1023 &a6xx_state->shaders[i], dumper);
1024 }
1025
1026 /* Read registers from behind the HLSQ aperture with the crashdumper */
a6xx_get_crashdumper_hlsq_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_registers * regs,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)1027 static void a6xx_get_crashdumper_hlsq_registers(struct msm_gpu *gpu,
1028 struct a6xx_gpu_state *a6xx_state,
1029 const struct a6xx_registers *regs,
1030 struct a6xx_gpu_state_obj *obj,
1031 struct a6xx_crashdumper *dumper)
1032
1033 {
1034 u64 *in = dumper->ptr;
1035 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
1036 int i, regcount = 0;
1037
1038 in += CRASHDUMP_WRITE(in, REG_A6XX_HLSQ_DBG_READ_SEL, (regs->val1 & 0xff) << 8);
1039
1040 for (i = 0; i < regs->count; i += 2) {
1041 u32 count = RANGE(regs->registers, i);
1042 u32 offset = REG_A6XX_HLSQ_DBG_AHB_READ_APERTURE +
1043 regs->registers[i] - (regs->val0 >> 2);
1044
1045 in += CRASHDUMP_READ(in, offset, count, out);
1046
1047 out += count * sizeof(u32);
1048 regcount += count;
1049 }
1050
1051 CRASHDUMP_FINI(in);
1052
1053 if (WARN_ON((regcount * sizeof(u32)) > A6XX_CD_DATA_SIZE))
1054 return;
1055
1056 if (a6xx_crashdumper_run(gpu, dumper))
1057 return;
1058
1059 obj->handle = regs;
1060 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
1061 regcount * sizeof(u32));
1062 }
1063
1064 /* Read a block of registers using the crashdumper */
a6xx_get_crashdumper_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_registers * regs,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)1065 static void a6xx_get_crashdumper_registers(struct msm_gpu *gpu,
1066 struct a6xx_gpu_state *a6xx_state,
1067 const struct a6xx_registers *regs,
1068 struct a6xx_gpu_state_obj *obj,
1069 struct a6xx_crashdumper *dumper)
1070
1071 {
1072 u64 *in = dumper->ptr;
1073 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
1074 int i, regcount = 0;
1075
1076 /* Skip unsupported registers on older generations */
1077 if (!adreno_is_a660_family(to_adreno_gpu(gpu)) &&
1078 (regs->registers == a660_registers))
1079 return;
1080
1081 /* Some blocks might need to program a selector register first */
1082 if (regs->val0)
1083 in += CRASHDUMP_WRITE(in, regs->val0, regs->val1);
1084
1085 for (i = 0; i < regs->count; i += 2) {
1086 u32 count = RANGE(regs->registers, i);
1087
1088 in += CRASHDUMP_READ(in, regs->registers[i], count, out);
1089
1090 out += count * sizeof(u32);
1091 regcount += count;
1092 }
1093
1094 CRASHDUMP_FINI(in);
1095
1096 if (WARN_ON((regcount * sizeof(u32)) > A6XX_CD_DATA_SIZE))
1097 return;
1098
1099 if (a6xx_crashdumper_run(gpu, dumper))
1100 return;
1101
1102 obj->handle = regs;
1103 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
1104 regcount * sizeof(u32));
1105 }
1106
a7xx_get_crashdumper_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct gen7_reg_list * regs,struct a6xx_gpu_state_obj * obj,struct a6xx_crashdumper * dumper)1107 static void a7xx_get_crashdumper_registers(struct msm_gpu *gpu,
1108 struct a6xx_gpu_state *a6xx_state,
1109 const struct gen7_reg_list *regs,
1110 struct a6xx_gpu_state_obj *obj,
1111 struct a6xx_crashdumper *dumper)
1112
1113 {
1114 u64 *in = dumper->ptr;
1115 u64 out = dumper->iova + A6XX_CD_DATA_OFFSET;
1116 int i, regcount = 0;
1117
1118 /* Some blocks might need to program a selector register first */
1119 if (regs->sel)
1120 in += CRASHDUMP_WRITE(in, regs->sel->cd_reg, regs->sel->val);
1121
1122 for (i = 0; regs->regs[i] != UINT_MAX; i += 2) {
1123 u32 count = RANGE(regs->regs, i);
1124
1125 in += CRASHDUMP_READ(in, regs->regs[i], count, out);
1126
1127 out += count * sizeof(u32);
1128 regcount += count;
1129 }
1130
1131 CRASHDUMP_FINI(in);
1132
1133 if (WARN_ON((regcount * sizeof(u32)) > A6XX_CD_DATA_SIZE))
1134 return;
1135
1136 if (a6xx_crashdumper_run(gpu, dumper))
1137 return;
1138
1139 obj->handle = regs->regs;
1140 obj->data = state_kmemdup(a6xx_state, dumper->ptr + A6XX_CD_DATA_OFFSET,
1141 regcount * sizeof(u32));
1142 }
1143
1144
1145 /* Read a block of registers via AHB */
a6xx_get_ahb_gpu_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_registers * regs,struct a6xx_gpu_state_obj * obj)1146 static void a6xx_get_ahb_gpu_registers(struct msm_gpu *gpu,
1147 struct a6xx_gpu_state *a6xx_state,
1148 const struct a6xx_registers *regs,
1149 struct a6xx_gpu_state_obj *obj)
1150 {
1151 int i, regcount = 0, index = 0;
1152
1153 /* Skip unsupported registers on older generations */
1154 if (!adreno_is_a660_family(to_adreno_gpu(gpu)) &&
1155 (regs->registers == a660_registers))
1156 return;
1157
1158 for (i = 0; i < regs->count; i += 2)
1159 regcount += RANGE(regs->registers, i);
1160
1161 obj->handle = (const void *) regs;
1162 obj->data = state_kcalloc(a6xx_state, regcount, sizeof(u32));
1163 if (!obj->data)
1164 return;
1165
1166 for (i = 0; i < regs->count; i += 2) {
1167 u32 count = RANGE(regs->registers, i);
1168 int j;
1169
1170 for (j = 0; j < count; j++)
1171 obj->data[index++] = gpu_read(gpu,
1172 regs->registers[i] + j);
1173 }
1174 }
1175
a7xx_get_ahb_gpu_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const u32 * regs,struct a6xx_gpu_state_obj * obj)1176 static void a7xx_get_ahb_gpu_registers(struct msm_gpu *gpu,
1177 struct a6xx_gpu_state *a6xx_state,
1178 const u32 *regs,
1179 struct a6xx_gpu_state_obj *obj)
1180 {
1181 int i, regcount = 0, index = 0;
1182
1183 for (i = 0; regs[i] != UINT_MAX; i += 2)
1184 regcount += RANGE(regs, i);
1185
1186 obj->handle = (const void *) regs;
1187 obj->data = state_kcalloc(a6xx_state, regcount, sizeof(u32));
1188 if (!obj->data)
1189 return;
1190
1191 for (i = 0; regs[i] != UINT_MAX; i += 2) {
1192 u32 count = RANGE(regs, i);
1193 int j;
1194
1195 for (j = 0; j < count; j++)
1196 obj->data[index++] = gpu_read(gpu, regs[i] + j);
1197 }
1198 }
1199
a7xx_get_ahb_gpu_reglist(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct gen7_reg_list * regs,struct a6xx_gpu_state_obj * obj)1200 static void a7xx_get_ahb_gpu_reglist(struct msm_gpu *gpu,
1201 struct a6xx_gpu_state *a6xx_state,
1202 const struct gen7_reg_list *regs,
1203 struct a6xx_gpu_state_obj *obj)
1204 {
1205 if (regs->sel)
1206 gpu_write(gpu, regs->sel->host_reg, regs->sel->val);
1207
1208 a7xx_get_ahb_gpu_registers(gpu, a6xx_state, regs->regs, obj);
1209 }
1210
1211 /* Read a block of GMU registers */
_a6xx_get_gmu_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_registers * regs,struct a6xx_gpu_state_obj * obj,bool rscc)1212 static void _a6xx_get_gmu_registers(struct msm_gpu *gpu,
1213 struct a6xx_gpu_state *a6xx_state,
1214 const struct a6xx_registers *regs,
1215 struct a6xx_gpu_state_obj *obj,
1216 bool rscc)
1217 {
1218 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1219 struct a6xx_gpu *a6xx_gpu = to_a6xx_gpu(adreno_gpu);
1220 struct a6xx_gmu *gmu = &a6xx_gpu->gmu;
1221 int i, regcount = 0, index = 0;
1222
1223 for (i = 0; i < regs->count; i += 2)
1224 regcount += RANGE(regs->registers, i);
1225
1226 obj->handle = (const void *) regs;
1227 obj->data = state_kcalloc(a6xx_state, regcount, sizeof(u32));
1228 if (!obj->data)
1229 return;
1230
1231 for (i = 0; i < regs->count; i += 2) {
1232 u32 count = RANGE(regs->registers, i);
1233 int j;
1234
1235 for (j = 0; j < count; j++) {
1236 u32 offset = regs->registers[i] + j;
1237 u32 val;
1238
1239 if (rscc)
1240 val = gmu_read_rscc(gmu, offset);
1241 else
1242 val = gmu_read(gmu, offset);
1243
1244 obj->data[index++] = val;
1245 }
1246 }
1247 }
1248
a6xx_get_gmu_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)1249 static void a6xx_get_gmu_registers(struct msm_gpu *gpu,
1250 struct a6xx_gpu_state *a6xx_state)
1251 {
1252 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1253 struct a6xx_gpu *a6xx_gpu = to_a6xx_gpu(adreno_gpu);
1254
1255 a6xx_state->gmu_registers = state_kcalloc(a6xx_state,
1256 4, sizeof(*a6xx_state->gmu_registers));
1257
1258 if (!a6xx_state->gmu_registers)
1259 return;
1260
1261 a6xx_state->nr_gmu_registers = 4;
1262
1263 /* Get the CX GMU registers from AHB */
1264 _a6xx_get_gmu_registers(gpu, a6xx_state, &a6xx_gmu_reglist[0],
1265 &a6xx_state->gmu_registers[0], false);
1266 _a6xx_get_gmu_registers(gpu, a6xx_state, &a6xx_gmu_reglist[1],
1267 &a6xx_state->gmu_registers[1], true);
1268
1269 if (adreno_is_a621(adreno_gpu))
1270 _a6xx_get_gmu_registers(gpu, a6xx_state, &a621_gpucc_reg,
1271 &a6xx_state->gmu_registers[2], false);
1272 else if (adreno_is_a623(adreno_gpu) || adreno_is_a663(adreno_gpu))
1273 _a6xx_get_gmu_registers(gpu, a6xx_state, &a623_gpucc_reg,
1274 &a6xx_state->gmu_registers[2], false);
1275 else
1276 _a6xx_get_gmu_registers(gpu, a6xx_state, &a6xx_gpucc_reg,
1277 &a6xx_state->gmu_registers[2], false);
1278
1279 if (!adreno_gpu->funcs->gx_is_on(adreno_gpu))
1280 return;
1281
1282 /* Set the fence to ALLOW mode so we can access the registers */
1283 gmu_write(&a6xx_gpu->gmu, REG_A6XX_GMU_AO_AHB_FENCE_CTRL, 0);
1284
1285 _a6xx_get_gmu_registers(gpu, a6xx_state, &a6xx_gmu_reglist[2],
1286 &a6xx_state->gmu_registers[3], false);
1287 }
1288
a6xx_snapshot_gmu_bo(struct a6xx_gpu_state * a6xx_state,struct a6xx_gmu_bo * bo)1289 static struct msm_gpu_state_bo *a6xx_snapshot_gmu_bo(
1290 struct a6xx_gpu_state *a6xx_state, struct a6xx_gmu_bo *bo)
1291 {
1292 struct msm_gpu_state_bo *snapshot;
1293
1294 if (!bo->size)
1295 return NULL;
1296
1297 snapshot = state_kcalloc(a6xx_state, 1, sizeof(*snapshot));
1298 if (!snapshot)
1299 return NULL;
1300
1301 snapshot->iova = bo->iova;
1302 snapshot->size = bo->size;
1303 snapshot->data = kvzalloc(snapshot->size, GFP_KERNEL);
1304 if (!snapshot->data)
1305 return NULL;
1306
1307 memcpy(snapshot->data, bo->virt, bo->size);
1308
1309 return snapshot;
1310 }
1311
a6xx_snapshot_gmu_hfi_history(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)1312 static void a6xx_snapshot_gmu_hfi_history(struct msm_gpu *gpu,
1313 struct a6xx_gpu_state *a6xx_state)
1314 {
1315 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1316 struct a6xx_gpu *a6xx_gpu = to_a6xx_gpu(adreno_gpu);
1317 struct a6xx_gmu *gmu = &a6xx_gpu->gmu;
1318 unsigned i, j;
1319
1320 BUILD_BUG_ON(ARRAY_SIZE(gmu->queues) != ARRAY_SIZE(a6xx_state->hfi_queue_history));
1321
1322 for (i = 0; i < ARRAY_SIZE(gmu->queues); i++) {
1323 struct a6xx_hfi_queue *queue = &gmu->queues[i];
1324 for (j = 0; j < HFI_HISTORY_SZ; j++) {
1325 unsigned idx = (j + queue->history_idx) % HFI_HISTORY_SZ;
1326 a6xx_state->hfi_queue_history[i][j] = queue->history[idx];
1327 }
1328 }
1329 }
1330
1331 #define A6XX_REGLIST_SIZE 1
1332 #define A6XX_GBIF_REGLIST_SIZE 1
a6xx_get_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)1333 static void a6xx_get_registers(struct msm_gpu *gpu,
1334 struct a6xx_gpu_state *a6xx_state,
1335 struct a6xx_crashdumper *dumper)
1336 {
1337 int i, count = A6XX_REGLIST_SIZE +
1338 ARRAY_SIZE(a6xx_reglist) +
1339 ARRAY_SIZE(a6xx_hlsq_reglist) + A6XX_GBIF_REGLIST_SIZE;
1340 int index = 0;
1341 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1342
1343 a6xx_state->registers = state_kcalloc(a6xx_state,
1344 count, sizeof(*a6xx_state->registers));
1345
1346 if (!a6xx_state->registers)
1347 return;
1348
1349 a6xx_state->nr_registers = count;
1350
1351 a6xx_get_ahb_gpu_registers(gpu,
1352 a6xx_state, &a6xx_ahb_reglist,
1353 &a6xx_state->registers[index++]);
1354
1355 if (a6xx_has_gbif(adreno_gpu))
1356 a6xx_get_ahb_gpu_registers(gpu,
1357 a6xx_state, &a6xx_gbif_reglist,
1358 &a6xx_state->registers[index++]);
1359 else
1360 a6xx_get_ahb_gpu_registers(gpu,
1361 a6xx_state, &a6xx_vbif_reglist,
1362 &a6xx_state->registers[index++]);
1363 if (!dumper) {
1364 /*
1365 * We can't use the crashdumper when the SMMU is stalled,
1366 * because the GPU has no memory access until we resume
1367 * translation (but we don't want to do that until after
1368 * we have captured as much useful GPU state as possible).
1369 * So instead collect registers via the CPU:
1370 */
1371 for (i = 0; i < ARRAY_SIZE(a6xx_reglist); i++)
1372 a6xx_get_ahb_gpu_registers(gpu,
1373 a6xx_state, &a6xx_reglist[i],
1374 &a6xx_state->registers[index++]);
1375 return;
1376 }
1377
1378 for (i = 0; i < ARRAY_SIZE(a6xx_reglist); i++)
1379 a6xx_get_crashdumper_registers(gpu,
1380 a6xx_state, &a6xx_reglist[i],
1381 &a6xx_state->registers[index++],
1382 dumper);
1383
1384 for (i = 0; i < ARRAY_SIZE(a6xx_hlsq_reglist); i++)
1385 a6xx_get_crashdumper_hlsq_registers(gpu,
1386 a6xx_state, &a6xx_hlsq_reglist[i],
1387 &a6xx_state->registers[index++],
1388 dumper);
1389 }
1390
1391 #define A7XX_PRE_CRASHDUMPER_SIZE 1
1392 #define A7XX_POST_CRASHDUMPER_SIZE 1
a7xx_get_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,struct a6xx_crashdumper * dumper)1393 static void a7xx_get_registers(struct msm_gpu *gpu,
1394 struct a6xx_gpu_state *a6xx_state,
1395 struct a6xx_crashdumper *dumper)
1396 {
1397 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1398 int i, count;
1399 int index = 0;
1400 const u32 *pre_crashdumper_regs;
1401 const struct gen7_reg_list *reglist;
1402
1403 if (adreno_gpu->info->family == ADRENO_7XX_GEN1) {
1404 if (adreno_is_a722(adreno_gpu)) {
1405 reglist = gen7_17_0_reg_list;
1406 pre_crashdumper_regs = gen7_9_0_pre_crashdumper_gpu_registers;
1407 } else {
1408 reglist = gen7_0_0_reg_list;
1409 pre_crashdumper_regs = gen7_0_0_pre_crashdumper_gpu_registers;
1410 }
1411 } else if (adreno_gpu->info->family == ADRENO_7XX_GEN2) {
1412 reglist = gen7_2_0_reg_list;
1413 pre_crashdumper_regs = gen7_0_0_pre_crashdumper_gpu_registers;
1414 } else {
1415 BUG_ON(adreno_gpu->info->family != ADRENO_7XX_GEN3);
1416 reglist = gen7_9_0_reg_list;
1417 pre_crashdumper_regs = gen7_9_0_pre_crashdumper_gpu_registers;
1418 }
1419
1420 count = A7XX_PRE_CRASHDUMPER_SIZE + A7XX_POST_CRASHDUMPER_SIZE;
1421
1422 /* The downstream reglist contains registers in other memory regions
1423 * (cx_misc/cx_mem and cx_dbgc) and we need to plumb through their
1424 * offsets and map them to read them on the CPU. For now only read the
1425 * first region which is the main one.
1426 */
1427 if (dumper) {
1428 for (i = 0; reglist[i].regs; i++)
1429 count++;
1430 } else {
1431 count++;
1432 }
1433
1434 a6xx_state->registers = state_kcalloc(a6xx_state,
1435 count, sizeof(*a6xx_state->registers));
1436
1437 if (!a6xx_state->registers)
1438 return;
1439
1440 a6xx_state->nr_registers = count;
1441
1442 a7xx_get_ahb_gpu_registers(gpu, a6xx_state, pre_crashdumper_regs,
1443 &a6xx_state->registers[index++]);
1444
1445 if (!dumper) {
1446 a7xx_get_ahb_gpu_reglist(gpu,
1447 a6xx_state, ®list[0],
1448 &a6xx_state->registers[index++]);
1449 return;
1450 }
1451
1452 for (i = 0; reglist[i].regs; i++)
1453 a7xx_get_crashdumper_registers(gpu,
1454 a6xx_state, ®list[i],
1455 &a6xx_state->registers[index++],
1456 dumper);
1457 }
1458
a7xx_get_post_crashdumper_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)1459 static void a7xx_get_post_crashdumper_registers(struct msm_gpu *gpu,
1460 struct a6xx_gpu_state *a6xx_state)
1461 {
1462 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1463 const u32 *regs;
1464
1465 BUG_ON(adreno_gpu->info->family > ADRENO_7XX_GEN3);
1466 regs = adreno_is_a722(adreno_gpu) ?
1467 gen7_17_0_post_crashdumper_registers :
1468 gen7_0_0_post_crashdumper_registers;
1469
1470 a7xx_get_ahb_gpu_registers(gpu,
1471 a6xx_state, regs,
1472 &a6xx_state->registers[a6xx_state->nr_registers - 1]);
1473 }
1474
a6xx_get_cp_roq_size(struct msm_gpu * gpu)1475 static u32 a6xx_get_cp_roq_size(struct msm_gpu *gpu)
1476 {
1477 /* The value at [16:31] is in 4dword units. Convert it to dwords */
1478 return gpu_read(gpu, REG_A6XX_CP_ROQ_THRESHOLDS_2) >> 14;
1479 }
1480
a7xx_get_cp_roq_size(struct msm_gpu * gpu)1481 static u32 a7xx_get_cp_roq_size(struct msm_gpu *gpu)
1482 {
1483 /*
1484 * The value at CP_ROQ_THRESHOLDS_2[20:31] is in 4dword units.
1485 * That register however is not directly accessible from APSS on A7xx.
1486 * Program the SQE_UCODE_DBG_ADDR with offset=0x70d3 and read the value.
1487 */
1488 gpu_write(gpu, REG_A6XX_CP_SQE_UCODE_DBG_ADDR, 0x70d3);
1489
1490 return 4 * (gpu_read(gpu, REG_A6XX_CP_SQE_UCODE_DBG_DATA) >> 20);
1491 }
1492
1493 /* Read a block of data from an indexed register pair */
a6xx_get_indexed_regs(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state,const struct a6xx_indexed_registers * indexed,struct a6xx_gpu_state_obj * obj)1494 static void a6xx_get_indexed_regs(struct msm_gpu *gpu,
1495 struct a6xx_gpu_state *a6xx_state,
1496 const struct a6xx_indexed_registers *indexed,
1497 struct a6xx_gpu_state_obj *obj)
1498 {
1499 u32 count = indexed->count;
1500 int i;
1501
1502 obj->handle = (const void *) indexed;
1503 if (indexed->count_fn)
1504 count = indexed->count_fn(gpu);
1505
1506 obj->data = state_kcalloc(a6xx_state, count, sizeof(u32));
1507 obj->count = count;
1508 if (!obj->data)
1509 return;
1510
1511 /* All the indexed banks start at address 0 */
1512 gpu_write(gpu, indexed->addr, 0);
1513
1514 /* Read the data - each read increments the internal address by 1 */
1515 for (i = 0; i < count; i++)
1516 obj->data[i] = gpu_read(gpu, indexed->data);
1517 }
1518
a6xx_get_indexed_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)1519 static void a6xx_get_indexed_registers(struct msm_gpu *gpu,
1520 struct a6xx_gpu_state *a6xx_state)
1521 {
1522 u32 mempool_size;
1523 int count = ARRAY_SIZE(a6xx_indexed_reglist) + 1;
1524 int i;
1525
1526 a6xx_state->indexed_regs = state_kcalloc(a6xx_state, count,
1527 sizeof(*a6xx_state->indexed_regs));
1528 if (!a6xx_state->indexed_regs)
1529 return;
1530
1531 for (i = 0; i < ARRAY_SIZE(a6xx_indexed_reglist); i++)
1532 a6xx_get_indexed_regs(gpu, a6xx_state, &a6xx_indexed_reglist[i],
1533 &a6xx_state->indexed_regs[i]);
1534
1535 if (adreno_is_a650_family(to_adreno_gpu(gpu))) {
1536 u32 val;
1537
1538 val = gpu_read(gpu, REG_A6XX_CP_CHICKEN_DBG);
1539 gpu_write(gpu, REG_A6XX_CP_CHICKEN_DBG, val | 4);
1540
1541 /* Get the contents of the CP mempool */
1542 a6xx_get_indexed_regs(gpu, a6xx_state, &a6xx_cp_mempool_indexed,
1543 &a6xx_state->indexed_regs[i]);
1544
1545 gpu_write(gpu, REG_A6XX_CP_CHICKEN_DBG, val);
1546 a6xx_state->nr_indexed_regs = count;
1547 return;
1548 }
1549
1550 /* Set the CP mempool size to 0 to stabilize it while dumping */
1551 mempool_size = gpu_read(gpu, REG_A6XX_CP_MEM_POOL_SIZE);
1552 gpu_write(gpu, REG_A6XX_CP_MEM_POOL_SIZE, 0);
1553
1554 /* Get the contents of the CP mempool */
1555 a6xx_get_indexed_regs(gpu, a6xx_state, &a6xx_cp_mempool_indexed,
1556 &a6xx_state->indexed_regs[i]);
1557
1558 /*
1559 * Offset 0x2000 in the mempool is the size - copy the saved size over
1560 * so the data is consistent
1561 */
1562 a6xx_state->indexed_regs[i].data[0x2000] = mempool_size;
1563
1564 /* Restore the size in the hardware */
1565 gpu_write(gpu, REG_A6XX_CP_MEM_POOL_SIZE, mempool_size);
1566
1567 a6xx_state->nr_indexed_regs = count;
1568 }
1569
a7xx_get_indexed_registers(struct msm_gpu * gpu,struct a6xx_gpu_state * a6xx_state)1570 static void a7xx_get_indexed_registers(struct msm_gpu *gpu,
1571 struct a6xx_gpu_state *a6xx_state)
1572 {
1573 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1574 const struct a6xx_indexed_registers *indexed_regs;
1575 const struct a6xx_indexed_registers *mempool_regs;
1576 int i, indexed_count, mempool_count;
1577 bool concurrent_binning;
1578
1579 if (adreno_is_a722(adreno_gpu)) {
1580 /*
1581 * Eliza has no BV or LPAC SQE — skip the BV/LPAC indexed
1582 * registers and the BV mempool
1583 */
1584 indexed_regs = gen7_17_0_cp_indexed_reglist;
1585 indexed_count = ARRAY_SIZE(gen7_17_0_cp_indexed_reglist);
1586 mempool_regs = a7xx_cp_mempool_indexed;
1587 mempool_count = ARRAY_SIZE(a7xx_cp_mempool_indexed);
1588 concurrent_binning = false;
1589 } else if (adreno_gpu->info->family <= ADRENO_7XX_GEN2) {
1590 indexed_regs = a7xx_indexed_reglist;
1591 indexed_count = ARRAY_SIZE(a7xx_indexed_reglist);
1592 mempool_regs = a7xx_cp_bv_mempool_indexed;
1593 mempool_count = ARRAY_SIZE(a7xx_cp_bv_mempool_indexed);
1594 concurrent_binning = true;
1595 } else {
1596 BUG_ON(adreno_gpu->info->family != ADRENO_7XX_GEN3);
1597 indexed_regs = gen7_9_0_cp_indexed_reg_list;
1598 indexed_count = ARRAY_SIZE(gen7_9_0_cp_indexed_reg_list);
1599 mempool_regs = a7xx_cp_bv_mempool_indexed;
1600 mempool_count = ARRAY_SIZE(a7xx_cp_bv_mempool_indexed);
1601 concurrent_binning = true;
1602 }
1603
1604 a6xx_state->indexed_regs = state_kcalloc(a6xx_state,
1605 indexed_count + mempool_count,
1606 sizeof(*a6xx_state->indexed_regs));
1607 if (!a6xx_state->indexed_regs)
1608 return;
1609
1610 a6xx_state->nr_indexed_regs = indexed_count + mempool_count;
1611
1612 /* First read the common regs */
1613 for (i = 0; i < indexed_count; i++)
1614 a6xx_get_indexed_regs(gpu, a6xx_state, &indexed_regs[i],
1615 &a6xx_state->indexed_regs[i]);
1616
1617 gpu_rmw(gpu, REG_A6XX_CP_CHICKEN_DBG, 0, BIT(2));
1618 if (concurrent_binning)
1619 gpu_rmw(gpu, REG_A7XX_CP_BV_CHICKEN_DBG, 0, BIT(2));
1620
1621 /* Get the contents of the CP_BV mempool */
1622 for (i = 0; i < mempool_count; i++)
1623 a6xx_get_indexed_regs(gpu, a6xx_state, &mempool_regs[i],
1624 &a6xx_state->indexed_regs[indexed_count + i]);
1625
1626 gpu_rmw(gpu, REG_A6XX_CP_CHICKEN_DBG, BIT(2), 0);
1627 if (concurrent_binning)
1628 gpu_rmw(gpu, REG_A7XX_CP_BV_CHICKEN_DBG, BIT(2), 0);
1629 return;
1630 }
1631
a6xx_gpu_state_get(struct msm_gpu * gpu)1632 struct msm_gpu_state *a6xx_gpu_state_get(struct msm_gpu *gpu)
1633 {
1634 struct a6xx_crashdumper _dumper = { 0 }, *dumper = NULL;
1635 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
1636 struct a6xx_gpu *a6xx_gpu = to_a6xx_gpu(adreno_gpu);
1637 struct a6xx_gpu_state *a6xx_state = kzalloc_obj(*a6xx_state);
1638 bool stalled;
1639
1640 if (!a6xx_state)
1641 return ERR_PTR(-ENOMEM);
1642
1643 INIT_LIST_HEAD(&a6xx_state->objs);
1644
1645 /* Get the generic state from the adreno core */
1646 adreno_gpu_state_get(gpu, &a6xx_state->base);
1647
1648 if (!adreno_has_gmu_wrapper(adreno_gpu) &&
1649 !adreno_has_rgmu(adreno_gpu)) {
1650 a6xx_get_gmu_registers(gpu, a6xx_state);
1651
1652 a6xx_state->gmu_log = a6xx_snapshot_gmu_bo(a6xx_state, &a6xx_gpu->gmu.log);
1653 a6xx_state->gmu_hfi = a6xx_snapshot_gmu_bo(a6xx_state, &a6xx_gpu->gmu.hfi);
1654 a6xx_state->gmu_debug = a6xx_snapshot_gmu_bo(a6xx_state, &a6xx_gpu->gmu.debug);
1655
1656 a6xx_snapshot_gmu_hfi_history(gpu, a6xx_state);
1657 }
1658
1659 /* If GX isn't on the rest of the data isn't going to be accessible */
1660 if (!adreno_gpu->funcs->gx_is_on(adreno_gpu))
1661 return &a6xx_state->base;
1662
1663 /* Halt SQE first */
1664 gpu_write(gpu, REG_A6XX_CP_SQE_CNTL, 3);
1665
1666 /* Get the banks of indexed registers */
1667 if (adreno_is_a7xx(adreno_gpu))
1668 a7xx_get_indexed_registers(gpu, a6xx_state);
1669 else
1670 a6xx_get_indexed_registers(gpu, a6xx_state);
1671
1672 stalled = !!(gpu_read(gpu, REG_A6XX_RBBM_STATUS3) &
1673 A6XX_RBBM_STATUS3_SMMU_STALLED_ON_FAULT);
1674 /*
1675 * Try to initialize the crashdumper, if we are not dumping state
1676 * with the SMMU stalled. The crashdumper needs memory access to
1677 * write out GPU state, so we need to skip this when the SMMU is
1678 * stalled in response to an iova fault
1679 */
1680 if (!stalled && !gpu->needs_hw_init &&
1681 !a6xx_crashdumper_init(gpu, &_dumper)) {
1682 dumper = &_dumper;
1683 }
1684
1685 if (adreno_is_a7xx(adreno_gpu)) {
1686 a7xx_get_registers(gpu, a6xx_state, dumper);
1687
1688 if (dumper) {
1689 a7xx_get_shaders(gpu, a6xx_state, dumper);
1690 a7xx_get_clusters(gpu, a6xx_state, dumper);
1691 a7xx_get_dbgahb_clusters(gpu, a6xx_state, dumper);
1692
1693 msm_gem_kernel_put(dumper->bo, gpu->vm);
1694 }
1695
1696 a7xx_get_post_crashdumper_registers(gpu, a6xx_state);
1697 } else {
1698 a6xx_get_registers(gpu, a6xx_state, dumper);
1699
1700 if (dumper) {
1701 a6xx_get_shaders(gpu, a6xx_state, dumper);
1702 a6xx_get_clusters(gpu, a6xx_state, dumper);
1703 a6xx_get_dbgahb_clusters(gpu, a6xx_state, dumper);
1704
1705 msm_gem_kernel_put(dumper->bo, gpu->vm);
1706 }
1707 }
1708
1709 if (snapshot_debugbus)
1710 a6xx_get_debugbus(gpu, a6xx_state);
1711
1712 a6xx_state->gpu_initialized = !gpu->needs_hw_init;
1713
1714 return &a6xx_state->base;
1715 }
1716
a6xx_gpu_state_destroy(struct kref * kref)1717 static void a6xx_gpu_state_destroy(struct kref *kref)
1718 {
1719 struct a6xx_state_memobj *obj, *tmp;
1720 struct msm_gpu_state *state = container_of(kref,
1721 struct msm_gpu_state, ref);
1722 struct a6xx_gpu_state *a6xx_state = container_of(state,
1723 struct a6xx_gpu_state, base);
1724
1725 if (a6xx_state->gmu_log)
1726 kvfree(a6xx_state->gmu_log->data);
1727
1728 if (a6xx_state->gmu_hfi)
1729 kvfree(a6xx_state->gmu_hfi->data);
1730
1731 if (a6xx_state->gmu_debug)
1732 kvfree(a6xx_state->gmu_debug->data);
1733
1734 list_for_each_entry_safe(obj, tmp, &a6xx_state->objs, node) {
1735 list_del(&obj->node);
1736 kvfree(obj);
1737 }
1738
1739 adreno_gpu_state_destroy(state);
1740 kfree(a6xx_state);
1741 }
1742
a6xx_gpu_state_put(struct msm_gpu_state * state)1743 int a6xx_gpu_state_put(struct msm_gpu_state *state)
1744 {
1745 if (IS_ERR_OR_NULL(state))
1746 return 1;
1747
1748 return kref_put(&state->ref, a6xx_gpu_state_destroy);
1749 }
1750
a6xx_show_registers(const u32 * registers,u32 * data,size_t count,struct drm_printer * p)1751 static void a6xx_show_registers(const u32 *registers, u32 *data, size_t count,
1752 struct drm_printer *p)
1753 {
1754 int i, index = 0;
1755
1756 if (!data)
1757 return;
1758
1759 for (i = 0; i < count; i += 2) {
1760 u32 count = RANGE(registers, i);
1761 u32 offset = registers[i];
1762 int j;
1763
1764 for (j = 0; j < count; index++, offset++, j++) {
1765 if (data[index] == 0xdeafbead)
1766 continue;
1767
1768 drm_printf(p, " - { offset: 0x%06x, value: 0x%08x }\n",
1769 offset << 2, data[index]);
1770 }
1771 }
1772 }
1773
a7xx_show_registers_indented(const u32 * registers,u32 * data,struct drm_printer * p,unsigned indent)1774 static void a7xx_show_registers_indented(const u32 *registers, u32 *data,
1775 struct drm_printer *p, unsigned indent)
1776 {
1777 int i, index = 0;
1778
1779 for (i = 0; registers[i] != UINT_MAX; i += 2) {
1780 u32 count = RANGE(registers, i);
1781 u32 offset = registers[i];
1782 int j;
1783
1784 for (j = 0; j < count; index++, offset++, j++) {
1785 int k;
1786
1787 if (data[index] == 0xdeafbead)
1788 continue;
1789
1790 for (k = 0; k < indent; k++)
1791 drm_printf(p, " ");
1792 drm_printf(p, "- { offset: 0x%06x, value: 0x%08x }\n",
1793 offset << 2, data[index]);
1794 }
1795 }
1796 }
1797
a7xx_show_registers(const u32 * registers,u32 * data,struct drm_printer * p)1798 static void a7xx_show_registers(const u32 *registers, u32 *data, struct drm_printer *p)
1799 {
1800 a7xx_show_registers_indented(registers, data, p, 1);
1801 }
1802
print_ascii85(struct drm_printer * p,size_t len,u32 * data)1803 static void print_ascii85(struct drm_printer *p, size_t len, u32 *data)
1804 {
1805 char out[ASCII85_BUFSZ];
1806 long i, l, datalen = 0;
1807
1808 for (i = 0; i < len >> 2; i++) {
1809 if (data[i])
1810 datalen = (i + 1) << 2;
1811 }
1812
1813 if (datalen == 0)
1814 return;
1815
1816 drm_puts(p, " data: !!ascii85 |\n");
1817 drm_puts(p, " ");
1818
1819
1820 l = ascii85_encode_len(datalen);
1821
1822 for (i = 0; i < l; i++)
1823 drm_puts(p, ascii85_encode(data[i], out));
1824
1825 drm_puts(p, "\n");
1826 }
1827
print_name(struct drm_printer * p,const char * fmt,const char * name)1828 static void print_name(struct drm_printer *p, const char *fmt, const char *name)
1829 {
1830 drm_puts(p, fmt);
1831 drm_puts(p, name);
1832 drm_puts(p, "\n");
1833 }
1834
a6xx_show_shader(struct a6xx_gpu_state_obj * obj,struct drm_printer * p)1835 static void a6xx_show_shader(struct a6xx_gpu_state_obj *obj,
1836 struct drm_printer *p)
1837 {
1838 const struct a6xx_shader_block *block = obj->handle;
1839 int i;
1840
1841 if (!obj->handle)
1842 return;
1843
1844 print_name(p, " - type: ", block->name);
1845
1846 for (i = 0; i < A6XX_NUM_SHADER_BANKS; i++) {
1847 drm_printf(p, " - bank: %d\n", i);
1848 drm_printf(p, " size: %d\n", block->size);
1849
1850 if (!obj->data)
1851 continue;
1852
1853 print_ascii85(p, block->size << 2,
1854 obj->data + (block->size * i));
1855 }
1856 }
1857
a7xx_show_shader(struct a6xx_gpu_state_obj * obj,struct drm_printer * p)1858 static void a7xx_show_shader(struct a6xx_gpu_state_obj *obj,
1859 struct drm_printer *p)
1860 {
1861 const struct gen7_shader_block *block = obj->handle;
1862 int i, j;
1863 u32 *data = obj->data;
1864
1865 if (!obj->handle)
1866 return;
1867
1868 print_name(p, " - type: ", a7xx_statetype_names[block->statetype]);
1869 print_name(p, " - pipe: ", a7xx_pipe_names[block->pipeid]);
1870 drm_printf(p, " - location: %d\n", block->location);
1871
1872 for (i = 0; i < block->num_sps; i++) {
1873 drm_printf(p, " - sp: %d\n", i);
1874
1875 for (j = 0; j < block->num_usptps; j++) {
1876 drm_printf(p, " - usptp: %d\n", j);
1877 drm_printf(p, " size: %d\n", block->size);
1878
1879 if (!obj->data)
1880 continue;
1881
1882 print_ascii85(p, block->size << 2, data);
1883
1884 data += block->size;
1885 }
1886 }
1887 }
1888
a6xx_show_cluster_data(const u32 * registers,int size,u32 * data,struct drm_printer * p)1889 static void a6xx_show_cluster_data(const u32 *registers, int size, u32 *data,
1890 struct drm_printer *p)
1891 {
1892 int ctx, index = 0;
1893
1894 for (ctx = 0; ctx < A6XX_NUM_CONTEXTS; ctx++) {
1895 int j;
1896
1897 drm_printf(p, " - context: %d\n", ctx);
1898
1899 for (j = 0; j < size; j += 2) {
1900 u32 count = RANGE(registers, j);
1901 u32 offset = registers[j];
1902 int k;
1903
1904 for (k = 0; k < count; index++, offset++, k++) {
1905 if (data[index] == 0xdeafbead)
1906 continue;
1907
1908 drm_printf(p, " - { offset: 0x%06x, value: 0x%08x }\n",
1909 offset << 2, data[index]);
1910 }
1911 }
1912 }
1913 }
1914
a6xx_show_dbgahb_cluster(struct a6xx_gpu_state_obj * obj,struct drm_printer * p)1915 static void a6xx_show_dbgahb_cluster(struct a6xx_gpu_state_obj *obj,
1916 struct drm_printer *p)
1917 {
1918 const struct a6xx_dbgahb_cluster *dbgahb = obj->handle;
1919
1920 if (dbgahb) {
1921 print_name(p, " - cluster-name: ", dbgahb->name);
1922 a6xx_show_cluster_data(dbgahb->registers, dbgahb->count,
1923 obj->data, p);
1924 }
1925 }
1926
a6xx_show_cluster(struct a6xx_gpu_state_obj * obj,struct drm_printer * p)1927 static void a6xx_show_cluster(struct a6xx_gpu_state_obj *obj,
1928 struct drm_printer *p)
1929 {
1930 const struct a6xx_cluster *cluster = obj->handle;
1931
1932 if (cluster) {
1933 print_name(p, " - cluster-name: ", cluster->name);
1934 a6xx_show_cluster_data(cluster->registers, cluster->count,
1935 obj->data, p);
1936 }
1937 }
1938
a7xx_show_dbgahb_cluster(struct a6xx_gpu_state_obj * obj,struct drm_printer * p)1939 static void a7xx_show_dbgahb_cluster(struct a6xx_gpu_state_obj *obj,
1940 struct drm_printer *p)
1941 {
1942 const struct gen7_sptp_cluster_registers *dbgahb = obj->handle;
1943
1944 if (dbgahb) {
1945 print_name(p, " - pipe: ", a7xx_pipe_names[dbgahb->pipe_id]);
1946 print_name(p, " - cluster-name: ", a7xx_cluster_names[dbgahb->cluster_id]);
1947 drm_printf(p, " - context: %d\n", dbgahb->context_id);
1948 drm_printf(p, " - location: %d\n", dbgahb->location_id);
1949 a7xx_show_registers_indented(dbgahb->regs, obj->data, p, 4);
1950 }
1951 }
1952
a7xx_show_cluster(struct a6xx_gpu_state_obj * obj,struct drm_printer * p)1953 static void a7xx_show_cluster(struct a6xx_gpu_state_obj *obj,
1954 struct drm_printer *p)
1955 {
1956 const struct gen7_cluster_registers *cluster = obj->handle;
1957
1958 if (cluster) {
1959 int context = (cluster->context_id == STATE_FORCE_CTXT_1) ? 1 : 0;
1960
1961 print_name(p, " - pipe: ", a7xx_pipe_names[cluster->pipe_id]);
1962 print_name(p, " - cluster-name: ", a7xx_cluster_names[cluster->cluster_id]);
1963 drm_printf(p, " - context: %d\n", context);
1964 a7xx_show_registers_indented(cluster->regs, obj->data, p, 4);
1965 }
1966 }
1967
a6xx_show_indexed_regs(struct a6xx_gpu_state_obj * obj,struct drm_printer * p)1968 static void a6xx_show_indexed_regs(struct a6xx_gpu_state_obj *obj,
1969 struct drm_printer *p)
1970 {
1971 const struct a6xx_indexed_registers *indexed = obj->handle;
1972
1973 if (!indexed)
1974 return;
1975
1976 print_name(p, " - regs-name: ", indexed->name);
1977 drm_printf(p, " dwords: %d\n", obj->count);
1978
1979 print_ascii85(p, obj->count << 2, obj->data);
1980 }
1981
a6xx_show_debugbus_block(const struct a6xx_debugbus_block * block,u32 * data,struct drm_printer * p)1982 static void a6xx_show_debugbus_block(const struct a6xx_debugbus_block *block,
1983 u32 *data, struct drm_printer *p)
1984 {
1985 if (block) {
1986 print_name(p, " - debugbus-block: ", block->name);
1987
1988 /*
1989 * count for regular debugbus data is in quadwords,
1990 * but print the size in dwords for consistency
1991 */
1992 drm_printf(p, " count: %d\n", block->count << 1);
1993
1994 print_ascii85(p, block->count << 3, data);
1995 }
1996 }
1997
a6xx_show_debugbus(struct a6xx_gpu_state * a6xx_state,struct drm_printer * p)1998 static void a6xx_show_debugbus(struct a6xx_gpu_state *a6xx_state,
1999 struct drm_printer *p)
2000 {
2001 int i;
2002
2003 for (i = 0; i < a6xx_state->nr_debugbus; i++) {
2004 struct a6xx_gpu_state_obj *obj = &a6xx_state->debugbus[i];
2005
2006 a6xx_show_debugbus_block(obj->handle, obj->data, p);
2007 }
2008
2009 if (a6xx_state->vbif_debugbus) {
2010 struct a6xx_gpu_state_obj *obj = a6xx_state->vbif_debugbus;
2011
2012 drm_puts(p, " - debugbus-block: A6XX_DBGBUS_VBIF\n");
2013 drm_printf(p, " count: %d\n", VBIF_DEBUGBUS_BLOCK_SIZE);
2014
2015 /* vbif debugbus data is in dwords. Confusing, huh? */
2016 print_ascii85(p, VBIF_DEBUGBUS_BLOCK_SIZE << 2, obj->data);
2017 }
2018
2019 for (i = 0; i < a6xx_state->nr_cx_debugbus; i++) {
2020 struct a6xx_gpu_state_obj *obj = &a6xx_state->cx_debugbus[i];
2021
2022 a6xx_show_debugbus_block(obj->handle, obj->data, p);
2023 }
2024 }
2025
a6xx_show(struct msm_gpu * gpu,struct msm_gpu_state * state,struct drm_printer * p)2026 void a6xx_show(struct msm_gpu *gpu, struct msm_gpu_state *state,
2027 struct drm_printer *p)
2028 {
2029 struct adreno_gpu *adreno_gpu = to_adreno_gpu(gpu);
2030 struct a6xx_gpu_state *a6xx_state = container_of(state,
2031 struct a6xx_gpu_state, base);
2032 int i;
2033
2034 if (IS_ERR_OR_NULL(state))
2035 return;
2036
2037 drm_printf(p, "gpu-initialized: %d\n", a6xx_state->gpu_initialized);
2038
2039 adreno_show(gpu, state, p);
2040
2041 drm_puts(p, "gmu-log:\n");
2042 if (a6xx_state->gmu_log) {
2043 struct msm_gpu_state_bo *gmu_log = a6xx_state->gmu_log;
2044
2045 drm_printf(p, " iova: 0x%016llx\n", gmu_log->iova);
2046 drm_printf(p, " size: %zu\n", gmu_log->size);
2047 adreno_show_object(p, &gmu_log->data, gmu_log->size,
2048 &gmu_log->encoded);
2049 }
2050
2051 drm_puts(p, "gmu-hfi:\n");
2052 if (a6xx_state->gmu_hfi) {
2053 struct msm_gpu_state_bo *gmu_hfi = a6xx_state->gmu_hfi;
2054 unsigned i, j;
2055
2056 drm_printf(p, " iova: 0x%016llx\n", gmu_hfi->iova);
2057 drm_printf(p, " size: %zu\n", gmu_hfi->size);
2058 for (i = 0; i < ARRAY_SIZE(a6xx_state->hfi_queue_history); i++) {
2059 drm_printf(p, " queue-history[%u]:", i);
2060 for (j = 0; j < HFI_HISTORY_SZ; j++) {
2061 drm_printf(p, " %d", a6xx_state->hfi_queue_history[i][j]);
2062 }
2063 drm_printf(p, "\n");
2064 }
2065 adreno_show_object(p, &gmu_hfi->data, gmu_hfi->size,
2066 &gmu_hfi->encoded);
2067 }
2068
2069 drm_puts(p, "gmu-debug:\n");
2070 if (a6xx_state->gmu_debug) {
2071 struct msm_gpu_state_bo *gmu_debug = a6xx_state->gmu_debug;
2072
2073 drm_printf(p, " iova: 0x%016llx\n", gmu_debug->iova);
2074 drm_printf(p, " size: %zu\n", gmu_debug->size);
2075 adreno_show_object(p, &gmu_debug->data, gmu_debug->size,
2076 &gmu_debug->encoded);
2077 }
2078
2079 drm_puts(p, "registers:\n");
2080 for (i = 0; i < a6xx_state->nr_registers; i++) {
2081 struct a6xx_gpu_state_obj *obj = &a6xx_state->registers[i];
2082
2083 if (!obj->handle)
2084 continue;
2085
2086 if (adreno_is_a7xx(adreno_gpu)) {
2087 a7xx_show_registers(obj->handle, obj->data, p);
2088 } else {
2089 const struct a6xx_registers *regs = obj->handle;
2090
2091 a6xx_show_registers(regs->registers, obj->data, regs->count, p);
2092 }
2093 }
2094
2095 drm_puts(p, "registers-gmu:\n");
2096 for (i = 0; i < a6xx_state->nr_gmu_registers; i++) {
2097 struct a6xx_gpu_state_obj *obj = &a6xx_state->gmu_registers[i];
2098 const struct a6xx_registers *regs = obj->handle;
2099
2100 if (!obj->handle)
2101 continue;
2102
2103 a6xx_show_registers(regs->registers, obj->data, regs->count, p);
2104 }
2105
2106 drm_puts(p, "indexed-registers:\n");
2107 for (i = 0; i < a6xx_state->nr_indexed_regs; i++)
2108 a6xx_show_indexed_regs(&a6xx_state->indexed_regs[i], p);
2109
2110 drm_puts(p, "shader-blocks:\n");
2111 for (i = 0; i < a6xx_state->nr_shaders; i++) {
2112 if (adreno_is_a7xx(adreno_gpu))
2113 a7xx_show_shader(&a6xx_state->shaders[i], p);
2114 else
2115 a6xx_show_shader(&a6xx_state->shaders[i], p);
2116 }
2117
2118 drm_puts(p, "clusters:\n");
2119 for (i = 0; i < a6xx_state->nr_clusters; i++) {
2120 if (adreno_is_a7xx(adreno_gpu))
2121 a7xx_show_cluster(&a6xx_state->clusters[i], p);
2122 else
2123 a6xx_show_cluster(&a6xx_state->clusters[i], p);
2124 }
2125
2126 for (i = 0; i < a6xx_state->nr_dbgahb_clusters; i++) {
2127 if (adreno_is_a7xx(adreno_gpu))
2128 a7xx_show_dbgahb_cluster(&a6xx_state->dbgahb_clusters[i], p);
2129 else
2130 a6xx_show_dbgahb_cluster(&a6xx_state->dbgahb_clusters[i], p);
2131 }
2132
2133 drm_puts(p, "debugbus:\n");
2134 a6xx_show_debugbus(a6xx_state, p);
2135 }
2136