File: local-stack-frame.ll

package info (click to toggle)
llvm-toolchain-17 1%3A17.0.6-22
  • links: PTS, VCS
  • area: main
  • in suites: forky, sid, trixie
  • size: 1,799,624 kB
  • sloc: cpp: 6,428,607; ansic: 1,383,196; asm: 793,408; python: 223,504; objc: 75,364; f90: 60,502; lisp: 33,869; pascal: 15,282; sh: 9,684; perl: 7,453; ml: 4,937; awk: 3,523; makefile: 2,889; javascript: 2,149; xml: 888; fortran: 619; cs: 573
file content (83 lines) | stat: -rw-r--r-- 3,505 bytes parent folder | download | duplicates (6)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
; RUN: llc < %s -march=nvptx -mcpu=sm_20 -verify-machineinstrs | FileCheck %s --check-prefix=PTX32
; RUN: llc < %s -march=nvptx64 -mcpu=sm_20 -verify-machineinstrs | FileCheck %s --check-prefix=PTX64
; RUN: %if ptxas && !ptxas-12.0 %{ llc < %s -march=nvptx -mcpu=sm_20 -verify-machineinstrs | %ptxas-verify %}
; RUN: %if ptxas %{ llc < %s -march=nvptx64 -mcpu=sm_20 -verify-machineinstrs | %ptxas-verify %}

; Ensure we access the local stack properly

; PTX32:        mov.u32          %SPL, __local_depot{{[0-9]+}};
; PTX32:        cvta.local.u32   %SP, %SPL;
; PTX32:        ld.param.u32     %r{{[0-9]+}}, [foo_param_0];
; PTX32:        st.volatile.u32  [%SP+0], %r{{[0-9]+}};
; PTX64:        mov.u64          %SPL, __local_depot{{[0-9]+}};
; PTX64:        cvta.local.u64   %SP, %SPL;
; PTX64:        ld.param.u32     %r{{[0-9]+}}, [foo_param_0];
; PTX64:        st.volatile.u32  [%SP+0], %r{{[0-9]+}};
define void @foo(i32 %a) {
  %local = alloca i32, align 4
  store volatile i32 %a, ptr %local
  ret void
}

; PTX32:        mov.u32          %SPL, __local_depot{{[0-9]+}};
; PTX32:        cvta.local.u32   %SP, %SPL;
; PTX32:        ld.param.u32     %r{{[0-9]+}}, [foo2_param_0];
; PTX32:        add.u32          %r[[SP_REG:[0-9]+]], %SPL, 0;
; PTX32:        st.local.u32  [%r[[SP_REG]]], %r{{[0-9]+}};
; PTX64:        mov.u64          %SPL, __local_depot{{[0-9]+}};
; PTX64:        cvta.local.u64   %SP, %SPL;
; PTX64:        ld.param.u32     %r{{[0-9]+}}, [foo2_param_0];
; PTX64:        add.u64          %rd[[SP_REG:[0-9]+]], %SPL, 0;
; PTX64:        st.local.u32  [%rd[[SP_REG]]], %r{{[0-9]+}};
define void @foo2(i32 %a) {
  %local = alloca i32, align 4
  store i32 %a, ptr %local
  call void @bar(ptr %local)
  ret void
}

declare void @bar(ptr %a)

!nvvm.annotations = !{!0}
!0 = !{ptr @foo2, !"kernel", i32 1}

; PTX32:        mov.u32          %SPL, __local_depot{{[0-9]+}};
; PTX32-NOT:    cvta.local.u32   %SP, %SPL;
; PTX32:        ld.param.u32     %r{{[0-9]+}}, [foo3_param_0];
; PTX32:        add.u32          %r{{[0-9]+}}, %SPL, 0;
; PTX32:        st.local.u32  [%r{{[0-9]+}}], %r{{[0-9]+}};
; PTX64:        mov.u64          %SPL, __local_depot{{[0-9]+}};
; PTX64-NOT:    cvta.local.u64   %SP, %SPL;
; PTX64:        ld.param.u32     %r{{[0-9]+}}, [foo3_param_0];
; PTX64:        add.u64          %rd{{[0-9]+}}, %SPL, 0;
; PTX64:        st.local.u32  [%rd{{[0-9]+}}], %r{{[0-9]+}};
define void @foo3(i32 %a) {
  %local = alloca [3 x i32], align 4
  %1 = getelementptr inbounds i32, ptr %local, i32 %a
  store i32 %a, ptr %1
  ret void
}

; PTX32:        cvta.local.u32   %SP, %SPL;
; PTX32:        add.u32          {{%r[0-9]+}}, %SP, 0;
; PTX32:        add.u32          {{%r[0-9]+}}, %SPL, 0;
; PTX32:        add.u32          {{%r[0-9]+}}, %SP, 4;
; PTX32:        add.u32          {{%r[0-9]+}}, %SPL, 4;
; PTX32:        st.local.u32     [{{%r[0-9]+}}], {{%r[0-9]+}}
; PTX32:        st.local.u32     [{{%r[0-9]+}}], {{%r[0-9]+}}
; PTX64:        cvta.local.u64   %SP, %SPL;
; PTX64:        add.u64          {{%rd[0-9]+}}, %SP, 0;
; PTX64:        add.u64          {{%rd[0-9]+}}, %SPL, 0;
; PTX64:        add.u64          {{%rd[0-9]+}}, %SP, 4;
; PTX64:        add.u64          {{%rd[0-9]+}}, %SPL, 4;
; PTX64:        st.local.u32     [{{%rd[0-9]+}}], {{%r[0-9]+}}
; PTX64:        st.local.u32     [{{%rd[0-9]+}}], {{%r[0-9]+}}
define void @foo4() {
  %A = alloca i32
  %B = alloca i32
  store i32 0, ptr %A
  store i32 0, ptr %B
  call void @bar(ptr %A)
  call void @bar(ptr %B)
  ret void
}