1 ; RUN: llc -march=amdgcn -mcpu=bonaire -verify-machineinstrs -enable-misched -enable-aa-sched-mi < %s | FileCheck -check-prefix=FUNC -check-prefix=CI %s
3 declare void @llvm.SI.tbuffer.store.i32(<16 x i8>, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32)
4 declare void @llvm.SI.tbuffer.store.v4i32(<16 x i8>, <4 x i32>, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32, i32)
5 declare void @llvm.AMDGPU.barrier.local() #2
8 @stored_lds_ptr = addrspace(3) global i32 addrspace(3)* undef, align 4
9 @stored_constant_ptr = addrspace(3) global i32 addrspace(2)* undef, align 8
10 @stored_global_ptr = addrspace(3) global i32 addrspace(1)* undef, align 8
12 ; FUNC-LABEL: @reorder_local_load_global_store_local_load
13 ; CI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:4
14 ; CI-NEXT: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:8
15 ; CI: buffer_store_dword
16 define void @reorder_local_load_global_store_local_load(i32 addrspace(1)* %out, i32 addrspace(1)* %gptr) #0 {
17 %ptr0 = load i32 addrspace(3)* addrspace(3)* @stored_lds_ptr, align 4
19 %ptr1 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 1
20 %ptr2 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 2
22 %tmp1 = load i32 addrspace(3)* %ptr1, align 4
23 store i32 99, i32 addrspace(1)* %gptr, align 4
24 %tmp2 = load i32 addrspace(3)* %ptr2, align 4
26 %add = add nsw i32 %tmp1, %tmp2
28 store i32 %add, i32 addrspace(1)* %out, align 4
32 ; FUNC-LABEL: @no_reorder_local_load_volatile_global_store_local_load
33 ; CI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:4
34 ; CI: buffer_store_dword
35 ; CI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:8
36 define void @no_reorder_local_load_volatile_global_store_local_load(i32 addrspace(1)* %out, i32 addrspace(1)* %gptr) #0 {
37 %ptr0 = load i32 addrspace(3)* addrspace(3)* @stored_lds_ptr, align 4
39 %ptr1 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 1
40 %ptr2 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 2
42 %tmp1 = load i32 addrspace(3)* %ptr1, align 4
43 store volatile i32 99, i32 addrspace(1)* %gptr, align 4
44 %tmp2 = load i32 addrspace(3)* %ptr2, align 4
46 %add = add nsw i32 %tmp1, %tmp2
48 store i32 %add, i32 addrspace(1)* %out, align 4
52 ; FUNC-LABEL: @no_reorder_barrier_local_load_global_store_local_load
53 ; CI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:4
54 ; CI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:8
55 ; CI: buffer_store_dword
56 define void @no_reorder_barrier_local_load_global_store_local_load(i32 addrspace(1)* %out, i32 addrspace(1)* %gptr) #0 {
57 %ptr0 = load i32 addrspace(3)* addrspace(3)* @stored_lds_ptr, align 4
59 %ptr1 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 1
60 %ptr2 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 2
62 %tmp1 = load i32 addrspace(3)* %ptr1, align 4
63 store i32 99, i32 addrspace(1)* %gptr, align 4
64 call void @llvm.AMDGPU.barrier.local() #2
65 %tmp2 = load i32 addrspace(3)* %ptr2, align 4
67 %add = add nsw i32 %tmp1, %tmp2
69 store i32 %add, i32 addrspace(1)* %out, align 4
73 ; Technically we could reorder these, but just comparing the
74 ; instruction type of the load is insufficient.
76 ; FUNC-LABEL: @no_reorder_constant_load_global_store_constant_load
77 ; CI: buffer_load_dword
78 ; CI: buffer_store_dword
79 ; CI: buffer_load_dword
80 ; CI: buffer_store_dword
81 define void @no_reorder_constant_load_global_store_constant_load(i32 addrspace(1)* %out, i32 addrspace(1)* %gptr) #0 {
82 %ptr0 = load i32 addrspace(2)* addrspace(3)* @stored_constant_ptr, align 8
84 %ptr1 = getelementptr inbounds i32 addrspace(2)* %ptr0, i64 1
85 %ptr2 = getelementptr inbounds i32 addrspace(2)* %ptr0, i64 2
87 %tmp1 = load i32 addrspace(2)* %ptr1, align 4
88 store i32 99, i32 addrspace(1)* %gptr, align 4
89 %tmp2 = load i32 addrspace(2)* %ptr2, align 4
91 %add = add nsw i32 %tmp1, %tmp2
93 store i32 %add, i32 addrspace(1)* %out, align 4
97 ; XXX: Should be able to reorder this, but the laods count as ordered
99 ; FUNC-LABEL: @reorder_constant_load_local_store_constant_load
100 ; CI: buffer_load_dword
102 ; CI: buffer_load_dword
103 ; CI: buffer_store_dword
104 define void @reorder_constant_load_local_store_constant_load(i32 addrspace(1)* %out, i32 addrspace(3)* %lptr) #0 {
105 %ptr0 = load i32 addrspace(2)* addrspace(3)* @stored_constant_ptr, align 8
107 %ptr1 = getelementptr inbounds i32 addrspace(2)* %ptr0, i64 1
108 %ptr2 = getelementptr inbounds i32 addrspace(2)* %ptr0, i64 2
110 %tmp1 = load i32 addrspace(2)* %ptr1, align 4
111 store i32 99, i32 addrspace(3)* %lptr, align 4
112 %tmp2 = load i32 addrspace(2)* %ptr2, align 4
114 %add = add nsw i32 %tmp1, %tmp2
116 store i32 %add, i32 addrspace(1)* %out, align 4
120 ; FUNC-LABEL: @reorder_smrd_load_local_store_smrd_load
125 ; CI: buffer_store_dword
126 define void @reorder_smrd_load_local_store_smrd_load(i32 addrspace(1)* %out, i32 addrspace(3)* noalias %lptr, i32 addrspace(2)* %ptr0) #0 {
127 %ptr1 = getelementptr inbounds i32 addrspace(2)* %ptr0, i64 1
128 %ptr2 = getelementptr inbounds i32 addrspace(2)* %ptr0, i64 2
130 %tmp1 = load i32 addrspace(2)* %ptr1, align 4
131 store i32 99, i32 addrspace(3)* %lptr, align 4
132 %tmp2 = load i32 addrspace(2)* %ptr2, align 4
134 %add = add nsw i32 %tmp1, %tmp2
136 store i32 %add, i32 addrspace(1)* %out, align 4
140 ; FUNC-LABEL: @reorder_global_load_local_store_global_load
141 ; CI: buffer_load_dword
142 ; CI: buffer_load_dword
144 ; CI: buffer_store_dword
145 define void @reorder_global_load_local_store_global_load(i32 addrspace(1)* %out, i32 addrspace(3)* %lptr, i32 addrspace(1)* %ptr0) #0 {
146 %ptr1 = getelementptr inbounds i32 addrspace(1)* %ptr0, i64 1
147 %ptr2 = getelementptr inbounds i32 addrspace(1)* %ptr0, i64 2
149 %tmp1 = load i32 addrspace(1)* %ptr1, align 4
150 store i32 99, i32 addrspace(3)* %lptr, align 4
151 %tmp2 = load i32 addrspace(1)* %ptr2, align 4
153 %add = add nsw i32 %tmp1, %tmp2
155 store i32 %add, i32 addrspace(1)* %out, align 4
159 ; FUNC-LABEL: @reorder_local_offsets
160 ; CI: ds_write_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:12
161 ; CI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:400
162 ; CI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:404
163 ; CI: ds_write_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:400
164 ; CI: ds_write_b32 {{v[0-9]+}}, {{v[0-9]+}} offset:404
165 ; CI: buffer_store_dword
167 define void @reorder_local_offsets(i32 addrspace(1)* nocapture %out, i32 addrspace(1)* noalias nocapture readnone %gptr, i32 addrspace(3)* noalias nocapture %ptr0) #0 {
168 %ptr1 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 3
169 %ptr2 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 100
170 %ptr3 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 101
172 store i32 123, i32 addrspace(3)* %ptr1, align 4
173 %tmp1 = load i32 addrspace(3)* %ptr2, align 4
174 %tmp2 = load i32 addrspace(3)* %ptr3, align 4
175 store i32 123, i32 addrspace(3)* %ptr2, align 4
176 %tmp3 = load i32 addrspace(3)* %ptr1, align 4
177 store i32 789, i32 addrspace(3)* %ptr3, align 4
179 %add.0 = add nsw i32 %tmp2, %tmp1
180 %add.1 = add nsw i32 %add.0, %tmp3
181 store i32 %add.1, i32 addrspace(1)* %out, align 4
185 ; FUNC-LABEL: @reorder_global_offsets
186 ; CI: buffer_store_dword {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, 0 offset:12
187 ; CI: buffer_load_dword {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, 0 offset:400
188 ; CI: buffer_load_dword {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, 0 offset:404
189 ; CI: buffer_store_dword {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, 0 offset:400
190 ; CI: buffer_store_dword {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, 0 offset:404
191 ; CI: buffer_store_dword
193 define void @reorder_global_offsets(i32 addrspace(1)* nocapture %out, i32 addrspace(1)* noalias nocapture readnone %gptr, i32 addrspace(1)* noalias nocapture %ptr0) #0 {
194 %ptr1 = getelementptr inbounds i32 addrspace(1)* %ptr0, i32 3
195 %ptr2 = getelementptr inbounds i32 addrspace(1)* %ptr0, i32 100
196 %ptr3 = getelementptr inbounds i32 addrspace(1)* %ptr0, i32 101
198 store i32 123, i32 addrspace(1)* %ptr1, align 4
199 %tmp1 = load i32 addrspace(1)* %ptr2, align 4
200 %tmp2 = load i32 addrspace(1)* %ptr3, align 4
201 store i32 123, i32 addrspace(1)* %ptr2, align 4
202 %tmp3 = load i32 addrspace(1)* %ptr1, align 4
203 store i32 789, i32 addrspace(1)* %ptr3, align 4
205 %add.0 = add nsw i32 %tmp2, %tmp1
206 %add.1 = add nsw i32 %add.0, %tmp3
207 store i32 %add.1, i32 addrspace(1)* %out, align 4
211 ; XFUNC-LABEL: @reorder_local_load_tbuffer_store_local_load
212 ; XCI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}}, 0x4
213 ; XCI: TBUFFER_STORE_FORMAT
214 ; XCI: ds_read_b32 {{v[0-9]+}}, {{v[0-9]+}}, 0x8
215 ; define void @reorder_local_load_tbuffer_store_local_load(i32 addrspace(1)* %out, i32 %a1, i32 %vaddr) #1 {
216 ; %ptr0 = load i32 addrspace(3)* addrspace(3)* @stored_lds_ptr, align 4
218 ; %ptr1 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 1
219 ; %ptr2 = getelementptr inbounds i32 addrspace(3)* %ptr0, i32 2
221 ; %tmp1 = load i32 addrspace(3)* %ptr1, align 4
223 ; %vdata = insertelement <4 x i32> undef, i32 %a1, i32 0
224 ; call void @llvm.SI.tbuffer.store.v4i32(<16 x i8> undef, <4 x i32> %vdata,
225 ; i32 4, i32 %vaddr, i32 0, i32 32, i32 14, i32 4, i32 1, i32 0, i32 1,
228 ; %tmp2 = load i32 addrspace(3)* %ptr2, align 4
230 ; %add = add nsw i32 %tmp1, %tmp2
232 ; store i32 %add, i32 addrspace(1)* %out, align 4
236 attributes #0 = { nounwind "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "stack-protector-buffer-size"="8" "unsafe-fp-math"="true" "use-soft-float"="false" }
237 attributes #1 = { "ShaderType"="1" nounwind "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "stack-protector-buffer-size"="8" "unsafe-fp-math"="true" "use-soft-float"="false" }
238 attributes #2 = { nounwind noduplicate }