1; RUN: llc -march=amdgcn -mcpu=verde -verify-machineinstrs < %s | FileCheck -check-prefix=CHECK -check-prefix=SI %s
2; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=CHECK -check-prefix=VI %s
3
4; Check that WQM isn't triggered by image load/store intrinsics.
5;
6;CHECK-LABEL: {{^}}test1:
7;CHECK-NOT: s_wqm
8define amdgpu_ps <4 x float> @test1(<8 x i32> inreg %rsrc, <4 x i32> %c) {
9main_body:
10  %tex = call <4 x float> @llvm.amdgcn.image.load.v4f32.v4i32.v8i32(<4 x i32> %c, <8 x i32> %rsrc, i32 15, i1 0, i1 0, i1 0, i1 0)
11  call void @llvm.amdgcn.image.store.v4f32.v4i32.v8i32(<4 x float> %tex, <4 x i32> %c, <8 x i32> %rsrc, i32 15, i1 0, i1 0, i1 0, i1 0)
12  ret <4 x float> %tex
13}
14
15; Check that WQM is triggered by image samples and left untouched for loads...
16;
17;CHECK-LABEL: {{^}}test2:
18;CHECK-NEXT: ; %main_body
19;CHECK-NEXT: s_wqm_b64 exec, exec
20;CHECK-NOT: exec
21define amdgpu_ps void @test2(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, float addrspace(1)* inreg %ptr, <4 x i32> %c) {
22main_body:
23  %c.1 = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
24  %c.2 = bitcast <4 x float> %c.1 to <4 x i32>
25  %c.3 = extractelement <4 x i32> %c.2, i32 0
26  %gep = getelementptr float, float addrspace(1)* %ptr, i32 %c.3
27  %data = load float, float addrspace(1)* %gep
28  call void @llvm.amdgcn.exp.f32(i32 0, i32 15, float %data, float undef, float undef, float undef, i1 true, i1 true) #1
29  ret void
30}
31
32; ... but disabled for stores (and, in this simple case, not re-enabled).
33;
34;CHECK-LABEL: {{^}}test3:
35;CHECK-NEXT: ; %main_body
36;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
37;CHECK-NEXT: s_wqm_b64 exec, exec
38;CHECK: s_and_b64 exec, exec, [[ORIG]]
39;CHECK: image_sample
40;CHECK: store
41;CHECK-NOT: exec
42;CHECK: .size test3
43define amdgpu_ps <4 x float> @test3(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, <4 x i32> %c) {
44main_body:
45  %tex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
46  %tex.1 = bitcast <4 x float> %tex to <4 x i32>
47  %tex.2 = extractelement <4 x i32> %tex.1, i32 0
48
49  call void @llvm.amdgcn.buffer.store.v4f32(<4 x float> %tex, <4 x i32> undef, i32 %tex.2, i32 0, i1 0, i1 0)
50
51  ret <4 x float> %tex
52}
53
54; Check that WQM is re-enabled when required.
55;
56;CHECK-LABEL: {{^}}test4:
57;CHECK-NEXT: ; %main_body
58;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
59;CHECK-NEXT: s_wqm_b64 exec, exec
60;CHECK: v_mul_lo_i32 [[MUL:v[0-9]+]], v0, v1
61;CHECK: s_and_b64 exec, exec, [[ORIG]]
62;CHECK: store
63;CHECK: s_wqm_b64 exec, exec
64;CHECK: image_sample
65;CHECK: image_sample
66define amdgpu_ps <4 x float> @test4(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, float addrspace(1)* inreg %ptr, i32 %c, i32 %d, float %data) {
67main_body:
68  %c.1 = mul i32 %c, %d
69
70  call void @llvm.amdgcn.buffer.store.v4f32(<4 x float> undef, <4 x i32> undef, i32 %c.1, i32 0, i1 0, i1 0)
71
72  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %c.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
73  %tex.1 = bitcast <4 x float> %tex to <4 x i32>
74  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
75  ret <4 x float> %dtex
76}
77
78; Check a case of one branch of an if-else requiring WQM, the other requiring
79; exact.
80;
81; Note: In this particular case, the save-and-restore could be avoided if the
82; analysis understood that the two branches of the if-else are mutually
83; exclusive.
84;
85;CHECK-LABEL: {{^}}test_control_flow_0:
86;CHECK-NEXT: ; %main_body
87;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
88;CHECK-NEXT: s_wqm_b64 exec, exec
89;CHECK: %ELSE
90;CHECK: s_and_saveexec_b64 [[SAVED:s\[[0-9]+:[0-9]+\]]], [[ORIG]]
91;CHECK: store
92;CHECK: s_mov_b64 exec, [[SAVED]]
93;CHECK: %IF
94;CHECK: image_sample
95;CHECK: image_sample
96define amdgpu_ps float @test_control_flow_0(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %c, i32 %z, float %data) {
97main_body:
98  %cmp = icmp eq i32 %z, 0
99  br i1 %cmp, label %IF, label %ELSE
100
101IF:
102  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
103  %tex.1 = bitcast <4 x float> %tex to <4 x i32>
104  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
105  %data.if = extractelement <4 x float> %dtex, i32 0
106  br label %END
107
108ELSE:
109  call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 %c, i32 0, i1 0, i1 0)
110  br label %END
111
112END:
113  %r = phi float [ %data.if, %IF ], [ %data, %ELSE ]
114  ret float %r
115}
116
117; Reverse branch order compared to the previous test.
118;
119;CHECK-LABEL: {{^}}test_control_flow_1:
120;CHECK-NEXT: ; %main_body
121;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
122;CHECK-NEXT: s_wqm_b64 exec, exec
123;CHECK: %IF
124;CHECK: image_sample
125;CHECK: image_sample
126;CHECK: %Flow
127;CHECK-NEXT: s_or_saveexec_b64 [[SAVED:s\[[0-9]+:[0-9]+\]]],
128;CHECK-NEXT: s_and_b64 exec, exec, [[ORIG]]
129;CHECK-NEXT: s_and_b64 [[SAVED]], exec, [[SAVED]]
130;CHECK-NEXT: s_xor_b64 exec, exec, [[SAVED]]
131;CHECK-NEXT: mask branch [[END_BB:BB[0-9]+_[0-9]+]]
132;CHECK-NEXT: BB{{[0-9]+_[0-9]+}}: ; %ELSE
133;CHECK: store_dword
134;CHECK: [[END_BB]]: ; %END
135;CHECK: s_or_b64 exec, exec,
136;CHECK: v_mov_b32_e32 v0
137;CHECK: ; return
138define amdgpu_ps float @test_control_flow_1(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %c, i32 %z, float %data) {
139main_body:
140  %cmp = icmp eq i32 %z, 0
141  br i1 %cmp, label %ELSE, label %IF
142
143IF:
144  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
145  %tex.1 = bitcast <4 x float> %tex to <4 x i32>
146  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
147  %data.if = extractelement <4 x float> %dtex, i32 0
148  br label %END
149
150ELSE:
151  call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 %c, i32 0, i1 0, i1 0)
152  br label %END
153
154END:
155  %r = phi float [ %data.if, %IF ], [ %data, %ELSE ]
156  ret float %r
157}
158
159; Check that branch conditions are properly marked as needing WQM...
160;
161;CHECK-LABEL: {{^}}test_control_flow_2:
162;CHECK-NEXT: ; %main_body
163;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
164;CHECK-NEXT: s_wqm_b64 exec, exec
165;CHECK: s_and_b64 exec, exec, [[ORIG]]
166;CHECK: store
167;CHECK: s_wqm_b64 exec, exec
168;CHECK: load
169;CHECK: s_and_b64 exec, exec, [[ORIG]]
170;CHECK: store
171;CHECK: s_wqm_b64 exec, exec
172;CHECK: v_cmp
173define amdgpu_ps <4 x float> @test_control_flow_2(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, <3 x i32> %idx, <2 x float> %data, i32 %coord) {
174main_body:
175  %idx.1 = extractelement <3 x i32> %idx, i32 0
176  %data.1 = extractelement <2 x float> %data, i32 0
177  call void @llvm.amdgcn.buffer.store.f32(float %data.1, <4 x i32> undef, i32 %idx.1, i32 0, i1 0, i1 0)
178
179  ; The load that determines the branch (and should therefore be WQM) is
180  ; surrounded by stores that require disabled WQM.
181  %idx.2 = extractelement <3 x i32> %idx, i32 1
182  %z = call float @llvm.amdgcn.buffer.load.f32(<4 x i32> undef, i32 %idx.2, i32 0, i1 0, i1 0)
183
184  %idx.3 = extractelement <3 x i32> %idx, i32 2
185  %data.3 = extractelement <2 x float> %data, i32 1
186  call void @llvm.amdgcn.buffer.store.f32(float %data.3, <4 x i32> undef, i32 %idx.3, i32 0, i1 0, i1 0)
187
188  %cc = fcmp ogt float %z, 0.0
189  br i1 %cc, label %IF, label %ELSE
190
191IF:
192  %coord.IF = mul i32 %coord, 3
193  br label %END
194
195ELSE:
196  %coord.ELSE = mul i32 %coord, 4
197  br label %END
198
199END:
200  %coord.END = phi i32 [ %coord.IF, %IF ], [ %coord.ELSE, %ELSE ]
201  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord.END, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
202  ret <4 x float> %tex
203}
204
205; ... but only if they really do need it.
206;
207;CHECK-LABEL: {{^}}test_control_flow_3:
208;CHECK-NEXT: ; %main_body
209;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
210;CHECK-NEXT: s_wqm_b64 exec, exec
211;CHECK: image_sample
212;CHECK: s_and_b64 exec, exec, [[ORIG]]
213;CHECK: image_sample
214;CHECK: v_cmp
215;CHECK: store
216define amdgpu_ps float @test_control_flow_3(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %idx, i32 %coord) {
217main_body:
218  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
219  %tex.1 = bitcast <4 x float> %tex to <4 x i32>
220  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
221  %dtex.1 = extractelement <4 x float> %dtex, i32 0
222
223  call void @llvm.amdgcn.buffer.store.f32(float %dtex.1, <4 x i32> undef, i32 %idx, i32 0, i1 0, i1 0)
224
225  %cc = fcmp ogt float %dtex.1, 0.0
226  br i1 %cc, label %IF, label %ELSE
227
228IF:
229  %tex.IF = fmul float %dtex.1, 3.0
230  br label %END
231
232ELSE:
233  %tex.ELSE = fmul float %dtex.1, 4.0
234  br label %END
235
236END:
237  %tex.END = phi float [ %tex.IF, %IF ], [ %tex.ELSE, %ELSE ]
238  ret float %tex.END
239}
240
241; Another test that failed at some point because of terminator handling.
242;
243;CHECK-LABEL: {{^}}test_control_flow_4:
244;CHECK-NEXT: ; %main_body
245;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
246;CHECK-NEXT: s_wqm_b64 exec, exec
247;CHECK: %IF
248;CHECK: s_and_saveexec_b64 [[SAVE:s\[[0-9]+:[0-9]+\]]],  [[ORIG]]
249;CHECK: load
250;CHECK: store
251;CHECK: s_mov_b64 exec, [[SAVE]]
252;CHECK: %END
253;CHECK: image_sample
254;CHECK: image_sample
255define amdgpu_ps <4 x float> @test_control_flow_4(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %coord, i32 %y, float %z) {
256main_body:
257  %cond = icmp eq i32 %y, 0
258  br i1 %cond, label %IF, label %END
259
260IF:
261  %data = call float @llvm.amdgcn.buffer.load.f32(<4 x i32> undef, i32 0, i32 0, i1 0, i1 0)
262  call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 1, i32 0, i1 0, i1 0)
263  br label %END
264
265END:
266  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
267  %tex.1 = bitcast <4 x float> %tex to <4 x i32>
268  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
269  ret <4 x float> %dtex
270}
271
272; Kill is performed in WQM mode so that uniform kill behaves correctly ...
273;
274;CHECK-LABEL: {{^}}test_kill_0:
275;CHECK-NEXT: ; %main_body
276;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
277;CHECK-NEXT: s_wqm_b64 exec, exec
278;CHECK: s_and_b64 exec, exec, [[ORIG]]
279;CHECK: image_sample
280;CHECK: buffer_store_dword
281;CHECK: s_wqm_b64 exec, exec
282;CHECK: v_cmpx_
283;CHECK: s_and_saveexec_b64 [[SAVE:s\[[0-9]+:[0-9]+\]]], [[ORIG]]
284;CHECK: buffer_store_dword
285;CHECK: s_mov_b64 exec, [[SAVE]]
286;CHECK: image_sample
287define amdgpu_ps <4 x float> @test_kill_0(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, float addrspace(1)* inreg %ptr, <2 x i32> %idx, <2 x float> %data, i32 %coord, i32 %coord2, float %z) {
288main_body:
289  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
290
291  %idx.0 = extractelement <2 x i32> %idx, i32 0
292  %data.0 = extractelement <2 x float> %data, i32 0
293  call void @llvm.amdgcn.buffer.store.f32(float %data.0, <4 x i32> undef, i32 %idx.0, i32 0, i1 0, i1 0)
294
295  call void @llvm.AMDGPU.kill(float %z)
296
297  %idx.1 = extractelement <2 x i32> %idx, i32 1
298  %data.1 = extractelement <2 x float> %data, i32 1
299  call void @llvm.amdgcn.buffer.store.f32(float %data.1, <4 x i32> undef, i32 %idx.1, i32 0, i1 0, i1 0)
300
301  %tex2 = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord2, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
302  %tex2.1 = bitcast <4 x float> %tex2 to <4 x i32>
303  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex2.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
304  %out = fadd <4 x float> %tex, %dtex
305
306  ret <4 x float> %out
307}
308
309; ... but only if WQM is necessary.
310;
311; CHECK-LABEL: {{^}}test_kill_1:
312; CHECK-NEXT: ; %main_body
313; CHECK: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
314; CHECK: s_wqm_b64 exec, exec
315; CHECK: image_sample
316; CHECK: s_and_b64 exec, exec, [[ORIG]]
317; CHECK: image_sample
318; CHECK: buffer_store_dword
319; CHECK-NOT: wqm
320; CHECK: v_cmpx_
321define amdgpu_ps <4 x float> @test_kill_1(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %idx, float %data, i32 %coord, i32 %coord2, float %z) {
322main_body:
323  %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
324  %tex.1 = bitcast <4 x float> %tex to <4 x i32>
325  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
326
327  call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 0, i32 0, i1 0, i1 0)
328
329  call void @llvm.AMDGPU.kill(float %z)
330
331  ret <4 x float> %dtex
332}
333
334; Check prolog shaders.
335;
336; CHECK-LABEL: {{^}}test_prolog_1:
337; CHECK: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
338; CHECK: s_wqm_b64 exec, exec
339; CHECK: v_add_f32_e32 v0,
340; CHECK: s_and_b64 exec, exec, [[ORIG]]
341define amdgpu_ps float @test_prolog_1(float %a, float %b) #4 {
342main_body:
343  %s = fadd float %a, %b
344  ret float %s
345}
346
347; CHECK-LABEL: {{^}}test_loop_vcc:
348; CHECK-NEXT: ; %entry
349; CHECK-NEXT: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec
350; CHECK: s_wqm_b64 exec, exec
351; CHECK: s_and_b64 exec, exec, [[LIVE]]
352; CHECK: image_store
353; CHECK: s_wqm_b64 exec, exec
354; CHECK-DAG: v_mov_b32_e32 [[CTR:v[0-9]+]], 0
355; CHECK-DAG: v_mov_b32_e32 [[SEVEN:v[0-9]+]], 0x40e00000
356
357; CHECK: [[LOOPHDR:BB[0-9]+_[0-9]+]]: ; %body
358; CHECK: v_add_f32_e32 [[CTR]], 2.0, [[CTR]]
359; CHECK: v_cmp_lt_f32_e32 vcc, [[SEVEN]], [[CTR]]
360; CHECK: s_cbranch_vccz [[LOOPHDR]]
361; CHECK: ; %break
362
363; CHECK: ; return
364define amdgpu_ps <4 x float> @test_loop_vcc(<4 x float> %in) nounwind {
365entry:
366  call void @llvm.amdgcn.image.store.v4f32.v4i32.v8i32(<4 x float> %in, <4 x i32> undef, <8 x i32> undef, i32 15, i1 0, i1 0, i1 0, i1 0)
367  br label %loop
368
369loop:
370  %ctr.iv = phi float [ 0.0, %entry ], [ %ctr.next, %body ]
371  %c.iv = phi <4 x float> [ %in, %entry ], [ %c.next, %body ]
372  %cc = fcmp ogt float %ctr.iv, 7.0
373  br i1 %cc, label %break, label %body
374
375body:
376  %c.i = bitcast <4 x float> %c.iv to <4 x i32>
377  %c.next = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %c.i, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
378  %ctr.next = fadd float %ctr.iv, 2.0
379  br label %loop
380
381break:
382  ret <4 x float> %c.iv
383}
384
385; Only intrinsic stores need exact execution -- other stores do not have
386; externally visible effects and may require WQM for correctness.
387;
388; CHECK-LABEL: {{^}}test_alloca:
389; CHECK: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec
390; CHECK: s_wqm_b64 exec, exec
391
392; CHECK: s_and_b64 exec, exec, [[LIVE]]
393; CHECK: buffer_store_dword {{v[0-9]+}}, off, {{s\[[0-9]+:[0-9]+\]}}, 0
394; CHECK: s_wqm_b64 exec, exec
395; CHECK: buffer_store_dword {{v[0-9]+}}, off, {{s\[[0-9]+:[0-9]+\]}}, {{s[0-9]+}} offset:4{{$}}
396; CHECK: s_and_b64 exec, exec, [[LIVE]]
397; CHECK: buffer_store_dword {{v[0-9]+}}, {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, 0 idxen
398; CHECK: s_wqm_b64 exec, exec
399; CHECK: buffer_load_dword {{v[0-9]+}}, {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, {{s[0-9]+}} offen
400
401; CHECK: s_and_b64 exec, exec, [[LIVE]]
402; CHECK: image_sample
403; CHECK: buffer_store_dwordx4
404define amdgpu_ps void @test_alloca(float %data, i32 %a, i32 %idx) nounwind {
405entry:
406  %array = alloca [32 x i32], align 4
407
408  call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 0, i32 0, i1 0, i1 0)
409
410  %s.gep = getelementptr [32 x i32], [32 x i32]* %array, i32 0, i32 0
411  store volatile i32 %a, i32* %s.gep, align 4
412
413  call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 1, i32 0, i1 0, i1 0)
414
415  %c.gep = getelementptr [32 x i32], [32 x i32]* %array, i32 0, i32 %idx
416  %c = load i32, i32* %c.gep, align 4
417
418  %t = call <4 x float> @llvm.SI.image.sample.i32(i32 %c, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
419
420  call void @llvm.amdgcn.buffer.store.v4f32(<4 x float> %t, <4 x i32> undef, i32 0, i32 0, i1 0, i1 0)
421
422  ret void
423}
424
425; Must return to exact at the end of a non-void returning shader,
426; otherwise the EXEC mask exported by the epilog will be wrong. This is true
427; even if the shader has no kills, because a kill could have happened in a
428; previous shader fragment.
429;
430; CHECK-LABEL: {{^}}test_nonvoid_return:
431; CHECK: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec
432; CHECK: s_wqm_b64 exec, exec
433;
434; CHECK: s_and_b64 exec, exec, [[LIVE]]
435; CHECK-NOT: exec
436define amdgpu_ps <4 x float> @test_nonvoid_return() nounwind {
437  %tex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> undef, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
438  %tex.i = bitcast <4 x float> %tex to <4 x i32>
439  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.i, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
440  ret <4 x float> %dtex
441}
442
443; CHECK-LABEL: {{^}}test_nonvoid_return_unreachable:
444; CHECK: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec
445; CHECK: s_wqm_b64 exec, exec
446;
447; CHECK: s_and_b64 exec, exec, [[LIVE]]
448; CHECK-NOT: exec
449define amdgpu_ps <4 x float> @test_nonvoid_return_unreachable(i32 inreg %c) nounwind {
450entry:
451  %tex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> undef, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
452  %tex.i = bitcast <4 x float> %tex to <4 x i32>
453  %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.i, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
454
455  %cc = icmp sgt i32 %c, 0
456  br i1 %cc, label %if, label %else
457
458if:
459  store volatile <4 x float> %dtex, <4 x float> addrspace(1)* undef
460  unreachable
461
462else:
463  ret <4 x float> %dtex
464}
465
466; Test awareness that s_wqm_b64 clobbers SCC.
467;
468; CHECK-LABEL: {{^}}test_scc:
469; CHECK: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec
470; CHECK: s_wqm_b64 exec, exec
471; CHECK: s_cmp_
472; CHECK-NEXT: s_cbranch_scc
473; CHECK: ; %if
474; CHECK: s_and_b64 exec, exec, [[ORIG]]
475; CHECK: image_sample
476; CHECK: ; %else
477; CHECK: s_and_b64 exec, exec, [[ORIG]]
478; CHECK: image_sample
479; CHECK: ; %end
480define amdgpu_ps <4 x float> @test_scc(i32 inreg %sel, i32 %idx) #1 {
481main_body:
482  %cc = icmp sgt i32 %sel, 0
483  br i1 %cc, label %if, label %else
484
485if:
486  %r.if = call <4 x float> @llvm.SI.image.sample.i32(i32 0, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
487  br label %end
488
489else:
490  %r.else = call <4 x float> @llvm.SI.image.sample.v2i32(<2 x i32> <i32 0, i32 1>, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0)
491  br label %end
492
493end:
494  %r = phi <4 x float> [ %r.if, %if ], [ %r.else, %else ]
495
496  call void @llvm.amdgcn.buffer.store.f32(float 1.0, <4 x i32> undef, i32 %idx, i32 0, i1 0, i1 0)
497
498  ret <4 x float> %r
499}
500
501declare void @llvm.amdgcn.exp.f32(i32, i32, float, float, float, float, i1, i1) #1
502declare void @llvm.amdgcn.image.store.v4f32.v4i32.v8i32(<4 x float>, <4 x i32>, <8 x i32>, i32, i1, i1, i1, i1) #1
503declare void @llvm.amdgcn.buffer.store.f32(float, <4 x i32>, i32, i32, i1, i1) #1
504declare void @llvm.amdgcn.buffer.store.v4f32(<4 x float>, <4 x i32>, i32, i32, i1, i1) #1
505
506declare <4 x float> @llvm.amdgcn.image.load.v4f32.v4i32.v8i32(<4 x i32>, <8 x i32>, i32, i1, i1, i1, i1) #2
507declare float @llvm.amdgcn.buffer.load.f32(<4 x i32>, i32, i32, i1, i1) #2
508
509declare <4 x float> @llvm.SI.image.sample.i32(i32, <8 x i32>, <4 x i32>, i32, i32, i32, i32, i32, i32, i32, i32) #3
510declare <4 x float> @llvm.SI.image.sample.v2i32(<2 x i32>, <8 x i32>, <4 x i32>, i32, i32, i32, i32, i32, i32, i32, i32) #3
511declare <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32>, <8 x i32>, <4 x i32>, i32, i32, i32, i32, i32, i32, i32, i32) #3
512
513declare void @llvm.AMDGPU.kill(float) #1
514
515attributes #1 = { nounwind }
516attributes #2 = { nounwind readonly }
517attributes #3 = { nounwind readnone }
518attributes #4 = { "amdgpu-ps-wqm-outputs" }
519