1; RUN: llc -march=amdgcn -mcpu=verde -verify-machineinstrs < %s | FileCheck -check-prefix=CHECK -check-prefix=SI %s 2; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=CHECK -check-prefix=VI %s 3 4; Check that WQM isn't triggered by image load/store intrinsics. 5; 6;CHECK-LABEL: {{^}}test1: 7;CHECK-NOT: s_wqm 8define amdgpu_ps <4 x float> @test1(<8 x i32> inreg %rsrc, <4 x i32> %c) { 9main_body: 10 %tex = call <4 x float> @llvm.amdgcn.image.load.v4f32.v4i32.v8i32(<4 x i32> %c, <8 x i32> %rsrc, i32 15, i1 0, i1 0, i1 0, i1 0) 11 call void @llvm.amdgcn.image.store.v4f32.v4i32.v8i32(<4 x float> %tex, <4 x i32> %c, <8 x i32> %rsrc, i32 15, i1 0, i1 0, i1 0, i1 0) 12 ret <4 x float> %tex 13} 14 15; Check that WQM is triggered by image samples and left untouched for loads... 16; 17;CHECK-LABEL: {{^}}test2: 18;CHECK-NEXT: ; %main_body 19;CHECK-NEXT: s_wqm_b64 exec, exec 20;CHECK-NOT: exec 21define amdgpu_ps void @test2(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, float addrspace(1)* inreg %ptr, <4 x i32> %c) { 22main_body: 23 %c.1 = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 24 %c.2 = bitcast <4 x float> %c.1 to <4 x i32> 25 %c.3 = extractelement <4 x i32> %c.2, i32 0 26 %gep = getelementptr float, float addrspace(1)* %ptr, i32 %c.3 27 %data = load float, float addrspace(1)* %gep 28 call void @llvm.amdgcn.exp.f32(i32 0, i32 15, float %data, float undef, float undef, float undef, i1 true, i1 true) #1 29 ret void 30} 31 32; ... but disabled for stores (and, in this simple case, not re-enabled). 33; 34;CHECK-LABEL: {{^}}test3: 35;CHECK-NEXT: ; %main_body 36;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 37;CHECK-NEXT: s_wqm_b64 exec, exec 38;CHECK: s_and_b64 exec, exec, [[ORIG]] 39;CHECK: image_sample 40;CHECK: store 41;CHECK-NOT: exec 42;CHECK: .size test3 43define amdgpu_ps <4 x float> @test3(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, <4 x i32> %c) { 44main_body: 45 %tex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 46 %tex.1 = bitcast <4 x float> %tex to <4 x i32> 47 %tex.2 = extractelement <4 x i32> %tex.1, i32 0 48 49 call void @llvm.amdgcn.buffer.store.v4f32(<4 x float> %tex, <4 x i32> undef, i32 %tex.2, i32 0, i1 0, i1 0) 50 51 ret <4 x float> %tex 52} 53 54; Check that WQM is re-enabled when required. 55; 56;CHECK-LABEL: {{^}}test4: 57;CHECK-NEXT: ; %main_body 58;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 59;CHECK-NEXT: s_wqm_b64 exec, exec 60;CHECK: v_mul_lo_i32 [[MUL:v[0-9]+]], v0, v1 61;CHECK: s_and_b64 exec, exec, [[ORIG]] 62;CHECK: store 63;CHECK: s_wqm_b64 exec, exec 64;CHECK: image_sample 65;CHECK: image_sample 66define amdgpu_ps <4 x float> @test4(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, float addrspace(1)* inreg %ptr, i32 %c, i32 %d, float %data) { 67main_body: 68 %c.1 = mul i32 %c, %d 69 70 call void @llvm.amdgcn.buffer.store.v4f32(<4 x float> undef, <4 x i32> undef, i32 %c.1, i32 0, i1 0, i1 0) 71 72 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %c.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 73 %tex.1 = bitcast <4 x float> %tex to <4 x i32> 74 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 75 ret <4 x float> %dtex 76} 77 78; Check a case of one branch of an if-else requiring WQM, the other requiring 79; exact. 80; 81; Note: In this particular case, the save-and-restore could be avoided if the 82; analysis understood that the two branches of the if-else are mutually 83; exclusive. 84; 85;CHECK-LABEL: {{^}}test_control_flow_0: 86;CHECK-NEXT: ; %main_body 87;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 88;CHECK-NEXT: s_wqm_b64 exec, exec 89;CHECK: %ELSE 90;CHECK: s_and_saveexec_b64 [[SAVED:s\[[0-9]+:[0-9]+\]]], [[ORIG]] 91;CHECK: store 92;CHECK: s_mov_b64 exec, [[SAVED]] 93;CHECK: %IF 94;CHECK: image_sample 95;CHECK: image_sample 96define amdgpu_ps float @test_control_flow_0(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %c, i32 %z, float %data) { 97main_body: 98 %cmp = icmp eq i32 %z, 0 99 br i1 %cmp, label %IF, label %ELSE 100 101IF: 102 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 103 %tex.1 = bitcast <4 x float> %tex to <4 x i32> 104 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 105 %data.if = extractelement <4 x float> %dtex, i32 0 106 br label %END 107 108ELSE: 109 call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 %c, i32 0, i1 0, i1 0) 110 br label %END 111 112END: 113 %r = phi float [ %data.if, %IF ], [ %data, %ELSE ] 114 ret float %r 115} 116 117; Reverse branch order compared to the previous test. 118; 119;CHECK-LABEL: {{^}}test_control_flow_1: 120;CHECK-NEXT: ; %main_body 121;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 122;CHECK-NEXT: s_wqm_b64 exec, exec 123;CHECK: %IF 124;CHECK: image_sample 125;CHECK: image_sample 126;CHECK: %Flow 127;CHECK-NEXT: s_or_saveexec_b64 [[SAVED:s\[[0-9]+:[0-9]+\]]], 128;CHECK-NEXT: s_and_b64 exec, exec, [[ORIG]] 129;CHECK-NEXT: s_and_b64 [[SAVED]], exec, [[SAVED]] 130;CHECK-NEXT: s_xor_b64 exec, exec, [[SAVED]] 131;CHECK-NEXT: mask branch [[END_BB:BB[0-9]+_[0-9]+]] 132;CHECK-NEXT: BB{{[0-9]+_[0-9]+}}: ; %ELSE 133;CHECK: store_dword 134;CHECK: [[END_BB]]: ; %END 135;CHECK: s_or_b64 exec, exec, 136;CHECK: v_mov_b32_e32 v0 137;CHECK: ; return 138define amdgpu_ps float @test_control_flow_1(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %c, i32 %z, float %data) { 139main_body: 140 %cmp = icmp eq i32 %z, 0 141 br i1 %cmp, label %ELSE, label %IF 142 143IF: 144 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %c, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 145 %tex.1 = bitcast <4 x float> %tex to <4 x i32> 146 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 147 %data.if = extractelement <4 x float> %dtex, i32 0 148 br label %END 149 150ELSE: 151 call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 %c, i32 0, i1 0, i1 0) 152 br label %END 153 154END: 155 %r = phi float [ %data.if, %IF ], [ %data, %ELSE ] 156 ret float %r 157} 158 159; Check that branch conditions are properly marked as needing WQM... 160; 161;CHECK-LABEL: {{^}}test_control_flow_2: 162;CHECK-NEXT: ; %main_body 163;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 164;CHECK-NEXT: s_wqm_b64 exec, exec 165;CHECK: s_and_b64 exec, exec, [[ORIG]] 166;CHECK: store 167;CHECK: s_wqm_b64 exec, exec 168;CHECK: load 169;CHECK: s_and_b64 exec, exec, [[ORIG]] 170;CHECK: store 171;CHECK: s_wqm_b64 exec, exec 172;CHECK: v_cmp 173define amdgpu_ps <4 x float> @test_control_flow_2(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, <3 x i32> %idx, <2 x float> %data, i32 %coord) { 174main_body: 175 %idx.1 = extractelement <3 x i32> %idx, i32 0 176 %data.1 = extractelement <2 x float> %data, i32 0 177 call void @llvm.amdgcn.buffer.store.f32(float %data.1, <4 x i32> undef, i32 %idx.1, i32 0, i1 0, i1 0) 178 179 ; The load that determines the branch (and should therefore be WQM) is 180 ; surrounded by stores that require disabled WQM. 181 %idx.2 = extractelement <3 x i32> %idx, i32 1 182 %z = call float @llvm.amdgcn.buffer.load.f32(<4 x i32> undef, i32 %idx.2, i32 0, i1 0, i1 0) 183 184 %idx.3 = extractelement <3 x i32> %idx, i32 2 185 %data.3 = extractelement <2 x float> %data, i32 1 186 call void @llvm.amdgcn.buffer.store.f32(float %data.3, <4 x i32> undef, i32 %idx.3, i32 0, i1 0, i1 0) 187 188 %cc = fcmp ogt float %z, 0.0 189 br i1 %cc, label %IF, label %ELSE 190 191IF: 192 %coord.IF = mul i32 %coord, 3 193 br label %END 194 195ELSE: 196 %coord.ELSE = mul i32 %coord, 4 197 br label %END 198 199END: 200 %coord.END = phi i32 [ %coord.IF, %IF ], [ %coord.ELSE, %ELSE ] 201 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord.END, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 202 ret <4 x float> %tex 203} 204 205; ... but only if they really do need it. 206; 207;CHECK-LABEL: {{^}}test_control_flow_3: 208;CHECK-NEXT: ; %main_body 209;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 210;CHECK-NEXT: s_wqm_b64 exec, exec 211;CHECK: image_sample 212;CHECK: s_and_b64 exec, exec, [[ORIG]] 213;CHECK: image_sample 214;CHECK: v_cmp 215;CHECK: store 216define amdgpu_ps float @test_control_flow_3(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %idx, i32 %coord) { 217main_body: 218 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 219 %tex.1 = bitcast <4 x float> %tex to <4 x i32> 220 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 221 %dtex.1 = extractelement <4 x float> %dtex, i32 0 222 223 call void @llvm.amdgcn.buffer.store.f32(float %dtex.1, <4 x i32> undef, i32 %idx, i32 0, i1 0, i1 0) 224 225 %cc = fcmp ogt float %dtex.1, 0.0 226 br i1 %cc, label %IF, label %ELSE 227 228IF: 229 %tex.IF = fmul float %dtex.1, 3.0 230 br label %END 231 232ELSE: 233 %tex.ELSE = fmul float %dtex.1, 4.0 234 br label %END 235 236END: 237 %tex.END = phi float [ %tex.IF, %IF ], [ %tex.ELSE, %ELSE ] 238 ret float %tex.END 239} 240 241; Another test that failed at some point because of terminator handling. 242; 243;CHECK-LABEL: {{^}}test_control_flow_4: 244;CHECK-NEXT: ; %main_body 245;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 246;CHECK-NEXT: s_wqm_b64 exec, exec 247;CHECK: %IF 248;CHECK: s_and_saveexec_b64 [[SAVE:s\[[0-9]+:[0-9]+\]]], [[ORIG]] 249;CHECK: load 250;CHECK: store 251;CHECK: s_mov_b64 exec, [[SAVE]] 252;CHECK: %END 253;CHECK: image_sample 254;CHECK: image_sample 255define amdgpu_ps <4 x float> @test_control_flow_4(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %coord, i32 %y, float %z) { 256main_body: 257 %cond = icmp eq i32 %y, 0 258 br i1 %cond, label %IF, label %END 259 260IF: 261 %data = call float @llvm.amdgcn.buffer.load.f32(<4 x i32> undef, i32 0, i32 0, i1 0, i1 0) 262 call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 1, i32 0, i1 0, i1 0) 263 br label %END 264 265END: 266 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 267 %tex.1 = bitcast <4 x float> %tex to <4 x i32> 268 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 269 ret <4 x float> %dtex 270} 271 272; Kill is performed in WQM mode so that uniform kill behaves correctly ... 273; 274;CHECK-LABEL: {{^}}test_kill_0: 275;CHECK-NEXT: ; %main_body 276;CHECK-NEXT: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 277;CHECK-NEXT: s_wqm_b64 exec, exec 278;CHECK: s_and_b64 exec, exec, [[ORIG]] 279;CHECK: image_sample 280;CHECK: buffer_store_dword 281;CHECK: s_wqm_b64 exec, exec 282;CHECK: v_cmpx_ 283;CHECK: s_and_saveexec_b64 [[SAVE:s\[[0-9]+:[0-9]+\]]], [[ORIG]] 284;CHECK: buffer_store_dword 285;CHECK: s_mov_b64 exec, [[SAVE]] 286;CHECK: image_sample 287define amdgpu_ps <4 x float> @test_kill_0(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, float addrspace(1)* inreg %ptr, <2 x i32> %idx, <2 x float> %data, i32 %coord, i32 %coord2, float %z) { 288main_body: 289 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 290 291 %idx.0 = extractelement <2 x i32> %idx, i32 0 292 %data.0 = extractelement <2 x float> %data, i32 0 293 call void @llvm.amdgcn.buffer.store.f32(float %data.0, <4 x i32> undef, i32 %idx.0, i32 0, i1 0, i1 0) 294 295 call void @llvm.AMDGPU.kill(float %z) 296 297 %idx.1 = extractelement <2 x i32> %idx, i32 1 298 %data.1 = extractelement <2 x float> %data, i32 1 299 call void @llvm.amdgcn.buffer.store.f32(float %data.1, <4 x i32> undef, i32 %idx.1, i32 0, i1 0, i1 0) 300 301 %tex2 = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord2, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 302 %tex2.1 = bitcast <4 x float> %tex2 to <4 x i32> 303 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex2.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 304 %out = fadd <4 x float> %tex, %dtex 305 306 ret <4 x float> %out 307} 308 309; ... but only if WQM is necessary. 310; 311; CHECK-LABEL: {{^}}test_kill_1: 312; CHECK-NEXT: ; %main_body 313; CHECK: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 314; CHECK: s_wqm_b64 exec, exec 315; CHECK: image_sample 316; CHECK: s_and_b64 exec, exec, [[ORIG]] 317; CHECK: image_sample 318; CHECK: buffer_store_dword 319; CHECK-NOT: wqm 320; CHECK: v_cmpx_ 321define amdgpu_ps <4 x float> @test_kill_1(<8 x i32> inreg %rsrc, <4 x i32> inreg %sampler, i32 %idx, float %data, i32 %coord, i32 %coord2, float %z) { 322main_body: 323 %tex = call <4 x float> @llvm.SI.image.sample.i32(i32 %coord, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 324 %tex.1 = bitcast <4 x float> %tex to <4 x i32> 325 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.1, <8 x i32> %rsrc, <4 x i32> %sampler, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 326 327 call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 0, i32 0, i1 0, i1 0) 328 329 call void @llvm.AMDGPU.kill(float %z) 330 331 ret <4 x float> %dtex 332} 333 334; Check prolog shaders. 335; 336; CHECK-LABEL: {{^}}test_prolog_1: 337; CHECK: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 338; CHECK: s_wqm_b64 exec, exec 339; CHECK: v_add_f32_e32 v0, 340; CHECK: s_and_b64 exec, exec, [[ORIG]] 341define amdgpu_ps float @test_prolog_1(float %a, float %b) #4 { 342main_body: 343 %s = fadd float %a, %b 344 ret float %s 345} 346 347; CHECK-LABEL: {{^}}test_loop_vcc: 348; CHECK-NEXT: ; %entry 349; CHECK-NEXT: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec 350; CHECK: s_wqm_b64 exec, exec 351; CHECK: s_and_b64 exec, exec, [[LIVE]] 352; CHECK: image_store 353; CHECK: s_wqm_b64 exec, exec 354; CHECK-DAG: v_mov_b32_e32 [[CTR:v[0-9]+]], 0 355; CHECK-DAG: v_mov_b32_e32 [[SEVEN:v[0-9]+]], 0x40e00000 356 357; CHECK: [[LOOPHDR:BB[0-9]+_[0-9]+]]: ; %body 358; CHECK: v_add_f32_e32 [[CTR]], 2.0, [[CTR]] 359; CHECK: v_cmp_lt_f32_e32 vcc, [[SEVEN]], [[CTR]] 360; CHECK: s_cbranch_vccz [[LOOPHDR]] 361; CHECK: ; %break 362 363; CHECK: ; return 364define amdgpu_ps <4 x float> @test_loop_vcc(<4 x float> %in) nounwind { 365entry: 366 call void @llvm.amdgcn.image.store.v4f32.v4i32.v8i32(<4 x float> %in, <4 x i32> undef, <8 x i32> undef, i32 15, i1 0, i1 0, i1 0, i1 0) 367 br label %loop 368 369loop: 370 %ctr.iv = phi float [ 0.0, %entry ], [ %ctr.next, %body ] 371 %c.iv = phi <4 x float> [ %in, %entry ], [ %c.next, %body ] 372 %cc = fcmp ogt float %ctr.iv, 7.0 373 br i1 %cc, label %break, label %body 374 375body: 376 %c.i = bitcast <4 x float> %c.iv to <4 x i32> 377 %c.next = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %c.i, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 378 %ctr.next = fadd float %ctr.iv, 2.0 379 br label %loop 380 381break: 382 ret <4 x float> %c.iv 383} 384 385; Only intrinsic stores need exact execution -- other stores do not have 386; externally visible effects and may require WQM for correctness. 387; 388; CHECK-LABEL: {{^}}test_alloca: 389; CHECK: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec 390; CHECK: s_wqm_b64 exec, exec 391 392; CHECK: s_and_b64 exec, exec, [[LIVE]] 393; CHECK: buffer_store_dword {{v[0-9]+}}, off, {{s\[[0-9]+:[0-9]+\]}}, 0 394; CHECK: s_wqm_b64 exec, exec 395; CHECK: buffer_store_dword {{v[0-9]+}}, off, {{s\[[0-9]+:[0-9]+\]}}, {{s[0-9]+}} offset:4{{$}} 396; CHECK: s_and_b64 exec, exec, [[LIVE]] 397; CHECK: buffer_store_dword {{v[0-9]+}}, {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, 0 idxen 398; CHECK: s_wqm_b64 exec, exec 399; CHECK: buffer_load_dword {{v[0-9]+}}, {{v[0-9]+}}, {{s\[[0-9]+:[0-9]+\]}}, {{s[0-9]+}} offen 400 401; CHECK: s_and_b64 exec, exec, [[LIVE]] 402; CHECK: image_sample 403; CHECK: buffer_store_dwordx4 404define amdgpu_ps void @test_alloca(float %data, i32 %a, i32 %idx) nounwind { 405entry: 406 %array = alloca [32 x i32], align 4 407 408 call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 0, i32 0, i1 0, i1 0) 409 410 %s.gep = getelementptr [32 x i32], [32 x i32]* %array, i32 0, i32 0 411 store volatile i32 %a, i32* %s.gep, align 4 412 413 call void @llvm.amdgcn.buffer.store.f32(float %data, <4 x i32> undef, i32 1, i32 0, i1 0, i1 0) 414 415 %c.gep = getelementptr [32 x i32], [32 x i32]* %array, i32 0, i32 %idx 416 %c = load i32, i32* %c.gep, align 4 417 418 %t = call <4 x float> @llvm.SI.image.sample.i32(i32 %c, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 419 420 call void @llvm.amdgcn.buffer.store.v4f32(<4 x float> %t, <4 x i32> undef, i32 0, i32 0, i1 0, i1 0) 421 422 ret void 423} 424 425; Must return to exact at the end of a non-void returning shader, 426; otherwise the EXEC mask exported by the epilog will be wrong. This is true 427; even if the shader has no kills, because a kill could have happened in a 428; previous shader fragment. 429; 430; CHECK-LABEL: {{^}}test_nonvoid_return: 431; CHECK: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec 432; CHECK: s_wqm_b64 exec, exec 433; 434; CHECK: s_and_b64 exec, exec, [[LIVE]] 435; CHECK-NOT: exec 436define amdgpu_ps <4 x float> @test_nonvoid_return() nounwind { 437 %tex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> undef, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 438 %tex.i = bitcast <4 x float> %tex to <4 x i32> 439 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.i, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 440 ret <4 x float> %dtex 441} 442 443; CHECK-LABEL: {{^}}test_nonvoid_return_unreachable: 444; CHECK: s_mov_b64 [[LIVE:s\[[0-9]+:[0-9]+\]]], exec 445; CHECK: s_wqm_b64 exec, exec 446; 447; CHECK: s_and_b64 exec, exec, [[LIVE]] 448; CHECK-NOT: exec 449define amdgpu_ps <4 x float> @test_nonvoid_return_unreachable(i32 inreg %c) nounwind { 450entry: 451 %tex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> undef, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 452 %tex.i = bitcast <4 x float> %tex to <4 x i32> 453 %dtex = call <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32> %tex.i, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 454 455 %cc = icmp sgt i32 %c, 0 456 br i1 %cc, label %if, label %else 457 458if: 459 store volatile <4 x float> %dtex, <4 x float> addrspace(1)* undef 460 unreachable 461 462else: 463 ret <4 x float> %dtex 464} 465 466; Test awareness that s_wqm_b64 clobbers SCC. 467; 468; CHECK-LABEL: {{^}}test_scc: 469; CHECK: s_mov_b64 [[ORIG:s\[[0-9]+:[0-9]+\]]], exec 470; CHECK: s_wqm_b64 exec, exec 471; CHECK: s_cmp_ 472; CHECK-NEXT: s_cbranch_scc 473; CHECK: ; %if 474; CHECK: s_and_b64 exec, exec, [[ORIG]] 475; CHECK: image_sample 476; CHECK: ; %else 477; CHECK: s_and_b64 exec, exec, [[ORIG]] 478; CHECK: image_sample 479; CHECK: ; %end 480define amdgpu_ps <4 x float> @test_scc(i32 inreg %sel, i32 %idx) #1 { 481main_body: 482 %cc = icmp sgt i32 %sel, 0 483 br i1 %cc, label %if, label %else 484 485if: 486 %r.if = call <4 x float> @llvm.SI.image.sample.i32(i32 0, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 487 br label %end 488 489else: 490 %r.else = call <4 x float> @llvm.SI.image.sample.v2i32(<2 x i32> <i32 0, i32 1>, <8 x i32> undef, <4 x i32> undef, i32 15, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0, i32 0) 491 br label %end 492 493end: 494 %r = phi <4 x float> [ %r.if, %if ], [ %r.else, %else ] 495 496 call void @llvm.amdgcn.buffer.store.f32(float 1.0, <4 x i32> undef, i32 %idx, i32 0, i1 0, i1 0) 497 498 ret <4 x float> %r 499} 500 501declare void @llvm.amdgcn.exp.f32(i32, i32, float, float, float, float, i1, i1) #1 502declare void @llvm.amdgcn.image.store.v4f32.v4i32.v8i32(<4 x float>, <4 x i32>, <8 x i32>, i32, i1, i1, i1, i1) #1 503declare void @llvm.amdgcn.buffer.store.f32(float, <4 x i32>, i32, i32, i1, i1) #1 504declare void @llvm.amdgcn.buffer.store.v4f32(<4 x float>, <4 x i32>, i32, i32, i1, i1) #1 505 506declare <4 x float> @llvm.amdgcn.image.load.v4f32.v4i32.v8i32(<4 x i32>, <8 x i32>, i32, i1, i1, i1, i1) #2 507declare float @llvm.amdgcn.buffer.load.f32(<4 x i32>, i32, i32, i1, i1) #2 508 509declare <4 x float> @llvm.SI.image.sample.i32(i32, <8 x i32>, <4 x i32>, i32, i32, i32, i32, i32, i32, i32, i32) #3 510declare <4 x float> @llvm.SI.image.sample.v2i32(<2 x i32>, <8 x i32>, <4 x i32>, i32, i32, i32, i32, i32, i32, i32, i32) #3 511declare <4 x float> @llvm.SI.image.sample.v4i32(<4 x i32>, <8 x i32>, <4 x i32>, i32, i32, i32, i32, i32, i32, i32, i32) #3 512 513declare void @llvm.AMDGPU.kill(float) #1 514 515attributes #1 = { nounwind } 516attributes #2 = { nounwind readonly } 517attributes #3 = { nounwind readnone } 518attributes #4 = { "amdgpu-ps-wqm-outputs" } 519