1; RUN: llc < %s -march=amdgcn -mcpu=verde -verify-machineinstrs | FileCheck --check-prefix=SI --check-prefix=FUNC %s 2; RUN: llc < %s -march=amdgcn -mcpu=tonga -mattr=-flat-for-global -verify-machineinstrs | FileCheck --check-prefix=SI --check-prefix=FUNC %s 3 4; FUNC-LABEL: {{^}}break_inserted_outside_of_loop: 5 6; SI: [[LOOP_LABEL:[A-Z0-9]+]]: 7; Lowered break instructin: 8; SI: s_or_b64 9; Lowered Loop instruction: 10; SI: s_andn2_b64 11; s_cbranch_execnz [[LOOP_LABEL]] 12; SI: s_endpgm 13define amdgpu_kernel void @break_inserted_outside_of_loop(i32 addrspace(1)* %out, i32 %a) { 14main_body: 15 %tid = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0) #0 16 %0 = and i32 %a, %tid 17 %1 = trunc i32 %0 to i1 18 br label %ENDIF 19 20ENDLOOP: 21 store i32 0, i32 addrspace(1)* %out 22 ret void 23 24ENDIF: 25 br i1 %1, label %ENDLOOP, label %ENDIF 26} 27 28 29; FUNC-LABEL: {{^}}phi_cond_outside_loop: 30 31; SI: s_mov_b64 [[LEFT:s\[[0-9]+:[0-9]+\]]], 0 32; SI: s_mov_b64 [[PHI:s\[[0-9]+:[0-9]+\]]], 0 33 34; SI: ; %else 35; SI: v_cmp_eq_u32_e64 [[TMP:s\[[0-9]+:[0-9]+\]]], 36 37; SI: ; %endif 38 39; SI: [[LOOP_LABEL:BB[0-9]+_[0-9]+]]: ; %loop 40; SI: s_and_b64 [[TMP1:s\[[0-9]+:[0-9]+\]]], exec, [[PHI]] 41; SI: s_or_b64 [[LEFT]], [[TMP1]], [[LEFT]] 42; SI: s_andn2_b64 exec, exec, [[LEFT]] 43; SI: s_cbranch_execnz [[LOOP_LABEL]] 44; SI: s_endpgm 45 46define amdgpu_kernel void @phi_cond_outside_loop(i32 %b) { 47entry: 48 %tid = call i32 @llvm.amdgcn.mbcnt.lo(i32 -1, i32 0) #0 49 %0 = icmp eq i32 %tid , 0 50 br i1 %0, label %if, label %else 51 52if: 53 br label %endif 54 55else: 56 %1 = icmp eq i32 %b, 0 57 br label %endif 58 59endif: 60 %2 = phi i1 [0, %if], [%1, %else] 61 br label %loop 62 63loop: 64 br i1 %2, label %exit, label %loop 65 66exit: 67 ret void 68} 69 70; FIXME: should emit s_endpgm 71; CHECK-LABEL: {{^}}switch_unreachable: 72; CHECK-NOT: s_endpgm 73; CHECK: .Lfunc_end2 74define amdgpu_kernel void @switch_unreachable(i32 addrspace(1)* %g, i8 addrspace(3)* %l, i32 %x) nounwind { 75centry: 76 switch i32 %x, label %sw.default [ 77 i32 0, label %sw.bb 78 i32 60, label %sw.bb 79 ] 80 81sw.bb: 82 unreachable 83 84sw.default: 85 unreachable 86 87sw.epilog: 88 ret void 89} 90 91declare float @llvm.fabs.f32(float) nounwind readnone 92 93; This broke the old AMDIL cfg structurizer 94; FUNC-LABEL: {{^}}loop_land_info_assert: 95; SI: v_cmp_lt_i32_e64 [[CMP4:s\[[0-9:]+\]]], s{{[0-9]+}}, 4{{$}} 96; SI: s_and_b64 [[CMP4M:s\[[0-9]+:[0-9]+\]]], exec, [[CMP4]] 97 98; SI: [[WHILELOOP:BB[0-9]+_[0-9]+]]: ; %while.cond 99; SI: s_cbranch_vccz [[FOR_COND_PH:BB[0-9]+_[0-9]+]] 100 101; SI: [[CONVEX_EXIT:BB[0-9_]+]] 102; SI: s_mov_b64 vcc, 103; SI-NEXT: s_cbranch_vccnz [[ENDPGM:BB[0-9]+_[0-9]+]] 104 105; SI: s_cbranch_vccnz [[WHILELOOP]] 106 107; SI: ; %if.else 108; SI: buffer_store_dword 109 110; SI: [[FOR_COND_PH]]: ; %for.cond.preheader 111; SI: s_cbranch_vccz [[ENDPGM]] 112 113; SI: [[ENDPGM]]: 114; SI-NEXT: s_endpgm 115define amdgpu_kernel void @loop_land_info_assert(i32 %c0, i32 %c1, i32 %c2, i32 %c3, i32 %x, i32 %y, i1 %arg) nounwind { 116entry: 117 %cmp = icmp sgt i32 %c0, 0 118 br label %while.cond.outer 119 120while.cond.outer: 121 %tmp = load float, float addrspace(1)* undef 122 br label %while.cond 123 124while.cond: 125 %cmp1 = icmp slt i32 %c1, 4 126 br i1 %cmp1, label %convex.exit, label %for.cond 127 128convex.exit: 129 %or = or i1 %cmp, %cmp1 130 br i1 %or, label %return, label %if.end 131 132if.end: 133 %tmp3 = call float @llvm.fabs.f32(float %tmp) nounwind readnone 134 %cmp2 = fcmp olt float %tmp3, 0x3E80000000000000 135 br i1 %cmp2, label %if.else, label %while.cond.outer 136 137if.else: 138 store volatile i32 3, i32 addrspace(1)* undef, align 4 139 br label %while.cond 140 141for.cond: 142 %cmp3 = icmp slt i32 %c3, 1000 143 br i1 %cmp3, label %for.body, label %return 144 145for.body: 146 br i1 %cmp3, label %self.loop, label %if.end.2 147 148if.end.2: 149 %or.cond2 = or i1 %cmp3, %arg 150 br i1 %or.cond2, label %return, label %for.cond 151 152self.loop: 153 br label %self.loop 154 155return: 156 ret void 157} 158 159declare i32 @llvm.amdgcn.mbcnt.lo(i32, i32) #0 160 161attributes #0 = { nounwind readnone } 162