1; RUN: llc -mtriple=amdgcn-- -mattr=-promote-alloca -amdgpu-sroa=0 -verify-machineinstrs < %s | FileCheck -check-prefix=GCN %s 2 3; GCN-LABEL: {{^}}store_fi_lifetime: 4; GCN: v_mov_b32_e32 [[FI:v[0-9]+]], 0{{$}} 5; GCN: buffer_store_dword [[FI]] 6define void @store_fi_lifetime(i32 addrspace(1)* %out, i32 %in) #0 { 7entry: 8 %b = alloca i8 9 call void @llvm.lifetime.start(i64 1, i8* %b) 10 store volatile i8* %b, i8* addrspace(1)* undef 11 call void @llvm.lifetime.end(i64 1, i8* %b) 12 ret void 13} 14 15; GCN-LABEL: {{^}}stored_fi_to_lds: 16; GCN: s_load_dword [[LDSPTR:s[0-9]+]] 17; GCN: v_mov_b32_e32 [[ZERO1:v[0-9]+]], 0{{$}} 18; GCN: buffer_store_dword v{{[0-9]+}}, [[ZERO1]] 19; GCN: v_mov_b32_e32 [[ZERO0:v[0-9]+]], 0{{$}} 20; GCN: v_mov_b32_e32 [[VLDSPTR:v[0-9]+]], [[LDSPTR]] 21; GCN: ds_write_b32 [[VLDSPTR]], [[ZERO0]] 22define void @stored_fi_to_lds(float* addrspace(3)* %ptr) #0 { 23 %tmp = alloca float 24 store float 4.0, float *%tmp 25 store float* %tmp, float* addrspace(3)* %ptr 26 ret void 27} 28 29; Offset is applied 30; GCN-LABEL: {{^}}stored_fi_to_lds_2_small_objects: 31; GCN-DAG: v_mov_b32_e32 [[ZERO:v[0-9]+]], 0{{$}} 32; GCN-DAG: buffer_store_dword v{{[0-9]+}}, [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen{{$}} 33; GCN-DAG: buffer_store_dword v{{[0-9]+}}, [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen offset:4{{$}} 34 35; GCN-DAG: s_load_dword [[LDSPTR:s[0-9]+]] 36 37; GCN-DAG: v_mov_b32_e32 [[VLDSPTR:v[0-9]+]], [[LDSPTR]] 38; GCN: ds_write_b32 [[VLDSPTR]], [[ZERO]] 39 40; GCN-DAG: v_mov_b32_e32 [[FI1:v[0-9]+]], 4{{$}} 41; GCN: ds_write_b32 [[VLDSPTR]], [[FI1]] 42define void @stored_fi_to_lds_2_small_objects(float* addrspace(3)* %ptr) #0 { 43 %tmp0 = alloca float 44 %tmp1 = alloca float 45 store float 4.0, float* %tmp0 46 store float 4.0, float* %tmp1 47 store volatile float* %tmp0, float* addrspace(3)* %ptr 48 store volatile float* %tmp1, float* addrspace(3)* %ptr 49 ret void 50} 51 52; Same frame index is used multiple times in the store 53; GCN-LABEL: {{^}}stored_fi_to_self: 54; GCN-DAG: v_mov_b32_e32 [[K:v[0-9]+]], 0x4d2{{$}} 55; GCN-DAG: v_mov_b32_e32 [[ZERO:v[0-9]+]], 0{{$}} 56; GCN: buffer_store_dword [[K]], [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen{{$}} 57; GCN: buffer_store_dword [[ZERO]], [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen{{$}} 58define void @stored_fi_to_self() #0 { 59 %tmp = alloca i32* 60 61 ; Avoid optimizing everything out 62 store volatile i32* inttoptr (i32 1234 to i32*), i32** %tmp 63 %bitcast = bitcast i32** %tmp to i32* 64 store volatile i32* %bitcast, i32** %tmp 65 ret void 66} 67 68; GCN-LABEL: {{^}}stored_fi_to_self_offset: 69; GCN-DAG: v_mov_b32_e32 [[ZERO:v[0-9]+]], 0{{$}} 70; GCN-DAG: v_mov_b32_e32 [[K0:v[0-9]+]], 32{{$}} 71; GCN: buffer_store_dword [[K0]], [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen{{$}} 72 73; GCN-DAG: v_mov_b32_e32 [[K1:v[0-9]+]], 0x4d2{{$}} 74; GCN: buffer_store_dword [[K1]], [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen offset:2048{{$}} 75 76; GCN: v_mov_b32_e32 [[OFFSETK:v[0-9]+]], 0x800{{$}} 77; GCN: buffer_store_dword [[OFFSETK]], [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen offset:2048{{$}} 78define void @stored_fi_to_self_offset() #0 { 79 %tmp0 = alloca [512 x i32] 80 %tmp1 = alloca i32* 81 82 ; Avoid optimizing everything out 83 %tmp0.cast = bitcast [512 x i32]* %tmp0 to i32* 84 store volatile i32 32, i32* %tmp0.cast 85 86 store volatile i32* inttoptr (i32 1234 to i32*), i32** %tmp1 87 88 %bitcast = bitcast i32** %tmp1 to i32* 89 store volatile i32* %bitcast, i32** %tmp1 90 ret void 91} 92 93; GCN-LABEL: {{^}}stored_fi_to_fi: 94; GCN: v_mov_b32_e32 [[ZERO:v[0-9]+]], 0{{$}} 95; GCN: buffer_store_dword v{{[0-9]+}}, [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen{{$}} 96; GCN: buffer_store_dword v{{[0-9]+}}, [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen offset:4{{$}} 97; GCN: buffer_store_dword v{{[0-9]+}}, [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen offset:8{{$}} 98 99; GCN: v_mov_b32_e32 [[FI1:v[0-9]+]], 4{{$}} 100; GCN: buffer_store_dword [[FI1]], [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen offset:8{{$}} 101 102; GCN: v_mov_b32_e32 [[FI2:v[0-9]+]], 8{{$}} 103; GCN: buffer_store_dword [[FI2]], [[ZERO]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen offset:4{{$}} 104define void @stored_fi_to_fi() #0 { 105 %tmp0 = alloca i32* 106 %tmp1 = alloca i32* 107 %tmp2 = alloca i32* 108 store volatile i32* inttoptr (i32 1234 to i32*), i32** %tmp0 109 store volatile i32* inttoptr (i32 5678 to i32*), i32** %tmp1 110 store volatile i32* inttoptr (i32 9999 to i32*), i32** %tmp2 111 112 %bitcast1 = bitcast i32** %tmp1 to i32* 113 %bitcast2 = bitcast i32** %tmp2 to i32* ; at offset 8 114 115 store volatile i32* %bitcast1, i32** %tmp2 ; store offset 4 at offset 8 116 store volatile i32* %bitcast2, i32** %tmp1 ; store offset 8 at offset 4 117 ret void 118} 119 120; GCN-LABEL: {{^}}stored_fi_to_global: 121; GCN: buffer_store_dword v{{[0-9]+}}, v{{[0-9]+}}, s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen 122; GCN: v_mov_b32_e32 [[FI:v[0-9]+]], 0{{$}} 123; GCN: buffer_store_dword [[FI]] 124define void @stored_fi_to_global(float* addrspace(1)* %ptr) #0 { 125 %tmp = alloca float 126 store float 0.0, float *%tmp 127 store float* %tmp, float* addrspace(1)* %ptr 128 ret void 129} 130 131; Offset is applied 132; GCN-LABEL: {{^}}stored_fi_to_global_2_small_objects: 133; GCN: buffer_store_dword v{{[0-9]+}}, v{{[0-9]+}}, s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen 134; GCN: buffer_store_dword v{{[0-9]+}}, v{{[0-9]+}}, s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen 135; GCN: buffer_store_dword v{{[0-9]+}}, v{{[0-9]+}}, s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen 136 137; GCN: v_mov_b32_e32 [[FI1:v[0-9]+]], 4{{$}} 138; GCN: buffer_store_dword [[FI1]], off, s{{\[[0-9]+:[0-9]+\]}}, 0{{$}} 139 140; GCN-DAG: v_mov_b32_e32 [[FI2:v[0-9]+]], 8{{$}} 141; GCN: buffer_store_dword [[FI2]], off, s{{\[[0-9]+:[0-9]+\]}}, 0{{$}} 142define void @stored_fi_to_global_2_small_objects(float* addrspace(1)* %ptr) #0 { 143 %tmp0 = alloca float 144 %tmp1 = alloca float 145 %tmp2 = alloca float 146 store volatile float 0.0, float *%tmp0 147 store volatile float 0.0, float *%tmp1 148 store volatile float 0.0, float *%tmp2 149 store volatile float* %tmp1, float* addrspace(1)* %ptr 150 store volatile float* %tmp2, float* addrspace(1)* %ptr 151 ret void 152} 153 154; GCN-LABEL: {{^}}stored_fi_to_global_huge_frame_offset: 155; GCN: v_mov_b32_e32 [[VAL_0:v[0-9]+]], 0{{$}} 156; GCN: v_mov_b32_e32 [[BASE_0:v[0-9]+]], 0{{$}} 157; GCN: buffer_store_dword [[VAL_0]], [[BASE_0]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen 158 159; GCN: v_mov_b32_e32 [[BASE_0_1:v[0-9]+]], 0{{$}} 160; GCN: v_add_i32_e32 [[BASE_1_OFF_0:v[0-9]+]], vcc, 0x3ffc, [[BASE_0_1]] 161 162; GCN: v_mov_b32_e32 [[K:v[0-9]+]], 0x3e7{{$}} 163; GCN: v_add_i32_e32 [[BASE_1_OFF_1:v[0-9]+]], vcc, 56, [[BASE_0_1]] 164; GCN: buffer_store_dword [[K]], [[BASE_1_OFF_0]], s{{\[[0-9]+:[0-9]+\]}}, s{{[0-9]+}} offen{{$}} 165 166; GCN: buffer_store_dword [[BASE_1_OFF_1]], off, s{{\[[0-9]+:[0-9]+\]}}, 0{{$}} 167define void @stored_fi_to_global_huge_frame_offset(i32* addrspace(1)* %ptr) #0 { 168 %tmp0 = alloca [4096 x i32] 169 %tmp1 = alloca [4096 x i32] 170 %gep0.tmp0 = getelementptr [4096 x i32], [4096 x i32]* %tmp0, i32 0, i32 0 171 store volatile i32 0, i32* %gep0.tmp0 172 %gep1.tmp0 = getelementptr [4096 x i32], [4096 x i32]* %tmp0, i32 0, i32 4095 173 store volatile i32 999, i32* %gep1.tmp0 174 %gep0.tmp1 = getelementptr [4096 x i32], [4096 x i32]* %tmp0, i32 0, i32 14 175 store i32* %gep0.tmp1, i32* addrspace(1)* %ptr 176 ret void 177} 178 179@g1 = external addrspace(1) global i32* 180 181; This was leaving a dead node around resulting in failing to select 182; on the leftover AssertZext's ValueType operand. 183 184; GCN-LABEL: {{^}}cannot_select_assertzext_valuetype: 185; GCN: s_getpc_b64 s{{\[}}[[PC_LO:[0-9]+]]:[[PC_HI:[0-9]+]]{{\]}} 186; GCN: s_add_u32 s{{[0-9]+}}, s[[PC_LO]], g1@gotpcrel32@lo+4 187; GCN: s_addc_u32 s{{[0-9]+}}, s[[PC_HI]], g1@gotpcrel32@hi+4 188; GCN: v_mov_b32_e32 [[FI:v[0-9]+]], 0{{$}} 189; GCN: buffer_store_dword [[FI]] 190define void @cannot_select_assertzext_valuetype(i32 addrspace(1)* %out, i32 %idx) #0 { 191entry: 192 %b = alloca i32, align 4 193 %tmp1 = load volatile i32*, i32* addrspace(1)* @g1, align 4 194 %arrayidx = getelementptr inbounds i32, i32* %tmp1, i32 %idx 195 %tmp2 = load i32, i32* %arrayidx, align 4 196 store volatile i32* %b, i32* addrspace(1)* undef 197 ret void 198} 199 200declare void @llvm.lifetime.start(i64, i8* nocapture) #1 201declare void @llvm.lifetime.end(i64, i8* nocapture) #1 202 203attributes #0 = { nounwind } 204attributes #1 = { argmemonly nounwind } 205