1; NOTE: Assertions have been autogenerated by utils/update_test_checks.py 2; RUN: opt %s -S -riscv-gather-scatter-lowering -mtriple=riscv64 -mattr=+m,+v -riscv-v-vector-bits-min=256 | FileCheck %s --check-prefixes=CHECK,V 3; RUN: opt %s -S -riscv-gather-scatter-lowering -mtriple=riscv64 -mattr=+m,+f,+zve32f -riscv-v-vector-bits-min=256 | FileCheck %s --check-prefixes=CHECK,ZVE32F 4 5%struct.foo = type { i32, i32, i32, i32 } 6 7; void gather(signed char * __restrict A, signed char * __restrict B) { 8; for (int i = 0; i != 1024; ++i) 9; A[i] += B[i * 5]; 10; } 11define void @gather(i8* noalias nocapture %A, i8* noalias nocapture readonly %B) { 12; CHECK-LABEL: @gather( 13; CHECK-NEXT: entry: 14; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 15; CHECK: vector.body: 16; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 17; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 18; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i8, i8* [[B:%.*]], i64 [[VEC_IND_SCALAR]] 19; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <32 x i8> @llvm.riscv.masked.strided.load.v32i8.p0i8.i64(<32 x i8> undef, i8* [[TMP0]], i64 5, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 20; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, i8* [[A:%.*]], i64 [[INDEX]] 21; CHECK-NEXT: [[TMP2:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 22; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, <32 x i8>* [[TMP2]], align 1 23; CHECK-NEXT: [[TMP3:%.*]] = add <32 x i8> [[WIDE_LOAD]], [[WIDE_MASKED_GATHER]] 24; CHECK-NEXT: [[TMP4:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 25; CHECK-NEXT: store <32 x i8> [[TMP3]], <32 x i8>* [[TMP4]], align 1 26; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32 27; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 160 28; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 29; CHECK-NEXT: br i1 [[TMP5]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 30; CHECK: for.cond.cleanup: 31; CHECK-NEXT: ret void 32; 33entry: 34 br label %vector.body 35 36vector.body: ; preds = %vector.body, %entry 37 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 38 %vec.ind = phi <32 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15, i64 16, i64 17, i64 18, i64 19, i64 20, i64 21, i64 22, i64 23, i64 24, i64 25, i64 26, i64 27, i64 28, i64 29, i64 30, i64 31>, %entry ], [ %vec.ind.next, %vector.body ] 39 %0 = mul nuw nsw <32 x i64> %vec.ind, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 40 %1 = getelementptr inbounds i8, i8* %B, <32 x i64> %0 41 %wide.masked.gather = call <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*> %1, i32 1, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <32 x i8> undef) 42 %2 = getelementptr inbounds i8, i8* %A, i64 %index 43 %3 = bitcast i8* %2 to <32 x i8>* 44 %wide.load = load <32 x i8>, <32 x i8>* %3, align 1 45 %4 = add <32 x i8> %wide.load, %wide.masked.gather 46 %5 = bitcast i8* %2 to <32 x i8>* 47 store <32 x i8> %4, <32 x i8>* %5, align 1 48 %index.next = add nuw i64 %index, 32 49 %vec.ind.next = add <32 x i64> %vec.ind, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 50 %6 = icmp eq i64 %index.next, 1024 51 br i1 %6, label %for.cond.cleanup, label %vector.body 52 53for.cond.cleanup: ; preds = %vector.body 54 ret void 55} 56 57define void @gather_masked(i8* noalias nocapture %A, i8* noalias nocapture readonly %B, <32 x i8> %maskedoff) { 58; CHECK-LABEL: @gather_masked( 59; CHECK-NEXT: entry: 60; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 61; CHECK: vector.body: 62; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 63; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 64; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i8, i8* [[B:%.*]], i64 [[VEC_IND_SCALAR]] 65; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <32 x i8> @llvm.riscv.masked.strided.load.v32i8.p0i8.i64(<32 x i8> [[MASKEDOFF:%.*]], i8* [[TMP0]], i64 5, <32 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true>) 66; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, i8* [[A:%.*]], i64 [[INDEX]] 67; CHECK-NEXT: [[TMP2:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 68; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, <32 x i8>* [[TMP2]], align 1 69; CHECK-NEXT: [[TMP3:%.*]] = add <32 x i8> [[WIDE_LOAD]], [[WIDE_MASKED_GATHER]] 70; CHECK-NEXT: [[TMP4:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 71; CHECK-NEXT: store <32 x i8> [[TMP3]], <32 x i8>* [[TMP4]], align 1 72; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32 73; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 160 74; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 75; CHECK-NEXT: br i1 [[TMP5]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 76; CHECK: for.cond.cleanup: 77; CHECK-NEXT: ret void 78; 79entry: 80 br label %vector.body 81 82vector.body: ; preds = %vector.body, %entry 83 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 84 %vec.ind = phi <32 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15, i64 16, i64 17, i64 18, i64 19, i64 20, i64 21, i64 22, i64 23, i64 24, i64 25, i64 26, i64 27, i64 28, i64 29, i64 30, i64 31>, %entry ], [ %vec.ind.next, %vector.body ] 85 %0 = mul nuw nsw <32 x i64> %vec.ind, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 86 %1 = getelementptr inbounds i8, i8* %B, <32 x i64> %0 87 %wide.masked.gather = call <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*> %1, i32 1, <32 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true>, <32 x i8> %maskedoff) 88 %2 = getelementptr inbounds i8, i8* %A, i64 %index 89 %3 = bitcast i8* %2 to <32 x i8>* 90 %wide.load = load <32 x i8>, <32 x i8>* %3, align 1 91 %4 = add <32 x i8> %wide.load, %wide.masked.gather 92 %5 = bitcast i8* %2 to <32 x i8>* 93 store <32 x i8> %4, <32 x i8>* %5, align 1 94 %index.next = add nuw i64 %index, 32 95 %vec.ind.next = add <32 x i64> %vec.ind, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 96 %6 = icmp eq i64 %index.next, 1024 97 br i1 %6, label %for.cond.cleanup, label %vector.body 98 99for.cond.cleanup: ; preds = %vector.body 100 ret void 101} 102 103define void @gather_negative_stride(i8* noalias nocapture %A, i8* noalias nocapture readonly %B) { 104; 105; CHECK-LABEL: @gather_negative_stride( 106; CHECK-NEXT: entry: 107; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 108; CHECK: vector.body: 109; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 110; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 155, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 111; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i8, i8* [[B:%.*]], i64 [[VEC_IND_SCALAR]] 112; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <32 x i8> @llvm.riscv.masked.strided.load.v32i8.p0i8.i64(<32 x i8> undef, i8* [[TMP0]], i64 -5, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 113; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, i8* [[A:%.*]], i64 [[INDEX]] 114; CHECK-NEXT: [[TMP2:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 115; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, <32 x i8>* [[TMP2]], align 1 116; CHECK-NEXT: [[TMP3:%.*]] = add <32 x i8> [[WIDE_LOAD]], [[WIDE_MASKED_GATHER]] 117; CHECK-NEXT: [[TMP4:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 118; CHECK-NEXT: store <32 x i8> [[TMP3]], <32 x i8>* [[TMP4]], align 1 119; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32 120; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 160 121; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 122; CHECK-NEXT: br i1 [[TMP5]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 123; CHECK: for.cond.cleanup: 124; CHECK-NEXT: ret void 125; 126entry: 127 br label %vector.body 128 129vector.body: ; preds = %vector.body, %entry 130 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 131 %vec.ind = phi <32 x i64> [ <i64 31, i64 30, i64 29, i64 28, i64 27, i64 26, i64 25, i64 24, i64 23, i64 22, i64 21, i64 20, i64 19, i64 18, i64 17, i64 16, i64 15, i64 14, i64 13, i64 12, i64 11, i64 10, i64 9, i64 8, i64 7, i64 6, i64 5, i64 4, i64 3, i64 2, i64 1, i64 0>, %entry ], [ %vec.ind.next, %vector.body ] 132 %0 = mul nuw nsw <32 x i64> %vec.ind, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 133 %1 = getelementptr inbounds i8, i8* %B, <32 x i64> %0 134 %wide.masked.gather = call <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*> %1, i32 1, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <32 x i8> undef) 135 %2 = getelementptr inbounds i8, i8* %A, i64 %index 136 %3 = bitcast i8* %2 to <32 x i8>* 137 %wide.load = load <32 x i8>, <32 x i8>* %3, align 1 138 %4 = add <32 x i8> %wide.load, %wide.masked.gather 139 %5 = bitcast i8* %2 to <32 x i8>* 140 store <32 x i8> %4, <32 x i8>* %5, align 1 141 %index.next = add nuw i64 %index, 32 142 %vec.ind.next = add <32 x i64> %vec.ind, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 143 %6 = icmp eq i64 %index.next, 1024 144 br i1 %6, label %for.cond.cleanup, label %vector.body 145 146for.cond.cleanup: ; preds = %vector.body 147 ret void 148} 149 150define void @gather_zero_stride(i8* noalias nocapture %A, i8* noalias nocapture readonly %B) { 151; 152; CHECK-LABEL: @gather_zero_stride( 153; CHECK-NEXT: entry: 154; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 155; CHECK: vector.body: 156; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 157; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 158; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i8, i8* [[B:%.*]], i64 [[VEC_IND_SCALAR]] 159; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <32 x i8> @llvm.riscv.masked.strided.load.v32i8.p0i8.i64(<32 x i8> undef, i8* [[TMP0]], i64 0, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 160; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, i8* [[A:%.*]], i64 [[INDEX]] 161; CHECK-NEXT: [[TMP2:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 162; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, <32 x i8>* [[TMP2]], align 1 163; CHECK-NEXT: [[TMP3:%.*]] = add <32 x i8> [[WIDE_LOAD]], [[WIDE_MASKED_GATHER]] 164; CHECK-NEXT: [[TMP4:%.*]] = bitcast i8* [[TMP1]] to <32 x i8>* 165; CHECK-NEXT: store <32 x i8> [[TMP3]], <32 x i8>* [[TMP4]], align 1 166; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32 167; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 160 168; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 169; CHECK-NEXT: br i1 [[TMP5]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 170; CHECK: for.cond.cleanup: 171; CHECK-NEXT: ret void 172; 173entry: 174 br label %vector.body 175 176vector.body: ; preds = %vector.body, %entry 177 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 178 %vec.ind = phi <32 x i64> [ zeroinitializer, %entry ], [ %vec.ind.next, %vector.body ] 179 %0 = mul nuw nsw <32 x i64> %vec.ind, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 180 %1 = getelementptr inbounds i8, i8* %B, <32 x i64> %0 181 %wide.masked.gather = call <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*> %1, i32 1, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <32 x i8> undef) 182 %2 = getelementptr inbounds i8, i8* %A, i64 %index 183 %3 = bitcast i8* %2 to <32 x i8>* 184 %wide.load = load <32 x i8>, <32 x i8>* %3, align 1 185 %4 = add <32 x i8> %wide.load, %wide.masked.gather 186 %5 = bitcast i8* %2 to <32 x i8>* 187 store <32 x i8> %4, <32 x i8>* %5, align 1 188 %index.next = add nuw i64 %index, 32 189 %vec.ind.next = add <32 x i64> %vec.ind, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 190 %6 = icmp eq i64 %index.next, 1024 191 br i1 %6, label %for.cond.cleanup, label %vector.body 192 193for.cond.cleanup: ; preds = %vector.body 194 ret void 195} 196 197;void scatter(signed char * __restrict A, signed char * __restrict B) { 198; for (int i = 0; i < 1024; ++i) 199; A[i * 5] += B[i]; 200;} 201define void @scatter(i8* noalias nocapture %A, i8* noalias nocapture readonly %B) { 202; 203; CHECK-LABEL: @scatter( 204; CHECK-NEXT: entry: 205; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 206; CHECK: vector.body: 207; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 208; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 209; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, i8* [[B:%.*]], i64 [[INDEX]] 210; CHECK-NEXT: [[TMP1:%.*]] = bitcast i8* [[TMP0]] to <32 x i8>* 211; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, <32 x i8>* [[TMP1]], align 1 212; CHECK-NEXT: [[TMP2:%.*]] = getelementptr i8, i8* [[A:%.*]], i64 [[VEC_IND_SCALAR]] 213; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <32 x i8> @llvm.riscv.masked.strided.load.v32i8.p0i8.i64(<32 x i8> undef, i8* [[TMP2]], i64 5, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 214; CHECK-NEXT: [[TMP3:%.*]] = add <32 x i8> [[WIDE_MASKED_GATHER]], [[WIDE_LOAD]] 215; CHECK-NEXT: call void @llvm.riscv.masked.strided.store.v32i8.p0i8.i64(<32 x i8> [[TMP3]], i8* [[TMP2]], i64 5, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 216; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32 217; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 160 218; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 219; CHECK-NEXT: br i1 [[TMP4]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 220; CHECK: for.cond.cleanup: 221; CHECK-NEXT: ret void 222; 223entry: 224 br label %vector.body 225 226vector.body: ; preds = %vector.body, %entry 227 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 228 %vec.ind = phi <32 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15, i64 16, i64 17, i64 18, i64 19, i64 20, i64 21, i64 22, i64 23, i64 24, i64 25, i64 26, i64 27, i64 28, i64 29, i64 30, i64 31>, %entry ], [ %vec.ind.next, %vector.body ] 229 %0 = getelementptr inbounds i8, i8* %B, i64 %index 230 %1 = bitcast i8* %0 to <32 x i8>* 231 %wide.load = load <32 x i8>, <32 x i8>* %1, align 1 232 %2 = mul nuw nsw <32 x i64> %vec.ind, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 233 %3 = getelementptr inbounds i8, i8* %A, <32 x i64> %2 234 %wide.masked.gather = call <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*> %3, i32 1, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <32 x i8> undef) 235 %4 = add <32 x i8> %wide.masked.gather, %wide.load 236 call void @llvm.masked.scatter.v32i8.v32p0i8(<32 x i8> %4, <32 x i8*> %3, i32 1, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 237 %index.next = add nuw i64 %index, 32 238 %vec.ind.next = add <32 x i64> %vec.ind, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 239 %5 = icmp eq i64 %index.next, 1024 240 br i1 %5, label %for.cond.cleanup, label %vector.body 241 242for.cond.cleanup: ; preds = %vector.body 243 ret void 244} 245 246define void @scatter_masked(i8* noalias nocapture %A, i8* noalias nocapture readonly %B, <32 x i8> %maskedoff) { 247; 248; CHECK-LABEL: @scatter_masked( 249; CHECK-NEXT: entry: 250; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 251; CHECK: vector.body: 252; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 253; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 254; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, i8* [[B:%.*]], i64 [[INDEX]] 255; CHECK-NEXT: [[TMP1:%.*]] = bitcast i8* [[TMP0]] to <32 x i8>* 256; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <32 x i8>, <32 x i8>* [[TMP1]], align 1 257; CHECK-NEXT: [[TMP2:%.*]] = getelementptr i8, i8* [[A:%.*]], i64 [[VEC_IND_SCALAR]] 258; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <32 x i8> @llvm.riscv.masked.strided.load.v32i8.p0i8.i64(<32 x i8> [[MASKEDOFF:%.*]], i8* [[TMP2]], i64 5, <32 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true>) 259; CHECK-NEXT: [[TMP3:%.*]] = add <32 x i8> [[WIDE_MASKED_GATHER]], [[WIDE_LOAD]] 260; CHECK-NEXT: call void @llvm.riscv.masked.strided.store.v32i8.p0i8.i64(<32 x i8> [[TMP3]], i8* [[TMP2]], i64 5, <32 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true>) 261; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32 262; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 160 263; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 264; CHECK-NEXT: br i1 [[TMP4]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 265; CHECK: for.cond.cleanup: 266; CHECK-NEXT: ret void 267; 268entry: 269 br label %vector.body 270 271vector.body: ; preds = %vector.body, %entry 272 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 273 %vec.ind = phi <32 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15, i64 16, i64 17, i64 18, i64 19, i64 20, i64 21, i64 22, i64 23, i64 24, i64 25, i64 26, i64 27, i64 28, i64 29, i64 30, i64 31>, %entry ], [ %vec.ind.next, %vector.body ] 274 %0 = getelementptr inbounds i8, i8* %B, i64 %index 275 %1 = bitcast i8* %0 to <32 x i8>* 276 %wide.load = load <32 x i8>, <32 x i8>* %1, align 1 277 %2 = mul nuw nsw <32 x i64> %vec.ind, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 278 %3 = getelementptr inbounds i8, i8* %A, <32 x i64> %2 279 %wide.masked.gather = call <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*> %3, i32 1, <32 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true>, <32 x i8> %maskedoff) 280 %4 = add <32 x i8> %wide.masked.gather, %wide.load 281 call void @llvm.masked.scatter.v32i8.v32p0i8(<32 x i8> %4, <32 x i8*> %3, i32 1, <32 x i1> <i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 true, i1 false, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 false, i1 true, i1 true, i1 false, i1 true, i1 false, i1 false, i1 false, i1 false, i1 false, i1 false, i1 true, i1 true, i1 true, i1 true>) 282 %index.next = add nuw i64 %index, 32 283 %vec.ind.next = add <32 x i64> %vec.ind, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 284 %5 = icmp eq i64 %index.next, 1024 285 br i1 %5, label %for.cond.cleanup, label %vector.body 286 287for.cond.cleanup: ; preds = %vector.body 288 ret void 289} 290 291; void gather_pow2(signed char * __restrict A, signed char * __restrict B) { 292; for (int i = 0; i != 1024; ++i) 293; A[i] += B[i * 4]; 294; } 295define void @gather_pow2(i32* noalias nocapture %A, i32* noalias nocapture readonly %B) { 296; 297; CHECK-LABEL: @gather_pow2( 298; CHECK-NEXT: entry: 299; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 300; CHECK: vector.body: 301; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 302; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 303; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i32, i32* [[B:%.*]], i64 [[VEC_IND_SCALAR]] 304; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP0]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 305; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, i32* [[A:%.*]], i64 [[INDEX]] 306; CHECK-NEXT: [[TMP2:%.*]] = bitcast i32* [[TMP1]] to <8 x i32>* 307; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, <8 x i32>* [[TMP2]], align 1 308; CHECK-NEXT: [[TMP3:%.*]] = add <8 x i32> [[WIDE_LOAD]], [[WIDE_MASKED_GATHER]] 309; CHECK-NEXT: [[TMP4:%.*]] = bitcast i32* [[TMP1]] to <8 x i32>* 310; CHECK-NEXT: store <8 x i32> [[TMP3]], <8 x i32>* [[TMP4]], align 1 311; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8 312; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 32 313; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 314; CHECK-NEXT: br i1 [[TMP5]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 315; CHECK: for.cond.cleanup: 316; CHECK-NEXT: ret void 317; 318entry: 319 br label %vector.body 320 321vector.body: ; preds = %vector.body, %entry 322 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 323 %vec.ind = phi <8 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>, %entry ], [ %vec.ind.next, %vector.body ] 324 %0 = shl nsw <8 x i64> %vec.ind, <i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2> 325 %1 = getelementptr inbounds i32, i32* %B, <8 x i64> %0 326 %wide.masked.gather = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %1, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 327 %2 = getelementptr inbounds i32, i32* %A, i64 %index 328 %3 = bitcast i32* %2 to <8 x i32>* 329 %wide.load = load <8 x i32>, <8 x i32>* %3, align 1 330 %4 = add <8 x i32> %wide.load, %wide.masked.gather 331 %5 = bitcast i32* %2 to <8 x i32>* 332 store <8 x i32> %4, <8 x i32>* %5, align 1 333 %index.next = add nuw i64 %index, 8 334 %vec.ind.next = add <8 x i64> %vec.ind, <i64 8, i64 8, i64 8, i64 8, i64 8, i64 8, i64 8, i64 8> 335 %6 = icmp eq i64 %index.next, 1024 336 br i1 %6, label %for.cond.cleanup, label %vector.body 337 338for.cond.cleanup: ; preds = %vector.body 339 ret void 340} 341 342;void scatter_pow2(signed char * __restrict A, signed char * __restrict B) { 343; for (int i = 0; i < 1024; ++i) 344; A[i * 4] += B[i]; 345;} 346define void @scatter_pow2(i32* noalias nocapture %A, i32* noalias nocapture readonly %B) { 347; 348; CHECK-LABEL: @scatter_pow2( 349; CHECK-NEXT: entry: 350; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 351; CHECK: vector.body: 352; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 353; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 354; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, i32* [[B:%.*]], i64 [[INDEX]] 355; CHECK-NEXT: [[TMP1:%.*]] = bitcast i32* [[TMP0]] to <8 x i32>* 356; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, <8 x i32>* [[TMP1]], align 1 357; CHECK-NEXT: [[TMP2:%.*]] = getelementptr i32, i32* [[A:%.*]], i64 [[VEC_IND_SCALAR]] 358; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP2]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 359; CHECK-NEXT: [[TMP3:%.*]] = add <8 x i32> [[WIDE_MASKED_GATHER]], [[WIDE_LOAD]] 360; CHECK-NEXT: call void @llvm.riscv.masked.strided.store.v8i32.p0i32.i64(<8 x i32> [[TMP3]], i32* [[TMP2]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 361; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8 362; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 32 363; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 364; CHECK-NEXT: br i1 [[TMP4]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 365; CHECK: for.cond.cleanup: 366; CHECK-NEXT: ret void 367; 368entry: 369 br label %vector.body 370 371vector.body: ; preds = %vector.body, %entry 372 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 373 %vec.ind = phi <8 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>, %entry ], [ %vec.ind.next, %vector.body ] 374 %0 = getelementptr inbounds i32, i32* %B, i64 %index 375 %1 = bitcast i32* %0 to <8 x i32>* 376 %wide.load = load <8 x i32>, <8 x i32>* %1, align 1 377 %2 = shl nuw nsw <8 x i64> %vec.ind, <i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2> 378 %3 = getelementptr inbounds i32, i32* %A, <8 x i64> %2 379 %wide.masked.gather = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %3, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 380 %4 = add <8 x i32> %wide.masked.gather, %wide.load 381 call void @llvm.masked.scatter.v8i32.v8p0i32(<8 x i32> %4, <8 x i32*> %3, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 382 %index.next = add nuw i64 %index, 8 383 %vec.ind.next = add <8 x i64> %vec.ind, <i64 8, i64 8, i64 8, i64 8, i64 8, i64 8, i64 8, i64 8> 384 %5 = icmp eq i64 %index.next, 1024 385 br i1 %5, label %for.cond.cleanup, label %vector.body 386 387for.cond.cleanup: ; preds = %vector.body 388 ret void 389} 390 391;struct foo { 392; int a, b, c, d; 393;}; 394; 395;void struct_gather(int * __restrict A, struct foo * __restrict B) { 396; for (int i = 0; i < 1024; ++i) 397; A[i] += B[i].b; 398;} 399define void @struct_gather(i32* noalias nocapture %A, %struct.foo* noalias nocapture readonly %B) { 400; 401; CHECK-LABEL: @struct_gather( 402; CHECK-NEXT: entry: 403; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 404; CHECK: vector.body: 405; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 406; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 407; CHECK-NEXT: [[VEC_IND_SCALAR1:%.*]] = phi i64 [ 8, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR2:%.*]], [[VECTOR_BODY]] ] 408; CHECK-NEXT: [[TMP0:%.*]] = getelementptr [[STRUCT_FOO:%.*]], %struct.foo* [[B:%.*]], i64 [[VEC_IND_SCALAR]], i32 1 409; CHECK-NEXT: [[TMP1:%.*]] = getelementptr [[STRUCT_FOO]], %struct.foo* [[B]], i64 [[VEC_IND_SCALAR1]], i32 1 410; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP0]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 411; CHECK-NEXT: [[WIDE_MASKED_GATHER9:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP1]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 412; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, i32* [[A:%.*]], i64 [[INDEX]] 413; CHECK-NEXT: [[TMP3:%.*]] = bitcast i32* [[TMP2]] to <8 x i32>* 414; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, <8 x i32>* [[TMP3]], align 4 415; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds i32, i32* [[TMP2]], i64 8 416; CHECK-NEXT: [[TMP5:%.*]] = bitcast i32* [[TMP4]] to <8 x i32>* 417; CHECK-NEXT: [[WIDE_LOAD10:%.*]] = load <8 x i32>, <8 x i32>* [[TMP5]], align 4 418; CHECK-NEXT: [[TMP6:%.*]] = add nsw <8 x i32> [[WIDE_LOAD]], [[WIDE_MASKED_GATHER]] 419; CHECK-NEXT: [[TMP7:%.*]] = add nsw <8 x i32> [[WIDE_LOAD10]], [[WIDE_MASKED_GATHER9]] 420; CHECK-NEXT: [[TMP8:%.*]] = bitcast i32* [[TMP2]] to <8 x i32>* 421; CHECK-NEXT: store <8 x i32> [[TMP6]], <8 x i32>* [[TMP8]], align 4 422; CHECK-NEXT: [[TMP9:%.*]] = bitcast i32* [[TMP4]] to <8 x i32>* 423; CHECK-NEXT: store <8 x i32> [[TMP7]], <8 x i32>* [[TMP9]], align 4 424; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16 425; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 16 426; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR2]] = add i64 [[VEC_IND_SCALAR1]], 16 427; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024 428; CHECK-NEXT: br i1 [[TMP10]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 429; CHECK: for.cond.cleanup: 430; CHECK-NEXT: ret void 431; 432entry: 433 br label %vector.body 434 435vector.body: ; preds = %vector.body, %entry 436 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 437 %vec.ind = phi <8 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>, %entry ], [ %vec.ind.next, %vector.body ] 438 %step.add = add <8 x i64> %vec.ind, <i64 8, i64 8, i64 8, i64 8, i64 8, i64 8, i64 8, i64 8> 439 %0 = getelementptr inbounds %struct.foo, %struct.foo* %B, <8 x i64> %vec.ind, i32 1 440 %1 = getelementptr inbounds %struct.foo, %struct.foo* %B, <8 x i64> %step.add, i32 1 441 %wide.masked.gather = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %0, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 442 %wide.masked.gather9 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %1, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 443 %2 = getelementptr inbounds i32, i32* %A, i64 %index 444 %3 = bitcast i32* %2 to <8 x i32>* 445 %wide.load = load <8 x i32>, <8 x i32>* %3, align 4 446 %4 = getelementptr inbounds i32, i32* %2, i64 8 447 %5 = bitcast i32* %4 to <8 x i32>* 448 %wide.load10 = load <8 x i32>, <8 x i32>* %5, align 4 449 %6 = add nsw <8 x i32> %wide.load, %wide.masked.gather 450 %7 = add nsw <8 x i32> %wide.load10, %wide.masked.gather9 451 %8 = bitcast i32* %2 to <8 x i32>* 452 store <8 x i32> %6, <8 x i32>* %8, align 4 453 %9 = bitcast i32* %4 to <8 x i32>* 454 store <8 x i32> %7, <8 x i32>* %9, align 4 455 %index.next = add nuw i64 %index, 16 456 %vec.ind.next = add <8 x i64> %vec.ind, <i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16> 457 %10 = icmp eq i64 %index.next, 1024 458 br i1 %10, label %for.cond.cleanup, label %vector.body 459 460for.cond.cleanup: ; preds = %vector.body 461 ret void 462} 463 464;void gather_unroll(int * __restrict A, int * __restrict B) { 465; for (int i = 0; i < 1024; i+= 4 ) { 466; A[i] += B[i * 4]; 467; A[i+1] += B[(i+1) * 4]; 468; A[i+2] += B[(i+2) * 4]; 469; A[i+3] += B[(i+3) * 4]; 470; } 471;} 472define void @gather_unroll(i32* noalias nocapture %A, i32* noalias nocapture readonly %B) { 473; 474; CHECK-LABEL: @gather_unroll( 475; CHECK-NEXT: entry: 476; CHECK-NEXT: br label [[VECTOR_BODY:%.*]] 477; CHECK: vector.body: 478; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ] 479; CHECK-NEXT: [[VEC_IND_SCALAR:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR:%.*]], [[VECTOR_BODY]] ] 480; CHECK-NEXT: [[VEC_IND_SCALAR1:%.*]] = phi i64 [ 0, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR2:%.*]], [[VECTOR_BODY]] ] 481; CHECK-NEXT: [[VEC_IND_SCALAR3:%.*]] = phi i64 [ 4, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR4:%.*]], [[VECTOR_BODY]] ] 482; CHECK-NEXT: [[VEC_IND_SCALAR5:%.*]] = phi i64 [ 1, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR6:%.*]], [[VECTOR_BODY]] ] 483; CHECK-NEXT: [[VEC_IND_SCALAR7:%.*]] = phi i64 [ 8, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR8:%.*]], [[VECTOR_BODY]] ] 484; CHECK-NEXT: [[VEC_IND_SCALAR9:%.*]] = phi i64 [ 2, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR10:%.*]], [[VECTOR_BODY]] ] 485; CHECK-NEXT: [[VEC_IND_SCALAR11:%.*]] = phi i64 [ 12, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR12:%.*]], [[VECTOR_BODY]] ] 486; CHECK-NEXT: [[VEC_IND_SCALAR13:%.*]] = phi i64 [ 3, [[ENTRY]] ], [ [[VEC_IND_NEXT_SCALAR14:%.*]], [[VECTOR_BODY]] ] 487; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i32, i32* [[B:%.*]], i64 [[VEC_IND_SCALAR]] 488; CHECK-NEXT: [[WIDE_MASKED_GATHER:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP0]], i64 64, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 489; CHECK-NEXT: [[TMP1:%.*]] = getelementptr i32, i32* [[A:%.*]], i64 [[VEC_IND_SCALAR1]] 490; CHECK-NEXT: [[WIDE_MASKED_GATHER52:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP1]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 491; CHECK-NEXT: [[TMP2:%.*]] = add nsw <8 x i32> [[WIDE_MASKED_GATHER52]], [[WIDE_MASKED_GATHER]] 492; CHECK-NEXT: call void @llvm.riscv.masked.strided.store.v8i32.p0i32.i64(<8 x i32> [[TMP2]], i32* [[TMP1]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 493; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i32, i32* [[B]], i64 [[VEC_IND_SCALAR3]] 494; CHECK-NEXT: [[WIDE_MASKED_GATHER53:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP3]], i64 64, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 495; CHECK-NEXT: [[TMP4:%.*]] = getelementptr i32, i32* [[A]], i64 [[VEC_IND_SCALAR5]] 496; CHECK-NEXT: [[WIDE_MASKED_GATHER54:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP4]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 497; CHECK-NEXT: [[TMP5:%.*]] = add nsw <8 x i32> [[WIDE_MASKED_GATHER54]], [[WIDE_MASKED_GATHER53]] 498; CHECK-NEXT: call void @llvm.riscv.masked.strided.store.v8i32.p0i32.i64(<8 x i32> [[TMP5]], i32* [[TMP4]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 499; CHECK-NEXT: [[TMP6:%.*]] = getelementptr i32, i32* [[B]], i64 [[VEC_IND_SCALAR7]] 500; CHECK-NEXT: [[WIDE_MASKED_GATHER55:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP6]], i64 64, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 501; CHECK-NEXT: [[TMP7:%.*]] = getelementptr i32, i32* [[A]], i64 [[VEC_IND_SCALAR9]] 502; CHECK-NEXT: [[WIDE_MASKED_GATHER56:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP7]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 503; CHECK-NEXT: [[TMP8:%.*]] = add nsw <8 x i32> [[WIDE_MASKED_GATHER56]], [[WIDE_MASKED_GATHER55]] 504; CHECK-NEXT: call void @llvm.riscv.masked.strided.store.v8i32.p0i32.i64(<8 x i32> [[TMP8]], i32* [[TMP7]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 505; CHECK-NEXT: [[TMP9:%.*]] = getelementptr i32, i32* [[B]], i64 [[VEC_IND_SCALAR11]] 506; CHECK-NEXT: [[WIDE_MASKED_GATHER57:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP9]], i64 64, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 507; CHECK-NEXT: [[TMP10:%.*]] = getelementptr i32, i32* [[A]], i64 [[VEC_IND_SCALAR13]] 508; CHECK-NEXT: [[WIDE_MASKED_GATHER58:%.*]] = call <8 x i32> @llvm.riscv.masked.strided.load.v8i32.p0i32.i64(<8 x i32> undef, i32* [[TMP10]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 509; CHECK-NEXT: [[TMP11:%.*]] = add nsw <8 x i32> [[WIDE_MASKED_GATHER58]], [[WIDE_MASKED_GATHER57]] 510; CHECK-NEXT: call void @llvm.riscv.masked.strided.store.v8i32.p0i32.i64(<8 x i32> [[TMP11]], i32* [[TMP10]], i64 16, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 511; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8 512; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR]] = add i64 [[VEC_IND_SCALAR]], 128 513; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR2]] = add i64 [[VEC_IND_SCALAR1]], 32 514; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR4]] = add i64 [[VEC_IND_SCALAR3]], 128 515; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR6]] = add i64 [[VEC_IND_SCALAR5]], 32 516; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR8]] = add i64 [[VEC_IND_SCALAR7]], 128 517; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR10]] = add i64 [[VEC_IND_SCALAR9]], 32 518; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR12]] = add i64 [[VEC_IND_SCALAR11]], 128 519; CHECK-NEXT: [[VEC_IND_NEXT_SCALAR14]] = add i64 [[VEC_IND_SCALAR13]], 32 520; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 256 521; CHECK-NEXT: br i1 [[TMP12]], label [[FOR_COND_CLEANUP:%.*]], label [[VECTOR_BODY]] 522; CHECK: for.cond.cleanup: 523; CHECK-NEXT: ret void 524; 525entry: 526 br label %vector.body 527 528vector.body: ; preds = %vector.body, %entry 529 %index = phi i64 [ 0, %entry ], [ %index.next, %vector.body ] 530 %vec.ind = phi <8 x i64> [ <i64 0, i64 4, i64 8, i64 12, i64 16, i64 20, i64 24, i64 28>, %entry ], [ %vec.ind.next, %vector.body ] 531 %0 = shl nuw nsw <8 x i64> %vec.ind, <i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2> 532 %1 = getelementptr inbounds i32, i32* %B, <8 x i64> %0 533 %wide.masked.gather = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %1, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 534 %2 = getelementptr inbounds i32, i32* %A, <8 x i64> %vec.ind 535 %wide.masked.gather52 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %2, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 536 %3 = add nsw <8 x i32> %wide.masked.gather52, %wide.masked.gather 537 call void @llvm.masked.scatter.v8i32.v8p0i32(<8 x i32> %3, <8 x i32*> %2, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 538 %4 = or <8 x i64> %vec.ind, <i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1, i64 1> 539 %5 = shl nsw <8 x i64> %4, <i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2> 540 %6 = getelementptr inbounds i32, i32* %B, <8 x i64> %5 541 %wide.masked.gather53 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %6, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 542 %7 = getelementptr inbounds i32, i32* %A, <8 x i64> %4 543 %wide.masked.gather54 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %7, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 544 %8 = add nsw <8 x i32> %wide.masked.gather54, %wide.masked.gather53 545 call void @llvm.masked.scatter.v8i32.v8p0i32(<8 x i32> %8, <8 x i32*> %7, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 546 %9 = or <8 x i64> %vec.ind, <i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2> 547 %10 = shl nsw <8 x i64> %9, <i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2> 548 %11 = getelementptr inbounds i32, i32* %B, <8 x i64> %10 549 %wide.masked.gather55 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %11, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 550 %12 = getelementptr inbounds i32, i32* %A, <8 x i64> %9 551 %wide.masked.gather56 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %12, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 552 %13 = add nsw <8 x i32> %wide.masked.gather56, %wide.masked.gather55 553 call void @llvm.masked.scatter.v8i32.v8p0i32(<8 x i32> %13, <8 x i32*> %12, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 554 %14 = or <8 x i64> %vec.ind, <i64 3, i64 3, i64 3, i64 3, i64 3, i64 3, i64 3, i64 3> 555 %15 = shl nsw <8 x i64> %14, <i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2, i64 2> 556 %16 = getelementptr inbounds i32, i32* %B, <8 x i64> %15 557 %wide.masked.gather57 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %16, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 558 %17 = getelementptr inbounds i32, i32* %A, <8 x i64> %14 559 %wide.masked.gather58 = call <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*> %17, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <8 x i32> undef) 560 %18 = add nsw <8 x i32> %wide.masked.gather58, %wide.masked.gather57 561 call void @llvm.masked.scatter.v8i32.v8p0i32(<8 x i32> %18, <8 x i32*> %17, i32 4, <8 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 562 %index.next = add nuw i64 %index, 8 563 %vec.ind.next = add <8 x i64> %vec.ind, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 564 %19 = icmp eq i64 %index.next, 256 565 br i1 %19, label %for.cond.cleanup, label %vector.body 566 567for.cond.cleanup: ; preds = %vector.body 568 ret void 569} 570 571declare <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*>, i32 immarg, <32 x i1>, <32 x i8>) 572declare <8 x i32> @llvm.masked.gather.v8i32.v8p0i32(<8 x i32*>, i32 immarg, <8 x i1>, <8 x i32>) 573declare void @llvm.masked.scatter.v32i8.v32p0i8(<32 x i8>, <32 x i8*>, i32 immarg, <32 x i1>) 574declare void @llvm.masked.scatter.v8i32.v8p0i32(<8 x i32>, <8 x i32*>, i32 immarg, <8 x i1>) 575 576; Make sure we don't crash in getTgtMemIntrinsic for a vector of pointers. 577define void @gather_of_pointers(i32** noalias nocapture %0, i32** noalias nocapture readonly %1) { 578; 579; V-LABEL: @gather_of_pointers( 580; V-NEXT: br label [[TMP3:%.*]] 581; V: 3: 582; V-NEXT: [[TMP4:%.*]] = phi i64 [ 0, [[TMP2:%.*]] ], [ [[TMP13:%.*]], [[TMP3]] ] 583; V-NEXT: [[DOTSCALAR:%.*]] = phi i64 [ 0, [[TMP2]] ], [ [[DOTSCALAR1:%.*]], [[TMP3]] ] 584; V-NEXT: [[DOTSCALAR2:%.*]] = phi i64 [ 10, [[TMP2]] ], [ [[DOTSCALAR3:%.*]], [[TMP3]] ] 585; V-NEXT: [[TMP5:%.*]] = getelementptr i32*, i32** [[TMP1:%.*]], i64 [[DOTSCALAR]] 586; V-NEXT: [[TMP6:%.*]] = getelementptr i32*, i32** [[TMP1]], i64 [[DOTSCALAR2]] 587; V-NEXT: [[TMP7:%.*]] = call <2 x i32*> @llvm.riscv.masked.strided.load.v2p0i32.p0p0i32.i64(<2 x i32*> undef, i32** [[TMP5]], i64 40, <2 x i1> <i1 true, i1 true>) 588; V-NEXT: [[TMP8:%.*]] = call <2 x i32*> @llvm.riscv.masked.strided.load.v2p0i32.p0p0i32.i64(<2 x i32*> undef, i32** [[TMP6]], i64 40, <2 x i1> <i1 true, i1 true>) 589; V-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32*, i32** [[TMP0:%.*]], i64 [[TMP4]] 590; V-NEXT: [[TMP10:%.*]] = bitcast i32** [[TMP9]] to <2 x i32*>* 591; V-NEXT: store <2 x i32*> [[TMP7]], <2 x i32*>* [[TMP10]], align 8 592; V-NEXT: [[TMP11:%.*]] = getelementptr inbounds i32*, i32** [[TMP9]], i64 2 593; V-NEXT: [[TMP12:%.*]] = bitcast i32** [[TMP11]] to <2 x i32*>* 594; V-NEXT: store <2 x i32*> [[TMP8]], <2 x i32*>* [[TMP12]], align 8 595; V-NEXT: [[TMP13]] = add nuw i64 [[TMP4]], 4 596; V-NEXT: [[DOTSCALAR1]] = add i64 [[DOTSCALAR]], 20 597; V-NEXT: [[DOTSCALAR3]] = add i64 [[DOTSCALAR2]], 20 598; V-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 1024 599; V-NEXT: br i1 [[TMP14]], label [[TMP15:%.*]], label [[TMP3]] 600; V: 15: 601; V-NEXT: ret void 602; 603; ZVE32F-LABEL: @gather_of_pointers( 604; ZVE32F-NEXT: br label [[TMP3:%.*]] 605; ZVE32F: 3: 606; ZVE32F-NEXT: [[TMP4:%.*]] = phi i64 [ 0, [[TMP2:%.*]] ], [ [[TMP17:%.*]], [[TMP3]] ] 607; ZVE32F-NEXT: [[TMP5:%.*]] = phi <2 x i64> [ <i64 0, i64 1>, [[TMP2]] ], [ [[TMP18:%.*]], [[TMP3]] ] 608; ZVE32F-NEXT: [[TMP6:%.*]] = mul nuw nsw <2 x i64> [[TMP5]], <i64 5, i64 5> 609; ZVE32F-NEXT: [[TMP7:%.*]] = mul <2 x i64> [[TMP5]], <i64 5, i64 5> 610; ZVE32F-NEXT: [[TMP8:%.*]] = add <2 x i64> [[TMP7]], <i64 10, i64 10> 611; ZVE32F-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32*, i32** [[TMP1:%.*]], <2 x i64> [[TMP6]] 612; ZVE32F-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32*, i32** [[TMP1]], <2 x i64> [[TMP8]] 613; ZVE32F-NEXT: [[TMP11:%.*]] = call <2 x i32*> @llvm.masked.gather.v2p0i32.v2p0p0i32(<2 x i32**> [[TMP9]], i32 8, <2 x i1> <i1 true, i1 true>, <2 x i32*> undef) 614; ZVE32F-NEXT: [[TMP12:%.*]] = call <2 x i32*> @llvm.masked.gather.v2p0i32.v2p0p0i32(<2 x i32**> [[TMP10]], i32 8, <2 x i1> <i1 true, i1 true>, <2 x i32*> undef) 615; ZVE32F-NEXT: [[TMP13:%.*]] = getelementptr inbounds i32*, i32** [[TMP0:%.*]], i64 [[TMP4]] 616; ZVE32F-NEXT: [[TMP14:%.*]] = bitcast i32** [[TMP13]] to <2 x i32*>* 617; ZVE32F-NEXT: store <2 x i32*> [[TMP11]], <2 x i32*>* [[TMP14]], align 8 618; ZVE32F-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32*, i32** [[TMP13]], i64 2 619; ZVE32F-NEXT: [[TMP16:%.*]] = bitcast i32** [[TMP15]] to <2 x i32*>* 620; ZVE32F-NEXT: store <2 x i32*> [[TMP12]], <2 x i32*>* [[TMP16]], align 8 621; ZVE32F-NEXT: [[TMP17]] = add nuw i64 [[TMP4]], 4 622; ZVE32F-NEXT: [[TMP18]] = add <2 x i64> [[TMP5]], <i64 4, i64 4> 623; ZVE32F-NEXT: [[TMP19:%.*]] = icmp eq i64 [[TMP17]], 1024 624; ZVE32F-NEXT: br i1 [[TMP19]], label [[TMP20:%.*]], label [[TMP3]] 625; ZVE32F: 20: 626; ZVE32F-NEXT: ret void 627; 628 br label %3 629 6303: ; preds = %3, %2 631 %4 = phi i64 [ 0, %2 ], [ %17, %3 ] 632 %5 = phi <2 x i64> [ <i64 0, i64 1>, %2 ], [ %18, %3 ] 633 %6 = mul nuw nsw <2 x i64> %5, <i64 5, i64 5> 634 %7 = mul <2 x i64> %5, <i64 5, i64 5> 635 %8 = add <2 x i64> %7, <i64 10, i64 10> 636 %9 = getelementptr inbounds i32*, i32** %1, <2 x i64> %6 637 %10 = getelementptr inbounds i32*, i32** %1, <2 x i64> %8 638 %11 = call <2 x i32*> @llvm.masked.gather.v2p0i32.v2p0p0i32(<2 x i32**> %9, i32 8, <2 x i1> <i1 true, i1 true>, <2 x i32*> undef) 639 %12 = call <2 x i32*> @llvm.masked.gather.v2p0i32.v2p0p0i32(<2 x i32**> %10, i32 8, <2 x i1> <i1 true, i1 true>, <2 x i32*> undef) 640 %13 = getelementptr inbounds i32*, i32** %0, i64 %4 641 %14 = bitcast i32** %13 to <2 x i32*>* 642 store <2 x i32*> %11, <2 x i32*>* %14, align 8 643 %15 = getelementptr inbounds i32*, i32** %13, i64 2 644 %16 = bitcast i32** %15 to <2 x i32*>* 645 store <2 x i32*> %12, <2 x i32*>* %16, align 8 646 %17 = add nuw i64 %4, 4 647 %18 = add <2 x i64> %5, <i64 4, i64 4> 648 %19 = icmp eq i64 %17, 1024 649 br i1 %19, label %20, label %3 650 65120: ; preds = %3 652 ret void 653} 654 655declare <2 x i32*> @llvm.masked.gather.v2p0i32.v2p0p0i32(<2 x i32**>, i32 immarg, <2 x i1>, <2 x i32*>) 656 657; Make sure we don't crash in getTgtMemIntrinsic for a vector of pointers. 658define void @scatter_of_pointers(i32** noalias nocapture %0, i32** noalias nocapture readonly %1) { 659; 660; V-LABEL: @scatter_of_pointers( 661; V-NEXT: br label [[TMP3:%.*]] 662; V: 3: 663; V-NEXT: [[TMP4:%.*]] = phi i64 [ 0, [[TMP2:%.*]] ], [ [[TMP13:%.*]], [[TMP3]] ] 664; V-NEXT: [[DOTSCALAR:%.*]] = phi i64 [ 0, [[TMP2]] ], [ [[DOTSCALAR1:%.*]], [[TMP3]] ] 665; V-NEXT: [[DOTSCALAR2:%.*]] = phi i64 [ 10, [[TMP2]] ], [ [[DOTSCALAR3:%.*]], [[TMP3]] ] 666; V-NEXT: [[TMP5:%.*]] = getelementptr inbounds i32*, i32** [[TMP1:%.*]], i64 [[TMP4]] 667; V-NEXT: [[TMP6:%.*]] = bitcast i32** [[TMP5]] to <2 x i32*>* 668; V-NEXT: [[TMP7:%.*]] = load <2 x i32*>, <2 x i32*>* [[TMP6]], align 8 669; V-NEXT: [[TMP8:%.*]] = getelementptr inbounds i32*, i32** [[TMP5]], i64 2 670; V-NEXT: [[TMP9:%.*]] = bitcast i32** [[TMP8]] to <2 x i32*>* 671; V-NEXT: [[TMP10:%.*]] = load <2 x i32*>, <2 x i32*>* [[TMP9]], align 8 672; V-NEXT: [[TMP11:%.*]] = getelementptr i32*, i32** [[TMP0:%.*]], i64 [[DOTSCALAR]] 673; V-NEXT: [[TMP12:%.*]] = getelementptr i32*, i32** [[TMP0]], i64 [[DOTSCALAR2]] 674; V-NEXT: call void @llvm.riscv.masked.strided.store.v2p0i32.p0p0i32.i64(<2 x i32*> [[TMP7]], i32** [[TMP11]], i64 40, <2 x i1> <i1 true, i1 true>) 675; V-NEXT: call void @llvm.riscv.masked.strided.store.v2p0i32.p0p0i32.i64(<2 x i32*> [[TMP10]], i32** [[TMP12]], i64 40, <2 x i1> <i1 true, i1 true>) 676; V-NEXT: [[TMP13]] = add nuw i64 [[TMP4]], 4 677; V-NEXT: [[DOTSCALAR1]] = add i64 [[DOTSCALAR]], 20 678; V-NEXT: [[DOTSCALAR3]] = add i64 [[DOTSCALAR2]], 20 679; V-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 1024 680; V-NEXT: br i1 [[TMP14]], label [[TMP15:%.*]], label [[TMP3]] 681; V: 15: 682; V-NEXT: ret void 683; 684; ZVE32F-LABEL: @scatter_of_pointers( 685; ZVE32F-NEXT: br label [[TMP3:%.*]] 686; ZVE32F: 3: 687; ZVE32F-NEXT: [[TMP4:%.*]] = phi i64 [ 0, [[TMP2:%.*]] ], [ [[TMP17:%.*]], [[TMP3]] ] 688; ZVE32F-NEXT: [[TMP5:%.*]] = phi <2 x i64> [ <i64 0, i64 1>, [[TMP2]] ], [ [[TMP18:%.*]], [[TMP3]] ] 689; ZVE32F-NEXT: [[TMP6:%.*]] = getelementptr inbounds i32*, i32** [[TMP1:%.*]], i64 [[TMP4]] 690; ZVE32F-NEXT: [[TMP7:%.*]] = bitcast i32** [[TMP6]] to <2 x i32*>* 691; ZVE32F-NEXT: [[TMP8:%.*]] = load <2 x i32*>, <2 x i32*>* [[TMP7]], align 8 692; ZVE32F-NEXT: [[TMP9:%.*]] = getelementptr inbounds i32*, i32** [[TMP6]], i64 2 693; ZVE32F-NEXT: [[TMP10:%.*]] = bitcast i32** [[TMP9]] to <2 x i32*>* 694; ZVE32F-NEXT: [[TMP11:%.*]] = load <2 x i32*>, <2 x i32*>* [[TMP10]], align 8 695; ZVE32F-NEXT: [[TMP12:%.*]] = mul nuw nsw <2 x i64> [[TMP5]], <i64 5, i64 5> 696; ZVE32F-NEXT: [[TMP13:%.*]] = mul <2 x i64> [[TMP5]], <i64 5, i64 5> 697; ZVE32F-NEXT: [[TMP14:%.*]] = add <2 x i64> [[TMP13]], <i64 10, i64 10> 698; ZVE32F-NEXT: [[TMP15:%.*]] = getelementptr inbounds i32*, i32** [[TMP0:%.*]], <2 x i64> [[TMP12]] 699; ZVE32F-NEXT: [[TMP16:%.*]] = getelementptr inbounds i32*, i32** [[TMP0]], <2 x i64> [[TMP14]] 700; ZVE32F-NEXT: call void @llvm.masked.scatter.v2p0i32.v2p0p0i32(<2 x i32*> [[TMP8]], <2 x i32**> [[TMP15]], i32 8, <2 x i1> <i1 true, i1 true>) 701; ZVE32F-NEXT: call void @llvm.masked.scatter.v2p0i32.v2p0p0i32(<2 x i32*> [[TMP11]], <2 x i32**> [[TMP16]], i32 8, <2 x i1> <i1 true, i1 true>) 702; ZVE32F-NEXT: [[TMP17]] = add nuw i64 [[TMP4]], 4 703; ZVE32F-NEXT: [[TMP18]] = add <2 x i64> [[TMP5]], <i64 4, i64 4> 704; ZVE32F-NEXT: [[TMP19:%.*]] = icmp eq i64 [[TMP17]], 1024 705; ZVE32F-NEXT: br i1 [[TMP19]], label [[TMP20:%.*]], label [[TMP3]] 706; ZVE32F: 20: 707; ZVE32F-NEXT: ret void 708; 709 br label %3 710 7113: ; preds = %3, %2 712 %4 = phi i64 [ 0, %2 ], [ %17, %3 ] 713 %5 = phi <2 x i64> [ <i64 0, i64 1>, %2 ], [ %18, %3 ] 714 %6 = getelementptr inbounds i32*, i32** %1, i64 %4 715 %7 = bitcast i32** %6 to <2 x i32*>* 716 %8 = load <2 x i32*>, <2 x i32*>* %7, align 8 717 %9 = getelementptr inbounds i32*, i32** %6, i64 2 718 %10 = bitcast i32** %9 to <2 x i32*>* 719 %11 = load <2 x i32*>, <2 x i32*>* %10, align 8 720 %12 = mul nuw nsw <2 x i64> %5, <i64 5, i64 5> 721 %13 = mul <2 x i64> %5, <i64 5, i64 5> 722 %14 = add <2 x i64> %13, <i64 10, i64 10> 723 %15 = getelementptr inbounds i32*, i32** %0, <2 x i64> %12 724 %16 = getelementptr inbounds i32*, i32** %0, <2 x i64> %14 725 call void @llvm.masked.scatter.v2p0i32.v2p0p0i32(<2 x i32*> %8, <2 x i32**> %15, i32 8, <2 x i1> <i1 true, i1 true>) 726 call void @llvm.masked.scatter.v2p0i32.v2p0p0i32(<2 x i32*> %11, <2 x i32**> %16, i32 8, <2 x i1> <i1 true, i1 true>) 727 %17 = add nuw i64 %4, 4 728 %18 = add <2 x i64> %5, <i64 4, i64 4> 729 %19 = icmp eq i64 %17, 1024 730 br i1 %19, label %20, label %3 731 73220: ; preds = %3 733 ret void 734} 735 736declare void @llvm.masked.scatter.v2p0i32.v2p0p0i32(<2 x i32*>, <2 x i32**>, i32 immarg, <2 x i1>) 737 738define void @strided_load_startval_add_with_splat(i8* noalias nocapture %0, i8* noalias nocapture readonly %1, i32 signext %2) { 739; 740; CHECK-LABEL: @strided_load_startval_add_with_splat( 741; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[TMP2:%.*]], 1024 742; CHECK-NEXT: br i1 [[TMP4]], label [[TMP31:%.*]], label [[TMP5:%.*]] 743; CHECK: 5: 744; CHECK-NEXT: [[TMP6:%.*]] = sext i32 [[TMP2]] to i64 745; CHECK-NEXT: [[TMP7:%.*]] = sub i32 1023, [[TMP2]] 746; CHECK-NEXT: [[TMP8:%.*]] = zext i32 [[TMP7]] to i64 747; CHECK-NEXT: [[TMP9:%.*]] = add nuw nsw i64 [[TMP8]], 1 748; CHECK-NEXT: [[TMP10:%.*]] = icmp ult i32 [[TMP7]], 31 749; CHECK-NEXT: br i1 [[TMP10]], label [[TMP29:%.*]], label [[TMP11:%.*]] 750; CHECK: 11: 751; CHECK-NEXT: [[TMP12:%.*]] = and i64 [[TMP9]], 8589934560 752; CHECK-NEXT: [[TMP13:%.*]] = add nsw i64 [[TMP12]], [[TMP6]] 753; CHECK-NEXT: [[TMP14:%.*]] = add i64 0, [[TMP6]] 754; CHECK-NEXT: [[START:%.*]] = mul i64 [[TMP14]], 5 755; CHECK-NEXT: br label [[TMP15:%.*]] 756; CHECK: 15: 757; CHECK-NEXT: [[TMP16:%.*]] = phi i64 [ 0, [[TMP11]] ], [ [[TMP25:%.*]], [[TMP15]] ] 758; CHECK-NEXT: [[DOTSCALAR:%.*]] = phi i64 [ [[START]], [[TMP11]] ], [ [[DOTSCALAR1:%.*]], [[TMP15]] ] 759; CHECK-NEXT: [[TMP17:%.*]] = add i64 [[TMP16]], [[TMP6]] 760; CHECK-NEXT: [[TMP18:%.*]] = getelementptr i8, i8* [[TMP1:%.*]], i64 [[DOTSCALAR]] 761; CHECK-NEXT: [[TMP19:%.*]] = call <32 x i8> @llvm.riscv.masked.strided.load.v32i8.p0i8.i64(<32 x i8> undef, i8* [[TMP18]], i64 5, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 762; CHECK-NEXT: [[TMP20:%.*]] = getelementptr inbounds i8, i8* [[TMP0:%.*]], i64 [[TMP17]] 763; CHECK-NEXT: [[TMP21:%.*]] = bitcast i8* [[TMP20]] to <32 x i8>* 764; CHECK-NEXT: [[TMP22:%.*]] = load <32 x i8>, <32 x i8>* [[TMP21]], align 1 765; CHECK-NEXT: [[TMP23:%.*]] = add <32 x i8> [[TMP22]], [[TMP19]] 766; CHECK-NEXT: [[TMP24:%.*]] = bitcast i8* [[TMP20]] to <32 x i8>* 767; CHECK-NEXT: store <32 x i8> [[TMP23]], <32 x i8>* [[TMP24]], align 1 768; CHECK-NEXT: [[TMP25]] = add nuw i64 [[TMP16]], 32 769; CHECK-NEXT: [[DOTSCALAR1]] = add i64 [[DOTSCALAR]], 160 770; CHECK-NEXT: [[TMP26:%.*]] = icmp eq i64 [[TMP25]], [[TMP12]] 771; CHECK-NEXT: br i1 [[TMP26]], label [[TMP27:%.*]], label [[TMP15]] 772; CHECK: 27: 773; CHECK-NEXT: [[TMP28:%.*]] = icmp eq i64 [[TMP9]], [[TMP12]] 774; CHECK-NEXT: br i1 [[TMP28]], label [[TMP31]], label [[TMP29]] 775; CHECK: 29: 776; CHECK-NEXT: [[TMP30:%.*]] = phi i64 [ [[TMP6]], [[TMP5]] ], [ [[TMP13]], [[TMP27]] ] 777; CHECK-NEXT: br label [[TMP32:%.*]] 778; CHECK: 31: 779; CHECK-NEXT: ret void 780; CHECK: 32: 781; CHECK-NEXT: [[TMP33:%.*]] = phi i64 [ [[TMP40:%.*]], [[TMP32]] ], [ [[TMP30]], [[TMP29]] ] 782; CHECK-NEXT: [[TMP34:%.*]] = mul nsw i64 [[TMP33]], 5 783; CHECK-NEXT: [[TMP35:%.*]] = getelementptr inbounds i8, i8* [[TMP1]], i64 [[TMP34]] 784; CHECK-NEXT: [[TMP36:%.*]] = load i8, i8* [[TMP35]], align 1 785; CHECK-NEXT: [[TMP37:%.*]] = getelementptr inbounds i8, i8* [[TMP0]], i64 [[TMP33]] 786; CHECK-NEXT: [[TMP38:%.*]] = load i8, i8* [[TMP37]], align 1 787; CHECK-NEXT: [[TMP39:%.*]] = add i8 [[TMP38]], [[TMP36]] 788; CHECK-NEXT: store i8 [[TMP39]], i8* [[TMP37]], align 1 789; CHECK-NEXT: [[TMP40]] = add nsw i64 [[TMP33]], 1 790; CHECK-NEXT: [[TMP41:%.*]] = trunc i64 [[TMP40]] to i32 791; CHECK-NEXT: [[TMP42:%.*]] = icmp eq i32 [[TMP41]], 1024 792; CHECK-NEXT: br i1 [[TMP42]], label [[TMP31]], label [[TMP32]] 793; 794 %4 = icmp eq i32 %2, 1024 795 br i1 %4, label %36, label %5 796 7975: ; preds = %3 798 %6 = sext i32 %2 to i64 799 %7 = sub i32 1023, %2 800 %8 = zext i32 %7 to i64 801 %9 = add nuw nsw i64 %8, 1 802 %10 = icmp ult i32 %7, 31 803 br i1 %10, label %34, label %11 804 80511: ; preds = %5 806 %12 = and i64 %9, 8589934560 807 %13 = add nsw i64 %12, %6 808 %14 = insertelement <32 x i64> poison, i64 %6, i64 0 809 %15 = shufflevector <32 x i64> %14, <32 x i64> poison, <32 x i32> zeroinitializer 810 %16 = add <32 x i64> %15, <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15, i64 16, i64 17, i64 18, i64 19, i64 20, i64 21, i64 22, i64 23, i64 24, i64 25, i64 26, i64 27, i64 28, i64 29, i64 30, i64 31> 811 br label %17 812 81317: ; preds = %17, %11 814 %18 = phi i64 [ 0, %11 ], [ %29, %17 ] 815 %19 = phi <32 x i64> [ %16, %11 ], [ %30, %17 ] 816 %20 = add i64 %18, %6 817 %21 = mul nsw <32 x i64> %19, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 818 %22 = getelementptr inbounds i8, i8* %1, <32 x i64> %21 819 %23 = call <32 x i8> @llvm.masked.gather.v32i8.v32p0i8(<32 x i8*> %22, i32 1, <32 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <32 x i8> undef) 820 %24 = getelementptr inbounds i8, i8* %0, i64 %20 821 %25 = bitcast i8* %24 to <32 x i8>* 822 %26 = load <32 x i8>, <32 x i8>* %25, align 1 823 %27 = add <32 x i8> %26, %23 824 %28 = bitcast i8* %24 to <32 x i8>* 825 store <32 x i8> %27, <32 x i8>* %28, align 1 826 %29 = add nuw i64 %18, 32 827 %30 = add <32 x i64> %19, <i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32, i64 32> 828 %31 = icmp eq i64 %29, %12 829 br i1 %31, label %32, label %17 830 83132: ; preds = %17 832 %33 = icmp eq i64 %9, %12 833 br i1 %33, label %36, label %34 834 83534: ; preds = %5, %32 836 %35 = phi i64 [ %6, %5 ], [ %13, %32 ] 837 br label %37 838 83936: ; preds = %37, %32, %3 840 ret void 841 84237: ; preds = %34, %37 843 %38 = phi i64 [ %45, %37 ], [ %35, %34 ] 844 %39 = mul nsw i64 %38, 5 845 %40 = getelementptr inbounds i8, i8* %1, i64 %39 846 %41 = load i8, i8* %40, align 1 847 %42 = getelementptr inbounds i8, i8* %0, i64 %38 848 %43 = load i8, i8* %42, align 1 849 %44 = add i8 %43, %41 850 store i8 %44, i8* %42, align 1 851 %45 = add nsw i64 %38, 1 852 %46 = trunc i64 %45 to i32 853 %47 = icmp eq i32 %46, 1024 854 br i1 %47, label %36, label %37 855} 856 857declare <16 x i8> @llvm.masked.gather.v16i8.v16p0i8(<16 x i8*>, i32 immarg, <16 x i1>, <16 x i8>) 858declare void @llvm.masked.scatter.v16i8.v16p0i8(<16 x i8>, <16 x i8*>, i32 immarg, <16 x i1>) 859 860define void @gather_no_scalar_remainder(i8* noalias nocapture noundef %arg, i8* noalias nocapture noundef readonly %arg1, i64 noundef %arg2) { 861; CHECK-LABEL: @gather_no_scalar_remainder( 862; CHECK-NEXT: bb: 863; CHECK-NEXT: [[I:%.*]] = shl i64 [[ARG2:%.*]], 4 864; CHECK-NEXT: [[I3:%.*]] = icmp eq i64 [[I]], 0 865; CHECK-NEXT: br i1 [[I3]], label [[BB16:%.*]], label [[BB2:%.*]] 866; CHECK: bb2: 867; CHECK-NEXT: br label [[BB4:%.*]] 868; CHECK: bb4: 869; CHECK-NEXT: [[I5:%.*]] = phi i64 [ [[I13:%.*]], [[BB4]] ], [ 0, [[BB2]] ] 870; CHECK-NEXT: [[I6_SCALAR:%.*]] = phi i64 [ 0, [[BB2]] ], [ [[I14_SCALAR:%.*]], [[BB4]] ] 871; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i8, i8* [[ARG1:%.*]], i64 [[I6_SCALAR]] 872; CHECK-NEXT: [[I9:%.*]] = call <16 x i8> @llvm.riscv.masked.strided.load.v16i8.p0i8.i64(<16 x i8> undef, i8* [[TMP0]], i64 5, <16 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>) 873; CHECK-NEXT: [[I10:%.*]] = getelementptr inbounds i8, i8* [[ARG:%.*]], i64 [[I5]] 874; CHECK-NEXT: [[CAST:%.*]] = bitcast i8* [[I10]] to <16 x i8>* 875; CHECK-NEXT: [[I11:%.*]] = load <16 x i8>, <16 x i8>* [[CAST]], align 1 876; CHECK-NEXT: [[I12:%.*]] = add <16 x i8> [[I11]], [[I9]] 877; CHECK-NEXT: [[CAST2:%.*]] = bitcast i8* [[I10]] to <16 x i8>* 878; CHECK-NEXT: store <16 x i8> [[I12]], <16 x i8>* [[CAST2]], align 1 879; CHECK-NEXT: [[I13]] = add nuw i64 [[I5]], 16 880; CHECK-NEXT: [[I14_SCALAR]] = add i64 [[I6_SCALAR]], 80 881; CHECK-NEXT: [[I15:%.*]] = icmp eq i64 [[I13]], [[I]] 882; CHECK-NEXT: br i1 [[I15]], label [[BB16]], label [[BB4]] 883; CHECK: bb16: 884; CHECK-NEXT: ret void 885; 886bb: 887 %i = shl i64 %arg2, 4 888 %i3 = icmp eq i64 %i, 0 889 br i1 %i3, label %bb16, label %bb2 890 891bb2: 892 br label %bb4 893 894bb4: ; preds = %bb4, %bb 895 %i5 = phi i64 [ %i13, %bb4 ], [ 0, %bb2 ] 896 %i6 = phi <16 x i64> [ %i14, %bb4 ], [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15>, %bb2 ] 897 %i7 = mul <16 x i64> %i6, <i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5, i64 5> 898 %i8 = getelementptr inbounds i8, i8* %arg1, <16 x i64> %i7 899 %i9 = call <16 x i8> @llvm.masked.gather.v16i8.v16p0i8(<16 x i8*> %i8, i32 1, <16 x i1> <i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true, i1 true>, <16 x i8> undef) 900 %i10 = getelementptr inbounds i8, i8* %arg, i64 %i5 901 %cast = bitcast i8* %i10 to <16 x i8>* 902 %i11 = load <16 x i8>, <16 x i8>* %cast, align 1 903 %i12 = add <16 x i8> %i11, %i9 904 %cast2 = bitcast i8* %i10 to <16 x i8>* 905 store <16 x i8> %i12, <16 x i8>* %cast2, align 1 906 %i13 = add nuw i64 %i5, 16 907 %i14 = add <16 x i64> %i6, <i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16, i64 16> 908 %i15 = icmp eq i64 %i13, %i 909 br i1 %i15, label %bb16, label %bb4 910 911bb16: ; preds = %bb4, %bb 912 ret void 913} 914