1; RUN: opt < %s -loop-vectorize -S | FileCheck %s --check-prefixes=COMMON,DEFAULT 2; RUN: opt < %s -loop-vectorize -prefer-predicate-over-epilog -S | FileCheck %s --check-prefixes=COMMON,CHECK-TF,CHECK-PREFER 3; RUN: opt < %s -loop-vectorize -disable-mve-tail-predication=false -S | FileCheck %s --check-prefixes=COMMON,CHECK-TF,CHECK-ENABLE-TP 4 5target datalayout = "e-m:e-p:32:32-Fi8-i64:64-v128:64:128-a:0:32-n32-S64" 6target triple = "thumbv8.1m.main-arm-unknown-eabihf" 7 8; This IR corresponds to this type of C-code: 9; 10; void f(char *a, char *b, char *c, int N) { 11; while (N-- > 0) 12; *c++ = *a++ + *b++; 13; } 14; 15define dso_local void @sgt_loopguard(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 16; COMMON-LABEL: @sgt_loopguard( 17; COMMON: vector.body: 18; CHECK-TF: masked.load 19; CHECK-TF: masked.load 20; CHECK-TF: masked.store 21entry: 22 %cmp5 = icmp sgt i32 %N, 0 23 br i1 %cmp5, label %while.body.preheader, label %while.end 24 25while.body.preheader: 26 br label %while.body 27 28while.body: 29 %N.addr.09 = phi i32 [ %dec, %while.body ], [ %N, %while.body.preheader ] 30 %c.addr.08 = phi i8* [ %incdec.ptr4, %while.body ], [ %c, %while.body.preheader ] 31 %b.addr.07 = phi i8* [ %incdec.ptr1, %while.body ], [ %b, %while.body.preheader ] 32 %a.addr.06 = phi i8* [ %incdec.ptr, %while.body ], [ %a, %while.body.preheader ] 33 %dec = add nsw i32 %N.addr.09, -1 34 %incdec.ptr = getelementptr inbounds i8, i8* %a.addr.06, i32 1 35 %0 = load i8, i8* %a.addr.06, align 1 36 %incdec.ptr1 = getelementptr inbounds i8, i8* %b.addr.07, i32 1 37 %1 = load i8, i8* %b.addr.07, align 1 38 %add = add i8 %1, %0 39 %incdec.ptr4 = getelementptr inbounds i8, i8* %c.addr.08, i32 1 40 store i8 %add, i8* %c.addr.08, align 1 41 %cmp = icmp sgt i32 %N.addr.09, 1 42 br i1 %cmp, label %while.body, label %while.end.loopexit 43 44while.end.loopexit: 45 br label %while.end 46 47while.end: 48 ret void 49} 50 51; No loop-guard: we need one for this to be valid. 52; 53define dso_local void @sgt_no_loopguard(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 54; COMMON-LABEL: @sgt_no_loopguard( 55; COMMON: vector.body: 56; 57; FIXME: I think this is currently miscompiled after D77635 58; 59; CHECK-TF: masked.load 60; CHECK-TF: masked.load 61; CHECK-TF: masked.store 62entry: 63 br label %while.body 64 65while.body: 66 %N.addr.09 = phi i32 [ %dec, %while.body ], [ %N, %entry ] 67 %c.addr.08 = phi i8* [ %incdec.ptr4, %while.body ], [ %c, %entry ] 68 %b.addr.07 = phi i8* [ %incdec.ptr1, %while.body ], [ %b, %entry ] 69 %a.addr.06 = phi i8* [ %incdec.ptr, %while.body ], [ %a, %entry ] 70 %dec = add nsw i32 %N.addr.09, -1 71 %incdec.ptr = getelementptr inbounds i8, i8* %a.addr.06, i32 1 72 %0 = load i8, i8* %a.addr.06, align 1 73 %incdec.ptr1 = getelementptr inbounds i8, i8* %b.addr.07, i32 1 74 %1 = load i8, i8* %b.addr.07, align 1 75 %add = add i8 %1, %0 76 %incdec.ptr4 = getelementptr inbounds i8, i8* %c.addr.08, i32 1 77 store i8 %add, i8* %c.addr.08, align 1 78 %cmp = icmp sgt i32 %N.addr.09, 1 79 br i1 %cmp, label %while.body, label %while.end.loopexit 80 81while.end.loopexit: 82 br label %while.end 83 84while.end: 85 ret void 86} 87 88define dso_local void @sgt_extra_use_cmp(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 89; COMMON-LABEL: @sgt_extra_use_cmp( 90; COMMON: vector.body: 91; CHECK-TF: masked.load 92; CHECK-TF: masked.load 93; CHECK-TF: masked.store 94entry: 95 br label %while.body 96 97while.body: 98 %N.addr.09 = phi i32 [ %dec, %while.body ], [ %N, %entry ] 99 %c.addr.08 = phi i8* [ %incdec.ptr4, %while.body ], [ %c, %entry ] 100 %b.addr.07 = phi i8* [ %incdec.ptr1, %while.body ], [ %b, %entry ] 101 %a.addr.06 = phi i8* [ %incdec.ptr, %while.body ], [ %a, %entry ] 102 %dec = add nsw i32 %N.addr.09, -1 103 %incdec.ptr = getelementptr inbounds i8, i8* %a.addr.06, i32 1 104 %0 = load i8, i8* %a.addr.06, align 1 105 %incdec.ptr1 = getelementptr inbounds i8, i8* %b.addr.07, i32 1 106 %1 = load i8, i8* %b.addr.07, align 1 107 %add = add i8 %1, %0 108 %incdec.ptr4 = getelementptr inbounds i8, i8* %c.addr.08, i32 1 109 store i8 %add, i8* %c.addr.08, align 1 110 %cmp = icmp sgt i32 %N.addr.09, 1 111 %select = select i1 %cmp, i8 %0, i8 %1 112 br i1 %cmp, label %while.body, label %while.end.loopexit 113 114while.end.loopexit: 115 br label %while.end 116 117while.end: 118 ret void 119} 120 121define dso_local void @sgt_const_tripcount(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 122; COMMON-LABEL: @sgt_const_tripcount( 123; COMMON: vector.body: 124; CHECK-TF: masked.load 125; CHECK-TF: masked.load 126; CHECK-TF: masked.store 127entry: 128 %cmp5 = icmp sgt i32 %N, 0 129 br i1 %cmp5, label %while.body.preheader, label %while.end 130 131while.body.preheader: 132 br label %while.body 133 134while.body: 135 %N.addr.09 = phi i32 [ %dec, %while.body ], [ 2049, %while.body.preheader ] 136 %c.addr.08 = phi i8* [ %incdec.ptr4, %while.body ], [ %c, %while.body.preheader ] 137 %b.addr.07 = phi i8* [ %incdec.ptr1, %while.body ], [ %b, %while.body.preheader ] 138 %a.addr.06 = phi i8* [ %incdec.ptr, %while.body ], [ %a, %while.body.preheader ] 139 %dec = add nsw i32 %N.addr.09, -1 140 %incdec.ptr = getelementptr inbounds i8, i8* %a.addr.06, i32 1 141 %0 = load i8, i8* %a.addr.06, align 1 142 %incdec.ptr1 = getelementptr inbounds i8, i8* %b.addr.07, i32 1 143 %1 = load i8, i8* %b.addr.07, align 1 144 %add = add i8 %1, %0 145 %incdec.ptr4 = getelementptr inbounds i8, i8* %c.addr.08, i32 1 146 store i8 %add, i8* %c.addr.08, align 1 147 %cmp = icmp sgt i32 %N.addr.09, 1 148 br i1 %cmp, label %while.body, label %while.end.loopexit 149 150while.end.loopexit: 151 br label %while.end 152 153while.end: 154 ret void 155} 156 157define dso_local void @sgt_no_guard_0_startval(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 158; COMMON-LABEL: @sgt_no_guard_0_startval( 159; COMMON-NOT: vector.body: 160entry: 161 br label %while.body 162 163while.body: 164 %N.addr.09 = phi i32 [ %dec, %while.body ], [ 0, %entry ] 165 %c.addr.08 = phi i8* [ %incdec.ptr4, %while.body ], [ %c, %entry ] 166 %b.addr.07 = phi i8* [ %incdec.ptr1, %while.body ], [ %b, %entry ] 167 %a.addr.06 = phi i8* [ %incdec.ptr, %while.body ], [ %a, %entry] 168 %dec = add nsw i32 %N.addr.09, -1 169 %incdec.ptr = getelementptr inbounds i8, i8* %a.addr.06, i32 1 170 %0 = load i8, i8* %a.addr.06, align 1 171 %incdec.ptr1 = getelementptr inbounds i8, i8* %b.addr.07, i32 1 172 %1 = load i8, i8* %b.addr.07, align 1 173 %add = add i8 %1, %0 174 %incdec.ptr4 = getelementptr inbounds i8, i8* %c.addr.08, i32 1 175 store i8 %add, i8* %c.addr.08, align 1 176 %cmp = icmp sgt i32 %N.addr.09, 1 177 br i1 %cmp, label %while.body, label %while.end.loopexit 178 179while.end.loopexit: 180 br label %while.end 181 182while.end: 183 ret void 184} 185 186define dso_local void @sgt_step_minus_two(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 187; COMMON-LABEL: @sgt_step_minus_two( 188; COMMON: vector.body: 189; CHECK-TF: masked.load 190; CHECK-TF: masked.load 191; CHECK-TF: masked.store 192entry: 193 %cmp5 = icmp sgt i32 %N, 0 194 br i1 %cmp5, label %while.body.preheader, label %while.end 195 196while.body.preheader: 197 br label %while.body 198 199while.body: 200 %N.addr.09 = phi i32 [ %dec, %while.body ], [ %N, %while.body.preheader ] 201 %c.addr.08 = phi i8* [ %incdec.ptr4, %while.body ], [ %c, %while.body.preheader ] 202 %b.addr.07 = phi i8* [ %incdec.ptr1, %while.body ], [ %b, %while.body.preheader ] 203 %a.addr.06 = phi i8* [ %incdec.ptr, %while.body ], [ %a, %while.body.preheader ] 204 %dec = add nsw i32 %N.addr.09, -2 205 %incdec.ptr = getelementptr inbounds i8, i8* %a.addr.06, i32 1 206 %0 = load i8, i8* %a.addr.06, align 1 207 %incdec.ptr1 = getelementptr inbounds i8, i8* %b.addr.07, i32 1 208 %1 = load i8, i8* %b.addr.07, align 1 209 %add = add i8 %1, %0 210 %incdec.ptr4 = getelementptr inbounds i8, i8* %c.addr.08, i32 1 211 store i8 %add, i8* %c.addr.08, align 1 212 %cmp = icmp sgt i32 %N.addr.09, 1 213 br i1 %cmp, label %while.body, label %while.end.loopexit 214 215while.end.loopexit: 216 br label %while.end 217 218while.end: 219 ret void 220} 221 222define dso_local void @sgt_step_not_constant(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N, i32 %S) local_unnamed_addr #0 { 223; COMMON-LABEL: @sgt_step_not_constant( 224; COMMON-NOT: vector.body: 225entry: 226 %cmp5 = icmp sgt i32 %N, 0 227 br i1 %cmp5, label %while.body.preheader, label %while.end 228 229while.body.preheader: 230 br label %while.body 231 232while.body: 233 %N.addr.09 = phi i32 [ %dec, %while.body ], [ %N, %while.body.preheader ] 234 %c.addr.08 = phi i8* [ %incdec.ptr4, %while.body ], [ %c, %while.body.preheader ] 235 %b.addr.07 = phi i8* [ %incdec.ptr1, %while.body ], [ %b, %while.body.preheader ] 236 %a.addr.06 = phi i8* [ %incdec.ptr, %while.body ], [ %a, %while.body.preheader ] 237 %dec = add nsw i32 %N.addr.09, %S 238 %incdec.ptr = getelementptr inbounds i8, i8* %a.addr.06, i32 1 239 %0 = load i8, i8* %a.addr.06, align 1 240 %incdec.ptr1 = getelementptr inbounds i8, i8* %b.addr.07, i32 1 241 %1 = load i8, i8* %b.addr.07, align 1 242 %add = add i8 %1, %0 243 %incdec.ptr4 = getelementptr inbounds i8, i8* %c.addr.08, i32 1 244 store i8 %add, i8* %c.addr.08, align 1 245 %cmp = icmp sgt i32 %N.addr.09, 1 246 br i1 %cmp, label %while.body, label %while.end.loopexit 247 248while.end.loopexit: 249 br label %while.end 250 251while.end: 252 ret void 253} 254 255define dso_local void @icmp_eq(i8* noalias nocapture readonly %A, i8* noalias nocapture readonly %B, i8* noalias nocapture %C, i32 %N) #0 { 256; COMMON-LABEL: @icmp_eq 257; COMMON: vector.body: 258; TODO 259entry: 260 %cmp6 = icmp eq i32 %N, 0 261 br i1 %cmp6, label %while.end, label %while.body.preheader 262 263while.body.preheader: 264 br label %while.body 265 266while.body: 267 %N.addr.010 = phi i32 [ %dec, %while.body ], [ %N, %while.body.preheader ] 268 %C.addr.09 = phi i8* [ %incdec.ptr4, %while.body ], [ %C, %while.body.preheader ] 269 %B.addr.08 = phi i8* [ %incdec.ptr1, %while.body ], [ %B, %while.body.preheader ] 270 %A.addr.07 = phi i8* [ %incdec.ptr, %while.body ], [ %A, %while.body.preheader ] 271 %incdec.ptr = getelementptr inbounds i8, i8* %A.addr.07, i32 1 272 %0 = load i8, i8* %A.addr.07, align 1 273 %incdec.ptr1 = getelementptr inbounds i8, i8* %B.addr.08, i32 1 274 %1 = load i8, i8* %B.addr.08, align 1 275 %add = add i8 %1, %0 276 %incdec.ptr4 = getelementptr inbounds i8, i8* %C.addr.09, i32 1 277 store i8 %add, i8* %C.addr.09, align 1 278 %dec = add i32 %N.addr.010, -1 279 %cmp = icmp eq i32 %dec, 0 280 br i1 %cmp, label %while.end.loopexit, label %while.body 281 282while.end.loopexit: 283 br label %while.end 284 285while.end: 286 ret void 287} 288 289; This IR corresponds to this type of C-code: 290; 291; void f(char *a, char *b, char * __restrict c, int N) { 292; #pragma clang loop vectorize_width(16) 293; for (int i = N; i>0; i--) 294; c[i] = a[i] + b[i]; 295; } 296; 297define dso_local void @sgt_for_loop(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 298; COMMON-LABEL: @sgt_for_loop( 299; COMMON : vector.body: 300; CHECK-PREFER: masked.load 301; CHECK-PREFER: masked.load 302; CHECK-PREFER: masked.store 303; 304; TODO: if tail-predication is requested, tail-folding isn't triggered because 305; the profitability check returns "Different strides found, can't tail-predicate", 306; investigate this. 307; 308; CHECK-ENABLE-TP-NOT: masked.load 309; CHECK-ENABLE-TP-NOT: masked.load 310; CHECK-ENABLE-TP-NOT: masked.store 311; 312entry: 313 %cmp5 = icmp sgt i32 %N, 0 314 br i1 %cmp5, label %for.body.preheader, label %for.end 315 316for.body.preheader: 317 br label %for.body 318 319for.body: 320 %i.011 = phi i32 [ %dec, %for.body ], [ %N, %for.body.preheader ] 321 %arrayidx = getelementptr inbounds i8, i8* %a, i32 %i.011 322 %0 = load i8, i8* %arrayidx, align 1 323 %arrayidx1 = getelementptr inbounds i8, i8* %b, i32 %i.011 324 %1 = load i8, i8* %arrayidx1, align 1 325 %add = add i8 %1, %0 326 %arrayidx4 = getelementptr inbounds i8, i8* %c, i32 %i.011 327 store i8 %add, i8* %arrayidx4, align 1 328 %dec = add nsw i32 %i.011, -1 329 %cmp = icmp sgt i32 %i.011, 1 330 br i1 %cmp, label %for.body, label %for.end, !llvm.loop !1 331 332for.end: 333 ret void 334} 335 336define dso_local void @sgt_for_loop_i64(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 337; COMMON-LABEL: @sgt_for_loop_i64( 338; COMMON: vector.body: 339; 340; CHECK-PREFER: masked.load 341; CHECK-PREFER: masked.load 342; CHECK-PREFER: masked.store 343; 344; With -disable-mve-tail-predication=false, the target hook returns 345; "preferPredicateOverEpilogue: hardware-loop is not profitable." 346; so here we don't expect the tail-folding. TODO: look into this. 347; 348; CHECK-ENABLE-TP-NOT: masked.load 349; CHECK-ENABLE-TP-NOT: masked.load 350; CHECK-ENABLE-TP-NOT: masked.store 351; 352entry: 353 %cmp14 = icmp sgt i32 %N, 0 354 br i1 %cmp14, label %for.body.preheader, label %for.cond.cleanup 355 356for.body.preheader: 357 %conv16 = zext i32 %N to i64 358 br label %for.body 359 360for.cond.cleanup.loopexit: 361 br label %for.cond.cleanup 362 363for.cond.cleanup: 364 ret void 365 366for.body: 367 %i.015 = phi i64 [ %dec, %for.body ], [ %conv16, %for.body.preheader ] 368 %idxprom = trunc i64 %i.015 to i32 369 %arrayidx = getelementptr inbounds i8, i8* %a, i32 %idxprom 370 %0 = load i8, i8* %arrayidx, align 1 371 %arrayidx4 = getelementptr inbounds i8, i8* %b, i32 %idxprom 372 %1 = load i8, i8* %arrayidx4, align 1 373 %add = add i8 %1, %0 374 %arrayidx8 = getelementptr inbounds i8, i8* %c, i32 %idxprom 375 store i8 %add, i8* %arrayidx8, align 1 376 %dec = add nsw i64 %i.015, -1 377 %cmp = icmp sgt i64 %i.015, 1 378 br i1 %cmp, label %for.body, label %for.cond.cleanup.loopexit, !llvm.loop !1 379} 380 381; This IR corresponds to this nested-loop: 382; 383; for (int i = 0; i<N; i++) 384; for (int j = i+1; j>0; j--) 385; c[j] = a[j] + b[j]; 386; 387; while the inner-loop looks similar to previous examples, we can't 388; transform this because the inner loop because isGuarded returns 389; false for the inner-loop. 390; 391define dso_local void @sgt_nested_loop(i8* noalias nocapture readonly %a, i8* noalias nocapture readonly %b, i8* noalias nocapture %c, i32 %N) local_unnamed_addr #0 { 392; COMMON-LABEL: @sgt_nested_loop( 393; DEFAULT-NOT: vector.body: 394; CHECK-TF-NOT: masked.load 395; CHECK-TF-NOT: masked.load 396; CHECK-TF-NOT: masked.store 397; COMMON: } 398; 399entry: 400 %cmp21 = icmp sgt i32 %N, 0 401 br i1 %cmp21, label %for.body.preheader, label %for.cond.cleanup 402 403for.body.preheader: 404 br label %for.body 405 406for.cond.loopexit: 407 %exitcond = icmp eq i32 %add, %N 408 br i1 %exitcond, label %for.cond.cleanup.loopexit, label %for.body 409 410for.cond.cleanup.loopexit: 411 br label %for.cond.cleanup 412 413for.cond.cleanup: 414 ret void 415 416for.body: 417 %i.022 = phi i32 [ %add, %for.cond.loopexit ], [ 0, %for.body.preheader ] 418 %add = add nuw nsw i32 %i.022, 1 419 br label %for.body4 420 421for.body4: ; preds = %for.body, %for.body4 422 %j.020 = phi i32 [ %add, %for.body ], [ %dec, %for.body4 ] 423 %arrayidx = getelementptr inbounds i8, i8* %a, i32 %j.020 424 %0 = load i8, i8* %arrayidx, align 1 425 %arrayidx5 = getelementptr inbounds i8, i8* %b, i32 %j.020 426 %1 = load i8, i8* %arrayidx5, align 1 427 %add7 = add i8 %1, %0 428 %arrayidx9 = getelementptr inbounds i8, i8* %c, i32 %j.020 429 store i8 %add7, i8* %arrayidx9, align 1 430 %dec = add nsw i32 %j.020, -1 431 %cmp2 = icmp sgt i32 %j.020, 1 432 br i1 %cmp2, label %for.body4, label %for.cond.loopexit 433} 434 435attributes #0 = { nofree norecurse nounwind "target-features"="+armv8.1-m.main,+mve.fp" } 436 437!1 = distinct !{!1, !2} 438!2 = !{!"llvm.loop.vectorize.width", i32 16} 439