1; NOTE: Assertions have been autogenerated by utils/update_test_checks.py 2; RUN: opt -O2 -expand-reductions -mattr=avx -S < %s | FileCheck %s 3 4; Test if SLP vector reduction patterns are recognized 5; and optionally converted to reduction intrinsics and 6; back to raw IR. 7 8target triple = "x86_64--" 9target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" 10 11define i32 @add_v4i32(i32* %p) #0 { 12; CHECK-LABEL: @add_v4i32( 13; CHECK-NEXT: entry: 14; CHECK-NEXT: [[TMP0:%.*]] = bitcast i32* [[P:%.*]] to <4 x i32>* 15; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, <4 x i32>* [[TMP0]], align 4, [[TBAA0:!tbaa !.*]] 16; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef> 17; CHECK-NEXT: [[BIN_RDX:%.*]] = add <4 x i32> [[TMP1]], [[RDX_SHUF]] 18; CHECK-NEXT: [[RDX_SHUF3:%.*]] = shufflevector <4 x i32> [[BIN_RDX]], <4 x i32> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef> 19; CHECK-NEXT: [[BIN_RDX4:%.*]] = add <4 x i32> [[BIN_RDX]], [[RDX_SHUF3]] 20; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x i32> [[BIN_RDX4]], i32 0 21; CHECK-NEXT: ret i32 [[TMP2]] 22; 23entry: 24 br label %for.cond 25 26for.cond: 27 %r.0 = phi i32 [ 0, %entry ], [ %add, %for.inc ] 28 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 29 %cmp = icmp slt i32 %i.0, 4 30 br i1 %cmp, label %for.body, label %for.cond.cleanup 31 32for.cond.cleanup: 33 br label %for.end 34 35for.body: 36 %idxprom = sext i32 %i.0 to i64 37 %arrayidx = getelementptr inbounds i32, i32* %p, i64 %idxprom 38 %0 = load i32, i32* %arrayidx, align 4, !tbaa !3 39 %add = add nsw i32 %r.0, %0 40 br label %for.inc 41 42for.inc: 43 %inc = add nsw i32 %i.0, 1 44 br label %for.cond 45 46for.end: 47 ret i32 %r.0 48} 49 50define signext i16 @mul_v8i16(i16* %p) #0 { 51; CHECK-LABEL: @mul_v8i16( 52; CHECK-NEXT: entry: 53; CHECK-NEXT: [[TMP0:%.*]] = bitcast i16* [[P:%.*]] to <8 x i16>* 54; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, <8 x i16>* [[TMP0]], align 2, [[TBAA4:!tbaa !.*]] 55; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 undef, i32 undef, i32 undef, i32 undef> 56; CHECK-NEXT: [[BIN_RDX:%.*]] = mul <8 x i16> [[TMP1]], [[RDX_SHUF]] 57; CHECK-NEXT: [[RDX_SHUF3:%.*]] = shufflevector <8 x i16> [[BIN_RDX]], <8 x i16> poison, <8 x i32> <i32 2, i32 3, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 58; CHECK-NEXT: [[BIN_RDX4:%.*]] = mul <8 x i16> [[BIN_RDX]], [[RDX_SHUF3]] 59; CHECK-NEXT: [[RDX_SHUF5:%.*]] = shufflevector <8 x i16> [[BIN_RDX4]], <8 x i16> poison, <8 x i32> <i32 1, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 60; CHECK-NEXT: [[BIN_RDX6:%.*]] = mul <8 x i16> [[BIN_RDX4]], [[RDX_SHUF5]] 61; CHECK-NEXT: [[TMP2:%.*]] = extractelement <8 x i16> [[BIN_RDX6]], i32 0 62; CHECK-NEXT: ret i16 [[TMP2]] 63; 64entry: 65 br label %for.cond 66 67for.cond: 68 %r.0 = phi i16 [ 1, %entry ], [ %conv2, %for.inc ] 69 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 70 %cmp = icmp slt i32 %i.0, 8 71 br i1 %cmp, label %for.body, label %for.cond.cleanup 72 73for.cond.cleanup: 74 br label %for.end 75 76for.body: 77 %idxprom = sext i32 %i.0 to i64 78 %arrayidx = getelementptr inbounds i16, i16* %p, i64 %idxprom 79 %0 = load i16, i16* %arrayidx, align 2, !tbaa !7 80 %conv = sext i16 %0 to i32 81 %conv1 = sext i16 %r.0 to i32 82 %mul = mul nsw i32 %conv1, %conv 83 %conv2 = trunc i32 %mul to i16 84 br label %for.inc 85 86for.inc: 87 %inc = add nsw i32 %i.0, 1 88 br label %for.cond 89 90for.end: 91 ret i16 %r.0 92} 93 94define signext i8 @or_v16i8(i8* %p) #0 { 95; CHECK-LABEL: @or_v16i8( 96; CHECK-NEXT: entry: 97; CHECK-NEXT: [[TMP0:%.*]] = bitcast i8* [[P:%.*]] to <16 x i8>* 98; CHECK-NEXT: [[TMP1:%.*]] = load <16 x i8>, <16 x i8>* [[TMP0]], align 1, [[TBAA6:!tbaa !.*]] 99; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 100; CHECK-NEXT: [[BIN_RDX:%.*]] = or <16 x i8> [[TMP1]], [[RDX_SHUF]] 101; CHECK-NEXT: [[RDX_SHUF4:%.*]] = shufflevector <16 x i8> [[BIN_RDX]], <16 x i8> poison, <16 x i32> <i32 4, i32 5, i32 6, i32 7, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 102; CHECK-NEXT: [[BIN_RDX5:%.*]] = or <16 x i8> [[BIN_RDX]], [[RDX_SHUF4]] 103; CHECK-NEXT: [[RDX_SHUF6:%.*]] = shufflevector <16 x i8> [[BIN_RDX5]], <16 x i8> poison, <16 x i32> <i32 2, i32 3, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 104; CHECK-NEXT: [[BIN_RDX7:%.*]] = or <16 x i8> [[BIN_RDX5]], [[RDX_SHUF6]] 105; CHECK-NEXT: [[RDX_SHUF8:%.*]] = shufflevector <16 x i8> [[BIN_RDX7]], <16 x i8> poison, <16 x i32> <i32 1, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 106; CHECK-NEXT: [[BIN_RDX9:%.*]] = or <16 x i8> [[BIN_RDX7]], [[RDX_SHUF8]] 107; CHECK-NEXT: [[TMP2:%.*]] = extractelement <16 x i8> [[BIN_RDX9]], i32 0 108; CHECK-NEXT: ret i8 [[TMP2]] 109; 110entry: 111 br label %for.cond 112 113for.cond: 114 %r.0 = phi i8 [ 0, %entry ], [ %conv2, %for.inc ] 115 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 116 %cmp = icmp slt i32 %i.0, 16 117 br i1 %cmp, label %for.body, label %for.cond.cleanup 118 119for.cond.cleanup: 120 br label %for.end 121 122for.body: 123 %idxprom = sext i32 %i.0 to i64 124 %arrayidx = getelementptr inbounds i8, i8* %p, i64 %idxprom 125 %0 = load i8, i8* %arrayidx, align 1, !tbaa !9 126 %conv = sext i8 %0 to i32 127 %conv1 = sext i8 %r.0 to i32 128 %or = or i32 %conv1, %conv 129 %conv2 = trunc i32 %or to i8 130 br label %for.inc 131 132for.inc: 133 %inc = add nsw i32 %i.0, 1 134 br label %for.cond 135 136for.end: 137 ret i8 %r.0 138} 139 140define i32 @smin_v4i32(i32* %p) #0 { 141; CHECK-LABEL: @smin_v4i32( 142; CHECK-NEXT: entry: 143; CHECK-NEXT: [[TMP0:%.*]] = bitcast i32* [[P:%.*]] to <4 x i32>* 144; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, <4 x i32>* [[TMP0]], align 4, [[TBAA0]] 145; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef> 146; CHECK-NEXT: [[RDX_MINMAX_CMP:%.*]] = icmp slt <4 x i32> [[TMP1]], [[RDX_SHUF]] 147; CHECK-NEXT: [[RDX_MINMAX_SELECT:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP]], <4 x i32> [[TMP1]], <4 x i32> [[RDX_SHUF]] 148; CHECK-NEXT: [[RDX_SHUF3:%.*]] = shufflevector <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef> 149; CHECK-NEXT: [[RDX_MINMAX_CMP4:%.*]] = icmp slt <4 x i32> [[RDX_MINMAX_SELECT]], [[RDX_SHUF3]] 150; CHECK-NEXT: [[RDX_MINMAX_SELECT5:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP4]], <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> [[RDX_SHUF3]] 151; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x i32> [[RDX_MINMAX_SELECT5]], i32 0 152; CHECK-NEXT: ret i32 [[TMP2]] 153; 154entry: 155 br label %for.cond 156 157for.cond: 158 %r.0 = phi i32 [ 2147483647, %entry ], [ %cond, %for.inc ] 159 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 160 %cmp = icmp slt i32 %i.0, 4 161 br i1 %cmp, label %for.body, label %for.cond.cleanup 162 163for.cond.cleanup: 164 br label %for.end 165 166for.body: 167 %idxprom = sext i32 %i.0 to i64 168 %arrayidx = getelementptr inbounds i32, i32* %p, i64 %idxprom 169 %0 = load i32, i32* %arrayidx, align 4, !tbaa !3 170 %cmp1 = icmp slt i32 %0, %r.0 171 br i1 %cmp1, label %cond.true, label %cond.false 172 173cond.true: 174 %idxprom2 = sext i32 %i.0 to i64 175 %arrayidx3 = getelementptr inbounds i32, i32* %p, i64 %idxprom2 176 %1 = load i32, i32* %arrayidx3, align 4, !tbaa !3 177 br label %cond.end 178 179cond.false: 180 br label %cond.end 181 182cond.end: 183 %cond = phi i32 [ %1, %cond.true ], [ %r.0, %cond.false ] 184 br label %for.inc 185 186for.inc: 187 %inc = add nsw i32 %i.0, 1 188 br label %for.cond 189 190for.end: 191 ret i32 %r.0 192} 193 194define i32 @umax_v4i32(i32* %p) #0 { 195; CHECK-LABEL: @umax_v4i32( 196; CHECK-NEXT: entry: 197; CHECK-NEXT: [[TMP0:%.*]] = bitcast i32* [[P:%.*]] to <4 x i32>* 198; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, <4 x i32>* [[TMP0]], align 4, [[TBAA0]] 199; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef> 200; CHECK-NEXT: [[RDX_MINMAX_CMP:%.*]] = icmp ugt <4 x i32> [[TMP1]], [[RDX_SHUF]] 201; CHECK-NEXT: [[RDX_MINMAX_SELECT:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP]], <4 x i32> [[TMP1]], <4 x i32> [[RDX_SHUF]] 202; CHECK-NEXT: [[RDX_SHUF3:%.*]] = shufflevector <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef> 203; CHECK-NEXT: [[RDX_MINMAX_CMP4:%.*]] = icmp ugt <4 x i32> [[RDX_MINMAX_SELECT]], [[RDX_SHUF3]] 204; CHECK-NEXT: [[RDX_MINMAX_SELECT5:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP4]], <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> [[RDX_SHUF3]] 205; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x i32> [[RDX_MINMAX_SELECT5]], i32 0 206; CHECK-NEXT: ret i32 [[TMP2]] 207; 208entry: 209 br label %for.cond 210 211for.cond: 212 %r.0 = phi i32 [ 0, %entry ], [ %cond, %for.inc ] 213 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 214 %cmp = icmp slt i32 %i.0, 4 215 br i1 %cmp, label %for.body, label %for.cond.cleanup 216 217for.cond.cleanup: 218 br label %for.end 219 220for.body: 221 %idxprom = sext i32 %i.0 to i64 222 %arrayidx = getelementptr inbounds i32, i32* %p, i64 %idxprom 223 %0 = load i32, i32* %arrayidx, align 4, !tbaa !3 224 %cmp1 = icmp ugt i32 %0, %r.0 225 br i1 %cmp1, label %cond.true, label %cond.false 226 227cond.true: 228 %idxprom2 = sext i32 %i.0 to i64 229 %arrayidx3 = getelementptr inbounds i32, i32* %p, i64 %idxprom2 230 %1 = load i32, i32* %arrayidx3, align 4, !tbaa !3 231 br label %cond.end 232 233cond.false: 234 br label %cond.end 235 236cond.end: 237 %cond = phi i32 [ %1, %cond.true ], [ %r.0, %cond.false ] 238 br label %for.inc 239 240for.inc: 241 %inc = add nsw i32 %i.0, 1 242 br label %for.cond 243 244for.end: 245 ret i32 %r.0 246} 247 248define float @fadd_v4i32(float* %p) #0 { 249; CHECK-LABEL: @fadd_v4i32( 250; CHECK-NEXT: entry: 251; CHECK-NEXT: [[TMP0:%.*]] = bitcast float* [[P:%.*]] to <4 x float>* 252; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, <4 x float>* [[TMP0]], align 4, [[TBAA7:!tbaa !.*]] 253; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef> 254; CHECK-NEXT: [[BIN_RDX:%.*]] = fadd fast <4 x float> [[TMP1]], [[RDX_SHUF]] 255; CHECK-NEXT: [[RDX_SHUF3:%.*]] = shufflevector <4 x float> [[BIN_RDX]], <4 x float> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef> 256; CHECK-NEXT: [[BIN_RDX4:%.*]] = fadd fast <4 x float> [[BIN_RDX]], [[RDX_SHUF3]] 257; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[BIN_RDX4]], i32 0 258; CHECK-NEXT: [[BIN_RDX5:%.*]] = fadd fast float -0.000000e+00, [[TMP2]] 259; CHECK-NEXT: [[OP_EXTRA:%.*]] = fadd fast float [[BIN_RDX5]], 4.200000e+01 260; CHECK-NEXT: ret float [[OP_EXTRA]] 261; 262entry: 263 br label %for.cond 264 265for.cond: 266 %r.0 = phi float [ 4.200000e+01, %entry ], [ %add, %for.inc ] 267 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 268 %cmp = icmp slt i32 %i.0, 4 269 br i1 %cmp, label %for.body, label %for.cond.cleanup 270 271for.cond.cleanup: 272 br label %for.end 273 274for.body: 275 %idxprom = sext i32 %i.0 to i64 276 %arrayidx = getelementptr inbounds float, float* %p, i64 %idxprom 277 %0 = load float, float* %arrayidx, align 4, !tbaa !10 278 %add = fadd fast float %r.0, %0 279 br label %for.inc 280 281for.inc: 282 %inc = add nsw i32 %i.0, 1 283 br label %for.cond 284 285for.end: 286 ret float %r.0 287} 288 289define float @fmul_v4i32(float* %p) #0 { 290; CHECK-LABEL: @fmul_v4i32( 291; CHECK-NEXT: entry: 292; CHECK-NEXT: [[TMP0:%.*]] = bitcast float* [[P:%.*]] to <4 x float>* 293; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, <4 x float>* [[TMP0]], align 4, [[TBAA7]] 294; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef> 295; CHECK-NEXT: [[BIN_RDX:%.*]] = fmul fast <4 x float> [[TMP1]], [[RDX_SHUF]] 296; CHECK-NEXT: [[RDX_SHUF3:%.*]] = shufflevector <4 x float> [[BIN_RDX]], <4 x float> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef> 297; CHECK-NEXT: [[BIN_RDX4:%.*]] = fmul fast <4 x float> [[BIN_RDX]], [[RDX_SHUF3]] 298; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[BIN_RDX4]], i32 0 299; CHECK-NEXT: [[BIN_RDX5:%.*]] = fmul fast float 1.000000e+00, [[TMP2]] 300; CHECK-NEXT: [[OP_EXTRA:%.*]] = fmul fast float [[BIN_RDX5]], 4.200000e+01 301; CHECK-NEXT: ret float [[OP_EXTRA]] 302; 303entry: 304 br label %for.cond 305 306for.cond: 307 %r.0 = phi float [ 4.200000e+01, %entry ], [ %mul, %for.inc ] 308 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 309 %cmp = icmp slt i32 %i.0, 4 310 br i1 %cmp, label %for.body, label %for.cond.cleanup 311 312for.cond.cleanup: 313 br label %for.end 314 315for.body: 316 %idxprom = sext i32 %i.0 to i64 317 %arrayidx = getelementptr inbounds float, float* %p, i64 %idxprom 318 %0 = load float, float* %arrayidx, align 4, !tbaa !10 319 %mul = fmul fast float %r.0, %0 320 br label %for.inc 321 322for.inc: 323 %inc = add nsw i32 %i.0, 1 324 br label %for.cond 325 326for.end: 327 ret float %r.0 328} 329 330define float @fmin_v4f32(float* %p) #0 { 331; CHECK-LABEL: @fmin_v4f32( 332; CHECK-NEXT: entry: 333; CHECK-NEXT: [[TMP0:%.*]] = bitcast float* [[P:%.*]] to <4 x float>* 334; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, <4 x float>* [[TMP0]], align 4, [[TBAA7]] 335; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef> 336; CHECK-NEXT: [[RDX_MINMAX_CMP:%.*]] = fcmp fast olt <4 x float> [[TMP1]], [[RDX_SHUF]] 337; CHECK-NEXT: [[RDX_MINMAX_SELECT:%.*]] = select fast <4 x i1> [[RDX_MINMAX_CMP]], <4 x float> [[TMP1]], <4 x float> [[RDX_SHUF]] 338; CHECK-NEXT: [[RDX_SHUF3:%.*]] = shufflevector <4 x float> [[RDX_MINMAX_SELECT]], <4 x float> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef> 339; CHECK-NEXT: [[RDX_MINMAX_CMP4:%.*]] = fcmp fast olt <4 x float> [[RDX_MINMAX_SELECT]], [[RDX_SHUF3]] 340; CHECK-NEXT: [[RDX_MINMAX_SELECT5:%.*]] = select fast <4 x i1> [[RDX_MINMAX_CMP4]], <4 x float> [[RDX_MINMAX_SELECT]], <4 x float> [[RDX_SHUF3]] 341; CHECK-NEXT: [[TMP2:%.*]] = extractelement <4 x float> [[RDX_MINMAX_SELECT5]], i32 0 342; CHECK-NEXT: ret float [[TMP2]] 343; 344entry: 345 br label %for.cond 346 347for.cond: 348 %r.0 = phi float [ 0x47EFFFFFE0000000, %entry ], [ %cond, %for.inc ] 349 %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ] 350 %cmp = icmp slt i32 %i.0, 4 351 br i1 %cmp, label %for.body, label %for.cond.cleanup 352 353for.cond.cleanup: 354 br label %for.end 355 356for.body: 357 %idxprom = sext i32 %i.0 to i64 358 %arrayidx = getelementptr inbounds float, float* %p, i64 %idxprom 359 %0 = load float, float* %arrayidx, align 4, !tbaa !10 360 %cmp1 = fcmp fast olt float %0, %r.0 361 br i1 %cmp1, label %cond.true, label %cond.false 362 363cond.true: 364 %idxprom2 = sext i32 %i.0 to i64 365 %arrayidx3 = getelementptr inbounds float, float* %p, i64 %idxprom2 366 %1 = load float, float* %arrayidx3, align 4, !tbaa !10 367 br label %cond.end 368 369cond.false: 370 br label %cond.end 371 372cond.end: 373 %cond = phi fast float [ %1, %cond.true ], [ %r.0, %cond.false ] 374 br label %for.inc 375 376for.inc: 377 %inc = add nsw i32 %i.0, 1 378 br label %for.cond 379 380for.end: 381 ret float %r.0 382} 383 384define available_externally float @max(float %a, float %b) { 385entry: 386 %a.addr = alloca float, align 4 387 %b.addr = alloca float, align 4 388 store float %a, float* %a.addr, align 4 389 store float %b, float* %b.addr, align 4 390 %0 = load float, float* %a.addr, align 4 391 %1 = load float, float* %b.addr, align 4 392 %cmp = fcmp nnan ninf nsz ogt float %0, %1 393 br i1 %cmp, label %cond.true, label %cond.false 394 395cond.true: ; preds = %entry 396 %2 = load float, float* %a.addr, align 4 397 br label %cond.end 398 399cond.false: ; preds = %entry 400 %3 = load float, float* %b.addr, align 4 401 br label %cond.end 402 403cond.end: ; preds = %cond.false, %cond.true 404 %cond = phi nnan ninf nsz float [ %2, %cond.true ], [ %3, %cond.false ] 405 ret float %cond 406} 407 408; PR23116 409 410define float @findMax(<8 x float>* byval(<8 x float>) align 16 %0) { 411; CHECK-LABEL: @findMax( 412; CHECK-NEXT: entry: 413; CHECK-NEXT: [[V:%.*]] = load <8 x float>, <8 x float>* [[TMP0:%.*]], align 16, [[TBAA0]] 414; CHECK-NEXT: [[RDX_SHUF:%.*]] = shufflevector <8 x float> [[V]], <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 undef, i32 undef, i32 undef, i32 undef> 415; CHECK-NEXT: [[RDX_MINMAX_CMP:%.*]] = fcmp nnan ninf nsz ogt <8 x float> [[V]], [[RDX_SHUF]] 416; CHECK-NEXT: [[RDX_MINMAX_SELECT:%.*]] = select nnan ninf nsz <8 x i1> [[RDX_MINMAX_CMP]], <8 x float> [[V]], <8 x float> [[RDX_SHUF]] 417; CHECK-NEXT: [[RDX_SHUF8:%.*]] = shufflevector <8 x float> [[RDX_MINMAX_SELECT]], <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 418; CHECK-NEXT: [[RDX_MINMAX_CMP9:%.*]] = fcmp nnan ninf nsz ogt <8 x float> [[RDX_MINMAX_SELECT]], [[RDX_SHUF8]] 419; CHECK-NEXT: [[RDX_MINMAX_SELECT10:%.*]] = select nnan ninf nsz <8 x i1> [[RDX_MINMAX_CMP9]], <8 x float> [[RDX_MINMAX_SELECT]], <8 x float> [[RDX_SHUF8]] 420; CHECK-NEXT: [[RDX_SHUF11:%.*]] = shufflevector <8 x float> [[RDX_MINMAX_SELECT10]], <8 x float> poison, <8 x i32> <i32 1, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef> 421; CHECK-NEXT: [[RDX_MINMAX_CMP12:%.*]] = fcmp nnan ninf nsz ogt <8 x float> [[RDX_MINMAX_SELECT10]], [[RDX_SHUF11]] 422; CHECK-NEXT: [[RDX_MINMAX_SELECT13:%.*]] = select nnan ninf nsz <8 x i1> [[RDX_MINMAX_CMP12]], <8 x float> [[RDX_MINMAX_SELECT10]], <8 x float> [[RDX_SHUF11]] 423; CHECK-NEXT: [[TMP1:%.*]] = extractelement <8 x float> [[RDX_MINMAX_SELECT13]], i32 0 424; CHECK-NEXT: ret float [[TMP1]] 425; 426entry: 427 %v.addr = alloca <8 x float>, align 32 428 %v = load <8 x float>, <8 x float>* %0, align 16, !tbaa !3 429 store <8 x float> %v, <8 x float>* %v.addr, align 32, !tbaa !3 430 %1 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 431 %vecext = extractelement <8 x float> %1, i32 0 432 %2 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 433 %vecext1 = extractelement <8 x float> %2, i32 1 434 %call = call nnan ninf nsz float @max(float %vecext, float %vecext1) 435 %3 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 436 %vecext2 = extractelement <8 x float> %3, i32 2 437 %call3 = call nnan ninf nsz float @max(float %call, float %vecext2) 438 %4 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 439 %vecext4 = extractelement <8 x float> %4, i32 3 440 %call5 = call nnan ninf nsz float @max(float %call3, float %vecext4) 441 %5 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 442 %vecext6 = extractelement <8 x float> %5, i32 4 443 %call7 = call nnan ninf nsz float @max(float %call5, float %vecext6) 444 %6 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 445 %vecext8 = extractelement <8 x float> %6, i32 5 446 %call9 = call nnan ninf nsz float @max(float %call7, float %vecext8) 447 %7 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 448 %vecext10 = extractelement <8 x float> %7, i32 6 449 %call11 = call nnan ninf nsz float @max(float %call9, float %vecext10) 450 %8 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3 451 %vecext12 = extractelement <8 x float> %8, i32 7 452 %call13 = call nnan ninf nsz float @max(float %call11, float %vecext12) 453 ret float %call13 454} 455 456attributes #0 = { nounwind ssp uwtable "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "frame-pointer"="all" "less-precise-fpmad"="false" "min-legal-vector-width"="0" "no-infs-fp-math"="true" "no-jump-tables"="false" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="penryn" "target-features"="+avx,+cx16,+cx8,+fxsr,+mmx,+popcnt,+sahf,+sse,+sse2,+sse3,+sse4.1,+sse4.2,+ssse3,+x87,+xsave" "unsafe-fp-math"="true" "use-soft-float"="false" } 457 458!0 = !{i32 1, !"wchar_size", i32 4} 459!1 = !{i32 7, !"PIC Level", i32 2} 460!2 = !{!"clang version 11.0.0 (https://github.com/llvm/llvm-project.git a9fe69c359de653015c39e413e48630d069abe27)"} 461!3 = !{!4, !4, i64 0} 462!4 = !{!"int", !5, i64 0} 463!5 = !{!"omnipotent char", !6, i64 0} 464!6 = !{!"Simple C/C++ TBAA"} 465!7 = !{!8, !8, i64 0} 466!8 = !{!"short", !5, i64 0} 467!9 = !{!5, !5, i64 0} 468!10 = !{!11, !11, i64 0} 469!11 = !{!"float", !5, i64 0} 470