1; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
2; RUN: opt -O2 -expand-reductions -mattr=avx -S < %s | FileCheck %s
3
4; Test if SLP vector reduction patterns are recognized
5; and optionally converted to reduction intrinsics and
6; back to raw IR.
7
8target triple = "x86_64--"
9target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
10
11define i32 @add_v4i32(i32* %p) #0 {
12; CHECK-LABEL: @add_v4i32(
13; CHECK-NEXT:  entry:
14; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i32* [[P:%.*]] to <4 x i32>*
15; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, <4 x i32>* [[TMP0]], align 4, [[TBAA0:!tbaa !.*]]
16; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef>
17; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[TMP1]], [[RDX_SHUF]]
18; CHECK-NEXT:    [[RDX_SHUF3:%.*]] = shufflevector <4 x i32> [[BIN_RDX]], <4 x i32> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef>
19; CHECK-NEXT:    [[BIN_RDX4:%.*]] = add <4 x i32> [[BIN_RDX]], [[RDX_SHUF3]]
20; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x i32> [[BIN_RDX4]], i32 0
21; CHECK-NEXT:    ret i32 [[TMP2]]
22;
23entry:
24  br label %for.cond
25
26for.cond:
27  %r.0 = phi i32 [ 0, %entry ], [ %add, %for.inc ]
28  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
29  %cmp = icmp slt i32 %i.0, 4
30  br i1 %cmp, label %for.body, label %for.cond.cleanup
31
32for.cond.cleanup:
33  br label %for.end
34
35for.body:
36  %idxprom = sext i32 %i.0 to i64
37  %arrayidx = getelementptr inbounds i32, i32* %p, i64 %idxprom
38  %0 = load i32, i32* %arrayidx, align 4, !tbaa !3
39  %add = add nsw i32 %r.0, %0
40  br label %for.inc
41
42for.inc:
43  %inc = add nsw i32 %i.0, 1
44  br label %for.cond
45
46for.end:
47  ret i32 %r.0
48}
49
50define signext i16 @mul_v8i16(i16* %p) #0 {
51; CHECK-LABEL: @mul_v8i16(
52; CHECK-NEXT:  entry:
53; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i16* [[P:%.*]] to <8 x i16>*
54; CHECK-NEXT:    [[TMP1:%.*]] = load <8 x i16>, <8 x i16>* [[TMP0]], align 2, [[TBAA4:!tbaa !.*]]
55; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 undef, i32 undef, i32 undef, i32 undef>
56; CHECK-NEXT:    [[BIN_RDX:%.*]] = mul <8 x i16> [[TMP1]], [[RDX_SHUF]]
57; CHECK-NEXT:    [[RDX_SHUF3:%.*]] = shufflevector <8 x i16> [[BIN_RDX]], <8 x i16> poison, <8 x i32> <i32 2, i32 3, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
58; CHECK-NEXT:    [[BIN_RDX4:%.*]] = mul <8 x i16> [[BIN_RDX]], [[RDX_SHUF3]]
59; CHECK-NEXT:    [[RDX_SHUF5:%.*]] = shufflevector <8 x i16> [[BIN_RDX4]], <8 x i16> poison, <8 x i32> <i32 1, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
60; CHECK-NEXT:    [[BIN_RDX6:%.*]] = mul <8 x i16> [[BIN_RDX4]], [[RDX_SHUF5]]
61; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <8 x i16> [[BIN_RDX6]], i32 0
62; CHECK-NEXT:    ret i16 [[TMP2]]
63;
64entry:
65  br label %for.cond
66
67for.cond:
68  %r.0 = phi i16 [ 1, %entry ], [ %conv2, %for.inc ]
69  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
70  %cmp = icmp slt i32 %i.0, 8
71  br i1 %cmp, label %for.body, label %for.cond.cleanup
72
73for.cond.cleanup:
74  br label %for.end
75
76for.body:
77  %idxprom = sext i32 %i.0 to i64
78  %arrayidx = getelementptr inbounds i16, i16* %p, i64 %idxprom
79  %0 = load i16, i16* %arrayidx, align 2, !tbaa !7
80  %conv = sext i16 %0 to i32
81  %conv1 = sext i16 %r.0 to i32
82  %mul = mul nsw i32 %conv1, %conv
83  %conv2 = trunc i32 %mul to i16
84  br label %for.inc
85
86for.inc:
87  %inc = add nsw i32 %i.0, 1
88  br label %for.cond
89
90for.end:
91  ret i16 %r.0
92}
93
94define signext i8 @or_v16i8(i8* %p) #0 {
95; CHECK-LABEL: @or_v16i8(
96; CHECK-NEXT:  entry:
97; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i8* [[P:%.*]] to <16 x i8>*
98; CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i8>, <16 x i8>* [[TMP0]], align 1, [[TBAA6:!tbaa !.*]]
99; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
100; CHECK-NEXT:    [[BIN_RDX:%.*]] = or <16 x i8> [[TMP1]], [[RDX_SHUF]]
101; CHECK-NEXT:    [[RDX_SHUF4:%.*]] = shufflevector <16 x i8> [[BIN_RDX]], <16 x i8> poison, <16 x i32> <i32 4, i32 5, i32 6, i32 7, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
102; CHECK-NEXT:    [[BIN_RDX5:%.*]] = or <16 x i8> [[BIN_RDX]], [[RDX_SHUF4]]
103; CHECK-NEXT:    [[RDX_SHUF6:%.*]] = shufflevector <16 x i8> [[BIN_RDX5]], <16 x i8> poison, <16 x i32> <i32 2, i32 3, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
104; CHECK-NEXT:    [[BIN_RDX7:%.*]] = or <16 x i8> [[BIN_RDX5]], [[RDX_SHUF6]]
105; CHECK-NEXT:    [[RDX_SHUF8:%.*]] = shufflevector <16 x i8> [[BIN_RDX7]], <16 x i8> poison, <16 x i32> <i32 1, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
106; CHECK-NEXT:    [[BIN_RDX9:%.*]] = or <16 x i8> [[BIN_RDX7]], [[RDX_SHUF8]]
107; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <16 x i8> [[BIN_RDX9]], i32 0
108; CHECK-NEXT:    ret i8 [[TMP2]]
109;
110entry:
111  br label %for.cond
112
113for.cond:
114  %r.0 = phi i8 [ 0, %entry ], [ %conv2, %for.inc ]
115  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
116  %cmp = icmp slt i32 %i.0, 16
117  br i1 %cmp, label %for.body, label %for.cond.cleanup
118
119for.cond.cleanup:
120  br label %for.end
121
122for.body:
123  %idxprom = sext i32 %i.0 to i64
124  %arrayidx = getelementptr inbounds i8, i8* %p, i64 %idxprom
125  %0 = load i8, i8* %arrayidx, align 1, !tbaa !9
126  %conv = sext i8 %0 to i32
127  %conv1 = sext i8 %r.0 to i32
128  %or = or i32 %conv1, %conv
129  %conv2 = trunc i32 %or to i8
130  br label %for.inc
131
132for.inc:
133  %inc = add nsw i32 %i.0, 1
134  br label %for.cond
135
136for.end:
137  ret i8 %r.0
138}
139
140define i32 @smin_v4i32(i32* %p) #0 {
141; CHECK-LABEL: @smin_v4i32(
142; CHECK-NEXT:  entry:
143; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i32* [[P:%.*]] to <4 x i32>*
144; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, <4 x i32>* [[TMP0]], align 4, [[TBAA0]]
145; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef>
146; CHECK-NEXT:    [[RDX_MINMAX_CMP:%.*]] = icmp slt <4 x i32> [[TMP1]], [[RDX_SHUF]]
147; CHECK-NEXT:    [[RDX_MINMAX_SELECT:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP]], <4 x i32> [[TMP1]], <4 x i32> [[RDX_SHUF]]
148; CHECK-NEXT:    [[RDX_SHUF3:%.*]] = shufflevector <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef>
149; CHECK-NEXT:    [[RDX_MINMAX_CMP4:%.*]] = icmp slt <4 x i32> [[RDX_MINMAX_SELECT]], [[RDX_SHUF3]]
150; CHECK-NEXT:    [[RDX_MINMAX_SELECT5:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP4]], <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> [[RDX_SHUF3]]
151; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x i32> [[RDX_MINMAX_SELECT5]], i32 0
152; CHECK-NEXT:    ret i32 [[TMP2]]
153;
154entry:
155  br label %for.cond
156
157for.cond:
158  %r.0 = phi i32 [ 2147483647, %entry ], [ %cond, %for.inc ]
159  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
160  %cmp = icmp slt i32 %i.0, 4
161  br i1 %cmp, label %for.body, label %for.cond.cleanup
162
163for.cond.cleanup:
164  br label %for.end
165
166for.body:
167  %idxprom = sext i32 %i.0 to i64
168  %arrayidx = getelementptr inbounds i32, i32* %p, i64 %idxprom
169  %0 = load i32, i32* %arrayidx, align 4, !tbaa !3
170  %cmp1 = icmp slt i32 %0, %r.0
171  br i1 %cmp1, label %cond.true, label %cond.false
172
173cond.true:
174  %idxprom2 = sext i32 %i.0 to i64
175  %arrayidx3 = getelementptr inbounds i32, i32* %p, i64 %idxprom2
176  %1 = load i32, i32* %arrayidx3, align 4, !tbaa !3
177  br label %cond.end
178
179cond.false:
180  br label %cond.end
181
182cond.end:
183  %cond = phi i32 [ %1, %cond.true ], [ %r.0, %cond.false ]
184  br label %for.inc
185
186for.inc:
187  %inc = add nsw i32 %i.0, 1
188  br label %for.cond
189
190for.end:
191  ret i32 %r.0
192}
193
194define i32 @umax_v4i32(i32* %p) #0 {
195; CHECK-LABEL: @umax_v4i32(
196; CHECK-NEXT:  entry:
197; CHECK-NEXT:    [[TMP0:%.*]] = bitcast i32* [[P:%.*]] to <4 x i32>*
198; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, <4 x i32>* [[TMP0]], align 4, [[TBAA0]]
199; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef>
200; CHECK-NEXT:    [[RDX_MINMAX_CMP:%.*]] = icmp ugt <4 x i32> [[TMP1]], [[RDX_SHUF]]
201; CHECK-NEXT:    [[RDX_MINMAX_SELECT:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP]], <4 x i32> [[TMP1]], <4 x i32> [[RDX_SHUF]]
202; CHECK-NEXT:    [[RDX_SHUF3:%.*]] = shufflevector <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef>
203; CHECK-NEXT:    [[RDX_MINMAX_CMP4:%.*]] = icmp ugt <4 x i32> [[RDX_MINMAX_SELECT]], [[RDX_SHUF3]]
204; CHECK-NEXT:    [[RDX_MINMAX_SELECT5:%.*]] = select <4 x i1> [[RDX_MINMAX_CMP4]], <4 x i32> [[RDX_MINMAX_SELECT]], <4 x i32> [[RDX_SHUF3]]
205; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x i32> [[RDX_MINMAX_SELECT5]], i32 0
206; CHECK-NEXT:    ret i32 [[TMP2]]
207;
208entry:
209  br label %for.cond
210
211for.cond:
212  %r.0 = phi i32 [ 0, %entry ], [ %cond, %for.inc ]
213  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
214  %cmp = icmp slt i32 %i.0, 4
215  br i1 %cmp, label %for.body, label %for.cond.cleanup
216
217for.cond.cleanup:
218  br label %for.end
219
220for.body:
221  %idxprom = sext i32 %i.0 to i64
222  %arrayidx = getelementptr inbounds i32, i32* %p, i64 %idxprom
223  %0 = load i32, i32* %arrayidx, align 4, !tbaa !3
224  %cmp1 = icmp ugt i32 %0, %r.0
225  br i1 %cmp1, label %cond.true, label %cond.false
226
227cond.true:
228  %idxprom2 = sext i32 %i.0 to i64
229  %arrayidx3 = getelementptr inbounds i32, i32* %p, i64 %idxprom2
230  %1 = load i32, i32* %arrayidx3, align 4, !tbaa !3
231  br label %cond.end
232
233cond.false:
234  br label %cond.end
235
236cond.end:
237  %cond = phi i32 [ %1, %cond.true ], [ %r.0, %cond.false ]
238  br label %for.inc
239
240for.inc:
241  %inc = add nsw i32 %i.0, 1
242  br label %for.cond
243
244for.end:
245  ret i32 %r.0
246}
247
248define float @fadd_v4i32(float* %p) #0 {
249; CHECK-LABEL: @fadd_v4i32(
250; CHECK-NEXT:  entry:
251; CHECK-NEXT:    [[TMP0:%.*]] = bitcast float* [[P:%.*]] to <4 x float>*
252; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, <4 x float>* [[TMP0]], align 4, [[TBAA7:!tbaa !.*]]
253; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef>
254; CHECK-NEXT:    [[BIN_RDX:%.*]] = fadd fast <4 x float> [[TMP1]], [[RDX_SHUF]]
255; CHECK-NEXT:    [[RDX_SHUF3:%.*]] = shufflevector <4 x float> [[BIN_RDX]], <4 x float> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef>
256; CHECK-NEXT:    [[BIN_RDX4:%.*]] = fadd fast <4 x float> [[BIN_RDX]], [[RDX_SHUF3]]
257; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[BIN_RDX4]], i32 0
258; CHECK-NEXT:    [[BIN_RDX5:%.*]] = fadd fast float -0.000000e+00, [[TMP2]]
259; CHECK-NEXT:    [[OP_EXTRA:%.*]] = fadd fast float [[BIN_RDX5]], 4.200000e+01
260; CHECK-NEXT:    ret float [[OP_EXTRA]]
261;
262entry:
263  br label %for.cond
264
265for.cond:
266  %r.0 = phi float [ 4.200000e+01, %entry ], [ %add, %for.inc ]
267  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
268  %cmp = icmp slt i32 %i.0, 4
269  br i1 %cmp, label %for.body, label %for.cond.cleanup
270
271for.cond.cleanup:
272  br label %for.end
273
274for.body:
275  %idxprom = sext i32 %i.0 to i64
276  %arrayidx = getelementptr inbounds float, float* %p, i64 %idxprom
277  %0 = load float, float* %arrayidx, align 4, !tbaa !10
278  %add = fadd fast float %r.0, %0
279  br label %for.inc
280
281for.inc:
282  %inc = add nsw i32 %i.0, 1
283  br label %for.cond
284
285for.end:
286  ret float %r.0
287}
288
289define float @fmul_v4i32(float* %p) #0 {
290; CHECK-LABEL: @fmul_v4i32(
291; CHECK-NEXT:  entry:
292; CHECK-NEXT:    [[TMP0:%.*]] = bitcast float* [[P:%.*]] to <4 x float>*
293; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, <4 x float>* [[TMP0]], align 4, [[TBAA7]]
294; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef>
295; CHECK-NEXT:    [[BIN_RDX:%.*]] = fmul fast <4 x float> [[TMP1]], [[RDX_SHUF]]
296; CHECK-NEXT:    [[RDX_SHUF3:%.*]] = shufflevector <4 x float> [[BIN_RDX]], <4 x float> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef>
297; CHECK-NEXT:    [[BIN_RDX4:%.*]] = fmul fast <4 x float> [[BIN_RDX]], [[RDX_SHUF3]]
298; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[BIN_RDX4]], i32 0
299; CHECK-NEXT:    [[BIN_RDX5:%.*]] = fmul fast float 1.000000e+00, [[TMP2]]
300; CHECK-NEXT:    [[OP_EXTRA:%.*]] = fmul fast float [[BIN_RDX5]], 4.200000e+01
301; CHECK-NEXT:    ret float [[OP_EXTRA]]
302;
303entry:
304  br label %for.cond
305
306for.cond:
307  %r.0 = phi float [ 4.200000e+01, %entry ], [ %mul, %for.inc ]
308  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
309  %cmp = icmp slt i32 %i.0, 4
310  br i1 %cmp, label %for.body, label %for.cond.cleanup
311
312for.cond.cleanup:
313  br label %for.end
314
315for.body:
316  %idxprom = sext i32 %i.0 to i64
317  %arrayidx = getelementptr inbounds float, float* %p, i64 %idxprom
318  %0 = load float, float* %arrayidx, align 4, !tbaa !10
319  %mul = fmul fast float %r.0, %0
320  br label %for.inc
321
322for.inc:
323  %inc = add nsw i32 %i.0, 1
324  br label %for.cond
325
326for.end:
327  ret float %r.0
328}
329
330define float @fmin_v4f32(float* %p) #0 {
331; CHECK-LABEL: @fmin_v4f32(
332; CHECK-NEXT:  entry:
333; CHECK-NEXT:    [[TMP0:%.*]] = bitcast float* [[P:%.*]] to <4 x float>*
334; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, <4 x float>* [[TMP0]], align 4, [[TBAA7]]
335; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 2, i32 3, i32 undef, i32 undef>
336; CHECK-NEXT:    [[RDX_MINMAX_CMP:%.*]] = fcmp fast olt <4 x float> [[TMP1]], [[RDX_SHUF]]
337; CHECK-NEXT:    [[RDX_MINMAX_SELECT:%.*]] = select fast <4 x i1> [[RDX_MINMAX_CMP]], <4 x float> [[TMP1]], <4 x float> [[RDX_SHUF]]
338; CHECK-NEXT:    [[RDX_SHUF3:%.*]] = shufflevector <4 x float> [[RDX_MINMAX_SELECT]], <4 x float> poison, <4 x i32> <i32 1, i32 undef, i32 undef, i32 undef>
339; CHECK-NEXT:    [[RDX_MINMAX_CMP4:%.*]] = fcmp fast olt <4 x float> [[RDX_MINMAX_SELECT]], [[RDX_SHUF3]]
340; CHECK-NEXT:    [[RDX_MINMAX_SELECT5:%.*]] = select fast <4 x i1> [[RDX_MINMAX_CMP4]], <4 x float> [[RDX_MINMAX_SELECT]], <4 x float> [[RDX_SHUF3]]
341; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x float> [[RDX_MINMAX_SELECT5]], i32 0
342; CHECK-NEXT:    ret float [[TMP2]]
343;
344entry:
345  br label %for.cond
346
347for.cond:
348  %r.0 = phi float [  0x47EFFFFFE0000000, %entry ], [ %cond, %for.inc ]
349  %i.0 = phi i32 [ 0, %entry ], [ %inc, %for.inc ]
350  %cmp = icmp slt i32 %i.0, 4
351  br i1 %cmp, label %for.body, label %for.cond.cleanup
352
353for.cond.cleanup:
354  br label %for.end
355
356for.body:
357  %idxprom = sext i32 %i.0 to i64
358  %arrayidx = getelementptr inbounds float, float* %p, i64 %idxprom
359  %0 = load float, float* %arrayidx, align 4, !tbaa !10
360  %cmp1 = fcmp fast olt float %0, %r.0
361  br i1 %cmp1, label %cond.true, label %cond.false
362
363cond.true:
364  %idxprom2 = sext i32 %i.0 to i64
365  %arrayidx3 = getelementptr inbounds float, float* %p, i64 %idxprom2
366  %1 = load float, float* %arrayidx3, align 4, !tbaa !10
367  br label %cond.end
368
369cond.false:
370  br label %cond.end
371
372cond.end:
373  %cond = phi fast float [ %1, %cond.true ], [ %r.0, %cond.false ]
374  br label %for.inc
375
376for.inc:
377  %inc = add nsw i32 %i.0, 1
378  br label %for.cond
379
380for.end:
381  ret float %r.0
382}
383
384define available_externally float @max(float %a, float %b) {
385entry:
386  %a.addr = alloca float, align 4
387  %b.addr = alloca float, align 4
388  store float %a, float* %a.addr, align 4
389  store float %b, float* %b.addr, align 4
390  %0 = load float, float* %a.addr, align 4
391  %1 = load float, float* %b.addr, align 4
392  %cmp = fcmp nnan ninf nsz ogt float %0, %1
393  br i1 %cmp, label %cond.true, label %cond.false
394
395cond.true:                                        ; preds = %entry
396  %2 = load float, float* %a.addr, align 4
397  br label %cond.end
398
399cond.false:                                       ; preds = %entry
400  %3 = load float, float* %b.addr, align 4
401  br label %cond.end
402
403cond.end:                                         ; preds = %cond.false, %cond.true
404  %cond = phi nnan ninf nsz float [ %2, %cond.true ], [ %3, %cond.false ]
405  ret float %cond
406}
407
408; PR23116
409
410define float @findMax(<8 x float>* byval(<8 x float>) align 16 %0) {
411; CHECK-LABEL: @findMax(
412; CHECK-NEXT:  entry:
413; CHECK-NEXT:    [[V:%.*]] = load <8 x float>, <8 x float>* [[TMP0:%.*]], align 16, [[TBAA0]]
414; CHECK-NEXT:    [[RDX_SHUF:%.*]] = shufflevector <8 x float> [[V]], <8 x float> poison, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 undef, i32 undef, i32 undef, i32 undef>
415; CHECK-NEXT:    [[RDX_MINMAX_CMP:%.*]] = fcmp nnan ninf nsz ogt <8 x float> [[V]], [[RDX_SHUF]]
416; CHECK-NEXT:    [[RDX_MINMAX_SELECT:%.*]] = select nnan ninf nsz <8 x i1> [[RDX_MINMAX_CMP]], <8 x float> [[V]], <8 x float> [[RDX_SHUF]]
417; CHECK-NEXT:    [[RDX_SHUF8:%.*]] = shufflevector <8 x float> [[RDX_MINMAX_SELECT]], <8 x float> poison, <8 x i32> <i32 2, i32 3, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
418; CHECK-NEXT:    [[RDX_MINMAX_CMP9:%.*]] = fcmp nnan ninf nsz ogt <8 x float> [[RDX_MINMAX_SELECT]], [[RDX_SHUF8]]
419; CHECK-NEXT:    [[RDX_MINMAX_SELECT10:%.*]] = select nnan ninf nsz <8 x i1> [[RDX_MINMAX_CMP9]], <8 x float> [[RDX_MINMAX_SELECT]], <8 x float> [[RDX_SHUF8]]
420; CHECK-NEXT:    [[RDX_SHUF11:%.*]] = shufflevector <8 x float> [[RDX_MINMAX_SELECT10]], <8 x float> poison, <8 x i32> <i32 1, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef, i32 undef>
421; CHECK-NEXT:    [[RDX_MINMAX_CMP12:%.*]] = fcmp nnan ninf nsz ogt <8 x float> [[RDX_MINMAX_SELECT10]], [[RDX_SHUF11]]
422; CHECK-NEXT:    [[RDX_MINMAX_SELECT13:%.*]] = select nnan ninf nsz <8 x i1> [[RDX_MINMAX_CMP12]], <8 x float> [[RDX_MINMAX_SELECT10]], <8 x float> [[RDX_SHUF11]]
423; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <8 x float> [[RDX_MINMAX_SELECT13]], i32 0
424; CHECK-NEXT:    ret float [[TMP1]]
425;
426entry:
427  %v.addr = alloca <8 x float>, align 32
428  %v = load <8 x float>, <8 x float>* %0, align 16, !tbaa !3
429  store <8 x float> %v, <8 x float>* %v.addr, align 32, !tbaa !3
430  %1 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
431  %vecext = extractelement <8 x float> %1, i32 0
432  %2 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
433  %vecext1 = extractelement <8 x float> %2, i32 1
434  %call = call nnan ninf nsz float @max(float %vecext, float %vecext1)
435  %3 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
436  %vecext2 = extractelement <8 x float> %3, i32 2
437  %call3 = call nnan ninf nsz float @max(float %call, float %vecext2)
438  %4 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
439  %vecext4 = extractelement <8 x float> %4, i32 3
440  %call5 = call nnan ninf nsz float @max(float %call3, float %vecext4)
441  %5 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
442  %vecext6 = extractelement <8 x float> %5, i32 4
443  %call7 = call nnan ninf nsz float @max(float %call5, float %vecext6)
444  %6 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
445  %vecext8 = extractelement <8 x float> %6, i32 5
446  %call9 = call nnan ninf nsz float @max(float %call7, float %vecext8)
447  %7 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
448  %vecext10 = extractelement <8 x float> %7, i32 6
449  %call11 = call nnan ninf nsz float @max(float %call9, float %vecext10)
450  %8 = load <8 x float>, <8 x float>* %v.addr, align 32, !tbaa !3
451  %vecext12 = extractelement <8 x float> %8, i32 7
452  %call13 = call nnan ninf nsz float @max(float %call11, float %vecext12)
453  ret float %call13
454}
455
456attributes #0 = { nounwind ssp uwtable "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "frame-pointer"="all" "less-precise-fpmad"="false" "min-legal-vector-width"="0" "no-infs-fp-math"="true" "no-jump-tables"="false" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="penryn" "target-features"="+avx,+cx16,+cx8,+fxsr,+mmx,+popcnt,+sahf,+sse,+sse2,+sse3,+sse4.1,+sse4.2,+ssse3,+x87,+xsave" "unsafe-fp-math"="true" "use-soft-float"="false" }
457
458!0 = !{i32 1, !"wchar_size", i32 4}
459!1 = !{i32 7, !"PIC Level", i32 2}
460!2 = !{!"clang version 11.0.0 (https://github.com/llvm/llvm-project.git a9fe69c359de653015c39e413e48630d069abe27)"}
461!3 = !{!4, !4, i64 0}
462!4 = !{!"int", !5, i64 0}
463!5 = !{!"omnipotent char", !6, i64 0}
464!6 = !{!"Simple C/C++ TBAA"}
465!7 = !{!8, !8, i64 0}
466!8 = !{!"short", !5, i64 0}
467!9 = !{!5, !5, i64 0}
468!10 = !{!11, !11, i64 0}
469!11 = !{!"float", !5, i64 0}
470