1 //===- DAGCombiner.cpp - Implement a DAG node combiner --------------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This pass combines dag nodes to form fewer, simpler DAG nodes.  It can be run
10 // both before and after the DAG is legalized.
11 //
12 // This pass is not a substitute for the LLVM IR instcombine pass. This pass is
13 // primarily intended to handle simplification opportunities that are implicit
14 // in the LLVM IR and exposed by the various codegen lowering phases.
15 //
16 //===----------------------------------------------------------------------===//
17 
18 #include "llvm/ADT/APFloat.h"
19 #include "llvm/ADT/APInt.h"
20 #include "llvm/ADT/ArrayRef.h"
21 #include "llvm/ADT/DenseMap.h"
22 #include "llvm/ADT/IntervalMap.h"
23 #include "llvm/ADT/None.h"
24 #include "llvm/ADT/Optional.h"
25 #include "llvm/ADT/STLExtras.h"
26 #include "llvm/ADT/SetVector.h"
27 #include "llvm/ADT/SmallBitVector.h"
28 #include "llvm/ADT/SmallPtrSet.h"
29 #include "llvm/ADT/SmallSet.h"
30 #include "llvm/ADT/SmallVector.h"
31 #include "llvm/ADT/Statistic.h"
32 #include "llvm/Analysis/AliasAnalysis.h"
33 #include "llvm/Analysis/MemoryLocation.h"
34 #include "llvm/CodeGen/DAGCombine.h"
35 #include "llvm/CodeGen/ISDOpcodes.h"
36 #include "llvm/CodeGen/MachineFrameInfo.h"
37 #include "llvm/CodeGen/MachineFunction.h"
38 #include "llvm/CodeGen/MachineMemOperand.h"
39 #include "llvm/CodeGen/RuntimeLibcalls.h"
40 #include "llvm/CodeGen/SelectionDAG.h"
41 #include "llvm/CodeGen/SelectionDAGAddressAnalysis.h"
42 #include "llvm/CodeGen/SelectionDAGNodes.h"
43 #include "llvm/CodeGen/SelectionDAGTargetInfo.h"
44 #include "llvm/CodeGen/TargetLowering.h"
45 #include "llvm/CodeGen/TargetRegisterInfo.h"
46 #include "llvm/CodeGen/TargetSubtargetInfo.h"
47 #include "llvm/CodeGen/ValueTypes.h"
48 #include "llvm/IR/Attributes.h"
49 #include "llvm/IR/Constant.h"
50 #include "llvm/IR/DataLayout.h"
51 #include "llvm/IR/DerivedTypes.h"
52 #include "llvm/IR/Function.h"
53 #include "llvm/IR/LLVMContext.h"
54 #include "llvm/IR/Metadata.h"
55 #include "llvm/Support/Casting.h"
56 #include "llvm/Support/CodeGen.h"
57 #include "llvm/Support/CommandLine.h"
58 #include "llvm/Support/Compiler.h"
59 #include "llvm/Support/Debug.h"
60 #include "llvm/Support/ErrorHandling.h"
61 #include "llvm/Support/KnownBits.h"
62 #include "llvm/Support/MachineValueType.h"
63 #include "llvm/Support/MathExtras.h"
64 #include "llvm/Support/raw_ostream.h"
65 #include "llvm/Target/TargetMachine.h"
66 #include "llvm/Target/TargetOptions.h"
67 #include <algorithm>
68 #include <cassert>
69 #include <cstdint>
70 #include <functional>
71 #include <iterator>
72 #include <string>
73 #include <tuple>
74 #include <utility>
75 
76 using namespace llvm;
77 
78 #define DEBUG_TYPE "dagcombine"
79 
80 STATISTIC(NodesCombined   , "Number of dag nodes combined");
81 STATISTIC(PreIndexedNodes , "Number of pre-indexed nodes created");
82 STATISTIC(PostIndexedNodes, "Number of post-indexed nodes created");
83 STATISTIC(OpsNarrowed     , "Number of load/op/store narrowed");
84 STATISTIC(LdStFP2Int      , "Number of fp load/store pairs transformed to int");
85 STATISTIC(SlicedLoads, "Number of load sliced");
86 STATISTIC(NumFPLogicOpsConv, "Number of logic ops converted to fp ops");
87 
88 static cl::opt<bool>
89 CombinerGlobalAA("combiner-global-alias-analysis", cl::Hidden,
90                  cl::desc("Enable DAG combiner's use of IR alias analysis"));
91 
92 static cl::opt<bool>
93 UseTBAA("combiner-use-tbaa", cl::Hidden, cl::init(true),
94         cl::desc("Enable DAG combiner's use of TBAA"));
95 
96 #ifndef NDEBUG
97 static cl::opt<std::string>
98 CombinerAAOnlyFunc("combiner-aa-only-func", cl::Hidden,
99                    cl::desc("Only use DAG-combiner alias analysis in this"
100                             " function"));
101 #endif
102 
103 /// Hidden option to stress test load slicing, i.e., when this option
104 /// is enabled, load slicing bypasses most of its profitability guards.
105 static cl::opt<bool>
106 StressLoadSlicing("combiner-stress-load-slicing", cl::Hidden,
107                   cl::desc("Bypass the profitability model of load slicing"),
108                   cl::init(false));
109 
110 static cl::opt<bool>
111   MaySplitLoadIndex("combiner-split-load-index", cl::Hidden, cl::init(true),
112                     cl::desc("DAG combiner may split indexing from loads"));
113 
114 static cl::opt<unsigned> TokenFactorInlineLimit(
115     "combiner-tokenfactor-inline-limit", cl::Hidden, cl::init(2048),
116     cl::desc("Limit the number of operands to inline for Token Factors"));
117 
118 namespace {
119 
120   class DAGCombiner {
121     SelectionDAG &DAG;
122     const TargetLowering &TLI;
123     CombineLevel Level;
124     CodeGenOpt::Level OptLevel;
125     bool LegalOperations = false;
126     bool LegalTypes = false;
127     bool ForCodeSize;
128 
129     /// Worklist of all of the nodes that need to be simplified.
130     ///
131     /// This must behave as a stack -- new nodes to process are pushed onto the
132     /// back and when processing we pop off of the back.
133     ///
134     /// The worklist will not contain duplicates but may contain null entries
135     /// due to nodes being deleted from the underlying DAG.
136     SmallVector<SDNode *, 64> Worklist;
137 
138     /// Mapping from an SDNode to its position on the worklist.
139     ///
140     /// This is used to find and remove nodes from the worklist (by nulling
141     /// them) when they are deleted from the underlying DAG. It relies on
142     /// stable indices of nodes within the worklist.
143     DenseMap<SDNode *, unsigned> WorklistMap;
144     /// This records all nodes attempted to add to the worklist since we
145     /// considered a new worklist entry. As we keep do not add duplicate nodes
146     /// in the worklist, this is different from the tail of the worklist.
147     SmallSetVector<SDNode *, 32> PruningList;
148 
149     /// Set of nodes which have been combined (at least once).
150     ///
151     /// This is used to allow us to reliably add any operands of a DAG node
152     /// which have not yet been combined to the worklist.
153     SmallPtrSet<SDNode *, 32> CombinedNodes;
154 
155     // AA - Used for DAG load/store alias analysis.
156     AliasAnalysis *AA;
157 
158     /// When an instruction is simplified, add all users of the instruction to
159     /// the work lists because they might get more simplified now.
160     void AddUsersToWorklist(SDNode *N) {
161       for (SDNode *Node : N->uses())
162         AddToWorklist(Node);
163     }
164 
165     // Prune potentially dangling nodes. This is called after
166     // any visit to a node, but should also be called during a visit after any
167     // failed combine which may have created a DAG node.
168     void clearAddedDanglingWorklistEntries() {
169       // Check any nodes added to the worklist to see if they are prunable.
170       while (!PruningList.empty()) {
171         auto *N = PruningList.pop_back_val();
172         if (N->use_empty())
173           recursivelyDeleteUnusedNodes(N);
174       }
175     }
176 
177     SDNode *getNextWorklistEntry() {
178       // Before we do any work, remove nodes that are not in use.
179       clearAddedDanglingWorklistEntries();
180       SDNode *N = nullptr;
181       // The Worklist holds the SDNodes in order, but it may contain null
182       // entries.
183       while (!N && !Worklist.empty()) {
184         N = Worklist.pop_back_val();
185       }
186 
187       if (N) {
188         bool GoodWorklistEntry = WorklistMap.erase(N);
189         (void)GoodWorklistEntry;
190         assert(GoodWorklistEntry &&
191                "Found a worklist entry without a corresponding map entry!");
192       }
193       return N;
194     }
195 
196     /// Call the node-specific routine that folds each particular type of node.
197     SDValue visit(SDNode *N);
198 
199   public:
200     DAGCombiner(SelectionDAG &D, AliasAnalysis *AA, CodeGenOpt::Level OL)
201         : DAG(D), TLI(D.getTargetLoweringInfo()), Level(BeforeLegalizeTypes),
202           OptLevel(OL), AA(AA) {
203       ForCodeSize = DAG.getMachineFunction().getFunction().hasOptSize();
204 
205       MaximumLegalStoreInBits = 0;
206       for (MVT VT : MVT::all_valuetypes())
207         if (EVT(VT).isSimple() && VT != MVT::Other &&
208             TLI.isTypeLegal(EVT(VT)) &&
209             VT.getSizeInBits() >= MaximumLegalStoreInBits)
210           MaximumLegalStoreInBits = VT.getSizeInBits();
211     }
212 
213     void ConsiderForPruning(SDNode *N) {
214       // Mark this for potential pruning.
215       PruningList.insert(N);
216     }
217 
218     /// Add to the worklist making sure its instance is at the back (next to be
219     /// processed.)
220     void AddToWorklist(SDNode *N) {
221       assert(N->getOpcode() != ISD::DELETED_NODE &&
222              "Deleted Node added to Worklist");
223 
224       // Skip handle nodes as they can't usefully be combined and confuse the
225       // zero-use deletion strategy.
226       if (N->getOpcode() == ISD::HANDLENODE)
227         return;
228 
229       ConsiderForPruning(N);
230 
231       if (WorklistMap.insert(std::make_pair(N, Worklist.size())).second)
232         Worklist.push_back(N);
233     }
234 
235     /// Remove all instances of N from the worklist.
236     void removeFromWorklist(SDNode *N) {
237       CombinedNodes.erase(N);
238       PruningList.remove(N);
239 
240       auto It = WorklistMap.find(N);
241       if (It == WorklistMap.end())
242         return; // Not in the worklist.
243 
244       // Null out the entry rather than erasing it to avoid a linear operation.
245       Worklist[It->second] = nullptr;
246       WorklistMap.erase(It);
247     }
248 
249     void deleteAndRecombine(SDNode *N);
250     bool recursivelyDeleteUnusedNodes(SDNode *N);
251 
252     /// Replaces all uses of the results of one DAG node with new values.
253     SDValue CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
254                       bool AddTo = true);
255 
256     /// Replaces all uses of the results of one DAG node with new values.
257     SDValue CombineTo(SDNode *N, SDValue Res, bool AddTo = true) {
258       return CombineTo(N, &Res, 1, AddTo);
259     }
260 
261     /// Replaces all uses of the results of one DAG node with new values.
262     SDValue CombineTo(SDNode *N, SDValue Res0, SDValue Res1,
263                       bool AddTo = true) {
264       SDValue To[] = { Res0, Res1 };
265       return CombineTo(N, To, 2, AddTo);
266     }
267 
268     void CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO);
269 
270   private:
271     unsigned MaximumLegalStoreInBits;
272 
273     /// Check the specified integer node value to see if it can be simplified or
274     /// if things it uses can be simplified by bit propagation.
275     /// If so, return true.
276     bool SimplifyDemandedBits(SDValue Op) {
277       unsigned BitWidth = Op.getScalarValueSizeInBits();
278       APInt DemandedBits = APInt::getAllOnesValue(BitWidth);
279       return SimplifyDemandedBits(Op, DemandedBits);
280     }
281 
282     bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits) {
283       EVT VT = Op.getValueType();
284       unsigned NumElts = VT.isVector() ? VT.getVectorNumElements() : 1;
285       APInt DemandedElts = APInt::getAllOnesValue(NumElts);
286       return SimplifyDemandedBits(Op, DemandedBits, DemandedElts);
287     }
288 
289     /// Check the specified vector node value to see if it can be simplified or
290     /// if things it uses can be simplified as it only uses some of the
291     /// elements. If so, return true.
292     bool SimplifyDemandedVectorElts(SDValue Op) {
293       unsigned NumElts = Op.getValueType().getVectorNumElements();
294       APInt DemandedElts = APInt::getAllOnesValue(NumElts);
295       return SimplifyDemandedVectorElts(Op, DemandedElts);
296     }
297 
298     bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits,
299                               const APInt &DemandedElts);
300     bool SimplifyDemandedVectorElts(SDValue Op, const APInt &DemandedElts,
301                                     bool AssumeSingleUse = false);
302 
303     bool CombineToPreIndexedLoadStore(SDNode *N);
304     bool CombineToPostIndexedLoadStore(SDNode *N);
305     SDValue SplitIndexingFromLoad(LoadSDNode *LD);
306     bool SliceUpLoad(SDNode *N);
307 
308     // Scalars have size 0 to distinguish from singleton vectors.
309     SDValue ForwardStoreValueToDirectLoad(LoadSDNode *LD);
310     bool getTruncatedStoreValue(StoreSDNode *ST, SDValue &Val);
311     bool extendLoadedValueToExtension(LoadSDNode *LD, SDValue &Val);
312 
313     /// Replace an ISD::EXTRACT_VECTOR_ELT of a load with a narrowed
314     ///   load.
315     ///
316     /// \param EVE ISD::EXTRACT_VECTOR_ELT to be replaced.
317     /// \param InVecVT type of the input vector to EVE with bitcasts resolved.
318     /// \param EltNo index of the vector element to load.
319     /// \param OriginalLoad load that EVE came from to be replaced.
320     /// \returns EVE on success SDValue() on failure.
321     SDValue scalarizeExtractedVectorLoad(SDNode *EVE, EVT InVecVT,
322                                          SDValue EltNo,
323                                          LoadSDNode *OriginalLoad);
324     void ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad);
325     SDValue PromoteOperand(SDValue Op, EVT PVT, bool &Replace);
326     SDValue SExtPromoteOperand(SDValue Op, EVT PVT);
327     SDValue ZExtPromoteOperand(SDValue Op, EVT PVT);
328     SDValue PromoteIntBinOp(SDValue Op);
329     SDValue PromoteIntShiftOp(SDValue Op);
330     SDValue PromoteExtend(SDValue Op);
331     bool PromoteLoad(SDValue Op);
332 
333     /// Call the node-specific routine that knows how to fold each
334     /// particular type of node. If that doesn't do anything, try the
335     /// target-specific DAG combines.
336     SDValue combine(SDNode *N);
337 
338     // Visitation implementation - Implement dag node combining for different
339     // node types.  The semantics are as follows:
340     // Return Value:
341     //   SDValue.getNode() == 0 - No change was made
342     //   SDValue.getNode() == N - N was replaced, is dead and has been handled.
343     //   otherwise              - N should be replaced by the returned Operand.
344     //
345     SDValue visitTokenFactor(SDNode *N);
346     SDValue visitMERGE_VALUES(SDNode *N);
347     SDValue visitADD(SDNode *N);
348     SDValue visitADDLike(SDNode *N);
349     SDValue visitADDLikeCommutative(SDValue N0, SDValue N1, SDNode *LocReference);
350     SDValue visitSUB(SDNode *N);
351     SDValue visitADDSAT(SDNode *N);
352     SDValue visitSUBSAT(SDNode *N);
353     SDValue visitADDC(SDNode *N);
354     SDValue visitADDO(SDNode *N);
355     SDValue visitUADDOLike(SDValue N0, SDValue N1, SDNode *N);
356     SDValue visitSUBC(SDNode *N);
357     SDValue visitSUBO(SDNode *N);
358     SDValue visitADDE(SDNode *N);
359     SDValue visitADDCARRY(SDNode *N);
360     SDValue visitADDCARRYLike(SDValue N0, SDValue N1, SDValue CarryIn, SDNode *N);
361     SDValue visitSUBE(SDNode *N);
362     SDValue visitSUBCARRY(SDNode *N);
363     SDValue visitMUL(SDNode *N);
364     SDValue useDivRem(SDNode *N);
365     SDValue visitSDIV(SDNode *N);
366     SDValue visitSDIVLike(SDValue N0, SDValue N1, SDNode *N);
367     SDValue visitUDIV(SDNode *N);
368     SDValue visitUDIVLike(SDValue N0, SDValue N1, SDNode *N);
369     SDValue visitREM(SDNode *N);
370     SDValue visitMULHU(SDNode *N);
371     SDValue visitMULHS(SDNode *N);
372     SDValue visitSMUL_LOHI(SDNode *N);
373     SDValue visitUMUL_LOHI(SDNode *N);
374     SDValue visitMULO(SDNode *N);
375     SDValue visitIMINMAX(SDNode *N);
376     SDValue visitAND(SDNode *N);
377     SDValue visitANDLike(SDValue N0, SDValue N1, SDNode *N);
378     SDValue visitOR(SDNode *N);
379     SDValue visitORLike(SDValue N0, SDValue N1, SDNode *N);
380     SDValue visitXOR(SDNode *N);
381     SDValue SimplifyVBinOp(SDNode *N);
382     SDValue visitSHL(SDNode *N);
383     SDValue visitSRA(SDNode *N);
384     SDValue visitSRL(SDNode *N);
385     SDValue visitFunnelShift(SDNode *N);
386     SDValue visitRotate(SDNode *N);
387     SDValue visitABS(SDNode *N);
388     SDValue visitBSWAP(SDNode *N);
389     SDValue visitBITREVERSE(SDNode *N);
390     SDValue visitCTLZ(SDNode *N);
391     SDValue visitCTLZ_ZERO_UNDEF(SDNode *N);
392     SDValue visitCTTZ(SDNode *N);
393     SDValue visitCTTZ_ZERO_UNDEF(SDNode *N);
394     SDValue visitCTPOP(SDNode *N);
395     SDValue visitSELECT(SDNode *N);
396     SDValue visitVSELECT(SDNode *N);
397     SDValue visitSELECT_CC(SDNode *N);
398     SDValue visitSETCC(SDNode *N);
399     SDValue visitSETCCCARRY(SDNode *N);
400     SDValue visitSIGN_EXTEND(SDNode *N);
401     SDValue visitZERO_EXTEND(SDNode *N);
402     SDValue visitANY_EXTEND(SDNode *N);
403     SDValue visitAssertExt(SDNode *N);
404     SDValue visitSIGN_EXTEND_INREG(SDNode *N);
405     SDValue visitSIGN_EXTEND_VECTOR_INREG(SDNode *N);
406     SDValue visitZERO_EXTEND_VECTOR_INREG(SDNode *N);
407     SDValue visitTRUNCATE(SDNode *N);
408     SDValue visitBITCAST(SDNode *N);
409     SDValue visitBUILD_PAIR(SDNode *N);
410     SDValue visitFADD(SDNode *N);
411     SDValue visitFSUB(SDNode *N);
412     SDValue visitFMUL(SDNode *N);
413     SDValue visitFMA(SDNode *N);
414     SDValue visitFDIV(SDNode *N);
415     SDValue visitFREM(SDNode *N);
416     SDValue visitFSQRT(SDNode *N);
417     SDValue visitFCOPYSIGN(SDNode *N);
418     SDValue visitFPOW(SDNode *N);
419     SDValue visitSINT_TO_FP(SDNode *N);
420     SDValue visitUINT_TO_FP(SDNode *N);
421     SDValue visitFP_TO_SINT(SDNode *N);
422     SDValue visitFP_TO_UINT(SDNode *N);
423     SDValue visitFP_ROUND(SDNode *N);
424     SDValue visitFP_ROUND_INREG(SDNode *N);
425     SDValue visitFP_EXTEND(SDNode *N);
426     SDValue visitFNEG(SDNode *N);
427     SDValue visitFABS(SDNode *N);
428     SDValue visitFCEIL(SDNode *N);
429     SDValue visitFTRUNC(SDNode *N);
430     SDValue visitFFLOOR(SDNode *N);
431     SDValue visitFMINNUM(SDNode *N);
432     SDValue visitFMAXNUM(SDNode *N);
433     SDValue visitFMINIMUM(SDNode *N);
434     SDValue visitFMAXIMUM(SDNode *N);
435     SDValue visitBRCOND(SDNode *N);
436     SDValue visitBR_CC(SDNode *N);
437     SDValue visitLOAD(SDNode *N);
438 
439     SDValue replaceStoreChain(StoreSDNode *ST, SDValue BetterChain);
440     SDValue replaceStoreOfFPConstant(StoreSDNode *ST);
441 
442     SDValue visitSTORE(SDNode *N);
443     SDValue visitLIFETIME_END(SDNode *N);
444     SDValue visitINSERT_VECTOR_ELT(SDNode *N);
445     SDValue visitEXTRACT_VECTOR_ELT(SDNode *N);
446     SDValue visitBUILD_VECTOR(SDNode *N);
447     SDValue visitCONCAT_VECTORS(SDNode *N);
448     SDValue visitEXTRACT_SUBVECTOR(SDNode *N);
449     SDValue visitVECTOR_SHUFFLE(SDNode *N);
450     SDValue visitSCALAR_TO_VECTOR(SDNode *N);
451     SDValue visitINSERT_SUBVECTOR(SDNode *N);
452     SDValue visitMLOAD(SDNode *N);
453     SDValue visitMSTORE(SDNode *N);
454     SDValue visitMGATHER(SDNode *N);
455     SDValue visitMSCATTER(SDNode *N);
456     SDValue visitFP_TO_FP16(SDNode *N);
457     SDValue visitFP16_TO_FP(SDNode *N);
458     SDValue visitVECREDUCE(SDNode *N);
459 
460     SDValue visitFADDForFMACombine(SDNode *N);
461     SDValue visitFSUBForFMACombine(SDNode *N);
462     SDValue visitFMULForFMADistributiveCombine(SDNode *N);
463 
464     SDValue XformToShuffleWithZero(SDNode *N);
465     bool reassociationCanBreakAddressingModePattern(unsigned Opc,
466                                                     const SDLoc &DL, SDValue N0,
467                                                     SDValue N1);
468     SDValue reassociateOpsCommutative(unsigned Opc, const SDLoc &DL, SDValue N0,
469                                       SDValue N1);
470     SDValue reassociateOps(unsigned Opc, const SDLoc &DL, SDValue N0,
471                            SDValue N1, SDNodeFlags Flags);
472 
473     SDValue visitShiftByConstant(SDNode *N, ConstantSDNode *Amt);
474 
475     SDValue foldSelectOfConstants(SDNode *N);
476     SDValue foldVSelectOfConstants(SDNode *N);
477     SDValue foldBinOpIntoSelect(SDNode *BO);
478     bool SimplifySelectOps(SDNode *SELECT, SDValue LHS, SDValue RHS);
479     SDValue hoistLogicOpWithSameOpcodeHands(SDNode *N);
480     SDValue SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2);
481     SDValue SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
482                              SDValue N2, SDValue N3, ISD::CondCode CC,
483                              bool NotExtCompare = false);
484     SDValue convertSelectOfFPConstantsToLoadOffset(
485         const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3,
486         ISD::CondCode CC);
487     SDValue foldSelectCCToShiftAnd(const SDLoc &DL, SDValue N0, SDValue N1,
488                                    SDValue N2, SDValue N3, ISD::CondCode CC);
489     SDValue foldLogicOfSetCCs(bool IsAnd, SDValue N0, SDValue N1,
490                               const SDLoc &DL);
491     SDValue unfoldMaskedMerge(SDNode *N);
492     SDValue unfoldExtremeBitClearingToShifts(SDNode *N);
493     SDValue SimplifySetCC(EVT VT, SDValue N0, SDValue N1, ISD::CondCode Cond,
494                           const SDLoc &DL, bool foldBooleans);
495     SDValue rebuildSetCC(SDValue N);
496 
497     bool isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
498                            SDValue &CC) const;
499     bool isOneUseSetCC(SDValue N) const;
500 
501     SDValue SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
502                                          unsigned HiOp);
503     SDValue CombineConsecutiveLoads(SDNode *N, EVT VT);
504     SDValue CombineExtLoad(SDNode *N);
505     SDValue CombineZExtLogicopShiftLoad(SDNode *N);
506     SDValue combineRepeatedFPDivisors(SDNode *N);
507     SDValue combineInsertEltToShuffle(SDNode *N, unsigned InsIndex);
508     SDValue ConstantFoldBITCASTofBUILD_VECTOR(SDNode *, EVT);
509     SDValue BuildSDIV(SDNode *N);
510     SDValue BuildSDIVPow2(SDNode *N);
511     SDValue BuildUDIV(SDNode *N);
512     SDValue BuildLogBase2(SDValue V, const SDLoc &DL);
513     SDValue BuildReciprocalEstimate(SDValue Op, SDNodeFlags Flags);
514     SDValue buildRsqrtEstimate(SDValue Op, SDNodeFlags Flags);
515     SDValue buildSqrtEstimate(SDValue Op, SDNodeFlags Flags);
516     SDValue buildSqrtEstimateImpl(SDValue Op, SDNodeFlags Flags, bool Recip);
517     SDValue buildSqrtNROneConst(SDValue Arg, SDValue Est, unsigned Iterations,
518                                 SDNodeFlags Flags, bool Reciprocal);
519     SDValue buildSqrtNRTwoConst(SDValue Arg, SDValue Est, unsigned Iterations,
520                                 SDNodeFlags Flags, bool Reciprocal);
521     SDValue MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
522                                bool DemandHighBits = true);
523     SDValue MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1);
524     SDNode *MatchRotatePosNeg(SDValue Shifted, SDValue Pos, SDValue Neg,
525                               SDValue InnerPos, SDValue InnerNeg,
526                               unsigned PosOpcode, unsigned NegOpcode,
527                               const SDLoc &DL);
528     SDNode *MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL);
529     SDValue MatchLoadCombine(SDNode *N);
530     SDValue MatchStoreCombine(StoreSDNode *N);
531     SDValue ReduceLoadWidth(SDNode *N);
532     SDValue ReduceLoadOpStoreWidth(SDNode *N);
533     SDValue splitMergedValStore(StoreSDNode *ST);
534     SDValue TransformFPLoadStorePair(SDNode *N);
535     SDValue convertBuildVecZextToZext(SDNode *N);
536     SDValue reduceBuildVecExtToExtBuildVec(SDNode *N);
537     SDValue reduceBuildVecToShuffle(SDNode *N);
538     SDValue createBuildVecShuffle(const SDLoc &DL, SDNode *N,
539                                   ArrayRef<int> VectorMask, SDValue VecIn1,
540                                   SDValue VecIn2, unsigned LeftIdx,
541                                   bool DidSplitVec);
542     SDValue matchVSelectOpSizesWithSetCC(SDNode *Cast);
543 
544     /// Walk up chain skipping non-aliasing memory nodes,
545     /// looking for aliasing nodes and adding them to the Aliases vector.
546     void GatherAllAliases(SDNode *N, SDValue OriginalChain,
547                           SmallVectorImpl<SDValue> &Aliases);
548 
549     /// Return true if there is any possibility that the two addresses overlap.
550     bool isAlias(SDNode *Op0, SDNode *Op1) const;
551 
552     /// Walk up chain skipping non-aliasing memory nodes, looking for a better
553     /// chain (aliasing node.)
554     SDValue FindBetterChain(SDNode *N, SDValue Chain);
555 
556     /// Try to replace a store and any possibly adjacent stores on
557     /// consecutive chains with better chains. Return true only if St is
558     /// replaced.
559     ///
560     /// Notice that other chains may still be replaced even if the function
561     /// returns false.
562     bool findBetterNeighborChains(StoreSDNode *St);
563 
564     // Helper for findBetterNeighborChains. Walk up store chain add additional
565     // chained stores that do not overlap and can be parallelized.
566     bool parallelizeChainedStores(StoreSDNode *St);
567 
568     /// Holds a pointer to an LSBaseSDNode as well as information on where it
569     /// is located in a sequence of memory operations connected by a chain.
570     struct MemOpLink {
571       // Ptr to the mem node.
572       LSBaseSDNode *MemNode;
573 
574       // Offset from the base ptr.
575       int64_t OffsetFromBase;
576 
577       MemOpLink(LSBaseSDNode *N, int64_t Offset)
578           : MemNode(N), OffsetFromBase(Offset) {}
579     };
580 
581     /// This is a helper function for visitMUL to check the profitability
582     /// of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
583     /// MulNode is the original multiply, AddNode is (add x, c1),
584     /// and ConstNode is c2.
585     bool isMulAddWithConstProfitable(SDNode *MulNode,
586                                      SDValue &AddNode,
587                                      SDValue &ConstNode);
588 
589     /// This is a helper function for visitAND and visitZERO_EXTEND.  Returns
590     /// true if the (and (load x) c) pattern matches an extload.  ExtVT returns
591     /// the type of the loaded value to be extended.
592     bool isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
593                           EVT LoadResultTy, EVT &ExtVT);
594 
595     /// Helper function to calculate whether the given Load/Store can have its
596     /// width reduced to ExtVT.
597     bool isLegalNarrowLdSt(LSBaseSDNode *LDSTN, ISD::LoadExtType ExtType,
598                            EVT &MemVT, unsigned ShAmt = 0);
599 
600     /// Used by BackwardsPropagateMask to find suitable loads.
601     bool SearchForAndLoads(SDNode *N, SmallVectorImpl<LoadSDNode*> &Loads,
602                            SmallPtrSetImpl<SDNode*> &NodesWithConsts,
603                            ConstantSDNode *Mask, SDNode *&NodeToMask);
604     /// Attempt to propagate a given AND node back to load leaves so that they
605     /// can be combined into narrow loads.
606     bool BackwardsPropagateMask(SDNode *N, SelectionDAG &DAG);
607 
608     /// Helper function for MergeConsecutiveStores which merges the
609     /// component store chains.
610     SDValue getMergeStoreChains(SmallVectorImpl<MemOpLink> &StoreNodes,
611                                 unsigned NumStores);
612 
613     /// This is a helper function for MergeConsecutiveStores. When the
614     /// source elements of the consecutive stores are all constants or
615     /// all extracted vector elements, try to merge them into one
616     /// larger store introducing bitcasts if necessary.  \return True
617     /// if a merged store was created.
618     bool MergeStoresOfConstantsOrVecElts(SmallVectorImpl<MemOpLink> &StoreNodes,
619                                          EVT MemVT, unsigned NumStores,
620                                          bool IsConstantSrc, bool UseVector,
621                                          bool UseTrunc);
622 
623     /// This is a helper function for MergeConsecutiveStores. Stores
624     /// that potentially may be merged with St are placed in
625     /// StoreNodes. RootNode is a chain predecessor to all store
626     /// candidates.
627     void getStoreMergeCandidates(StoreSDNode *St,
628                                  SmallVectorImpl<MemOpLink> &StoreNodes,
629                                  SDNode *&Root);
630 
631     /// Helper function for MergeConsecutiveStores. Checks if
632     /// candidate stores have indirect dependency through their
633     /// operands. RootNode is the predecessor to all stores calculated
634     /// by getStoreMergeCandidates and is used to prune the dependency check.
635     /// \return True if safe to merge.
636     bool checkMergeStoreCandidatesForDependencies(
637         SmallVectorImpl<MemOpLink> &StoreNodes, unsigned NumStores,
638         SDNode *RootNode);
639 
640     /// Merge consecutive store operations into a wide store.
641     /// This optimization uses wide integers or vectors when possible.
642     /// \return number of stores that were merged into a merged store (the
643     /// affected nodes are stored as a prefix in \p StoreNodes).
644     bool MergeConsecutiveStores(StoreSDNode *St);
645 
646     /// Try to transform a truncation where C is a constant:
647     ///     (trunc (and X, C)) -> (and (trunc X), (trunc C))
648     ///
649     /// \p N needs to be a truncation and its first operand an AND. Other
650     /// requirements are checked by the function (e.g. that trunc is
651     /// single-use) and if missed an empty SDValue is returned.
652     SDValue distributeTruncateThroughAnd(SDNode *N);
653 
654     /// Helper function to determine whether the target supports operation
655     /// given by \p Opcode for type \p VT, that is, whether the operation
656     /// is legal or custom before legalizing operations, and whether is
657     /// legal (but not custom) after legalization.
658     bool hasOperation(unsigned Opcode, EVT VT) {
659       if (LegalOperations)
660         return TLI.isOperationLegal(Opcode, VT);
661       return TLI.isOperationLegalOrCustom(Opcode, VT);
662     }
663 
664   public:
665     /// Runs the dag combiner on all nodes in the work list
666     void Run(CombineLevel AtLevel);
667 
668     SelectionDAG &getDAG() const { return DAG; }
669 
670     /// Returns a type large enough to hold any valid shift amount - before type
671     /// legalization these can be huge.
672     EVT getShiftAmountTy(EVT LHSTy) {
673       assert(LHSTy.isInteger() && "Shift amount is not an integer type!");
674       return TLI.getShiftAmountTy(LHSTy, DAG.getDataLayout(), LegalTypes);
675     }
676 
677     /// This method returns true if we are running before type legalization or
678     /// if the specified VT is legal.
679     bool isTypeLegal(const EVT &VT) {
680       if (!LegalTypes) return true;
681       return TLI.isTypeLegal(VT);
682     }
683 
684     /// Convenience wrapper around TargetLowering::getSetCCResultType
685     EVT getSetCCResultType(EVT VT) const {
686       return TLI.getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
687     }
688 
689     void ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs,
690                          SDValue OrigLoad, SDValue ExtLoad,
691                          ISD::NodeType ExtType);
692   };
693 
694 /// This class is a DAGUpdateListener that removes any deleted
695 /// nodes from the worklist.
696 class WorklistRemover : public SelectionDAG::DAGUpdateListener {
697   DAGCombiner &DC;
698 
699 public:
700   explicit WorklistRemover(DAGCombiner &dc)
701     : SelectionDAG::DAGUpdateListener(dc.getDAG()), DC(dc) {}
702 
703   void NodeDeleted(SDNode *N, SDNode *E) override {
704     DC.removeFromWorklist(N);
705   }
706 };
707 
708 class WorklistInserter : public SelectionDAG::DAGUpdateListener {
709   DAGCombiner &DC;
710 
711 public:
712   explicit WorklistInserter(DAGCombiner &dc)
713       : SelectionDAG::DAGUpdateListener(dc.getDAG()), DC(dc) {}
714 
715   // FIXME: Ideally we could add N to the worklist, but this causes exponential
716   //        compile time costs in large DAGs, e.g. Halide.
717   void NodeInserted(SDNode *N) override { DC.ConsiderForPruning(N); }
718 };
719 
720 } // end anonymous namespace
721 
722 //===----------------------------------------------------------------------===//
723 //  TargetLowering::DAGCombinerInfo implementation
724 //===----------------------------------------------------------------------===//
725 
726 void TargetLowering::DAGCombinerInfo::AddToWorklist(SDNode *N) {
727   ((DAGCombiner*)DC)->AddToWorklist(N);
728 }
729 
730 SDValue TargetLowering::DAGCombinerInfo::
731 CombineTo(SDNode *N, ArrayRef<SDValue> To, bool AddTo) {
732   return ((DAGCombiner*)DC)->CombineTo(N, &To[0], To.size(), AddTo);
733 }
734 
735 SDValue TargetLowering::DAGCombinerInfo::
736 CombineTo(SDNode *N, SDValue Res, bool AddTo) {
737   return ((DAGCombiner*)DC)->CombineTo(N, Res, AddTo);
738 }
739 
740 SDValue TargetLowering::DAGCombinerInfo::
741 CombineTo(SDNode *N, SDValue Res0, SDValue Res1, bool AddTo) {
742   return ((DAGCombiner*)DC)->CombineTo(N, Res0, Res1, AddTo);
743 }
744 
745 void TargetLowering::DAGCombinerInfo::
746 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
747   return ((DAGCombiner*)DC)->CommitTargetLoweringOpt(TLO);
748 }
749 
750 //===----------------------------------------------------------------------===//
751 // Helper Functions
752 //===----------------------------------------------------------------------===//
753 
754 void DAGCombiner::deleteAndRecombine(SDNode *N) {
755   removeFromWorklist(N);
756 
757   // If the operands of this node are only used by the node, they will now be
758   // dead. Make sure to re-visit them and recursively delete dead nodes.
759   for (const SDValue &Op : N->ops())
760     // For an operand generating multiple values, one of the values may
761     // become dead allowing further simplification (e.g. split index
762     // arithmetic from an indexed load).
763     if (Op->hasOneUse() || Op->getNumValues() > 1)
764       AddToWorklist(Op.getNode());
765 
766   DAG.DeleteNode(N);
767 }
768 
769 /// Return 1 if we can compute the negated form of the specified expression for
770 /// the same cost as the expression itself, or 2 if we can compute the negated
771 /// form more cheaply than the expression itself.
772 static char isNegatibleForFree(SDValue Op, bool LegalOperations,
773                                const TargetLowering &TLI,
774                                const TargetOptions *Options,
775                                bool ForCodeSize,
776                                unsigned Depth = 0) {
777   // fneg is removable even if it has multiple uses.
778   if (Op.getOpcode() == ISD::FNEG)
779     return 2;
780 
781   // Don't allow anything with multiple uses unless we know it is free.
782   EVT VT = Op.getValueType();
783   const SDNodeFlags Flags = Op->getFlags();
784   if (!Op.hasOneUse() &&
785       !(Op.getOpcode() == ISD::FP_EXTEND &&
786         TLI.isFPExtFree(VT, Op.getOperand(0).getValueType())))
787     return 0;
788 
789   // Don't recurse exponentially.
790   if (Depth > 6)
791     return 0;
792 
793   switch (Op.getOpcode()) {
794   default: return false;
795   case ISD::ConstantFP: {
796     if (!LegalOperations)
797       return 1;
798 
799     // Don't invert constant FP values after legalization unless the target says
800     // the negated constant is legal.
801     return TLI.isOperationLegal(ISD::ConstantFP, VT) ||
802            TLI.isFPImmLegal(neg(cast<ConstantFPSDNode>(Op)->getValueAPF()), VT,
803                             ForCodeSize);
804   }
805   case ISD::BUILD_VECTOR: {
806     // Only permit BUILD_VECTOR of constants.
807     if (llvm::any_of(Op->op_values(), [&](SDValue N) {
808           return !N.isUndef() && !isa<ConstantFPSDNode>(N);
809         }))
810       return 0;
811     if (!LegalOperations)
812       return 1;
813     if (TLI.isOperationLegal(ISD::ConstantFP, VT) &&
814         TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
815       return 1;
816     return llvm::all_of(Op->op_values(), [&](SDValue N) {
817       return N.isUndef() ||
818              TLI.isFPImmLegal(neg(cast<ConstantFPSDNode>(N)->getValueAPF()), VT,
819                               ForCodeSize);
820     });
821   }
822   case ISD::FADD:
823     if (!Options->UnsafeFPMath && !Flags.hasNoSignedZeros())
824       return 0;
825 
826     // After operation legalization, it might not be legal to create new FSUBs.
827     if (LegalOperations && !TLI.isOperationLegalOrCustom(ISD::FSUB, VT))
828       return 0;
829 
830     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
831     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
832                                     Options, ForCodeSize, Depth + 1))
833       return V;
834     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
835     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
836                               ForCodeSize, Depth + 1);
837   case ISD::FSUB:
838     // We can't turn -(A-B) into B-A when we honor signed zeros.
839     if (!Options->NoSignedZerosFPMath && !Flags.hasNoSignedZeros())
840       return 0;
841 
842     // fold (fneg (fsub A, B)) -> (fsub B, A)
843     return 1;
844 
845   case ISD::FMUL:
846   case ISD::FDIV:
847     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y) or (fmul X, (fneg Y))
848     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
849                                     Options, ForCodeSize, Depth + 1))
850       return V;
851 
852     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
853                               ForCodeSize, Depth + 1);
854 
855   case ISD::FP_EXTEND:
856   case ISD::FP_ROUND:
857   case ISD::FSIN:
858     return isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI, Options,
859                               ForCodeSize, Depth + 1);
860   }
861 }
862 
863 /// If isNegatibleForFree returns true, return the newly negated expression.
864 static SDValue GetNegatedExpression(SDValue Op, SelectionDAG &DAG,
865                                     bool LegalOperations, bool ForCodeSize,
866                                     unsigned Depth = 0) {
867   // fneg is removable even if it has multiple uses.
868   if (Op.getOpcode() == ISD::FNEG)
869     return Op.getOperand(0);
870 
871   assert(Depth <= 6 && "GetNegatedExpression doesn't match isNegatibleForFree");
872   const TargetOptions &Options = DAG.getTarget().Options;
873   const SDNodeFlags Flags = Op->getFlags();
874 
875   switch (Op.getOpcode()) {
876   default: llvm_unreachable("Unknown code");
877   case ISD::ConstantFP: {
878     APFloat V = cast<ConstantFPSDNode>(Op)->getValueAPF();
879     V.changeSign();
880     return DAG.getConstantFP(V, SDLoc(Op), Op.getValueType());
881   }
882   case ISD::BUILD_VECTOR: {
883     SmallVector<SDValue, 4> Ops;
884     for (SDValue C : Op->op_values()) {
885       if (C.isUndef()) {
886         Ops.push_back(C);
887         continue;
888       }
889       APFloat V = cast<ConstantFPSDNode>(C)->getValueAPF();
890       V.changeSign();
891       Ops.push_back(DAG.getConstantFP(V, SDLoc(Op), C.getValueType()));
892     }
893     return DAG.getBuildVector(Op.getValueType(), SDLoc(Op), Ops);
894   }
895   case ISD::FADD:
896     assert(Options.UnsafeFPMath || Flags.hasNoSignedZeros());
897 
898     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
899     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
900                            DAG.getTargetLoweringInfo(), &Options, ForCodeSize,
901                            Depth + 1))
902       return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
903                          GetNegatedExpression(Op.getOperand(0), DAG,
904                                               LegalOperations, ForCodeSize,
905                                               Depth + 1),
906                          Op.getOperand(1), Flags);
907     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
908     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
909                        GetNegatedExpression(Op.getOperand(1), DAG,
910                                             LegalOperations, ForCodeSize,
911                                             Depth + 1),
912                        Op.getOperand(0), Flags);
913   case ISD::FSUB:
914     // fold (fneg (fsub 0, B)) -> B
915     if (ConstantFPSDNode *N0CFP =
916             isConstOrConstSplatFP(Op.getOperand(0), /*AllowUndefs*/ true))
917       if (N0CFP->isZero())
918         return Op.getOperand(1);
919 
920     // fold (fneg (fsub A, B)) -> (fsub B, A)
921     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
922                        Op.getOperand(1), Op.getOperand(0), Flags);
923 
924   case ISD::FMUL:
925   case ISD::FDIV:
926     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y)
927     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
928                            DAG.getTargetLoweringInfo(), &Options, ForCodeSize,
929                            Depth + 1))
930       return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
931                          GetNegatedExpression(Op.getOperand(0), DAG,
932                                               LegalOperations, ForCodeSize,
933                                               Depth + 1),
934                          Op.getOperand(1), Flags);
935 
936     // fold (fneg (fmul X, Y)) -> (fmul X, (fneg Y))
937     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
938                        Op.getOperand(0),
939                        GetNegatedExpression(Op.getOperand(1), DAG,
940                                             LegalOperations, ForCodeSize,
941                                             Depth + 1), Flags);
942 
943   case ISD::FP_EXTEND:
944   case ISD::FSIN:
945     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
946                        GetNegatedExpression(Op.getOperand(0), DAG,
947                                             LegalOperations, ForCodeSize,
948                                             Depth + 1));
949   case ISD::FP_ROUND:
950     return DAG.getNode(ISD::FP_ROUND, SDLoc(Op), Op.getValueType(),
951                        GetNegatedExpression(Op.getOperand(0), DAG,
952                                             LegalOperations, ForCodeSize,
953                                             Depth + 1),
954                        Op.getOperand(1));
955   }
956 }
957 
958 // APInts must be the same size for most operations, this helper
959 // function zero extends the shorter of the pair so that they match.
960 // We provide an Offset so that we can create bitwidths that won't overflow.
961 static void zeroExtendToMatch(APInt &LHS, APInt &RHS, unsigned Offset = 0) {
962   unsigned Bits = Offset + std::max(LHS.getBitWidth(), RHS.getBitWidth());
963   LHS = LHS.zextOrSelf(Bits);
964   RHS = RHS.zextOrSelf(Bits);
965 }
966 
967 // Return true if this node is a setcc, or is a select_cc
968 // that selects between the target values used for true and false, making it
969 // equivalent to a setcc. Also, set the incoming LHS, RHS, and CC references to
970 // the appropriate nodes based on the type of node we are checking. This
971 // simplifies life a bit for the callers.
972 bool DAGCombiner::isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
973                                     SDValue &CC) const {
974   if (N.getOpcode() == ISD::SETCC) {
975     LHS = N.getOperand(0);
976     RHS = N.getOperand(1);
977     CC  = N.getOperand(2);
978     return true;
979   }
980 
981   if (N.getOpcode() != ISD::SELECT_CC ||
982       !TLI.isConstTrueVal(N.getOperand(2).getNode()) ||
983       !TLI.isConstFalseVal(N.getOperand(3).getNode()))
984     return false;
985 
986   if (TLI.getBooleanContents(N.getValueType()) ==
987       TargetLowering::UndefinedBooleanContent)
988     return false;
989 
990   LHS = N.getOperand(0);
991   RHS = N.getOperand(1);
992   CC  = N.getOperand(4);
993   return true;
994 }
995 
996 /// Return true if this is a SetCC-equivalent operation with only one use.
997 /// If this is true, it allows the users to invert the operation for free when
998 /// it is profitable to do so.
999 bool DAGCombiner::isOneUseSetCC(SDValue N) const {
1000   SDValue N0, N1, N2;
1001   if (isSetCCEquivalent(N, N0, N1, N2) && N.getNode()->hasOneUse())
1002     return true;
1003   return false;
1004 }
1005 
1006 // Returns the SDNode if it is a constant float BuildVector
1007 // or constant float.
1008 static SDNode *isConstantFPBuildVectorOrConstantFP(SDValue N) {
1009   if (isa<ConstantFPSDNode>(N))
1010     return N.getNode();
1011   if (ISD::isBuildVectorOfConstantFPSDNodes(N.getNode()))
1012     return N.getNode();
1013   return nullptr;
1014 }
1015 
1016 // Determines if it is a constant integer or a build vector of constant
1017 // integers (and undefs).
1018 // Do not permit build vector implicit truncation.
1019 static bool isConstantOrConstantVector(SDValue N, bool NoOpaques = false) {
1020   if (ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N))
1021     return !(Const->isOpaque() && NoOpaques);
1022   if (N.getOpcode() != ISD::BUILD_VECTOR)
1023     return false;
1024   unsigned BitWidth = N.getScalarValueSizeInBits();
1025   for (const SDValue &Op : N->op_values()) {
1026     if (Op.isUndef())
1027       continue;
1028     ConstantSDNode *Const = dyn_cast<ConstantSDNode>(Op);
1029     if (!Const || Const->getAPIntValue().getBitWidth() != BitWidth ||
1030         (Const->isOpaque() && NoOpaques))
1031       return false;
1032   }
1033   return true;
1034 }
1035 
1036 // Determines if a BUILD_VECTOR is composed of all-constants possibly mixed with
1037 // undef's.
1038 static bool isAnyConstantBuildVector(SDValue V, bool NoOpaques = false) {
1039   if (V.getOpcode() != ISD::BUILD_VECTOR)
1040     return false;
1041   return isConstantOrConstantVector(V, NoOpaques) ||
1042          ISD::isBuildVectorOfConstantFPSDNodes(V.getNode());
1043 }
1044 
1045 bool DAGCombiner::reassociationCanBreakAddressingModePattern(unsigned Opc,
1046                                                              const SDLoc &DL,
1047                                                              SDValue N0,
1048                                                              SDValue N1) {
1049   // Currently this only tries to ensure we don't undo the GEP splits done by
1050   // CodeGenPrepare when shouldConsiderGEPOffsetSplit is true. To ensure this,
1051   // we check if the following transformation would be problematic:
1052   // (load/store (add, (add, x, offset1), offset2)) ->
1053   // (load/store (add, x, offset1+offset2)).
1054 
1055   if (Opc != ISD::ADD || N0.getOpcode() != ISD::ADD)
1056     return false;
1057 
1058   if (N0.hasOneUse())
1059     return false;
1060 
1061   auto *C1 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
1062   auto *C2 = dyn_cast<ConstantSDNode>(N1);
1063   if (!C1 || !C2)
1064     return false;
1065 
1066   const APInt &C1APIntVal = C1->getAPIntValue();
1067   const APInt &C2APIntVal = C2->getAPIntValue();
1068   if (C1APIntVal.getBitWidth() > 64 || C2APIntVal.getBitWidth() > 64)
1069     return false;
1070 
1071   const APInt CombinedValueIntVal = C1APIntVal + C2APIntVal;
1072   if (CombinedValueIntVal.getBitWidth() > 64)
1073     return false;
1074   const int64_t CombinedValue = CombinedValueIntVal.getSExtValue();
1075 
1076   for (SDNode *Node : N0->uses()) {
1077     auto LoadStore = dyn_cast<MemSDNode>(Node);
1078     if (LoadStore) {
1079       // Is x[offset2] already not a legal addressing mode? If so then
1080       // reassociating the constants breaks nothing (we test offset2 because
1081       // that's the one we hope to fold into the load or store).
1082       TargetLoweringBase::AddrMode AM;
1083       AM.HasBaseReg = true;
1084       AM.BaseOffs = C2APIntVal.getSExtValue();
1085       EVT VT = LoadStore->getMemoryVT();
1086       unsigned AS = LoadStore->getAddressSpace();
1087       Type *AccessTy = VT.getTypeForEVT(*DAG.getContext());
1088       if (!TLI.isLegalAddressingMode(DAG.getDataLayout(), AM, AccessTy, AS))
1089         continue;
1090 
1091       // Would x[offset1+offset2] still be a legal addressing mode?
1092       AM.BaseOffs = CombinedValue;
1093       if (!TLI.isLegalAddressingMode(DAG.getDataLayout(), AM, AccessTy, AS))
1094         return true;
1095     }
1096   }
1097 
1098   return false;
1099 }
1100 
1101 // Helper for DAGCombiner::reassociateOps. Try to reassociate an expression
1102 // such as (Opc N0, N1), if \p N0 is the same kind of operation as \p Opc.
1103 SDValue DAGCombiner::reassociateOpsCommutative(unsigned Opc, const SDLoc &DL,
1104                                                SDValue N0, SDValue N1) {
1105   EVT VT = N0.getValueType();
1106 
1107   if (N0.getOpcode() != Opc)
1108     return SDValue();
1109 
1110   // Don't reassociate reductions.
1111   if (N0->getFlags().hasVectorReduction())
1112     return SDValue();
1113 
1114   if (SDNode *C1 = DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1))) {
1115     if (SDNode *C2 = DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
1116       // Reassociate: (op (op x, c1), c2) -> (op x, (op c1, c2))
1117       if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, C1, C2))
1118         return DAG.getNode(Opc, DL, VT, N0.getOperand(0), OpNode);
1119       return SDValue();
1120     }
1121     if (N0.hasOneUse()) {
1122       // Reassociate: (op (op x, c1), y) -> (op (op x, y), c1)
1123       //              iff (op x, c1) has one use
1124       SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0.getOperand(0), N1);
1125       if (!OpNode.getNode())
1126         return SDValue();
1127       AddToWorklist(OpNode.getNode());
1128       return DAG.getNode(Opc, DL, VT, OpNode, N0.getOperand(1));
1129     }
1130   }
1131   return SDValue();
1132 }
1133 
1134 // Try to reassociate commutative binops.
1135 SDValue DAGCombiner::reassociateOps(unsigned Opc, const SDLoc &DL, SDValue N0,
1136                                     SDValue N1, SDNodeFlags Flags) {
1137   assert(TLI.isCommutativeBinOp(Opc) && "Operation not commutative.");
1138   // Don't reassociate reductions.
1139   if (Flags.hasVectorReduction())
1140     return SDValue();
1141 
1142   // Floating-point reassociation is not allowed without loose FP math.
1143   if (N0.getValueType().isFloatingPoint() ||
1144       N1.getValueType().isFloatingPoint())
1145     if (!Flags.hasAllowReassociation() || !Flags.hasNoSignedZeros())
1146       return SDValue();
1147 
1148   if (SDValue Combined = reassociateOpsCommutative(Opc, DL, N0, N1))
1149     return Combined;
1150   if (SDValue Combined = reassociateOpsCommutative(Opc, DL, N1, N0))
1151     return Combined;
1152   return SDValue();
1153 }
1154 
1155 SDValue DAGCombiner::CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
1156                                bool AddTo) {
1157   assert(N->getNumValues() == NumTo && "Broken CombineTo call!");
1158   ++NodesCombined;
1159   LLVM_DEBUG(dbgs() << "\nReplacing.1 "; N->dump(&DAG); dbgs() << "\nWith: ";
1160              To[0].getNode()->dump(&DAG);
1161              dbgs() << " and " << NumTo - 1 << " other values\n");
1162   for (unsigned i = 0, e = NumTo; i != e; ++i)
1163     assert((!To[i].getNode() ||
1164             N->getValueType(i) == To[i].getValueType()) &&
1165            "Cannot combine value to value of different type!");
1166 
1167   WorklistRemover DeadNodes(*this);
1168   DAG.ReplaceAllUsesWith(N, To);
1169   if (AddTo) {
1170     // Push the new nodes and any users onto the worklist
1171     for (unsigned i = 0, e = NumTo; i != e; ++i) {
1172       if (To[i].getNode()) {
1173         AddToWorklist(To[i].getNode());
1174         AddUsersToWorklist(To[i].getNode());
1175       }
1176     }
1177   }
1178 
1179   // Finally, if the node is now dead, remove it from the graph.  The node
1180   // may not be dead if the replacement process recursively simplified to
1181   // something else needing this node.
1182   if (N->use_empty())
1183     deleteAndRecombine(N);
1184   return SDValue(N, 0);
1185 }
1186 
1187 void DAGCombiner::
1188 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
1189   // Replace all uses.  If any nodes become isomorphic to other nodes and
1190   // are deleted, make sure to remove them from our worklist.
1191   WorklistRemover DeadNodes(*this);
1192   DAG.ReplaceAllUsesOfValueWith(TLO.Old, TLO.New);
1193 
1194   // Push the new node and any (possibly new) users onto the worklist.
1195   AddToWorklist(TLO.New.getNode());
1196   AddUsersToWorklist(TLO.New.getNode());
1197 
1198   // Finally, if the node is now dead, remove it from the graph.  The node
1199   // may not be dead if the replacement process recursively simplified to
1200   // something else needing this node.
1201   if (TLO.Old.getNode()->use_empty())
1202     deleteAndRecombine(TLO.Old.getNode());
1203 }
1204 
1205 /// Check the specified integer node value to see if it can be simplified or if
1206 /// things it uses can be simplified by bit propagation. If so, return true.
1207 bool DAGCombiner::SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits,
1208                                        const APInt &DemandedElts) {
1209   TargetLowering::TargetLoweringOpt TLO(DAG, LegalTypes, LegalOperations);
1210   KnownBits Known;
1211   if (!TLI.SimplifyDemandedBits(Op, DemandedBits, DemandedElts, Known, TLO))
1212     return false;
1213 
1214   // Revisit the node.
1215   AddToWorklist(Op.getNode());
1216 
1217   // Replace the old value with the new one.
1218   ++NodesCombined;
1219   LLVM_DEBUG(dbgs() << "\nReplacing.2 "; TLO.Old.getNode()->dump(&DAG);
1220              dbgs() << "\nWith: "; TLO.New.getNode()->dump(&DAG);
1221              dbgs() << '\n');
1222 
1223   CommitTargetLoweringOpt(TLO);
1224   return true;
1225 }
1226 
1227 /// Check the specified vector node value to see if it can be simplified or
1228 /// if things it uses can be simplified as it only uses some of the elements.
1229 /// If so, return true.
1230 bool DAGCombiner::SimplifyDemandedVectorElts(SDValue Op,
1231                                              const APInt &DemandedElts,
1232                                              bool AssumeSingleUse) {
1233   TargetLowering::TargetLoweringOpt TLO(DAG, LegalTypes, LegalOperations);
1234   APInt KnownUndef, KnownZero;
1235   if (!TLI.SimplifyDemandedVectorElts(Op, DemandedElts, KnownUndef, KnownZero,
1236                                       TLO, 0, AssumeSingleUse))
1237     return false;
1238 
1239   // Revisit the node.
1240   AddToWorklist(Op.getNode());
1241 
1242   // Replace the old value with the new one.
1243   ++NodesCombined;
1244   LLVM_DEBUG(dbgs() << "\nReplacing.2 "; TLO.Old.getNode()->dump(&DAG);
1245              dbgs() << "\nWith: "; TLO.New.getNode()->dump(&DAG);
1246              dbgs() << '\n');
1247 
1248   CommitTargetLoweringOpt(TLO);
1249   return true;
1250 }
1251 
1252 void DAGCombiner::ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad) {
1253   SDLoc DL(Load);
1254   EVT VT = Load->getValueType(0);
1255   SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, VT, SDValue(ExtLoad, 0));
1256 
1257   LLVM_DEBUG(dbgs() << "\nReplacing.9 "; Load->dump(&DAG); dbgs() << "\nWith: ";
1258              Trunc.getNode()->dump(&DAG); dbgs() << '\n');
1259   WorklistRemover DeadNodes(*this);
1260   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 0), Trunc);
1261   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), SDValue(ExtLoad, 1));
1262   deleteAndRecombine(Load);
1263   AddToWorklist(Trunc.getNode());
1264 }
1265 
1266 SDValue DAGCombiner::PromoteOperand(SDValue Op, EVT PVT, bool &Replace) {
1267   Replace = false;
1268   SDLoc DL(Op);
1269   if (ISD::isUNINDEXEDLoad(Op.getNode())) {
1270     LoadSDNode *LD = cast<LoadSDNode>(Op);
1271     EVT MemVT = LD->getMemoryVT();
1272     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD) ? ISD::EXTLOAD
1273                                                       : LD->getExtensionType();
1274     Replace = true;
1275     return DAG.getExtLoad(ExtType, DL, PVT,
1276                           LD->getChain(), LD->getBasePtr(),
1277                           MemVT, LD->getMemOperand());
1278   }
1279 
1280   unsigned Opc = Op.getOpcode();
1281   switch (Opc) {
1282   default: break;
1283   case ISD::AssertSext:
1284     if (SDValue Op0 = SExtPromoteOperand(Op.getOperand(0), PVT))
1285       return DAG.getNode(ISD::AssertSext, DL, PVT, Op0, Op.getOperand(1));
1286     break;
1287   case ISD::AssertZext:
1288     if (SDValue Op0 = ZExtPromoteOperand(Op.getOperand(0), PVT))
1289       return DAG.getNode(ISD::AssertZext, DL, PVT, Op0, Op.getOperand(1));
1290     break;
1291   case ISD::Constant: {
1292     unsigned ExtOpc =
1293       Op.getValueType().isByteSized() ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
1294     return DAG.getNode(ExtOpc, DL, PVT, Op);
1295   }
1296   }
1297 
1298   if (!TLI.isOperationLegal(ISD::ANY_EXTEND, PVT))
1299     return SDValue();
1300   return DAG.getNode(ISD::ANY_EXTEND, DL, PVT, Op);
1301 }
1302 
1303 SDValue DAGCombiner::SExtPromoteOperand(SDValue Op, EVT PVT) {
1304   if (!TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, PVT))
1305     return SDValue();
1306   EVT OldVT = Op.getValueType();
1307   SDLoc DL(Op);
1308   bool Replace = false;
1309   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1310   if (!NewOp.getNode())
1311     return SDValue();
1312   AddToWorklist(NewOp.getNode());
1313 
1314   if (Replace)
1315     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1316   return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, NewOp.getValueType(), NewOp,
1317                      DAG.getValueType(OldVT));
1318 }
1319 
1320 SDValue DAGCombiner::ZExtPromoteOperand(SDValue Op, EVT PVT) {
1321   EVT OldVT = Op.getValueType();
1322   SDLoc DL(Op);
1323   bool Replace = false;
1324   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1325   if (!NewOp.getNode())
1326     return SDValue();
1327   AddToWorklist(NewOp.getNode());
1328 
1329   if (Replace)
1330     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1331   return DAG.getZeroExtendInReg(NewOp, DL, OldVT);
1332 }
1333 
1334 /// Promote the specified integer binary operation if the target indicates it is
1335 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1336 /// i32 since i16 instructions are longer.
1337 SDValue DAGCombiner::PromoteIntBinOp(SDValue Op) {
1338   if (!LegalOperations)
1339     return SDValue();
1340 
1341   EVT VT = Op.getValueType();
1342   if (VT.isVector() || !VT.isInteger())
1343     return SDValue();
1344 
1345   // If operation type is 'undesirable', e.g. i16 on x86, consider
1346   // promoting it.
1347   unsigned Opc = Op.getOpcode();
1348   if (TLI.isTypeDesirableForOp(Opc, VT))
1349     return SDValue();
1350 
1351   EVT PVT = VT;
1352   // Consult target whether it is a good idea to promote this operation and
1353   // what's the right type to promote it to.
1354   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1355     assert(PVT != VT && "Don't know what type to promote to!");
1356 
1357     LLVM_DEBUG(dbgs() << "\nPromoting "; Op.getNode()->dump(&DAG));
1358 
1359     bool Replace0 = false;
1360     SDValue N0 = Op.getOperand(0);
1361     SDValue NN0 = PromoteOperand(N0, PVT, Replace0);
1362 
1363     bool Replace1 = false;
1364     SDValue N1 = Op.getOperand(1);
1365     SDValue NN1 = PromoteOperand(N1, PVT, Replace1);
1366     SDLoc DL(Op);
1367 
1368     SDValue RV =
1369         DAG.getNode(ISD::TRUNCATE, DL, VT, DAG.getNode(Opc, DL, PVT, NN0, NN1));
1370 
1371     // We are always replacing N0/N1's use in N and only need
1372     // additional replacements if there are additional uses.
1373     Replace0 &= !N0->hasOneUse();
1374     Replace1 &= (N0 != N1) && !N1->hasOneUse();
1375 
1376     // Combine Op here so it is preserved past replacements.
1377     CombineTo(Op.getNode(), RV);
1378 
1379     // If operands have a use ordering, make sure we deal with
1380     // predecessor first.
1381     if (Replace0 && Replace1 && N0.getNode()->isPredecessorOf(N1.getNode())) {
1382       std::swap(N0, N1);
1383       std::swap(NN0, NN1);
1384     }
1385 
1386     if (Replace0) {
1387       AddToWorklist(NN0.getNode());
1388       ReplaceLoadWithPromotedLoad(N0.getNode(), NN0.getNode());
1389     }
1390     if (Replace1) {
1391       AddToWorklist(NN1.getNode());
1392       ReplaceLoadWithPromotedLoad(N1.getNode(), NN1.getNode());
1393     }
1394     return Op;
1395   }
1396   return SDValue();
1397 }
1398 
1399 /// Promote the specified integer shift operation if the target indicates it is
1400 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1401 /// i32 since i16 instructions are longer.
1402 SDValue DAGCombiner::PromoteIntShiftOp(SDValue Op) {
1403   if (!LegalOperations)
1404     return SDValue();
1405 
1406   EVT VT = Op.getValueType();
1407   if (VT.isVector() || !VT.isInteger())
1408     return SDValue();
1409 
1410   // If operation type is 'undesirable', e.g. i16 on x86, consider
1411   // promoting it.
1412   unsigned Opc = Op.getOpcode();
1413   if (TLI.isTypeDesirableForOp(Opc, VT))
1414     return SDValue();
1415 
1416   EVT PVT = VT;
1417   // Consult target whether it is a good idea to promote this operation and
1418   // what's the right type to promote it to.
1419   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1420     assert(PVT != VT && "Don't know what type to promote to!");
1421 
1422     LLVM_DEBUG(dbgs() << "\nPromoting "; Op.getNode()->dump(&DAG));
1423 
1424     bool Replace = false;
1425     SDValue N0 = Op.getOperand(0);
1426     SDValue N1 = Op.getOperand(1);
1427     if (Opc == ISD::SRA)
1428       N0 = SExtPromoteOperand(N0, PVT);
1429     else if (Opc == ISD::SRL)
1430       N0 = ZExtPromoteOperand(N0, PVT);
1431     else
1432       N0 = PromoteOperand(N0, PVT, Replace);
1433 
1434     if (!N0.getNode())
1435       return SDValue();
1436 
1437     SDLoc DL(Op);
1438     SDValue RV =
1439         DAG.getNode(ISD::TRUNCATE, DL, VT, DAG.getNode(Opc, DL, PVT, N0, N1));
1440 
1441     AddToWorklist(N0.getNode());
1442     if (Replace)
1443       ReplaceLoadWithPromotedLoad(Op.getOperand(0).getNode(), N0.getNode());
1444 
1445     // Deal with Op being deleted.
1446     if (Op && Op.getOpcode() != ISD::DELETED_NODE)
1447       return RV;
1448   }
1449   return SDValue();
1450 }
1451 
1452 SDValue DAGCombiner::PromoteExtend(SDValue Op) {
1453   if (!LegalOperations)
1454     return SDValue();
1455 
1456   EVT VT = Op.getValueType();
1457   if (VT.isVector() || !VT.isInteger())
1458     return SDValue();
1459 
1460   // If operation type is 'undesirable', e.g. i16 on x86, consider
1461   // promoting it.
1462   unsigned Opc = Op.getOpcode();
1463   if (TLI.isTypeDesirableForOp(Opc, VT))
1464     return SDValue();
1465 
1466   EVT PVT = VT;
1467   // Consult target whether it is a good idea to promote this operation and
1468   // what's the right type to promote it to.
1469   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1470     assert(PVT != VT && "Don't know what type to promote to!");
1471     // fold (aext (aext x)) -> (aext x)
1472     // fold (aext (zext x)) -> (zext x)
1473     // fold (aext (sext x)) -> (sext x)
1474     LLVM_DEBUG(dbgs() << "\nPromoting "; Op.getNode()->dump(&DAG));
1475     return DAG.getNode(Op.getOpcode(), SDLoc(Op), VT, Op.getOperand(0));
1476   }
1477   return SDValue();
1478 }
1479 
1480 bool DAGCombiner::PromoteLoad(SDValue Op) {
1481   if (!LegalOperations)
1482     return false;
1483 
1484   if (!ISD::isUNINDEXEDLoad(Op.getNode()))
1485     return false;
1486 
1487   EVT VT = Op.getValueType();
1488   if (VT.isVector() || !VT.isInteger())
1489     return false;
1490 
1491   // If operation type is 'undesirable', e.g. i16 on x86, consider
1492   // promoting it.
1493   unsigned Opc = Op.getOpcode();
1494   if (TLI.isTypeDesirableForOp(Opc, VT))
1495     return false;
1496 
1497   EVT PVT = VT;
1498   // Consult target whether it is a good idea to promote this operation and
1499   // what's the right type to promote it to.
1500   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1501     assert(PVT != VT && "Don't know what type to promote to!");
1502 
1503     SDLoc DL(Op);
1504     SDNode *N = Op.getNode();
1505     LoadSDNode *LD = cast<LoadSDNode>(N);
1506     EVT MemVT = LD->getMemoryVT();
1507     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD) ? ISD::EXTLOAD
1508                                                       : LD->getExtensionType();
1509     SDValue NewLD = DAG.getExtLoad(ExtType, DL, PVT,
1510                                    LD->getChain(), LD->getBasePtr(),
1511                                    MemVT, LD->getMemOperand());
1512     SDValue Result = DAG.getNode(ISD::TRUNCATE, DL, VT, NewLD);
1513 
1514     LLVM_DEBUG(dbgs() << "\nPromoting "; N->dump(&DAG); dbgs() << "\nTo: ";
1515                Result.getNode()->dump(&DAG); dbgs() << '\n');
1516     WorklistRemover DeadNodes(*this);
1517     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result);
1518     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), NewLD.getValue(1));
1519     deleteAndRecombine(N);
1520     AddToWorklist(Result.getNode());
1521     return true;
1522   }
1523   return false;
1524 }
1525 
1526 /// Recursively delete a node which has no uses and any operands for
1527 /// which it is the only use.
1528 ///
1529 /// Note that this both deletes the nodes and removes them from the worklist.
1530 /// It also adds any nodes who have had a user deleted to the worklist as they
1531 /// may now have only one use and subject to other combines.
1532 bool DAGCombiner::recursivelyDeleteUnusedNodes(SDNode *N) {
1533   if (!N->use_empty())
1534     return false;
1535 
1536   SmallSetVector<SDNode *, 16> Nodes;
1537   Nodes.insert(N);
1538   do {
1539     N = Nodes.pop_back_val();
1540     if (!N)
1541       continue;
1542 
1543     if (N->use_empty()) {
1544       for (const SDValue &ChildN : N->op_values())
1545         Nodes.insert(ChildN.getNode());
1546 
1547       removeFromWorklist(N);
1548       DAG.DeleteNode(N);
1549     } else {
1550       AddToWorklist(N);
1551     }
1552   } while (!Nodes.empty());
1553   return true;
1554 }
1555 
1556 //===----------------------------------------------------------------------===//
1557 //  Main DAG Combiner implementation
1558 //===----------------------------------------------------------------------===//
1559 
1560 void DAGCombiner::Run(CombineLevel AtLevel) {
1561   // set the instance variables, so that the various visit routines may use it.
1562   Level = AtLevel;
1563   LegalOperations = Level >= AfterLegalizeVectorOps;
1564   LegalTypes = Level >= AfterLegalizeTypes;
1565 
1566   WorklistInserter AddNodes(*this);
1567 
1568   // Add all the dag nodes to the worklist.
1569   for (SDNode &Node : DAG.allnodes())
1570     AddToWorklist(&Node);
1571 
1572   // Create a dummy node (which is not added to allnodes), that adds a reference
1573   // to the root node, preventing it from being deleted, and tracking any
1574   // changes of the root.
1575   HandleSDNode Dummy(DAG.getRoot());
1576 
1577   // While we have a valid worklist entry node, try to combine it.
1578   while (SDNode *N = getNextWorklistEntry()) {
1579     // If N has no uses, it is dead.  Make sure to revisit all N's operands once
1580     // N is deleted from the DAG, since they too may now be dead or may have a
1581     // reduced number of uses, allowing other xforms.
1582     if (recursivelyDeleteUnusedNodes(N))
1583       continue;
1584 
1585     WorklistRemover DeadNodes(*this);
1586 
1587     // If this combine is running after legalizing the DAG, re-legalize any
1588     // nodes pulled off the worklist.
1589     if (Level == AfterLegalizeDAG) {
1590       SmallSetVector<SDNode *, 16> UpdatedNodes;
1591       bool NIsValid = DAG.LegalizeOp(N, UpdatedNodes);
1592 
1593       for (SDNode *LN : UpdatedNodes) {
1594         AddToWorklist(LN);
1595         AddUsersToWorklist(LN);
1596       }
1597       if (!NIsValid)
1598         continue;
1599     }
1600 
1601     LLVM_DEBUG(dbgs() << "\nCombining: "; N->dump(&DAG));
1602 
1603     // Add any operands of the new node which have not yet been combined to the
1604     // worklist as well. Because the worklist uniques things already, this
1605     // won't repeatedly process the same operand.
1606     CombinedNodes.insert(N);
1607     for (const SDValue &ChildN : N->op_values())
1608       if (!CombinedNodes.count(ChildN.getNode()))
1609         AddToWorklist(ChildN.getNode());
1610 
1611     SDValue RV = combine(N);
1612 
1613     if (!RV.getNode())
1614       continue;
1615 
1616     ++NodesCombined;
1617 
1618     // If we get back the same node we passed in, rather than a new node or
1619     // zero, we know that the node must have defined multiple values and
1620     // CombineTo was used.  Since CombineTo takes care of the worklist
1621     // mechanics for us, we have no work to do in this case.
1622     if (RV.getNode() == N)
1623       continue;
1624 
1625     assert(N->getOpcode() != ISD::DELETED_NODE &&
1626            RV.getOpcode() != ISD::DELETED_NODE &&
1627            "Node was deleted but visit returned new node!");
1628 
1629     LLVM_DEBUG(dbgs() << " ... into: "; RV.getNode()->dump(&DAG));
1630 
1631     if (N->getNumValues() == RV.getNode()->getNumValues())
1632       DAG.ReplaceAllUsesWith(N, RV.getNode());
1633     else {
1634       assert(N->getValueType(0) == RV.getValueType() &&
1635              N->getNumValues() == 1 && "Type mismatch");
1636       DAG.ReplaceAllUsesWith(N, &RV);
1637     }
1638 
1639     // Push the new node and any users onto the worklist
1640     AddToWorklist(RV.getNode());
1641     AddUsersToWorklist(RV.getNode());
1642 
1643     // Finally, if the node is now dead, remove it from the graph.  The node
1644     // may not be dead if the replacement process recursively simplified to
1645     // something else needing this node. This will also take care of adding any
1646     // operands which have lost a user to the worklist.
1647     recursivelyDeleteUnusedNodes(N);
1648   }
1649 
1650   // If the root changed (e.g. it was a dead load, update the root).
1651   DAG.setRoot(Dummy.getValue());
1652   DAG.RemoveDeadNodes();
1653 }
1654 
1655 SDValue DAGCombiner::visit(SDNode *N) {
1656   switch (N->getOpcode()) {
1657   default: break;
1658   case ISD::TokenFactor:        return visitTokenFactor(N);
1659   case ISD::MERGE_VALUES:       return visitMERGE_VALUES(N);
1660   case ISD::ADD:                return visitADD(N);
1661   case ISD::SUB:                return visitSUB(N);
1662   case ISD::SADDSAT:
1663   case ISD::UADDSAT:            return visitADDSAT(N);
1664   case ISD::SSUBSAT:
1665   case ISD::USUBSAT:            return visitSUBSAT(N);
1666   case ISD::ADDC:               return visitADDC(N);
1667   case ISD::SADDO:
1668   case ISD::UADDO:              return visitADDO(N);
1669   case ISD::SUBC:               return visitSUBC(N);
1670   case ISD::SSUBO:
1671   case ISD::USUBO:              return visitSUBO(N);
1672   case ISD::ADDE:               return visitADDE(N);
1673   case ISD::ADDCARRY:           return visitADDCARRY(N);
1674   case ISD::SUBE:               return visitSUBE(N);
1675   case ISD::SUBCARRY:           return visitSUBCARRY(N);
1676   case ISD::MUL:                return visitMUL(N);
1677   case ISD::SDIV:               return visitSDIV(N);
1678   case ISD::UDIV:               return visitUDIV(N);
1679   case ISD::SREM:
1680   case ISD::UREM:               return visitREM(N);
1681   case ISD::MULHU:              return visitMULHU(N);
1682   case ISD::MULHS:              return visitMULHS(N);
1683   case ISD::SMUL_LOHI:          return visitSMUL_LOHI(N);
1684   case ISD::UMUL_LOHI:          return visitUMUL_LOHI(N);
1685   case ISD::SMULO:
1686   case ISD::UMULO:              return visitMULO(N);
1687   case ISD::SMIN:
1688   case ISD::SMAX:
1689   case ISD::UMIN:
1690   case ISD::UMAX:               return visitIMINMAX(N);
1691   case ISD::AND:                return visitAND(N);
1692   case ISD::OR:                 return visitOR(N);
1693   case ISD::XOR:                return visitXOR(N);
1694   case ISD::SHL:                return visitSHL(N);
1695   case ISD::SRA:                return visitSRA(N);
1696   case ISD::SRL:                return visitSRL(N);
1697   case ISD::ROTR:
1698   case ISD::ROTL:               return visitRotate(N);
1699   case ISD::FSHL:
1700   case ISD::FSHR:               return visitFunnelShift(N);
1701   case ISD::ABS:                return visitABS(N);
1702   case ISD::BSWAP:              return visitBSWAP(N);
1703   case ISD::BITREVERSE:         return visitBITREVERSE(N);
1704   case ISD::CTLZ:               return visitCTLZ(N);
1705   case ISD::CTLZ_ZERO_UNDEF:    return visitCTLZ_ZERO_UNDEF(N);
1706   case ISD::CTTZ:               return visitCTTZ(N);
1707   case ISD::CTTZ_ZERO_UNDEF:    return visitCTTZ_ZERO_UNDEF(N);
1708   case ISD::CTPOP:              return visitCTPOP(N);
1709   case ISD::SELECT:             return visitSELECT(N);
1710   case ISD::VSELECT:            return visitVSELECT(N);
1711   case ISD::SELECT_CC:          return visitSELECT_CC(N);
1712   case ISD::SETCC:              return visitSETCC(N);
1713   case ISD::SETCCCARRY:         return visitSETCCCARRY(N);
1714   case ISD::SIGN_EXTEND:        return visitSIGN_EXTEND(N);
1715   case ISD::ZERO_EXTEND:        return visitZERO_EXTEND(N);
1716   case ISD::ANY_EXTEND:         return visitANY_EXTEND(N);
1717   case ISD::AssertSext:
1718   case ISD::AssertZext:         return visitAssertExt(N);
1719   case ISD::SIGN_EXTEND_INREG:  return visitSIGN_EXTEND_INREG(N);
1720   case ISD::SIGN_EXTEND_VECTOR_INREG: return visitSIGN_EXTEND_VECTOR_INREG(N);
1721   case ISD::ZERO_EXTEND_VECTOR_INREG: return visitZERO_EXTEND_VECTOR_INREG(N);
1722   case ISD::TRUNCATE:           return visitTRUNCATE(N);
1723   case ISD::BITCAST:            return visitBITCAST(N);
1724   case ISD::BUILD_PAIR:         return visitBUILD_PAIR(N);
1725   case ISD::FADD:               return visitFADD(N);
1726   case ISD::FSUB:               return visitFSUB(N);
1727   case ISD::FMUL:               return visitFMUL(N);
1728   case ISD::FMA:                return visitFMA(N);
1729   case ISD::FDIV:               return visitFDIV(N);
1730   case ISD::FREM:               return visitFREM(N);
1731   case ISD::FSQRT:              return visitFSQRT(N);
1732   case ISD::FCOPYSIGN:          return visitFCOPYSIGN(N);
1733   case ISD::FPOW:               return visitFPOW(N);
1734   case ISD::SINT_TO_FP:         return visitSINT_TO_FP(N);
1735   case ISD::UINT_TO_FP:         return visitUINT_TO_FP(N);
1736   case ISD::FP_TO_SINT:         return visitFP_TO_SINT(N);
1737   case ISD::FP_TO_UINT:         return visitFP_TO_UINT(N);
1738   case ISD::FP_ROUND:           return visitFP_ROUND(N);
1739   case ISD::FP_ROUND_INREG:     return visitFP_ROUND_INREG(N);
1740   case ISD::FP_EXTEND:          return visitFP_EXTEND(N);
1741   case ISD::FNEG:               return visitFNEG(N);
1742   case ISD::FABS:               return visitFABS(N);
1743   case ISD::FFLOOR:             return visitFFLOOR(N);
1744   case ISD::FMINNUM:            return visitFMINNUM(N);
1745   case ISD::FMAXNUM:            return visitFMAXNUM(N);
1746   case ISD::FMINIMUM:           return visitFMINIMUM(N);
1747   case ISD::FMAXIMUM:           return visitFMAXIMUM(N);
1748   case ISD::FCEIL:              return visitFCEIL(N);
1749   case ISD::FTRUNC:             return visitFTRUNC(N);
1750   case ISD::BRCOND:             return visitBRCOND(N);
1751   case ISD::BR_CC:              return visitBR_CC(N);
1752   case ISD::LOAD:               return visitLOAD(N);
1753   case ISD::STORE:              return visitSTORE(N);
1754   case ISD::INSERT_VECTOR_ELT:  return visitINSERT_VECTOR_ELT(N);
1755   case ISD::EXTRACT_VECTOR_ELT: return visitEXTRACT_VECTOR_ELT(N);
1756   case ISD::BUILD_VECTOR:       return visitBUILD_VECTOR(N);
1757   case ISD::CONCAT_VECTORS:     return visitCONCAT_VECTORS(N);
1758   case ISD::EXTRACT_SUBVECTOR:  return visitEXTRACT_SUBVECTOR(N);
1759   case ISD::VECTOR_SHUFFLE:     return visitVECTOR_SHUFFLE(N);
1760   case ISD::SCALAR_TO_VECTOR:   return visitSCALAR_TO_VECTOR(N);
1761   case ISD::INSERT_SUBVECTOR:   return visitINSERT_SUBVECTOR(N);
1762   case ISD::MGATHER:            return visitMGATHER(N);
1763   case ISD::MLOAD:              return visitMLOAD(N);
1764   case ISD::MSCATTER:           return visitMSCATTER(N);
1765   case ISD::MSTORE:             return visitMSTORE(N);
1766   case ISD::LIFETIME_END:       return visitLIFETIME_END(N);
1767   case ISD::FP_TO_FP16:         return visitFP_TO_FP16(N);
1768   case ISD::FP16_TO_FP:         return visitFP16_TO_FP(N);
1769   case ISD::VECREDUCE_FADD:
1770   case ISD::VECREDUCE_FMUL:
1771   case ISD::VECREDUCE_ADD:
1772   case ISD::VECREDUCE_MUL:
1773   case ISD::VECREDUCE_AND:
1774   case ISD::VECREDUCE_OR:
1775   case ISD::VECREDUCE_XOR:
1776   case ISD::VECREDUCE_SMAX:
1777   case ISD::VECREDUCE_SMIN:
1778   case ISD::VECREDUCE_UMAX:
1779   case ISD::VECREDUCE_UMIN:
1780   case ISD::VECREDUCE_FMAX:
1781   case ISD::VECREDUCE_FMIN:     return visitVECREDUCE(N);
1782   }
1783   return SDValue();
1784 }
1785 
1786 SDValue DAGCombiner::combine(SDNode *N) {
1787   SDValue RV = visit(N);
1788 
1789   // If nothing happened, try a target-specific DAG combine.
1790   if (!RV.getNode()) {
1791     assert(N->getOpcode() != ISD::DELETED_NODE &&
1792            "Node was deleted but visit returned NULL!");
1793 
1794     if (N->getOpcode() >= ISD::BUILTIN_OP_END ||
1795         TLI.hasTargetDAGCombine((ISD::NodeType)N->getOpcode())) {
1796 
1797       // Expose the DAG combiner to the target combiner impls.
1798       TargetLowering::DAGCombinerInfo
1799         DagCombineInfo(DAG, Level, false, this);
1800 
1801       RV = TLI.PerformDAGCombine(N, DagCombineInfo);
1802     }
1803   }
1804 
1805   // If nothing happened still, try promoting the operation.
1806   if (!RV.getNode()) {
1807     switch (N->getOpcode()) {
1808     default: break;
1809     case ISD::ADD:
1810     case ISD::SUB:
1811     case ISD::MUL:
1812     case ISD::AND:
1813     case ISD::OR:
1814     case ISD::XOR:
1815       RV = PromoteIntBinOp(SDValue(N, 0));
1816       break;
1817     case ISD::SHL:
1818     case ISD::SRA:
1819     case ISD::SRL:
1820       RV = PromoteIntShiftOp(SDValue(N, 0));
1821       break;
1822     case ISD::SIGN_EXTEND:
1823     case ISD::ZERO_EXTEND:
1824     case ISD::ANY_EXTEND:
1825       RV = PromoteExtend(SDValue(N, 0));
1826       break;
1827     case ISD::LOAD:
1828       if (PromoteLoad(SDValue(N, 0)))
1829         RV = SDValue(N, 0);
1830       break;
1831     }
1832   }
1833 
1834   // If N is a commutative binary node, try to eliminate it if the commuted
1835   // version is already present in the DAG.
1836   if (!RV.getNode() && TLI.isCommutativeBinOp(N->getOpcode()) &&
1837       N->getNumValues() == 1) {
1838     SDValue N0 = N->getOperand(0);
1839     SDValue N1 = N->getOperand(1);
1840 
1841     // Constant operands are canonicalized to RHS.
1842     if (N0 != N1 && (isa<ConstantSDNode>(N0) || !isa<ConstantSDNode>(N1))) {
1843       SDValue Ops[] = {N1, N0};
1844       SDNode *CSENode = DAG.getNodeIfExists(N->getOpcode(), N->getVTList(), Ops,
1845                                             N->getFlags());
1846       if (CSENode)
1847         return SDValue(CSENode, 0);
1848     }
1849   }
1850 
1851   return RV;
1852 }
1853 
1854 /// Given a node, return its input chain if it has one, otherwise return a null
1855 /// sd operand.
1856 static SDValue getInputChainForNode(SDNode *N) {
1857   if (unsigned NumOps = N->getNumOperands()) {
1858     if (N->getOperand(0).getValueType() == MVT::Other)
1859       return N->getOperand(0);
1860     if (N->getOperand(NumOps-1).getValueType() == MVT::Other)
1861       return N->getOperand(NumOps-1);
1862     for (unsigned i = 1; i < NumOps-1; ++i)
1863       if (N->getOperand(i).getValueType() == MVT::Other)
1864         return N->getOperand(i);
1865   }
1866   return SDValue();
1867 }
1868 
1869 SDValue DAGCombiner::visitTokenFactor(SDNode *N) {
1870   // If N has two operands, where one has an input chain equal to the other,
1871   // the 'other' chain is redundant.
1872   if (N->getNumOperands() == 2) {
1873     if (getInputChainForNode(N->getOperand(0).getNode()) == N->getOperand(1))
1874       return N->getOperand(0);
1875     if (getInputChainForNode(N->getOperand(1).getNode()) == N->getOperand(0))
1876       return N->getOperand(1);
1877   }
1878 
1879   // Don't simplify token factors if optnone.
1880   if (OptLevel == CodeGenOpt::None)
1881     return SDValue();
1882 
1883   // If the sole user is a token factor, we should make sure we have a
1884   // chance to merge them together. This prevents TF chains from inhibiting
1885   // optimizations.
1886   if (N->hasOneUse() && N->use_begin()->getOpcode() == ISD::TokenFactor)
1887     AddToWorklist(*(N->use_begin()));
1888 
1889   SmallVector<SDNode *, 8> TFs;     // List of token factors to visit.
1890   SmallVector<SDValue, 8> Ops;      // Ops for replacing token factor.
1891   SmallPtrSet<SDNode*, 16> SeenOps;
1892   bool Changed = false;             // If we should replace this token factor.
1893 
1894   // Start out with this token factor.
1895   TFs.push_back(N);
1896 
1897   // Iterate through token factors.  The TFs grows when new token factors are
1898   // encountered.
1899   for (unsigned i = 0; i < TFs.size(); ++i) {
1900     // Limit number of nodes to inline, to avoid quadratic compile times.
1901     // We have to add the outstanding Token Factors to Ops, otherwise we might
1902     // drop Ops from the resulting Token Factors.
1903     if (Ops.size() > TokenFactorInlineLimit) {
1904       for (unsigned j = i; j < TFs.size(); j++)
1905         Ops.emplace_back(TFs[j], 0);
1906       // Drop unprocessed Token Factors from TFs, so we do not add them to the
1907       // combiner worklist later.
1908       TFs.resize(i);
1909       break;
1910     }
1911 
1912     SDNode *TF = TFs[i];
1913     // Check each of the operands.
1914     for (const SDValue &Op : TF->op_values()) {
1915       switch (Op.getOpcode()) {
1916       case ISD::EntryToken:
1917         // Entry tokens don't need to be added to the list. They are
1918         // redundant.
1919         Changed = true;
1920         break;
1921 
1922       case ISD::TokenFactor:
1923         if (Op.hasOneUse() && !is_contained(TFs, Op.getNode())) {
1924           // Queue up for processing.
1925           TFs.push_back(Op.getNode());
1926           Changed = true;
1927           break;
1928         }
1929         LLVM_FALLTHROUGH;
1930 
1931       default:
1932         // Only add if it isn't already in the list.
1933         if (SeenOps.insert(Op.getNode()).second)
1934           Ops.push_back(Op);
1935         else
1936           Changed = true;
1937         break;
1938       }
1939     }
1940   }
1941 
1942   // Re-visit inlined Token Factors, to clean them up in case they have been
1943   // removed. Skip the first Token Factor, as this is the current node.
1944   for (unsigned i = 1, e = TFs.size(); i < e; i++)
1945     AddToWorklist(TFs[i]);
1946 
1947   // Remove Nodes that are chained to another node in the list. Do so
1948   // by walking up chains breath-first stopping when we've seen
1949   // another operand. In general we must climb to the EntryNode, but we can exit
1950   // early if we find all remaining work is associated with just one operand as
1951   // no further pruning is possible.
1952 
1953   // List of nodes to search through and original Ops from which they originate.
1954   SmallVector<std::pair<SDNode *, unsigned>, 8> Worklist;
1955   SmallVector<unsigned, 8> OpWorkCount; // Count of work for each Op.
1956   SmallPtrSet<SDNode *, 16> SeenChains;
1957   bool DidPruneOps = false;
1958 
1959   unsigned NumLeftToConsider = 0;
1960   for (const SDValue &Op : Ops) {
1961     Worklist.push_back(std::make_pair(Op.getNode(), NumLeftToConsider++));
1962     OpWorkCount.push_back(1);
1963   }
1964 
1965   auto AddToWorklist = [&](unsigned CurIdx, SDNode *Op, unsigned OpNumber) {
1966     // If this is an Op, we can remove the op from the list. Remark any
1967     // search associated with it as from the current OpNumber.
1968     if (SeenOps.count(Op) != 0) {
1969       Changed = true;
1970       DidPruneOps = true;
1971       unsigned OrigOpNumber = 0;
1972       while (OrigOpNumber < Ops.size() && Ops[OrigOpNumber].getNode() != Op)
1973         OrigOpNumber++;
1974       assert((OrigOpNumber != Ops.size()) &&
1975              "expected to find TokenFactor Operand");
1976       // Re-mark worklist from OrigOpNumber to OpNumber
1977       for (unsigned i = CurIdx + 1; i < Worklist.size(); ++i) {
1978         if (Worklist[i].second == OrigOpNumber) {
1979           Worklist[i].second = OpNumber;
1980         }
1981       }
1982       OpWorkCount[OpNumber] += OpWorkCount[OrigOpNumber];
1983       OpWorkCount[OrigOpNumber] = 0;
1984       NumLeftToConsider--;
1985     }
1986     // Add if it's a new chain
1987     if (SeenChains.insert(Op).second) {
1988       OpWorkCount[OpNumber]++;
1989       Worklist.push_back(std::make_pair(Op, OpNumber));
1990     }
1991   };
1992 
1993   for (unsigned i = 0; i < Worklist.size() && i < 1024; ++i) {
1994     // We need at least be consider at least 2 Ops to prune.
1995     if (NumLeftToConsider <= 1)
1996       break;
1997     auto CurNode = Worklist[i].first;
1998     auto CurOpNumber = Worklist[i].second;
1999     assert((OpWorkCount[CurOpNumber] > 0) &&
2000            "Node should not appear in worklist");
2001     switch (CurNode->getOpcode()) {
2002     case ISD::EntryToken:
2003       // Hitting EntryToken is the only way for the search to terminate without
2004       // hitting
2005       // another operand's search. Prevent us from marking this operand
2006       // considered.
2007       NumLeftToConsider++;
2008       break;
2009     case ISD::TokenFactor:
2010       for (const SDValue &Op : CurNode->op_values())
2011         AddToWorklist(i, Op.getNode(), CurOpNumber);
2012       break;
2013     case ISD::LIFETIME_START:
2014     case ISD::LIFETIME_END:
2015     case ISD::CopyFromReg:
2016     case ISD::CopyToReg:
2017       AddToWorklist(i, CurNode->getOperand(0).getNode(), CurOpNumber);
2018       break;
2019     default:
2020       if (auto *MemNode = dyn_cast<MemSDNode>(CurNode))
2021         AddToWorklist(i, MemNode->getChain().getNode(), CurOpNumber);
2022       break;
2023     }
2024     OpWorkCount[CurOpNumber]--;
2025     if (OpWorkCount[CurOpNumber] == 0)
2026       NumLeftToConsider--;
2027   }
2028 
2029   // If we've changed things around then replace token factor.
2030   if (Changed) {
2031     SDValue Result;
2032     if (Ops.empty()) {
2033       // The entry token is the only possible outcome.
2034       Result = DAG.getEntryNode();
2035     } else {
2036       if (DidPruneOps) {
2037         SmallVector<SDValue, 8> PrunedOps;
2038         //
2039         for (const SDValue &Op : Ops) {
2040           if (SeenChains.count(Op.getNode()) == 0)
2041             PrunedOps.push_back(Op);
2042         }
2043         Result = DAG.getTokenFactor(SDLoc(N), PrunedOps);
2044       } else {
2045         Result = DAG.getTokenFactor(SDLoc(N), Ops);
2046       }
2047     }
2048     return Result;
2049   }
2050   return SDValue();
2051 }
2052 
2053 /// MERGE_VALUES can always be eliminated.
2054 SDValue DAGCombiner::visitMERGE_VALUES(SDNode *N) {
2055   WorklistRemover DeadNodes(*this);
2056   // Replacing results may cause a different MERGE_VALUES to suddenly
2057   // be CSE'd with N, and carry its uses with it. Iterate until no
2058   // uses remain, to ensure that the node can be safely deleted.
2059   // First add the users of this node to the work list so that they
2060   // can be tried again once they have new operands.
2061   AddUsersToWorklist(N);
2062   do {
2063     // Do as a single replacement to avoid rewalking use lists.
2064     SmallVector<SDValue, 8> Ops;
2065     for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i)
2066       Ops.push_back(N->getOperand(i));
2067     DAG.ReplaceAllUsesWith(N, Ops.data());
2068   } while (!N->use_empty());
2069   deleteAndRecombine(N);
2070   return SDValue(N, 0);   // Return N so it doesn't get rechecked!
2071 }
2072 
2073 /// If \p N is a ConstantSDNode with isOpaque() == false return it casted to a
2074 /// ConstantSDNode pointer else nullptr.
2075 static ConstantSDNode *getAsNonOpaqueConstant(SDValue N) {
2076   ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N);
2077   return Const != nullptr && !Const->isOpaque() ? Const : nullptr;
2078 }
2079 
2080 SDValue DAGCombiner::foldBinOpIntoSelect(SDNode *BO) {
2081   assert(TLI.isBinOp(BO->getOpcode()) && BO->getNumValues() == 1 &&
2082          "Unexpected binary operator");
2083 
2084   // Don't do this unless the old select is going away. We want to eliminate the
2085   // binary operator, not replace a binop with a select.
2086   // TODO: Handle ISD::SELECT_CC.
2087   unsigned SelOpNo = 0;
2088   SDValue Sel = BO->getOperand(0);
2089   if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse()) {
2090     SelOpNo = 1;
2091     Sel = BO->getOperand(1);
2092   }
2093 
2094   if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse())
2095     return SDValue();
2096 
2097   SDValue CT = Sel.getOperand(1);
2098   if (!isConstantOrConstantVector(CT, true) &&
2099       !isConstantFPBuildVectorOrConstantFP(CT))
2100     return SDValue();
2101 
2102   SDValue CF = Sel.getOperand(2);
2103   if (!isConstantOrConstantVector(CF, true) &&
2104       !isConstantFPBuildVectorOrConstantFP(CF))
2105     return SDValue();
2106 
2107   // Bail out if any constants are opaque because we can't constant fold those.
2108   // The exception is "and" and "or" with either 0 or -1 in which case we can
2109   // propagate non constant operands into select. I.e.:
2110   // and (select Cond, 0, -1), X --> select Cond, 0, X
2111   // or X, (select Cond, -1, 0) --> select Cond, -1, X
2112   auto BinOpcode = BO->getOpcode();
2113   bool CanFoldNonConst =
2114       (BinOpcode == ISD::AND || BinOpcode == ISD::OR) &&
2115       (isNullOrNullSplat(CT) || isAllOnesOrAllOnesSplat(CT)) &&
2116       (isNullOrNullSplat(CF) || isAllOnesOrAllOnesSplat(CF));
2117 
2118   SDValue CBO = BO->getOperand(SelOpNo ^ 1);
2119   if (!CanFoldNonConst &&
2120       !isConstantOrConstantVector(CBO, true) &&
2121       !isConstantFPBuildVectorOrConstantFP(CBO))
2122     return SDValue();
2123 
2124   EVT VT = Sel.getValueType();
2125 
2126   // In case of shift value and shift amount may have different VT. For instance
2127   // on x86 shift amount is i8 regardles of LHS type. Bail out if we have
2128   // swapped operands and value types do not match. NB: x86 is fine if operands
2129   // are not swapped with shift amount VT being not bigger than shifted value.
2130   // TODO: that is possible to check for a shift operation, correct VTs and
2131   // still perform optimization on x86 if needed.
2132   if (SelOpNo && VT != CBO.getValueType())
2133     return SDValue();
2134 
2135   // We have a select-of-constants followed by a binary operator with a
2136   // constant. Eliminate the binop by pulling the constant math into the select.
2137   // Example: add (select Cond, CT, CF), CBO --> select Cond, CT + CBO, CF + CBO
2138   SDLoc DL(Sel);
2139   SDValue NewCT = SelOpNo ? DAG.getNode(BinOpcode, DL, VT, CBO, CT)
2140                           : DAG.getNode(BinOpcode, DL, VT, CT, CBO);
2141   if (!CanFoldNonConst && !NewCT.isUndef() &&
2142       !isConstantOrConstantVector(NewCT, true) &&
2143       !isConstantFPBuildVectorOrConstantFP(NewCT))
2144     return SDValue();
2145 
2146   SDValue NewCF = SelOpNo ? DAG.getNode(BinOpcode, DL, VT, CBO, CF)
2147                           : DAG.getNode(BinOpcode, DL, VT, CF, CBO);
2148   if (!CanFoldNonConst && !NewCF.isUndef() &&
2149       !isConstantOrConstantVector(NewCF, true) &&
2150       !isConstantFPBuildVectorOrConstantFP(NewCF))
2151     return SDValue();
2152 
2153   SDValue SelectOp = DAG.getSelect(DL, VT, Sel.getOperand(0), NewCT, NewCF);
2154   SelectOp->setFlags(BO->getFlags());
2155   return SelectOp;
2156 }
2157 
2158 static SDValue foldAddSubBoolOfMaskedVal(SDNode *N, SelectionDAG &DAG) {
2159   assert((N->getOpcode() == ISD::ADD || N->getOpcode() == ISD::SUB) &&
2160          "Expecting add or sub");
2161 
2162   // Match a constant operand and a zext operand for the math instruction:
2163   // add Z, C
2164   // sub C, Z
2165   bool IsAdd = N->getOpcode() == ISD::ADD;
2166   SDValue C = IsAdd ? N->getOperand(1) : N->getOperand(0);
2167   SDValue Z = IsAdd ? N->getOperand(0) : N->getOperand(1);
2168   auto *CN = dyn_cast<ConstantSDNode>(C);
2169   if (!CN || Z.getOpcode() != ISD::ZERO_EXTEND)
2170     return SDValue();
2171 
2172   // Match the zext operand as a setcc of a boolean.
2173   if (Z.getOperand(0).getOpcode() != ISD::SETCC ||
2174       Z.getOperand(0).getValueType() != MVT::i1)
2175     return SDValue();
2176 
2177   // Match the compare as: setcc (X & 1), 0, eq.
2178   SDValue SetCC = Z.getOperand(0);
2179   ISD::CondCode CC = cast<CondCodeSDNode>(SetCC->getOperand(2))->get();
2180   if (CC != ISD::SETEQ || !isNullConstant(SetCC.getOperand(1)) ||
2181       SetCC.getOperand(0).getOpcode() != ISD::AND ||
2182       !isOneConstant(SetCC.getOperand(0).getOperand(1)))
2183     return SDValue();
2184 
2185   // We are adding/subtracting a constant and an inverted low bit. Turn that
2186   // into a subtract/add of the low bit with incremented/decremented constant:
2187   // add (zext i1 (seteq (X & 1), 0)), C --> sub C+1, (zext (X & 1))
2188   // sub C, (zext i1 (seteq (X & 1), 0)) --> add C-1, (zext (X & 1))
2189   EVT VT = C.getValueType();
2190   SDLoc DL(N);
2191   SDValue LowBit = DAG.getZExtOrTrunc(SetCC.getOperand(0), DL, VT);
2192   SDValue C1 = IsAdd ? DAG.getConstant(CN->getAPIntValue() + 1, DL, VT) :
2193                        DAG.getConstant(CN->getAPIntValue() - 1, DL, VT);
2194   return DAG.getNode(IsAdd ? ISD::SUB : ISD::ADD, DL, VT, C1, LowBit);
2195 }
2196 
2197 /// Try to fold a 'not' shifted sign-bit with add/sub with constant operand into
2198 /// a shift and add with a different constant.
2199 static SDValue foldAddSubOfSignBit(SDNode *N, SelectionDAG &DAG) {
2200   assert((N->getOpcode() == ISD::ADD || N->getOpcode() == ISD::SUB) &&
2201          "Expecting add or sub");
2202 
2203   // We need a constant operand for the add/sub, and the other operand is a
2204   // logical shift right: add (srl), C or sub C, (srl).
2205   // TODO - support non-uniform vector amounts.
2206   bool IsAdd = N->getOpcode() == ISD::ADD;
2207   SDValue ConstantOp = IsAdd ? N->getOperand(1) : N->getOperand(0);
2208   SDValue ShiftOp = IsAdd ? N->getOperand(0) : N->getOperand(1);
2209   ConstantSDNode *C = isConstOrConstSplat(ConstantOp);
2210   if (!C || ShiftOp.getOpcode() != ISD::SRL)
2211     return SDValue();
2212 
2213   // The shift must be of a 'not' value.
2214   SDValue Not = ShiftOp.getOperand(0);
2215   if (!Not.hasOneUse() || !isBitwiseNot(Not))
2216     return SDValue();
2217 
2218   // The shift must be moving the sign bit to the least-significant-bit.
2219   EVT VT = ShiftOp.getValueType();
2220   SDValue ShAmt = ShiftOp.getOperand(1);
2221   ConstantSDNode *ShAmtC = isConstOrConstSplat(ShAmt);
2222   if (!ShAmtC || ShAmtC->getAPIntValue() != (VT.getScalarSizeInBits() - 1))
2223     return SDValue();
2224 
2225   // Eliminate the 'not' by adjusting the shift and add/sub constant:
2226   // add (srl (not X), 31), C --> add (sra X, 31), (C + 1)
2227   // sub C, (srl (not X), 31) --> add (srl X, 31), (C - 1)
2228   SDLoc DL(N);
2229   auto ShOpcode = IsAdd ? ISD::SRA : ISD::SRL;
2230   SDValue NewShift = DAG.getNode(ShOpcode, DL, VT, Not.getOperand(0), ShAmt);
2231   APInt NewC = IsAdd ? C->getAPIntValue() + 1 : C->getAPIntValue() - 1;
2232   return DAG.getNode(ISD::ADD, DL, VT, NewShift, DAG.getConstant(NewC, DL, VT));
2233 }
2234 
2235 /// Try to fold a node that behaves like an ADD (note that N isn't necessarily
2236 /// an ISD::ADD here, it could for example be an ISD::OR if we know that there
2237 /// are no common bits set in the operands).
2238 SDValue DAGCombiner::visitADDLike(SDNode *N) {
2239   SDValue N0 = N->getOperand(0);
2240   SDValue N1 = N->getOperand(1);
2241   EVT VT = N0.getValueType();
2242   SDLoc DL(N);
2243 
2244   // fold vector ops
2245   if (VT.isVector()) {
2246     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2247       return FoldedVOp;
2248 
2249     // fold (add x, 0) -> x, vector edition
2250     if (ISD::isBuildVectorAllZeros(N1.getNode()))
2251       return N0;
2252     if (ISD::isBuildVectorAllZeros(N0.getNode()))
2253       return N1;
2254   }
2255 
2256   // fold (add x, undef) -> undef
2257   if (N0.isUndef())
2258     return N0;
2259 
2260   if (N1.isUndef())
2261     return N1;
2262 
2263   if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
2264     // canonicalize constant to RHS
2265     if (!DAG.isConstantIntBuildVectorOrConstantInt(N1))
2266       return DAG.getNode(ISD::ADD, DL, VT, N1, N0);
2267     // fold (add c1, c2) -> c1+c2
2268     return DAG.FoldConstantArithmetic(ISD::ADD, DL, VT, N0.getNode(),
2269                                       N1.getNode());
2270   }
2271 
2272   // fold (add x, 0) -> x
2273   if (isNullConstant(N1))
2274     return N0;
2275 
2276   if (isConstantOrConstantVector(N1, /* NoOpaque */ true)) {
2277     // fold ((A-c1)+c2) -> (A+(c2-c1))
2278     if (N0.getOpcode() == ISD::SUB &&
2279         isConstantOrConstantVector(N0.getOperand(1), /* NoOpaque */ true)) {
2280       SDValue Sub = DAG.FoldConstantArithmetic(ISD::SUB, DL, VT, N1.getNode(),
2281                                                N0.getOperand(1).getNode());
2282       assert(Sub && "Constant folding failed");
2283       return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0), Sub);
2284     }
2285 
2286     // fold ((c1-A)+c2) -> (c1+c2)-A
2287     if (N0.getOpcode() == ISD::SUB &&
2288         isConstantOrConstantVector(N0.getOperand(0), /* NoOpaque */ true)) {
2289       SDValue Add = DAG.FoldConstantArithmetic(ISD::ADD, DL, VT, N1.getNode(),
2290                                                N0.getOperand(0).getNode());
2291       assert(Add && "Constant folding failed");
2292       return DAG.getNode(ISD::SUB, DL, VT, Add, N0.getOperand(1));
2293     }
2294 
2295     // add (sext i1 X), 1 -> zext (not i1 X)
2296     // We don't transform this pattern:
2297     //   add (zext i1 X), -1 -> sext (not i1 X)
2298     // because most (?) targets generate better code for the zext form.
2299     if (N0.getOpcode() == ISD::SIGN_EXTEND && N0.hasOneUse() &&
2300         isOneOrOneSplat(N1)) {
2301       SDValue X = N0.getOperand(0);
2302       if ((!LegalOperations ||
2303            (TLI.isOperationLegal(ISD::XOR, X.getValueType()) &&
2304             TLI.isOperationLegal(ISD::ZERO_EXTEND, VT))) &&
2305           X.getScalarValueSizeInBits() == 1) {
2306         SDValue Not = DAG.getNOT(DL, X, X.getValueType());
2307         return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Not);
2308       }
2309     }
2310 
2311     // Undo the add -> or combine to merge constant offsets from a frame index.
2312     if (N0.getOpcode() == ISD::OR &&
2313         isa<FrameIndexSDNode>(N0.getOperand(0)) &&
2314         isa<ConstantSDNode>(N0.getOperand(1)) &&
2315         DAG.haveNoCommonBitsSet(N0.getOperand(0), N0.getOperand(1))) {
2316       SDValue Add0 = DAG.getNode(ISD::ADD, DL, VT, N1, N0.getOperand(1));
2317       return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0), Add0);
2318     }
2319   }
2320 
2321   if (SDValue NewSel = foldBinOpIntoSelect(N))
2322     return NewSel;
2323 
2324   // reassociate add
2325   if (!reassociationCanBreakAddressingModePattern(ISD::ADD, DL, N0, N1)) {
2326     if (SDValue RADD = reassociateOps(ISD::ADD, DL, N0, N1, N->getFlags()))
2327       return RADD;
2328   }
2329   // fold ((0-A) + B) -> B-A
2330   if (N0.getOpcode() == ISD::SUB && isNullOrNullSplat(N0.getOperand(0)))
2331     return DAG.getNode(ISD::SUB, DL, VT, N1, N0.getOperand(1));
2332 
2333   // fold (A + (0-B)) -> A-B
2334   if (N1.getOpcode() == ISD::SUB && isNullOrNullSplat(N1.getOperand(0)))
2335     return DAG.getNode(ISD::SUB, DL, VT, N0, N1.getOperand(1));
2336 
2337   // fold (A+(B-A)) -> B
2338   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(1))
2339     return N1.getOperand(0);
2340 
2341   // fold ((B-A)+A) -> B
2342   if (N0.getOpcode() == ISD::SUB && N1 == N0.getOperand(1))
2343     return N0.getOperand(0);
2344 
2345   // fold ((A-B)+(C-A)) -> (C-B)
2346   if (N0.getOpcode() == ISD::SUB && N1.getOpcode() == ISD::SUB &&
2347       N0.getOperand(0) == N1.getOperand(1))
2348     return DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(0),
2349                        N0.getOperand(1));
2350 
2351   // fold ((A-B)+(B-C)) -> (A-C)
2352   if (N0.getOpcode() == ISD::SUB && N1.getOpcode() == ISD::SUB &&
2353       N0.getOperand(1) == N1.getOperand(0))
2354     return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0),
2355                        N1.getOperand(1));
2356 
2357   // fold (A+(B-(A+C))) to (B-C)
2358   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
2359       N0 == N1.getOperand(1).getOperand(0))
2360     return DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(0),
2361                        N1.getOperand(1).getOperand(1));
2362 
2363   // fold (A+(B-(C+A))) to (B-C)
2364   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
2365       N0 == N1.getOperand(1).getOperand(1))
2366     return DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(0),
2367                        N1.getOperand(1).getOperand(0));
2368 
2369   // fold (A+((B-A)+or-C)) to (B+or-C)
2370   if ((N1.getOpcode() == ISD::SUB || N1.getOpcode() == ISD::ADD) &&
2371       N1.getOperand(0).getOpcode() == ISD::SUB &&
2372       N0 == N1.getOperand(0).getOperand(1))
2373     return DAG.getNode(N1.getOpcode(), DL, VT, N1.getOperand(0).getOperand(0),
2374                        N1.getOperand(1));
2375 
2376   // fold (A-B)+(C-D) to (A+C)-(B+D) when A or C is constant
2377   if (N0.getOpcode() == ISD::SUB && N1.getOpcode() == ISD::SUB) {
2378     SDValue N00 = N0.getOperand(0);
2379     SDValue N01 = N0.getOperand(1);
2380     SDValue N10 = N1.getOperand(0);
2381     SDValue N11 = N1.getOperand(1);
2382 
2383     if (isConstantOrConstantVector(N00) || isConstantOrConstantVector(N10))
2384       return DAG.getNode(ISD::SUB, DL, VT,
2385                          DAG.getNode(ISD::ADD, SDLoc(N0), VT, N00, N10),
2386                          DAG.getNode(ISD::ADD, SDLoc(N1), VT, N01, N11));
2387   }
2388 
2389   // fold (add (umax X, C), -C) --> (usubsat X, C)
2390   if (N0.getOpcode() == ISD::UMAX && hasOperation(ISD::USUBSAT, VT)) {
2391     auto MatchUSUBSAT = [](ConstantSDNode *Max, ConstantSDNode *Op) {
2392       return (!Max && !Op) ||
2393              (Max && Op && Max->getAPIntValue() == (-Op->getAPIntValue()));
2394     };
2395     if (ISD::matchBinaryPredicate(N0.getOperand(1), N1, MatchUSUBSAT,
2396                                   /*AllowUndefs*/ true))
2397       return DAG.getNode(ISD::USUBSAT, DL, VT, N0.getOperand(0),
2398                          N0.getOperand(1));
2399   }
2400 
2401   if (SimplifyDemandedBits(SDValue(N, 0)))
2402     return SDValue(N, 0);
2403 
2404   if (isOneOrOneSplat(N1)) {
2405     // fold (add (xor a, -1), 1) -> (sub 0, a)
2406     if (isBitwiseNot(N0))
2407       return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT),
2408                          N0.getOperand(0));
2409 
2410     // fold (add (add (xor a, -1), b), 1) -> (sub b, a)
2411     if (N0.getOpcode() == ISD::ADD ||
2412         N0.getOpcode() == ISD::UADDO ||
2413         N0.getOpcode() == ISD::SADDO) {
2414       SDValue A, Xor;
2415 
2416       if (isBitwiseNot(N0.getOperand(0))) {
2417         A = N0.getOperand(1);
2418         Xor = N0.getOperand(0);
2419       } else if (isBitwiseNot(N0.getOperand(1))) {
2420         A = N0.getOperand(0);
2421         Xor = N0.getOperand(1);
2422       }
2423 
2424       if (Xor)
2425         return DAG.getNode(ISD::SUB, DL, VT, A, Xor.getOperand(0));
2426     }
2427   }
2428 
2429   // (x - y) + -1  ->  add (xor y, -1), x
2430   if (N0.hasOneUse() && N0.getOpcode() == ISD::SUB &&
2431       isAllOnesOrAllOnesSplat(N1)) {
2432     SDValue Xor = DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(1), N1);
2433     return DAG.getNode(ISD::ADD, DL, VT, Xor, N0.getOperand(0));
2434   }
2435 
2436   if (SDValue Combined = visitADDLikeCommutative(N0, N1, N))
2437     return Combined;
2438 
2439   if (SDValue Combined = visitADDLikeCommutative(N1, N0, N))
2440     return Combined;
2441 
2442   return SDValue();
2443 }
2444 
2445 SDValue DAGCombiner::visitADD(SDNode *N) {
2446   SDValue N0 = N->getOperand(0);
2447   SDValue N1 = N->getOperand(1);
2448   EVT VT = N0.getValueType();
2449   SDLoc DL(N);
2450 
2451   if (SDValue Combined = visitADDLike(N))
2452     return Combined;
2453 
2454   if (SDValue V = foldAddSubBoolOfMaskedVal(N, DAG))
2455     return V;
2456 
2457   if (SDValue V = foldAddSubOfSignBit(N, DAG))
2458     return V;
2459 
2460   // fold (a+b) -> (a|b) iff a and b share no bits.
2461   if ((!LegalOperations || TLI.isOperationLegal(ISD::OR, VT)) &&
2462       DAG.haveNoCommonBitsSet(N0, N1))
2463     return DAG.getNode(ISD::OR, DL, VT, N0, N1);
2464 
2465   return SDValue();
2466 }
2467 
2468 SDValue DAGCombiner::visitADDSAT(SDNode *N) {
2469   unsigned Opcode = N->getOpcode();
2470   SDValue N0 = N->getOperand(0);
2471   SDValue N1 = N->getOperand(1);
2472   EVT VT = N0.getValueType();
2473   SDLoc DL(N);
2474 
2475   // fold vector ops
2476   if (VT.isVector()) {
2477     // TODO SimplifyVBinOp
2478 
2479     // fold (add_sat x, 0) -> x, vector edition
2480     if (ISD::isBuildVectorAllZeros(N1.getNode()))
2481       return N0;
2482     if (ISD::isBuildVectorAllZeros(N0.getNode()))
2483       return N1;
2484   }
2485 
2486   // fold (add_sat x, undef) -> -1
2487   if (N0.isUndef() || N1.isUndef())
2488     return DAG.getAllOnesConstant(DL, VT);
2489 
2490   if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
2491     // canonicalize constant to RHS
2492     if (!DAG.isConstantIntBuildVectorOrConstantInt(N1))
2493       return DAG.getNode(Opcode, DL, VT, N1, N0);
2494     // fold (add_sat c1, c2) -> c3
2495     return DAG.FoldConstantArithmetic(Opcode, DL, VT, N0.getNode(),
2496                                       N1.getNode());
2497   }
2498 
2499   // fold (add_sat x, 0) -> x
2500   if (isNullConstant(N1))
2501     return N0;
2502 
2503   // If it cannot overflow, transform into an add.
2504   if (Opcode == ISD::UADDSAT)
2505     if (DAG.computeOverflowKind(N0, N1) == SelectionDAG::OFK_Never)
2506       return DAG.getNode(ISD::ADD, DL, VT, N0, N1);
2507 
2508   return SDValue();
2509 }
2510 
2511 static SDValue getAsCarry(const TargetLowering &TLI, SDValue V) {
2512   bool Masked = false;
2513 
2514   // First, peel away TRUNCATE/ZERO_EXTEND/AND nodes due to legalization.
2515   while (true) {
2516     if (V.getOpcode() == ISD::TRUNCATE || V.getOpcode() == ISD::ZERO_EXTEND) {
2517       V = V.getOperand(0);
2518       continue;
2519     }
2520 
2521     if (V.getOpcode() == ISD::AND && isOneConstant(V.getOperand(1))) {
2522       Masked = true;
2523       V = V.getOperand(0);
2524       continue;
2525     }
2526 
2527     break;
2528   }
2529 
2530   // If this is not a carry, return.
2531   if (V.getResNo() != 1)
2532     return SDValue();
2533 
2534   if (V.getOpcode() != ISD::ADDCARRY && V.getOpcode() != ISD::SUBCARRY &&
2535       V.getOpcode() != ISD::UADDO && V.getOpcode() != ISD::USUBO)
2536     return SDValue();
2537 
2538   EVT VT = V.getNode()->getValueType(0);
2539   if (!TLI.isOperationLegalOrCustom(V.getOpcode(), VT))
2540     return SDValue();
2541 
2542   // If the result is masked, then no matter what kind of bool it is we can
2543   // return. If it isn't, then we need to make sure the bool type is either 0 or
2544   // 1 and not other values.
2545   if (Masked ||
2546       TLI.getBooleanContents(V.getValueType()) ==
2547           TargetLoweringBase::ZeroOrOneBooleanContent)
2548     return V;
2549 
2550   return SDValue();
2551 }
2552 
2553 /// Given the operands of an add/sub operation, see if the 2nd operand is a
2554 /// masked 0/1 whose source operand is actually known to be 0/-1. If so, invert
2555 /// the opcode and bypass the mask operation.
2556 static SDValue foldAddSubMasked1(bool IsAdd, SDValue N0, SDValue N1,
2557                                  SelectionDAG &DAG, const SDLoc &DL) {
2558   if (N1.getOpcode() != ISD::AND || !isOneOrOneSplat(N1->getOperand(1)))
2559     return SDValue();
2560 
2561   EVT VT = N0.getValueType();
2562   if (DAG.ComputeNumSignBits(N1.getOperand(0)) != VT.getScalarSizeInBits())
2563     return SDValue();
2564 
2565   // add N0, (and (AssertSext X, i1), 1) --> sub N0, X
2566   // sub N0, (and (AssertSext X, i1), 1) --> add N0, X
2567   return DAG.getNode(IsAdd ? ISD::SUB : ISD::ADD, DL, VT, N0, N1.getOperand(0));
2568 }
2569 
2570 /// Helper for doing combines based on N0 and N1 being added to each other.
2571 SDValue DAGCombiner::visitADDLikeCommutative(SDValue N0, SDValue N1,
2572                                           SDNode *LocReference) {
2573   EVT VT = N0.getValueType();
2574   SDLoc DL(LocReference);
2575 
2576   // fold (add x, shl(0 - y, n)) -> sub(x, shl(y, n))
2577   if (N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::SUB &&
2578       isNullOrNullSplat(N1.getOperand(0).getOperand(0)))
2579     return DAG.getNode(ISD::SUB, DL, VT, N0,
2580                        DAG.getNode(ISD::SHL, DL, VT,
2581                                    N1.getOperand(0).getOperand(1),
2582                                    N1.getOperand(1)));
2583 
2584   if (SDValue V = foldAddSubMasked1(true, N0, N1, DAG, DL))
2585     return V;
2586 
2587   // Hoist one-use subtraction by non-opaque constant:
2588   //   (x - C) + y  ->  (x + y) - C
2589   // This is necessary because SUB(X,C) -> ADD(X,-C) doesn't work for vectors.
2590   if (N0.hasOneUse() && N0.getOpcode() == ISD::SUB &&
2591       isConstantOrConstantVector(N0.getOperand(1), /*NoOpaques=*/true)) {
2592     SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0), N1);
2593     return DAG.getNode(ISD::SUB, DL, VT, Add, N0.getOperand(1));
2594   }
2595   // Hoist one-use subtraction from non-opaque constant:
2596   //   (C - x) + y  ->  (y - x) + C
2597   if (N0.hasOneUse() && N0.getOpcode() == ISD::SUB &&
2598       isConstantOrConstantVector(N0.getOperand(0), /*NoOpaques=*/true)) {
2599     SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N1, N0.getOperand(1));
2600     return DAG.getNode(ISD::ADD, DL, VT, Sub, N0.getOperand(0));
2601   }
2602 
2603   // If the target's bool is represented as 0/1, prefer to make this 'sub 0/1'
2604   // rather than 'add 0/-1' (the zext should get folded).
2605   // add (sext i1 Y), X --> sub X, (zext i1 Y)
2606   if (N0.getOpcode() == ISD::SIGN_EXTEND &&
2607       N0.getOperand(0).getScalarValueSizeInBits() == 1 &&
2608       TLI.getBooleanContents(VT) == TargetLowering::ZeroOrOneBooleanContent) {
2609     SDValue ZExt = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0));
2610     return DAG.getNode(ISD::SUB, DL, VT, N1, ZExt);
2611   }
2612 
2613   // add X, (sextinreg Y i1) -> sub X, (and Y 1)
2614   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
2615     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
2616     if (TN->getVT() == MVT::i1) {
2617       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
2618                                  DAG.getConstant(1, DL, VT));
2619       return DAG.getNode(ISD::SUB, DL, VT, N0, ZExt);
2620     }
2621   }
2622 
2623   // (add X, (addcarry Y, 0, Carry)) -> (addcarry X, Y, Carry)
2624   if (N1.getOpcode() == ISD::ADDCARRY && isNullConstant(N1.getOperand(1)) &&
2625       N1.getResNo() == 0)
2626     return DAG.getNode(ISD::ADDCARRY, DL, N1->getVTList(),
2627                        N0, N1.getOperand(0), N1.getOperand(2));
2628 
2629   // (add X, Carry) -> (addcarry X, 0, Carry)
2630   if (TLI.isOperationLegalOrCustom(ISD::ADDCARRY, VT))
2631     if (SDValue Carry = getAsCarry(TLI, N1))
2632       return DAG.getNode(ISD::ADDCARRY, DL,
2633                          DAG.getVTList(VT, Carry.getValueType()), N0,
2634                          DAG.getConstant(0, DL, VT), Carry);
2635 
2636   return SDValue();
2637 }
2638 
2639 SDValue DAGCombiner::visitADDC(SDNode *N) {
2640   SDValue N0 = N->getOperand(0);
2641   SDValue N1 = N->getOperand(1);
2642   EVT VT = N0.getValueType();
2643   SDLoc DL(N);
2644 
2645   // If the flag result is dead, turn this into an ADD.
2646   if (!N->hasAnyUseOfValue(1))
2647     return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1),
2648                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2649 
2650   // canonicalize constant to RHS.
2651   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
2652   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
2653   if (N0C && !N1C)
2654     return DAG.getNode(ISD::ADDC, DL, N->getVTList(), N1, N0);
2655 
2656   // fold (addc x, 0) -> x + no carry out
2657   if (isNullConstant(N1))
2658     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE,
2659                                         DL, MVT::Glue));
2660 
2661   // If it cannot overflow, transform into an add.
2662   if (DAG.computeOverflowKind(N0, N1) == SelectionDAG::OFK_Never)
2663     return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1),
2664                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2665 
2666   return SDValue();
2667 }
2668 
2669 static SDValue flipBoolean(SDValue V, const SDLoc &DL,
2670                            SelectionDAG &DAG, const TargetLowering &TLI) {
2671   EVT VT = V.getValueType();
2672 
2673   SDValue Cst;
2674   switch (TLI.getBooleanContents(VT)) {
2675   case TargetLowering::ZeroOrOneBooleanContent:
2676   case TargetLowering::UndefinedBooleanContent:
2677     Cst = DAG.getConstant(1, DL, VT);
2678     break;
2679   case TargetLowering::ZeroOrNegativeOneBooleanContent:
2680     Cst = DAG.getConstant(-1, DL, VT);
2681     break;
2682   }
2683 
2684   return DAG.getNode(ISD::XOR, DL, VT, V, Cst);
2685 }
2686 
2687 static SDValue extractBooleanFlip(SDValue V, const TargetLowering &TLI) {
2688   if (V.getOpcode() != ISD::XOR)
2689     return SDValue();
2690 
2691   ConstantSDNode *Const = isConstOrConstSplat(V.getOperand(1), false);
2692   if (!Const)
2693     return SDValue();
2694 
2695   EVT VT = V.getValueType();
2696 
2697   bool IsFlip = false;
2698   switch(TLI.getBooleanContents(VT)) {
2699     case TargetLowering::ZeroOrOneBooleanContent:
2700       IsFlip = Const->isOne();
2701       break;
2702     case TargetLowering::ZeroOrNegativeOneBooleanContent:
2703       IsFlip = Const->isAllOnesValue();
2704       break;
2705     case TargetLowering::UndefinedBooleanContent:
2706       IsFlip = (Const->getAPIntValue() & 0x01) == 1;
2707       break;
2708   }
2709 
2710   if (IsFlip)
2711     return V.getOperand(0);
2712   return SDValue();
2713 }
2714 
2715 SDValue DAGCombiner::visitADDO(SDNode *N) {
2716   SDValue N0 = N->getOperand(0);
2717   SDValue N1 = N->getOperand(1);
2718   EVT VT = N0.getValueType();
2719   bool IsSigned = (ISD::SADDO == N->getOpcode());
2720 
2721   EVT CarryVT = N->getValueType(1);
2722   SDLoc DL(N);
2723 
2724   // If the flag result is dead, turn this into an ADD.
2725   if (!N->hasAnyUseOfValue(1))
2726     return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1),
2727                      DAG.getUNDEF(CarryVT));
2728 
2729   // canonicalize constant to RHS.
2730   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2731       !DAG.isConstantIntBuildVectorOrConstantInt(N1))
2732     return DAG.getNode(N->getOpcode(), DL, N->getVTList(), N1, N0);
2733 
2734   // fold (addo x, 0) -> x + no carry out
2735   if (isNullOrNullSplat(N1))
2736     return CombineTo(N, N0, DAG.getConstant(0, DL, CarryVT));
2737 
2738   if (!IsSigned) {
2739     // If it cannot overflow, transform into an add.
2740     if (DAG.computeOverflowKind(N0, N1) == SelectionDAG::OFK_Never)
2741       return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1),
2742                        DAG.getConstant(0, DL, CarryVT));
2743 
2744     // fold (uaddo (xor a, -1), 1) -> (usub 0, a) and flip carry.
2745     if (isBitwiseNot(N0) && isOneOrOneSplat(N1)) {
2746       SDValue Sub = DAG.getNode(ISD::USUBO, DL, N->getVTList(),
2747                                 DAG.getConstant(0, DL, VT), N0.getOperand(0));
2748       return CombineTo(N, Sub,
2749                        flipBoolean(Sub.getValue(1), DL, DAG, TLI));
2750     }
2751 
2752     if (SDValue Combined = visitUADDOLike(N0, N1, N))
2753       return Combined;
2754 
2755     if (SDValue Combined = visitUADDOLike(N1, N0, N))
2756       return Combined;
2757   }
2758 
2759   return SDValue();
2760 }
2761 
2762 SDValue DAGCombiner::visitUADDOLike(SDValue N0, SDValue N1, SDNode *N) {
2763   EVT VT = N0.getValueType();
2764   if (VT.isVector())
2765     return SDValue();
2766 
2767   // (uaddo X, (addcarry Y, 0, Carry)) -> (addcarry X, Y, Carry)
2768   // If Y + 1 cannot overflow.
2769   if (N1.getOpcode() == ISD::ADDCARRY && isNullConstant(N1.getOperand(1))) {
2770     SDValue Y = N1.getOperand(0);
2771     SDValue One = DAG.getConstant(1, SDLoc(N), Y.getValueType());
2772     if (DAG.computeOverflowKind(Y, One) == SelectionDAG::OFK_Never)
2773       return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(), N0, Y,
2774                          N1.getOperand(2));
2775   }
2776 
2777   // (uaddo X, Carry) -> (addcarry X, 0, Carry)
2778   if (TLI.isOperationLegalOrCustom(ISD::ADDCARRY, VT))
2779     if (SDValue Carry = getAsCarry(TLI, N1))
2780       return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(), N0,
2781                          DAG.getConstant(0, SDLoc(N), VT), Carry);
2782 
2783   return SDValue();
2784 }
2785 
2786 SDValue DAGCombiner::visitADDE(SDNode *N) {
2787   SDValue N0 = N->getOperand(0);
2788   SDValue N1 = N->getOperand(1);
2789   SDValue CarryIn = N->getOperand(2);
2790 
2791   // canonicalize constant to RHS
2792   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
2793   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
2794   if (N0C && !N1C)
2795     return DAG.getNode(ISD::ADDE, SDLoc(N), N->getVTList(),
2796                        N1, N0, CarryIn);
2797 
2798   // fold (adde x, y, false) -> (addc x, y)
2799   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
2800     return DAG.getNode(ISD::ADDC, SDLoc(N), N->getVTList(), N0, N1);
2801 
2802   return SDValue();
2803 }
2804 
2805 SDValue DAGCombiner::visitADDCARRY(SDNode *N) {
2806   SDValue N0 = N->getOperand(0);
2807   SDValue N1 = N->getOperand(1);
2808   SDValue CarryIn = N->getOperand(2);
2809   SDLoc DL(N);
2810 
2811   // canonicalize constant to RHS
2812   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
2813   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
2814   if (N0C && !N1C)
2815     return DAG.getNode(ISD::ADDCARRY, DL, N->getVTList(), N1, N0, CarryIn);
2816 
2817   // fold (addcarry x, y, false) -> (uaddo x, y)
2818   if (isNullConstant(CarryIn)) {
2819     if (!LegalOperations ||
2820         TLI.isOperationLegalOrCustom(ISD::UADDO, N->getValueType(0)))
2821       return DAG.getNode(ISD::UADDO, DL, N->getVTList(), N0, N1);
2822   }
2823 
2824   EVT CarryVT = CarryIn.getValueType();
2825 
2826   // fold (addcarry 0, 0, X) -> (and (ext/trunc X), 1) and no carry.
2827   if (isNullConstant(N0) && isNullConstant(N1)) {
2828     EVT VT = N0.getValueType();
2829     SDValue CarryExt = DAG.getBoolExtOrTrunc(CarryIn, DL, VT, CarryVT);
2830     AddToWorklist(CarryExt.getNode());
2831     return CombineTo(N, DAG.getNode(ISD::AND, DL, VT, CarryExt,
2832                                     DAG.getConstant(1, DL, VT)),
2833                      DAG.getConstant(0, DL, CarryVT));
2834   }
2835 
2836   // fold (addcarry (xor a, -1), 0, !b) -> (subcarry 0, a, b) and flip carry.
2837   if (isBitwiseNot(N0) && isNullConstant(N1)) {
2838     if (SDValue B = extractBooleanFlip(CarryIn, TLI)) {
2839       SDValue Sub = DAG.getNode(ISD::SUBCARRY, DL, N->getVTList(),
2840                                 DAG.getConstant(0, DL, N0.getValueType()),
2841                                 N0.getOperand(0), B);
2842       return CombineTo(N, Sub,
2843                        flipBoolean(Sub.getValue(1), DL, DAG, TLI));
2844     }
2845   }
2846 
2847   if (SDValue Combined = visitADDCARRYLike(N0, N1, CarryIn, N))
2848     return Combined;
2849 
2850   if (SDValue Combined = visitADDCARRYLike(N1, N0, CarryIn, N))
2851     return Combined;
2852 
2853   return SDValue();
2854 }
2855 
2856 SDValue DAGCombiner::visitADDCARRYLike(SDValue N0, SDValue N1, SDValue CarryIn,
2857                                        SDNode *N) {
2858   // Iff the flag result is dead:
2859   // (addcarry (add|uaddo X, Y), 0, Carry) -> (addcarry X, Y, Carry)
2860   if ((N0.getOpcode() == ISD::ADD ||
2861        (N0.getOpcode() == ISD::UADDO && N0.getResNo() == 0)) &&
2862       isNullConstant(N1) && !N->hasAnyUseOfValue(1))
2863     return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(),
2864                        N0.getOperand(0), N0.getOperand(1), CarryIn);
2865 
2866   /**
2867    * When one of the addcarry argument is itself a carry, we may be facing
2868    * a diamond carry propagation. In which case we try to transform the DAG
2869    * to ensure linear carry propagation if that is possible.
2870    *
2871    * We are trying to get:
2872    *   (addcarry X, 0, (addcarry A, B, Z):Carry)
2873    */
2874   if (auto Y = getAsCarry(TLI, N1)) {
2875     /**
2876      *            (uaddo A, B)
2877      *             /       \
2878      *          Carry      Sum
2879      *            |          \
2880      *            | (addcarry *, 0, Z)
2881      *            |       /
2882      *             \   Carry
2883      *              |   /
2884      * (addcarry X, *, *)
2885      */
2886     if (Y.getOpcode() == ISD::UADDO &&
2887         CarryIn.getResNo() == 1 &&
2888         CarryIn.getOpcode() == ISD::ADDCARRY &&
2889         isNullConstant(CarryIn.getOperand(1)) &&
2890         CarryIn.getOperand(0) == Y.getValue(0)) {
2891       auto NewY = DAG.getNode(ISD::ADDCARRY, SDLoc(N), Y->getVTList(),
2892                               Y.getOperand(0), Y.getOperand(1),
2893                               CarryIn.getOperand(2));
2894       AddToWorklist(NewY.getNode());
2895       return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(), N0,
2896                          DAG.getConstant(0, SDLoc(N), N0.getValueType()),
2897                          NewY.getValue(1));
2898     }
2899   }
2900 
2901   return SDValue();
2902 }
2903 
2904 // Since it may not be valid to emit a fold to zero for vector initializers
2905 // check if we can before folding.
2906 static SDValue tryFoldToZero(const SDLoc &DL, const TargetLowering &TLI, EVT VT,
2907                              SelectionDAG &DAG, bool LegalOperations) {
2908   if (!VT.isVector())
2909     return DAG.getConstant(0, DL, VT);
2910   if (!LegalOperations || TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
2911     return DAG.getConstant(0, DL, VT);
2912   return SDValue();
2913 }
2914 
2915 SDValue DAGCombiner::visitSUB(SDNode *N) {
2916   SDValue N0 = N->getOperand(0);
2917   SDValue N1 = N->getOperand(1);
2918   EVT VT = N0.getValueType();
2919   SDLoc DL(N);
2920 
2921   // fold vector ops
2922   if (VT.isVector()) {
2923     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2924       return FoldedVOp;
2925 
2926     // fold (sub x, 0) -> x, vector edition
2927     if (ISD::isBuildVectorAllZeros(N1.getNode()))
2928       return N0;
2929   }
2930 
2931   // fold (sub x, x) -> 0
2932   // FIXME: Refactor this and xor and other similar operations together.
2933   if (N0 == N1)
2934     return tryFoldToZero(DL, TLI, VT, DAG, LegalOperations);
2935   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2936       DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
2937     // fold (sub c1, c2) -> c1-c2
2938     return DAG.FoldConstantArithmetic(ISD::SUB, DL, VT, N0.getNode(),
2939                                       N1.getNode());
2940   }
2941 
2942   if (SDValue NewSel = foldBinOpIntoSelect(N))
2943     return NewSel;
2944 
2945   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
2946 
2947   // fold (sub x, c) -> (add x, -c)
2948   if (N1C) {
2949     return DAG.getNode(ISD::ADD, DL, VT, N0,
2950                        DAG.getConstant(-N1C->getAPIntValue(), DL, VT));
2951   }
2952 
2953   if (isNullOrNullSplat(N0)) {
2954     unsigned BitWidth = VT.getScalarSizeInBits();
2955     // Right-shifting everything out but the sign bit followed by negation is
2956     // the same as flipping arithmetic/logical shift type without the negation:
2957     // -(X >>u 31) -> (X >>s 31)
2958     // -(X >>s 31) -> (X >>u 31)
2959     if (N1->getOpcode() == ISD::SRA || N1->getOpcode() == ISD::SRL) {
2960       ConstantSDNode *ShiftAmt = isConstOrConstSplat(N1.getOperand(1));
2961       if (ShiftAmt && ShiftAmt->getAPIntValue() == (BitWidth - 1)) {
2962         auto NewSh = N1->getOpcode() == ISD::SRA ? ISD::SRL : ISD::SRA;
2963         if (!LegalOperations || TLI.isOperationLegal(NewSh, VT))
2964           return DAG.getNode(NewSh, DL, VT, N1.getOperand(0), N1.getOperand(1));
2965       }
2966     }
2967 
2968     // 0 - X --> 0 if the sub is NUW.
2969     if (N->getFlags().hasNoUnsignedWrap())
2970       return N0;
2971 
2972     if (DAG.MaskedValueIsZero(N1, ~APInt::getSignMask(BitWidth))) {
2973       // N1 is either 0 or the minimum signed value. If the sub is NSW, then
2974       // N1 must be 0 because negating the minimum signed value is undefined.
2975       if (N->getFlags().hasNoSignedWrap())
2976         return N0;
2977 
2978       // 0 - X --> X if X is 0 or the minimum signed value.
2979       return N1;
2980     }
2981   }
2982 
2983   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1)
2984   if (isAllOnesOrAllOnesSplat(N0))
2985     return DAG.getNode(ISD::XOR, DL, VT, N1, N0);
2986 
2987   // fold (A - (0-B)) -> A+B
2988   if (N1.getOpcode() == ISD::SUB && isNullOrNullSplat(N1.getOperand(0)))
2989     return DAG.getNode(ISD::ADD, DL, VT, N0, N1.getOperand(1));
2990 
2991   // fold A-(A-B) -> B
2992   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(0))
2993     return N1.getOperand(1);
2994 
2995   // fold (A+B)-A -> B
2996   if (N0.getOpcode() == ISD::ADD && N0.getOperand(0) == N1)
2997     return N0.getOperand(1);
2998 
2999   // fold (A+B)-B -> A
3000   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1) == N1)
3001     return N0.getOperand(0);
3002 
3003   // fold (A+C1)-C2 -> A+(C1-C2)
3004   if (N0.getOpcode() == ISD::ADD &&
3005       isConstantOrConstantVector(N1, /* NoOpaques */ true) &&
3006       isConstantOrConstantVector(N0.getOperand(1), /* NoOpaques */ true)) {
3007     SDValue NewC = DAG.FoldConstantArithmetic(
3008         ISD::SUB, DL, VT, N0.getOperand(1).getNode(), N1.getNode());
3009     assert(NewC && "Constant folding failed");
3010     return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0), NewC);
3011   }
3012 
3013   // fold C2-(A+C1) -> (C2-C1)-A
3014   if (N1.getOpcode() == ISD::ADD) {
3015     SDValue N11 = N1.getOperand(1);
3016     if (isConstantOrConstantVector(N0, /* NoOpaques */ true) &&
3017         isConstantOrConstantVector(N11, /* NoOpaques */ true)) {
3018       SDValue NewC = DAG.FoldConstantArithmetic(ISD::SUB, DL, VT, N0.getNode(),
3019                                                 N11.getNode());
3020       assert(NewC && "Constant folding failed");
3021       return DAG.getNode(ISD::SUB, DL, VT, NewC, N1.getOperand(0));
3022     }
3023   }
3024 
3025   // fold (A-C1)-C2 -> A-(C1+C2)
3026   if (N0.getOpcode() == ISD::SUB &&
3027       isConstantOrConstantVector(N1, /* NoOpaques */ true) &&
3028       isConstantOrConstantVector(N0.getOperand(1), /* NoOpaques */ true)) {
3029     SDValue NewC = DAG.FoldConstantArithmetic(
3030         ISD::ADD, DL, VT, N0.getOperand(1).getNode(), N1.getNode());
3031     assert(NewC && "Constant folding failed");
3032     return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0), NewC);
3033   }
3034 
3035   // fold (c1-A)-c2 -> (c1-c2)-A
3036   if (N0.getOpcode() == ISD::SUB &&
3037       isConstantOrConstantVector(N1, /* NoOpaques */ true) &&
3038       isConstantOrConstantVector(N0.getOperand(0), /* NoOpaques */ true)) {
3039     SDValue NewC = DAG.FoldConstantArithmetic(
3040         ISD::SUB, DL, VT, N0.getOperand(0).getNode(), N1.getNode());
3041     assert(NewC && "Constant folding failed");
3042     return DAG.getNode(ISD::SUB, DL, VT, NewC, N0.getOperand(1));
3043   }
3044 
3045   // fold ((A+(B+or-C))-B) -> A+or-C
3046   if (N0.getOpcode() == ISD::ADD &&
3047       (N0.getOperand(1).getOpcode() == ISD::SUB ||
3048        N0.getOperand(1).getOpcode() == ISD::ADD) &&
3049       N0.getOperand(1).getOperand(0) == N1)
3050     return DAG.getNode(N0.getOperand(1).getOpcode(), DL, VT, N0.getOperand(0),
3051                        N0.getOperand(1).getOperand(1));
3052 
3053   // fold ((A+(C+B))-B) -> A+C
3054   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1).getOpcode() == ISD::ADD &&
3055       N0.getOperand(1).getOperand(1) == N1)
3056     return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0),
3057                        N0.getOperand(1).getOperand(0));
3058 
3059   // fold ((A-(B-C))-C) -> A-B
3060   if (N0.getOpcode() == ISD::SUB && N0.getOperand(1).getOpcode() == ISD::SUB &&
3061       N0.getOperand(1).getOperand(1) == N1)
3062     return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0),
3063                        N0.getOperand(1).getOperand(0));
3064 
3065   // fold (A-(B-C)) -> A+(C-B)
3066   if (N1.getOpcode() == ISD::SUB && N1.hasOneUse())
3067     return DAG.getNode(ISD::ADD, DL, VT, N0,
3068                        DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(1),
3069                                    N1.getOperand(0)));
3070 
3071   // fold (X - (-Y * Z)) -> (X + (Y * Z))
3072   if (N1.getOpcode() == ISD::MUL && N1.hasOneUse()) {
3073     if (N1.getOperand(0).getOpcode() == ISD::SUB &&
3074         isNullOrNullSplat(N1.getOperand(0).getOperand(0))) {
3075       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT,
3076                                 N1.getOperand(0).getOperand(1),
3077                                 N1.getOperand(1));
3078       return DAG.getNode(ISD::ADD, DL, VT, N0, Mul);
3079     }
3080     if (N1.getOperand(1).getOpcode() == ISD::SUB &&
3081         isNullOrNullSplat(N1.getOperand(1).getOperand(0))) {
3082       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT,
3083                                 N1.getOperand(0),
3084                                 N1.getOperand(1).getOperand(1));
3085       return DAG.getNode(ISD::ADD, DL, VT, N0, Mul);
3086     }
3087   }
3088 
3089   // If either operand of a sub is undef, the result is undef
3090   if (N0.isUndef())
3091     return N0;
3092   if (N1.isUndef())
3093     return N1;
3094 
3095   if (SDValue V = foldAddSubBoolOfMaskedVal(N, DAG))
3096     return V;
3097 
3098   if (SDValue V = foldAddSubOfSignBit(N, DAG))
3099     return V;
3100 
3101   if (SDValue V = foldAddSubMasked1(false, N0, N1, DAG, SDLoc(N)))
3102     return V;
3103 
3104   // (x - y) - 1  ->  add (xor y, -1), x
3105   if (N0.hasOneUse() && N0.getOpcode() == ISD::SUB && isOneOrOneSplat(N1)) {
3106     SDValue Xor = DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(1),
3107                               DAG.getAllOnesConstant(DL, VT));
3108     return DAG.getNode(ISD::ADD, DL, VT, Xor, N0.getOperand(0));
3109   }
3110 
3111   // Hoist one-use addition by non-opaque constant:
3112   //   (x + C) - y  ->  (x - y) + C
3113   if (N0.hasOneUse() && N0.getOpcode() == ISD::ADD &&
3114       isConstantOrConstantVector(N0.getOperand(1), /*NoOpaques=*/true)) {
3115     SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0), N1);
3116     return DAG.getNode(ISD::ADD, DL, VT, Sub, N0.getOperand(1));
3117   }
3118   // y - (x + C)  ->  (y - x) - C
3119   if (N1.hasOneUse() && N1.getOpcode() == ISD::ADD &&
3120       isConstantOrConstantVector(N1.getOperand(1), /*NoOpaques=*/true)) {
3121     SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, N1.getOperand(0));
3122     return DAG.getNode(ISD::SUB, DL, VT, Sub, N1.getOperand(1));
3123   }
3124   // (x - C) - y  ->  (x - y) - C
3125   // This is necessary because SUB(X,C) -> ADD(X,-C) doesn't work for vectors.
3126   if (N0.hasOneUse() && N0.getOpcode() == ISD::SUB &&
3127       isConstantOrConstantVector(N0.getOperand(1), /*NoOpaques=*/true)) {
3128     SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0), N1);
3129     return DAG.getNode(ISD::SUB, DL, VT, Sub, N0.getOperand(1));
3130   }
3131   // (C - x) - y  ->  C - (x + y)
3132   if (N0.hasOneUse() && N0.getOpcode() == ISD::SUB &&
3133       isConstantOrConstantVector(N0.getOperand(0), /*NoOpaques=*/true)) {
3134     SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(1), N1);
3135     return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0), Add);
3136   }
3137 
3138   // If the target's bool is represented as 0/-1, prefer to make this 'add 0/-1'
3139   // rather than 'sub 0/1' (the sext should get folded).
3140   // sub X, (zext i1 Y) --> add X, (sext i1 Y)
3141   if (N1.getOpcode() == ISD::ZERO_EXTEND &&
3142       N1.getOperand(0).getScalarValueSizeInBits() == 1 &&
3143       TLI.getBooleanContents(VT) ==
3144           TargetLowering::ZeroOrNegativeOneBooleanContent) {
3145     SDValue SExt = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, N1.getOperand(0));
3146     return DAG.getNode(ISD::ADD, DL, VT, N0, SExt);
3147   }
3148 
3149   // fold Y = sra (X, size(X)-1); sub (xor (X, Y), Y) -> (abs X)
3150   if (TLI.isOperationLegalOrCustom(ISD::ABS, VT)) {
3151     if (N0.getOpcode() == ISD::XOR && N1.getOpcode() == ISD::SRA) {
3152       SDValue X0 = N0.getOperand(0), X1 = N0.getOperand(1);
3153       SDValue S0 = N1.getOperand(0);
3154       if ((X0 == S0 && X1 == N1) || (X0 == N1 && X1 == S0)) {
3155         unsigned OpSizeInBits = VT.getScalarSizeInBits();
3156         if (ConstantSDNode *C = isConstOrConstSplat(N1.getOperand(1)))
3157           if (C->getAPIntValue() == (OpSizeInBits - 1))
3158             return DAG.getNode(ISD::ABS, SDLoc(N), VT, S0);
3159       }
3160     }
3161   }
3162 
3163   // If the relocation model supports it, consider symbol offsets.
3164   if (GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(N0))
3165     if (!LegalOperations && TLI.isOffsetFoldingLegal(GA)) {
3166       // fold (sub Sym, c) -> Sym-c
3167       if (N1C && GA->getOpcode() == ISD::GlobalAddress)
3168         return DAG.getGlobalAddress(GA->getGlobal(), SDLoc(N1C), VT,
3169                                     GA->getOffset() -
3170                                         (uint64_t)N1C->getSExtValue());
3171       // fold (sub Sym+c1, Sym+c2) -> c1-c2
3172       if (GlobalAddressSDNode *GB = dyn_cast<GlobalAddressSDNode>(N1))
3173         if (GA->getGlobal() == GB->getGlobal())
3174           return DAG.getConstant((uint64_t)GA->getOffset() - GB->getOffset(),
3175                                  DL, VT);
3176     }
3177 
3178   // sub X, (sextinreg Y i1) -> add X, (and Y 1)
3179   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
3180     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
3181     if (TN->getVT() == MVT::i1) {
3182       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
3183                                  DAG.getConstant(1, DL, VT));
3184       return DAG.getNode(ISD::ADD, DL, VT, N0, ZExt);
3185     }
3186   }
3187 
3188   // Prefer an add for more folding potential and possibly better codegen:
3189   // sub N0, (lshr N10, width-1) --> add N0, (ashr N10, width-1)
3190   if (!LegalOperations && N1.getOpcode() == ISD::SRL && N1.hasOneUse()) {
3191     SDValue ShAmt = N1.getOperand(1);
3192     ConstantSDNode *ShAmtC = isConstOrConstSplat(ShAmt);
3193     if (ShAmtC &&
3194         ShAmtC->getAPIntValue() == (N1.getScalarValueSizeInBits() - 1)) {
3195       SDValue SRA = DAG.getNode(ISD::SRA, DL, VT, N1.getOperand(0), ShAmt);
3196       return DAG.getNode(ISD::ADD, DL, VT, N0, SRA);
3197     }
3198   }
3199 
3200   return SDValue();
3201 }
3202 
3203 SDValue DAGCombiner::visitSUBSAT(SDNode *N) {
3204   SDValue N0 = N->getOperand(0);
3205   SDValue N1 = N->getOperand(1);
3206   EVT VT = N0.getValueType();
3207   SDLoc DL(N);
3208 
3209   // fold vector ops
3210   if (VT.isVector()) {
3211     // TODO SimplifyVBinOp
3212 
3213     // fold (sub_sat x, 0) -> x, vector edition
3214     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3215       return N0;
3216   }
3217 
3218   // fold (sub_sat x, undef) -> 0
3219   if (N0.isUndef() || N1.isUndef())
3220     return DAG.getConstant(0, DL, VT);
3221 
3222   // fold (sub_sat x, x) -> 0
3223   if (N0 == N1)
3224     return DAG.getConstant(0, DL, VT);
3225 
3226   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3227       DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
3228     // fold (sub_sat c1, c2) -> c3
3229     return DAG.FoldConstantArithmetic(N->getOpcode(), DL, VT, N0.getNode(),
3230                                       N1.getNode());
3231   }
3232 
3233   // fold (sub_sat x, 0) -> x
3234   if (isNullConstant(N1))
3235     return N0;
3236 
3237   return SDValue();
3238 }
3239 
3240 SDValue DAGCombiner::visitSUBC(SDNode *N) {
3241   SDValue N0 = N->getOperand(0);
3242   SDValue N1 = N->getOperand(1);
3243   EVT VT = N0.getValueType();
3244   SDLoc DL(N);
3245 
3246   // If the flag result is dead, turn this into an SUB.
3247   if (!N->hasAnyUseOfValue(1))
3248     return CombineTo(N, DAG.getNode(ISD::SUB, DL, VT, N0, N1),
3249                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
3250 
3251   // fold (subc x, x) -> 0 + no borrow
3252   if (N0 == N1)
3253     return CombineTo(N, DAG.getConstant(0, DL, VT),
3254                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
3255 
3256   // fold (subc x, 0) -> x + no borrow
3257   if (isNullConstant(N1))
3258     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
3259 
3260   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1) + no borrow
3261   if (isAllOnesConstant(N0))
3262     return CombineTo(N, DAG.getNode(ISD::XOR, DL, VT, N1, N0),
3263                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
3264 
3265   return SDValue();
3266 }
3267 
3268 SDValue DAGCombiner::visitSUBO(SDNode *N) {
3269   SDValue N0 = N->getOperand(0);
3270   SDValue N1 = N->getOperand(1);
3271   EVT VT = N0.getValueType();
3272   bool IsSigned = (ISD::SSUBO == N->getOpcode());
3273 
3274   EVT CarryVT = N->getValueType(1);
3275   SDLoc DL(N);
3276 
3277   // If the flag result is dead, turn this into an SUB.
3278   if (!N->hasAnyUseOfValue(1))
3279     return CombineTo(N, DAG.getNode(ISD::SUB, DL, VT, N0, N1),
3280                      DAG.getUNDEF(CarryVT));
3281 
3282   // fold (subo x, x) -> 0 + no borrow
3283   if (N0 == N1)
3284     return CombineTo(N, DAG.getConstant(0, DL, VT),
3285                      DAG.getConstant(0, DL, CarryVT));
3286 
3287   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
3288 
3289   // fold (subox, c) -> (addo x, -c)
3290   if (IsSigned && N1C && !N1C->getAPIntValue().isMinSignedValue()) {
3291     return DAG.getNode(ISD::SADDO, DL, N->getVTList(), N0,
3292                        DAG.getConstant(-N1C->getAPIntValue(), DL, VT));
3293   }
3294 
3295   // fold (subo x, 0) -> x + no borrow
3296   if (isNullOrNullSplat(N1))
3297     return CombineTo(N, N0, DAG.getConstant(0, DL, CarryVT));
3298 
3299   // Canonicalize (usubo -1, x) -> ~x, i.e. (xor x, -1) + no borrow
3300   if (!IsSigned && isAllOnesOrAllOnesSplat(N0))
3301     return CombineTo(N, DAG.getNode(ISD::XOR, DL, VT, N1, N0),
3302                      DAG.getConstant(0, DL, CarryVT));
3303 
3304   return SDValue();
3305 }
3306 
3307 SDValue DAGCombiner::visitSUBE(SDNode *N) {
3308   SDValue N0 = N->getOperand(0);
3309   SDValue N1 = N->getOperand(1);
3310   SDValue CarryIn = N->getOperand(2);
3311 
3312   // fold (sube x, y, false) -> (subc x, y)
3313   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
3314     return DAG.getNode(ISD::SUBC, SDLoc(N), N->getVTList(), N0, N1);
3315 
3316   return SDValue();
3317 }
3318 
3319 SDValue DAGCombiner::visitSUBCARRY(SDNode *N) {
3320   SDValue N0 = N->getOperand(0);
3321   SDValue N1 = N->getOperand(1);
3322   SDValue CarryIn = N->getOperand(2);
3323 
3324   // fold (subcarry x, y, false) -> (usubo x, y)
3325   if (isNullConstant(CarryIn)) {
3326     if (!LegalOperations ||
3327         TLI.isOperationLegalOrCustom(ISD::USUBO, N->getValueType(0)))
3328       return DAG.getNode(ISD::USUBO, SDLoc(N), N->getVTList(), N0, N1);
3329   }
3330 
3331   return SDValue();
3332 }
3333 
3334 SDValue DAGCombiner::visitMUL(SDNode *N) {
3335   SDValue N0 = N->getOperand(0);
3336   SDValue N1 = N->getOperand(1);
3337   EVT VT = N0.getValueType();
3338 
3339   // fold (mul x, undef) -> 0
3340   if (N0.isUndef() || N1.isUndef())
3341     return DAG.getConstant(0, SDLoc(N), VT);
3342 
3343   bool N0IsConst = false;
3344   bool N1IsConst = false;
3345   bool N1IsOpaqueConst = false;
3346   bool N0IsOpaqueConst = false;
3347   APInt ConstValue0, ConstValue1;
3348   // fold vector ops
3349   if (VT.isVector()) {
3350     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3351       return FoldedVOp;
3352 
3353     N0IsConst = ISD::isConstantSplatVector(N0.getNode(), ConstValue0);
3354     N1IsConst = ISD::isConstantSplatVector(N1.getNode(), ConstValue1);
3355     assert((!N0IsConst ||
3356             ConstValue0.getBitWidth() == VT.getScalarSizeInBits()) &&
3357            "Splat APInt should be element width");
3358     assert((!N1IsConst ||
3359             ConstValue1.getBitWidth() == VT.getScalarSizeInBits()) &&
3360            "Splat APInt should be element width");
3361   } else {
3362     N0IsConst = isa<ConstantSDNode>(N0);
3363     if (N0IsConst) {
3364       ConstValue0 = cast<ConstantSDNode>(N0)->getAPIntValue();
3365       N0IsOpaqueConst = cast<ConstantSDNode>(N0)->isOpaque();
3366     }
3367     N1IsConst = isa<ConstantSDNode>(N1);
3368     if (N1IsConst) {
3369       ConstValue1 = cast<ConstantSDNode>(N1)->getAPIntValue();
3370       N1IsOpaqueConst = cast<ConstantSDNode>(N1)->isOpaque();
3371     }
3372   }
3373 
3374   // fold (mul c1, c2) -> c1*c2
3375   if (N0IsConst && N1IsConst && !N0IsOpaqueConst && !N1IsOpaqueConst)
3376     return DAG.FoldConstantArithmetic(ISD::MUL, SDLoc(N), VT,
3377                                       N0.getNode(), N1.getNode());
3378 
3379   // canonicalize constant to RHS (vector doesn't have to splat)
3380   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3381      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
3382     return DAG.getNode(ISD::MUL, SDLoc(N), VT, N1, N0);
3383   // fold (mul x, 0) -> 0
3384   if (N1IsConst && ConstValue1.isNullValue())
3385     return N1;
3386   // fold (mul x, 1) -> x
3387   if (N1IsConst && ConstValue1.isOneValue())
3388     return N0;
3389 
3390   if (SDValue NewSel = foldBinOpIntoSelect(N))
3391     return NewSel;
3392 
3393   // fold (mul x, -1) -> 0-x
3394   if (N1IsConst && ConstValue1.isAllOnesValue()) {
3395     SDLoc DL(N);
3396     return DAG.getNode(ISD::SUB, DL, VT,
3397                        DAG.getConstant(0, DL, VT), N0);
3398   }
3399   // fold (mul x, (1 << c)) -> x << c
3400   if (isConstantOrConstantVector(N1, /*NoOpaques*/ true) &&
3401       DAG.isKnownToBeAPowerOfTwo(N1) &&
3402       (!VT.isVector() || Level <= AfterLegalizeVectorOps)) {
3403     SDLoc DL(N);
3404     SDValue LogBase2 = BuildLogBase2(N1, DL);
3405     EVT ShiftVT = getShiftAmountTy(N0.getValueType());
3406     SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ShiftVT);
3407     return DAG.getNode(ISD::SHL, DL, VT, N0, Trunc);
3408   }
3409   // fold (mul x, -(1 << c)) -> -(x << c) or (-x) << c
3410   if (N1IsConst && !N1IsOpaqueConst && (-ConstValue1).isPowerOf2()) {
3411     unsigned Log2Val = (-ConstValue1).logBase2();
3412     SDLoc DL(N);
3413     // FIXME: If the input is something that is easily negated (e.g. a
3414     // single-use add), we should put the negate there.
3415     return DAG.getNode(ISD::SUB, DL, VT,
3416                        DAG.getConstant(0, DL, VT),
3417                        DAG.getNode(ISD::SHL, DL, VT, N0,
3418                             DAG.getConstant(Log2Val, DL,
3419                                       getShiftAmountTy(N0.getValueType()))));
3420   }
3421 
3422   // Try to transform multiply-by-(power-of-2 +/- 1) into shift and add/sub.
3423   // mul x, (2^N + 1) --> add (shl x, N), x
3424   // mul x, (2^N - 1) --> sub (shl x, N), x
3425   // Examples: x * 33 --> (x << 5) + x
3426   //           x * 15 --> (x << 4) - x
3427   //           x * -33 --> -((x << 5) + x)
3428   //           x * -15 --> -((x << 4) - x) ; this reduces --> x - (x << 4)
3429   if (N1IsConst && TLI.decomposeMulByConstant(VT, N1)) {
3430     // TODO: We could handle more general decomposition of any constant by
3431     //       having the target set a limit on number of ops and making a
3432     //       callback to determine that sequence (similar to sqrt expansion).
3433     unsigned MathOp = ISD::DELETED_NODE;
3434     APInt MulC = ConstValue1.abs();
3435     if ((MulC - 1).isPowerOf2())
3436       MathOp = ISD::ADD;
3437     else if ((MulC + 1).isPowerOf2())
3438       MathOp = ISD::SUB;
3439 
3440     if (MathOp != ISD::DELETED_NODE) {
3441       unsigned ShAmt =
3442           MathOp == ISD::ADD ? (MulC - 1).logBase2() : (MulC + 1).logBase2();
3443       assert(ShAmt < VT.getScalarSizeInBits() &&
3444              "multiply-by-constant generated out of bounds shift");
3445       SDLoc DL(N);
3446       SDValue Shl =
3447           DAG.getNode(ISD::SHL, DL, VT, N0, DAG.getConstant(ShAmt, DL, VT));
3448       SDValue R = DAG.getNode(MathOp, DL, VT, Shl, N0);
3449       if (ConstValue1.isNegative())
3450         R = DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), R);
3451       return R;
3452     }
3453   }
3454 
3455   // (mul (shl X, c1), c2) -> (mul X, c2 << c1)
3456   if (N0.getOpcode() == ISD::SHL &&
3457       isConstantOrConstantVector(N1, /* NoOpaques */ true) &&
3458       isConstantOrConstantVector(N0.getOperand(1), /* NoOpaques */ true)) {
3459     SDValue C3 = DAG.getNode(ISD::SHL, SDLoc(N), VT, N1, N0.getOperand(1));
3460     if (isConstantOrConstantVector(C3))
3461       return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), C3);
3462   }
3463 
3464   // Change (mul (shl X, C), Y) -> (shl (mul X, Y), C) when the shift has one
3465   // use.
3466   {
3467     SDValue Sh(nullptr, 0), Y(nullptr, 0);
3468 
3469     // Check for both (mul (shl X, C), Y)  and  (mul Y, (shl X, C)).
3470     if (N0.getOpcode() == ISD::SHL &&
3471         isConstantOrConstantVector(N0.getOperand(1)) &&
3472         N0.getNode()->hasOneUse()) {
3473       Sh = N0; Y = N1;
3474     } else if (N1.getOpcode() == ISD::SHL &&
3475                isConstantOrConstantVector(N1.getOperand(1)) &&
3476                N1.getNode()->hasOneUse()) {
3477       Sh = N1; Y = N0;
3478     }
3479 
3480     if (Sh.getNode()) {
3481       SDValue Mul = DAG.getNode(ISD::MUL, SDLoc(N), VT, Sh.getOperand(0), Y);
3482       return DAG.getNode(ISD::SHL, SDLoc(N), VT, Mul, Sh.getOperand(1));
3483     }
3484   }
3485 
3486   // fold (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2)
3487   if (DAG.isConstantIntBuildVectorOrConstantInt(N1) &&
3488       N0.getOpcode() == ISD::ADD &&
3489       DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1)) &&
3490       isMulAddWithConstProfitable(N, N0, N1))
3491       return DAG.getNode(ISD::ADD, SDLoc(N), VT,
3492                          DAG.getNode(ISD::MUL, SDLoc(N0), VT,
3493                                      N0.getOperand(0), N1),
3494                          DAG.getNode(ISD::MUL, SDLoc(N1), VT,
3495                                      N0.getOperand(1), N1));
3496 
3497   // reassociate mul
3498   if (SDValue RMUL = reassociateOps(ISD::MUL, SDLoc(N), N0, N1, N->getFlags()))
3499     return RMUL;
3500 
3501   return SDValue();
3502 }
3503 
3504 /// Return true if divmod libcall is available.
3505 static bool isDivRemLibcallAvailable(SDNode *Node, bool isSigned,
3506                                      const TargetLowering &TLI) {
3507   RTLIB::Libcall LC;
3508   EVT NodeType = Node->getValueType(0);
3509   if (!NodeType.isSimple())
3510     return false;
3511   switch (NodeType.getSimpleVT().SimpleTy) {
3512   default: return false; // No libcall for vector types.
3513   case MVT::i8:   LC= isSigned ? RTLIB::SDIVREM_I8  : RTLIB::UDIVREM_I8;  break;
3514   case MVT::i16:  LC= isSigned ? RTLIB::SDIVREM_I16 : RTLIB::UDIVREM_I16; break;
3515   case MVT::i32:  LC= isSigned ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32; break;
3516   case MVT::i64:  LC= isSigned ? RTLIB::SDIVREM_I64 : RTLIB::UDIVREM_I64; break;
3517   case MVT::i128: LC= isSigned ? RTLIB::SDIVREM_I128:RTLIB::UDIVREM_I128; break;
3518   }
3519 
3520   return TLI.getLibcallName(LC) != nullptr;
3521 }
3522 
3523 /// Issue divrem if both quotient and remainder are needed.
3524 SDValue DAGCombiner::useDivRem(SDNode *Node) {
3525   if (Node->use_empty())
3526     return SDValue(); // This is a dead node, leave it alone.
3527 
3528   unsigned Opcode = Node->getOpcode();
3529   bool isSigned = (Opcode == ISD::SDIV) || (Opcode == ISD::SREM);
3530   unsigned DivRemOpc = isSigned ? ISD::SDIVREM : ISD::UDIVREM;
3531 
3532   // DivMod lib calls can still work on non-legal types if using lib-calls.
3533   EVT VT = Node->getValueType(0);
3534   if (VT.isVector() || !VT.isInteger())
3535     return SDValue();
3536 
3537   if (!TLI.isTypeLegal(VT) && !TLI.isOperationCustom(DivRemOpc, VT))
3538     return SDValue();
3539 
3540   // If DIVREM is going to get expanded into a libcall,
3541   // but there is no libcall available, then don't combine.
3542   if (!TLI.isOperationLegalOrCustom(DivRemOpc, VT) &&
3543       !isDivRemLibcallAvailable(Node, isSigned, TLI))
3544     return SDValue();
3545 
3546   // If div is legal, it's better to do the normal expansion
3547   unsigned OtherOpcode = 0;
3548   if ((Opcode == ISD::SDIV) || (Opcode == ISD::UDIV)) {
3549     OtherOpcode = isSigned ? ISD::SREM : ISD::UREM;
3550     if (TLI.isOperationLegalOrCustom(Opcode, VT))
3551       return SDValue();
3552   } else {
3553     OtherOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
3554     if (TLI.isOperationLegalOrCustom(OtherOpcode, VT))
3555       return SDValue();
3556   }
3557 
3558   SDValue Op0 = Node->getOperand(0);
3559   SDValue Op1 = Node->getOperand(1);
3560   SDValue combined;
3561   for (SDNode::use_iterator UI = Op0.getNode()->use_begin(),
3562          UE = Op0.getNode()->use_end(); UI != UE; ++UI) {
3563     SDNode *User = *UI;
3564     if (User == Node || User->getOpcode() == ISD::DELETED_NODE ||
3565         User->use_empty())
3566       continue;
3567     // Convert the other matching node(s), too;
3568     // otherwise, the DIVREM may get target-legalized into something
3569     // target-specific that we won't be able to recognize.
3570     unsigned UserOpc = User->getOpcode();
3571     if ((UserOpc == Opcode || UserOpc == OtherOpcode || UserOpc == DivRemOpc) &&
3572         User->getOperand(0) == Op0 &&
3573         User->getOperand(1) == Op1) {
3574       if (!combined) {
3575         if (UserOpc == OtherOpcode) {
3576           SDVTList VTs = DAG.getVTList(VT, VT);
3577           combined = DAG.getNode(DivRemOpc, SDLoc(Node), VTs, Op0, Op1);
3578         } else if (UserOpc == DivRemOpc) {
3579           combined = SDValue(User, 0);
3580         } else {
3581           assert(UserOpc == Opcode);
3582           continue;
3583         }
3584       }
3585       if (UserOpc == ISD::SDIV || UserOpc == ISD::UDIV)
3586         CombineTo(User, combined);
3587       else if (UserOpc == ISD::SREM || UserOpc == ISD::UREM)
3588         CombineTo(User, combined.getValue(1));
3589     }
3590   }
3591   return combined;
3592 }
3593 
3594 static SDValue simplifyDivRem(SDNode *N, SelectionDAG &DAG) {
3595   SDValue N0 = N->getOperand(0);
3596   SDValue N1 = N->getOperand(1);
3597   EVT VT = N->getValueType(0);
3598   SDLoc DL(N);
3599 
3600   unsigned Opc = N->getOpcode();
3601   bool IsDiv = (ISD::SDIV == Opc) || (ISD::UDIV == Opc);
3602   ConstantSDNode *N1C = isConstOrConstSplat(N1);
3603 
3604   // X / undef -> undef
3605   // X % undef -> undef
3606   // X / 0 -> undef
3607   // X % 0 -> undef
3608   // NOTE: This includes vectors where any divisor element is zero/undef.
3609   if (DAG.isUndef(Opc, {N0, N1}))
3610     return DAG.getUNDEF(VT);
3611 
3612   // undef / X -> 0
3613   // undef % X -> 0
3614   if (N0.isUndef())
3615     return DAG.getConstant(0, DL, VT);
3616 
3617   // 0 / X -> 0
3618   // 0 % X -> 0
3619   ConstantSDNode *N0C = isConstOrConstSplat(N0);
3620   if (N0C && N0C->isNullValue())
3621     return N0;
3622 
3623   // X / X -> 1
3624   // X % X -> 0
3625   if (N0 == N1)
3626     return DAG.getConstant(IsDiv ? 1 : 0, DL, VT);
3627 
3628   // X / 1 -> X
3629   // X % 1 -> 0
3630   // If this is a boolean op (single-bit element type), we can't have
3631   // division-by-zero or remainder-by-zero, so assume the divisor is 1.
3632   // TODO: Similarly, if we're zero-extending a boolean divisor, then assume
3633   // it's a 1.
3634   if ((N1C && N1C->isOne()) || (VT.getScalarType() == MVT::i1))
3635     return IsDiv ? N0 : DAG.getConstant(0, DL, VT);
3636 
3637   return SDValue();
3638 }
3639 
3640 SDValue DAGCombiner::visitSDIV(SDNode *N) {
3641   SDValue N0 = N->getOperand(0);
3642   SDValue N1 = N->getOperand(1);
3643   EVT VT = N->getValueType(0);
3644   EVT CCVT = getSetCCResultType(VT);
3645 
3646   // fold vector ops
3647   if (VT.isVector())
3648     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3649       return FoldedVOp;
3650 
3651   SDLoc DL(N);
3652 
3653   // fold (sdiv c1, c2) -> c1/c2
3654   ConstantSDNode *N0C = isConstOrConstSplat(N0);
3655   ConstantSDNode *N1C = isConstOrConstSplat(N1);
3656   if (N0C && N1C && !N0C->isOpaque() && !N1C->isOpaque())
3657     return DAG.FoldConstantArithmetic(ISD::SDIV, DL, VT, N0C, N1C);
3658   // fold (sdiv X, -1) -> 0-X
3659   if (N1C && N1C->isAllOnesValue())
3660     return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), N0);
3661   // fold (sdiv X, MIN_SIGNED) -> select(X == MIN_SIGNED, 1, 0)
3662   if (N1C && N1C->getAPIntValue().isMinSignedValue())
3663     return DAG.getSelect(DL, VT, DAG.getSetCC(DL, CCVT, N0, N1, ISD::SETEQ),
3664                          DAG.getConstant(1, DL, VT),
3665                          DAG.getConstant(0, DL, VT));
3666 
3667   if (SDValue V = simplifyDivRem(N, DAG))
3668     return V;
3669 
3670   if (SDValue NewSel = foldBinOpIntoSelect(N))
3671     return NewSel;
3672 
3673   // If we know the sign bits of both operands are zero, strength reduce to a
3674   // udiv instead.  Handles (X&15) /s 4 -> X&15 >> 2
3675   if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
3676     return DAG.getNode(ISD::UDIV, DL, N1.getValueType(), N0, N1);
3677 
3678   if (SDValue V = visitSDIVLike(N0, N1, N)) {
3679     // If the corresponding remainder node exists, update its users with
3680     // (Dividend - (Quotient * Divisor).
3681     if (SDNode *RemNode = DAG.getNodeIfExists(ISD::SREM, N->getVTList(),
3682                                               { N0, N1 })) {
3683       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, V, N1);
3684       SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul);
3685       AddToWorklist(Mul.getNode());
3686       AddToWorklist(Sub.getNode());
3687       CombineTo(RemNode, Sub);
3688     }
3689     return V;
3690   }
3691 
3692   // sdiv, srem -> sdivrem
3693   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is
3694   // true.  Otherwise, we break the simplification logic in visitREM().
3695   AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes();
3696   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
3697     if (SDValue DivRem = useDivRem(N))
3698         return DivRem;
3699 
3700   return SDValue();
3701 }
3702 
3703 SDValue DAGCombiner::visitSDIVLike(SDValue N0, SDValue N1, SDNode *N) {
3704   SDLoc DL(N);
3705   EVT VT = N->getValueType(0);
3706   EVT CCVT = getSetCCResultType(VT);
3707   unsigned BitWidth = VT.getScalarSizeInBits();
3708 
3709   // Helper for determining whether a value is a power-2 constant scalar or a
3710   // vector of such elements.
3711   auto IsPowerOfTwo = [](ConstantSDNode *C) {
3712     if (C->isNullValue() || C->isOpaque())
3713       return false;
3714     if (C->getAPIntValue().isPowerOf2())
3715       return true;
3716     if ((-C->getAPIntValue()).isPowerOf2())
3717       return true;
3718     return false;
3719   };
3720 
3721   // fold (sdiv X, pow2) -> simple ops after legalize
3722   // FIXME: We check for the exact bit here because the generic lowering gives
3723   // better results in that case. The target-specific lowering should learn how
3724   // to handle exact sdivs efficiently.
3725   if (!N->getFlags().hasExact() && ISD::matchUnaryPredicate(N1, IsPowerOfTwo)) {
3726     // Target-specific implementation of sdiv x, pow2.
3727     if (SDValue Res = BuildSDIVPow2(N))
3728       return Res;
3729 
3730     // Create constants that are functions of the shift amount value.
3731     EVT ShiftAmtTy = getShiftAmountTy(N0.getValueType());
3732     SDValue Bits = DAG.getConstant(BitWidth, DL, ShiftAmtTy);
3733     SDValue C1 = DAG.getNode(ISD::CTTZ, DL, VT, N1);
3734     C1 = DAG.getZExtOrTrunc(C1, DL, ShiftAmtTy);
3735     SDValue Inexact = DAG.getNode(ISD::SUB, DL, ShiftAmtTy, Bits, C1);
3736     if (!isConstantOrConstantVector(Inexact))
3737       return SDValue();
3738 
3739     // Splat the sign bit into the register
3740     SDValue Sign = DAG.getNode(ISD::SRA, DL, VT, N0,
3741                                DAG.getConstant(BitWidth - 1, DL, ShiftAmtTy));
3742     AddToWorklist(Sign.getNode());
3743 
3744     // Add (N0 < 0) ? abs2 - 1 : 0;
3745     SDValue Srl = DAG.getNode(ISD::SRL, DL, VT, Sign, Inexact);
3746     AddToWorklist(Srl.getNode());
3747     SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N0, Srl);
3748     AddToWorklist(Add.getNode());
3749     SDValue Sra = DAG.getNode(ISD::SRA, DL, VT, Add, C1);
3750     AddToWorklist(Sra.getNode());
3751 
3752     // Special case: (sdiv X, 1) -> X
3753     // Special Case: (sdiv X, -1) -> 0-X
3754     SDValue One = DAG.getConstant(1, DL, VT);
3755     SDValue AllOnes = DAG.getAllOnesConstant(DL, VT);
3756     SDValue IsOne = DAG.getSetCC(DL, CCVT, N1, One, ISD::SETEQ);
3757     SDValue IsAllOnes = DAG.getSetCC(DL, CCVT, N1, AllOnes, ISD::SETEQ);
3758     SDValue IsOneOrAllOnes = DAG.getNode(ISD::OR, DL, CCVT, IsOne, IsAllOnes);
3759     Sra = DAG.getSelect(DL, VT, IsOneOrAllOnes, N0, Sra);
3760 
3761     // If dividing by a positive value, we're done. Otherwise, the result must
3762     // be negated.
3763     SDValue Zero = DAG.getConstant(0, DL, VT);
3764     SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, Zero, Sra);
3765 
3766     // FIXME: Use SELECT_CC once we improve SELECT_CC constant-folding.
3767     SDValue IsNeg = DAG.getSetCC(DL, CCVT, N1, Zero, ISD::SETLT);
3768     SDValue Res = DAG.getSelect(DL, VT, IsNeg, Sub, Sra);
3769     return Res;
3770   }
3771 
3772   // If integer divide is expensive and we satisfy the requirements, emit an
3773   // alternate sequence.  Targets may check function attributes for size/speed
3774   // trade-offs.
3775   AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes();
3776   if (isConstantOrConstantVector(N1) &&
3777       !TLI.isIntDivCheap(N->getValueType(0), Attr))
3778     if (SDValue Op = BuildSDIV(N))
3779       return Op;
3780 
3781   return SDValue();
3782 }
3783 
3784 SDValue DAGCombiner::visitUDIV(SDNode *N) {
3785   SDValue N0 = N->getOperand(0);
3786   SDValue N1 = N->getOperand(1);
3787   EVT VT = N->getValueType(0);
3788   EVT CCVT = getSetCCResultType(VT);
3789 
3790   // fold vector ops
3791   if (VT.isVector())
3792     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3793       return FoldedVOp;
3794 
3795   SDLoc DL(N);
3796 
3797   // fold (udiv c1, c2) -> c1/c2
3798   ConstantSDNode *N0C = isConstOrConstSplat(N0);
3799   ConstantSDNode *N1C = isConstOrConstSplat(N1);
3800   if (N0C && N1C)
3801     if (SDValue Folded = DAG.FoldConstantArithmetic(ISD::UDIV, DL, VT,
3802                                                     N0C, N1C))
3803       return Folded;
3804   // fold (udiv X, -1) -> select(X == -1, 1, 0)
3805   if (N1C && N1C->getAPIntValue().isAllOnesValue())
3806     return DAG.getSelect(DL, VT, DAG.getSetCC(DL, CCVT, N0, N1, ISD::SETEQ),
3807                          DAG.getConstant(1, DL, VT),
3808                          DAG.getConstant(0, DL, VT));
3809 
3810   if (SDValue V = simplifyDivRem(N, DAG))
3811     return V;
3812 
3813   if (SDValue NewSel = foldBinOpIntoSelect(N))
3814     return NewSel;
3815 
3816   if (SDValue V = visitUDIVLike(N0, N1, N)) {
3817     // If the corresponding remainder node exists, update its users with
3818     // (Dividend - (Quotient * Divisor).
3819     if (SDNode *RemNode = DAG.getNodeIfExists(ISD::UREM, N->getVTList(),
3820                                               { N0, N1 })) {
3821       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, V, N1);
3822       SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul);
3823       AddToWorklist(Mul.getNode());
3824       AddToWorklist(Sub.getNode());
3825       CombineTo(RemNode, Sub);
3826     }
3827     return V;
3828   }
3829 
3830   // sdiv, srem -> sdivrem
3831   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is
3832   // true.  Otherwise, we break the simplification logic in visitREM().
3833   AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes();
3834   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
3835     if (SDValue DivRem = useDivRem(N))
3836         return DivRem;
3837 
3838   return SDValue();
3839 }
3840 
3841 SDValue DAGCombiner::visitUDIVLike(SDValue N0, SDValue N1, SDNode *N) {
3842   SDLoc DL(N);
3843   EVT VT = N->getValueType(0);
3844 
3845   // fold (udiv x, (1 << c)) -> x >>u c
3846   if (isConstantOrConstantVector(N1, /*NoOpaques*/ true) &&
3847       DAG.isKnownToBeAPowerOfTwo(N1)) {
3848     SDValue LogBase2 = BuildLogBase2(N1, DL);
3849     AddToWorklist(LogBase2.getNode());
3850 
3851     EVT ShiftVT = getShiftAmountTy(N0.getValueType());
3852     SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ShiftVT);
3853     AddToWorklist(Trunc.getNode());
3854     return DAG.getNode(ISD::SRL, DL, VT, N0, Trunc);
3855   }
3856 
3857   // fold (udiv x, (shl c, y)) -> x >>u (log2(c)+y) iff c is power of 2
3858   if (N1.getOpcode() == ISD::SHL) {
3859     SDValue N10 = N1.getOperand(0);
3860     if (isConstantOrConstantVector(N10, /*NoOpaques*/ true) &&
3861         DAG.isKnownToBeAPowerOfTwo(N10)) {
3862       SDValue LogBase2 = BuildLogBase2(N10, DL);
3863       AddToWorklist(LogBase2.getNode());
3864 
3865       EVT ADDVT = N1.getOperand(1).getValueType();
3866       SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ADDVT);
3867       AddToWorklist(Trunc.getNode());
3868       SDValue Add = DAG.getNode(ISD::ADD, DL, ADDVT, N1.getOperand(1), Trunc);
3869       AddToWorklist(Add.getNode());
3870       return DAG.getNode(ISD::SRL, DL, VT, N0, Add);
3871     }
3872   }
3873 
3874   // fold (udiv x, c) -> alternate
3875   AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes();
3876   if (isConstantOrConstantVector(N1) &&
3877       !TLI.isIntDivCheap(N->getValueType(0), Attr))
3878     if (SDValue Op = BuildUDIV(N))
3879       return Op;
3880 
3881   return SDValue();
3882 }
3883 
3884 // handles ISD::SREM and ISD::UREM
3885 SDValue DAGCombiner::visitREM(SDNode *N) {
3886   unsigned Opcode = N->getOpcode();
3887   SDValue N0 = N->getOperand(0);
3888   SDValue N1 = N->getOperand(1);
3889   EVT VT = N->getValueType(0);
3890   EVT CCVT = getSetCCResultType(VT);
3891 
3892   bool isSigned = (Opcode == ISD::SREM);
3893   SDLoc DL(N);
3894 
3895   // fold (rem c1, c2) -> c1%c2
3896   ConstantSDNode *N0C = isConstOrConstSplat(N0);
3897   ConstantSDNode *N1C = isConstOrConstSplat(N1);
3898   if (N0C && N1C)
3899     if (SDValue Folded = DAG.FoldConstantArithmetic(Opcode, DL, VT, N0C, N1C))
3900       return Folded;
3901   // fold (urem X, -1) -> select(X == -1, 0, x)
3902   if (!isSigned && N1C && N1C->getAPIntValue().isAllOnesValue())
3903     return DAG.getSelect(DL, VT, DAG.getSetCC(DL, CCVT, N0, N1, ISD::SETEQ),
3904                          DAG.getConstant(0, DL, VT), N0);
3905 
3906   if (SDValue V = simplifyDivRem(N, DAG))
3907     return V;
3908 
3909   if (SDValue NewSel = foldBinOpIntoSelect(N))
3910     return NewSel;
3911 
3912   if (isSigned) {
3913     // If we know the sign bits of both operands are zero, strength reduce to a
3914     // urem instead.  Handles (X & 0x0FFFFFFF) %s 16 -> X&15
3915     if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
3916       return DAG.getNode(ISD::UREM, DL, VT, N0, N1);
3917   } else {
3918     SDValue NegOne = DAG.getAllOnesConstant(DL, VT);
3919     if (DAG.isKnownToBeAPowerOfTwo(N1)) {
3920       // fold (urem x, pow2) -> (and x, pow2-1)
3921       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N1, NegOne);
3922       AddToWorklist(Add.getNode());
3923       return DAG.getNode(ISD::AND, DL, VT, N0, Add);
3924     }
3925     if (N1.getOpcode() == ISD::SHL &&
3926         DAG.isKnownToBeAPowerOfTwo(N1.getOperand(0))) {
3927       // fold (urem x, (shl pow2, y)) -> (and x, (add (shl pow2, y), -1))
3928       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N1, NegOne);
3929       AddToWorklist(Add.getNode());
3930       return DAG.getNode(ISD::AND, DL, VT, N0, Add);
3931     }
3932   }
3933 
3934   AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes();
3935 
3936   // If X/C can be simplified by the division-by-constant logic, lower
3937   // X%C to the equivalent of X-X/C*C.
3938   // Reuse the SDIVLike/UDIVLike combines - to avoid mangling nodes, the
3939   // speculative DIV must not cause a DIVREM conversion.  We guard against this
3940   // by skipping the simplification if isIntDivCheap().  When div is not cheap,
3941   // combine will not return a DIVREM.  Regardless, checking cheapness here
3942   // makes sense since the simplification results in fatter code.
3943   if (DAG.isKnownNeverZero(N1) && !TLI.isIntDivCheap(VT, Attr)) {
3944     SDValue OptimizedDiv =
3945         isSigned ? visitSDIVLike(N0, N1, N) : visitUDIVLike(N0, N1, N);
3946     if (OptimizedDiv.getNode()) {
3947       // If the equivalent Div node also exists, update its users.
3948       unsigned DivOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
3949       if (SDNode *DivNode = DAG.getNodeIfExists(DivOpcode, N->getVTList(),
3950                                                 { N0, N1 }))
3951         CombineTo(DivNode, OptimizedDiv);
3952       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, OptimizedDiv, N1);
3953       SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul);
3954       AddToWorklist(OptimizedDiv.getNode());
3955       AddToWorklist(Mul.getNode());
3956       return Sub;
3957     }
3958   }
3959 
3960   // sdiv, srem -> sdivrem
3961   if (SDValue DivRem = useDivRem(N))
3962     return DivRem.getValue(1);
3963 
3964   return SDValue();
3965 }
3966 
3967 SDValue DAGCombiner::visitMULHS(SDNode *N) {
3968   SDValue N0 = N->getOperand(0);
3969   SDValue N1 = N->getOperand(1);
3970   EVT VT = N->getValueType(0);
3971   SDLoc DL(N);
3972 
3973   if (VT.isVector()) {
3974     // fold (mulhs x, 0) -> 0
3975     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3976       return N1;
3977     if (ISD::isBuildVectorAllZeros(N0.getNode()))
3978       return N0;
3979   }
3980 
3981   // fold (mulhs x, 0) -> 0
3982   if (isNullConstant(N1))
3983     return N1;
3984   // fold (mulhs x, 1) -> (sra x, size(x)-1)
3985   if (isOneConstant(N1))
3986     return DAG.getNode(ISD::SRA, DL, N0.getValueType(), N0,
3987                        DAG.getConstant(N0.getValueSizeInBits() - 1, DL,
3988                                        getShiftAmountTy(N0.getValueType())));
3989 
3990   // fold (mulhs x, undef) -> 0
3991   if (N0.isUndef() || N1.isUndef())
3992     return DAG.getConstant(0, DL, VT);
3993 
3994   // If the type twice as wide is legal, transform the mulhs to a wider multiply
3995   // plus a shift.
3996   if (VT.isSimple() && !VT.isVector()) {
3997     MVT Simple = VT.getSimpleVT();
3998     unsigned SimpleSize = Simple.getSizeInBits();
3999     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
4000     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
4001       N0 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N0);
4002       N1 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N1);
4003       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
4004       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
4005             DAG.getConstant(SimpleSize, DL,
4006                             getShiftAmountTy(N1.getValueType())));
4007       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
4008     }
4009   }
4010 
4011   return SDValue();
4012 }
4013 
4014 SDValue DAGCombiner::visitMULHU(SDNode *N) {
4015   SDValue N0 = N->getOperand(0);
4016   SDValue N1 = N->getOperand(1);
4017   EVT VT = N->getValueType(0);
4018   SDLoc DL(N);
4019 
4020   if (VT.isVector()) {
4021     // fold (mulhu x, 0) -> 0
4022     if (ISD::isBuildVectorAllZeros(N1.getNode()))
4023       return N1;
4024     if (ISD::isBuildVectorAllZeros(N0.getNode()))
4025       return N0;
4026   }
4027 
4028   // fold (mulhu x, 0) -> 0
4029   if (isNullConstant(N1))
4030     return N1;
4031   // fold (mulhu x, 1) -> 0
4032   if (isOneConstant(N1))
4033     return DAG.getConstant(0, DL, N0.getValueType());
4034   // fold (mulhu x, undef) -> 0
4035   if (N0.isUndef() || N1.isUndef())
4036     return DAG.getConstant(0, DL, VT);
4037 
4038   // fold (mulhu x, (1 << c)) -> x >> (bitwidth - c)
4039   if (isConstantOrConstantVector(N1, /*NoOpaques*/ true) &&
4040       DAG.isKnownToBeAPowerOfTwo(N1) && hasOperation(ISD::SRL, VT)) {
4041     unsigned NumEltBits = VT.getScalarSizeInBits();
4042     SDValue LogBase2 = BuildLogBase2(N1, DL);
4043     SDValue SRLAmt = DAG.getNode(
4044         ISD::SUB, DL, VT, DAG.getConstant(NumEltBits, DL, VT), LogBase2);
4045     EVT ShiftVT = getShiftAmountTy(N0.getValueType());
4046     SDValue Trunc = DAG.getZExtOrTrunc(SRLAmt, DL, ShiftVT);
4047     return DAG.getNode(ISD::SRL, DL, VT, N0, Trunc);
4048   }
4049 
4050   // If the type twice as wide is legal, transform the mulhu to a wider multiply
4051   // plus a shift.
4052   if (VT.isSimple() && !VT.isVector()) {
4053     MVT Simple = VT.getSimpleVT();
4054     unsigned SimpleSize = Simple.getSizeInBits();
4055     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
4056     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
4057       N0 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N0);
4058       N1 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N1);
4059       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
4060       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
4061             DAG.getConstant(SimpleSize, DL,
4062                             getShiftAmountTy(N1.getValueType())));
4063       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
4064     }
4065   }
4066 
4067   return SDValue();
4068 }
4069 
4070 /// Perform optimizations common to nodes that compute two values. LoOp and HiOp
4071 /// give the opcodes for the two computations that are being performed. Return
4072 /// true if a simplification was made.
4073 SDValue DAGCombiner::SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
4074                                                 unsigned HiOp) {
4075   // If the high half is not needed, just compute the low half.
4076   bool HiExists = N->hasAnyUseOfValue(1);
4077   if (!HiExists && (!LegalOperations ||
4078                     TLI.isOperationLegalOrCustom(LoOp, N->getValueType(0)))) {
4079     SDValue Res = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
4080     return CombineTo(N, Res, Res);
4081   }
4082 
4083   // If the low half is not needed, just compute the high half.
4084   bool LoExists = N->hasAnyUseOfValue(0);
4085   if (!LoExists && (!LegalOperations ||
4086                     TLI.isOperationLegalOrCustom(HiOp, N->getValueType(1)))) {
4087     SDValue Res = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
4088     return CombineTo(N, Res, Res);
4089   }
4090 
4091   // If both halves are used, return as it is.
4092   if (LoExists && HiExists)
4093     return SDValue();
4094 
4095   // If the two computed results can be simplified separately, separate them.
4096   if (LoExists) {
4097     SDValue Lo = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
4098     AddToWorklist(Lo.getNode());
4099     SDValue LoOpt = combine(Lo.getNode());
4100     if (LoOpt.getNode() && LoOpt.getNode() != Lo.getNode() &&
4101         (!LegalOperations ||
4102          TLI.isOperationLegalOrCustom(LoOpt.getOpcode(), LoOpt.getValueType())))
4103       return CombineTo(N, LoOpt, LoOpt);
4104   }
4105 
4106   if (HiExists) {
4107     SDValue Hi = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
4108     AddToWorklist(Hi.getNode());
4109     SDValue HiOpt = combine(Hi.getNode());
4110     if (HiOpt.getNode() && HiOpt != Hi &&
4111         (!LegalOperations ||
4112          TLI.isOperationLegalOrCustom(HiOpt.getOpcode(), HiOpt.getValueType())))
4113       return CombineTo(N, HiOpt, HiOpt);
4114   }
4115 
4116   return SDValue();
4117 }
4118 
4119 SDValue DAGCombiner::visitSMUL_LOHI(SDNode *N) {
4120   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHS))
4121     return Res;
4122 
4123   EVT VT = N->getValueType(0);
4124   SDLoc DL(N);
4125 
4126   // If the type is twice as wide is legal, transform the mulhu to a wider
4127   // multiply plus a shift.
4128   if (VT.isSimple() && !VT.isVector()) {
4129     MVT Simple = VT.getSimpleVT();
4130     unsigned SimpleSize = Simple.getSizeInBits();
4131     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
4132     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
4133       SDValue Lo = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(0));
4134       SDValue Hi = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(1));
4135       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
4136       // Compute the high part as N1.
4137       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
4138             DAG.getConstant(SimpleSize, DL,
4139                             getShiftAmountTy(Lo.getValueType())));
4140       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
4141       // Compute the low part as N0.
4142       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
4143       return CombineTo(N, Lo, Hi);
4144     }
4145   }
4146 
4147   return SDValue();
4148 }
4149 
4150 SDValue DAGCombiner::visitUMUL_LOHI(SDNode *N) {
4151   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHU))
4152     return Res;
4153 
4154   EVT VT = N->getValueType(0);
4155   SDLoc DL(N);
4156 
4157   // If the type is twice as wide is legal, transform the mulhu to a wider
4158   // multiply plus a shift.
4159   if (VT.isSimple() && !VT.isVector()) {
4160     MVT Simple = VT.getSimpleVT();
4161     unsigned SimpleSize = Simple.getSizeInBits();
4162     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
4163     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
4164       SDValue Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(0));
4165       SDValue Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(1));
4166       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
4167       // Compute the high part as N1.
4168       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
4169             DAG.getConstant(SimpleSize, DL,
4170                             getShiftAmountTy(Lo.getValueType())));
4171       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
4172       // Compute the low part as N0.
4173       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
4174       return CombineTo(N, Lo, Hi);
4175     }
4176   }
4177 
4178   return SDValue();
4179 }
4180 
4181 SDValue DAGCombiner::visitMULO(SDNode *N) {
4182   bool IsSigned = (ISD::SMULO == N->getOpcode());
4183 
4184   // (mulo x, 2) -> (addo x, x)
4185   if (ConstantSDNode *C2 = isConstOrConstSplat(N->getOperand(1)))
4186     if (C2->getAPIntValue() == 2)
4187       return DAG.getNode(IsSigned ? ISD::SADDO : ISD::UADDO, SDLoc(N),
4188                          N->getVTList(), N->getOperand(0), N->getOperand(0));
4189 
4190   return SDValue();
4191 }
4192 
4193 SDValue DAGCombiner::visitIMINMAX(SDNode *N) {
4194   SDValue N0 = N->getOperand(0);
4195   SDValue N1 = N->getOperand(1);
4196   EVT VT = N0.getValueType();
4197 
4198   // fold vector ops
4199   if (VT.isVector())
4200     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4201       return FoldedVOp;
4202 
4203   // fold operation with constant operands.
4204   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4205   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
4206   if (N0C && N1C)
4207     return DAG.FoldConstantArithmetic(N->getOpcode(), SDLoc(N), VT, N0C, N1C);
4208 
4209   // canonicalize constant to RHS
4210   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
4211      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
4212     return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N1, N0);
4213 
4214   // Is sign bits are zero, flip between UMIN/UMAX and SMIN/SMAX.
4215   // Only do this if the current op isn't legal and the flipped is.
4216   unsigned Opcode = N->getOpcode();
4217   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
4218   if (!TLI.isOperationLegal(Opcode, VT) &&
4219       (N0.isUndef() || DAG.SignBitIsZero(N0)) &&
4220       (N1.isUndef() || DAG.SignBitIsZero(N1))) {
4221     unsigned AltOpcode;
4222     switch (Opcode) {
4223     case ISD::SMIN: AltOpcode = ISD::UMIN; break;
4224     case ISD::SMAX: AltOpcode = ISD::UMAX; break;
4225     case ISD::UMIN: AltOpcode = ISD::SMIN; break;
4226     case ISD::UMAX: AltOpcode = ISD::SMAX; break;
4227     default: llvm_unreachable("Unknown MINMAX opcode");
4228     }
4229     if (TLI.isOperationLegal(AltOpcode, VT))
4230       return DAG.getNode(AltOpcode, SDLoc(N), VT, N0, N1);
4231   }
4232 
4233   return SDValue();
4234 }
4235 
4236 /// If this is a bitwise logic instruction and both operands have the same
4237 /// opcode, try to sink the other opcode after the logic instruction.
4238 SDValue DAGCombiner::hoistLogicOpWithSameOpcodeHands(SDNode *N) {
4239   SDValue N0 = N->getOperand(0), N1 = N->getOperand(1);
4240   EVT VT = N0.getValueType();
4241   unsigned LogicOpcode = N->getOpcode();
4242   unsigned HandOpcode = N0.getOpcode();
4243   assert((LogicOpcode == ISD::AND || LogicOpcode == ISD::OR ||
4244           LogicOpcode == ISD::XOR) && "Expected logic opcode");
4245   assert(HandOpcode == N1.getOpcode() && "Bad input!");
4246 
4247   // Bail early if none of these transforms apply.
4248   if (N0.getNumOperands() == 0)
4249     return SDValue();
4250 
4251   // FIXME: We should check number of uses of the operands to not increase
4252   //        the instruction count for all transforms.
4253 
4254   // Handle size-changing casts.
4255   SDValue X = N0.getOperand(0);
4256   SDValue Y = N1.getOperand(0);
4257   EVT XVT = X.getValueType();
4258   SDLoc DL(N);
4259   if (HandOpcode == ISD::ANY_EXTEND || HandOpcode == ISD::ZERO_EXTEND ||
4260       HandOpcode == ISD::SIGN_EXTEND) {
4261     // If both operands have other uses, this transform would create extra
4262     // instructions without eliminating anything.
4263     if (!N0.hasOneUse() && !N1.hasOneUse())
4264       return SDValue();
4265     // We need matching integer source types.
4266     if (XVT != Y.getValueType())
4267       return SDValue();
4268     // Don't create an illegal op during or after legalization. Don't ever
4269     // create an unsupported vector op.
4270     if ((VT.isVector() || LegalOperations) &&
4271         !TLI.isOperationLegalOrCustom(LogicOpcode, XVT))
4272       return SDValue();
4273     // Avoid infinite looping with PromoteIntBinOp.
4274     // TODO: Should we apply desirable/legal constraints to all opcodes?
4275     if (HandOpcode == ISD::ANY_EXTEND && LegalTypes &&
4276         !TLI.isTypeDesirableForOp(LogicOpcode, XVT))
4277       return SDValue();
4278     // logic_op (hand_op X), (hand_op Y) --> hand_op (logic_op X, Y)
4279     SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y);
4280     return DAG.getNode(HandOpcode, DL, VT, Logic);
4281   }
4282 
4283   // logic_op (truncate x), (truncate y) --> truncate (logic_op x, y)
4284   if (HandOpcode == ISD::TRUNCATE) {
4285     // If both operands have other uses, this transform would create extra
4286     // instructions without eliminating anything.
4287     if (!N0.hasOneUse() && !N1.hasOneUse())
4288       return SDValue();
4289     // We need matching source types.
4290     if (XVT != Y.getValueType())
4291       return SDValue();
4292     // Don't create an illegal op during or after legalization.
4293     if (LegalOperations && !TLI.isOperationLegal(LogicOpcode, XVT))
4294       return SDValue();
4295     // Be extra careful sinking truncate. If it's free, there's no benefit in
4296     // widening a binop. Also, don't create a logic op on an illegal type.
4297     if (TLI.isZExtFree(VT, XVT) && TLI.isTruncateFree(XVT, VT))
4298       return SDValue();
4299     if (!TLI.isTypeLegal(XVT))
4300       return SDValue();
4301     SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y);
4302     return DAG.getNode(HandOpcode, DL, VT, Logic);
4303   }
4304 
4305   // For binops SHL/SRL/SRA/AND:
4306   //   logic_op (OP x, z), (OP y, z) --> OP (logic_op x, y), z
4307   if ((HandOpcode == ISD::SHL || HandOpcode == ISD::SRL ||
4308        HandOpcode == ISD::SRA || HandOpcode == ISD::AND) &&
4309       N0.getOperand(1) == N1.getOperand(1)) {
4310     // If either operand has other uses, this transform is not an improvement.
4311     if (!N0.hasOneUse() || !N1.hasOneUse())
4312       return SDValue();
4313     SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y);
4314     return DAG.getNode(HandOpcode, DL, VT, Logic, N0.getOperand(1));
4315   }
4316 
4317   // Unary ops: logic_op (bswap x), (bswap y) --> bswap (logic_op x, y)
4318   if (HandOpcode == ISD::BSWAP) {
4319     // If either operand has other uses, this transform is not an improvement.
4320     if (!N0.hasOneUse() || !N1.hasOneUse())
4321       return SDValue();
4322     SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y);
4323     return DAG.getNode(HandOpcode, DL, VT, Logic);
4324   }
4325 
4326   // Simplify xor/and/or (bitcast(A), bitcast(B)) -> bitcast(op (A,B))
4327   // Only perform this optimization up until type legalization, before
4328   // LegalizeVectorOprs. LegalizeVectorOprs promotes vector operations by
4329   // adding bitcasts. For example (xor v4i32) is promoted to (v2i64), and
4330   // we don't want to undo this promotion.
4331   // We also handle SCALAR_TO_VECTOR because xor/or/and operations are cheaper
4332   // on scalars.
4333   if ((HandOpcode == ISD::BITCAST || HandOpcode == ISD::SCALAR_TO_VECTOR) &&
4334        Level <= AfterLegalizeTypes) {
4335     // Input types must be integer and the same.
4336     if (XVT.isInteger() && XVT == Y.getValueType()) {
4337       SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y);
4338       return DAG.getNode(HandOpcode, DL, VT, Logic);
4339     }
4340   }
4341 
4342   // Xor/and/or are indifferent to the swizzle operation (shuffle of one value).
4343   // Simplify xor/and/or (shuff(A), shuff(B)) -> shuff(op (A,B))
4344   // If both shuffles use the same mask, and both shuffle within a single
4345   // vector, then it is worthwhile to move the swizzle after the operation.
4346   // The type-legalizer generates this pattern when loading illegal
4347   // vector types from memory. In many cases this allows additional shuffle
4348   // optimizations.
4349   // There are other cases where moving the shuffle after the xor/and/or
4350   // is profitable even if shuffles don't perform a swizzle.
4351   // If both shuffles use the same mask, and both shuffles have the same first
4352   // or second operand, then it might still be profitable to move the shuffle
4353   // after the xor/and/or operation.
4354   if (HandOpcode == ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG) {
4355     auto *SVN0 = cast<ShuffleVectorSDNode>(N0);
4356     auto *SVN1 = cast<ShuffleVectorSDNode>(N1);
4357     assert(X.getValueType() == Y.getValueType() &&
4358            "Inputs to shuffles are not the same type");
4359 
4360     // Check that both shuffles use the same mask. The masks are known to be of
4361     // the same length because the result vector type is the same.
4362     // Check also that shuffles have only one use to avoid introducing extra
4363     // instructions.
4364     if (!SVN0->hasOneUse() || !SVN1->hasOneUse() ||
4365         !SVN0->getMask().equals(SVN1->getMask()))
4366       return SDValue();
4367 
4368     // Don't try to fold this node if it requires introducing a
4369     // build vector of all zeros that might be illegal at this stage.
4370     SDValue ShOp = N0.getOperand(1);
4371     if (LogicOpcode == ISD::XOR && !ShOp.isUndef())
4372       ShOp = tryFoldToZero(DL, TLI, VT, DAG, LegalOperations);
4373 
4374     // (logic_op (shuf (A, C), shuf (B, C))) --> shuf (logic_op (A, B), C)
4375     if (N0.getOperand(1) == N1.getOperand(1) && ShOp.getNode()) {
4376       SDValue Logic = DAG.getNode(LogicOpcode, DL, VT,
4377                                   N0.getOperand(0), N1.getOperand(0));
4378       return DAG.getVectorShuffle(VT, DL, Logic, ShOp, SVN0->getMask());
4379     }
4380 
4381     // Don't try to fold this node if it requires introducing a
4382     // build vector of all zeros that might be illegal at this stage.
4383     ShOp = N0.getOperand(0);
4384     if (LogicOpcode == ISD::XOR && !ShOp.isUndef())
4385       ShOp = tryFoldToZero(DL, TLI, VT, DAG, LegalOperations);
4386 
4387     // (logic_op (shuf (C, A), shuf (C, B))) --> shuf (C, logic_op (A, B))
4388     if (N0.getOperand(0) == N1.getOperand(0) && ShOp.getNode()) {
4389       SDValue Logic = DAG.getNode(LogicOpcode, DL, VT, N0.getOperand(1),
4390                                   N1.getOperand(1));
4391       return DAG.getVectorShuffle(VT, DL, ShOp, Logic, SVN0->getMask());
4392     }
4393   }
4394 
4395   return SDValue();
4396 }
4397 
4398 /// Try to make (and/or setcc (LL, LR), setcc (RL, RR)) more efficient.
4399 SDValue DAGCombiner::foldLogicOfSetCCs(bool IsAnd, SDValue N0, SDValue N1,
4400                                        const SDLoc &DL) {
4401   SDValue LL, LR, RL, RR, N0CC, N1CC;
4402   if (!isSetCCEquivalent(N0, LL, LR, N0CC) ||
4403       !isSetCCEquivalent(N1, RL, RR, N1CC))
4404     return SDValue();
4405 
4406   assert(N0.getValueType() == N1.getValueType() &&
4407          "Unexpected operand types for bitwise logic op");
4408   assert(LL.getValueType() == LR.getValueType() &&
4409          RL.getValueType() == RR.getValueType() &&
4410          "Unexpected operand types for setcc");
4411 
4412   // If we're here post-legalization or the logic op type is not i1, the logic
4413   // op type must match a setcc result type. Also, all folds require new
4414   // operations on the left and right operands, so those types must match.
4415   EVT VT = N0.getValueType();
4416   EVT OpVT = LL.getValueType();
4417   if (LegalOperations || VT.getScalarType() != MVT::i1)
4418     if (VT != getSetCCResultType(OpVT))
4419       return SDValue();
4420   if (OpVT != RL.getValueType())
4421     return SDValue();
4422 
4423   ISD::CondCode CC0 = cast<CondCodeSDNode>(N0CC)->get();
4424   ISD::CondCode CC1 = cast<CondCodeSDNode>(N1CC)->get();
4425   bool IsInteger = OpVT.isInteger();
4426   if (LR == RR && CC0 == CC1 && IsInteger) {
4427     bool IsZero = isNullOrNullSplat(LR);
4428     bool IsNeg1 = isAllOnesOrAllOnesSplat(LR);
4429 
4430     // All bits clear?
4431     bool AndEqZero = IsAnd && CC1 == ISD::SETEQ && IsZero;
4432     // All sign bits clear?
4433     bool AndGtNeg1 = IsAnd && CC1 == ISD::SETGT && IsNeg1;
4434     // Any bits set?
4435     bool OrNeZero = !IsAnd && CC1 == ISD::SETNE && IsZero;
4436     // Any sign bits set?
4437     bool OrLtZero = !IsAnd && CC1 == ISD::SETLT && IsZero;
4438 
4439     // (and (seteq X,  0), (seteq Y,  0)) --> (seteq (or X, Y),  0)
4440     // (and (setgt X, -1), (setgt Y, -1)) --> (setgt (or X, Y), -1)
4441     // (or  (setne X,  0), (setne Y,  0)) --> (setne (or X, Y),  0)
4442     // (or  (setlt X,  0), (setlt Y,  0)) --> (setlt (or X, Y),  0)
4443     if (AndEqZero || AndGtNeg1 || OrNeZero || OrLtZero) {
4444       SDValue Or = DAG.getNode(ISD::OR, SDLoc(N0), OpVT, LL, RL);
4445       AddToWorklist(Or.getNode());
4446       return DAG.getSetCC(DL, VT, Or, LR, CC1);
4447     }
4448 
4449     // All bits set?
4450     bool AndEqNeg1 = IsAnd && CC1 == ISD::SETEQ && IsNeg1;
4451     // All sign bits set?
4452     bool AndLtZero = IsAnd && CC1 == ISD::SETLT && IsZero;
4453     // Any bits clear?
4454     bool OrNeNeg1 = !IsAnd && CC1 == ISD::SETNE && IsNeg1;
4455     // Any sign bits clear?
4456     bool OrGtNeg1 = !IsAnd && CC1 == ISD::SETGT && IsNeg1;
4457 
4458     // (and (seteq X, -1), (seteq Y, -1)) --> (seteq (and X, Y), -1)
4459     // (and (setlt X,  0), (setlt Y,  0)) --> (setlt (and X, Y),  0)
4460     // (or  (setne X, -1), (setne Y, -1)) --> (setne (and X, Y), -1)
4461     // (or  (setgt X, -1), (setgt Y  -1)) --> (setgt (and X, Y), -1)
4462     if (AndEqNeg1 || AndLtZero || OrNeNeg1 || OrGtNeg1) {
4463       SDValue And = DAG.getNode(ISD::AND, SDLoc(N0), OpVT, LL, RL);
4464       AddToWorklist(And.getNode());
4465       return DAG.getSetCC(DL, VT, And, LR, CC1);
4466     }
4467   }
4468 
4469   // TODO: What is the 'or' equivalent of this fold?
4470   // (and (setne X, 0), (setne X, -1)) --> (setuge (add X, 1), 2)
4471   if (IsAnd && LL == RL && CC0 == CC1 && OpVT.getScalarSizeInBits() > 1 &&
4472       IsInteger && CC0 == ISD::SETNE &&
4473       ((isNullConstant(LR) && isAllOnesConstant(RR)) ||
4474        (isAllOnesConstant(LR) && isNullConstant(RR)))) {
4475     SDValue One = DAG.getConstant(1, DL, OpVT);
4476     SDValue Two = DAG.getConstant(2, DL, OpVT);
4477     SDValue Add = DAG.getNode(ISD::ADD, SDLoc(N0), OpVT, LL, One);
4478     AddToWorklist(Add.getNode());
4479     return DAG.getSetCC(DL, VT, Add, Two, ISD::SETUGE);
4480   }
4481 
4482   // Try more general transforms if the predicates match and the only user of
4483   // the compares is the 'and' or 'or'.
4484   if (IsInteger && TLI.convertSetCCLogicToBitwiseLogic(OpVT) && CC0 == CC1 &&
4485       N0.hasOneUse() && N1.hasOneUse()) {
4486     // and (seteq A, B), (seteq C, D) --> seteq (or (xor A, B), (xor C, D)), 0
4487     // or  (setne A, B), (setne C, D) --> setne (or (xor A, B), (xor C, D)), 0
4488     if ((IsAnd && CC1 == ISD::SETEQ) || (!IsAnd && CC1 == ISD::SETNE)) {
4489       SDValue XorL = DAG.getNode(ISD::XOR, SDLoc(N0), OpVT, LL, LR);
4490       SDValue XorR = DAG.getNode(ISD::XOR, SDLoc(N1), OpVT, RL, RR);
4491       SDValue Or = DAG.getNode(ISD::OR, DL, OpVT, XorL, XorR);
4492       SDValue Zero = DAG.getConstant(0, DL, OpVT);
4493       return DAG.getSetCC(DL, VT, Or, Zero, CC1);
4494     }
4495 
4496     // Turn compare of constants whose difference is 1 bit into add+and+setcc.
4497     // TODO - support non-uniform vector amounts.
4498     if ((IsAnd && CC1 == ISD::SETNE) || (!IsAnd && CC1 == ISD::SETEQ)) {
4499       // Match a shared variable operand and 2 non-opaque constant operands.
4500       ConstantSDNode *C0 = isConstOrConstSplat(LR);
4501       ConstantSDNode *C1 = isConstOrConstSplat(RR);
4502       if (LL == RL && C0 && C1 && !C0->isOpaque() && !C1->isOpaque()) {
4503         // Canonicalize larger constant as C0.
4504         if (C1->getAPIntValue().ugt(C0->getAPIntValue()))
4505           std::swap(C0, C1);
4506 
4507         // The difference of the constants must be a single bit.
4508         const APInt &C0Val = C0->getAPIntValue();
4509         const APInt &C1Val = C1->getAPIntValue();
4510         if ((C0Val - C1Val).isPowerOf2()) {
4511           // and/or (setcc X, C0, ne), (setcc X, C1, ne/eq) -->
4512           // setcc ((add X, -C1), ~(C0 - C1)), 0, ne/eq
4513           SDValue OffsetC = DAG.getConstant(-C1Val, DL, OpVT);
4514           SDValue Add = DAG.getNode(ISD::ADD, DL, OpVT, LL, OffsetC);
4515           SDValue MaskC = DAG.getConstant(~(C0Val - C1Val), DL, OpVT);
4516           SDValue And = DAG.getNode(ISD::AND, DL, OpVT, Add, MaskC);
4517           SDValue Zero = DAG.getConstant(0, DL, OpVT);
4518           return DAG.getSetCC(DL, VT, And, Zero, CC0);
4519         }
4520       }
4521     }
4522   }
4523 
4524   // Canonicalize equivalent operands to LL == RL.
4525   if (LL == RR && LR == RL) {
4526     CC1 = ISD::getSetCCSwappedOperands(CC1);
4527     std::swap(RL, RR);
4528   }
4529 
4530   // (and (setcc X, Y, CC0), (setcc X, Y, CC1)) --> (setcc X, Y, NewCC)
4531   // (or  (setcc X, Y, CC0), (setcc X, Y, CC1)) --> (setcc X, Y, NewCC)
4532   if (LL == RL && LR == RR) {
4533     ISD::CondCode NewCC = IsAnd ? ISD::getSetCCAndOperation(CC0, CC1, IsInteger)
4534                                 : ISD::getSetCCOrOperation(CC0, CC1, IsInteger);
4535     if (NewCC != ISD::SETCC_INVALID &&
4536         (!LegalOperations ||
4537          (TLI.isCondCodeLegal(NewCC, LL.getSimpleValueType()) &&
4538           TLI.isOperationLegal(ISD::SETCC, OpVT))))
4539       return DAG.getSetCC(DL, VT, LL, LR, NewCC);
4540   }
4541 
4542   return SDValue();
4543 }
4544 
4545 /// This contains all DAGCombine rules which reduce two values combined by
4546 /// an And operation to a single value. This makes them reusable in the context
4547 /// of visitSELECT(). Rules involving constants are not included as
4548 /// visitSELECT() already handles those cases.
4549 SDValue DAGCombiner::visitANDLike(SDValue N0, SDValue N1, SDNode *N) {
4550   EVT VT = N1.getValueType();
4551   SDLoc DL(N);
4552 
4553   // fold (and x, undef) -> 0
4554   if (N0.isUndef() || N1.isUndef())
4555     return DAG.getConstant(0, DL, VT);
4556 
4557   if (SDValue V = foldLogicOfSetCCs(true, N0, N1, DL))
4558     return V;
4559 
4560   if (N0.getOpcode() == ISD::ADD && N1.getOpcode() == ISD::SRL &&
4561       VT.getSizeInBits() <= 64) {
4562     if (ConstantSDNode *ADDI = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
4563       if (ConstantSDNode *SRLI = dyn_cast<ConstantSDNode>(N1.getOperand(1))) {
4564         // Look for (and (add x, c1), (lshr y, c2)). If C1 wasn't a legal
4565         // immediate for an add, but it is legal if its top c2 bits are set,
4566         // transform the ADD so the immediate doesn't need to be materialized
4567         // in a register.
4568         APInt ADDC = ADDI->getAPIntValue();
4569         APInt SRLC = SRLI->getAPIntValue();
4570         if (ADDC.getMinSignedBits() <= 64 &&
4571             SRLC.ult(VT.getSizeInBits()) &&
4572             !TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
4573           APInt Mask = APInt::getHighBitsSet(VT.getSizeInBits(),
4574                                              SRLC.getZExtValue());
4575           if (DAG.MaskedValueIsZero(N0.getOperand(1), Mask)) {
4576             ADDC |= Mask;
4577             if (TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
4578               SDLoc DL0(N0);
4579               SDValue NewAdd =
4580                 DAG.getNode(ISD::ADD, DL0, VT,
4581                             N0.getOperand(0), DAG.getConstant(ADDC, DL, VT));
4582               CombineTo(N0.getNode(), NewAdd);
4583               // Return N so it doesn't get rechecked!
4584               return SDValue(N, 0);
4585             }
4586           }
4587         }
4588       }
4589     }
4590   }
4591 
4592   // Reduce bit extract of low half of an integer to the narrower type.
4593   // (and (srl i64:x, K), KMask) ->
4594   //   (i64 zero_extend (and (srl (i32 (trunc i64:x)), K)), KMask)
4595   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
4596     if (ConstantSDNode *CAnd = dyn_cast<ConstantSDNode>(N1)) {
4597       if (ConstantSDNode *CShift = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
4598         unsigned Size = VT.getSizeInBits();
4599         const APInt &AndMask = CAnd->getAPIntValue();
4600         unsigned ShiftBits = CShift->getZExtValue();
4601 
4602         // Bail out, this node will probably disappear anyway.
4603         if (ShiftBits == 0)
4604           return SDValue();
4605 
4606         unsigned MaskBits = AndMask.countTrailingOnes();
4607         EVT HalfVT = EVT::getIntegerVT(*DAG.getContext(), Size / 2);
4608 
4609         if (AndMask.isMask() &&
4610             // Required bits must not span the two halves of the integer and
4611             // must fit in the half size type.
4612             (ShiftBits + MaskBits <= Size / 2) &&
4613             TLI.isNarrowingProfitable(VT, HalfVT) &&
4614             TLI.isTypeDesirableForOp(ISD::AND, HalfVT) &&
4615             TLI.isTypeDesirableForOp(ISD::SRL, HalfVT) &&
4616             TLI.isTruncateFree(VT, HalfVT) &&
4617             TLI.isZExtFree(HalfVT, VT)) {
4618           // The isNarrowingProfitable is to avoid regressions on PPC and
4619           // AArch64 which match a few 64-bit bit insert / bit extract patterns
4620           // on downstream users of this. Those patterns could probably be
4621           // extended to handle extensions mixed in.
4622 
4623           SDValue SL(N0);
4624           assert(MaskBits <= Size);
4625 
4626           // Extracting the highest bit of the low half.
4627           EVT ShiftVT = TLI.getShiftAmountTy(HalfVT, DAG.getDataLayout());
4628           SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, HalfVT,
4629                                       N0.getOperand(0));
4630 
4631           SDValue NewMask = DAG.getConstant(AndMask.trunc(Size / 2), SL, HalfVT);
4632           SDValue ShiftK = DAG.getConstant(ShiftBits, SL, ShiftVT);
4633           SDValue Shift = DAG.getNode(ISD::SRL, SL, HalfVT, Trunc, ShiftK);
4634           SDValue And = DAG.getNode(ISD::AND, SL, HalfVT, Shift, NewMask);
4635           return DAG.getNode(ISD::ZERO_EXTEND, SL, VT, And);
4636         }
4637       }
4638     }
4639   }
4640 
4641   return SDValue();
4642 }
4643 
4644 bool DAGCombiner::isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
4645                                    EVT LoadResultTy, EVT &ExtVT) {
4646   if (!AndC->getAPIntValue().isMask())
4647     return false;
4648 
4649   unsigned ActiveBits = AndC->getAPIntValue().countTrailingOnes();
4650 
4651   ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits);
4652   EVT LoadedVT = LoadN->getMemoryVT();
4653 
4654   if (ExtVT == LoadedVT &&
4655       (!LegalOperations ||
4656        TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))) {
4657     // ZEXTLOAD will match without needing to change the size of the value being
4658     // loaded.
4659     return true;
4660   }
4661 
4662   // Do not change the width of a volatile load.
4663   if (LoadN->isVolatile())
4664     return false;
4665 
4666   // Do not generate loads of non-round integer types since these can
4667   // be expensive (and would be wrong if the type is not byte sized).
4668   if (!LoadedVT.bitsGT(ExtVT) || !ExtVT.isRound())
4669     return false;
4670 
4671   if (LegalOperations &&
4672       !TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))
4673     return false;
4674 
4675   if (!TLI.shouldReduceLoadWidth(LoadN, ISD::ZEXTLOAD, ExtVT))
4676     return false;
4677 
4678   return true;
4679 }
4680 
4681 bool DAGCombiner::isLegalNarrowLdSt(LSBaseSDNode *LDST,
4682                                     ISD::LoadExtType ExtType, EVT &MemVT,
4683                                     unsigned ShAmt) {
4684   if (!LDST)
4685     return false;
4686   // Only allow byte offsets.
4687   if (ShAmt % 8)
4688     return false;
4689 
4690   // Do not generate loads of non-round integer types since these can
4691   // be expensive (and would be wrong if the type is not byte sized).
4692   if (!MemVT.isRound())
4693     return false;
4694 
4695   // Don't change the width of a volatile load.
4696   if (LDST->isVolatile())
4697     return false;
4698 
4699   // Verify that we are actually reducing a load width here.
4700   if (LDST->getMemoryVT().getSizeInBits() < MemVT.getSizeInBits())
4701     return false;
4702 
4703   // Ensure that this isn't going to produce an unsupported unaligned access.
4704   if (ShAmt &&
4705       !TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), MemVT,
4706                               LDST->getAddressSpace(), ShAmt / 8,
4707                               LDST->getMemOperand()->getFlags()))
4708     return false;
4709 
4710   // It's not possible to generate a constant of extended or untyped type.
4711   EVT PtrType = LDST->getBasePtr().getValueType();
4712   if (PtrType == MVT::Untyped || PtrType.isExtended())
4713     return false;
4714 
4715   if (isa<LoadSDNode>(LDST)) {
4716     LoadSDNode *Load = cast<LoadSDNode>(LDST);
4717     // Don't transform one with multiple uses, this would require adding a new
4718     // load.
4719     if (!SDValue(Load, 0).hasOneUse())
4720       return false;
4721 
4722     if (LegalOperations &&
4723         !TLI.isLoadExtLegal(ExtType, Load->getValueType(0), MemVT))
4724       return false;
4725 
4726     // For the transform to be legal, the load must produce only two values
4727     // (the value loaded and the chain).  Don't transform a pre-increment
4728     // load, for example, which produces an extra value.  Otherwise the
4729     // transformation is not equivalent, and the downstream logic to replace
4730     // uses gets things wrong.
4731     if (Load->getNumValues() > 2)
4732       return false;
4733 
4734     // If the load that we're shrinking is an extload and we're not just
4735     // discarding the extension we can't simply shrink the load. Bail.
4736     // TODO: It would be possible to merge the extensions in some cases.
4737     if (Load->getExtensionType() != ISD::NON_EXTLOAD &&
4738         Load->getMemoryVT().getSizeInBits() < MemVT.getSizeInBits() + ShAmt)
4739       return false;
4740 
4741     if (!TLI.shouldReduceLoadWidth(Load, ExtType, MemVT))
4742       return false;
4743   } else {
4744     assert(isa<StoreSDNode>(LDST) && "It is not a Load nor a Store SDNode");
4745     StoreSDNode *Store = cast<StoreSDNode>(LDST);
4746     // Can't write outside the original store
4747     if (Store->getMemoryVT().getSizeInBits() < MemVT.getSizeInBits() + ShAmt)
4748       return false;
4749 
4750     if (LegalOperations &&
4751         !TLI.isTruncStoreLegal(Store->getValue().getValueType(), MemVT))
4752       return false;
4753   }
4754   return true;
4755 }
4756 
4757 bool DAGCombiner::SearchForAndLoads(SDNode *N,
4758                                     SmallVectorImpl<LoadSDNode*> &Loads,
4759                                     SmallPtrSetImpl<SDNode*> &NodesWithConsts,
4760                                     ConstantSDNode *Mask,
4761                                     SDNode *&NodeToMask) {
4762   // Recursively search for the operands, looking for loads which can be
4763   // narrowed.
4764   for (SDValue Op : N->op_values()) {
4765     if (Op.getValueType().isVector())
4766       return false;
4767 
4768     // Some constants may need fixing up later if they are too large.
4769     if (auto *C = dyn_cast<ConstantSDNode>(Op)) {
4770       if ((N->getOpcode() == ISD::OR || N->getOpcode() == ISD::XOR) &&
4771           (Mask->getAPIntValue() & C->getAPIntValue()) != C->getAPIntValue())
4772         NodesWithConsts.insert(N);
4773       continue;
4774     }
4775 
4776     if (!Op.hasOneUse())
4777       return false;
4778 
4779     switch(Op.getOpcode()) {
4780     case ISD::LOAD: {
4781       auto *Load = cast<LoadSDNode>(Op);
4782       EVT ExtVT;
4783       if (isAndLoadExtLoad(Mask, Load, Load->getValueType(0), ExtVT) &&
4784           isLegalNarrowLdSt(Load, ISD::ZEXTLOAD, ExtVT)) {
4785 
4786         // ZEXTLOAD is already small enough.
4787         if (Load->getExtensionType() == ISD::ZEXTLOAD &&
4788             ExtVT.bitsGE(Load->getMemoryVT()))
4789           continue;
4790 
4791         // Use LE to convert equal sized loads to zext.
4792         if (ExtVT.bitsLE(Load->getMemoryVT()))
4793           Loads.push_back(Load);
4794 
4795         continue;
4796       }
4797       return false;
4798     }
4799     case ISD::ZERO_EXTEND:
4800     case ISD::AssertZext: {
4801       unsigned ActiveBits = Mask->getAPIntValue().countTrailingOnes();
4802       EVT ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits);
4803       EVT VT = Op.getOpcode() == ISD::AssertZext ?
4804         cast<VTSDNode>(Op.getOperand(1))->getVT() :
4805         Op.getOperand(0).getValueType();
4806 
4807       // We can accept extending nodes if the mask is wider or an equal
4808       // width to the original type.
4809       if (ExtVT.bitsGE(VT))
4810         continue;
4811       break;
4812     }
4813     case ISD::OR:
4814     case ISD::XOR:
4815     case ISD::AND:
4816       if (!SearchForAndLoads(Op.getNode(), Loads, NodesWithConsts, Mask,
4817                              NodeToMask))
4818         return false;
4819       continue;
4820     }
4821 
4822     // Allow one node which will masked along with any loads found.
4823     if (NodeToMask)
4824       return false;
4825 
4826     // Also ensure that the node to be masked only produces one data result.
4827     NodeToMask = Op.getNode();
4828     if (NodeToMask->getNumValues() > 1) {
4829       bool HasValue = false;
4830       for (unsigned i = 0, e = NodeToMask->getNumValues(); i < e; ++i) {
4831         MVT VT = SDValue(NodeToMask, i).getSimpleValueType();
4832         if (VT != MVT::Glue && VT != MVT::Other) {
4833           if (HasValue) {
4834             NodeToMask = nullptr;
4835             return false;
4836           }
4837           HasValue = true;
4838         }
4839       }
4840       assert(HasValue && "Node to be masked has no data result?");
4841     }
4842   }
4843   return true;
4844 }
4845 
4846 bool DAGCombiner::BackwardsPropagateMask(SDNode *N, SelectionDAG &DAG) {
4847   auto *Mask = dyn_cast<ConstantSDNode>(N->getOperand(1));
4848   if (!Mask)
4849     return false;
4850 
4851   if (!Mask->getAPIntValue().isMask())
4852     return false;
4853 
4854   // No need to do anything if the and directly uses a load.
4855   if (isa<LoadSDNode>(N->getOperand(0)))
4856     return false;
4857 
4858   SmallVector<LoadSDNode*, 8> Loads;
4859   SmallPtrSet<SDNode*, 2> NodesWithConsts;
4860   SDNode *FixupNode = nullptr;
4861   if (SearchForAndLoads(N, Loads, NodesWithConsts, Mask, FixupNode)) {
4862     if (Loads.size() == 0)
4863       return false;
4864 
4865     LLVM_DEBUG(dbgs() << "Backwards propagate AND: "; N->dump());
4866     SDValue MaskOp = N->getOperand(1);
4867 
4868     // If it exists, fixup the single node we allow in the tree that needs
4869     // masking.
4870     if (FixupNode) {
4871       LLVM_DEBUG(dbgs() << "First, need to fix up: "; FixupNode->dump());
4872       SDValue And = DAG.getNode(ISD::AND, SDLoc(FixupNode),
4873                                 FixupNode->getValueType(0),
4874                                 SDValue(FixupNode, 0), MaskOp);
4875       DAG.ReplaceAllUsesOfValueWith(SDValue(FixupNode, 0), And);
4876       if (And.getOpcode() == ISD ::AND)
4877         DAG.UpdateNodeOperands(And.getNode(), SDValue(FixupNode, 0), MaskOp);
4878     }
4879 
4880     // Narrow any constants that need it.
4881     for (auto *LogicN : NodesWithConsts) {
4882       SDValue Op0 = LogicN->getOperand(0);
4883       SDValue Op1 = LogicN->getOperand(1);
4884 
4885       if (isa<ConstantSDNode>(Op0))
4886           std::swap(Op0, Op1);
4887 
4888       SDValue And = DAG.getNode(ISD::AND, SDLoc(Op1), Op1.getValueType(),
4889                                 Op1, MaskOp);
4890 
4891       DAG.UpdateNodeOperands(LogicN, Op0, And);
4892     }
4893 
4894     // Create narrow loads.
4895     for (auto *Load : Loads) {
4896       LLVM_DEBUG(dbgs() << "Propagate AND back to: "; Load->dump());
4897       SDValue And = DAG.getNode(ISD::AND, SDLoc(Load), Load->getValueType(0),
4898                                 SDValue(Load, 0), MaskOp);
4899       DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 0), And);
4900       if (And.getOpcode() == ISD ::AND)
4901         And = SDValue(
4902             DAG.UpdateNodeOperands(And.getNode(), SDValue(Load, 0), MaskOp), 0);
4903       SDValue NewLoad = ReduceLoadWidth(And.getNode());
4904       assert(NewLoad &&
4905              "Shouldn't be masking the load if it can't be narrowed");
4906       CombineTo(Load, NewLoad, NewLoad.getValue(1));
4907     }
4908     DAG.ReplaceAllUsesWith(N, N->getOperand(0).getNode());
4909     return true;
4910   }
4911   return false;
4912 }
4913 
4914 // Unfold
4915 //    x &  (-1 'logical shift' y)
4916 // To
4917 //    (x 'opposite logical shift' y) 'logical shift' y
4918 // if it is better for performance.
4919 SDValue DAGCombiner::unfoldExtremeBitClearingToShifts(SDNode *N) {
4920   assert(N->getOpcode() == ISD::AND);
4921 
4922   SDValue N0 = N->getOperand(0);
4923   SDValue N1 = N->getOperand(1);
4924 
4925   // Do we actually prefer shifts over mask?
4926   if (!TLI.shouldFoldMaskToVariableShiftPair(N0))
4927     return SDValue();
4928 
4929   // Try to match  (-1 '[outer] logical shift' y)
4930   unsigned OuterShift;
4931   unsigned InnerShift; // The opposite direction to the OuterShift.
4932   SDValue Y;           // Shift amount.
4933   auto matchMask = [&OuterShift, &InnerShift, &Y](SDValue M) -> bool {
4934     if (!M.hasOneUse())
4935       return false;
4936     OuterShift = M->getOpcode();
4937     if (OuterShift == ISD::SHL)
4938       InnerShift = ISD::SRL;
4939     else if (OuterShift == ISD::SRL)
4940       InnerShift = ISD::SHL;
4941     else
4942       return false;
4943     if (!isAllOnesConstant(M->getOperand(0)))
4944       return false;
4945     Y = M->getOperand(1);
4946     return true;
4947   };
4948 
4949   SDValue X;
4950   if (matchMask(N1))
4951     X = N0;
4952   else if (matchMask(N0))
4953     X = N1;
4954   else
4955     return SDValue();
4956 
4957   SDLoc DL(N);
4958   EVT VT = N->getValueType(0);
4959 
4960   //     tmp = x   'opposite logical shift' y
4961   SDValue T0 = DAG.getNode(InnerShift, DL, VT, X, Y);
4962   //     ret = tmp 'logical shift' y
4963   SDValue T1 = DAG.getNode(OuterShift, DL, VT, T0, Y);
4964 
4965   return T1;
4966 }
4967 
4968 SDValue DAGCombiner::visitAND(SDNode *N) {
4969   SDValue N0 = N->getOperand(0);
4970   SDValue N1 = N->getOperand(1);
4971   EVT VT = N1.getValueType();
4972 
4973   // x & x --> x
4974   if (N0 == N1)
4975     return N0;
4976 
4977   // fold vector ops
4978   if (VT.isVector()) {
4979     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4980       return FoldedVOp;
4981 
4982     // fold (and x, 0) -> 0, vector edition
4983     if (ISD::isBuildVectorAllZeros(N0.getNode()))
4984       // do not return N0, because undef node may exist in N0
4985       return DAG.getConstant(APInt::getNullValue(N0.getScalarValueSizeInBits()),
4986                              SDLoc(N), N0.getValueType());
4987     if (ISD::isBuildVectorAllZeros(N1.getNode()))
4988       // do not return N1, because undef node may exist in N1
4989       return DAG.getConstant(APInt::getNullValue(N1.getScalarValueSizeInBits()),
4990                              SDLoc(N), N1.getValueType());
4991 
4992     // fold (and x, -1) -> x, vector edition
4993     if (ISD::isBuildVectorAllOnes(N0.getNode()))
4994       return N1;
4995     if (ISD::isBuildVectorAllOnes(N1.getNode()))
4996       return N0;
4997   }
4998 
4999   // fold (and c1, c2) -> c1&c2
5000   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
5001   ConstantSDNode *N1C = isConstOrConstSplat(N1);
5002   if (N0C && N1C && !N1C->isOpaque())
5003     return DAG.FoldConstantArithmetic(ISD::AND, SDLoc(N), VT, N0C, N1C);
5004   // canonicalize constant to RHS
5005   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
5006       !DAG.isConstantIntBuildVectorOrConstantInt(N1))
5007     return DAG.getNode(ISD::AND, SDLoc(N), VT, N1, N0);
5008   // fold (and x, -1) -> x
5009   if (isAllOnesConstant(N1))
5010     return N0;
5011   // if (and x, c) is known to be zero, return 0
5012   unsigned BitWidth = VT.getScalarSizeInBits();
5013   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
5014                                    APInt::getAllOnesValue(BitWidth)))
5015     return DAG.getConstant(0, SDLoc(N), VT);
5016 
5017   if (SDValue NewSel = foldBinOpIntoSelect(N))
5018     return NewSel;
5019 
5020   // reassociate and
5021   if (SDValue RAND = reassociateOps(ISD::AND, SDLoc(N), N0, N1, N->getFlags()))
5022     return RAND;
5023 
5024   // Try to convert a constant mask AND into a shuffle clear mask.
5025   if (VT.isVector())
5026     if (SDValue Shuffle = XformToShuffleWithZero(N))
5027       return Shuffle;
5028 
5029   // fold (and (or x, C), D) -> D if (C & D) == D
5030   auto MatchSubset = [](ConstantSDNode *LHS, ConstantSDNode *RHS) {
5031     return RHS->getAPIntValue().isSubsetOf(LHS->getAPIntValue());
5032   };
5033   if (N0.getOpcode() == ISD::OR &&
5034       ISD::matchBinaryPredicate(N0.getOperand(1), N1, MatchSubset))
5035     return N1;
5036   // fold (and (any_ext V), c) -> (zero_ext V) if 'and' only clears top bits.
5037   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
5038     SDValue N0Op0 = N0.getOperand(0);
5039     APInt Mask = ~N1C->getAPIntValue();
5040     Mask = Mask.trunc(N0Op0.getScalarValueSizeInBits());
5041     if (DAG.MaskedValueIsZero(N0Op0, Mask)) {
5042       SDValue Zext = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N),
5043                                  N0.getValueType(), N0Op0);
5044 
5045       // Replace uses of the AND with uses of the Zero extend node.
5046       CombineTo(N, Zext);
5047 
5048       // We actually want to replace all uses of the any_extend with the
5049       // zero_extend, to avoid duplicating things.  This will later cause this
5050       // AND to be folded.
5051       CombineTo(N0.getNode(), Zext);
5052       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
5053     }
5054   }
5055   // similarly fold (and (X (load ([non_ext|any_ext|zero_ext] V))), c) ->
5056   // (X (load ([non_ext|zero_ext] V))) if 'and' only clears top bits which must
5057   // already be zero by virtue of the width of the base type of the load.
5058   //
5059   // the 'X' node here can either be nothing or an extract_vector_elt to catch
5060   // more cases.
5061   if ((N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
5062        N0.getValueSizeInBits() == N0.getOperand(0).getScalarValueSizeInBits() &&
5063        N0.getOperand(0).getOpcode() == ISD::LOAD &&
5064        N0.getOperand(0).getResNo() == 0) ||
5065       (N0.getOpcode() == ISD::LOAD && N0.getResNo() == 0)) {
5066     LoadSDNode *Load = cast<LoadSDNode>( (N0.getOpcode() == ISD::LOAD) ?
5067                                          N0 : N0.getOperand(0) );
5068 
5069     // Get the constant (if applicable) the zero'th operand is being ANDed with.
5070     // This can be a pure constant or a vector splat, in which case we treat the
5071     // vector as a scalar and use the splat value.
5072     APInt Constant = APInt::getNullValue(1);
5073     if (const ConstantSDNode *C = dyn_cast<ConstantSDNode>(N1)) {
5074       Constant = C->getAPIntValue();
5075     } else if (BuildVectorSDNode *Vector = dyn_cast<BuildVectorSDNode>(N1)) {
5076       APInt SplatValue, SplatUndef;
5077       unsigned SplatBitSize;
5078       bool HasAnyUndefs;
5079       bool IsSplat = Vector->isConstantSplat(SplatValue, SplatUndef,
5080                                              SplatBitSize, HasAnyUndefs);
5081       if (IsSplat) {
5082         // Undef bits can contribute to a possible optimisation if set, so
5083         // set them.
5084         SplatValue |= SplatUndef;
5085 
5086         // The splat value may be something like "0x00FFFFFF", which means 0 for
5087         // the first vector value and FF for the rest, repeating. We need a mask
5088         // that will apply equally to all members of the vector, so AND all the
5089         // lanes of the constant together.
5090         unsigned EltBitWidth = Vector->getValueType(0).getScalarSizeInBits();
5091 
5092         // If the splat value has been compressed to a bitlength lower
5093         // than the size of the vector lane, we need to re-expand it to
5094         // the lane size.
5095         if (EltBitWidth > SplatBitSize)
5096           for (SplatValue = SplatValue.zextOrTrunc(EltBitWidth);
5097                SplatBitSize < EltBitWidth; SplatBitSize = SplatBitSize * 2)
5098             SplatValue |= SplatValue.shl(SplatBitSize);
5099 
5100         // Make sure that variable 'Constant' is only set if 'SplatBitSize' is a
5101         // multiple of 'BitWidth'. Otherwise, we could propagate a wrong value.
5102         if ((SplatBitSize % EltBitWidth) == 0) {
5103           Constant = APInt::getAllOnesValue(EltBitWidth);
5104           for (unsigned i = 0, n = (SplatBitSize / EltBitWidth); i < n; ++i)
5105             Constant &= SplatValue.extractBits(EltBitWidth, i * EltBitWidth);
5106         }
5107       }
5108     }
5109 
5110     // If we want to change an EXTLOAD to a ZEXTLOAD, ensure a ZEXTLOAD is
5111     // actually legal and isn't going to get expanded, else this is a false
5112     // optimisation.
5113     bool CanZextLoadProfitably = TLI.isLoadExtLegal(ISD::ZEXTLOAD,
5114                                                     Load->getValueType(0),
5115                                                     Load->getMemoryVT());
5116 
5117     // Resize the constant to the same size as the original memory access before
5118     // extension. If it is still the AllOnesValue then this AND is completely
5119     // unneeded.
5120     Constant = Constant.zextOrTrunc(Load->getMemoryVT().getScalarSizeInBits());
5121 
5122     bool B;
5123     switch (Load->getExtensionType()) {
5124     default: B = false; break;
5125     case ISD::EXTLOAD: B = CanZextLoadProfitably; break;
5126     case ISD::ZEXTLOAD:
5127     case ISD::NON_EXTLOAD: B = true; break;
5128     }
5129 
5130     if (B && Constant.isAllOnesValue()) {
5131       // If the load type was an EXTLOAD, convert to ZEXTLOAD in order to
5132       // preserve semantics once we get rid of the AND.
5133       SDValue NewLoad(Load, 0);
5134 
5135       // Fold the AND away. NewLoad may get replaced immediately.
5136       CombineTo(N, (N0.getNode() == Load) ? NewLoad : N0);
5137 
5138       if (Load->getExtensionType() == ISD::EXTLOAD) {
5139         NewLoad = DAG.getLoad(Load->getAddressingMode(), ISD::ZEXTLOAD,
5140                               Load->getValueType(0), SDLoc(Load),
5141                               Load->getChain(), Load->getBasePtr(),
5142                               Load->getOffset(), Load->getMemoryVT(),
5143                               Load->getMemOperand());
5144         // Replace uses of the EXTLOAD with the new ZEXTLOAD.
5145         if (Load->getNumValues() == 3) {
5146           // PRE/POST_INC loads have 3 values.
5147           SDValue To[] = { NewLoad.getValue(0), NewLoad.getValue(1),
5148                            NewLoad.getValue(2) };
5149           CombineTo(Load, To, 3, true);
5150         } else {
5151           CombineTo(Load, NewLoad.getValue(0), NewLoad.getValue(1));
5152         }
5153       }
5154 
5155       return SDValue(N, 0); // Return N so it doesn't get rechecked!
5156     }
5157   }
5158 
5159   // fold (and (load x), 255) -> (zextload x, i8)
5160   // fold (and (extload x, i16), 255) -> (zextload x, i8)
5161   // fold (and (any_ext (extload x, i16)), 255) -> (zextload x, i8)
5162   if (!VT.isVector() && N1C && (N0.getOpcode() == ISD::LOAD ||
5163                                 (N0.getOpcode() == ISD::ANY_EXTEND &&
5164                                  N0.getOperand(0).getOpcode() == ISD::LOAD))) {
5165     if (SDValue Res = ReduceLoadWidth(N)) {
5166       LoadSDNode *LN0 = N0->getOpcode() == ISD::ANY_EXTEND
5167         ? cast<LoadSDNode>(N0.getOperand(0)) : cast<LoadSDNode>(N0);
5168       AddToWorklist(N);
5169       DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 0), Res);
5170       return SDValue(N, 0);
5171     }
5172   }
5173 
5174   if (Level >= AfterLegalizeTypes) {
5175     // Attempt to propagate the AND back up to the leaves which, if they're
5176     // loads, can be combined to narrow loads and the AND node can be removed.
5177     // Perform after legalization so that extend nodes will already be
5178     // combined into the loads.
5179     if (BackwardsPropagateMask(N, DAG)) {
5180       return SDValue(N, 0);
5181     }
5182   }
5183 
5184   if (SDValue Combined = visitANDLike(N0, N1, N))
5185     return Combined;
5186 
5187   // Simplify: (and (op x...), (op y...))  -> (op (and x, y))
5188   if (N0.getOpcode() == N1.getOpcode())
5189     if (SDValue V = hoistLogicOpWithSameOpcodeHands(N))
5190       return V;
5191 
5192   // Masking the negated extension of a boolean is just the zero-extended
5193   // boolean:
5194   // and (sub 0, zext(bool X)), 1 --> zext(bool X)
5195   // and (sub 0, sext(bool X)), 1 --> zext(bool X)
5196   //
5197   // Note: the SimplifyDemandedBits fold below can make an information-losing
5198   // transform, and then we have no way to find this better fold.
5199   if (N1C && N1C->isOne() && N0.getOpcode() == ISD::SUB) {
5200     if (isNullOrNullSplat(N0.getOperand(0))) {
5201       SDValue SubRHS = N0.getOperand(1);
5202       if (SubRHS.getOpcode() == ISD::ZERO_EXTEND &&
5203           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
5204         return SubRHS;
5205       if (SubRHS.getOpcode() == ISD::SIGN_EXTEND &&
5206           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
5207         return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, SubRHS.getOperand(0));
5208     }
5209   }
5210 
5211   // fold (and (sign_extend_inreg x, i16 to i32), 1) -> (and x, 1)
5212   // fold (and (sra)) -> (and (srl)) when possible.
5213   if (SimplifyDemandedBits(SDValue(N, 0)))
5214     return SDValue(N, 0);
5215 
5216   // fold (zext_inreg (extload x)) -> (zextload x)
5217   // fold (zext_inreg (sextload x)) -> (zextload x) iff load has one use
5218   if (ISD::isUNINDEXEDLoad(N0.getNode()) &&
5219       (ISD::isEXTLoad(N0.getNode()) ||
5220        (ISD::isSEXTLoad(N0.getNode()) && N0.hasOneUse()))) {
5221     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
5222     EVT MemVT = LN0->getMemoryVT();
5223     // If we zero all the possible extended bits, then we can turn this into
5224     // a zextload if we are running before legalize or the operation is legal.
5225     unsigned ExtBitSize = N1.getScalarValueSizeInBits();
5226     unsigned MemBitSize = MemVT.getScalarSizeInBits();
5227     APInt ExtBits = APInt::getHighBitsSet(ExtBitSize, ExtBitSize - MemBitSize);
5228     if (DAG.MaskedValueIsZero(N1, ExtBits) &&
5229         ((!LegalOperations && !LN0->isVolatile()) ||
5230          TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) {
5231       SDValue ExtLoad =
5232           DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT, LN0->getChain(),
5233                          LN0->getBasePtr(), MemVT, LN0->getMemOperand());
5234       AddToWorklist(N);
5235       CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
5236       return SDValue(N, 0); // Return N so it doesn't get rechecked!
5237     }
5238   }
5239 
5240   // fold (and (or (srl N, 8), (shl N, 8)), 0xffff) -> (srl (bswap N), const)
5241   if (N1C && N1C->getAPIntValue() == 0xffff && N0.getOpcode() == ISD::OR) {
5242     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
5243                                            N0.getOperand(1), false))
5244       return BSwap;
5245   }
5246 
5247   if (SDValue Shifts = unfoldExtremeBitClearingToShifts(N))
5248     return Shifts;
5249 
5250   return SDValue();
5251 }
5252 
5253 /// Match (a >> 8) | (a << 8) as (bswap a) >> 16.
5254 SDValue DAGCombiner::MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
5255                                         bool DemandHighBits) {
5256   if (!LegalOperations)
5257     return SDValue();
5258 
5259   EVT VT = N->getValueType(0);
5260   if (VT != MVT::i64 && VT != MVT::i32 && VT != MVT::i16)
5261     return SDValue();
5262   if (!TLI.isOperationLegalOrCustom(ISD::BSWAP, VT))
5263     return SDValue();
5264 
5265   // Recognize (and (shl a, 8), 0xff00), (and (srl a, 8), 0xff)
5266   bool LookPassAnd0 = false;
5267   bool LookPassAnd1 = false;
5268   if (N0.getOpcode() == ISD::AND && N0.getOperand(0).getOpcode() == ISD::SRL)
5269       std::swap(N0, N1);
5270   if (N1.getOpcode() == ISD::AND && N1.getOperand(0).getOpcode() == ISD::SHL)
5271       std::swap(N0, N1);
5272   if (N0.getOpcode() == ISD::AND) {
5273     if (!N0.getNode()->hasOneUse())
5274       return SDValue();
5275     ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
5276     // Also handle 0xffff since the LHS is guaranteed to have zeros there.
5277     // This is needed for X86.
5278     if (!N01C || (N01C->getZExtValue() != 0xFF00 &&
5279                   N01C->getZExtValue() != 0xFFFF))
5280       return SDValue();
5281     N0 = N0.getOperand(0);
5282     LookPassAnd0 = true;
5283   }
5284 
5285   if (N1.getOpcode() == ISD::AND) {
5286     if (!N1.getNode()->hasOneUse())
5287       return SDValue();
5288     ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
5289     if (!N11C || N11C->getZExtValue() != 0xFF)
5290       return SDValue();
5291     N1 = N1.getOperand(0);
5292     LookPassAnd1 = true;
5293   }
5294 
5295   if (N0.getOpcode() == ISD::SRL && N1.getOpcode() == ISD::SHL)
5296     std::swap(N0, N1);
5297   if (N0.getOpcode() != ISD::SHL || N1.getOpcode() != ISD::SRL)
5298     return SDValue();
5299   if (!N0.getNode()->hasOneUse() || !N1.getNode()->hasOneUse())
5300     return SDValue();
5301 
5302   ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
5303   ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
5304   if (!N01C || !N11C)
5305     return SDValue();
5306   if (N01C->getZExtValue() != 8 || N11C->getZExtValue() != 8)
5307     return SDValue();
5308 
5309   // Look for (shl (and a, 0xff), 8), (srl (and a, 0xff00), 8)
5310   SDValue N00 = N0->getOperand(0);
5311   if (!LookPassAnd0 && N00.getOpcode() == ISD::AND) {
5312     if (!N00.getNode()->hasOneUse())
5313       return SDValue();
5314     ConstantSDNode *N001C = dyn_cast<ConstantSDNode>(N00.getOperand(1));
5315     if (!N001C || N001C->getZExtValue() != 0xFF)
5316       return SDValue();
5317     N00 = N00.getOperand(0);
5318     LookPassAnd0 = true;
5319   }
5320 
5321   SDValue N10 = N1->getOperand(0);
5322   if (!LookPassAnd1 && N10.getOpcode() == ISD::AND) {
5323     if (!N10.getNode()->hasOneUse())
5324       return SDValue();
5325     ConstantSDNode *N101C = dyn_cast<ConstantSDNode>(N10.getOperand(1));
5326     // Also allow 0xFFFF since the bits will be shifted out. This is needed
5327     // for X86.
5328     if (!N101C || (N101C->getZExtValue() != 0xFF00 &&
5329                    N101C->getZExtValue() != 0xFFFF))
5330       return SDValue();
5331     N10 = N10.getOperand(0);
5332     LookPassAnd1 = true;
5333   }
5334 
5335   if (N00 != N10)
5336     return SDValue();
5337 
5338   // Make sure everything beyond the low halfword gets set to zero since the SRL
5339   // 16 will clear the top bits.
5340   unsigned OpSizeInBits = VT.getSizeInBits();
5341   if (DemandHighBits && OpSizeInBits > 16) {
5342     // If the left-shift isn't masked out then the only way this is a bswap is
5343     // if all bits beyond the low 8 are 0. In that case the entire pattern
5344     // reduces to a left shift anyway: leave it for other parts of the combiner.
5345     if (!LookPassAnd0)
5346       return SDValue();
5347 
5348     // However, if the right shift isn't masked out then it might be because
5349     // it's not needed. See if we can spot that too.
5350     if (!LookPassAnd1 &&
5351         !DAG.MaskedValueIsZero(
5352             N10, APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - 16)))
5353       return SDValue();
5354   }
5355 
5356   SDValue Res = DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N00);
5357   if (OpSizeInBits > 16) {
5358     SDLoc DL(N);
5359     Res = DAG.getNode(ISD::SRL, DL, VT, Res,
5360                       DAG.getConstant(OpSizeInBits - 16, DL,
5361                                       getShiftAmountTy(VT)));
5362   }
5363   return Res;
5364 }
5365 
5366 /// Return true if the specified node is an element that makes up a 32-bit
5367 /// packed halfword byteswap.
5368 /// ((x & 0x000000ff) << 8) |
5369 /// ((x & 0x0000ff00) >> 8) |
5370 /// ((x & 0x00ff0000) << 8) |
5371 /// ((x & 0xff000000) >> 8)
5372 static bool isBSwapHWordElement(SDValue N, MutableArrayRef<SDNode *> Parts) {
5373   if (!N.getNode()->hasOneUse())
5374     return false;
5375 
5376   unsigned Opc = N.getOpcode();
5377   if (Opc != ISD::AND && Opc != ISD::SHL && Opc != ISD::SRL)
5378     return false;
5379 
5380   SDValue N0 = N.getOperand(0);
5381   unsigned Opc0 = N0.getOpcode();
5382   if (Opc0 != ISD::AND && Opc0 != ISD::SHL && Opc0 != ISD::SRL)
5383     return false;
5384 
5385   ConstantSDNode *N1C = nullptr;
5386   // SHL or SRL: look upstream for AND mask operand
5387   if (Opc == ISD::AND)
5388     N1C = dyn_cast<ConstantSDNode>(N.getOperand(1));
5389   else if (Opc0 == ISD::AND)
5390     N1C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
5391   if (!N1C)
5392     return false;
5393 
5394   unsigned MaskByteOffset;
5395   switch (N1C->getZExtValue()) {
5396   default:
5397     return false;
5398   case 0xFF:       MaskByteOffset = 0; break;
5399   case 0xFF00:     MaskByteOffset = 1; break;
5400   case 0xFFFF:
5401     // In case demanded bits didn't clear the bits that will be shifted out.
5402     // This is needed for X86.
5403     if (Opc == ISD::SRL || (Opc == ISD::AND && Opc0 == ISD::SHL)) {
5404       MaskByteOffset = 1;
5405       break;
5406     }
5407     return false;
5408   case 0xFF0000:   MaskByteOffset = 2; break;
5409   case 0xFF000000: MaskByteOffset = 3; break;
5410   }
5411 
5412   // Look for (x & 0xff) << 8 as well as ((x << 8) & 0xff00).
5413   if (Opc == ISD::AND) {
5414     if (MaskByteOffset == 0 || MaskByteOffset == 2) {
5415       // (x >> 8) & 0xff
5416       // (x >> 8) & 0xff0000
5417       if (Opc0 != ISD::SRL)
5418         return false;
5419       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
5420       if (!C || C->getZExtValue() != 8)
5421         return false;
5422     } else {
5423       // (x << 8) & 0xff00
5424       // (x << 8) & 0xff000000
5425       if (Opc0 != ISD::SHL)
5426         return false;
5427       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
5428       if (!C || C->getZExtValue() != 8)
5429         return false;
5430     }
5431   } else if (Opc == ISD::SHL) {
5432     // (x & 0xff) << 8
5433     // (x & 0xff0000) << 8
5434     if (MaskByteOffset != 0 && MaskByteOffset != 2)
5435       return false;
5436     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
5437     if (!C || C->getZExtValue() != 8)
5438       return false;
5439   } else { // Opc == ISD::SRL
5440     // (x & 0xff00) >> 8
5441     // (x & 0xff000000) >> 8
5442     if (MaskByteOffset != 1 && MaskByteOffset != 3)
5443       return false;
5444     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
5445     if (!C || C->getZExtValue() != 8)
5446       return false;
5447   }
5448 
5449   if (Parts[MaskByteOffset])
5450     return false;
5451 
5452   Parts[MaskByteOffset] = N0.getOperand(0).getNode();
5453   return true;
5454 }
5455 
5456 /// Match a 32-bit packed halfword bswap. That is
5457 /// ((x & 0x000000ff) << 8) |
5458 /// ((x & 0x0000ff00) >> 8) |
5459 /// ((x & 0x00ff0000) << 8) |
5460 /// ((x & 0xff000000) >> 8)
5461 /// => (rotl (bswap x), 16)
5462 SDValue DAGCombiner::MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1) {
5463   if (!LegalOperations)
5464     return SDValue();
5465 
5466   EVT VT = N->getValueType(0);
5467   if (VT != MVT::i32)
5468     return SDValue();
5469   if (!TLI.isOperationLegalOrCustom(ISD::BSWAP, VT))
5470     return SDValue();
5471 
5472   // Look for either
5473   // (or (or (and), (and)), (or (and), (and)))
5474   // (or (or (or (and), (and)), (and)), (and))
5475   if (N0.getOpcode() != ISD::OR)
5476     return SDValue();
5477   SDValue N00 = N0.getOperand(0);
5478   SDValue N01 = N0.getOperand(1);
5479   SDNode *Parts[4] = {};
5480 
5481   if (N1.getOpcode() == ISD::OR &&
5482       N00.getNumOperands() == 2 && N01.getNumOperands() == 2) {
5483     // (or (or (and), (and)), (or (and), (and)))
5484     if (!isBSwapHWordElement(N00, Parts))
5485       return SDValue();
5486 
5487     if (!isBSwapHWordElement(N01, Parts))
5488       return SDValue();
5489     SDValue N10 = N1.getOperand(0);
5490     if (!isBSwapHWordElement(N10, Parts))
5491       return SDValue();
5492     SDValue N11 = N1.getOperand(1);
5493     if (!isBSwapHWordElement(N11, Parts))
5494       return SDValue();
5495   } else {
5496     // (or (or (or (and), (and)), (and)), (and))
5497     if (!isBSwapHWordElement(N1, Parts))
5498       return SDValue();
5499     if (!isBSwapHWordElement(N01, Parts))
5500       return SDValue();
5501     if (N00.getOpcode() != ISD::OR)
5502       return SDValue();
5503     SDValue N000 = N00.getOperand(0);
5504     if (!isBSwapHWordElement(N000, Parts))
5505       return SDValue();
5506     SDValue N001 = N00.getOperand(1);
5507     if (!isBSwapHWordElement(N001, Parts))
5508       return SDValue();
5509   }
5510 
5511   // Make sure the parts are all coming from the same node.
5512   if (Parts[0] != Parts[1] || Parts[0] != Parts[2] || Parts[0] != Parts[3])
5513     return SDValue();
5514 
5515   SDLoc DL(N);
5516   SDValue BSwap = DAG.getNode(ISD::BSWAP, DL, VT,
5517                               SDValue(Parts[0], 0));
5518 
5519   // Result of the bswap should be rotated by 16. If it's not legal, then
5520   // do  (x << 16) | (x >> 16).
5521   SDValue ShAmt = DAG.getConstant(16, DL, getShiftAmountTy(VT));
5522   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT))
5523     return DAG.getNode(ISD::ROTL, DL, VT, BSwap, ShAmt);
5524   if (TLI.isOperationLegalOrCustom(ISD::ROTR, VT))
5525     return DAG.getNode(ISD::ROTR, DL, VT, BSwap, ShAmt);
5526   return DAG.getNode(ISD::OR, DL, VT,
5527                      DAG.getNode(ISD::SHL, DL, VT, BSwap, ShAmt),
5528                      DAG.getNode(ISD::SRL, DL, VT, BSwap, ShAmt));
5529 }
5530 
5531 /// This contains all DAGCombine rules which reduce two values combined by
5532 /// an Or operation to a single value \see visitANDLike().
5533 SDValue DAGCombiner::visitORLike(SDValue N0, SDValue N1, SDNode *N) {
5534   EVT VT = N1.getValueType();
5535   SDLoc DL(N);
5536 
5537   // fold (or x, undef) -> -1
5538   if (!LegalOperations && (N0.isUndef() || N1.isUndef()))
5539     return DAG.getAllOnesConstant(DL, VT);
5540 
5541   if (SDValue V = foldLogicOfSetCCs(false, N0, N1, DL))
5542     return V;
5543 
5544   // (or (and X, C1), (and Y, C2))  -> (and (or X, Y), C3) if possible.
5545   if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND &&
5546       // Don't increase # computations.
5547       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
5548     // We can only do this xform if we know that bits from X that are set in C2
5549     // but not in C1 are already zero.  Likewise for Y.
5550     if (const ConstantSDNode *N0O1C =
5551         getAsNonOpaqueConstant(N0.getOperand(1))) {
5552       if (const ConstantSDNode *N1O1C =
5553           getAsNonOpaqueConstant(N1.getOperand(1))) {
5554         // We can only do this xform if we know that bits from X that are set in
5555         // C2 but not in C1 are already zero.  Likewise for Y.
5556         const APInt &LHSMask = N0O1C->getAPIntValue();
5557         const APInt &RHSMask = N1O1C->getAPIntValue();
5558 
5559         if (DAG.MaskedValueIsZero(N0.getOperand(0), RHSMask&~LHSMask) &&
5560             DAG.MaskedValueIsZero(N1.getOperand(0), LHSMask&~RHSMask)) {
5561           SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
5562                                   N0.getOperand(0), N1.getOperand(0));
5563           return DAG.getNode(ISD::AND, DL, VT, X,
5564                              DAG.getConstant(LHSMask | RHSMask, DL, VT));
5565         }
5566       }
5567     }
5568   }
5569 
5570   // (or (and X, M), (and X, N)) -> (and X, (or M, N))
5571   if (N0.getOpcode() == ISD::AND &&
5572       N1.getOpcode() == ISD::AND &&
5573       N0.getOperand(0) == N1.getOperand(0) &&
5574       // Don't increase # computations.
5575       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
5576     SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
5577                             N0.getOperand(1), N1.getOperand(1));
5578     return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), X);
5579   }
5580 
5581   return SDValue();
5582 }
5583 
5584 /// OR combines for which the commuted variant will be tried as well.
5585 static SDValue visitORCommutative(
5586     SelectionDAG &DAG, SDValue N0, SDValue N1, SDNode *N) {
5587   EVT VT = N0.getValueType();
5588   if (N0.getOpcode() == ISD::AND) {
5589     // fold (or (and X, (xor Y, -1)), Y) -> (or X, Y)
5590     if (isBitwiseNot(N0.getOperand(1)) && N0.getOperand(1).getOperand(0) == N1)
5591       return DAG.getNode(ISD::OR, SDLoc(N), VT, N0.getOperand(0), N1);
5592 
5593     // fold (or (and (xor Y, -1), X), Y) -> (or X, Y)
5594     if (isBitwiseNot(N0.getOperand(0)) && N0.getOperand(0).getOperand(0) == N1)
5595       return DAG.getNode(ISD::OR, SDLoc(N), VT, N0.getOperand(1), N1);
5596   }
5597 
5598   return SDValue();
5599 }
5600 
5601 SDValue DAGCombiner::visitOR(SDNode *N) {
5602   SDValue N0 = N->getOperand(0);
5603   SDValue N1 = N->getOperand(1);
5604   EVT VT = N1.getValueType();
5605 
5606   // x | x --> x
5607   if (N0 == N1)
5608     return N0;
5609 
5610   // fold vector ops
5611   if (VT.isVector()) {
5612     if (SDValue FoldedVOp = SimplifyVBinOp(N))
5613       return FoldedVOp;
5614 
5615     // fold (or x, 0) -> x, vector edition
5616     if (ISD::isBuildVectorAllZeros(N0.getNode()))
5617       return N1;
5618     if (ISD::isBuildVectorAllZeros(N1.getNode()))
5619       return N0;
5620 
5621     // fold (or x, -1) -> -1, vector edition
5622     if (ISD::isBuildVectorAllOnes(N0.getNode()))
5623       // do not return N0, because undef node may exist in N0
5624       return DAG.getAllOnesConstant(SDLoc(N), N0.getValueType());
5625     if (ISD::isBuildVectorAllOnes(N1.getNode()))
5626       // do not return N1, because undef node may exist in N1
5627       return DAG.getAllOnesConstant(SDLoc(N), N1.getValueType());
5628 
5629     // fold (or (shuf A, V_0, MA), (shuf B, V_0, MB)) -> (shuf A, B, Mask)
5630     // Do this only if the resulting shuffle is legal.
5631     if (isa<ShuffleVectorSDNode>(N0) &&
5632         isa<ShuffleVectorSDNode>(N1) &&
5633         // Avoid folding a node with illegal type.
5634         TLI.isTypeLegal(VT)) {
5635       bool ZeroN00 = ISD::isBuildVectorAllZeros(N0.getOperand(0).getNode());
5636       bool ZeroN01 = ISD::isBuildVectorAllZeros(N0.getOperand(1).getNode());
5637       bool ZeroN10 = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
5638       bool ZeroN11 = ISD::isBuildVectorAllZeros(N1.getOperand(1).getNode());
5639       // Ensure both shuffles have a zero input.
5640       if ((ZeroN00 != ZeroN01) && (ZeroN10 != ZeroN11)) {
5641         assert((!ZeroN00 || !ZeroN01) && "Both inputs zero!");
5642         assert((!ZeroN10 || !ZeroN11) && "Both inputs zero!");
5643         const ShuffleVectorSDNode *SV0 = cast<ShuffleVectorSDNode>(N0);
5644         const ShuffleVectorSDNode *SV1 = cast<ShuffleVectorSDNode>(N1);
5645         bool CanFold = true;
5646         int NumElts = VT.getVectorNumElements();
5647         SmallVector<int, 4> Mask(NumElts);
5648 
5649         for (int i = 0; i != NumElts; ++i) {
5650           int M0 = SV0->getMaskElt(i);
5651           int M1 = SV1->getMaskElt(i);
5652 
5653           // Determine if either index is pointing to a zero vector.
5654           bool M0Zero = M0 < 0 || (ZeroN00 == (M0 < NumElts));
5655           bool M1Zero = M1 < 0 || (ZeroN10 == (M1 < NumElts));
5656 
5657           // If one element is zero and the otherside is undef, keep undef.
5658           // This also handles the case that both are undef.
5659           if ((M0Zero && M1 < 0) || (M1Zero && M0 < 0)) {
5660             Mask[i] = -1;
5661             continue;
5662           }
5663 
5664           // Make sure only one of the elements is zero.
5665           if (M0Zero == M1Zero) {
5666             CanFold = false;
5667             break;
5668           }
5669 
5670           assert((M0 >= 0 || M1 >= 0) && "Undef index!");
5671 
5672           // We have a zero and non-zero element. If the non-zero came from
5673           // SV0 make the index a LHS index. If it came from SV1, make it
5674           // a RHS index. We need to mod by NumElts because we don't care
5675           // which operand it came from in the original shuffles.
5676           Mask[i] = M1Zero ? M0 % NumElts : (M1 % NumElts) + NumElts;
5677         }
5678 
5679         if (CanFold) {
5680           SDValue NewLHS = ZeroN00 ? N0.getOperand(1) : N0.getOperand(0);
5681           SDValue NewRHS = ZeroN10 ? N1.getOperand(1) : N1.getOperand(0);
5682 
5683           bool LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
5684           if (!LegalMask) {
5685             std::swap(NewLHS, NewRHS);
5686             ShuffleVectorSDNode::commuteMask(Mask);
5687             LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
5688           }
5689 
5690           if (LegalMask)
5691             return DAG.getVectorShuffle(VT, SDLoc(N), NewLHS, NewRHS, Mask);
5692         }
5693       }
5694     }
5695   }
5696 
5697   // fold (or c1, c2) -> c1|c2
5698   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
5699   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
5700   if (N0C && N1C && !N1C->isOpaque())
5701     return DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N), VT, N0C, N1C);
5702   // canonicalize constant to RHS
5703   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
5704      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
5705     return DAG.getNode(ISD::OR, SDLoc(N), VT, N1, N0);
5706   // fold (or x, 0) -> x
5707   if (isNullConstant(N1))
5708     return N0;
5709   // fold (or x, -1) -> -1
5710   if (isAllOnesConstant(N1))
5711     return N1;
5712 
5713   if (SDValue NewSel = foldBinOpIntoSelect(N))
5714     return NewSel;
5715 
5716   // fold (or x, c) -> c iff (x & ~c) == 0
5717   if (N1C && DAG.MaskedValueIsZero(N0, ~N1C->getAPIntValue()))
5718     return N1;
5719 
5720   if (SDValue Combined = visitORLike(N0, N1, N))
5721     return Combined;
5722 
5723   // Recognize halfword bswaps as (bswap + rotl 16) or (bswap + shl 16)
5724   if (SDValue BSwap = MatchBSwapHWord(N, N0, N1))
5725     return BSwap;
5726   if (SDValue BSwap = MatchBSwapHWordLow(N, N0, N1))
5727     return BSwap;
5728 
5729   // reassociate or
5730   if (SDValue ROR = reassociateOps(ISD::OR, SDLoc(N), N0, N1, N->getFlags()))
5731     return ROR;
5732 
5733   // Canonicalize (or (and X, c1), c2) -> (and (or X, c2), c1|c2)
5734   // iff (c1 & c2) != 0 or c1/c2 are undef.
5735   auto MatchIntersect = [](ConstantSDNode *C1, ConstantSDNode *C2) {
5736     return !C1 || !C2 || C1->getAPIntValue().intersects(C2->getAPIntValue());
5737   };
5738   if (N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
5739       ISD::matchBinaryPredicate(N0.getOperand(1), N1, MatchIntersect, true)) {
5740     if (SDValue COR = DAG.FoldConstantArithmetic(
5741             ISD::OR, SDLoc(N1), VT, N1.getNode(), N0.getOperand(1).getNode())) {
5742       SDValue IOR = DAG.getNode(ISD::OR, SDLoc(N0), VT, N0.getOperand(0), N1);
5743       AddToWorklist(IOR.getNode());
5744       return DAG.getNode(ISD::AND, SDLoc(N), VT, COR, IOR);
5745     }
5746   }
5747 
5748   if (SDValue Combined = visitORCommutative(DAG, N0, N1, N))
5749     return Combined;
5750   if (SDValue Combined = visitORCommutative(DAG, N1, N0, N))
5751     return Combined;
5752 
5753   // Simplify: (or (op x...), (op y...))  -> (op (or x, y))
5754   if (N0.getOpcode() == N1.getOpcode())
5755     if (SDValue V = hoistLogicOpWithSameOpcodeHands(N))
5756       return V;
5757 
5758   // See if this is some rotate idiom.
5759   if (SDNode *Rot = MatchRotate(N0, N1, SDLoc(N)))
5760     return SDValue(Rot, 0);
5761 
5762   if (SDValue Load = MatchLoadCombine(N))
5763     return Load;
5764 
5765   // Simplify the operands using demanded-bits information.
5766   if (SimplifyDemandedBits(SDValue(N, 0)))
5767     return SDValue(N, 0);
5768 
5769   // If OR can be rewritten into ADD, try combines based on ADD.
5770   if ((!LegalOperations || TLI.isOperationLegal(ISD::ADD, VT)) &&
5771       DAG.haveNoCommonBitsSet(N0, N1))
5772     if (SDValue Combined = visitADDLike(N))
5773       return Combined;
5774 
5775   return SDValue();
5776 }
5777 
5778 static SDValue stripConstantMask(SelectionDAG &DAG, SDValue Op, SDValue &Mask) {
5779   if (Op.getOpcode() == ISD::AND &&
5780       DAG.isConstantIntBuildVectorOrConstantInt(Op.getOperand(1))) {
5781     Mask = Op.getOperand(1);
5782     return Op.getOperand(0);
5783   }
5784   return Op;
5785 }
5786 
5787 /// Match "(X shl/srl V1) & V2" where V2 may not be present.
5788 static bool matchRotateHalf(SelectionDAG &DAG, SDValue Op, SDValue &Shift,
5789                             SDValue &Mask) {
5790   Op = stripConstantMask(DAG, Op, Mask);
5791   if (Op.getOpcode() == ISD::SRL || Op.getOpcode() == ISD::SHL) {
5792     Shift = Op;
5793     return true;
5794   }
5795   return false;
5796 }
5797 
5798 /// Helper function for visitOR to extract the needed side of a rotate idiom
5799 /// from a shl/srl/mul/udiv.  This is meant to handle cases where
5800 /// InstCombine merged some outside op with one of the shifts from
5801 /// the rotate pattern.
5802 /// \returns An empty \c SDValue if the needed shift couldn't be extracted.
5803 /// Otherwise, returns an expansion of \p ExtractFrom based on the following
5804 /// patterns:
5805 ///
5806 ///   (or (mul v c0) (shrl (mul v c1) c2)):
5807 ///     expands (mul v c0) -> (shl (mul v c1) c3)
5808 ///
5809 ///   (or (udiv v c0) (shl (udiv v c1) c2)):
5810 ///     expands (udiv v c0) -> (shrl (udiv v c1) c3)
5811 ///
5812 ///   (or (shl v c0) (shrl (shl v c1) c2)):
5813 ///     expands (shl v c0) -> (shl (shl v c1) c3)
5814 ///
5815 ///   (or (shrl v c0) (shl (shrl v c1) c2)):
5816 ///     expands (shrl v c0) -> (shrl (shrl v c1) c3)
5817 ///
5818 /// Such that in all cases, c3+c2==bitwidth(op v c1).
5819 static SDValue extractShiftForRotate(SelectionDAG &DAG, SDValue OppShift,
5820                                      SDValue ExtractFrom, SDValue &Mask,
5821                                      const SDLoc &DL) {
5822   assert(OppShift && ExtractFrom && "Empty SDValue");
5823   assert(
5824       (OppShift.getOpcode() == ISD::SHL || OppShift.getOpcode() == ISD::SRL) &&
5825       "Existing shift must be valid as a rotate half");
5826 
5827   ExtractFrom = stripConstantMask(DAG, ExtractFrom, Mask);
5828   // Preconditions:
5829   //    (or (op0 v c0) (shiftl/r (op0 v c1) c2))
5830   //
5831   // Find opcode of the needed shift to be extracted from (op0 v c0).
5832   unsigned Opcode = ISD::DELETED_NODE;
5833   bool IsMulOrDiv = false;
5834   // Set Opcode and IsMulOrDiv if the extract opcode matches the needed shift
5835   // opcode or its arithmetic (mul or udiv) variant.
5836   auto SelectOpcode = [&](unsigned NeededShift, unsigned MulOrDivVariant) {
5837     IsMulOrDiv = ExtractFrom.getOpcode() == MulOrDivVariant;
5838     if (!IsMulOrDiv && ExtractFrom.getOpcode() != NeededShift)
5839       return false;
5840     Opcode = NeededShift;
5841     return true;
5842   };
5843   // op0 must be either the needed shift opcode or the mul/udiv equivalent
5844   // that the needed shift can be extracted from.
5845   if ((OppShift.getOpcode() != ISD::SRL || !SelectOpcode(ISD::SHL, ISD::MUL)) &&
5846       (OppShift.getOpcode() != ISD::SHL || !SelectOpcode(ISD::SRL, ISD::UDIV)))
5847     return SDValue();
5848 
5849   // op0 must be the same opcode on both sides, have the same LHS argument,
5850   // and produce the same value type.
5851   SDValue OppShiftLHS = OppShift.getOperand(0);
5852   EVT ShiftedVT = OppShiftLHS.getValueType();
5853   if (OppShiftLHS.getOpcode() != ExtractFrom.getOpcode() ||
5854       OppShiftLHS.getOperand(0) != ExtractFrom.getOperand(0) ||
5855       ShiftedVT != ExtractFrom.getValueType())
5856     return SDValue();
5857 
5858   // Amount of the existing shift.
5859   ConstantSDNode *OppShiftCst = isConstOrConstSplat(OppShift.getOperand(1));
5860   // Constant mul/udiv/shift amount from the RHS of the shift's LHS op.
5861   ConstantSDNode *OppLHSCst = isConstOrConstSplat(OppShiftLHS.getOperand(1));
5862   // Constant mul/udiv/shift amount from the RHS of the ExtractFrom op.
5863   ConstantSDNode *ExtractFromCst =
5864       isConstOrConstSplat(ExtractFrom.getOperand(1));
5865   // TODO: We should be able to handle non-uniform constant vectors for these values
5866   // Check that we have constant values.
5867   if (!OppShiftCst || !OppShiftCst->getAPIntValue() ||
5868       !OppLHSCst || !OppLHSCst->getAPIntValue() ||
5869       !ExtractFromCst || !ExtractFromCst->getAPIntValue())
5870     return SDValue();
5871 
5872   // Compute the shift amount we need to extract to complete the rotate.
5873   const unsigned VTWidth = ShiftedVT.getScalarSizeInBits();
5874   if (OppShiftCst->getAPIntValue().ugt(VTWidth))
5875     return SDValue();
5876   APInt NeededShiftAmt = VTWidth - OppShiftCst->getAPIntValue();
5877   // Normalize the bitwidth of the two mul/udiv/shift constant operands.
5878   APInt ExtractFromAmt = ExtractFromCst->getAPIntValue();
5879   APInt OppLHSAmt = OppLHSCst->getAPIntValue();
5880   zeroExtendToMatch(ExtractFromAmt, OppLHSAmt);
5881 
5882   // Now try extract the needed shift from the ExtractFrom op and see if the
5883   // result matches up with the existing shift's LHS op.
5884   if (IsMulOrDiv) {
5885     // Op to extract from is a mul or udiv by a constant.
5886     // Check:
5887     //     c2 / (1 << (bitwidth(op0 v c0) - c1)) == c0
5888     //     c2 % (1 << (bitwidth(op0 v c0) - c1)) == 0
5889     const APInt ExtractDiv = APInt::getOneBitSet(ExtractFromAmt.getBitWidth(),
5890                                                  NeededShiftAmt.getZExtValue());
5891     APInt ResultAmt;
5892     APInt Rem;
5893     APInt::udivrem(ExtractFromAmt, ExtractDiv, ResultAmt, Rem);
5894     if (Rem != 0 || ResultAmt != OppLHSAmt)
5895       return SDValue();
5896   } else {
5897     // Op to extract from is a shift by a constant.
5898     // Check:
5899     //      c2 - (bitwidth(op0 v c0) - c1) == c0
5900     if (OppLHSAmt != ExtractFromAmt - NeededShiftAmt.zextOrTrunc(
5901                                           ExtractFromAmt.getBitWidth()))
5902       return SDValue();
5903   }
5904 
5905   // Return the expanded shift op that should allow a rotate to be formed.
5906   EVT ShiftVT = OppShift.getOperand(1).getValueType();
5907   EVT ResVT = ExtractFrom.getValueType();
5908   SDValue NewShiftNode = DAG.getConstant(NeededShiftAmt, DL, ShiftVT);
5909   return DAG.getNode(Opcode, DL, ResVT, OppShiftLHS, NewShiftNode);
5910 }
5911 
5912 // Return true if we can prove that, whenever Neg and Pos are both in the
5913 // range [0, EltSize), Neg == (Pos == 0 ? 0 : EltSize - Pos).  This means that
5914 // for two opposing shifts shift1 and shift2 and a value X with OpBits bits:
5915 //
5916 //     (or (shift1 X, Neg), (shift2 X, Pos))
5917 //
5918 // reduces to a rotate in direction shift2 by Pos or (equivalently) a rotate
5919 // in direction shift1 by Neg.  The range [0, EltSize) means that we only need
5920 // to consider shift amounts with defined behavior.
5921 static bool matchRotateSub(SDValue Pos, SDValue Neg, unsigned EltSize,
5922                            SelectionDAG &DAG) {
5923   // If EltSize is a power of 2 then:
5924   //
5925   //  (a) (Pos == 0 ? 0 : EltSize - Pos) == (EltSize - Pos) & (EltSize - 1)
5926   //  (b) Neg == Neg & (EltSize - 1) whenever Neg is in [0, EltSize).
5927   //
5928   // So if EltSize is a power of 2 and Neg is (and Neg', EltSize-1), we check
5929   // for the stronger condition:
5930   //
5931   //     Neg & (EltSize - 1) == (EltSize - Pos) & (EltSize - 1)    [A]
5932   //
5933   // for all Neg and Pos.  Since Neg & (EltSize - 1) == Neg' & (EltSize - 1)
5934   // we can just replace Neg with Neg' for the rest of the function.
5935   //
5936   // In other cases we check for the even stronger condition:
5937   //
5938   //     Neg == EltSize - Pos                                    [B]
5939   //
5940   // for all Neg and Pos.  Note that the (or ...) then invokes undefined
5941   // behavior if Pos == 0 (and consequently Neg == EltSize).
5942   //
5943   // We could actually use [A] whenever EltSize is a power of 2, but the
5944   // only extra cases that it would match are those uninteresting ones
5945   // where Neg and Pos are never in range at the same time.  E.g. for
5946   // EltSize == 32, using [A] would allow a Neg of the form (sub 64, Pos)
5947   // as well as (sub 32, Pos), but:
5948   //
5949   //     (or (shift1 X, (sub 64, Pos)), (shift2 X, Pos))
5950   //
5951   // always invokes undefined behavior for 32-bit X.
5952   //
5953   // Below, Mask == EltSize - 1 when using [A] and is all-ones otherwise.
5954   unsigned MaskLoBits = 0;
5955   if (Neg.getOpcode() == ISD::AND && isPowerOf2_64(EltSize)) {
5956     if (ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(1))) {
5957       KnownBits Known = DAG.computeKnownBits(Neg.getOperand(0));
5958       unsigned Bits = Log2_64(EltSize);
5959       if (NegC->getAPIntValue().getActiveBits() <= Bits &&
5960           ((NegC->getAPIntValue() | Known.Zero).countTrailingOnes() >= Bits)) {
5961         Neg = Neg.getOperand(0);
5962         MaskLoBits = Bits;
5963       }
5964     }
5965   }
5966 
5967   // Check whether Neg has the form (sub NegC, NegOp1) for some NegC and NegOp1.
5968   if (Neg.getOpcode() != ISD::SUB)
5969     return false;
5970   ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(0));
5971   if (!NegC)
5972     return false;
5973   SDValue NegOp1 = Neg.getOperand(1);
5974 
5975   // On the RHS of [A], if Pos is Pos' & (EltSize - 1), just replace Pos with
5976   // Pos'.  The truncation is redundant for the purpose of the equality.
5977   if (MaskLoBits && Pos.getOpcode() == ISD::AND) {
5978     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1))) {
5979       KnownBits Known = DAG.computeKnownBits(Pos.getOperand(0));
5980       if (PosC->getAPIntValue().getActiveBits() <= MaskLoBits &&
5981           ((PosC->getAPIntValue() | Known.Zero).countTrailingOnes() >=
5982            MaskLoBits))
5983         Pos = Pos.getOperand(0);
5984     }
5985   }
5986 
5987   // The condition we need is now:
5988   //
5989   //     (NegC - NegOp1) & Mask == (EltSize - Pos) & Mask
5990   //
5991   // If NegOp1 == Pos then we need:
5992   //
5993   //              EltSize & Mask == NegC & Mask
5994   //
5995   // (because "x & Mask" is a truncation and distributes through subtraction).
5996   APInt Width;
5997   if (Pos == NegOp1)
5998     Width = NegC->getAPIntValue();
5999 
6000   // Check for cases where Pos has the form (add NegOp1, PosC) for some PosC.
6001   // Then the condition we want to prove becomes:
6002   //
6003   //     (NegC - NegOp1) & Mask == (EltSize - (NegOp1 + PosC)) & Mask
6004   //
6005   // which, again because "x & Mask" is a truncation, becomes:
6006   //
6007   //                NegC & Mask == (EltSize - PosC) & Mask
6008   //             EltSize & Mask == (NegC + PosC) & Mask
6009   else if (Pos.getOpcode() == ISD::ADD && Pos.getOperand(0) == NegOp1) {
6010     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1)))
6011       Width = PosC->getAPIntValue() + NegC->getAPIntValue();
6012     else
6013       return false;
6014   } else
6015     return false;
6016 
6017   // Now we just need to check that EltSize & Mask == Width & Mask.
6018   if (MaskLoBits)
6019     // EltSize & Mask is 0 since Mask is EltSize - 1.
6020     return Width.getLoBits(MaskLoBits) == 0;
6021   return Width == EltSize;
6022 }
6023 
6024 // A subroutine of MatchRotate used once we have found an OR of two opposite
6025 // shifts of Shifted.  If Neg == <operand size> - Pos then the OR reduces
6026 // to both (PosOpcode Shifted, Pos) and (NegOpcode Shifted, Neg), with the
6027 // former being preferred if supported.  InnerPos and InnerNeg are Pos and
6028 // Neg with outer conversions stripped away.
6029 SDNode *DAGCombiner::MatchRotatePosNeg(SDValue Shifted, SDValue Pos,
6030                                        SDValue Neg, SDValue InnerPos,
6031                                        SDValue InnerNeg, unsigned PosOpcode,
6032                                        unsigned NegOpcode, const SDLoc &DL) {
6033   // fold (or (shl x, (*ext y)),
6034   //          (srl x, (*ext (sub 32, y)))) ->
6035   //   (rotl x, y) or (rotr x, (sub 32, y))
6036   //
6037   // fold (or (shl x, (*ext (sub 32, y))),
6038   //          (srl x, (*ext y))) ->
6039   //   (rotr x, y) or (rotl x, (sub 32, y))
6040   EVT VT = Shifted.getValueType();
6041   if (matchRotateSub(InnerPos, InnerNeg, VT.getScalarSizeInBits(), DAG)) {
6042     bool HasPos = TLI.isOperationLegalOrCustom(PosOpcode, VT);
6043     return DAG.getNode(HasPos ? PosOpcode : NegOpcode, DL, VT, Shifted,
6044                        HasPos ? Pos : Neg).getNode();
6045   }
6046 
6047   return nullptr;
6048 }
6049 
6050 // MatchRotate - Handle an 'or' of two operands.  If this is one of the many
6051 // idioms for rotate, and if the target supports rotation instructions, generate
6052 // a rot[lr].
6053 SDNode *DAGCombiner::MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL) {
6054   // Must be a legal type.  Expanded 'n promoted things won't work with rotates.
6055   EVT VT = LHS.getValueType();
6056   if (!TLI.isTypeLegal(VT)) return nullptr;
6057 
6058   // The target must have at least one rotate flavor.
6059   bool HasROTL = hasOperation(ISD::ROTL, VT);
6060   bool HasROTR = hasOperation(ISD::ROTR, VT);
6061   if (!HasROTL && !HasROTR) return nullptr;
6062 
6063   // Check for truncated rotate.
6064   if (LHS.getOpcode() == ISD::TRUNCATE && RHS.getOpcode() == ISD::TRUNCATE &&
6065       LHS.getOperand(0).getValueType() == RHS.getOperand(0).getValueType()) {
6066     assert(LHS.getValueType() == RHS.getValueType());
6067     if (SDNode *Rot = MatchRotate(LHS.getOperand(0), RHS.getOperand(0), DL)) {
6068       return DAG.getNode(ISD::TRUNCATE, SDLoc(LHS), LHS.getValueType(),
6069                          SDValue(Rot, 0)).getNode();
6070     }
6071   }
6072 
6073   // Match "(X shl/srl V1) & V2" where V2 may not be present.
6074   SDValue LHSShift;   // The shift.
6075   SDValue LHSMask;    // AND value if any.
6076   matchRotateHalf(DAG, LHS, LHSShift, LHSMask);
6077 
6078   SDValue RHSShift;   // The shift.
6079   SDValue RHSMask;    // AND value if any.
6080   matchRotateHalf(DAG, RHS, RHSShift, RHSMask);
6081 
6082   // If neither side matched a rotate half, bail
6083   if (!LHSShift && !RHSShift)
6084     return nullptr;
6085 
6086   // InstCombine may have combined a constant shl, srl, mul, or udiv with one
6087   // side of the rotate, so try to handle that here. In all cases we need to
6088   // pass the matched shift from the opposite side to compute the opcode and
6089   // needed shift amount to extract.  We still want to do this if both sides
6090   // matched a rotate half because one half may be a potential overshift that
6091   // can be broken down (ie if InstCombine merged two shl or srl ops into a
6092   // single one).
6093 
6094   // Have LHS side of the rotate, try to extract the needed shift from the RHS.
6095   if (LHSShift)
6096     if (SDValue NewRHSShift =
6097             extractShiftForRotate(DAG, LHSShift, RHS, RHSMask, DL))
6098       RHSShift = NewRHSShift;
6099   // Have RHS side of the rotate, try to extract the needed shift from the LHS.
6100   if (RHSShift)
6101     if (SDValue NewLHSShift =
6102             extractShiftForRotate(DAG, RHSShift, LHS, LHSMask, DL))
6103       LHSShift = NewLHSShift;
6104 
6105   // If a side is still missing, nothing else we can do.
6106   if (!RHSShift || !LHSShift)
6107     return nullptr;
6108 
6109   // At this point we've matched or extracted a shift op on each side.
6110 
6111   if (LHSShift.getOperand(0) != RHSShift.getOperand(0))
6112     return nullptr;   // Not shifting the same value.
6113 
6114   if (LHSShift.getOpcode() == RHSShift.getOpcode())
6115     return nullptr;   // Shifts must disagree.
6116 
6117   // Canonicalize shl to left side in a shl/srl pair.
6118   if (RHSShift.getOpcode() == ISD::SHL) {
6119     std::swap(LHS, RHS);
6120     std::swap(LHSShift, RHSShift);
6121     std::swap(LHSMask, RHSMask);
6122   }
6123 
6124   unsigned EltSizeInBits = VT.getScalarSizeInBits();
6125   SDValue LHSShiftArg = LHSShift.getOperand(0);
6126   SDValue LHSShiftAmt = LHSShift.getOperand(1);
6127   SDValue RHSShiftArg = RHSShift.getOperand(0);
6128   SDValue RHSShiftAmt = RHSShift.getOperand(1);
6129 
6130   // fold (or (shl x, C1), (srl x, C2)) -> (rotl x, C1)
6131   // fold (or (shl x, C1), (srl x, C2)) -> (rotr x, C2)
6132   auto MatchRotateSum = [EltSizeInBits](ConstantSDNode *LHS,
6133                                         ConstantSDNode *RHS) {
6134     return (LHS->getAPIntValue() + RHS->getAPIntValue()) == EltSizeInBits;
6135   };
6136   if (ISD::matchBinaryPredicate(LHSShiftAmt, RHSShiftAmt, MatchRotateSum)) {
6137     SDValue Rot = DAG.getNode(HasROTL ? ISD::ROTL : ISD::ROTR, DL, VT,
6138                               LHSShiftArg, HasROTL ? LHSShiftAmt : RHSShiftAmt);
6139 
6140     // If there is an AND of either shifted operand, apply it to the result.
6141     if (LHSMask.getNode() || RHSMask.getNode()) {
6142       SDValue AllOnes = DAG.getAllOnesConstant(DL, VT);
6143       SDValue Mask = AllOnes;
6144 
6145       if (LHSMask.getNode()) {
6146         SDValue RHSBits = DAG.getNode(ISD::SRL, DL, VT, AllOnes, RHSShiftAmt);
6147         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
6148                            DAG.getNode(ISD::OR, DL, VT, LHSMask, RHSBits));
6149       }
6150       if (RHSMask.getNode()) {
6151         SDValue LHSBits = DAG.getNode(ISD::SHL, DL, VT, AllOnes, LHSShiftAmt);
6152         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
6153                            DAG.getNode(ISD::OR, DL, VT, RHSMask, LHSBits));
6154       }
6155 
6156       Rot = DAG.getNode(ISD::AND, DL, VT, Rot, Mask);
6157     }
6158 
6159     return Rot.getNode();
6160   }
6161 
6162   // If there is a mask here, and we have a variable shift, we can't be sure
6163   // that we're masking out the right stuff.
6164   if (LHSMask.getNode() || RHSMask.getNode())
6165     return nullptr;
6166 
6167   // If the shift amount is sign/zext/any-extended just peel it off.
6168   SDValue LExtOp0 = LHSShiftAmt;
6169   SDValue RExtOp0 = RHSShiftAmt;
6170   if ((LHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
6171        LHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
6172        LHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
6173        LHSShiftAmt.getOpcode() == ISD::TRUNCATE) &&
6174       (RHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
6175        RHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
6176        RHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
6177        RHSShiftAmt.getOpcode() == ISD::TRUNCATE)) {
6178     LExtOp0 = LHSShiftAmt.getOperand(0);
6179     RExtOp0 = RHSShiftAmt.getOperand(0);
6180   }
6181 
6182   SDNode *TryL = MatchRotatePosNeg(LHSShiftArg, LHSShiftAmt, RHSShiftAmt,
6183                                    LExtOp0, RExtOp0, ISD::ROTL, ISD::ROTR, DL);
6184   if (TryL)
6185     return TryL;
6186 
6187   SDNode *TryR = MatchRotatePosNeg(RHSShiftArg, RHSShiftAmt, LHSShiftAmt,
6188                                    RExtOp0, LExtOp0, ISD::ROTR, ISD::ROTL, DL);
6189   if (TryR)
6190     return TryR;
6191 
6192   return nullptr;
6193 }
6194 
6195 namespace {
6196 
6197 /// Represents known origin of an individual byte in load combine pattern. The
6198 /// value of the byte is either constant zero or comes from memory.
6199 struct ByteProvider {
6200   // For constant zero providers Load is set to nullptr. For memory providers
6201   // Load represents the node which loads the byte from memory.
6202   // ByteOffset is the offset of the byte in the value produced by the load.
6203   LoadSDNode *Load = nullptr;
6204   unsigned ByteOffset = 0;
6205 
6206   ByteProvider() = default;
6207 
6208   static ByteProvider getMemory(LoadSDNode *Load, unsigned ByteOffset) {
6209     return ByteProvider(Load, ByteOffset);
6210   }
6211 
6212   static ByteProvider getConstantZero() { return ByteProvider(nullptr, 0); }
6213 
6214   bool isConstantZero() const { return !Load; }
6215   bool isMemory() const { return Load; }
6216 
6217   bool operator==(const ByteProvider &Other) const {
6218     return Other.Load == Load && Other.ByteOffset == ByteOffset;
6219   }
6220 
6221 private:
6222   ByteProvider(LoadSDNode *Load, unsigned ByteOffset)
6223       : Load(Load), ByteOffset(ByteOffset) {}
6224 };
6225 
6226 } // end anonymous namespace
6227 
6228 /// Recursively traverses the expression calculating the origin of the requested
6229 /// byte of the given value. Returns None if the provider can't be calculated.
6230 ///
6231 /// For all the values except the root of the expression verifies that the value
6232 /// has exactly one use and if it's not true return None. This way if the origin
6233 /// of the byte is returned it's guaranteed that the values which contribute to
6234 /// the byte are not used outside of this expression.
6235 ///
6236 /// Because the parts of the expression are not allowed to have more than one
6237 /// use this function iterates over trees, not DAGs. So it never visits the same
6238 /// node more than once.
6239 static const Optional<ByteProvider>
6240 calculateByteProvider(SDValue Op, unsigned Index, unsigned Depth,
6241                       bool Root = false) {
6242   // Typical i64 by i8 pattern requires recursion up to 8 calls depth
6243   if (Depth == 10)
6244     return None;
6245 
6246   if (!Root && !Op.hasOneUse())
6247     return None;
6248 
6249   assert(Op.getValueType().isScalarInteger() && "can't handle other types");
6250   unsigned BitWidth = Op.getValueSizeInBits();
6251   if (BitWidth % 8 != 0)
6252     return None;
6253   unsigned ByteWidth = BitWidth / 8;
6254   assert(Index < ByteWidth && "invalid index requested");
6255   (void) ByteWidth;
6256 
6257   switch (Op.getOpcode()) {
6258   case ISD::OR: {
6259     auto LHS = calculateByteProvider(Op->getOperand(0), Index, Depth + 1);
6260     if (!LHS)
6261       return None;
6262     auto RHS = calculateByteProvider(Op->getOperand(1), Index, Depth + 1);
6263     if (!RHS)
6264       return None;
6265 
6266     if (LHS->isConstantZero())
6267       return RHS;
6268     if (RHS->isConstantZero())
6269       return LHS;
6270     return None;
6271   }
6272   case ISD::SHL: {
6273     auto ShiftOp = dyn_cast<ConstantSDNode>(Op->getOperand(1));
6274     if (!ShiftOp)
6275       return None;
6276 
6277     uint64_t BitShift = ShiftOp->getZExtValue();
6278     if (BitShift % 8 != 0)
6279       return None;
6280     uint64_t ByteShift = BitShift / 8;
6281 
6282     return Index < ByteShift
6283                ? ByteProvider::getConstantZero()
6284                : calculateByteProvider(Op->getOperand(0), Index - ByteShift,
6285                                        Depth + 1);
6286   }
6287   case ISD::ANY_EXTEND:
6288   case ISD::SIGN_EXTEND:
6289   case ISD::ZERO_EXTEND: {
6290     SDValue NarrowOp = Op->getOperand(0);
6291     unsigned NarrowBitWidth = NarrowOp.getScalarValueSizeInBits();
6292     if (NarrowBitWidth % 8 != 0)
6293       return None;
6294     uint64_t NarrowByteWidth = NarrowBitWidth / 8;
6295 
6296     if (Index >= NarrowByteWidth)
6297       return Op.getOpcode() == ISD::ZERO_EXTEND
6298                  ? Optional<ByteProvider>(ByteProvider::getConstantZero())
6299                  : None;
6300     return calculateByteProvider(NarrowOp, Index, Depth + 1);
6301   }
6302   case ISD::BSWAP:
6303     return calculateByteProvider(Op->getOperand(0), ByteWidth - Index - 1,
6304                                  Depth + 1);
6305   case ISD::LOAD: {
6306     auto L = cast<LoadSDNode>(Op.getNode());
6307     if (L->isVolatile() || L->isIndexed())
6308       return None;
6309 
6310     unsigned NarrowBitWidth = L->getMemoryVT().getSizeInBits();
6311     if (NarrowBitWidth % 8 != 0)
6312       return None;
6313     uint64_t NarrowByteWidth = NarrowBitWidth / 8;
6314 
6315     if (Index >= NarrowByteWidth)
6316       return L->getExtensionType() == ISD::ZEXTLOAD
6317                  ? Optional<ByteProvider>(ByteProvider::getConstantZero())
6318                  : None;
6319     return ByteProvider::getMemory(L, Index);
6320   }
6321   }
6322 
6323   return None;
6324 }
6325 
6326 static unsigned LittleEndianByteAt(unsigned BW, unsigned i) {
6327   return i;
6328 }
6329 
6330 static unsigned BigEndianByteAt(unsigned BW, unsigned i) {
6331   return BW - i - 1;
6332 }
6333 
6334 // Check if the bytes offsets we are looking at match with either big or
6335 // little endian value loaded. Return true for big endian, false for little
6336 // endian, and None if match failed.
6337 static Optional<bool> isBigEndian(const SmallVector<int64_t, 4> &ByteOffsets,
6338                                   int64_t FirstOffset) {
6339   // The endian can be decided only when it is 2 bytes at least.
6340   unsigned Width = ByteOffsets.size();
6341   if (Width < 2)
6342     return None;
6343 
6344   bool BigEndian = true, LittleEndian = true;
6345   for (unsigned i = 0; i < Width; i++) {
6346     int64_t CurrentByteOffset = ByteOffsets[i] - FirstOffset;
6347     LittleEndian &= CurrentByteOffset == LittleEndianByteAt(Width, i);
6348     BigEndian &= CurrentByteOffset == BigEndianByteAt(Width, i);
6349     if (!BigEndian && !LittleEndian)
6350       return None;
6351   }
6352 
6353   assert((BigEndian != LittleEndian) && "It should be either big endian or"
6354                                         "little endian");
6355   return BigEndian;
6356 }
6357 
6358 static SDValue stripTruncAndExt(SDValue Value) {
6359   switch (Value.getOpcode()) {
6360   case ISD::TRUNCATE:
6361   case ISD::ZERO_EXTEND:
6362   case ISD::SIGN_EXTEND:
6363   case ISD::ANY_EXTEND:
6364     return stripTruncAndExt(Value.getOperand(0));
6365   }
6366   return Value;
6367 }
6368 
6369 /// Match a pattern where a wide type scalar value is stored by several narrow
6370 /// stores. Fold it into a single store or a BSWAP and a store if the targets
6371 /// supports it.
6372 ///
6373 /// Assuming little endian target:
6374 ///  i8 *p = ...
6375 ///  i32 val = ...
6376 ///  p[0] = (val >> 0) & 0xFF;
6377 ///  p[1] = (val >> 8) & 0xFF;
6378 ///  p[2] = (val >> 16) & 0xFF;
6379 ///  p[3] = (val >> 24) & 0xFF;
6380 /// =>
6381 ///  *((i32)p) = val;
6382 ///
6383 ///  i8 *p = ...
6384 ///  i32 val = ...
6385 ///  p[0] = (val >> 24) & 0xFF;
6386 ///  p[1] = (val >> 16) & 0xFF;
6387 ///  p[2] = (val >> 8) & 0xFF;
6388 ///  p[3] = (val >> 0) & 0xFF;
6389 /// =>
6390 ///  *((i32)p) = BSWAP(val);
6391 SDValue DAGCombiner::MatchStoreCombine(StoreSDNode *N) {
6392   // Collect all the stores in the chain.
6393   SDValue Chain;
6394   SmallVector<StoreSDNode *, 8> Stores;
6395   for (StoreSDNode *Store = N; Store; Store = dyn_cast<StoreSDNode>(Chain)) {
6396     if (Store->getMemoryVT() != MVT::i8 ||
6397         Store->isVolatile() || Store->isIndexed())
6398       return SDValue();
6399     Stores.push_back(Store);
6400     Chain = Store->getChain();
6401   }
6402   // Handle the simple type only.
6403   unsigned Width = Stores.size();
6404   EVT VT = EVT::getIntegerVT(
6405     *DAG.getContext(), Width * N->getMemoryVT().getSizeInBits());
6406   if (VT != MVT::i16 && VT != MVT::i32 && VT != MVT::i64)
6407     return SDValue();
6408 
6409   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
6410   if (LegalOperations && !TLI.isOperationLegal(ISD::STORE, VT))
6411     return SDValue();
6412 
6413   // Check if all the bytes of the combined value we are looking at are stored
6414   // to the same base address. Collect bytes offsets from Base address into
6415   // ByteOffsets.
6416   SDValue CombinedValue;
6417   SmallVector<int64_t, 4> ByteOffsets(Width, INT64_MAX);
6418   int64_t FirstOffset = INT64_MAX;
6419   StoreSDNode *FirstStore = nullptr;
6420   Optional<BaseIndexOffset> Base;
6421   for (auto Store : Stores) {
6422     // All the stores store different byte of the CombinedValue. A truncate is
6423     // required to get that byte value.
6424     SDValue Trunc = Store->getValue();
6425     if (Trunc.getOpcode() != ISD::TRUNCATE)
6426       return SDValue();
6427     // A shift operation is required to get the right byte offset, except the
6428     // first byte.
6429     int64_t Offset = 0;
6430     SDValue Value = Trunc.getOperand(0);
6431     if (Value.getOpcode() == ISD::SRL ||
6432         Value.getOpcode() == ISD::SRA) {
6433       ConstantSDNode *ShiftOffset =
6434         dyn_cast<ConstantSDNode>(Value.getOperand(1));
6435       // Trying to match the following pattern. The shift offset must be
6436       // a constant and a multiple of 8. It is the byte offset in "y".
6437       //
6438       // x = srl y, offset
6439       // i8 z = trunc x
6440       // store z, ...
6441       if (!ShiftOffset || (ShiftOffset->getSExtValue() % 8))
6442         return SDValue();
6443 
6444      Offset = ShiftOffset->getSExtValue()/8;
6445      Value = Value.getOperand(0);
6446     }
6447 
6448     // Stores must share the same combined value with different offsets.
6449     if (!CombinedValue)
6450       CombinedValue = Value;
6451     else if (stripTruncAndExt(CombinedValue) != stripTruncAndExt(Value))
6452       return SDValue();
6453 
6454     // The trunc and all the extend operation should be stripped to get the
6455     // real value we are stored.
6456     else if (CombinedValue.getValueType() != VT) {
6457       if (Value.getValueType() == VT ||
6458           Value.getValueSizeInBits() > CombinedValue.getValueSizeInBits())
6459         CombinedValue = Value;
6460       // Give up if the combined value type is smaller than the store size.
6461       if (CombinedValue.getValueSizeInBits() < VT.getSizeInBits())
6462         return SDValue();
6463     }
6464 
6465     // Stores must share the same base address
6466     BaseIndexOffset Ptr = BaseIndexOffset::match(Store, DAG);
6467     int64_t ByteOffsetFromBase = 0;
6468     if (!Base)
6469       Base = Ptr;
6470     else if (!Base->equalBaseIndex(Ptr, DAG, ByteOffsetFromBase))
6471       return SDValue();
6472 
6473     // Remember the first byte store
6474     if (ByteOffsetFromBase < FirstOffset) {
6475       FirstStore = Store;
6476       FirstOffset = ByteOffsetFromBase;
6477     }
6478     // Map the offset in the store and the offset in the combined value, and
6479     // early return if it has been set before.
6480     if (Offset < 0 || Offset >= Width || ByteOffsets[Offset] != INT64_MAX)
6481       return SDValue();
6482     ByteOffsets[Offset] = ByteOffsetFromBase;
6483   }
6484 
6485   assert(FirstOffset != INT64_MAX && "First byte offset must be set");
6486   assert(FirstStore && "First store must be set");
6487 
6488   // Check if the bytes of the combined value we are looking at match with
6489   // either big or little endian value store.
6490   Optional<bool> IsBigEndian = isBigEndian(ByteOffsets, FirstOffset);
6491   if (!IsBigEndian.hasValue())
6492     return SDValue();
6493 
6494   // The node we are looking at matches with the pattern, check if we can
6495   // replace it with a single bswap if needed and store.
6496 
6497   // If the store needs byte swap check if the target supports it
6498   bool NeedsBswap = DAG.getDataLayout().isBigEndian() != *IsBigEndian;
6499 
6500   // Before legalize we can introduce illegal bswaps which will be later
6501   // converted to an explicit bswap sequence. This way we end up with a single
6502   // store and byte shuffling instead of several stores and byte shuffling.
6503   if (NeedsBswap && LegalOperations && !TLI.isOperationLegal(ISD::BSWAP, VT))
6504     return SDValue();
6505 
6506   // Check that a store of the wide type is both allowed and fast on the target
6507   bool Fast = false;
6508   bool Allowed =
6509       TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT,
6510                              *FirstStore->getMemOperand(), &Fast);
6511   if (!Allowed || !Fast)
6512     return SDValue();
6513 
6514   if (VT != CombinedValue.getValueType()) {
6515     assert(CombinedValue.getValueType().getSizeInBits() > VT.getSizeInBits() &&
6516            "Get unexpected store value to combine");
6517     CombinedValue = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT,
6518                              CombinedValue);
6519   }
6520 
6521   if (NeedsBswap)
6522     CombinedValue = DAG.getNode(ISD::BSWAP, SDLoc(N), VT, CombinedValue);
6523 
6524   SDValue NewStore =
6525     DAG.getStore(Chain, SDLoc(N),  CombinedValue, FirstStore->getBasePtr(),
6526                  FirstStore->getPointerInfo(), FirstStore->getAlignment());
6527 
6528   // Rely on other DAG combine rules to remove the other individual stores.
6529   DAG.ReplaceAllUsesWith(N, NewStore.getNode());
6530   return NewStore;
6531 }
6532 
6533 /// Match a pattern where a wide type scalar value is loaded by several narrow
6534 /// loads and combined by shifts and ors. Fold it into a single load or a load
6535 /// and a BSWAP if the targets supports it.
6536 ///
6537 /// Assuming little endian target:
6538 ///  i8 *a = ...
6539 ///  i32 val = a[0] | (a[1] << 8) | (a[2] << 16) | (a[3] << 24)
6540 /// =>
6541 ///  i32 val = *((i32)a)
6542 ///
6543 ///  i8 *a = ...
6544 ///  i32 val = (a[0] << 24) | (a[1] << 16) | (a[2] << 8) | a[3]
6545 /// =>
6546 ///  i32 val = BSWAP(*((i32)a))
6547 ///
6548 /// TODO: This rule matches complex patterns with OR node roots and doesn't
6549 /// interact well with the worklist mechanism. When a part of the pattern is
6550 /// updated (e.g. one of the loads) its direct users are put into the worklist,
6551 /// but the root node of the pattern which triggers the load combine is not
6552 /// necessarily a direct user of the changed node. For example, once the address
6553 /// of t28 load is reassociated load combine won't be triggered:
6554 ///             t25: i32 = add t4, Constant:i32<2>
6555 ///           t26: i64 = sign_extend t25
6556 ///        t27: i64 = add t2, t26
6557 ///       t28: i8,ch = load<LD1[%tmp9]> t0, t27, undef:i64
6558 ///     t29: i32 = zero_extend t28
6559 ///   t32: i32 = shl t29, Constant:i8<8>
6560 /// t33: i32 = or t23, t32
6561 /// As a possible fix visitLoad can check if the load can be a part of a load
6562 /// combine pattern and add corresponding OR roots to the worklist.
6563 SDValue DAGCombiner::MatchLoadCombine(SDNode *N) {
6564   assert(N->getOpcode() == ISD::OR &&
6565          "Can only match load combining against OR nodes");
6566 
6567   // Handles simple types only
6568   EVT VT = N->getValueType(0);
6569   if (VT != MVT::i16 && VT != MVT::i32 && VT != MVT::i64)
6570     return SDValue();
6571   unsigned ByteWidth = VT.getSizeInBits() / 8;
6572 
6573   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
6574   // Before legalize we can introduce too wide illegal loads which will be later
6575   // split into legal sized loads. This enables us to combine i64 load by i8
6576   // patterns to a couple of i32 loads on 32 bit targets.
6577   if (LegalOperations && !TLI.isOperationLegal(ISD::LOAD, VT))
6578     return SDValue();
6579 
6580   bool IsBigEndianTarget = DAG.getDataLayout().isBigEndian();
6581   auto MemoryByteOffset = [&] (ByteProvider P) {
6582     assert(P.isMemory() && "Must be a memory byte provider");
6583     unsigned LoadBitWidth = P.Load->getMemoryVT().getSizeInBits();
6584     assert(LoadBitWidth % 8 == 0 &&
6585            "can only analyze providers for individual bytes not bit");
6586     unsigned LoadByteWidth = LoadBitWidth / 8;
6587     return IsBigEndianTarget
6588             ? BigEndianByteAt(LoadByteWidth, P.ByteOffset)
6589             : LittleEndianByteAt(LoadByteWidth, P.ByteOffset);
6590   };
6591 
6592   Optional<BaseIndexOffset> Base;
6593   SDValue Chain;
6594 
6595   SmallPtrSet<LoadSDNode *, 8> Loads;
6596   Optional<ByteProvider> FirstByteProvider;
6597   int64_t FirstOffset = INT64_MAX;
6598 
6599   // Check if all the bytes of the OR we are looking at are loaded from the same
6600   // base address. Collect bytes offsets from Base address in ByteOffsets.
6601   SmallVector<int64_t, 4> ByteOffsets(ByteWidth);
6602   for (unsigned i = 0; i < ByteWidth; i++) {
6603     auto P = calculateByteProvider(SDValue(N, 0), i, 0, /*Root=*/true);
6604     if (!P || !P->isMemory()) // All the bytes must be loaded from memory
6605       return SDValue();
6606 
6607     LoadSDNode *L = P->Load;
6608     assert(L->hasNUsesOfValue(1, 0) && !L->isVolatile() && !L->isIndexed() &&
6609            "Must be enforced by calculateByteProvider");
6610     assert(L->getOffset().isUndef() && "Unindexed load must have undef offset");
6611 
6612     // All loads must share the same chain
6613     SDValue LChain = L->getChain();
6614     if (!Chain)
6615       Chain = LChain;
6616     else if (Chain != LChain)
6617       return SDValue();
6618 
6619     // Loads must share the same base address
6620     BaseIndexOffset Ptr = BaseIndexOffset::match(L, DAG);
6621     int64_t ByteOffsetFromBase = 0;
6622     if (!Base)
6623       Base = Ptr;
6624     else if (!Base->equalBaseIndex(Ptr, DAG, ByteOffsetFromBase))
6625       return SDValue();
6626 
6627     // Calculate the offset of the current byte from the base address
6628     ByteOffsetFromBase += MemoryByteOffset(*P);
6629     ByteOffsets[i] = ByteOffsetFromBase;
6630 
6631     // Remember the first byte load
6632     if (ByteOffsetFromBase < FirstOffset) {
6633       FirstByteProvider = P;
6634       FirstOffset = ByteOffsetFromBase;
6635     }
6636 
6637     Loads.insert(L);
6638   }
6639   assert(!Loads.empty() && "All the bytes of the value must be loaded from "
6640          "memory, so there must be at least one load which produces the value");
6641   assert(Base && "Base address of the accessed memory location must be set");
6642   assert(FirstOffset != INT64_MAX && "First byte offset must be set");
6643 
6644   // Check if the bytes of the OR we are looking at match with either big or
6645   // little endian value load
6646   Optional<bool> IsBigEndian = isBigEndian(ByteOffsets, FirstOffset);
6647   if (!IsBigEndian.hasValue())
6648     return SDValue();
6649 
6650   assert(FirstByteProvider && "must be set");
6651 
6652   // Ensure that the first byte is loaded from zero offset of the first load.
6653   // So the combined value can be loaded from the first load address.
6654   if (MemoryByteOffset(*FirstByteProvider) != 0)
6655     return SDValue();
6656   LoadSDNode *FirstLoad = FirstByteProvider->Load;
6657 
6658   // The node we are looking at matches with the pattern, check if we can
6659   // replace it with a single load and bswap if needed.
6660 
6661   // If the load needs byte swap check if the target supports it
6662   bool NeedsBswap = IsBigEndianTarget != *IsBigEndian;
6663 
6664   // Before legalize we can introduce illegal bswaps which will be later
6665   // converted to an explicit bswap sequence. This way we end up with a single
6666   // load and byte shuffling instead of several loads and byte shuffling.
6667   if (NeedsBswap && LegalOperations && !TLI.isOperationLegal(ISD::BSWAP, VT))
6668     return SDValue();
6669 
6670   // Check that a load of the wide type is both allowed and fast on the target
6671   bool Fast = false;
6672   bool Allowed = TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(),
6673                                         VT, *FirstLoad->getMemOperand(), &Fast);
6674   if (!Allowed || !Fast)
6675     return SDValue();
6676 
6677   SDValue NewLoad =
6678       DAG.getLoad(VT, SDLoc(N), Chain, FirstLoad->getBasePtr(),
6679                   FirstLoad->getPointerInfo(), FirstLoad->getAlignment());
6680 
6681   // Transfer chain users from old loads to the new load.
6682   for (LoadSDNode *L : Loads)
6683     DAG.ReplaceAllUsesOfValueWith(SDValue(L, 1), SDValue(NewLoad.getNode(), 1));
6684 
6685   return NeedsBswap ? DAG.getNode(ISD::BSWAP, SDLoc(N), VT, NewLoad) : NewLoad;
6686 }
6687 
6688 // If the target has andn, bsl, or a similar bit-select instruction,
6689 // we want to unfold masked merge, with canonical pattern of:
6690 //   |        A  |  |B|
6691 //   ((x ^ y) & m) ^ y
6692 //    |  D  |
6693 // Into:
6694 //   (x & m) | (y & ~m)
6695 // If y is a constant, and the 'andn' does not work with immediates,
6696 // we unfold into a different pattern:
6697 //   ~(~x & m) & (m | y)
6698 // NOTE: we don't unfold the pattern if 'xor' is actually a 'not', because at
6699 //       the very least that breaks andnpd / andnps patterns, and because those
6700 //       patterns are simplified in IR and shouldn't be created in the DAG
6701 SDValue DAGCombiner::unfoldMaskedMerge(SDNode *N) {
6702   assert(N->getOpcode() == ISD::XOR);
6703 
6704   // Don't touch 'not' (i.e. where y = -1).
6705   if (isAllOnesOrAllOnesSplat(N->getOperand(1)))
6706     return SDValue();
6707 
6708   EVT VT = N->getValueType(0);
6709 
6710   // There are 3 commutable operators in the pattern,
6711   // so we have to deal with 8 possible variants of the basic pattern.
6712   SDValue X, Y, M;
6713   auto matchAndXor = [&X, &Y, &M](SDValue And, unsigned XorIdx, SDValue Other) {
6714     if (And.getOpcode() != ISD::AND || !And.hasOneUse())
6715       return false;
6716     SDValue Xor = And.getOperand(XorIdx);
6717     if (Xor.getOpcode() != ISD::XOR || !Xor.hasOneUse())
6718       return false;
6719     SDValue Xor0 = Xor.getOperand(0);
6720     SDValue Xor1 = Xor.getOperand(1);
6721     // Don't touch 'not' (i.e. where y = -1).
6722     if (isAllOnesOrAllOnesSplat(Xor1))
6723       return false;
6724     if (Other == Xor0)
6725       std::swap(Xor0, Xor1);
6726     if (Other != Xor1)
6727       return false;
6728     X = Xor0;
6729     Y = Xor1;
6730     M = And.getOperand(XorIdx ? 0 : 1);
6731     return true;
6732   };
6733 
6734   SDValue N0 = N->getOperand(0);
6735   SDValue N1 = N->getOperand(1);
6736   if (!matchAndXor(N0, 0, N1) && !matchAndXor(N0, 1, N1) &&
6737       !matchAndXor(N1, 0, N0) && !matchAndXor(N1, 1, N0))
6738     return SDValue();
6739 
6740   // Don't do anything if the mask is constant. This should not be reachable.
6741   // InstCombine should have already unfolded this pattern, and DAGCombiner
6742   // probably shouldn't produce it, too.
6743   if (isa<ConstantSDNode>(M.getNode()))
6744     return SDValue();
6745 
6746   // We can transform if the target has AndNot
6747   if (!TLI.hasAndNot(M))
6748     return SDValue();
6749 
6750   SDLoc DL(N);
6751 
6752   // If Y is a constant, check that 'andn' works with immediates.
6753   if (!TLI.hasAndNot(Y)) {
6754     assert(TLI.hasAndNot(X) && "Only mask is a variable? Unreachable.");
6755     // If not, we need to do a bit more work to make sure andn is still used.
6756     SDValue NotX = DAG.getNOT(DL, X, VT);
6757     SDValue LHS = DAG.getNode(ISD::AND, DL, VT, NotX, M);
6758     SDValue NotLHS = DAG.getNOT(DL, LHS, VT);
6759     SDValue RHS = DAG.getNode(ISD::OR, DL, VT, M, Y);
6760     return DAG.getNode(ISD::AND, DL, VT, NotLHS, RHS);
6761   }
6762 
6763   SDValue LHS = DAG.getNode(ISD::AND, DL, VT, X, M);
6764   SDValue NotM = DAG.getNOT(DL, M, VT);
6765   SDValue RHS = DAG.getNode(ISD::AND, DL, VT, Y, NotM);
6766 
6767   return DAG.getNode(ISD::OR, DL, VT, LHS, RHS);
6768 }
6769 
6770 SDValue DAGCombiner::visitXOR(SDNode *N) {
6771   SDValue N0 = N->getOperand(0);
6772   SDValue N1 = N->getOperand(1);
6773   EVT VT = N0.getValueType();
6774 
6775   // fold vector ops
6776   if (VT.isVector()) {
6777     if (SDValue FoldedVOp = SimplifyVBinOp(N))
6778       return FoldedVOp;
6779 
6780     // fold (xor x, 0) -> x, vector edition
6781     if (ISD::isBuildVectorAllZeros(N0.getNode()))
6782       return N1;
6783     if (ISD::isBuildVectorAllZeros(N1.getNode()))
6784       return N0;
6785   }
6786 
6787   // fold (xor undef, undef) -> 0. This is a common idiom (misuse).
6788   SDLoc DL(N);
6789   if (N0.isUndef() && N1.isUndef())
6790     return DAG.getConstant(0, DL, VT);
6791   // fold (xor x, undef) -> undef
6792   if (N0.isUndef())
6793     return N0;
6794   if (N1.isUndef())
6795     return N1;
6796   // fold (xor c1, c2) -> c1^c2
6797   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
6798   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
6799   if (N0C && N1C)
6800     return DAG.FoldConstantArithmetic(ISD::XOR, DL, VT, N0C, N1C);
6801   // canonicalize constant to RHS
6802   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
6803      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
6804     return DAG.getNode(ISD::XOR, DL, VT, N1, N0);
6805   // fold (xor x, 0) -> x
6806   if (isNullConstant(N1))
6807     return N0;
6808 
6809   if (SDValue NewSel = foldBinOpIntoSelect(N))
6810     return NewSel;
6811 
6812   // reassociate xor
6813   if (SDValue RXOR = reassociateOps(ISD::XOR, DL, N0, N1, N->getFlags()))
6814     return RXOR;
6815 
6816   // fold !(x cc y) -> (x !cc y)
6817   unsigned N0Opcode = N0.getOpcode();
6818   SDValue LHS, RHS, CC;
6819   if (TLI.isConstTrueVal(N1.getNode()) && isSetCCEquivalent(N0, LHS, RHS, CC)) {
6820     ISD::CondCode NotCC = ISD::getSetCCInverse(cast<CondCodeSDNode>(CC)->get(),
6821                                                LHS.getValueType().isInteger());
6822     if (!LegalOperations ||
6823         TLI.isCondCodeLegal(NotCC, LHS.getSimpleValueType())) {
6824       switch (N0Opcode) {
6825       default:
6826         llvm_unreachable("Unhandled SetCC Equivalent!");
6827       case ISD::SETCC:
6828         return DAG.getSetCC(SDLoc(N0), VT, LHS, RHS, NotCC);
6829       case ISD::SELECT_CC:
6830         return DAG.getSelectCC(SDLoc(N0), LHS, RHS, N0.getOperand(2),
6831                                N0.getOperand(3), NotCC);
6832       }
6833     }
6834   }
6835 
6836   // fold (not (zext (setcc x, y))) -> (zext (not (setcc x, y)))
6837   if (isOneConstant(N1) && N0Opcode == ISD::ZERO_EXTEND && N0.hasOneUse() &&
6838       isSetCCEquivalent(N0.getOperand(0), LHS, RHS, CC)){
6839     SDValue V = N0.getOperand(0);
6840     SDLoc DL0(N0);
6841     V = DAG.getNode(ISD::XOR, DL0, V.getValueType(), V,
6842                     DAG.getConstant(1, DL0, V.getValueType()));
6843     AddToWorklist(V.getNode());
6844     return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, V);
6845   }
6846 
6847   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are setcc
6848   if (isOneConstant(N1) && VT == MVT::i1 && N0.hasOneUse() &&
6849       (N0Opcode == ISD::OR || N0Opcode == ISD::AND)) {
6850     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
6851     if (isOneUseSetCC(RHS) || isOneUseSetCC(LHS)) {
6852       unsigned NewOpcode = N0Opcode == ISD::AND ? ISD::OR : ISD::AND;
6853       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
6854       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
6855       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
6856       return DAG.getNode(NewOpcode, DL, VT, LHS, RHS);
6857     }
6858   }
6859   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are constants
6860   if (isAllOnesConstant(N1) && N0.hasOneUse() &&
6861       (N0Opcode == ISD::OR || N0Opcode == ISD::AND)) {
6862     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
6863     if (isa<ConstantSDNode>(RHS) || isa<ConstantSDNode>(LHS)) {
6864       unsigned NewOpcode = N0Opcode == ISD::AND ? ISD::OR : ISD::AND;
6865       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
6866       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
6867       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
6868       return DAG.getNode(NewOpcode, DL, VT, LHS, RHS);
6869     }
6870   }
6871 
6872   // fold (not (neg x)) -> (add X, -1)
6873   // FIXME: This can be generalized to (not (sub Y, X)) -> (add X, ~Y) if
6874   // Y is a constant or the subtract has a single use.
6875   if (isAllOnesConstant(N1) && N0.getOpcode() == ISD::SUB &&
6876       isNullConstant(N0.getOperand(0))) {
6877     return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(1),
6878                        DAG.getAllOnesConstant(DL, VT));
6879   }
6880 
6881   // fold (xor (and x, y), y) -> (and (not x), y)
6882   if (N0Opcode == ISD::AND && N0.hasOneUse() && N0->getOperand(1) == N1) {
6883     SDValue X = N0.getOperand(0);
6884     SDValue NotX = DAG.getNOT(SDLoc(X), X, VT);
6885     AddToWorklist(NotX.getNode());
6886     return DAG.getNode(ISD::AND, DL, VT, NotX, N1);
6887   }
6888 
6889   if ((N0Opcode == ISD::SRL || N0Opcode == ISD::SHL) && N0.hasOneUse()) {
6890     ConstantSDNode *XorC = isConstOrConstSplat(N1);
6891     ConstantSDNode *ShiftC = isConstOrConstSplat(N0.getOperand(1));
6892     unsigned BitWidth = VT.getScalarSizeInBits();
6893     if (XorC && ShiftC) {
6894       // Don't crash on an oversized shift. We can not guarantee that a bogus
6895       // shift has been simplified to undef.
6896       uint64_t ShiftAmt = ShiftC->getLimitedValue();
6897       if (ShiftAmt < BitWidth) {
6898         APInt Ones = APInt::getAllOnesValue(BitWidth);
6899         Ones = N0Opcode == ISD::SHL ? Ones.shl(ShiftAmt) : Ones.lshr(ShiftAmt);
6900         if (XorC->getAPIntValue() == Ones) {
6901           // If the xor constant is a shifted -1, do a 'not' before the shift:
6902           // xor (X << ShiftC), XorC --> (not X) << ShiftC
6903           // xor (X >> ShiftC), XorC --> (not X) >> ShiftC
6904           SDValue Not = DAG.getNOT(DL, N0.getOperand(0), VT);
6905           return DAG.getNode(N0Opcode, DL, VT, Not, N0.getOperand(1));
6906         }
6907       }
6908     }
6909   }
6910 
6911   // fold Y = sra (X, size(X)-1); xor (add (X, Y), Y) -> (abs X)
6912   if (TLI.isOperationLegalOrCustom(ISD::ABS, VT)) {
6913     SDValue A = N0Opcode == ISD::ADD ? N0 : N1;
6914     SDValue S = N0Opcode == ISD::SRA ? N0 : N1;
6915     if (A.getOpcode() == ISD::ADD && S.getOpcode() == ISD::SRA) {
6916       SDValue A0 = A.getOperand(0), A1 = A.getOperand(1);
6917       SDValue S0 = S.getOperand(0);
6918       if ((A0 == S && A1 == S0) || (A1 == S && A0 == S0)) {
6919         unsigned OpSizeInBits = VT.getScalarSizeInBits();
6920         if (ConstantSDNode *C = isConstOrConstSplat(S.getOperand(1)))
6921           if (C->getAPIntValue() == (OpSizeInBits - 1))
6922             return DAG.getNode(ISD::ABS, DL, VT, S0);
6923       }
6924     }
6925   }
6926 
6927   // fold (xor x, x) -> 0
6928   if (N0 == N1)
6929     return tryFoldToZero(DL, TLI, VT, DAG, LegalOperations);
6930 
6931   // fold (xor (shl 1, x), -1) -> (rotl ~1, x)
6932   // Here is a concrete example of this equivalence:
6933   // i16   x ==  14
6934   // i16 shl ==   1 << 14  == 16384 == 0b0100000000000000
6935   // i16 xor == ~(1 << 14) == 49151 == 0b1011111111111111
6936   //
6937   // =>
6938   //
6939   // i16     ~1      == 0b1111111111111110
6940   // i16 rol(~1, 14) == 0b1011111111111111
6941   //
6942   // Some additional tips to help conceptualize this transform:
6943   // - Try to see the operation as placing a single zero in a value of all ones.
6944   // - There exists no value for x which would allow the result to contain zero.
6945   // - Values of x larger than the bitwidth are undefined and do not require a
6946   //   consistent result.
6947   // - Pushing the zero left requires shifting one bits in from the right.
6948   // A rotate left of ~1 is a nice way of achieving the desired result.
6949   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT) && N0Opcode == ISD::SHL &&
6950       isAllOnesConstant(N1) && isOneConstant(N0.getOperand(0))) {
6951     return DAG.getNode(ISD::ROTL, DL, VT, DAG.getConstant(~1, DL, VT),
6952                        N0.getOperand(1));
6953   }
6954 
6955   // Simplify: xor (op x...), (op y...)  -> (op (xor x, y))
6956   if (N0Opcode == N1.getOpcode())
6957     if (SDValue V = hoistLogicOpWithSameOpcodeHands(N))
6958       return V;
6959 
6960   // Unfold  ((x ^ y) & m) ^ y  into  (x & m) | (y & ~m)  if profitable
6961   if (SDValue MM = unfoldMaskedMerge(N))
6962     return MM;
6963 
6964   // Simplify the expression using non-local knowledge.
6965   if (SimplifyDemandedBits(SDValue(N, 0)))
6966     return SDValue(N, 0);
6967 
6968   return SDValue();
6969 }
6970 
6971 /// Handle transforms common to the three shifts, when the shift amount is a
6972 /// constant.
6973 /// We are looking for: (shift being one of shl/sra/srl)
6974 ///   shift (binop X, C0), C1
6975 /// And want to transform into:
6976 ///   binop (shift X, C1), (shift C0, C1)
6977 SDValue DAGCombiner::visitShiftByConstant(SDNode *N, ConstantSDNode *Amt) {
6978   // Do not turn a 'not' into a regular xor.
6979   if (isBitwiseNot(N->getOperand(0)))
6980     return SDValue();
6981 
6982   // The inner binop must be one-use, since we want to replace it.
6983   SDNode *LHS = N->getOperand(0).getNode();
6984   if (!LHS->hasOneUse()) return SDValue();
6985 
6986   // We want to pull some binops through shifts, so that we have (and (shift))
6987   // instead of (shift (and)), likewise for add, or, xor, etc.  This sort of
6988   // thing happens with address calculations, so it's important to canonicalize
6989   // it.
6990   switch (LHS->getOpcode()) {
6991   default:
6992     return SDValue();
6993   case ISD::OR:
6994   case ISD::XOR:
6995   case ISD::AND:
6996     break;
6997   case ISD::ADD:
6998     if (N->getOpcode() != ISD::SHL)
6999       return SDValue(); // only shl(add) not sr[al](add).
7000     break;
7001   }
7002 
7003   // We require the RHS of the binop to be a constant and not opaque as well.
7004   ConstantSDNode *BinOpCst = getAsNonOpaqueConstant(LHS->getOperand(1));
7005   if (!BinOpCst)
7006     return SDValue();
7007 
7008   // FIXME: disable this unless the input to the binop is a shift by a constant
7009   // or is copy/select. Enable this in other cases when figure out it's exactly
7010   // profitable.
7011   SDValue BinOpLHSVal = LHS->getOperand(0);
7012   bool IsShiftByConstant = (BinOpLHSVal.getOpcode() == ISD::SHL ||
7013                             BinOpLHSVal.getOpcode() == ISD::SRA ||
7014                             BinOpLHSVal.getOpcode() == ISD::SRL) &&
7015                            isa<ConstantSDNode>(BinOpLHSVal.getOperand(1));
7016   bool IsCopyOrSelect = BinOpLHSVal.getOpcode() == ISD::CopyFromReg ||
7017                         BinOpLHSVal.getOpcode() == ISD::SELECT;
7018 
7019   if (!IsShiftByConstant && !IsCopyOrSelect)
7020     return SDValue();
7021 
7022   if (IsCopyOrSelect && N->hasOneUse())
7023     return SDValue();
7024 
7025   EVT VT = N->getValueType(0);
7026 
7027   if (!TLI.isDesirableToCommuteWithShift(N, Level))
7028     return SDValue();
7029 
7030   // Fold the constants, shifting the binop RHS by the shift amount.
7031   SDValue NewRHS = DAG.getNode(N->getOpcode(), SDLoc(LHS->getOperand(1)),
7032                                N->getValueType(0),
7033                                LHS->getOperand(1), N->getOperand(1));
7034   assert(isa<ConstantSDNode>(NewRHS) && "Folding was not successful!");
7035 
7036   // Create the new shift.
7037   SDValue NewShift = DAG.getNode(N->getOpcode(),
7038                                  SDLoc(LHS->getOperand(0)),
7039                                  VT, LHS->getOperand(0), N->getOperand(1));
7040 
7041   // Create the new binop.
7042   return DAG.getNode(LHS->getOpcode(), SDLoc(N), VT, NewShift, NewRHS);
7043 }
7044 
7045 SDValue DAGCombiner::distributeTruncateThroughAnd(SDNode *N) {
7046   assert(N->getOpcode() == ISD::TRUNCATE);
7047   assert(N->getOperand(0).getOpcode() == ISD::AND);
7048 
7049   // (truncate:TruncVT (and N00, N01C)) -> (and (truncate:TruncVT N00), TruncC)
7050   EVT TruncVT = N->getValueType(0);
7051   if (N->hasOneUse() && N->getOperand(0).hasOneUse() &&
7052       TLI.isTypeDesirableForOp(ISD::AND, TruncVT)) {
7053     SDValue N01 = N->getOperand(0).getOperand(1);
7054     if (isConstantOrConstantVector(N01, /* NoOpaques */ true)) {
7055       SDLoc DL(N);
7056       SDValue N00 = N->getOperand(0).getOperand(0);
7057       SDValue Trunc00 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N00);
7058       SDValue Trunc01 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N01);
7059       AddToWorklist(Trunc00.getNode());
7060       AddToWorklist(Trunc01.getNode());
7061       return DAG.getNode(ISD::AND, DL, TruncVT, Trunc00, Trunc01);
7062     }
7063   }
7064 
7065   return SDValue();
7066 }
7067 
7068 SDValue DAGCombiner::visitRotate(SDNode *N) {
7069   SDLoc dl(N);
7070   SDValue N0 = N->getOperand(0);
7071   SDValue N1 = N->getOperand(1);
7072   EVT VT = N->getValueType(0);
7073   unsigned Bitsize = VT.getScalarSizeInBits();
7074 
7075   // fold (rot x, 0) -> x
7076   if (isNullOrNullSplat(N1))
7077     return N0;
7078 
7079   // fold (rot x, c) -> x iff (c % BitSize) == 0
7080   if (isPowerOf2_32(Bitsize) && Bitsize > 1) {
7081     APInt ModuloMask(N1.getScalarValueSizeInBits(), Bitsize - 1);
7082     if (DAG.MaskedValueIsZero(N1, ModuloMask))
7083       return N0;
7084   }
7085 
7086   // fold (rot x, c) -> (rot x, c % BitSize)
7087   // TODO - support non-uniform vector amounts.
7088   if (ConstantSDNode *Cst = isConstOrConstSplat(N1)) {
7089     if (Cst->getAPIntValue().uge(Bitsize)) {
7090       uint64_t RotAmt = Cst->getAPIntValue().urem(Bitsize);
7091       return DAG.getNode(N->getOpcode(), dl, VT, N0,
7092                          DAG.getConstant(RotAmt, dl, N1.getValueType()));
7093     }
7094   }
7095 
7096   // fold (rot* x, (trunc (and y, c))) -> (rot* x, (and (trunc y), (trunc c))).
7097   if (N1.getOpcode() == ISD::TRUNCATE &&
7098       N1.getOperand(0).getOpcode() == ISD::AND) {
7099     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
7100       return DAG.getNode(N->getOpcode(), dl, VT, N0, NewOp1);
7101   }
7102 
7103   unsigned NextOp = N0.getOpcode();
7104   // fold (rot* (rot* x, c2), c1) -> (rot* x, c1 +- c2 % bitsize)
7105   if (NextOp == ISD::ROTL || NextOp == ISD::ROTR) {
7106     SDNode *C1 = DAG.isConstantIntBuildVectorOrConstantInt(N1);
7107     SDNode *C2 = DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1));
7108     if (C1 && C2 && C1->getValueType(0) == C2->getValueType(0)) {
7109       EVT ShiftVT = C1->getValueType(0);
7110       bool SameSide = (N->getOpcode() == NextOp);
7111       unsigned CombineOp = SameSide ? ISD::ADD : ISD::SUB;
7112       if (SDValue CombinedShift =
7113               DAG.FoldConstantArithmetic(CombineOp, dl, ShiftVT, C1, C2)) {
7114         SDValue BitsizeC = DAG.getConstant(Bitsize, dl, ShiftVT);
7115         SDValue CombinedShiftNorm = DAG.FoldConstantArithmetic(
7116             ISD::SREM, dl, ShiftVT, CombinedShift.getNode(),
7117             BitsizeC.getNode());
7118         return DAG.getNode(N->getOpcode(), dl, VT, N0->getOperand(0),
7119                            CombinedShiftNorm);
7120       }
7121     }
7122   }
7123   return SDValue();
7124 }
7125 
7126 SDValue DAGCombiner::visitSHL(SDNode *N) {
7127   SDValue N0 = N->getOperand(0);
7128   SDValue N1 = N->getOperand(1);
7129   if (SDValue V = DAG.simplifyShift(N0, N1))
7130     return V;
7131 
7132   EVT VT = N0.getValueType();
7133   EVT ShiftVT = N1.getValueType();
7134   unsigned OpSizeInBits = VT.getScalarSizeInBits();
7135 
7136   // fold vector ops
7137   if (VT.isVector()) {
7138     if (SDValue FoldedVOp = SimplifyVBinOp(N))
7139       return FoldedVOp;
7140 
7141     BuildVectorSDNode *N1CV = dyn_cast<BuildVectorSDNode>(N1);
7142     // If setcc produces all-one true value then:
7143     // (shl (and (setcc) N01CV) N1CV) -> (and (setcc) N01CV<<N1CV)
7144     if (N1CV && N1CV->isConstant()) {
7145       if (N0.getOpcode() == ISD::AND) {
7146         SDValue N00 = N0->getOperand(0);
7147         SDValue N01 = N0->getOperand(1);
7148         BuildVectorSDNode *N01CV = dyn_cast<BuildVectorSDNode>(N01);
7149 
7150         if (N01CV && N01CV->isConstant() && N00.getOpcode() == ISD::SETCC &&
7151             TLI.getBooleanContents(N00.getOperand(0).getValueType()) ==
7152                 TargetLowering::ZeroOrNegativeOneBooleanContent) {
7153           if (SDValue C = DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT,
7154                                                      N01CV, N1CV))
7155             return DAG.getNode(ISD::AND, SDLoc(N), VT, N00, C);
7156         }
7157       }
7158     }
7159   }
7160 
7161   ConstantSDNode *N1C = isConstOrConstSplat(N1);
7162 
7163   // fold (shl c1, c2) -> c1<<c2
7164   // TODO - support non-uniform vector shift amounts.
7165   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
7166   if (N0C && N1C && !N1C->isOpaque())
7167     return DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT, N0C, N1C);
7168 
7169   if (SDValue NewSel = foldBinOpIntoSelect(N))
7170     return NewSel;
7171 
7172   // if (shl x, c) is known to be zero, return 0
7173   if (DAG.MaskedValueIsZero(SDValue(N, 0),
7174                             APInt::getAllOnesValue(OpSizeInBits)))
7175     return DAG.getConstant(0, SDLoc(N), VT);
7176 
7177   // fold (shl x, (trunc (and y, c))) -> (shl x, (and (trunc y), (trunc c))).
7178   if (N1.getOpcode() == ISD::TRUNCATE &&
7179       N1.getOperand(0).getOpcode() == ISD::AND) {
7180     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
7181       return DAG.getNode(ISD::SHL, SDLoc(N), VT, N0, NewOp1);
7182   }
7183 
7184   // TODO - support non-uniform vector shift amounts.
7185   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
7186     return SDValue(N, 0);
7187 
7188   // fold (shl (shl x, c1), c2) -> 0 or (shl x, (add c1, c2))
7189   if (N0.getOpcode() == ISD::SHL) {
7190     auto MatchOutOfRange = [OpSizeInBits](ConstantSDNode *LHS,
7191                                           ConstantSDNode *RHS) {
7192       APInt c1 = LHS->getAPIntValue();
7193       APInt c2 = RHS->getAPIntValue();
7194       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
7195       return (c1 + c2).uge(OpSizeInBits);
7196     };
7197     if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchOutOfRange))
7198       return DAG.getConstant(0, SDLoc(N), VT);
7199 
7200     auto MatchInRange = [OpSizeInBits](ConstantSDNode *LHS,
7201                                        ConstantSDNode *RHS) {
7202       APInt c1 = LHS->getAPIntValue();
7203       APInt c2 = RHS->getAPIntValue();
7204       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
7205       return (c1 + c2).ult(OpSizeInBits);
7206     };
7207     if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchInRange)) {
7208       SDLoc DL(N);
7209       SDValue Sum = DAG.getNode(ISD::ADD, DL, ShiftVT, N1, N0.getOperand(1));
7210       return DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0), Sum);
7211     }
7212   }
7213 
7214   // fold (shl (ext (shl x, c1)), c2) -> (shl (ext x), (add c1, c2))
7215   // For this to be valid, the second form must not preserve any of the bits
7216   // that are shifted out by the inner shift in the first form.  This means
7217   // the outer shift size must be >= the number of bits added by the ext.
7218   // As a corollary, we don't care what kind of ext it is.
7219   if ((N0.getOpcode() == ISD::ZERO_EXTEND ||
7220        N0.getOpcode() == ISD::ANY_EXTEND ||
7221        N0.getOpcode() == ISD::SIGN_EXTEND) &&
7222       N0.getOperand(0).getOpcode() == ISD::SHL) {
7223     SDValue N0Op0 = N0.getOperand(0);
7224     SDValue InnerShiftAmt = N0Op0.getOperand(1);
7225     EVT InnerVT = N0Op0.getValueType();
7226     uint64_t InnerBitwidth = InnerVT.getScalarSizeInBits();
7227 
7228     auto MatchOutOfRange = [OpSizeInBits, InnerBitwidth](ConstantSDNode *LHS,
7229                                                          ConstantSDNode *RHS) {
7230       APInt c1 = LHS->getAPIntValue();
7231       APInt c2 = RHS->getAPIntValue();
7232       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
7233       return c2.uge(OpSizeInBits - InnerBitwidth) &&
7234              (c1 + c2).uge(OpSizeInBits);
7235     };
7236     if (ISD::matchBinaryPredicate(InnerShiftAmt, N1, MatchOutOfRange,
7237                                   /*AllowUndefs*/ false,
7238                                   /*AllowTypeMismatch*/ true))
7239       return DAG.getConstant(0, SDLoc(N), VT);
7240 
7241     auto MatchInRange = [OpSizeInBits, InnerBitwidth](ConstantSDNode *LHS,
7242                                                       ConstantSDNode *RHS) {
7243       APInt c1 = LHS->getAPIntValue();
7244       APInt c2 = RHS->getAPIntValue();
7245       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
7246       return c2.uge(OpSizeInBits - InnerBitwidth) &&
7247              (c1 + c2).ult(OpSizeInBits);
7248     };
7249     if (ISD::matchBinaryPredicate(InnerShiftAmt, N1, MatchInRange,
7250                                   /*AllowUndefs*/ false,
7251                                   /*AllowTypeMismatch*/ true)) {
7252       SDLoc DL(N);
7253       SDValue Ext = DAG.getNode(N0.getOpcode(), DL, VT, N0Op0.getOperand(0));
7254       SDValue Sum = DAG.getZExtOrTrunc(InnerShiftAmt, DL, ShiftVT);
7255       Sum = DAG.getNode(ISD::ADD, DL, ShiftVT, Sum, N1);
7256       return DAG.getNode(ISD::SHL, DL, VT, Ext, Sum);
7257     }
7258   }
7259 
7260   // fold (shl (zext (srl x, C)), C) -> (zext (shl (srl x, C), C))
7261   // Only fold this if the inner zext has no other uses to avoid increasing
7262   // the total number of instructions.
7263   if (N0.getOpcode() == ISD::ZERO_EXTEND && N0.hasOneUse() &&
7264       N0.getOperand(0).getOpcode() == ISD::SRL) {
7265     SDValue N0Op0 = N0.getOperand(0);
7266     SDValue InnerShiftAmt = N0Op0.getOperand(1);
7267 
7268     auto MatchEqual = [VT](ConstantSDNode *LHS, ConstantSDNode *RHS) {
7269       APInt c1 = LHS->getAPIntValue();
7270       APInt c2 = RHS->getAPIntValue();
7271       zeroExtendToMatch(c1, c2);
7272       return c1.ult(VT.getScalarSizeInBits()) && (c1 == c2);
7273     };
7274     if (ISD::matchBinaryPredicate(InnerShiftAmt, N1, MatchEqual,
7275                                   /*AllowUndefs*/ false,
7276                                   /*AllowTypeMismatch*/ true)) {
7277       SDLoc DL(N);
7278       EVT InnerShiftAmtVT = N0Op0.getOperand(1).getValueType();
7279       SDValue NewSHL = DAG.getZExtOrTrunc(N1, DL, InnerShiftAmtVT);
7280       NewSHL = DAG.getNode(ISD::SHL, DL, N0Op0.getValueType(), N0Op0, NewSHL);
7281       AddToWorklist(NewSHL.getNode());
7282       return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N0), VT, NewSHL);
7283     }
7284   }
7285 
7286   // fold (shl (sr[la] exact X,  C1), C2) -> (shl    X, (C2-C1)) if C1 <= C2
7287   // fold (shl (sr[la] exact X,  C1), C2) -> (sr[la] X, (C2-C1)) if C1  > C2
7288   // TODO - support non-uniform vector shift amounts.
7289   if (N1C && (N0.getOpcode() == ISD::SRL || N0.getOpcode() == ISD::SRA) &&
7290       N0->getFlags().hasExact()) {
7291     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
7292       uint64_t C1 = N0C1->getZExtValue();
7293       uint64_t C2 = N1C->getZExtValue();
7294       SDLoc DL(N);
7295       if (C1 <= C2)
7296         return DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
7297                            DAG.getConstant(C2 - C1, DL, ShiftVT));
7298       return DAG.getNode(N0.getOpcode(), DL, VT, N0.getOperand(0),
7299                          DAG.getConstant(C1 - C2, DL, ShiftVT));
7300     }
7301   }
7302 
7303   // fold (shl (srl x, c1), c2) -> (and (shl x, (sub c2, c1), MASK) or
7304   //                               (and (srl x, (sub c1, c2), MASK)
7305   // Only fold this if the inner shift has no other uses -- if it does, folding
7306   // this will increase the total number of instructions.
7307   // TODO - drop hasOneUse requirement if c1 == c2?
7308   // TODO - support non-uniform vector shift amounts.
7309   if (N1C && N0.getOpcode() == ISD::SRL && N0.hasOneUse() &&
7310       TLI.shouldFoldConstantShiftPairToMask(N, Level)) {
7311     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
7312       if (N0C1->getAPIntValue().ult(OpSizeInBits)) {
7313         uint64_t c1 = N0C1->getZExtValue();
7314         uint64_t c2 = N1C->getZExtValue();
7315         APInt Mask = APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - c1);
7316         SDValue Shift;
7317         if (c2 > c1) {
7318           Mask <<= c2 - c1;
7319           SDLoc DL(N);
7320           Shift = DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
7321                               DAG.getConstant(c2 - c1, DL, ShiftVT));
7322         } else {
7323           Mask.lshrInPlace(c1 - c2);
7324           SDLoc DL(N);
7325           Shift = DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0),
7326                               DAG.getConstant(c1 - c2, DL, ShiftVT));
7327         }
7328         SDLoc DL(N0);
7329         return DAG.getNode(ISD::AND, DL, VT, Shift,
7330                            DAG.getConstant(Mask, DL, VT));
7331       }
7332     }
7333   }
7334 
7335   // fold (shl (sra x, c1), c1) -> (and x, (shl -1, c1))
7336   if (N0.getOpcode() == ISD::SRA && N1 == N0.getOperand(1) &&
7337       isConstantOrConstantVector(N1, /* No Opaques */ true)) {
7338     SDLoc DL(N);
7339     SDValue AllBits = DAG.getAllOnesConstant(DL, VT);
7340     SDValue HiBitsMask = DAG.getNode(ISD::SHL, DL, VT, AllBits, N1);
7341     return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), HiBitsMask);
7342   }
7343 
7344   // fold (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
7345   // fold (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
7346   // Variant of version done on multiply, except mul by a power of 2 is turned
7347   // into a shift.
7348   if ((N0.getOpcode() == ISD::ADD || N0.getOpcode() == ISD::OR) &&
7349       N0.getNode()->hasOneUse() &&
7350       isConstantOrConstantVector(N1, /* No Opaques */ true) &&
7351       isConstantOrConstantVector(N0.getOperand(1), /* No Opaques */ true) &&
7352       TLI.isDesirableToCommuteWithShift(N, Level)) {
7353     SDValue Shl0 = DAG.getNode(ISD::SHL, SDLoc(N0), VT, N0.getOperand(0), N1);
7354     SDValue Shl1 = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1);
7355     AddToWorklist(Shl0.getNode());
7356     AddToWorklist(Shl1.getNode());
7357     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, Shl0, Shl1);
7358   }
7359 
7360   // fold (shl (mul x, c1), c2) -> (mul x, c1 << c2)
7361   if (N0.getOpcode() == ISD::MUL && N0.getNode()->hasOneUse() &&
7362       isConstantOrConstantVector(N1, /* No Opaques */ true) &&
7363       isConstantOrConstantVector(N0.getOperand(1), /* No Opaques */ true)) {
7364     SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1);
7365     if (isConstantOrConstantVector(Shl))
7366       return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), Shl);
7367   }
7368 
7369   if (N1C && !N1C->isOpaque())
7370     if (SDValue NewSHL = visitShiftByConstant(N, N1C))
7371       return NewSHL;
7372 
7373   return SDValue();
7374 }
7375 
7376 SDValue DAGCombiner::visitSRA(SDNode *N) {
7377   SDValue N0 = N->getOperand(0);
7378   SDValue N1 = N->getOperand(1);
7379   if (SDValue V = DAG.simplifyShift(N0, N1))
7380     return V;
7381 
7382   EVT VT = N0.getValueType();
7383   unsigned OpSizeInBits = VT.getScalarSizeInBits();
7384 
7385   // Arithmetic shifting an all-sign-bit value is a no-op.
7386   // fold (sra 0, x) -> 0
7387   // fold (sra -1, x) -> -1
7388   if (DAG.ComputeNumSignBits(N0) == OpSizeInBits)
7389     return N0;
7390 
7391   // fold vector ops
7392   if (VT.isVector())
7393     if (SDValue FoldedVOp = SimplifyVBinOp(N))
7394       return FoldedVOp;
7395 
7396   ConstantSDNode *N1C = isConstOrConstSplat(N1);
7397 
7398   // fold (sra c1, c2) -> (sra c1, c2)
7399   // TODO - support non-uniform vector shift amounts.
7400   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
7401   if (N0C && N1C && !N1C->isOpaque())
7402     return DAG.FoldConstantArithmetic(ISD::SRA, SDLoc(N), VT, N0C, N1C);
7403 
7404   if (SDValue NewSel = foldBinOpIntoSelect(N))
7405     return NewSel;
7406 
7407   // fold (sra (shl x, c1), c1) -> sext_inreg for some c1 and target supports
7408   // sext_inreg.
7409   if (N1C && N0.getOpcode() == ISD::SHL && N1 == N0.getOperand(1)) {
7410     unsigned LowBits = OpSizeInBits - (unsigned)N1C->getZExtValue();
7411     EVT ExtVT = EVT::getIntegerVT(*DAG.getContext(), LowBits);
7412     if (VT.isVector())
7413       ExtVT = EVT::getVectorVT(*DAG.getContext(),
7414                                ExtVT, VT.getVectorNumElements());
7415     if ((!LegalOperations ||
7416          TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, ExtVT)))
7417       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
7418                          N0.getOperand(0), DAG.getValueType(ExtVT));
7419   }
7420 
7421   // fold (sra (sra x, c1), c2) -> (sra x, (add c1, c2))
7422   // clamp (add c1, c2) to max shift.
7423   if (N0.getOpcode() == ISD::SRA) {
7424     SDLoc DL(N);
7425     EVT ShiftVT = N1.getValueType();
7426     EVT ShiftSVT = ShiftVT.getScalarType();
7427     SmallVector<SDValue, 16> ShiftValues;
7428 
7429     auto SumOfShifts = [&](ConstantSDNode *LHS, ConstantSDNode *RHS) {
7430       APInt c1 = LHS->getAPIntValue();
7431       APInt c2 = RHS->getAPIntValue();
7432       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
7433       APInt Sum = c1 + c2;
7434       unsigned ShiftSum =
7435           Sum.uge(OpSizeInBits) ? (OpSizeInBits - 1) : Sum.getZExtValue();
7436       ShiftValues.push_back(DAG.getConstant(ShiftSum, DL, ShiftSVT));
7437       return true;
7438     };
7439     if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), SumOfShifts)) {
7440       SDValue ShiftValue;
7441       if (VT.isVector())
7442         ShiftValue = DAG.getBuildVector(ShiftVT, DL, ShiftValues);
7443       else
7444         ShiftValue = ShiftValues[0];
7445       return DAG.getNode(ISD::SRA, DL, VT, N0.getOperand(0), ShiftValue);
7446     }
7447   }
7448 
7449   // fold (sra (shl X, m), (sub result_size, n))
7450   // -> (sign_extend (trunc (shl X, (sub (sub result_size, n), m)))) for
7451   // result_size - n != m.
7452   // If truncate is free for the target sext(shl) is likely to result in better
7453   // code.
7454   if (N0.getOpcode() == ISD::SHL && N1C) {
7455     // Get the two constanst of the shifts, CN0 = m, CN = n.
7456     const ConstantSDNode *N01C = isConstOrConstSplat(N0.getOperand(1));
7457     if (N01C) {
7458       LLVMContext &Ctx = *DAG.getContext();
7459       // Determine what the truncate's result bitsize and type would be.
7460       EVT TruncVT = EVT::getIntegerVT(Ctx, OpSizeInBits - N1C->getZExtValue());
7461 
7462       if (VT.isVector())
7463         TruncVT = EVT::getVectorVT(Ctx, TruncVT, VT.getVectorNumElements());
7464 
7465       // Determine the residual right-shift amount.
7466       int ShiftAmt = N1C->getZExtValue() - N01C->getZExtValue();
7467 
7468       // If the shift is not a no-op (in which case this should be just a sign
7469       // extend already), the truncated to type is legal, sign_extend is legal
7470       // on that type, and the truncate to that type is both legal and free,
7471       // perform the transform.
7472       if ((ShiftAmt > 0) &&
7473           TLI.isOperationLegalOrCustom(ISD::SIGN_EXTEND, TruncVT) &&
7474           TLI.isOperationLegalOrCustom(ISD::TRUNCATE, VT) &&
7475           TLI.isTruncateFree(VT, TruncVT)) {
7476         SDLoc DL(N);
7477         SDValue Amt = DAG.getConstant(ShiftAmt, DL,
7478             getShiftAmountTy(N0.getOperand(0).getValueType()));
7479         SDValue Shift = DAG.getNode(ISD::SRL, DL, VT,
7480                                     N0.getOperand(0), Amt);
7481         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, TruncVT,
7482                                     Shift);
7483         return DAG.getNode(ISD::SIGN_EXTEND, DL,
7484                            N->getValueType(0), Trunc);
7485       }
7486     }
7487   }
7488 
7489   // fold (sra x, (trunc (and y, c))) -> (sra x, (and (trunc y), (trunc c))).
7490   if (N1.getOpcode() == ISD::TRUNCATE &&
7491       N1.getOperand(0).getOpcode() == ISD::AND) {
7492     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
7493       return DAG.getNode(ISD::SRA, SDLoc(N), VT, N0, NewOp1);
7494   }
7495 
7496   // fold (sra (trunc (sra x, c1)), c2) -> (trunc (sra x, c1 + c2))
7497   // fold (sra (trunc (srl x, c1)), c2) -> (trunc (sra x, c1 + c2))
7498   //      if c1 is equal to the number of bits the trunc removes
7499   // TODO - support non-uniform vector shift amounts.
7500   if (N0.getOpcode() == ISD::TRUNCATE &&
7501       (N0.getOperand(0).getOpcode() == ISD::SRL ||
7502        N0.getOperand(0).getOpcode() == ISD::SRA) &&
7503       N0.getOperand(0).hasOneUse() &&
7504       N0.getOperand(0).getOperand(1).hasOneUse() && N1C) {
7505     SDValue N0Op0 = N0.getOperand(0);
7506     if (ConstantSDNode *LargeShift = isConstOrConstSplat(N0Op0.getOperand(1))) {
7507       EVT LargeVT = N0Op0.getValueType();
7508       unsigned TruncBits = LargeVT.getScalarSizeInBits() - OpSizeInBits;
7509       if (LargeShift->getAPIntValue() == TruncBits) {
7510         SDLoc DL(N);
7511         SDValue Amt = DAG.getConstant(N1C->getZExtValue() + TruncBits, DL,
7512                                       getShiftAmountTy(LargeVT));
7513         SDValue SRA =
7514             DAG.getNode(ISD::SRA, DL, LargeVT, N0Op0.getOperand(0), Amt);
7515         return DAG.getNode(ISD::TRUNCATE, DL, VT, SRA);
7516       }
7517     }
7518   }
7519 
7520   // Simplify, based on bits shifted out of the LHS.
7521   // TODO - support non-uniform vector shift amounts.
7522   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
7523     return SDValue(N, 0);
7524 
7525   // If the sign bit is known to be zero, switch this to a SRL.
7526   if (DAG.SignBitIsZero(N0))
7527     return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, N1);
7528 
7529   if (N1C && !N1C->isOpaque())
7530     if (SDValue NewSRA = visitShiftByConstant(N, N1C))
7531       return NewSRA;
7532 
7533   return SDValue();
7534 }
7535 
7536 SDValue DAGCombiner::visitSRL(SDNode *N) {
7537   SDValue N0 = N->getOperand(0);
7538   SDValue N1 = N->getOperand(1);
7539   if (SDValue V = DAG.simplifyShift(N0, N1))
7540     return V;
7541 
7542   EVT VT = N0.getValueType();
7543   unsigned OpSizeInBits = VT.getScalarSizeInBits();
7544 
7545   // fold vector ops
7546   if (VT.isVector())
7547     if (SDValue FoldedVOp = SimplifyVBinOp(N))
7548       return FoldedVOp;
7549 
7550   ConstantSDNode *N1C = isConstOrConstSplat(N1);
7551 
7552   // fold (srl c1, c2) -> c1 >>u c2
7553   // TODO - support non-uniform vector shift amounts.
7554   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
7555   if (N0C && N1C && !N1C->isOpaque())
7556     return DAG.FoldConstantArithmetic(ISD::SRL, SDLoc(N), VT, N0C, N1C);
7557 
7558   if (SDValue NewSel = foldBinOpIntoSelect(N))
7559     return NewSel;
7560 
7561   // if (srl x, c) is known to be zero, return 0
7562   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
7563                                    APInt::getAllOnesValue(OpSizeInBits)))
7564     return DAG.getConstant(0, SDLoc(N), VT);
7565 
7566   // fold (srl (srl x, c1), c2) -> 0 or (srl x, (add c1, c2))
7567   if (N0.getOpcode() == ISD::SRL) {
7568     auto MatchOutOfRange = [OpSizeInBits](ConstantSDNode *LHS,
7569                                           ConstantSDNode *RHS) {
7570       APInt c1 = LHS->getAPIntValue();
7571       APInt c2 = RHS->getAPIntValue();
7572       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
7573       return (c1 + c2).uge(OpSizeInBits);
7574     };
7575     if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchOutOfRange))
7576       return DAG.getConstant(0, SDLoc(N), VT);
7577 
7578     auto MatchInRange = [OpSizeInBits](ConstantSDNode *LHS,
7579                                        ConstantSDNode *RHS) {
7580       APInt c1 = LHS->getAPIntValue();
7581       APInt c2 = RHS->getAPIntValue();
7582       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
7583       return (c1 + c2).ult(OpSizeInBits);
7584     };
7585     if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchInRange)) {
7586       SDLoc DL(N);
7587       EVT ShiftVT = N1.getValueType();
7588       SDValue Sum = DAG.getNode(ISD::ADD, DL, ShiftVT, N1, N0.getOperand(1));
7589       return DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0), Sum);
7590     }
7591   }
7592 
7593   // fold (srl (trunc (srl x, c1)), c2) -> 0 or (trunc (srl x, (add c1, c2)))
7594   // TODO - support non-uniform vector shift amounts.
7595   if (N1C && N0.getOpcode() == ISD::TRUNCATE &&
7596       N0.getOperand(0).getOpcode() == ISD::SRL) {
7597     if (auto N001C = isConstOrConstSplat(N0.getOperand(0).getOperand(1))) {
7598       uint64_t c1 = N001C->getZExtValue();
7599       uint64_t c2 = N1C->getZExtValue();
7600       EVT InnerShiftVT = N0.getOperand(0).getValueType();
7601       EVT ShiftCountVT = N0.getOperand(0).getOperand(1).getValueType();
7602       uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits();
7603       // This is only valid if the OpSizeInBits + c1 = size of inner shift.
7604       if (c1 + OpSizeInBits == InnerShiftSize) {
7605         SDLoc DL(N0);
7606         if (c1 + c2 >= InnerShiftSize)
7607           return DAG.getConstant(0, DL, VT);
7608         return DAG.getNode(ISD::TRUNCATE, DL, VT,
7609                            DAG.getNode(ISD::SRL, DL, InnerShiftVT,
7610                                        N0.getOperand(0).getOperand(0),
7611                                        DAG.getConstant(c1 + c2, DL,
7612                                                        ShiftCountVT)));
7613       }
7614     }
7615   }
7616 
7617   // fold (srl (shl x, c), c) -> (and x, cst2)
7618   // TODO - (srl (shl x, c1), c2).
7619   if (N0.getOpcode() == ISD::SHL && N0.getOperand(1) == N1 &&
7620       isConstantOrConstantVector(N1, /* NoOpaques */ true)) {
7621     SDLoc DL(N);
7622     SDValue Mask =
7623         DAG.getNode(ISD::SRL, DL, VT, DAG.getAllOnesConstant(DL, VT), N1);
7624     AddToWorklist(Mask.getNode());
7625     return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), Mask);
7626   }
7627 
7628   // fold (srl (anyextend x), c) -> (and (anyextend (srl x, c)), mask)
7629   // TODO - support non-uniform vector shift amounts.
7630   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
7631     // Shifting in all undef bits?
7632     EVT SmallVT = N0.getOperand(0).getValueType();
7633     unsigned BitSize = SmallVT.getScalarSizeInBits();
7634     if (N1C->getAPIntValue().uge(BitSize))
7635       return DAG.getUNDEF(VT);
7636 
7637     if (!LegalTypes || TLI.isTypeDesirableForOp(ISD::SRL, SmallVT)) {
7638       uint64_t ShiftAmt = N1C->getZExtValue();
7639       SDLoc DL0(N0);
7640       SDValue SmallShift = DAG.getNode(ISD::SRL, DL0, SmallVT,
7641                                        N0.getOperand(0),
7642                           DAG.getConstant(ShiftAmt, DL0,
7643                                           getShiftAmountTy(SmallVT)));
7644       AddToWorklist(SmallShift.getNode());
7645       APInt Mask = APInt::getLowBitsSet(OpSizeInBits, OpSizeInBits - ShiftAmt);
7646       SDLoc DL(N);
7647       return DAG.getNode(ISD::AND, DL, VT,
7648                          DAG.getNode(ISD::ANY_EXTEND, DL, VT, SmallShift),
7649                          DAG.getConstant(Mask, DL, VT));
7650     }
7651   }
7652 
7653   // fold (srl (sra X, Y), 31) -> (srl X, 31).  This srl only looks at the sign
7654   // bit, which is unmodified by sra.
7655   if (N1C && N1C->getAPIntValue() == (OpSizeInBits - 1)) {
7656     if (N0.getOpcode() == ISD::SRA)
7657       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0.getOperand(0), N1);
7658   }
7659 
7660   // fold (srl (ctlz x), "5") -> x  iff x has one bit set (the low bit).
7661   if (N1C && N0.getOpcode() == ISD::CTLZ &&
7662       N1C->getAPIntValue() == Log2_32(OpSizeInBits)) {
7663     KnownBits Known = DAG.computeKnownBits(N0.getOperand(0));
7664 
7665     // If any of the input bits are KnownOne, then the input couldn't be all
7666     // zeros, thus the result of the srl will always be zero.
7667     if (Known.One.getBoolValue()) return DAG.getConstant(0, SDLoc(N0), VT);
7668 
7669     // If all of the bits input the to ctlz node are known to be zero, then
7670     // the result of the ctlz is "32" and the result of the shift is one.
7671     APInt UnknownBits = ~Known.Zero;
7672     if (UnknownBits == 0) return DAG.getConstant(1, SDLoc(N0), VT);
7673 
7674     // Otherwise, check to see if there is exactly one bit input to the ctlz.
7675     if (UnknownBits.isPowerOf2()) {
7676       // Okay, we know that only that the single bit specified by UnknownBits
7677       // could be set on input to the CTLZ node. If this bit is set, the SRL
7678       // will return 0, if it is clear, it returns 1. Change the CTLZ/SRL pair
7679       // to an SRL/XOR pair, which is likely to simplify more.
7680       unsigned ShAmt = UnknownBits.countTrailingZeros();
7681       SDValue Op = N0.getOperand(0);
7682 
7683       if (ShAmt) {
7684         SDLoc DL(N0);
7685         Op = DAG.getNode(ISD::SRL, DL, VT, Op,
7686                   DAG.getConstant(ShAmt, DL,
7687                                   getShiftAmountTy(Op.getValueType())));
7688         AddToWorklist(Op.getNode());
7689       }
7690 
7691       SDLoc DL(N);
7692       return DAG.getNode(ISD::XOR, DL, VT,
7693                          Op, DAG.getConstant(1, DL, VT));
7694     }
7695   }
7696 
7697   // fold (srl x, (trunc (and y, c))) -> (srl x, (and (trunc y), (trunc c))).
7698   if (N1.getOpcode() == ISD::TRUNCATE &&
7699       N1.getOperand(0).getOpcode() == ISD::AND) {
7700     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
7701       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, NewOp1);
7702   }
7703 
7704   // fold operands of srl based on knowledge that the low bits are not
7705   // demanded.
7706   // TODO - support non-uniform vector shift amounts.
7707   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
7708     return SDValue(N, 0);
7709 
7710   if (N1C && !N1C->isOpaque())
7711     if (SDValue NewSRL = visitShiftByConstant(N, N1C))
7712       return NewSRL;
7713 
7714   // Attempt to convert a srl of a load into a narrower zero-extending load.
7715   if (SDValue NarrowLoad = ReduceLoadWidth(N))
7716     return NarrowLoad;
7717 
7718   // Here is a common situation. We want to optimize:
7719   //
7720   //   %a = ...
7721   //   %b = and i32 %a, 2
7722   //   %c = srl i32 %b, 1
7723   //   brcond i32 %c ...
7724   //
7725   // into
7726   //
7727   //   %a = ...
7728   //   %b = and %a, 2
7729   //   %c = setcc eq %b, 0
7730   //   brcond %c ...
7731   //
7732   // However when after the source operand of SRL is optimized into AND, the SRL
7733   // itself may not be optimized further. Look for it and add the BRCOND into
7734   // the worklist.
7735   if (N->hasOneUse()) {
7736     SDNode *Use = *N->use_begin();
7737     if (Use->getOpcode() == ISD::BRCOND)
7738       AddToWorklist(Use);
7739     else if (Use->getOpcode() == ISD::TRUNCATE && Use->hasOneUse()) {
7740       // Also look pass the truncate.
7741       Use = *Use->use_begin();
7742       if (Use->getOpcode() == ISD::BRCOND)
7743         AddToWorklist(Use);
7744     }
7745   }
7746 
7747   return SDValue();
7748 }
7749 
7750 SDValue DAGCombiner::visitFunnelShift(SDNode *N) {
7751   EVT VT = N->getValueType(0);
7752   SDValue N0 = N->getOperand(0);
7753   SDValue N1 = N->getOperand(1);
7754   SDValue N2 = N->getOperand(2);
7755   bool IsFSHL = N->getOpcode() == ISD::FSHL;
7756   unsigned BitWidth = VT.getScalarSizeInBits();
7757 
7758   // fold (fshl N0, N1, 0) -> N0
7759   // fold (fshr N0, N1, 0) -> N1
7760   if (isPowerOf2_32(BitWidth))
7761     if (DAG.MaskedValueIsZero(
7762             N2, APInt(N2.getScalarValueSizeInBits(), BitWidth - 1)))
7763       return IsFSHL ? N0 : N1;
7764 
7765   auto IsUndefOrZero = [](SDValue V) {
7766     return V.isUndef() || isNullOrNullSplat(V, /*AllowUndefs*/ true);
7767   };
7768 
7769   // TODO - support non-uniform vector shift amounts.
7770   if (ConstantSDNode *Cst = isConstOrConstSplat(N2)) {
7771     EVT ShAmtTy = N2.getValueType();
7772 
7773     // fold (fsh* N0, N1, c) -> (fsh* N0, N1, c % BitWidth)
7774     if (Cst->getAPIntValue().uge(BitWidth)) {
7775       uint64_t RotAmt = Cst->getAPIntValue().urem(BitWidth);
7776       return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N0, N1,
7777                          DAG.getConstant(RotAmt, SDLoc(N), ShAmtTy));
7778     }
7779 
7780     unsigned ShAmt = Cst->getZExtValue();
7781     if (ShAmt == 0)
7782       return IsFSHL ? N0 : N1;
7783 
7784     // fold fshl(undef_or_zero, N1, C) -> lshr(N1, BW-C)
7785     // fold fshr(undef_or_zero, N1, C) -> lshr(N1, C)
7786     // fold fshl(N0, undef_or_zero, C) -> shl(N0, C)
7787     // fold fshr(N0, undef_or_zero, C) -> shl(N0, BW-C)
7788     if (IsUndefOrZero(N0))
7789       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N1,
7790                          DAG.getConstant(IsFSHL ? BitWidth - ShAmt : ShAmt,
7791                                          SDLoc(N), ShAmtTy));
7792     if (IsUndefOrZero(N1))
7793       return DAG.getNode(ISD::SHL, SDLoc(N), VT, N0,
7794                          DAG.getConstant(IsFSHL ? ShAmt : BitWidth - ShAmt,
7795                                          SDLoc(N), ShAmtTy));
7796   }
7797 
7798   // fold fshr(undef_or_zero, N1, N2) -> lshr(N1, N2)
7799   // fold fshl(N0, undef_or_zero, N2) -> shl(N0, N2)
7800   // iff We know the shift amount is in range.
7801   // TODO: when is it worth doing SUB(BW, N2) as well?
7802   if (isPowerOf2_32(BitWidth)) {
7803     APInt ModuloBits(N2.getScalarValueSizeInBits(), BitWidth - 1);
7804     if (IsUndefOrZero(N0) && !IsFSHL && DAG.MaskedValueIsZero(N2, ~ModuloBits))
7805       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N1, N2);
7806     if (IsUndefOrZero(N1) && IsFSHL && DAG.MaskedValueIsZero(N2, ~ModuloBits))
7807       return DAG.getNode(ISD::SHL, SDLoc(N), VT, N0, N2);
7808   }
7809 
7810   // fold (fshl N0, N0, N2) -> (rotl N0, N2)
7811   // fold (fshr N0, N0, N2) -> (rotr N0, N2)
7812   // TODO: Investigate flipping this rotate if only one is legal, if funnel shift
7813   // is legal as well we might be better off avoiding non-constant (BW - N2).
7814   unsigned RotOpc = IsFSHL ? ISD::ROTL : ISD::ROTR;
7815   if (N0 == N1 && hasOperation(RotOpc, VT))
7816     return DAG.getNode(RotOpc, SDLoc(N), VT, N0, N2);
7817 
7818   // Simplify, based on bits shifted out of N0/N1.
7819   if (SimplifyDemandedBits(SDValue(N, 0)))
7820     return SDValue(N, 0);
7821 
7822   return SDValue();
7823 }
7824 
7825 SDValue DAGCombiner::visitABS(SDNode *N) {
7826   SDValue N0 = N->getOperand(0);
7827   EVT VT = N->getValueType(0);
7828 
7829   // fold (abs c1) -> c2
7830   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7831     return DAG.getNode(ISD::ABS, SDLoc(N), VT, N0);
7832   // fold (abs (abs x)) -> (abs x)
7833   if (N0.getOpcode() == ISD::ABS)
7834     return N0;
7835   // fold (abs x) -> x iff not-negative
7836   if (DAG.SignBitIsZero(N0))
7837     return N0;
7838   return SDValue();
7839 }
7840 
7841 SDValue DAGCombiner::visitBSWAP(SDNode *N) {
7842   SDValue N0 = N->getOperand(0);
7843   EVT VT = N->getValueType(0);
7844 
7845   // fold (bswap c1) -> c2
7846   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7847     return DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N0);
7848   // fold (bswap (bswap x)) -> x
7849   if (N0.getOpcode() == ISD::BSWAP)
7850     return N0->getOperand(0);
7851   return SDValue();
7852 }
7853 
7854 SDValue DAGCombiner::visitBITREVERSE(SDNode *N) {
7855   SDValue N0 = N->getOperand(0);
7856   EVT VT = N->getValueType(0);
7857 
7858   // fold (bitreverse c1) -> c2
7859   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7860     return DAG.getNode(ISD::BITREVERSE, SDLoc(N), VT, N0);
7861   // fold (bitreverse (bitreverse x)) -> x
7862   if (N0.getOpcode() == ISD::BITREVERSE)
7863     return N0.getOperand(0);
7864   return SDValue();
7865 }
7866 
7867 SDValue DAGCombiner::visitCTLZ(SDNode *N) {
7868   SDValue N0 = N->getOperand(0);
7869   EVT VT = N->getValueType(0);
7870 
7871   // fold (ctlz c1) -> c2
7872   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7873     return DAG.getNode(ISD::CTLZ, SDLoc(N), VT, N0);
7874 
7875   // If the value is known never to be zero, switch to the undef version.
7876   if (!LegalOperations || TLI.isOperationLegal(ISD::CTLZ_ZERO_UNDEF, VT)) {
7877     if (DAG.isKnownNeverZero(N0))
7878       return DAG.getNode(ISD::CTLZ_ZERO_UNDEF, SDLoc(N), VT, N0);
7879   }
7880 
7881   return SDValue();
7882 }
7883 
7884 SDValue DAGCombiner::visitCTLZ_ZERO_UNDEF(SDNode *N) {
7885   SDValue N0 = N->getOperand(0);
7886   EVT VT = N->getValueType(0);
7887 
7888   // fold (ctlz_zero_undef c1) -> c2
7889   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7890     return DAG.getNode(ISD::CTLZ_ZERO_UNDEF, SDLoc(N), VT, N0);
7891   return SDValue();
7892 }
7893 
7894 SDValue DAGCombiner::visitCTTZ(SDNode *N) {
7895   SDValue N0 = N->getOperand(0);
7896   EVT VT = N->getValueType(0);
7897 
7898   // fold (cttz c1) -> c2
7899   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7900     return DAG.getNode(ISD::CTTZ, SDLoc(N), VT, N0);
7901 
7902   // If the value is known never to be zero, switch to the undef version.
7903   if (!LegalOperations || TLI.isOperationLegal(ISD::CTTZ_ZERO_UNDEF, VT)) {
7904     if (DAG.isKnownNeverZero(N0))
7905       return DAG.getNode(ISD::CTTZ_ZERO_UNDEF, SDLoc(N), VT, N0);
7906   }
7907 
7908   return SDValue();
7909 }
7910 
7911 SDValue DAGCombiner::visitCTTZ_ZERO_UNDEF(SDNode *N) {
7912   SDValue N0 = N->getOperand(0);
7913   EVT VT = N->getValueType(0);
7914 
7915   // fold (cttz_zero_undef c1) -> c2
7916   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7917     return DAG.getNode(ISD::CTTZ_ZERO_UNDEF, SDLoc(N), VT, N0);
7918   return SDValue();
7919 }
7920 
7921 SDValue DAGCombiner::visitCTPOP(SDNode *N) {
7922   SDValue N0 = N->getOperand(0);
7923   EVT VT = N->getValueType(0);
7924 
7925   // fold (ctpop c1) -> c2
7926   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7927     return DAG.getNode(ISD::CTPOP, SDLoc(N), VT, N0);
7928   return SDValue();
7929 }
7930 
7931 // FIXME: This should be checking for no signed zeros on individual operands, as
7932 // well as no nans.
7933 static bool isLegalToCombineMinNumMaxNum(SelectionDAG &DAG, SDValue LHS,
7934                                          SDValue RHS,
7935                                          const TargetLowering &TLI) {
7936   const TargetOptions &Options = DAG.getTarget().Options;
7937   EVT VT = LHS.getValueType();
7938 
7939   return Options.NoSignedZerosFPMath && VT.isFloatingPoint() &&
7940          TLI.isProfitableToCombineMinNumMaxNum(VT) &&
7941          DAG.isKnownNeverNaN(LHS) && DAG.isKnownNeverNaN(RHS);
7942 }
7943 
7944 /// Generate Min/Max node
7945 static SDValue combineMinNumMaxNum(const SDLoc &DL, EVT VT, SDValue LHS,
7946                                    SDValue RHS, SDValue True, SDValue False,
7947                                    ISD::CondCode CC, const TargetLowering &TLI,
7948                                    SelectionDAG &DAG) {
7949   if (!(LHS == True && RHS == False) && !(LHS == False && RHS == True))
7950     return SDValue();
7951 
7952   EVT TransformVT = TLI.getTypeToTransformTo(*DAG.getContext(), VT);
7953   switch (CC) {
7954   case ISD::SETOLT:
7955   case ISD::SETOLE:
7956   case ISD::SETLT:
7957   case ISD::SETLE:
7958   case ISD::SETULT:
7959   case ISD::SETULE: {
7960     // Since it's known never nan to get here already, either fminnum or
7961     // fminnum_ieee are OK. Try the ieee version first, since it's fminnum is
7962     // expanded in terms of it.
7963     unsigned IEEEOpcode = (LHS == True) ? ISD::FMINNUM_IEEE : ISD::FMAXNUM_IEEE;
7964     if (TLI.isOperationLegalOrCustom(IEEEOpcode, VT))
7965       return DAG.getNode(IEEEOpcode, DL, VT, LHS, RHS);
7966 
7967     unsigned Opcode = (LHS == True) ? ISD::FMINNUM : ISD::FMAXNUM;
7968     if (TLI.isOperationLegalOrCustom(Opcode, TransformVT))
7969       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
7970     return SDValue();
7971   }
7972   case ISD::SETOGT:
7973   case ISD::SETOGE:
7974   case ISD::SETGT:
7975   case ISD::SETGE:
7976   case ISD::SETUGT:
7977   case ISD::SETUGE: {
7978     unsigned IEEEOpcode = (LHS == True) ? ISD::FMAXNUM_IEEE : ISD::FMINNUM_IEEE;
7979     if (TLI.isOperationLegalOrCustom(IEEEOpcode, VT))
7980       return DAG.getNode(IEEEOpcode, DL, VT, LHS, RHS);
7981 
7982     unsigned Opcode = (LHS == True) ? ISD::FMAXNUM : ISD::FMINNUM;
7983     if (TLI.isOperationLegalOrCustom(Opcode, TransformVT))
7984       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
7985     return SDValue();
7986   }
7987   default:
7988     return SDValue();
7989   }
7990 }
7991 
7992 SDValue DAGCombiner::foldSelectOfConstants(SDNode *N) {
7993   SDValue Cond = N->getOperand(0);
7994   SDValue N1 = N->getOperand(1);
7995   SDValue N2 = N->getOperand(2);
7996   EVT VT = N->getValueType(0);
7997   EVT CondVT = Cond.getValueType();
7998   SDLoc DL(N);
7999 
8000   if (!VT.isInteger())
8001     return SDValue();
8002 
8003   auto *C1 = dyn_cast<ConstantSDNode>(N1);
8004   auto *C2 = dyn_cast<ConstantSDNode>(N2);
8005   if (!C1 || !C2)
8006     return SDValue();
8007 
8008   // Only do this before legalization to avoid conflicting with target-specific
8009   // transforms in the other direction (create a select from a zext/sext). There
8010   // is also a target-independent combine here in DAGCombiner in the other
8011   // direction for (select Cond, -1, 0) when the condition is not i1.
8012   if (CondVT == MVT::i1 && !LegalOperations) {
8013     if (C1->isNullValue() && C2->isOne()) {
8014       // select Cond, 0, 1 --> zext (!Cond)
8015       SDValue NotCond = DAG.getNOT(DL, Cond, MVT::i1);
8016       if (VT != MVT::i1)
8017         NotCond = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, NotCond);
8018       return NotCond;
8019     }
8020     if (C1->isNullValue() && C2->isAllOnesValue()) {
8021       // select Cond, 0, -1 --> sext (!Cond)
8022       SDValue NotCond = DAG.getNOT(DL, Cond, MVT::i1);
8023       if (VT != MVT::i1)
8024         NotCond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, NotCond);
8025       return NotCond;
8026     }
8027     if (C1->isOne() && C2->isNullValue()) {
8028       // select Cond, 1, 0 --> zext (Cond)
8029       if (VT != MVT::i1)
8030         Cond = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Cond);
8031       return Cond;
8032     }
8033     if (C1->isAllOnesValue() && C2->isNullValue()) {
8034       // select Cond, -1, 0 --> sext (Cond)
8035       if (VT != MVT::i1)
8036         Cond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cond);
8037       return Cond;
8038     }
8039 
8040     // For any constants that differ by 1, we can transform the select into an
8041     // extend and add. Use a target hook because some targets may prefer to
8042     // transform in the other direction.
8043     if (TLI.convertSelectOfConstantsToMath(VT)) {
8044       if (C1->getAPIntValue() - 1 == C2->getAPIntValue()) {
8045         // select Cond, C1, C1-1 --> add (zext Cond), C1-1
8046         if (VT != MVT::i1)
8047           Cond = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Cond);
8048         return DAG.getNode(ISD::ADD, DL, VT, Cond, N2);
8049       }
8050       if (C1->getAPIntValue() + 1 == C2->getAPIntValue()) {
8051         // select Cond, C1, C1+1 --> add (sext Cond), C1+1
8052         if (VT != MVT::i1)
8053           Cond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cond);
8054         return DAG.getNode(ISD::ADD, DL, VT, Cond, N2);
8055       }
8056     }
8057 
8058     return SDValue();
8059   }
8060 
8061   // fold (select Cond, 0, 1) -> (xor Cond, 1)
8062   // We can't do this reliably if integer based booleans have different contents
8063   // to floating point based booleans. This is because we can't tell whether we
8064   // have an integer-based boolean or a floating-point-based boolean unless we
8065   // can find the SETCC that produced it and inspect its operands. This is
8066   // fairly easy if C is the SETCC node, but it can potentially be
8067   // undiscoverable (or not reasonably discoverable). For example, it could be
8068   // in another basic block or it could require searching a complicated
8069   // expression.
8070   if (CondVT.isInteger() &&
8071       TLI.getBooleanContents(/*isVec*/false, /*isFloat*/true) ==
8072           TargetLowering::ZeroOrOneBooleanContent &&
8073       TLI.getBooleanContents(/*isVec*/false, /*isFloat*/false) ==
8074           TargetLowering::ZeroOrOneBooleanContent &&
8075       C1->isNullValue() && C2->isOne()) {
8076     SDValue NotCond =
8077         DAG.getNode(ISD::XOR, DL, CondVT, Cond, DAG.getConstant(1, DL, CondVT));
8078     if (VT.bitsEq(CondVT))
8079       return NotCond;
8080     return DAG.getZExtOrTrunc(NotCond, DL, VT);
8081   }
8082 
8083   return SDValue();
8084 }
8085 
8086 SDValue DAGCombiner::visitSELECT(SDNode *N) {
8087   SDValue N0 = N->getOperand(0);
8088   SDValue N1 = N->getOperand(1);
8089   SDValue N2 = N->getOperand(2);
8090   EVT VT = N->getValueType(0);
8091   EVT VT0 = N0.getValueType();
8092   SDLoc DL(N);
8093   SDNodeFlags Flags = N->getFlags();
8094 
8095   if (SDValue V = DAG.simplifySelect(N0, N1, N2))
8096     return V;
8097 
8098   // fold (select X, X, Y) -> (or X, Y)
8099   // fold (select X, 1, Y) -> (or C, Y)
8100   if (VT == VT0 && VT == MVT::i1 && (N0 == N1 || isOneConstant(N1)))
8101     return DAG.getNode(ISD::OR, DL, VT, N0, N2);
8102 
8103   if (SDValue V = foldSelectOfConstants(N))
8104     return V;
8105 
8106   // fold (select C, 0, X) -> (and (not C), X)
8107   if (VT == VT0 && VT == MVT::i1 && isNullConstant(N1)) {
8108     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
8109     AddToWorklist(NOTNode.getNode());
8110     return DAG.getNode(ISD::AND, DL, VT, NOTNode, N2);
8111   }
8112   // fold (select C, X, 1) -> (or (not C), X)
8113   if (VT == VT0 && VT == MVT::i1 && isOneConstant(N2)) {
8114     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
8115     AddToWorklist(NOTNode.getNode());
8116     return DAG.getNode(ISD::OR, DL, VT, NOTNode, N1);
8117   }
8118   // fold (select X, Y, X) -> (and X, Y)
8119   // fold (select X, Y, 0) -> (and X, Y)
8120   if (VT == VT0 && VT == MVT::i1 && (N0 == N2 || isNullConstant(N2)))
8121     return DAG.getNode(ISD::AND, DL, VT, N0, N1);
8122 
8123   // If we can fold this based on the true/false value, do so.
8124   if (SimplifySelectOps(N, N1, N2))
8125     return SDValue(N, 0); // Don't revisit N.
8126 
8127   if (VT0 == MVT::i1) {
8128     // The code in this block deals with the following 2 equivalences:
8129     //    select(C0|C1, x, y) <=> select(C0, x, select(C1, x, y))
8130     //    select(C0&C1, x, y) <=> select(C0, select(C1, x, y), y)
8131     // The target can specify its preferred form with the
8132     // shouldNormalizeToSelectSequence() callback. However we always transform
8133     // to the right anyway if we find the inner select exists in the DAG anyway
8134     // and we always transform to the left side if we know that we can further
8135     // optimize the combination of the conditions.
8136     bool normalizeToSequence =
8137         TLI.shouldNormalizeToSelectSequence(*DAG.getContext(), VT);
8138     // select (and Cond0, Cond1), X, Y
8139     //   -> select Cond0, (select Cond1, X, Y), Y
8140     if (N0->getOpcode() == ISD::AND && N0->hasOneUse()) {
8141       SDValue Cond0 = N0->getOperand(0);
8142       SDValue Cond1 = N0->getOperand(1);
8143       SDValue InnerSelect =
8144           DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Cond1, N1, N2, Flags);
8145       if (normalizeToSequence || !InnerSelect.use_empty())
8146         return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Cond0,
8147                            InnerSelect, N2, Flags);
8148       // Cleanup on failure.
8149       if (InnerSelect.use_empty())
8150         recursivelyDeleteUnusedNodes(InnerSelect.getNode());
8151     }
8152     // select (or Cond0, Cond1), X, Y -> select Cond0, X, (select Cond1, X, Y)
8153     if (N0->getOpcode() == ISD::OR && N0->hasOneUse()) {
8154       SDValue Cond0 = N0->getOperand(0);
8155       SDValue Cond1 = N0->getOperand(1);
8156       SDValue InnerSelect = DAG.getNode(ISD::SELECT, DL, N1.getValueType(),
8157                                         Cond1, N1, N2, Flags);
8158       if (normalizeToSequence || !InnerSelect.use_empty())
8159         return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Cond0, N1,
8160                            InnerSelect, Flags);
8161       // Cleanup on failure.
8162       if (InnerSelect.use_empty())
8163         recursivelyDeleteUnusedNodes(InnerSelect.getNode());
8164     }
8165 
8166     // select Cond0, (select Cond1, X, Y), Y -> select (and Cond0, Cond1), X, Y
8167     if (N1->getOpcode() == ISD::SELECT && N1->hasOneUse()) {
8168       SDValue N1_0 = N1->getOperand(0);
8169       SDValue N1_1 = N1->getOperand(1);
8170       SDValue N1_2 = N1->getOperand(2);
8171       if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
8172         // Create the actual and node if we can generate good code for it.
8173         if (!normalizeToSequence) {
8174           SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N1_0);
8175           return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), And, N1_1,
8176                              N2, Flags);
8177         }
8178         // Otherwise see if we can optimize the "and" to a better pattern.
8179         if (SDValue Combined = visitANDLike(N0, N1_0, N)) {
8180           return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Combined, N1_1,
8181                              N2, Flags);
8182         }
8183       }
8184     }
8185     // select Cond0, X, (select Cond1, X, Y) -> select (or Cond0, Cond1), X, Y
8186     if (N2->getOpcode() == ISD::SELECT && N2->hasOneUse()) {
8187       SDValue N2_0 = N2->getOperand(0);
8188       SDValue N2_1 = N2->getOperand(1);
8189       SDValue N2_2 = N2->getOperand(2);
8190       if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
8191         // Create the actual or node if we can generate good code for it.
8192         if (!normalizeToSequence) {
8193           SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N2_0);
8194           return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Or, N1,
8195                              N2_2, Flags);
8196         }
8197         // Otherwise see if we can optimize to a better pattern.
8198         if (SDValue Combined = visitORLike(N0, N2_0, N))
8199           return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Combined, N1,
8200                              N2_2, Flags);
8201       }
8202     }
8203   }
8204 
8205   // select (not Cond), N1, N2 -> select Cond, N2, N1
8206   if (SDValue F = extractBooleanFlip(N0, TLI)) {
8207     SDValue SelectOp = DAG.getSelect(DL, VT, F, N2, N1);
8208     SelectOp->setFlags(Flags);
8209     return SelectOp;
8210   }
8211 
8212   // Fold selects based on a setcc into other things, such as min/max/abs.
8213   if (N0.getOpcode() == ISD::SETCC) {
8214     SDValue Cond0 = N0.getOperand(0), Cond1 = N0.getOperand(1);
8215     ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
8216 
8217     // select (fcmp lt x, y), x, y -> fminnum x, y
8218     // select (fcmp gt x, y), x, y -> fmaxnum x, y
8219     //
8220     // This is OK if we don't care what happens if either operand is a NaN.
8221     if (N0.hasOneUse() && isLegalToCombineMinNumMaxNum(DAG, N1, N2, TLI))
8222       if (SDValue FMinMax = combineMinNumMaxNum(DL, VT, Cond0, Cond1, N1, N2,
8223                                                 CC, TLI, DAG))
8224         return FMinMax;
8225 
8226     // Use 'unsigned add with overflow' to optimize an unsigned saturating add.
8227     // This is conservatively limited to pre-legal-operations to give targets
8228     // a chance to reverse the transform if they want to do that. Also, it is
8229     // unlikely that the pattern would be formed late, so it's probably not
8230     // worth going through the other checks.
8231     if (!LegalOperations && TLI.isOperationLegalOrCustom(ISD::UADDO, VT) &&
8232         CC == ISD::SETUGT && N0.hasOneUse() && isAllOnesConstant(N1) &&
8233         N2.getOpcode() == ISD::ADD && Cond0 == N2.getOperand(0)) {
8234       auto *C = dyn_cast<ConstantSDNode>(N2.getOperand(1));
8235       auto *NotC = dyn_cast<ConstantSDNode>(Cond1);
8236       if (C && NotC && C->getAPIntValue() == ~NotC->getAPIntValue()) {
8237         // select (setcc Cond0, ~C, ugt), -1, (add Cond0, C) -->
8238         // uaddo Cond0, C; select uaddo.1, -1, uaddo.0
8239         //
8240         // The IR equivalent of this transform would have this form:
8241         //   %a = add %x, C
8242         //   %c = icmp ugt %x, ~C
8243         //   %r = select %c, -1, %a
8244         //   =>
8245         //   %u = call {iN,i1} llvm.uadd.with.overflow(%x, C)
8246         //   %u0 = extractvalue %u, 0
8247         //   %u1 = extractvalue %u, 1
8248         //   %r = select %u1, -1, %u0
8249         SDVTList VTs = DAG.getVTList(VT, VT0);
8250         SDValue UAO = DAG.getNode(ISD::UADDO, DL, VTs, Cond0, N2.getOperand(1));
8251         return DAG.getSelect(DL, VT, UAO.getValue(1), N1, UAO.getValue(0));
8252       }
8253     }
8254 
8255     if (TLI.isOperationLegal(ISD::SELECT_CC, VT) ||
8256         (!LegalOperations &&
8257          TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT))) {
8258       // Any flags available in a select/setcc fold will be on the setcc as they
8259       // migrated from fcmp
8260       Flags = N0.getNode()->getFlags();
8261       SDValue SelectNode = DAG.getNode(ISD::SELECT_CC, DL, VT, Cond0, Cond1, N1,
8262                                        N2, N0.getOperand(2));
8263       SelectNode->setFlags(Flags);
8264       return SelectNode;
8265     }
8266 
8267     return SimplifySelect(DL, N0, N1, N2);
8268   }
8269 
8270   return SDValue();
8271 }
8272 
8273 static
8274 std::pair<SDValue, SDValue> SplitVSETCC(const SDNode *N, SelectionDAG &DAG) {
8275   SDLoc DL(N);
8276   EVT LoVT, HiVT;
8277   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(N->getValueType(0));
8278 
8279   // Split the inputs.
8280   SDValue Lo, Hi, LL, LH, RL, RH;
8281   std::tie(LL, LH) = DAG.SplitVectorOperand(N, 0);
8282   std::tie(RL, RH) = DAG.SplitVectorOperand(N, 1);
8283 
8284   Lo = DAG.getNode(N->getOpcode(), DL, LoVT, LL, RL, N->getOperand(2));
8285   Hi = DAG.getNode(N->getOpcode(), DL, HiVT, LH, RH, N->getOperand(2));
8286 
8287   return std::make_pair(Lo, Hi);
8288 }
8289 
8290 // This function assumes all the vselect's arguments are CONCAT_VECTOR
8291 // nodes and that the condition is a BV of ConstantSDNodes (or undefs).
8292 static SDValue ConvertSelectToConcatVector(SDNode *N, SelectionDAG &DAG) {
8293   SDLoc DL(N);
8294   SDValue Cond = N->getOperand(0);
8295   SDValue LHS = N->getOperand(1);
8296   SDValue RHS = N->getOperand(2);
8297   EVT VT = N->getValueType(0);
8298   int NumElems = VT.getVectorNumElements();
8299   assert(LHS.getOpcode() == ISD::CONCAT_VECTORS &&
8300          RHS.getOpcode() == ISD::CONCAT_VECTORS &&
8301          Cond.getOpcode() == ISD::BUILD_VECTOR);
8302 
8303   // CONCAT_VECTOR can take an arbitrary number of arguments. We only care about
8304   // binary ones here.
8305   if (LHS->getNumOperands() != 2 || RHS->getNumOperands() != 2)
8306     return SDValue();
8307 
8308   // We're sure we have an even number of elements due to the
8309   // concat_vectors we have as arguments to vselect.
8310   // Skip BV elements until we find one that's not an UNDEF
8311   // After we find an UNDEF element, keep looping until we get to half the
8312   // length of the BV and see if all the non-undef nodes are the same.
8313   ConstantSDNode *BottomHalf = nullptr;
8314   for (int i = 0; i < NumElems / 2; ++i) {
8315     if (Cond->getOperand(i)->isUndef())
8316       continue;
8317 
8318     if (BottomHalf == nullptr)
8319       BottomHalf = cast<ConstantSDNode>(Cond.getOperand(i));
8320     else if (Cond->getOperand(i).getNode() != BottomHalf)
8321       return SDValue();
8322   }
8323 
8324   // Do the same for the second half of the BuildVector
8325   ConstantSDNode *TopHalf = nullptr;
8326   for (int i = NumElems / 2; i < NumElems; ++i) {
8327     if (Cond->getOperand(i)->isUndef())
8328       continue;
8329 
8330     if (TopHalf == nullptr)
8331       TopHalf = cast<ConstantSDNode>(Cond.getOperand(i));
8332     else if (Cond->getOperand(i).getNode() != TopHalf)
8333       return SDValue();
8334   }
8335 
8336   assert(TopHalf && BottomHalf &&
8337          "One half of the selector was all UNDEFs and the other was all the "
8338          "same value. This should have been addressed before this function.");
8339   return DAG.getNode(
8340       ISD::CONCAT_VECTORS, DL, VT,
8341       BottomHalf->isNullValue() ? RHS->getOperand(0) : LHS->getOperand(0),
8342       TopHalf->isNullValue() ? RHS->getOperand(1) : LHS->getOperand(1));
8343 }
8344 
8345 SDValue DAGCombiner::visitMSCATTER(SDNode *N) {
8346   MaskedScatterSDNode *MSC = cast<MaskedScatterSDNode>(N);
8347   SDValue Mask = MSC->getMask();
8348   SDValue Data = MSC->getValue();
8349   SDValue Chain = MSC->getChain();
8350   SDLoc DL(N);
8351 
8352   // Zap scatters with a zero mask.
8353   if (ISD::isBuildVectorAllZeros(Mask.getNode()))
8354     return Chain;
8355 
8356   if (Level >= AfterLegalizeTypes)
8357     return SDValue();
8358 
8359   // If the MSCATTER data type requires splitting and the mask is provided by a
8360   // SETCC, then split both nodes and its operands before legalization. This
8361   // prevents the type legalizer from unrolling SETCC into scalar comparisons
8362   // and enables future optimizations (e.g. min/max pattern matching on X86).
8363   if (Mask.getOpcode() != ISD::SETCC)
8364     return SDValue();
8365 
8366   // Check if any splitting is required.
8367   if (TLI.getTypeAction(*DAG.getContext(), Data.getValueType()) !=
8368       TargetLowering::TypeSplitVector)
8369     return SDValue();
8370   SDValue MaskLo, MaskHi;
8371   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
8372 
8373   EVT LoVT, HiVT;
8374   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MSC->getValueType(0));
8375 
8376   EVT MemoryVT = MSC->getMemoryVT();
8377   unsigned Alignment = MSC->getOriginalAlignment();
8378 
8379   EVT LoMemVT, HiMemVT;
8380   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
8381 
8382   SDValue DataLo, DataHi;
8383   std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
8384 
8385   SDValue Scale = MSC->getScale();
8386   SDValue BasePtr = MSC->getBasePtr();
8387   SDValue IndexLo, IndexHi;
8388   std::tie(IndexLo, IndexHi) = DAG.SplitVector(MSC->getIndex(), DL);
8389 
8390   MachineMemOperand *MMO = DAG.getMachineFunction().
8391     getMachineMemOperand(MSC->getPointerInfo(),
8392                           MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
8393                           Alignment, MSC->getAAInfo(), MSC->getRanges());
8394 
8395   SDValue OpsLo[] = { Chain, DataLo, MaskLo, BasePtr, IndexLo, Scale };
8396   SDValue Lo = DAG.getMaskedScatter(DAG.getVTList(MVT::Other),
8397                                     DataLo.getValueType(), DL, OpsLo, MMO);
8398 
8399   // The order of the Scatter operation after split is well defined. The "Hi"
8400   // part comes after the "Lo". So these two operations should be chained one
8401   // after another.
8402   SDValue OpsHi[] = { Lo, DataHi, MaskHi, BasePtr, IndexHi, Scale };
8403   return DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataHi.getValueType(),
8404                               DL, OpsHi, MMO);
8405 }
8406 
8407 SDValue DAGCombiner::visitMSTORE(SDNode *N) {
8408   MaskedStoreSDNode *MST = cast<MaskedStoreSDNode>(N);
8409   SDValue Mask = MST->getMask();
8410   SDValue Data = MST->getValue();
8411   SDValue Chain = MST->getChain();
8412   EVT VT = Data.getValueType();
8413   SDLoc DL(N);
8414 
8415   // Zap masked stores with a zero mask.
8416   if (ISD::isBuildVectorAllZeros(Mask.getNode()))
8417     return Chain;
8418 
8419   if (Level >= AfterLegalizeTypes)
8420     return SDValue();
8421 
8422   // If the MSTORE data type requires splitting and the mask is provided by a
8423   // SETCC, then split both nodes and its operands before legalization. This
8424   // prevents the type legalizer from unrolling SETCC into scalar comparisons
8425   // and enables future optimizations (e.g. min/max pattern matching on X86).
8426   if (Mask.getOpcode() == ISD::SETCC) {
8427     // Check if any splitting is required.
8428     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
8429         TargetLowering::TypeSplitVector)
8430       return SDValue();
8431 
8432     SDValue MaskLo, MaskHi, Lo, Hi;
8433     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
8434 
8435     SDValue Ptr   = MST->getBasePtr();
8436 
8437     EVT MemoryVT = MST->getMemoryVT();
8438     unsigned Alignment = MST->getOriginalAlignment();
8439 
8440     EVT LoMemVT, HiMemVT;
8441     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
8442 
8443     SDValue DataLo, DataHi;
8444     std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
8445 
8446     MachineMemOperand *MMO = DAG.getMachineFunction().
8447       getMachineMemOperand(MST->getPointerInfo(),
8448                            MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
8449                            Alignment, MST->getAAInfo(), MST->getRanges());
8450 
8451     Lo = DAG.getMaskedStore(Chain, DL, DataLo, Ptr, MaskLo, LoMemVT, MMO,
8452                             MST->isTruncatingStore(),
8453                             MST->isCompressingStore());
8454 
8455     Ptr = TLI.IncrementMemoryAddress(Ptr, MaskLo, DL, LoMemVT, DAG,
8456                                      MST->isCompressingStore());
8457     unsigned HiOffset = LoMemVT.getStoreSize();
8458 
8459     MMO = DAG.getMachineFunction().getMachineMemOperand(
8460         MST->getPointerInfo().getWithOffset(HiOffset),
8461         MachineMemOperand::MOStore, HiMemVT.getStoreSize(), Alignment,
8462         MST->getAAInfo(), MST->getRanges());
8463 
8464     Hi = DAG.getMaskedStore(Chain, DL, DataHi, Ptr, MaskHi, HiMemVT, MMO,
8465                             MST->isTruncatingStore(),
8466                             MST->isCompressingStore());
8467 
8468     AddToWorklist(Lo.getNode());
8469     AddToWorklist(Hi.getNode());
8470 
8471     return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi);
8472   }
8473   return SDValue();
8474 }
8475 
8476 SDValue DAGCombiner::visitMGATHER(SDNode *N) {
8477   MaskedGatherSDNode *MGT = cast<MaskedGatherSDNode>(N);
8478   SDValue Mask = MGT->getMask();
8479   SDLoc DL(N);
8480 
8481   // Zap gathers with a zero mask.
8482   if (ISD::isBuildVectorAllZeros(Mask.getNode()))
8483     return CombineTo(N, MGT->getPassThru(), MGT->getChain());
8484 
8485   if (Level >= AfterLegalizeTypes)
8486     return SDValue();
8487 
8488   // If the MGATHER result requires splitting and the mask is provided by a
8489   // SETCC, then split both nodes and its operands before legalization. This
8490   // prevents the type legalizer from unrolling SETCC into scalar comparisons
8491   // and enables future optimizations (e.g. min/max pattern matching on X86).
8492 
8493   if (Mask.getOpcode() != ISD::SETCC)
8494     return SDValue();
8495 
8496   EVT VT = N->getValueType(0);
8497 
8498   // Check if any splitting is required.
8499   if (TLI.getTypeAction(*DAG.getContext(), VT) !=
8500       TargetLowering::TypeSplitVector)
8501     return SDValue();
8502 
8503   SDValue MaskLo, MaskHi, Lo, Hi;
8504   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
8505 
8506   SDValue PassThru = MGT->getPassThru();
8507   SDValue PassThruLo, PassThruHi;
8508   std::tie(PassThruLo, PassThruHi) = DAG.SplitVector(PassThru, DL);
8509 
8510   EVT LoVT, HiVT;
8511   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
8512 
8513   SDValue Chain = MGT->getChain();
8514   EVT MemoryVT = MGT->getMemoryVT();
8515   unsigned Alignment = MGT->getOriginalAlignment();
8516 
8517   EVT LoMemVT, HiMemVT;
8518   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
8519 
8520   SDValue Scale = MGT->getScale();
8521   SDValue BasePtr = MGT->getBasePtr();
8522   SDValue Index = MGT->getIndex();
8523   SDValue IndexLo, IndexHi;
8524   std::tie(IndexLo, IndexHi) = DAG.SplitVector(Index, DL);
8525 
8526   MachineMemOperand *MMO = DAG.getMachineFunction().
8527     getMachineMemOperand(MGT->getPointerInfo(),
8528                           MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
8529                           Alignment, MGT->getAAInfo(), MGT->getRanges());
8530 
8531   SDValue OpsLo[] = { Chain, PassThruLo, MaskLo, BasePtr, IndexLo, Scale };
8532   Lo = DAG.getMaskedGather(DAG.getVTList(LoVT, MVT::Other), LoVT, DL, OpsLo,
8533                            MMO);
8534 
8535   SDValue OpsHi[] = { Chain, PassThruHi, MaskHi, BasePtr, IndexHi, Scale };
8536   Hi = DAG.getMaskedGather(DAG.getVTList(HiVT, MVT::Other), HiVT, DL, OpsHi,
8537                            MMO);
8538 
8539   AddToWorklist(Lo.getNode());
8540   AddToWorklist(Hi.getNode());
8541 
8542   // Build a factor node to remember that this load is independent of the
8543   // other one.
8544   Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
8545                       Hi.getValue(1));
8546 
8547   // Legalized the chain result - switch anything that used the old chain to
8548   // use the new one.
8549   DAG.ReplaceAllUsesOfValueWith(SDValue(MGT, 1), Chain);
8550 
8551   SDValue GatherRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
8552 
8553   SDValue RetOps[] = { GatherRes, Chain };
8554   return DAG.getMergeValues(RetOps, DL);
8555 }
8556 
8557 SDValue DAGCombiner::visitMLOAD(SDNode *N) {
8558   MaskedLoadSDNode *MLD = cast<MaskedLoadSDNode>(N);
8559   SDValue Mask = MLD->getMask();
8560   SDLoc DL(N);
8561 
8562   // Zap masked loads with a zero mask.
8563   if (ISD::isBuildVectorAllZeros(Mask.getNode()))
8564     return CombineTo(N, MLD->getPassThru(), MLD->getChain());
8565 
8566   if (Level >= AfterLegalizeTypes)
8567     return SDValue();
8568 
8569   // If the MLOAD result requires splitting and the mask is provided by a
8570   // SETCC, then split both nodes and its operands before legalization. This
8571   // prevents the type legalizer from unrolling SETCC into scalar comparisons
8572   // and enables future optimizations (e.g. min/max pattern matching on X86).
8573   if (Mask.getOpcode() == ISD::SETCC) {
8574     EVT VT = N->getValueType(0);
8575 
8576     // Check if any splitting is required.
8577     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
8578         TargetLowering::TypeSplitVector)
8579       return SDValue();
8580 
8581     SDValue MaskLo, MaskHi, Lo, Hi;
8582     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
8583 
8584     SDValue PassThru = MLD->getPassThru();
8585     SDValue PassThruLo, PassThruHi;
8586     std::tie(PassThruLo, PassThruHi) = DAG.SplitVector(PassThru, DL);
8587 
8588     EVT LoVT, HiVT;
8589     std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MLD->getValueType(0));
8590 
8591     SDValue Chain = MLD->getChain();
8592     SDValue Ptr   = MLD->getBasePtr();
8593     EVT MemoryVT = MLD->getMemoryVT();
8594     unsigned Alignment = MLD->getOriginalAlignment();
8595 
8596     EVT LoMemVT, HiMemVT;
8597     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
8598 
8599     MachineMemOperand *MMO = DAG.getMachineFunction().
8600     getMachineMemOperand(MLD->getPointerInfo(),
8601                          MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
8602                          Alignment, MLD->getAAInfo(), MLD->getRanges());
8603 
8604     Lo = DAG.getMaskedLoad(LoVT, DL, Chain, Ptr, MaskLo, PassThruLo, LoMemVT,
8605                            MMO, ISD::NON_EXTLOAD, MLD->isExpandingLoad());
8606 
8607     Ptr = TLI.IncrementMemoryAddress(Ptr, MaskLo, DL, LoMemVT, DAG,
8608                                      MLD->isExpandingLoad());
8609     unsigned HiOffset = LoMemVT.getStoreSize();
8610 
8611     MMO = DAG.getMachineFunction().getMachineMemOperand(
8612         MLD->getPointerInfo().getWithOffset(HiOffset),
8613         MachineMemOperand::MOLoad, HiMemVT.getStoreSize(), Alignment,
8614         MLD->getAAInfo(), MLD->getRanges());
8615 
8616     Hi = DAG.getMaskedLoad(HiVT, DL, Chain, Ptr, MaskHi, PassThruHi, HiMemVT,
8617                            MMO, ISD::NON_EXTLOAD, MLD->isExpandingLoad());
8618 
8619     AddToWorklist(Lo.getNode());
8620     AddToWorklist(Hi.getNode());
8621 
8622     // Build a factor node to remember that this load is independent of the
8623     // other one.
8624     Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
8625                         Hi.getValue(1));
8626 
8627     // Legalized the chain result - switch anything that used the old chain to
8628     // use the new one.
8629     DAG.ReplaceAllUsesOfValueWith(SDValue(MLD, 1), Chain);
8630 
8631     SDValue LoadRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
8632 
8633     SDValue RetOps[] = { LoadRes, Chain };
8634     return DAG.getMergeValues(RetOps, DL);
8635   }
8636   return SDValue();
8637 }
8638 
8639 /// A vector select of 2 constant vectors can be simplified to math/logic to
8640 /// avoid a variable select instruction and possibly avoid constant loads.
8641 SDValue DAGCombiner::foldVSelectOfConstants(SDNode *N) {
8642   SDValue Cond = N->getOperand(0);
8643   SDValue N1 = N->getOperand(1);
8644   SDValue N2 = N->getOperand(2);
8645   EVT VT = N->getValueType(0);
8646   if (!Cond.hasOneUse() || Cond.getScalarValueSizeInBits() != 1 ||
8647       !TLI.convertSelectOfConstantsToMath(VT) ||
8648       !ISD::isBuildVectorOfConstantSDNodes(N1.getNode()) ||
8649       !ISD::isBuildVectorOfConstantSDNodes(N2.getNode()))
8650     return SDValue();
8651 
8652   // Check if we can use the condition value to increment/decrement a single
8653   // constant value. This simplifies a select to an add and removes a constant
8654   // load/materialization from the general case.
8655   bool AllAddOne = true;
8656   bool AllSubOne = true;
8657   unsigned Elts = VT.getVectorNumElements();
8658   for (unsigned i = 0; i != Elts; ++i) {
8659     SDValue N1Elt = N1.getOperand(i);
8660     SDValue N2Elt = N2.getOperand(i);
8661     if (N1Elt.isUndef() || N2Elt.isUndef())
8662       continue;
8663 
8664     const APInt &C1 = cast<ConstantSDNode>(N1Elt)->getAPIntValue();
8665     const APInt &C2 = cast<ConstantSDNode>(N2Elt)->getAPIntValue();
8666     if (C1 != C2 + 1)
8667       AllAddOne = false;
8668     if (C1 != C2 - 1)
8669       AllSubOne = false;
8670   }
8671 
8672   // Further simplifications for the extra-special cases where the constants are
8673   // all 0 or all -1 should be implemented as folds of these patterns.
8674   SDLoc DL(N);
8675   if (AllAddOne || AllSubOne) {
8676     // vselect <N x i1> Cond, C+1, C --> add (zext Cond), C
8677     // vselect <N x i1> Cond, C-1, C --> add (sext Cond), C
8678     auto ExtendOpcode = AllAddOne ? ISD::ZERO_EXTEND : ISD::SIGN_EXTEND;
8679     SDValue ExtendedCond = DAG.getNode(ExtendOpcode, DL, VT, Cond);
8680     return DAG.getNode(ISD::ADD, DL, VT, ExtendedCond, N2);
8681   }
8682 
8683   // The general case for select-of-constants:
8684   // vselect <N x i1> Cond, C1, C2 --> xor (and (sext Cond), (C1^C2)), C2
8685   // ...but that only makes sense if a vselect is slower than 2 logic ops, so
8686   // leave that to a machine-specific pass.
8687   return SDValue();
8688 }
8689 
8690 SDValue DAGCombiner::visitVSELECT(SDNode *N) {
8691   SDValue N0 = N->getOperand(0);
8692   SDValue N1 = N->getOperand(1);
8693   SDValue N2 = N->getOperand(2);
8694   EVT VT = N->getValueType(0);
8695   SDLoc DL(N);
8696 
8697   if (SDValue V = DAG.simplifySelect(N0, N1, N2))
8698     return V;
8699 
8700   // vselect (not Cond), N1, N2 -> vselect Cond, N2, N1
8701   if (SDValue F = extractBooleanFlip(N0, TLI))
8702     return DAG.getSelect(DL, VT, F, N2, N1);
8703 
8704   // Canonicalize integer abs.
8705   // vselect (setg[te] X,  0),  X, -X ->
8706   // vselect (setgt    X, -1),  X, -X ->
8707   // vselect (setl[te] X,  0), -X,  X ->
8708   // Y = sra (X, size(X)-1); xor (add (X, Y), Y)
8709   if (N0.getOpcode() == ISD::SETCC) {
8710     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
8711     ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
8712     bool isAbs = false;
8713     bool RHSIsAllZeros = ISD::isBuildVectorAllZeros(RHS.getNode());
8714 
8715     if (((RHSIsAllZeros && (CC == ISD::SETGT || CC == ISD::SETGE)) ||
8716          (ISD::isBuildVectorAllOnes(RHS.getNode()) && CC == ISD::SETGT)) &&
8717         N1 == LHS && N2.getOpcode() == ISD::SUB && N1 == N2.getOperand(1))
8718       isAbs = ISD::isBuildVectorAllZeros(N2.getOperand(0).getNode());
8719     else if ((RHSIsAllZeros && (CC == ISD::SETLT || CC == ISD::SETLE)) &&
8720              N2 == LHS && N1.getOpcode() == ISD::SUB && N2 == N1.getOperand(1))
8721       isAbs = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
8722 
8723     if (isAbs) {
8724       EVT VT = LHS.getValueType();
8725       if (TLI.isOperationLegalOrCustom(ISD::ABS, VT))
8726         return DAG.getNode(ISD::ABS, DL, VT, LHS);
8727 
8728       SDValue Shift = DAG.getNode(
8729           ISD::SRA, DL, VT, LHS,
8730           DAG.getConstant(VT.getScalarSizeInBits() - 1, DL, VT));
8731       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, LHS, Shift);
8732       AddToWorklist(Shift.getNode());
8733       AddToWorklist(Add.getNode());
8734       return DAG.getNode(ISD::XOR, DL, VT, Add, Shift);
8735     }
8736 
8737     // vselect x, y (fcmp lt x, y) -> fminnum x, y
8738     // vselect x, y (fcmp gt x, y) -> fmaxnum x, y
8739     //
8740     // This is OK if we don't care about what happens if either operand is a
8741     // NaN.
8742     //
8743     if (N0.hasOneUse() && isLegalToCombineMinNumMaxNum(DAG, N0.getOperand(0),
8744                                                        N0.getOperand(1), TLI)) {
8745       if (SDValue FMinMax = combineMinNumMaxNum(
8746               DL, VT, N0.getOperand(0), N0.getOperand(1), N1, N2, CC, TLI, DAG))
8747         return FMinMax;
8748     }
8749 
8750     // If this select has a condition (setcc) with narrower operands than the
8751     // select, try to widen the compare to match the select width.
8752     // TODO: This should be extended to handle any constant.
8753     // TODO: This could be extended to handle non-loading patterns, but that
8754     //       requires thorough testing to avoid regressions.
8755     if (isNullOrNullSplat(RHS)) {
8756       EVT NarrowVT = LHS.getValueType();
8757       EVT WideVT = N1.getValueType().changeVectorElementTypeToInteger();
8758       EVT SetCCVT = getSetCCResultType(LHS.getValueType());
8759       unsigned SetCCWidth = SetCCVT.getScalarSizeInBits();
8760       unsigned WideWidth = WideVT.getScalarSizeInBits();
8761       bool IsSigned = isSignedIntSetCC(CC);
8762       auto LoadExtOpcode = IsSigned ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
8763       if (LHS.getOpcode() == ISD::LOAD && LHS.hasOneUse() &&
8764           SetCCWidth != 1 && SetCCWidth < WideWidth &&
8765           TLI.isLoadExtLegalOrCustom(LoadExtOpcode, WideVT, NarrowVT) &&
8766           TLI.isOperationLegalOrCustom(ISD::SETCC, WideVT)) {
8767         // Both compare operands can be widened for free. The LHS can use an
8768         // extended load, and the RHS is a constant:
8769         //   vselect (ext (setcc load(X), C)), N1, N2 -->
8770         //   vselect (setcc extload(X), C'), N1, N2
8771         auto ExtOpcode = IsSigned ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
8772         SDValue WideLHS = DAG.getNode(ExtOpcode, DL, WideVT, LHS);
8773         SDValue WideRHS = DAG.getNode(ExtOpcode, DL, WideVT, RHS);
8774         EVT WideSetCCVT = getSetCCResultType(WideVT);
8775         SDValue WideSetCC = DAG.getSetCC(DL, WideSetCCVT, WideLHS, WideRHS, CC);
8776         return DAG.getSelect(DL, N1.getValueType(), WideSetCC, N1, N2);
8777       }
8778     }
8779   }
8780 
8781   if (SimplifySelectOps(N, N1, N2))
8782     return SDValue(N, 0);  // Don't revisit N.
8783 
8784   // Fold (vselect (build_vector all_ones), N1, N2) -> N1
8785   if (ISD::isBuildVectorAllOnes(N0.getNode()))
8786     return N1;
8787   // Fold (vselect (build_vector all_zeros), N1, N2) -> N2
8788   if (ISD::isBuildVectorAllZeros(N0.getNode()))
8789     return N2;
8790 
8791   // The ConvertSelectToConcatVector function is assuming both the above
8792   // checks for (vselect (build_vector all{ones,zeros) ...) have been made
8793   // and addressed.
8794   if (N1.getOpcode() == ISD::CONCAT_VECTORS &&
8795       N2.getOpcode() == ISD::CONCAT_VECTORS &&
8796       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())) {
8797     if (SDValue CV = ConvertSelectToConcatVector(N, DAG))
8798       return CV;
8799   }
8800 
8801   if (SDValue V = foldVSelectOfConstants(N))
8802     return V;
8803 
8804   return SDValue();
8805 }
8806 
8807 SDValue DAGCombiner::visitSELECT_CC(SDNode *N) {
8808   SDValue N0 = N->getOperand(0);
8809   SDValue N1 = N->getOperand(1);
8810   SDValue N2 = N->getOperand(2);
8811   SDValue N3 = N->getOperand(3);
8812   SDValue N4 = N->getOperand(4);
8813   ISD::CondCode CC = cast<CondCodeSDNode>(N4)->get();
8814 
8815   // fold select_cc lhs, rhs, x, x, cc -> x
8816   if (N2 == N3)
8817     return N2;
8818 
8819   // Determine if the condition we're dealing with is constant
8820   if (SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()), N0, N1,
8821                                   CC, SDLoc(N), false)) {
8822     AddToWorklist(SCC.getNode());
8823 
8824     if (ConstantSDNode *SCCC = dyn_cast<ConstantSDNode>(SCC.getNode())) {
8825       if (!SCCC->isNullValue())
8826         return N2;    // cond always true -> true val
8827       else
8828         return N3;    // cond always false -> false val
8829     } else if (SCC->isUndef()) {
8830       // When the condition is UNDEF, just return the first operand. This is
8831       // coherent the DAG creation, no setcc node is created in this case
8832       return N2;
8833     } else if (SCC.getOpcode() == ISD::SETCC) {
8834       // Fold to a simpler select_cc
8835       SDValue SelectOp = DAG.getNode(
8836           ISD::SELECT_CC, SDLoc(N), N2.getValueType(), SCC.getOperand(0),
8837           SCC.getOperand(1), N2, N3, SCC.getOperand(2));
8838       SelectOp->setFlags(SCC->getFlags());
8839       return SelectOp;
8840     }
8841   }
8842 
8843   // If we can fold this based on the true/false value, do so.
8844   if (SimplifySelectOps(N, N2, N3))
8845     return SDValue(N, 0);  // Don't revisit N.
8846 
8847   // fold select_cc into other things, such as min/max/abs
8848   return SimplifySelectCC(SDLoc(N), N0, N1, N2, N3, CC);
8849 }
8850 
8851 SDValue DAGCombiner::visitSETCC(SDNode *N) {
8852   // setcc is very commonly used as an argument to brcond. This pattern
8853   // also lend itself to numerous combines and, as a result, it is desired
8854   // we keep the argument to a brcond as a setcc as much as possible.
8855   bool PreferSetCC =
8856       N->hasOneUse() && N->use_begin()->getOpcode() == ISD::BRCOND;
8857 
8858   SDValue Combined = SimplifySetCC(
8859       N->getValueType(0), N->getOperand(0), N->getOperand(1),
8860       cast<CondCodeSDNode>(N->getOperand(2))->get(), SDLoc(N), !PreferSetCC);
8861 
8862   if (!Combined)
8863     return SDValue();
8864 
8865   // If we prefer to have a setcc, and we don't, we'll try our best to
8866   // recreate one using rebuildSetCC.
8867   if (PreferSetCC && Combined.getOpcode() != ISD::SETCC) {
8868     SDValue NewSetCC = rebuildSetCC(Combined);
8869 
8870     // We don't have anything interesting to combine to.
8871     if (NewSetCC.getNode() == N)
8872       return SDValue();
8873 
8874     if (NewSetCC)
8875       return NewSetCC;
8876   }
8877 
8878   return Combined;
8879 }
8880 
8881 SDValue DAGCombiner::visitSETCCCARRY(SDNode *N) {
8882   SDValue LHS = N->getOperand(0);
8883   SDValue RHS = N->getOperand(1);
8884   SDValue Carry = N->getOperand(2);
8885   SDValue Cond = N->getOperand(3);
8886 
8887   // If Carry is false, fold to a regular SETCC.
8888   if (isNullConstant(Carry))
8889     return DAG.getNode(ISD::SETCC, SDLoc(N), N->getVTList(), LHS, RHS, Cond);
8890 
8891   return SDValue();
8892 }
8893 
8894 /// Try to fold a sext/zext/aext dag node into a ConstantSDNode or
8895 /// a build_vector of constants.
8896 /// This function is called by the DAGCombiner when visiting sext/zext/aext
8897 /// dag nodes (see for example method DAGCombiner::visitSIGN_EXTEND).
8898 /// Vector extends are not folded if operations are legal; this is to
8899 /// avoid introducing illegal build_vector dag nodes.
8900 static SDValue tryToFoldExtendOfConstant(SDNode *N, const TargetLowering &TLI,
8901                                          SelectionDAG &DAG, bool LegalTypes) {
8902   unsigned Opcode = N->getOpcode();
8903   SDValue N0 = N->getOperand(0);
8904   EVT VT = N->getValueType(0);
8905 
8906   assert((Opcode == ISD::SIGN_EXTEND || Opcode == ISD::ZERO_EXTEND ||
8907          Opcode == ISD::ANY_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG ||
8908          Opcode == ISD::ZERO_EXTEND_VECTOR_INREG)
8909          && "Expected EXTEND dag node in input!");
8910 
8911   // fold (sext c1) -> c1
8912   // fold (zext c1) -> c1
8913   // fold (aext c1) -> c1
8914   if (isa<ConstantSDNode>(N0))
8915     return DAG.getNode(Opcode, SDLoc(N), VT, N0);
8916 
8917   // fold (sext (build_vector AllConstants) -> (build_vector AllConstants)
8918   // fold (zext (build_vector AllConstants) -> (build_vector AllConstants)
8919   // fold (aext (build_vector AllConstants) -> (build_vector AllConstants)
8920   EVT SVT = VT.getScalarType();
8921   if (!(VT.isVector() && (!LegalTypes || TLI.isTypeLegal(SVT)) &&
8922       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())))
8923     return SDValue();
8924 
8925   // We can fold this node into a build_vector.
8926   unsigned VTBits = SVT.getSizeInBits();
8927   unsigned EVTBits = N0->getValueType(0).getScalarSizeInBits();
8928   SmallVector<SDValue, 8> Elts;
8929   unsigned NumElts = VT.getVectorNumElements();
8930   SDLoc DL(N);
8931 
8932   // For zero-extensions, UNDEF elements still guarantee to have the upper
8933   // bits set to zero.
8934   bool IsZext =
8935       Opcode == ISD::ZERO_EXTEND || Opcode == ISD::ZERO_EXTEND_VECTOR_INREG;
8936 
8937   for (unsigned i = 0; i != NumElts; ++i) {
8938     SDValue Op = N0.getOperand(i);
8939     if (Op.isUndef()) {
8940       Elts.push_back(IsZext ? DAG.getConstant(0, DL, SVT) : DAG.getUNDEF(SVT));
8941       continue;
8942     }
8943 
8944     SDLoc DL(Op);
8945     // Get the constant value and if needed trunc it to the size of the type.
8946     // Nodes like build_vector might have constants wider than the scalar type.
8947     APInt C = cast<ConstantSDNode>(Op)->getAPIntValue().zextOrTrunc(EVTBits);
8948     if (Opcode == ISD::SIGN_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG)
8949       Elts.push_back(DAG.getConstant(C.sext(VTBits), DL, SVT));
8950     else
8951       Elts.push_back(DAG.getConstant(C.zext(VTBits), DL, SVT));
8952   }
8953 
8954   return DAG.getBuildVector(VT, DL, Elts);
8955 }
8956 
8957 // ExtendUsesToFormExtLoad - Trying to extend uses of a load to enable this:
8958 // "fold ({s|z|a}ext (load x)) -> ({s|z|a}ext (truncate ({s|z|a}extload x)))"
8959 // transformation. Returns true if extension are possible and the above
8960 // mentioned transformation is profitable.
8961 static bool ExtendUsesToFormExtLoad(EVT VT, SDNode *N, SDValue N0,
8962                                     unsigned ExtOpc,
8963                                     SmallVectorImpl<SDNode *> &ExtendNodes,
8964                                     const TargetLowering &TLI) {
8965   bool HasCopyToRegUses = false;
8966   bool isTruncFree = TLI.isTruncateFree(VT, N0.getValueType());
8967   for (SDNode::use_iterator UI = N0.getNode()->use_begin(),
8968                             UE = N0.getNode()->use_end();
8969        UI != UE; ++UI) {
8970     SDNode *User = *UI;
8971     if (User == N)
8972       continue;
8973     if (UI.getUse().getResNo() != N0.getResNo())
8974       continue;
8975     // FIXME: Only extend SETCC N, N and SETCC N, c for now.
8976     if (ExtOpc != ISD::ANY_EXTEND && User->getOpcode() == ISD::SETCC) {
8977       ISD::CondCode CC = cast<CondCodeSDNode>(User->getOperand(2))->get();
8978       if (ExtOpc == ISD::ZERO_EXTEND && ISD::isSignedIntSetCC(CC))
8979         // Sign bits will be lost after a zext.
8980         return false;
8981       bool Add = false;
8982       for (unsigned i = 0; i != 2; ++i) {
8983         SDValue UseOp = User->getOperand(i);
8984         if (UseOp == N0)
8985           continue;
8986         if (!isa<ConstantSDNode>(UseOp))
8987           return false;
8988         Add = true;
8989       }
8990       if (Add)
8991         ExtendNodes.push_back(User);
8992       continue;
8993     }
8994     // If truncates aren't free and there are users we can't
8995     // extend, it isn't worthwhile.
8996     if (!isTruncFree)
8997       return false;
8998     // Remember if this value is live-out.
8999     if (User->getOpcode() == ISD::CopyToReg)
9000       HasCopyToRegUses = true;
9001   }
9002 
9003   if (HasCopyToRegUses) {
9004     bool BothLiveOut = false;
9005     for (SDNode::use_iterator UI = N->use_begin(), UE = N->use_end();
9006          UI != UE; ++UI) {
9007       SDUse &Use = UI.getUse();
9008       if (Use.getResNo() == 0 && Use.getUser()->getOpcode() == ISD::CopyToReg) {
9009         BothLiveOut = true;
9010         break;
9011       }
9012     }
9013     if (BothLiveOut)
9014       // Both unextended and extended values are live out. There had better be
9015       // a good reason for the transformation.
9016       return ExtendNodes.size();
9017   }
9018   return true;
9019 }
9020 
9021 void DAGCombiner::ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs,
9022                                   SDValue OrigLoad, SDValue ExtLoad,
9023                                   ISD::NodeType ExtType) {
9024   // Extend SetCC uses if necessary.
9025   SDLoc DL(ExtLoad);
9026   for (SDNode *SetCC : SetCCs) {
9027     SmallVector<SDValue, 4> Ops;
9028 
9029     for (unsigned j = 0; j != 2; ++j) {
9030       SDValue SOp = SetCC->getOperand(j);
9031       if (SOp == OrigLoad)
9032         Ops.push_back(ExtLoad);
9033       else
9034         Ops.push_back(DAG.getNode(ExtType, DL, ExtLoad->getValueType(0), SOp));
9035     }
9036 
9037     Ops.push_back(SetCC->getOperand(2));
9038     CombineTo(SetCC, DAG.getNode(ISD::SETCC, DL, SetCC->getValueType(0), Ops));
9039   }
9040 }
9041 
9042 // FIXME: Bring more similar combines here, common to sext/zext (maybe aext?).
9043 SDValue DAGCombiner::CombineExtLoad(SDNode *N) {
9044   SDValue N0 = N->getOperand(0);
9045   EVT DstVT = N->getValueType(0);
9046   EVT SrcVT = N0.getValueType();
9047 
9048   assert((N->getOpcode() == ISD::SIGN_EXTEND ||
9049           N->getOpcode() == ISD::ZERO_EXTEND) &&
9050          "Unexpected node type (not an extend)!");
9051 
9052   // fold (sext (load x)) to multiple smaller sextloads; same for zext.
9053   // For example, on a target with legal v4i32, but illegal v8i32, turn:
9054   //   (v8i32 (sext (v8i16 (load x))))
9055   // into:
9056   //   (v8i32 (concat_vectors (v4i32 (sextload x)),
9057   //                          (v4i32 (sextload (x + 16)))))
9058   // Where uses of the original load, i.e.:
9059   //   (v8i16 (load x))
9060   // are replaced with:
9061   //   (v8i16 (truncate
9062   //     (v8i32 (concat_vectors (v4i32 (sextload x)),
9063   //                            (v4i32 (sextload (x + 16)))))))
9064   //
9065   // This combine is only applicable to illegal, but splittable, vectors.
9066   // All legal types, and illegal non-vector types, are handled elsewhere.
9067   // This combine is controlled by TargetLowering::isVectorLoadExtDesirable.
9068   //
9069   if (N0->getOpcode() != ISD::LOAD)
9070     return SDValue();
9071 
9072   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9073 
9074   if (!ISD::isNON_EXTLoad(LN0) || !ISD::isUNINDEXEDLoad(LN0) ||
9075       !N0.hasOneUse() || LN0->isVolatile() || !DstVT.isVector() ||
9076       !DstVT.isPow2VectorType() || !TLI.isVectorLoadExtDesirable(SDValue(N, 0)))
9077     return SDValue();
9078 
9079   SmallVector<SDNode *, 4> SetCCs;
9080   if (!ExtendUsesToFormExtLoad(DstVT, N, N0, N->getOpcode(), SetCCs, TLI))
9081     return SDValue();
9082 
9083   ISD::LoadExtType ExtType =
9084       N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
9085 
9086   // Try to split the vector types to get down to legal types.
9087   EVT SplitSrcVT = SrcVT;
9088   EVT SplitDstVT = DstVT;
9089   while (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT) &&
9090          SplitSrcVT.getVectorNumElements() > 1) {
9091     SplitDstVT = DAG.GetSplitDestVTs(SplitDstVT).first;
9092     SplitSrcVT = DAG.GetSplitDestVTs(SplitSrcVT).first;
9093   }
9094 
9095   if (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT))
9096     return SDValue();
9097 
9098   SDLoc DL(N);
9099   const unsigned NumSplits =
9100       DstVT.getVectorNumElements() / SplitDstVT.getVectorNumElements();
9101   const unsigned Stride = SplitSrcVT.getStoreSize();
9102   SmallVector<SDValue, 4> Loads;
9103   SmallVector<SDValue, 4> Chains;
9104 
9105   SDValue BasePtr = LN0->getBasePtr();
9106   for (unsigned Idx = 0; Idx < NumSplits; Idx++) {
9107     const unsigned Offset = Idx * Stride;
9108     const unsigned Align = MinAlign(LN0->getAlignment(), Offset);
9109 
9110     SDValue SplitLoad = DAG.getExtLoad(
9111         ExtType, SDLoc(LN0), SplitDstVT, LN0->getChain(), BasePtr,
9112         LN0->getPointerInfo().getWithOffset(Offset), SplitSrcVT, Align,
9113         LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
9114 
9115     BasePtr = DAG.getNode(ISD::ADD, DL, BasePtr.getValueType(), BasePtr,
9116                           DAG.getConstant(Stride, DL, BasePtr.getValueType()));
9117 
9118     Loads.push_back(SplitLoad.getValue(0));
9119     Chains.push_back(SplitLoad.getValue(1));
9120   }
9121 
9122   SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
9123   SDValue NewValue = DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Loads);
9124 
9125   // Simplify TF.
9126   AddToWorklist(NewChain.getNode());
9127 
9128   CombineTo(N, NewValue);
9129 
9130   // Replace uses of the original load (before extension)
9131   // with a truncate of the concatenated sextloaded vectors.
9132   SDValue Trunc =
9133       DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), NewValue);
9134   ExtendSetCCUses(SetCCs, N0, NewValue, (ISD::NodeType)N->getOpcode());
9135   CombineTo(N0.getNode(), Trunc, NewChain);
9136   return SDValue(N, 0); // Return N so it doesn't get rechecked!
9137 }
9138 
9139 // fold (zext (and/or/xor (shl/shr (load x), cst), cst)) ->
9140 //      (and/or/xor (shl/shr (zextload x), (zext cst)), (zext cst))
9141 SDValue DAGCombiner::CombineZExtLogicopShiftLoad(SDNode *N) {
9142   assert(N->getOpcode() == ISD::ZERO_EXTEND);
9143   EVT VT = N->getValueType(0);
9144   EVT OrigVT = N->getOperand(0).getValueType();
9145   if (TLI.isZExtFree(OrigVT, VT))
9146     return SDValue();
9147 
9148   // and/or/xor
9149   SDValue N0 = N->getOperand(0);
9150   if (!(N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
9151         N0.getOpcode() == ISD::XOR) ||
9152       N0.getOperand(1).getOpcode() != ISD::Constant ||
9153       (LegalOperations && !TLI.isOperationLegal(N0.getOpcode(), VT)))
9154     return SDValue();
9155 
9156   // shl/shr
9157   SDValue N1 = N0->getOperand(0);
9158   if (!(N1.getOpcode() == ISD::SHL || N1.getOpcode() == ISD::SRL) ||
9159       N1.getOperand(1).getOpcode() != ISD::Constant ||
9160       (LegalOperations && !TLI.isOperationLegal(N1.getOpcode(), VT)))
9161     return SDValue();
9162 
9163   // load
9164   if (!isa<LoadSDNode>(N1.getOperand(0)))
9165     return SDValue();
9166   LoadSDNode *Load = cast<LoadSDNode>(N1.getOperand(0));
9167   EVT MemVT = Load->getMemoryVT();
9168   if (!TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT) ||
9169       Load->getExtensionType() == ISD::SEXTLOAD || Load->isIndexed())
9170     return SDValue();
9171 
9172 
9173   // If the shift op is SHL, the logic op must be AND, otherwise the result
9174   // will be wrong.
9175   if (N1.getOpcode() == ISD::SHL && N0.getOpcode() != ISD::AND)
9176     return SDValue();
9177 
9178   if (!N0.hasOneUse() || !N1.hasOneUse())
9179     return SDValue();
9180 
9181   SmallVector<SDNode*, 4> SetCCs;
9182   if (!ExtendUsesToFormExtLoad(VT, N1.getNode(), N1.getOperand(0),
9183                                ISD::ZERO_EXTEND, SetCCs, TLI))
9184     return SDValue();
9185 
9186   // Actually do the transformation.
9187   SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(Load), VT,
9188                                    Load->getChain(), Load->getBasePtr(),
9189                                    Load->getMemoryVT(), Load->getMemOperand());
9190 
9191   SDLoc DL1(N1);
9192   SDValue Shift = DAG.getNode(N1.getOpcode(), DL1, VT, ExtLoad,
9193                               N1.getOperand(1));
9194 
9195   APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
9196   Mask = Mask.zext(VT.getSizeInBits());
9197   SDLoc DL0(N0);
9198   SDValue And = DAG.getNode(N0.getOpcode(), DL0, VT, Shift,
9199                             DAG.getConstant(Mask, DL0, VT));
9200 
9201   ExtendSetCCUses(SetCCs, N1.getOperand(0), ExtLoad, ISD::ZERO_EXTEND);
9202   CombineTo(N, And);
9203   if (SDValue(Load, 0).hasOneUse()) {
9204     DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), ExtLoad.getValue(1));
9205   } else {
9206     SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(Load),
9207                                 Load->getValueType(0), ExtLoad);
9208     CombineTo(Load, Trunc, ExtLoad.getValue(1));
9209   }
9210 
9211   // N0 is dead at this point.
9212   recursivelyDeleteUnusedNodes(N0.getNode());
9213 
9214   return SDValue(N,0); // Return N so it doesn't get rechecked!
9215 }
9216 
9217 /// If we're narrowing or widening the result of a vector select and the final
9218 /// size is the same size as a setcc (compare) feeding the select, then try to
9219 /// apply the cast operation to the select's operands because matching vector
9220 /// sizes for a select condition and other operands should be more efficient.
9221 SDValue DAGCombiner::matchVSelectOpSizesWithSetCC(SDNode *Cast) {
9222   unsigned CastOpcode = Cast->getOpcode();
9223   assert((CastOpcode == ISD::SIGN_EXTEND || CastOpcode == ISD::ZERO_EXTEND ||
9224           CastOpcode == ISD::TRUNCATE || CastOpcode == ISD::FP_EXTEND ||
9225           CastOpcode == ISD::FP_ROUND) &&
9226          "Unexpected opcode for vector select narrowing/widening");
9227 
9228   // We only do this transform before legal ops because the pattern may be
9229   // obfuscated by target-specific operations after legalization. Do not create
9230   // an illegal select op, however, because that may be difficult to lower.
9231   EVT VT = Cast->getValueType(0);
9232   if (LegalOperations || !TLI.isOperationLegalOrCustom(ISD::VSELECT, VT))
9233     return SDValue();
9234 
9235   SDValue VSel = Cast->getOperand(0);
9236   if (VSel.getOpcode() != ISD::VSELECT || !VSel.hasOneUse() ||
9237       VSel.getOperand(0).getOpcode() != ISD::SETCC)
9238     return SDValue();
9239 
9240   // Does the setcc have the same vector size as the casted select?
9241   SDValue SetCC = VSel.getOperand(0);
9242   EVT SetCCVT = getSetCCResultType(SetCC.getOperand(0).getValueType());
9243   if (SetCCVT.getSizeInBits() != VT.getSizeInBits())
9244     return SDValue();
9245 
9246   // cast (vsel (setcc X), A, B) --> vsel (setcc X), (cast A), (cast B)
9247   SDValue A = VSel.getOperand(1);
9248   SDValue B = VSel.getOperand(2);
9249   SDValue CastA, CastB;
9250   SDLoc DL(Cast);
9251   if (CastOpcode == ISD::FP_ROUND) {
9252     // FP_ROUND (fptrunc) has an extra flag operand to pass along.
9253     CastA = DAG.getNode(CastOpcode, DL, VT, A, Cast->getOperand(1));
9254     CastB = DAG.getNode(CastOpcode, DL, VT, B, Cast->getOperand(1));
9255   } else {
9256     CastA = DAG.getNode(CastOpcode, DL, VT, A);
9257     CastB = DAG.getNode(CastOpcode, DL, VT, B);
9258   }
9259   return DAG.getNode(ISD::VSELECT, DL, VT, SetCC, CastA, CastB);
9260 }
9261 
9262 // fold ([s|z]ext ([s|z]extload x)) -> ([s|z]ext (truncate ([s|z]extload x)))
9263 // fold ([s|z]ext (     extload x)) -> ([s|z]ext (truncate ([s|z]extload x)))
9264 static SDValue tryToFoldExtOfExtload(SelectionDAG &DAG, DAGCombiner &Combiner,
9265                                      const TargetLowering &TLI, EVT VT,
9266                                      bool LegalOperations, SDNode *N,
9267                                      SDValue N0, ISD::LoadExtType ExtLoadType) {
9268   SDNode *N0Node = N0.getNode();
9269   bool isAExtLoad = (ExtLoadType == ISD::SEXTLOAD) ? ISD::isSEXTLoad(N0Node)
9270                                                    : ISD::isZEXTLoad(N0Node);
9271   if ((!isAExtLoad && !ISD::isEXTLoad(N0Node)) ||
9272       !ISD::isUNINDEXEDLoad(N0Node) || !N0.hasOneUse())
9273     return SDValue();
9274 
9275   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9276   EVT MemVT = LN0->getMemoryVT();
9277   if ((LegalOperations || LN0->isVolatile() || VT.isVector()) &&
9278       !TLI.isLoadExtLegal(ExtLoadType, VT, MemVT))
9279     return SDValue();
9280 
9281   SDValue ExtLoad =
9282       DAG.getExtLoad(ExtLoadType, SDLoc(LN0), VT, LN0->getChain(),
9283                      LN0->getBasePtr(), MemVT, LN0->getMemOperand());
9284   Combiner.CombineTo(N, ExtLoad);
9285   DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1));
9286   if (LN0->use_empty())
9287     Combiner.recursivelyDeleteUnusedNodes(LN0);
9288   return SDValue(N, 0); // Return N so it doesn't get rechecked!
9289 }
9290 
9291 // fold ([s|z]ext (load x)) -> ([s|z]ext (truncate ([s|z]extload x)))
9292 // Only generate vector extloads when 1) they're legal, and 2) they are
9293 // deemed desirable by the target.
9294 static SDValue tryToFoldExtOfLoad(SelectionDAG &DAG, DAGCombiner &Combiner,
9295                                   const TargetLowering &TLI, EVT VT,
9296                                   bool LegalOperations, SDNode *N, SDValue N0,
9297                                   ISD::LoadExtType ExtLoadType,
9298                                   ISD::NodeType ExtOpc) {
9299   if (!ISD::isNON_EXTLoad(N0.getNode()) ||
9300       !ISD::isUNINDEXEDLoad(N0.getNode()) ||
9301       ((LegalOperations || VT.isVector() ||
9302         cast<LoadSDNode>(N0)->isVolatile()) &&
9303        !TLI.isLoadExtLegal(ExtLoadType, VT, N0.getValueType())))
9304     return {};
9305 
9306   bool DoXform = true;
9307   SmallVector<SDNode *, 4> SetCCs;
9308   if (!N0.hasOneUse())
9309     DoXform = ExtendUsesToFormExtLoad(VT, N, N0, ExtOpc, SetCCs, TLI);
9310   if (VT.isVector())
9311     DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0));
9312   if (!DoXform)
9313     return {};
9314 
9315   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9316   SDValue ExtLoad = DAG.getExtLoad(ExtLoadType, SDLoc(LN0), VT, LN0->getChain(),
9317                                    LN0->getBasePtr(), N0.getValueType(),
9318                                    LN0->getMemOperand());
9319   Combiner.ExtendSetCCUses(SetCCs, N0, ExtLoad, ExtOpc);
9320   // If the load value is used only by N, replace it via CombineTo N.
9321   bool NoReplaceTrunc = SDValue(LN0, 0).hasOneUse();
9322   Combiner.CombineTo(N, ExtLoad);
9323   if (NoReplaceTrunc) {
9324     DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1));
9325     Combiner.recursivelyDeleteUnusedNodes(LN0);
9326   } else {
9327     SDValue Trunc =
9328         DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), ExtLoad);
9329     Combiner.CombineTo(LN0, Trunc, ExtLoad.getValue(1));
9330   }
9331   return SDValue(N, 0); // Return N so it doesn't get rechecked!
9332 }
9333 
9334 static SDValue foldExtendedSignBitTest(SDNode *N, SelectionDAG &DAG,
9335                                        bool LegalOperations) {
9336   assert((N->getOpcode() == ISD::SIGN_EXTEND ||
9337           N->getOpcode() == ISD::ZERO_EXTEND) && "Expected sext or zext");
9338 
9339   SDValue SetCC = N->getOperand(0);
9340   if (LegalOperations || SetCC.getOpcode() != ISD::SETCC ||
9341       !SetCC.hasOneUse() || SetCC.getValueType() != MVT::i1)
9342     return SDValue();
9343 
9344   SDValue X = SetCC.getOperand(0);
9345   SDValue Ones = SetCC.getOperand(1);
9346   ISD::CondCode CC = cast<CondCodeSDNode>(SetCC.getOperand(2))->get();
9347   EVT VT = N->getValueType(0);
9348   EVT XVT = X.getValueType();
9349   // setge X, C is canonicalized to setgt, so we do not need to match that
9350   // pattern. The setlt sibling is folded in SimplifySelectCC() because it does
9351   // not require the 'not' op.
9352   if (CC == ISD::SETGT && isAllOnesConstant(Ones) && VT == XVT) {
9353     // Invert and smear/shift the sign bit:
9354     // sext i1 (setgt iN X, -1) --> sra (not X), (N - 1)
9355     // zext i1 (setgt iN X, -1) --> srl (not X), (N - 1)
9356     SDLoc DL(N);
9357     SDValue NotX = DAG.getNOT(DL, X, VT);
9358     SDValue ShiftAmount = DAG.getConstant(VT.getSizeInBits() - 1, DL, VT);
9359     auto ShiftOpcode = N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SRA : ISD::SRL;
9360     return DAG.getNode(ShiftOpcode, DL, VT, NotX, ShiftAmount);
9361   }
9362   return SDValue();
9363 }
9364 
9365 SDValue DAGCombiner::visitSIGN_EXTEND(SDNode *N) {
9366   SDValue N0 = N->getOperand(0);
9367   EVT VT = N->getValueType(0);
9368   SDLoc DL(N);
9369 
9370   if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes))
9371     return Res;
9372 
9373   // fold (sext (sext x)) -> (sext x)
9374   // fold (sext (aext x)) -> (sext x)
9375   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
9376     return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, N0.getOperand(0));
9377 
9378   if (N0.getOpcode() == ISD::TRUNCATE) {
9379     // fold (sext (truncate (load x))) -> (sext (smaller load x))
9380     // fold (sext (truncate (srl (load x), c))) -> (sext (smaller load (x+c/n)))
9381     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
9382       SDNode *oye = N0.getOperand(0).getNode();
9383       if (NarrowLoad.getNode() != N0.getNode()) {
9384         CombineTo(N0.getNode(), NarrowLoad);
9385         // CombineTo deleted the truncate, if needed, but not what's under it.
9386         AddToWorklist(oye);
9387       }
9388       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9389     }
9390 
9391     // See if the value being truncated is already sign extended.  If so, just
9392     // eliminate the trunc/sext pair.
9393     SDValue Op = N0.getOperand(0);
9394     unsigned OpBits   = Op.getScalarValueSizeInBits();
9395     unsigned MidBits  = N0.getScalarValueSizeInBits();
9396     unsigned DestBits = VT.getScalarSizeInBits();
9397     unsigned NumSignBits = DAG.ComputeNumSignBits(Op);
9398 
9399     if (OpBits == DestBits) {
9400       // Op is i32, Mid is i8, and Dest is i32.  If Op has more than 24 sign
9401       // bits, it is already ready.
9402       if (NumSignBits > DestBits-MidBits)
9403         return Op;
9404     } else if (OpBits < DestBits) {
9405       // Op is i32, Mid is i8, and Dest is i64.  If Op has more than 24 sign
9406       // bits, just sext from i32.
9407       if (NumSignBits > OpBits-MidBits)
9408         return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Op);
9409     } else {
9410       // Op is i64, Mid is i8, and Dest is i32.  If Op has more than 56 sign
9411       // bits, just truncate to i32.
9412       if (NumSignBits > OpBits-MidBits)
9413         return DAG.getNode(ISD::TRUNCATE, DL, VT, Op);
9414     }
9415 
9416     // fold (sext (truncate x)) -> (sextinreg x).
9417     if (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG,
9418                                                  N0.getValueType())) {
9419       if (OpBits < DestBits)
9420         Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N0), VT, Op);
9421       else if (OpBits > DestBits)
9422         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), VT, Op);
9423       return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, Op,
9424                          DAG.getValueType(N0.getValueType()));
9425     }
9426   }
9427 
9428   // Try to simplify (sext (load x)).
9429   if (SDValue foldedExt =
9430           tryToFoldExtOfLoad(DAG, *this, TLI, VT, LegalOperations, N, N0,
9431                              ISD::SEXTLOAD, ISD::SIGN_EXTEND))
9432     return foldedExt;
9433 
9434   // fold (sext (load x)) to multiple smaller sextloads.
9435   // Only on illegal but splittable vectors.
9436   if (SDValue ExtLoad = CombineExtLoad(N))
9437     return ExtLoad;
9438 
9439   // Try to simplify (sext (sextload x)).
9440   if (SDValue foldedExt = tryToFoldExtOfExtload(
9441           DAG, *this, TLI, VT, LegalOperations, N, N0, ISD::SEXTLOAD))
9442     return foldedExt;
9443 
9444   // fold (sext (and/or/xor (load x), cst)) ->
9445   //      (and/or/xor (sextload x), (sext cst))
9446   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
9447        N0.getOpcode() == ISD::XOR) &&
9448       isa<LoadSDNode>(N0.getOperand(0)) &&
9449       N0.getOperand(1).getOpcode() == ISD::Constant &&
9450       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
9451     LoadSDNode *LN00 = cast<LoadSDNode>(N0.getOperand(0));
9452     EVT MemVT = LN00->getMemoryVT();
9453     if (TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, MemVT) &&
9454       LN00->getExtensionType() != ISD::ZEXTLOAD && LN00->isUnindexed()) {
9455       SmallVector<SDNode*, 4> SetCCs;
9456       bool DoXform = ExtendUsesToFormExtLoad(VT, N0.getNode(), N0.getOperand(0),
9457                                              ISD::SIGN_EXTEND, SetCCs, TLI);
9458       if (DoXform) {
9459         SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(LN00), VT,
9460                                          LN00->getChain(), LN00->getBasePtr(),
9461                                          LN00->getMemoryVT(),
9462                                          LN00->getMemOperand());
9463         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
9464         Mask = Mask.sext(VT.getSizeInBits());
9465         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
9466                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
9467         ExtendSetCCUses(SetCCs, N0.getOperand(0), ExtLoad, ISD::SIGN_EXTEND);
9468         bool NoReplaceTruncAnd = !N0.hasOneUse();
9469         bool NoReplaceTrunc = SDValue(LN00, 0).hasOneUse();
9470         CombineTo(N, And);
9471         // If N0 has multiple uses, change other uses as well.
9472         if (NoReplaceTruncAnd) {
9473           SDValue TruncAnd =
9474               DAG.getNode(ISD::TRUNCATE, DL, N0.getValueType(), And);
9475           CombineTo(N0.getNode(), TruncAnd);
9476         }
9477         if (NoReplaceTrunc) {
9478           DAG.ReplaceAllUsesOfValueWith(SDValue(LN00, 1), ExtLoad.getValue(1));
9479         } else {
9480           SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(LN00),
9481                                       LN00->getValueType(0), ExtLoad);
9482           CombineTo(LN00, Trunc, ExtLoad.getValue(1));
9483         }
9484         return SDValue(N,0); // Return N so it doesn't get rechecked!
9485       }
9486     }
9487   }
9488 
9489   if (SDValue V = foldExtendedSignBitTest(N, DAG, LegalOperations))
9490     return V;
9491 
9492   if (N0.getOpcode() == ISD::SETCC) {
9493     SDValue N00 = N0.getOperand(0);
9494     SDValue N01 = N0.getOperand(1);
9495     ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
9496     EVT N00VT = N0.getOperand(0).getValueType();
9497 
9498     // sext(setcc) -> sext_in_reg(vsetcc) for vectors.
9499     // Only do this before legalize for now.
9500     if (VT.isVector() && !LegalOperations &&
9501         TLI.getBooleanContents(N00VT) ==
9502             TargetLowering::ZeroOrNegativeOneBooleanContent) {
9503       // On some architectures (such as SSE/NEON/etc) the SETCC result type is
9504       // of the same size as the compared operands. Only optimize sext(setcc())
9505       // if this is the case.
9506       EVT SVT = getSetCCResultType(N00VT);
9507 
9508       // If we already have the desired type, don't change it.
9509       if (SVT != N0.getValueType()) {
9510         // We know that the # elements of the results is the same as the
9511         // # elements of the compare (and the # elements of the compare result
9512         // for that matter).  Check to see that they are the same size.  If so,
9513         // we know that the element size of the sext'd result matches the
9514         // element size of the compare operands.
9515         if (VT.getSizeInBits() == SVT.getSizeInBits())
9516           return DAG.getSetCC(DL, VT, N00, N01, CC);
9517 
9518         // If the desired elements are smaller or larger than the source
9519         // elements, we can use a matching integer vector type and then
9520         // truncate/sign extend.
9521         EVT MatchingVecType = N00VT.changeVectorElementTypeToInteger();
9522         if (SVT == MatchingVecType) {
9523           SDValue VsetCC = DAG.getSetCC(DL, MatchingVecType, N00, N01, CC);
9524           return DAG.getSExtOrTrunc(VsetCC, DL, VT);
9525         }
9526       }
9527     }
9528 
9529     // sext(setcc x, y, cc) -> (select (setcc x, y, cc), T, 0)
9530     // Here, T can be 1 or -1, depending on the type of the setcc and
9531     // getBooleanContents().
9532     unsigned SetCCWidth = N0.getScalarValueSizeInBits();
9533 
9534     // To determine the "true" side of the select, we need to know the high bit
9535     // of the value returned by the setcc if it evaluates to true.
9536     // If the type of the setcc is i1, then the true case of the select is just
9537     // sext(i1 1), that is, -1.
9538     // If the type of the setcc is larger (say, i8) then the value of the high
9539     // bit depends on getBooleanContents(), so ask TLI for a real "true" value
9540     // of the appropriate width.
9541     SDValue ExtTrueVal = (SetCCWidth == 1)
9542                              ? DAG.getAllOnesConstant(DL, VT)
9543                              : DAG.getBoolConstant(true, DL, VT, N00VT);
9544     SDValue Zero = DAG.getConstant(0, DL, VT);
9545     if (SDValue SCC =
9546             SimplifySelectCC(DL, N00, N01, ExtTrueVal, Zero, CC, true))
9547       return SCC;
9548 
9549     if (!VT.isVector() && !TLI.convertSelectOfConstantsToMath(VT)) {
9550       EVT SetCCVT = getSetCCResultType(N00VT);
9551       // Don't do this transform for i1 because there's a select transform
9552       // that would reverse it.
9553       // TODO: We should not do this transform at all without a target hook
9554       // because a sext is likely cheaper than a select?
9555       if (SetCCVT.getScalarSizeInBits() != 1 &&
9556           (!LegalOperations || TLI.isOperationLegal(ISD::SETCC, N00VT))) {
9557         SDValue SetCC = DAG.getSetCC(DL, SetCCVT, N00, N01, CC);
9558         return DAG.getSelect(DL, VT, SetCC, ExtTrueVal, Zero);
9559       }
9560     }
9561   }
9562 
9563   // fold (sext x) -> (zext x) if the sign bit is known zero.
9564   if ((!LegalOperations || TLI.isOperationLegal(ISD::ZERO_EXTEND, VT)) &&
9565       DAG.SignBitIsZero(N0))
9566     return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0);
9567 
9568   if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N))
9569     return NewVSel;
9570 
9571   // Eliminate this sign extend by doing a negation in the destination type:
9572   // sext i32 (0 - (zext i8 X to i32)) to i64 --> 0 - (zext i8 X to i64)
9573   if (N0.getOpcode() == ISD::SUB && N0.hasOneUse() &&
9574       isNullOrNullSplat(N0.getOperand(0)) &&
9575       N0.getOperand(1).getOpcode() == ISD::ZERO_EXTEND &&
9576       TLI.isOperationLegalOrCustom(ISD::SUB, VT)) {
9577     SDValue Zext = DAG.getZExtOrTrunc(N0.getOperand(1).getOperand(0), DL, VT);
9578     return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), Zext);
9579   }
9580   // Eliminate this sign extend by doing a decrement in the destination type:
9581   // sext i32 ((zext i8 X to i32) + (-1)) to i64 --> (zext i8 X to i64) + (-1)
9582   if (N0.getOpcode() == ISD::ADD && N0.hasOneUse() &&
9583       isAllOnesOrAllOnesSplat(N0.getOperand(1)) &&
9584       N0.getOperand(0).getOpcode() == ISD::ZERO_EXTEND &&
9585       TLI.isOperationLegalOrCustom(ISD::ADD, VT)) {
9586     SDValue Zext = DAG.getZExtOrTrunc(N0.getOperand(0).getOperand(0), DL, VT);
9587     return DAG.getNode(ISD::ADD, DL, VT, Zext, DAG.getAllOnesConstant(DL, VT));
9588   }
9589 
9590   return SDValue();
9591 }
9592 
9593 // isTruncateOf - If N is a truncate of some other value, return true, record
9594 // the value being truncated in Op and which of Op's bits are zero/one in Known.
9595 // This function computes KnownBits to avoid a duplicated call to
9596 // computeKnownBits in the caller.
9597 static bool isTruncateOf(SelectionDAG &DAG, SDValue N, SDValue &Op,
9598                          KnownBits &Known) {
9599   if (N->getOpcode() == ISD::TRUNCATE) {
9600     Op = N->getOperand(0);
9601     Known = DAG.computeKnownBits(Op);
9602     return true;
9603   }
9604 
9605   if (N.getOpcode() != ISD::SETCC ||
9606       N.getValueType().getScalarType() != MVT::i1 ||
9607       cast<CondCodeSDNode>(N.getOperand(2))->get() != ISD::SETNE)
9608     return false;
9609 
9610   SDValue Op0 = N->getOperand(0);
9611   SDValue Op1 = N->getOperand(1);
9612   assert(Op0.getValueType() == Op1.getValueType());
9613 
9614   if (isNullOrNullSplat(Op0))
9615     Op = Op1;
9616   else if (isNullOrNullSplat(Op1))
9617     Op = Op0;
9618   else
9619     return false;
9620 
9621   Known = DAG.computeKnownBits(Op);
9622 
9623   return (Known.Zero | 1).isAllOnesValue();
9624 }
9625 
9626 SDValue DAGCombiner::visitZERO_EXTEND(SDNode *N) {
9627   SDValue N0 = N->getOperand(0);
9628   EVT VT = N->getValueType(0);
9629 
9630   if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes))
9631     return Res;
9632 
9633   // fold (zext (zext x)) -> (zext x)
9634   // fold (zext (aext x)) -> (zext x)
9635   if (N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
9636     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT,
9637                        N0.getOperand(0));
9638 
9639   // fold (zext (truncate x)) -> (zext x) or
9640   //      (zext (truncate x)) -> (truncate x)
9641   // This is valid when the truncated bits of x are already zero.
9642   SDValue Op;
9643   KnownBits Known;
9644   if (isTruncateOf(DAG, N0, Op, Known)) {
9645     APInt TruncatedBits =
9646       (Op.getScalarValueSizeInBits() == N0.getScalarValueSizeInBits()) ?
9647       APInt(Op.getScalarValueSizeInBits(), 0) :
9648       APInt::getBitsSet(Op.getScalarValueSizeInBits(),
9649                         N0.getScalarValueSizeInBits(),
9650                         std::min(Op.getScalarValueSizeInBits(),
9651                                  VT.getScalarSizeInBits()));
9652     if (TruncatedBits.isSubsetOf(Known.Zero))
9653       return DAG.getZExtOrTrunc(Op, SDLoc(N), VT);
9654   }
9655 
9656   // fold (zext (truncate x)) -> (and x, mask)
9657   if (N0.getOpcode() == ISD::TRUNCATE) {
9658     // fold (zext (truncate (load x))) -> (zext (smaller load x))
9659     // fold (zext (truncate (srl (load x), c))) -> (zext (smaller load (x+c/n)))
9660     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
9661       SDNode *oye = N0.getOperand(0).getNode();
9662       if (NarrowLoad.getNode() != N0.getNode()) {
9663         CombineTo(N0.getNode(), NarrowLoad);
9664         // CombineTo deleted the truncate, if needed, but not what's under it.
9665         AddToWorklist(oye);
9666       }
9667       return SDValue(N, 0); // Return N so it doesn't get rechecked!
9668     }
9669 
9670     EVT SrcVT = N0.getOperand(0).getValueType();
9671     EVT MinVT = N0.getValueType();
9672 
9673     // Try to mask before the extension to avoid having to generate a larger mask,
9674     // possibly over several sub-vectors.
9675     if (SrcVT.bitsLT(VT) && VT.isVector()) {
9676       if (!LegalOperations || (TLI.isOperationLegal(ISD::AND, SrcVT) &&
9677                                TLI.isOperationLegal(ISD::ZERO_EXTEND, VT))) {
9678         SDValue Op = N0.getOperand(0);
9679         Op = DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
9680         AddToWorklist(Op.getNode());
9681         SDValue ZExtOrTrunc = DAG.getZExtOrTrunc(Op, SDLoc(N), VT);
9682         // Transfer the debug info; the new node is equivalent to N0.
9683         DAG.transferDbgValues(N0, ZExtOrTrunc);
9684         return ZExtOrTrunc;
9685       }
9686     }
9687 
9688     if (!LegalOperations || TLI.isOperationLegal(ISD::AND, VT)) {
9689       SDValue Op = DAG.getAnyExtOrTrunc(N0.getOperand(0), SDLoc(N), VT);
9690       AddToWorklist(Op.getNode());
9691       SDValue And = DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
9692       // We may safely transfer the debug info describing the truncate node over
9693       // to the equivalent and operation.
9694       DAG.transferDbgValues(N0, And);
9695       return And;
9696     }
9697   }
9698 
9699   // Fold (zext (and (trunc x), cst)) -> (and x, cst),
9700   // if either of the casts is not free.
9701   if (N0.getOpcode() == ISD::AND &&
9702       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
9703       N0.getOperand(1).getOpcode() == ISD::Constant &&
9704       (!TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
9705                            N0.getValueType()) ||
9706        !TLI.isZExtFree(N0.getValueType(), VT))) {
9707     SDValue X = N0.getOperand(0).getOperand(0);
9708     X = DAG.getAnyExtOrTrunc(X, SDLoc(X), VT);
9709     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
9710     Mask = Mask.zext(VT.getSizeInBits());
9711     SDLoc DL(N);
9712     return DAG.getNode(ISD::AND, DL, VT,
9713                        X, DAG.getConstant(Mask, DL, VT));
9714   }
9715 
9716   // Try to simplify (zext (load x)).
9717   if (SDValue foldedExt =
9718           tryToFoldExtOfLoad(DAG, *this, TLI, VT, LegalOperations, N, N0,
9719                              ISD::ZEXTLOAD, ISD::ZERO_EXTEND))
9720     return foldedExt;
9721 
9722   // fold (zext (load x)) to multiple smaller zextloads.
9723   // Only on illegal but splittable vectors.
9724   if (SDValue ExtLoad = CombineExtLoad(N))
9725     return ExtLoad;
9726 
9727   // fold (zext (and/or/xor (load x), cst)) ->
9728   //      (and/or/xor (zextload x), (zext cst))
9729   // Unless (and (load x) cst) will match as a zextload already and has
9730   // additional users.
9731   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
9732        N0.getOpcode() == ISD::XOR) &&
9733       isa<LoadSDNode>(N0.getOperand(0)) &&
9734       N0.getOperand(1).getOpcode() == ISD::Constant &&
9735       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
9736     LoadSDNode *LN00 = cast<LoadSDNode>(N0.getOperand(0));
9737     EVT MemVT = LN00->getMemoryVT();
9738     if (TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT) &&
9739         LN00->getExtensionType() != ISD::SEXTLOAD && LN00->isUnindexed()) {
9740       bool DoXform = true;
9741       SmallVector<SDNode*, 4> SetCCs;
9742       if (!N0.hasOneUse()) {
9743         if (N0.getOpcode() == ISD::AND) {
9744           auto *AndC = cast<ConstantSDNode>(N0.getOperand(1));
9745           EVT LoadResultTy = AndC->getValueType(0);
9746           EVT ExtVT;
9747           if (isAndLoadExtLoad(AndC, LN00, LoadResultTy, ExtVT))
9748             DoXform = false;
9749         }
9750       }
9751       if (DoXform)
9752         DoXform = ExtendUsesToFormExtLoad(VT, N0.getNode(), N0.getOperand(0),
9753                                           ISD::ZERO_EXTEND, SetCCs, TLI);
9754       if (DoXform) {
9755         SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN00), VT,
9756                                          LN00->getChain(), LN00->getBasePtr(),
9757                                          LN00->getMemoryVT(),
9758                                          LN00->getMemOperand());
9759         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
9760         Mask = Mask.zext(VT.getSizeInBits());
9761         SDLoc DL(N);
9762         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
9763                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
9764         ExtendSetCCUses(SetCCs, N0.getOperand(0), ExtLoad, ISD::ZERO_EXTEND);
9765         bool NoReplaceTruncAnd = !N0.hasOneUse();
9766         bool NoReplaceTrunc = SDValue(LN00, 0).hasOneUse();
9767         CombineTo(N, And);
9768         // If N0 has multiple uses, change other uses as well.
9769         if (NoReplaceTruncAnd) {
9770           SDValue TruncAnd =
9771               DAG.getNode(ISD::TRUNCATE, DL, N0.getValueType(), And);
9772           CombineTo(N0.getNode(), TruncAnd);
9773         }
9774         if (NoReplaceTrunc) {
9775           DAG.ReplaceAllUsesOfValueWith(SDValue(LN00, 1), ExtLoad.getValue(1));
9776         } else {
9777           SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(LN00),
9778                                       LN00->getValueType(0), ExtLoad);
9779           CombineTo(LN00, Trunc, ExtLoad.getValue(1));
9780         }
9781         return SDValue(N,0); // Return N so it doesn't get rechecked!
9782       }
9783     }
9784   }
9785 
9786   // fold (zext (and/or/xor (shl/shr (load x), cst), cst)) ->
9787   //      (and/or/xor (shl/shr (zextload x), (zext cst)), (zext cst))
9788   if (SDValue ZExtLoad = CombineZExtLogicopShiftLoad(N))
9789     return ZExtLoad;
9790 
9791   // Try to simplify (zext (zextload x)).
9792   if (SDValue foldedExt = tryToFoldExtOfExtload(
9793           DAG, *this, TLI, VT, LegalOperations, N, N0, ISD::ZEXTLOAD))
9794     return foldedExt;
9795 
9796   if (SDValue V = foldExtendedSignBitTest(N, DAG, LegalOperations))
9797     return V;
9798 
9799   if (N0.getOpcode() == ISD::SETCC) {
9800     // Only do this before legalize for now.
9801     if (!LegalOperations && VT.isVector() &&
9802         N0.getValueType().getVectorElementType() == MVT::i1) {
9803       EVT N00VT = N0.getOperand(0).getValueType();
9804       if (getSetCCResultType(N00VT) == N0.getValueType())
9805         return SDValue();
9806 
9807       // We know that the # elements of the results is the same as the #
9808       // elements of the compare (and the # elements of the compare result for
9809       // that matter). Check to see that they are the same size. If so, we know
9810       // that the element size of the sext'd result matches the element size of
9811       // the compare operands.
9812       SDLoc DL(N);
9813       SDValue VecOnes = DAG.getConstant(1, DL, VT);
9814       if (VT.getSizeInBits() == N00VT.getSizeInBits()) {
9815         // zext(setcc) -> (and (vsetcc), (1, 1, ...) for vectors.
9816         SDValue VSetCC = DAG.getNode(ISD::SETCC, DL, VT, N0.getOperand(0),
9817                                      N0.getOperand(1), N0.getOperand(2));
9818         return DAG.getNode(ISD::AND, DL, VT, VSetCC, VecOnes);
9819       }
9820 
9821       // If the desired elements are smaller or larger than the source
9822       // elements we can use a matching integer vector type and then
9823       // truncate/sign extend.
9824       EVT MatchingVectorType = N00VT.changeVectorElementTypeToInteger();
9825       SDValue VsetCC =
9826           DAG.getNode(ISD::SETCC, DL, MatchingVectorType, N0.getOperand(0),
9827                       N0.getOperand(1), N0.getOperand(2));
9828       return DAG.getNode(ISD::AND, DL, VT, DAG.getSExtOrTrunc(VsetCC, DL, VT),
9829                          VecOnes);
9830     }
9831 
9832     // zext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
9833     SDLoc DL(N);
9834     if (SDValue SCC = SimplifySelectCC(
9835             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
9836             DAG.getConstant(0, DL, VT),
9837             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
9838       return SCC;
9839   }
9840 
9841   // (zext (shl (zext x), cst)) -> (shl (zext x), cst)
9842   if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL) &&
9843       isa<ConstantSDNode>(N0.getOperand(1)) &&
9844       N0.getOperand(0).getOpcode() == ISD::ZERO_EXTEND &&
9845       N0.hasOneUse()) {
9846     SDValue ShAmt = N0.getOperand(1);
9847     if (N0.getOpcode() == ISD::SHL) {
9848       SDValue InnerZExt = N0.getOperand(0);
9849       // If the original shl may be shifting out bits, do not perform this
9850       // transformation.
9851       unsigned KnownZeroBits = InnerZExt.getValueSizeInBits() -
9852         InnerZExt.getOperand(0).getValueSizeInBits();
9853       if (cast<ConstantSDNode>(ShAmt)->getAPIntValue().ugt(KnownZeroBits))
9854         return SDValue();
9855     }
9856 
9857     SDLoc DL(N);
9858 
9859     // Ensure that the shift amount is wide enough for the shifted value.
9860     if (VT.getSizeInBits() >= 256)
9861       ShAmt = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, ShAmt);
9862 
9863     return DAG.getNode(N0.getOpcode(), DL, VT,
9864                        DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0)),
9865                        ShAmt);
9866   }
9867 
9868   if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N))
9869     return NewVSel;
9870 
9871   return SDValue();
9872 }
9873 
9874 SDValue DAGCombiner::visitANY_EXTEND(SDNode *N) {
9875   SDValue N0 = N->getOperand(0);
9876   EVT VT = N->getValueType(0);
9877 
9878   if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes))
9879     return Res;
9880 
9881   // fold (aext (aext x)) -> (aext x)
9882   // fold (aext (zext x)) -> (zext x)
9883   // fold (aext (sext x)) -> (sext x)
9884   if (N0.getOpcode() == ISD::ANY_EXTEND  ||
9885       N0.getOpcode() == ISD::ZERO_EXTEND ||
9886       N0.getOpcode() == ISD::SIGN_EXTEND)
9887     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
9888 
9889   // fold (aext (truncate (load x))) -> (aext (smaller load x))
9890   // fold (aext (truncate (srl (load x), c))) -> (aext (small load (x+c/n)))
9891   if (N0.getOpcode() == ISD::TRUNCATE) {
9892     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
9893       SDNode *oye = N0.getOperand(0).getNode();
9894       if (NarrowLoad.getNode() != N0.getNode()) {
9895         CombineTo(N0.getNode(), NarrowLoad);
9896         // CombineTo deleted the truncate, if needed, but not what's under it.
9897         AddToWorklist(oye);
9898       }
9899       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9900     }
9901   }
9902 
9903   // fold (aext (truncate x))
9904   if (N0.getOpcode() == ISD::TRUNCATE)
9905     return DAG.getAnyExtOrTrunc(N0.getOperand(0), SDLoc(N), VT);
9906 
9907   // Fold (aext (and (trunc x), cst)) -> (and x, cst)
9908   // if the trunc is not free.
9909   if (N0.getOpcode() == ISD::AND &&
9910       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
9911       N0.getOperand(1).getOpcode() == ISD::Constant &&
9912       !TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
9913                           N0.getValueType())) {
9914     SDLoc DL(N);
9915     SDValue X = N0.getOperand(0).getOperand(0);
9916     X = DAG.getAnyExtOrTrunc(X, DL, VT);
9917     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
9918     Mask = Mask.zext(VT.getSizeInBits());
9919     return DAG.getNode(ISD::AND, DL, VT,
9920                        X, DAG.getConstant(Mask, DL, VT));
9921   }
9922 
9923   // fold (aext (load x)) -> (aext (truncate (extload x)))
9924   // None of the supported targets knows how to perform load and any_ext
9925   // on vectors in one instruction.  We only perform this transformation on
9926   // scalars.
9927   if (ISD::isNON_EXTLoad(N0.getNode()) && !VT.isVector() &&
9928       ISD::isUNINDEXEDLoad(N0.getNode()) &&
9929       TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
9930     bool DoXform = true;
9931     SmallVector<SDNode*, 4> SetCCs;
9932     if (!N0.hasOneUse())
9933       DoXform = ExtendUsesToFormExtLoad(VT, N, N0, ISD::ANY_EXTEND, SetCCs,
9934                                         TLI);
9935     if (DoXform) {
9936       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9937       SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
9938                                        LN0->getChain(),
9939                                        LN0->getBasePtr(), N0.getValueType(),
9940                                        LN0->getMemOperand());
9941       ExtendSetCCUses(SetCCs, N0, ExtLoad, ISD::ANY_EXTEND);
9942       // If the load value is used only by N, replace it via CombineTo N.
9943       bool NoReplaceTrunc = N0.hasOneUse();
9944       CombineTo(N, ExtLoad);
9945       if (NoReplaceTrunc) {
9946         DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1));
9947         recursivelyDeleteUnusedNodes(LN0);
9948       } else {
9949         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
9950                                     N0.getValueType(), ExtLoad);
9951         CombineTo(LN0, Trunc, ExtLoad.getValue(1));
9952       }
9953       return SDValue(N, 0); // Return N so it doesn't get rechecked!
9954     }
9955   }
9956 
9957   // fold (aext (zextload x)) -> (aext (truncate (zextload x)))
9958   // fold (aext (sextload x)) -> (aext (truncate (sextload x)))
9959   // fold (aext ( extload x)) -> (aext (truncate (extload  x)))
9960   if (N0.getOpcode() == ISD::LOAD && !ISD::isNON_EXTLoad(N0.getNode()) &&
9961       ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) {
9962     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9963     ISD::LoadExtType ExtType = LN0->getExtensionType();
9964     EVT MemVT = LN0->getMemoryVT();
9965     if (!LegalOperations || TLI.isLoadExtLegal(ExtType, VT, MemVT)) {
9966       SDValue ExtLoad = DAG.getExtLoad(ExtType, SDLoc(N),
9967                                        VT, LN0->getChain(), LN0->getBasePtr(),
9968                                        MemVT, LN0->getMemOperand());
9969       CombineTo(N, ExtLoad);
9970       DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1));
9971       recursivelyDeleteUnusedNodes(LN0);
9972       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9973     }
9974   }
9975 
9976   if (N0.getOpcode() == ISD::SETCC) {
9977     // For vectors:
9978     // aext(setcc) -> vsetcc
9979     // aext(setcc) -> truncate(vsetcc)
9980     // aext(setcc) -> aext(vsetcc)
9981     // Only do this before legalize for now.
9982     if (VT.isVector() && !LegalOperations) {
9983       EVT N00VT = N0.getOperand(0).getValueType();
9984       if (getSetCCResultType(N00VT) == N0.getValueType())
9985         return SDValue();
9986 
9987       // We know that the # elements of the results is the same as the
9988       // # elements of the compare (and the # elements of the compare result
9989       // for that matter).  Check to see that they are the same size.  If so,
9990       // we know that the element size of the sext'd result matches the
9991       // element size of the compare operands.
9992       if (VT.getSizeInBits() == N00VT.getSizeInBits())
9993         return DAG.getSetCC(SDLoc(N), VT, N0.getOperand(0),
9994                              N0.getOperand(1),
9995                              cast<CondCodeSDNode>(N0.getOperand(2))->get());
9996 
9997       // If the desired elements are smaller or larger than the source
9998       // elements we can use a matching integer vector type and then
9999       // truncate/any extend
10000       EVT MatchingVectorType = N00VT.changeVectorElementTypeToInteger();
10001       SDValue VsetCC =
10002         DAG.getSetCC(SDLoc(N), MatchingVectorType, N0.getOperand(0),
10003                       N0.getOperand(1),
10004                       cast<CondCodeSDNode>(N0.getOperand(2))->get());
10005       return DAG.getAnyExtOrTrunc(VsetCC, SDLoc(N), VT);
10006     }
10007 
10008     // aext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
10009     SDLoc DL(N);
10010     if (SDValue SCC = SimplifySelectCC(
10011             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
10012             DAG.getConstant(0, DL, VT),
10013             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
10014       return SCC;
10015   }
10016 
10017   return SDValue();
10018 }
10019 
10020 SDValue DAGCombiner::visitAssertExt(SDNode *N) {
10021   unsigned Opcode = N->getOpcode();
10022   SDValue N0 = N->getOperand(0);
10023   SDValue N1 = N->getOperand(1);
10024   EVT AssertVT = cast<VTSDNode>(N1)->getVT();
10025 
10026   // fold (assert?ext (assert?ext x, vt), vt) -> (assert?ext x, vt)
10027   if (N0.getOpcode() == Opcode &&
10028       AssertVT == cast<VTSDNode>(N0.getOperand(1))->getVT())
10029     return N0;
10030 
10031   if (N0.getOpcode() == ISD::TRUNCATE && N0.hasOneUse() &&
10032       N0.getOperand(0).getOpcode() == Opcode) {
10033     // We have an assert, truncate, assert sandwich. Make one stronger assert
10034     // by asserting on the smallest asserted type to the larger source type.
10035     // This eliminates the later assert:
10036     // assert (trunc (assert X, i8) to iN), i1 --> trunc (assert X, i1) to iN
10037     // assert (trunc (assert X, i1) to iN), i8 --> trunc (assert X, i1) to iN
10038     SDValue BigA = N0.getOperand(0);
10039     EVT BigA_AssertVT = cast<VTSDNode>(BigA.getOperand(1))->getVT();
10040     assert(BigA_AssertVT.bitsLE(N0.getValueType()) &&
10041            "Asserting zero/sign-extended bits to a type larger than the "
10042            "truncated destination does not provide information");
10043 
10044     SDLoc DL(N);
10045     EVT MinAssertVT = AssertVT.bitsLT(BigA_AssertVT) ? AssertVT : BigA_AssertVT;
10046     SDValue MinAssertVTVal = DAG.getValueType(MinAssertVT);
10047     SDValue NewAssert = DAG.getNode(Opcode, DL, BigA.getValueType(),
10048                                     BigA.getOperand(0), MinAssertVTVal);
10049     return DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), NewAssert);
10050   }
10051 
10052   // If we have (AssertZext (truncate (AssertSext X, iX)), iY) and Y is smaller
10053   // than X. Just move the AssertZext in front of the truncate and drop the
10054   // AssertSExt.
10055   if (N0.getOpcode() == ISD::TRUNCATE && N0.hasOneUse() &&
10056       N0.getOperand(0).getOpcode() == ISD::AssertSext &&
10057       Opcode == ISD::AssertZext) {
10058     SDValue BigA = N0.getOperand(0);
10059     EVT BigA_AssertVT = cast<VTSDNode>(BigA.getOperand(1))->getVT();
10060     assert(BigA_AssertVT.bitsLE(N0.getValueType()) &&
10061            "Asserting zero/sign-extended bits to a type larger than the "
10062            "truncated destination does not provide information");
10063 
10064     if (AssertVT.bitsLT(BigA_AssertVT)) {
10065       SDLoc DL(N);
10066       SDValue NewAssert = DAG.getNode(Opcode, DL, BigA.getValueType(),
10067                                       BigA.getOperand(0), N1);
10068       return DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), NewAssert);
10069     }
10070   }
10071 
10072   return SDValue();
10073 }
10074 
10075 /// If the result of a wider load is shifted to right of N  bits and then
10076 /// truncated to a narrower type and where N is a multiple of number of bits of
10077 /// the narrower type, transform it to a narrower load from address + N / num of
10078 /// bits of new type. Also narrow the load if the result is masked with an AND
10079 /// to effectively produce a smaller type. If the result is to be extended, also
10080 /// fold the extension to form a extending load.
10081 SDValue DAGCombiner::ReduceLoadWidth(SDNode *N) {
10082   unsigned Opc = N->getOpcode();
10083 
10084   ISD::LoadExtType ExtType = ISD::NON_EXTLOAD;
10085   SDValue N0 = N->getOperand(0);
10086   EVT VT = N->getValueType(0);
10087   EVT ExtVT = VT;
10088 
10089   // This transformation isn't valid for vector loads.
10090   if (VT.isVector())
10091     return SDValue();
10092 
10093   unsigned ShAmt = 0;
10094   bool HasShiftedOffset = false;
10095   // Special case: SIGN_EXTEND_INREG is basically truncating to ExtVT then
10096   // extended to VT.
10097   if (Opc == ISD::SIGN_EXTEND_INREG) {
10098     ExtType = ISD::SEXTLOAD;
10099     ExtVT = cast<VTSDNode>(N->getOperand(1))->getVT();
10100   } else if (Opc == ISD::SRL) {
10101     // Another special-case: SRL is basically zero-extending a narrower value,
10102     // or it maybe shifting a higher subword, half or byte into the lowest
10103     // bits.
10104     ExtType = ISD::ZEXTLOAD;
10105     N0 = SDValue(N, 0);
10106 
10107     auto *LN0 = dyn_cast<LoadSDNode>(N0.getOperand(0));
10108     auto *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
10109     if (!N01 || !LN0)
10110       return SDValue();
10111 
10112     uint64_t ShiftAmt = N01->getZExtValue();
10113     uint64_t MemoryWidth = LN0->getMemoryVT().getSizeInBits();
10114     if (LN0->getExtensionType() != ISD::SEXTLOAD && MemoryWidth > ShiftAmt)
10115       ExtVT = EVT::getIntegerVT(*DAG.getContext(), MemoryWidth - ShiftAmt);
10116     else
10117       ExtVT = EVT::getIntegerVT(*DAG.getContext(),
10118                                 VT.getSizeInBits() - ShiftAmt);
10119   } else if (Opc == ISD::AND) {
10120     // An AND with a constant mask is the same as a truncate + zero-extend.
10121     auto AndC = dyn_cast<ConstantSDNode>(N->getOperand(1));
10122     if (!AndC)
10123       return SDValue();
10124 
10125     const APInt &Mask = AndC->getAPIntValue();
10126     unsigned ActiveBits = 0;
10127     if (Mask.isMask()) {
10128       ActiveBits = Mask.countTrailingOnes();
10129     } else if (Mask.isShiftedMask()) {
10130       ShAmt = Mask.countTrailingZeros();
10131       APInt ShiftedMask = Mask.lshr(ShAmt);
10132       ActiveBits = ShiftedMask.countTrailingOnes();
10133       HasShiftedOffset = true;
10134     } else
10135       return SDValue();
10136 
10137     ExtType = ISD::ZEXTLOAD;
10138     ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits);
10139   }
10140 
10141   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
10142     SDValue SRL = N0;
10143     if (auto *ConstShift = dyn_cast<ConstantSDNode>(SRL.getOperand(1))) {
10144       ShAmt = ConstShift->getZExtValue();
10145       unsigned EVTBits = ExtVT.getSizeInBits();
10146       // Is the shift amount a multiple of size of VT?
10147       if ((ShAmt & (EVTBits-1)) == 0) {
10148         N0 = N0.getOperand(0);
10149         // Is the load width a multiple of size of VT?
10150         if ((N0.getValueSizeInBits() & (EVTBits-1)) != 0)
10151           return SDValue();
10152       }
10153 
10154       // At this point, we must have a load or else we can't do the transform.
10155       if (!isa<LoadSDNode>(N0)) return SDValue();
10156 
10157       auto *LN0 = cast<LoadSDNode>(N0);
10158 
10159       // Because a SRL must be assumed to *need* to zero-extend the high bits
10160       // (as opposed to anyext the high bits), we can't combine the zextload
10161       // lowering of SRL and an sextload.
10162       if (LN0->getExtensionType() == ISD::SEXTLOAD)
10163         return SDValue();
10164 
10165       // If the shift amount is larger than the input type then we're not
10166       // accessing any of the loaded bytes.  If the load was a zextload/extload
10167       // then the result of the shift+trunc is zero/undef (handled elsewhere).
10168       if (ShAmt >= LN0->getMemoryVT().getSizeInBits())
10169         return SDValue();
10170 
10171       // If the SRL is only used by a masking AND, we may be able to adjust
10172       // the ExtVT to make the AND redundant.
10173       SDNode *Mask = *(SRL->use_begin());
10174       if (Mask->getOpcode() == ISD::AND &&
10175           isa<ConstantSDNode>(Mask->getOperand(1))) {
10176         const APInt &ShiftMask =
10177           cast<ConstantSDNode>(Mask->getOperand(1))->getAPIntValue();
10178         if (ShiftMask.isMask()) {
10179           EVT MaskedVT = EVT::getIntegerVT(*DAG.getContext(),
10180                                            ShiftMask.countTrailingOnes());
10181           // If the mask is smaller, recompute the type.
10182           if ((ExtVT.getSizeInBits() > MaskedVT.getSizeInBits()) &&
10183               TLI.isLoadExtLegal(ExtType, N0.getValueType(), MaskedVT))
10184             ExtVT = MaskedVT;
10185         }
10186       }
10187     }
10188   }
10189 
10190   // If the load is shifted left (and the result isn't shifted back right),
10191   // we can fold the truncate through the shift.
10192   unsigned ShLeftAmt = 0;
10193   if (ShAmt == 0 && N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
10194       ExtVT == VT && TLI.isNarrowingProfitable(N0.getValueType(), VT)) {
10195     if (ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
10196       ShLeftAmt = N01->getZExtValue();
10197       N0 = N0.getOperand(0);
10198     }
10199   }
10200 
10201   // If we haven't found a load, we can't narrow it.
10202   if (!isa<LoadSDNode>(N0))
10203     return SDValue();
10204 
10205   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
10206   if (!isLegalNarrowLdSt(LN0, ExtType, ExtVT, ShAmt))
10207     return SDValue();
10208 
10209   auto AdjustBigEndianShift = [&](unsigned ShAmt) {
10210     unsigned LVTStoreBits = LN0->getMemoryVT().getStoreSizeInBits();
10211     unsigned EVTStoreBits = ExtVT.getStoreSizeInBits();
10212     return LVTStoreBits - EVTStoreBits - ShAmt;
10213   };
10214 
10215   // For big endian targets, we need to adjust the offset to the pointer to
10216   // load the correct bytes.
10217   if (DAG.getDataLayout().isBigEndian())
10218     ShAmt = AdjustBigEndianShift(ShAmt);
10219 
10220   EVT PtrType = N0.getOperand(1).getValueType();
10221   uint64_t PtrOff = ShAmt / 8;
10222   unsigned NewAlign = MinAlign(LN0->getAlignment(), PtrOff);
10223   SDLoc DL(LN0);
10224   // The original load itself didn't wrap, so an offset within it doesn't.
10225   SDNodeFlags Flags;
10226   Flags.setNoUnsignedWrap(true);
10227   SDValue NewPtr = DAG.getNode(ISD::ADD, DL,
10228                                PtrType, LN0->getBasePtr(),
10229                                DAG.getConstant(PtrOff, DL, PtrType),
10230                                Flags);
10231   AddToWorklist(NewPtr.getNode());
10232 
10233   SDValue Load;
10234   if (ExtType == ISD::NON_EXTLOAD)
10235     Load = DAG.getLoad(VT, SDLoc(N0), LN0->getChain(), NewPtr,
10236                        LN0->getPointerInfo().getWithOffset(PtrOff), NewAlign,
10237                        LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
10238   else
10239     Load = DAG.getExtLoad(ExtType, SDLoc(N0), VT, LN0->getChain(), NewPtr,
10240                           LN0->getPointerInfo().getWithOffset(PtrOff), ExtVT,
10241                           NewAlign, LN0->getMemOperand()->getFlags(),
10242                           LN0->getAAInfo());
10243 
10244   // Replace the old load's chain with the new load's chain.
10245   WorklistRemover DeadNodes(*this);
10246   DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
10247 
10248   // Shift the result left, if we've swallowed a left shift.
10249   SDValue Result = Load;
10250   if (ShLeftAmt != 0) {
10251     EVT ShImmTy = getShiftAmountTy(Result.getValueType());
10252     if (!isUIntN(ShImmTy.getSizeInBits(), ShLeftAmt))
10253       ShImmTy = VT;
10254     // If the shift amount is as large as the result size (but, presumably,
10255     // no larger than the source) then the useful bits of the result are
10256     // zero; we can't simply return the shortened shift, because the result
10257     // of that operation is undefined.
10258     SDLoc DL(N0);
10259     if (ShLeftAmt >= VT.getSizeInBits())
10260       Result = DAG.getConstant(0, DL, VT);
10261     else
10262       Result = DAG.getNode(ISD::SHL, DL, VT,
10263                           Result, DAG.getConstant(ShLeftAmt, DL, ShImmTy));
10264   }
10265 
10266   if (HasShiftedOffset) {
10267     // Recalculate the shift amount after it has been altered to calculate
10268     // the offset.
10269     if (DAG.getDataLayout().isBigEndian())
10270       ShAmt = AdjustBigEndianShift(ShAmt);
10271 
10272     // We're using a shifted mask, so the load now has an offset. This means
10273     // that data has been loaded into the lower bytes than it would have been
10274     // before, so we need to shl the loaded data into the correct position in the
10275     // register.
10276     SDValue ShiftC = DAG.getConstant(ShAmt, DL, VT);
10277     Result = DAG.getNode(ISD::SHL, DL, VT, Result, ShiftC);
10278     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result);
10279   }
10280 
10281   // Return the new loaded value.
10282   return Result;
10283 }
10284 
10285 SDValue DAGCombiner::visitSIGN_EXTEND_INREG(SDNode *N) {
10286   SDValue N0 = N->getOperand(0);
10287   SDValue N1 = N->getOperand(1);
10288   EVT VT = N->getValueType(0);
10289   EVT EVT = cast<VTSDNode>(N1)->getVT();
10290   unsigned VTBits = VT.getScalarSizeInBits();
10291   unsigned EVTBits = EVT.getScalarSizeInBits();
10292 
10293   if (N0.isUndef())
10294     return DAG.getUNDEF(VT);
10295 
10296   // fold (sext_in_reg c1) -> c1
10297   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
10298     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, N0, N1);
10299 
10300   // If the input is already sign extended, just drop the extension.
10301   if (DAG.ComputeNumSignBits(N0) >= VTBits-EVTBits+1)
10302     return N0;
10303 
10304   // fold (sext_in_reg (sext_in_reg x, VT2), VT1) -> (sext_in_reg x, minVT) pt2
10305   if (N0.getOpcode() == ISD::SIGN_EXTEND_INREG &&
10306       EVT.bitsLT(cast<VTSDNode>(N0.getOperand(1))->getVT()))
10307     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
10308                        N0.getOperand(0), N1);
10309 
10310   // fold (sext_in_reg (sext x)) -> (sext x)
10311   // fold (sext_in_reg (aext x)) -> (sext x)
10312   // if x is small enough or if we know that x has more than 1 sign bit and the
10313   // sign_extend_inreg is extending from one of them.
10314   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) {
10315     SDValue N00 = N0.getOperand(0);
10316     unsigned N00Bits = N00.getScalarValueSizeInBits();
10317     if ((N00Bits <= EVTBits ||
10318          (N00Bits - DAG.ComputeNumSignBits(N00)) < EVTBits) &&
10319         (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT)))
10320       return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00);
10321   }
10322 
10323   // fold (sext_in_reg (*_extend_vector_inreg x)) -> (sext_vector_inreg x)
10324   if ((N0.getOpcode() == ISD::ANY_EXTEND_VECTOR_INREG ||
10325        N0.getOpcode() == ISD::SIGN_EXTEND_VECTOR_INREG ||
10326        N0.getOpcode() == ISD::ZERO_EXTEND_VECTOR_INREG) &&
10327       N0.getOperand(0).getScalarValueSizeInBits() == EVTBits) {
10328     if (!LegalOperations ||
10329         TLI.isOperationLegal(ISD::SIGN_EXTEND_VECTOR_INREG, VT))
10330       return DAG.getNode(ISD::SIGN_EXTEND_VECTOR_INREG, SDLoc(N), VT,
10331                          N0.getOperand(0));
10332   }
10333 
10334   // fold (sext_in_reg (zext x)) -> (sext x)
10335   // iff we are extending the source sign bit.
10336   if (N0.getOpcode() == ISD::ZERO_EXTEND) {
10337     SDValue N00 = N0.getOperand(0);
10338     if (N00.getScalarValueSizeInBits() == EVTBits &&
10339         (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT)))
10340       return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00, N1);
10341   }
10342 
10343   // fold (sext_in_reg x) -> (zext_in_reg x) if the sign bit is known zero.
10344   if (DAG.MaskedValueIsZero(N0, APInt::getOneBitSet(VTBits, EVTBits - 1)))
10345     return DAG.getZeroExtendInReg(N0, SDLoc(N), EVT.getScalarType());
10346 
10347   // fold operands of sext_in_reg based on knowledge that the top bits are not
10348   // demanded.
10349   if (SimplifyDemandedBits(SDValue(N, 0)))
10350     return SDValue(N, 0);
10351 
10352   // fold (sext_in_reg (load x)) -> (smaller sextload x)
10353   // fold (sext_in_reg (srl (load x), c)) -> (smaller sextload (x+c/evtbits))
10354   if (SDValue NarrowLoad = ReduceLoadWidth(N))
10355     return NarrowLoad;
10356 
10357   // fold (sext_in_reg (srl X, 24), i8) -> (sra X, 24)
10358   // fold (sext_in_reg (srl X, 23), i8) -> (sra X, 23) iff possible.
10359   // We already fold "(sext_in_reg (srl X, 25), i8) -> srl X, 25" above.
10360   if (N0.getOpcode() == ISD::SRL) {
10361     if (auto *ShAmt = dyn_cast<ConstantSDNode>(N0.getOperand(1)))
10362       if (ShAmt->getAPIntValue().ule(VTBits - EVTBits)) {
10363         // We can turn this into an SRA iff the input to the SRL is already sign
10364         // extended enough.
10365         unsigned InSignBits = DAG.ComputeNumSignBits(N0.getOperand(0));
10366         if (((VTBits - EVTBits) - ShAmt->getZExtValue()) < InSignBits)
10367           return DAG.getNode(ISD::SRA, SDLoc(N), VT, N0.getOperand(0),
10368                              N0.getOperand(1));
10369       }
10370   }
10371 
10372   // fold (sext_inreg (extload x)) -> (sextload x)
10373   // If sextload is not supported by target, we can only do the combine when
10374   // load has one use. Doing otherwise can block folding the extload with other
10375   // extends that the target does support.
10376   if (ISD::isEXTLoad(N0.getNode()) &&
10377       ISD::isUNINDEXEDLoad(N0.getNode()) &&
10378       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
10379       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile() &&
10380         N0.hasOneUse()) ||
10381        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
10382     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
10383     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
10384                                      LN0->getChain(),
10385                                      LN0->getBasePtr(), EVT,
10386                                      LN0->getMemOperand());
10387     CombineTo(N, ExtLoad);
10388     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
10389     AddToWorklist(ExtLoad.getNode());
10390     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10391   }
10392   // fold (sext_inreg (zextload x)) -> (sextload x) iff load has one use
10393   if (ISD::isZEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
10394       N0.hasOneUse() &&
10395       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
10396       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
10397        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
10398     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
10399     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
10400                                      LN0->getChain(),
10401                                      LN0->getBasePtr(), EVT,
10402                                      LN0->getMemOperand());
10403     CombineTo(N, ExtLoad);
10404     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
10405     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10406   }
10407 
10408   // Form (sext_inreg (bswap >> 16)) or (sext_inreg (rotl (bswap) 16))
10409   if (EVTBits <= 16 && N0.getOpcode() == ISD::OR) {
10410     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
10411                                            N0.getOperand(1), false))
10412       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
10413                          BSwap, N1);
10414   }
10415 
10416   return SDValue();
10417 }
10418 
10419 SDValue DAGCombiner::visitSIGN_EXTEND_VECTOR_INREG(SDNode *N) {
10420   SDValue N0 = N->getOperand(0);
10421   EVT VT = N->getValueType(0);
10422 
10423   if (N0.isUndef())
10424     return DAG.getUNDEF(VT);
10425 
10426   if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes))
10427     return Res;
10428 
10429   if (SimplifyDemandedVectorElts(SDValue(N, 0)))
10430     return SDValue(N, 0);
10431 
10432   return SDValue();
10433 }
10434 
10435 SDValue DAGCombiner::visitZERO_EXTEND_VECTOR_INREG(SDNode *N) {
10436   SDValue N0 = N->getOperand(0);
10437   EVT VT = N->getValueType(0);
10438 
10439   if (N0.isUndef())
10440     return DAG.getUNDEF(VT);
10441 
10442   if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes))
10443     return Res;
10444 
10445   if (SimplifyDemandedVectorElts(SDValue(N, 0)))
10446     return SDValue(N, 0);
10447 
10448   return SDValue();
10449 }
10450 
10451 SDValue DAGCombiner::visitTRUNCATE(SDNode *N) {
10452   SDValue N0 = N->getOperand(0);
10453   EVT VT = N->getValueType(0);
10454   EVT SrcVT = N0.getValueType();
10455   bool isLE = DAG.getDataLayout().isLittleEndian();
10456 
10457   // noop truncate
10458   if (SrcVT == VT)
10459     return N0;
10460 
10461   // fold (truncate (truncate x)) -> (truncate x)
10462   if (N0.getOpcode() == ISD::TRUNCATE)
10463     return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
10464 
10465   // fold (truncate c1) -> c1
10466   if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
10467     SDValue C = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0);
10468     if (C.getNode() != N)
10469       return C;
10470   }
10471 
10472   // fold (truncate (ext x)) -> (ext x) or (truncate x) or x
10473   if (N0.getOpcode() == ISD::ZERO_EXTEND ||
10474       N0.getOpcode() == ISD::SIGN_EXTEND ||
10475       N0.getOpcode() == ISD::ANY_EXTEND) {
10476     // if the source is smaller than the dest, we still need an extend.
10477     if (N0.getOperand(0).getValueType().bitsLT(VT))
10478       return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
10479     // if the source is larger than the dest, than we just need the truncate.
10480     if (N0.getOperand(0).getValueType().bitsGT(VT))
10481       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
10482     // if the source and dest are the same type, we can drop both the extend
10483     // and the truncate.
10484     return N0.getOperand(0);
10485   }
10486 
10487   // If this is anyext(trunc), don't fold it, allow ourselves to be folded.
10488   if (N->hasOneUse() && (N->use_begin()->getOpcode() == ISD::ANY_EXTEND))
10489     return SDValue();
10490 
10491   // Fold extract-and-trunc into a narrow extract. For example:
10492   //   i64 x = EXTRACT_VECTOR_ELT(v2i64 val, i32 1)
10493   //   i32 y = TRUNCATE(i64 x)
10494   //        -- becomes --
10495   //   v16i8 b = BITCAST (v2i64 val)
10496   //   i8 x = EXTRACT_VECTOR_ELT(v16i8 b, i32 8)
10497   //
10498   // Note: We only run this optimization after type legalization (which often
10499   // creates this pattern) and before operation legalization after which
10500   // we need to be more careful about the vector instructions that we generate.
10501   if (N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
10502       LegalTypes && !LegalOperations && N0->hasOneUse() && VT != MVT::i1) {
10503     EVT VecTy = N0.getOperand(0).getValueType();
10504     EVT ExTy = N0.getValueType();
10505     EVT TrTy = N->getValueType(0);
10506 
10507     unsigned NumElem = VecTy.getVectorNumElements();
10508     unsigned SizeRatio = ExTy.getSizeInBits()/TrTy.getSizeInBits();
10509 
10510     EVT NVT = EVT::getVectorVT(*DAG.getContext(), TrTy, SizeRatio * NumElem);
10511     assert(NVT.getSizeInBits() == VecTy.getSizeInBits() && "Invalid Size");
10512 
10513     SDValue EltNo = N0->getOperand(1);
10514     if (isa<ConstantSDNode>(EltNo) && isTypeLegal(NVT)) {
10515       int Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
10516       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
10517       int Index = isLE ? (Elt*SizeRatio) : (Elt*SizeRatio + (SizeRatio-1));
10518 
10519       SDLoc DL(N);
10520       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, TrTy,
10521                          DAG.getBitcast(NVT, N0.getOperand(0)),
10522                          DAG.getConstant(Index, DL, IndexTy));
10523     }
10524   }
10525 
10526   // trunc (select c, a, b) -> select c, (trunc a), (trunc b)
10527   if (N0.getOpcode() == ISD::SELECT && N0.hasOneUse()) {
10528     if ((!LegalOperations || TLI.isOperationLegal(ISD::SELECT, SrcVT)) &&
10529         TLI.isTruncateFree(SrcVT, VT)) {
10530       SDLoc SL(N0);
10531       SDValue Cond = N0.getOperand(0);
10532       SDValue TruncOp0 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1));
10533       SDValue TruncOp1 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(2));
10534       return DAG.getNode(ISD::SELECT, SDLoc(N), VT, Cond, TruncOp0, TruncOp1);
10535     }
10536   }
10537 
10538   // trunc (shl x, K) -> shl (trunc x), K => K < VT.getScalarSizeInBits()
10539   if (N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
10540       (!LegalOperations || TLI.isOperationLegal(ISD::SHL, VT)) &&
10541       TLI.isTypeDesirableForOp(ISD::SHL, VT)) {
10542     SDValue Amt = N0.getOperand(1);
10543     KnownBits Known = DAG.computeKnownBits(Amt);
10544     unsigned Size = VT.getScalarSizeInBits();
10545     if (Known.getBitWidth() - Known.countMinLeadingZeros() <= Log2_32(Size)) {
10546       SDLoc SL(N);
10547       EVT AmtVT = TLI.getShiftAmountTy(VT, DAG.getDataLayout());
10548 
10549       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0));
10550       if (AmtVT != Amt.getValueType()) {
10551         Amt = DAG.getZExtOrTrunc(Amt, SL, AmtVT);
10552         AddToWorklist(Amt.getNode());
10553       }
10554       return DAG.getNode(ISD::SHL, SL, VT, Trunc, Amt);
10555     }
10556   }
10557 
10558   // Attempt to pre-truncate BUILD_VECTOR sources.
10559   if (N0.getOpcode() == ISD::BUILD_VECTOR && !LegalOperations &&
10560       TLI.isTruncateFree(SrcVT.getScalarType(), VT.getScalarType())) {
10561     SDLoc DL(N);
10562     EVT SVT = VT.getScalarType();
10563     SmallVector<SDValue, 8> TruncOps;
10564     for (const SDValue &Op : N0->op_values()) {
10565       SDValue TruncOp = DAG.getNode(ISD::TRUNCATE, DL, SVT, Op);
10566       TruncOps.push_back(TruncOp);
10567     }
10568     return DAG.getBuildVector(VT, DL, TruncOps);
10569   }
10570 
10571   // Fold a series of buildvector, bitcast, and truncate if possible.
10572   // For example fold
10573   //   (2xi32 trunc (bitcast ((4xi32)buildvector x, x, y, y) 2xi64)) to
10574   //   (2xi32 (buildvector x, y)).
10575   if (Level == AfterLegalizeVectorOps && VT.isVector() &&
10576       N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
10577       N0.getOperand(0).getOpcode() == ISD::BUILD_VECTOR &&
10578       N0.getOperand(0).hasOneUse()) {
10579     SDValue BuildVect = N0.getOperand(0);
10580     EVT BuildVectEltTy = BuildVect.getValueType().getVectorElementType();
10581     EVT TruncVecEltTy = VT.getVectorElementType();
10582 
10583     // Check that the element types match.
10584     if (BuildVectEltTy == TruncVecEltTy) {
10585       // Now we only need to compute the offset of the truncated elements.
10586       unsigned BuildVecNumElts =  BuildVect.getNumOperands();
10587       unsigned TruncVecNumElts = VT.getVectorNumElements();
10588       unsigned TruncEltOffset = BuildVecNumElts / TruncVecNumElts;
10589 
10590       assert((BuildVecNumElts % TruncVecNumElts) == 0 &&
10591              "Invalid number of elements");
10592 
10593       SmallVector<SDValue, 8> Opnds;
10594       for (unsigned i = 0, e = BuildVecNumElts; i != e; i += TruncEltOffset)
10595         Opnds.push_back(BuildVect.getOperand(i));
10596 
10597       return DAG.getBuildVector(VT, SDLoc(N), Opnds);
10598     }
10599   }
10600 
10601   // See if we can simplify the input to this truncate through knowledge that
10602   // only the low bits are being used.
10603   // For example "trunc (or (shl x, 8), y)" // -> trunc y
10604   // Currently we only perform this optimization on scalars because vectors
10605   // may have different active low bits.
10606   if (!VT.isVector()) {
10607     APInt Mask =
10608         APInt::getLowBitsSet(N0.getValueSizeInBits(), VT.getSizeInBits());
10609     if (SDValue Shorter = DAG.GetDemandedBits(N0, Mask))
10610       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Shorter);
10611   }
10612 
10613   // fold (truncate (load x)) -> (smaller load x)
10614   // fold (truncate (srl (load x), c)) -> (smaller load (x+c/evtbits))
10615   if (!LegalTypes || TLI.isTypeDesirableForOp(N0.getOpcode(), VT)) {
10616     if (SDValue Reduced = ReduceLoadWidth(N))
10617       return Reduced;
10618 
10619     // Handle the case where the load remains an extending load even
10620     // after truncation.
10621     if (N0.hasOneUse() && ISD::isUNINDEXEDLoad(N0.getNode())) {
10622       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
10623       if (!LN0->isVolatile() &&
10624           LN0->getMemoryVT().getStoreSizeInBits() < VT.getSizeInBits()) {
10625         SDValue NewLoad = DAG.getExtLoad(LN0->getExtensionType(), SDLoc(LN0),
10626                                          VT, LN0->getChain(), LN0->getBasePtr(),
10627                                          LN0->getMemoryVT(),
10628                                          LN0->getMemOperand());
10629         DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLoad.getValue(1));
10630         return NewLoad;
10631       }
10632     }
10633   }
10634 
10635   // fold (trunc (concat ... x ...)) -> (concat ..., (trunc x), ...)),
10636   // where ... are all 'undef'.
10637   if (N0.getOpcode() == ISD::CONCAT_VECTORS && !LegalTypes) {
10638     SmallVector<EVT, 8> VTs;
10639     SDValue V;
10640     unsigned Idx = 0;
10641     unsigned NumDefs = 0;
10642 
10643     for (unsigned i = 0, e = N0.getNumOperands(); i != e; ++i) {
10644       SDValue X = N0.getOperand(i);
10645       if (!X.isUndef()) {
10646         V = X;
10647         Idx = i;
10648         NumDefs++;
10649       }
10650       // Stop if more than one members are non-undef.
10651       if (NumDefs > 1)
10652         break;
10653       VTs.push_back(EVT::getVectorVT(*DAG.getContext(),
10654                                      VT.getVectorElementType(),
10655                                      X.getValueType().getVectorNumElements()));
10656     }
10657 
10658     if (NumDefs == 0)
10659       return DAG.getUNDEF(VT);
10660 
10661     if (NumDefs == 1) {
10662       assert(V.getNode() && "The single defined operand is empty!");
10663       SmallVector<SDValue, 8> Opnds;
10664       for (unsigned i = 0, e = VTs.size(); i != e; ++i) {
10665         if (i != Idx) {
10666           Opnds.push_back(DAG.getUNDEF(VTs[i]));
10667           continue;
10668         }
10669         SDValue NV = DAG.getNode(ISD::TRUNCATE, SDLoc(V), VTs[i], V);
10670         AddToWorklist(NV.getNode());
10671         Opnds.push_back(NV);
10672       }
10673       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Opnds);
10674     }
10675   }
10676 
10677   // Fold truncate of a bitcast of a vector to an extract of the low vector
10678   // element.
10679   //
10680   // e.g. trunc (i64 (bitcast v2i32:x)) -> extract_vector_elt v2i32:x, idx
10681   if (N0.getOpcode() == ISD::BITCAST && !VT.isVector()) {
10682     SDValue VecSrc = N0.getOperand(0);
10683     EVT SrcVT = VecSrc.getValueType();
10684     if (SrcVT.isVector() && SrcVT.getScalarType() == VT &&
10685         (!LegalOperations ||
10686          TLI.isOperationLegal(ISD::EXTRACT_VECTOR_ELT, SrcVT))) {
10687       SDLoc SL(N);
10688 
10689       EVT IdxVT = TLI.getVectorIdxTy(DAG.getDataLayout());
10690       unsigned Idx = isLE ? 0 : SrcVT.getVectorNumElements() - 1;
10691       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, VT,
10692                          VecSrc, DAG.getConstant(Idx, SL, IdxVT));
10693     }
10694   }
10695 
10696   // Simplify the operands using demanded-bits information.
10697   if (!VT.isVector() &&
10698       SimplifyDemandedBits(SDValue(N, 0)))
10699     return SDValue(N, 0);
10700 
10701   // (trunc adde(X, Y, Carry)) -> (adde trunc(X), trunc(Y), Carry)
10702   // (trunc addcarry(X, Y, Carry)) -> (addcarry trunc(X), trunc(Y), Carry)
10703   // When the adde's carry is not used.
10704   if ((N0.getOpcode() == ISD::ADDE || N0.getOpcode() == ISD::ADDCARRY) &&
10705       N0.hasOneUse() && !N0.getNode()->hasAnyUseOfValue(1) &&
10706       // We only do for addcarry before legalize operation
10707       ((!LegalOperations && N0.getOpcode() == ISD::ADDCARRY) ||
10708        TLI.isOperationLegal(N0.getOpcode(), VT))) {
10709     SDLoc SL(N);
10710     auto X = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0));
10711     auto Y = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1));
10712     auto VTs = DAG.getVTList(VT, N0->getValueType(1));
10713     return DAG.getNode(N0.getOpcode(), SL, VTs, X, Y, N0.getOperand(2));
10714   }
10715 
10716   // fold (truncate (extract_subvector(ext x))) ->
10717   //      (extract_subvector x)
10718   // TODO: This can be generalized to cover cases where the truncate and extract
10719   // do not fully cancel each other out.
10720   if (!LegalTypes && N0.getOpcode() == ISD::EXTRACT_SUBVECTOR) {
10721     SDValue N00 = N0.getOperand(0);
10722     if (N00.getOpcode() == ISD::SIGN_EXTEND ||
10723         N00.getOpcode() == ISD::ZERO_EXTEND ||
10724         N00.getOpcode() == ISD::ANY_EXTEND) {
10725       if (N00.getOperand(0)->getValueType(0).getVectorElementType() ==
10726           VT.getVectorElementType())
10727         return DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(N0->getOperand(0)), VT,
10728                            N00.getOperand(0), N0.getOperand(1));
10729     }
10730   }
10731 
10732   if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N))
10733     return NewVSel;
10734 
10735   // Narrow a suitable binary operation with a non-opaque constant operand by
10736   // moving it ahead of the truncate. This is limited to pre-legalization
10737   // because targets may prefer a wider type during later combines and invert
10738   // this transform.
10739   switch (N0.getOpcode()) {
10740   case ISD::ADD:
10741   case ISD::SUB:
10742   case ISD::MUL:
10743   case ISD::AND:
10744   case ISD::OR:
10745   case ISD::XOR:
10746     if (!LegalOperations && N0.hasOneUse() &&
10747         (isConstantOrConstantVector(N0.getOperand(0), true) ||
10748          isConstantOrConstantVector(N0.getOperand(1), true))) {
10749       // TODO: We already restricted this to pre-legalization, but for vectors
10750       // we are extra cautious to not create an unsupported operation.
10751       // Target-specific changes are likely needed to avoid regressions here.
10752       if (VT.isScalarInteger() || TLI.isOperationLegal(N0.getOpcode(), VT)) {
10753         SDLoc DL(N);
10754         SDValue NarrowL = DAG.getNode(ISD::TRUNCATE, DL, VT, N0.getOperand(0));
10755         SDValue NarrowR = DAG.getNode(ISD::TRUNCATE, DL, VT, N0.getOperand(1));
10756         return DAG.getNode(N0.getOpcode(), DL, VT, NarrowL, NarrowR);
10757       }
10758     }
10759   }
10760 
10761   return SDValue();
10762 }
10763 
10764 static SDNode *getBuildPairElt(SDNode *N, unsigned i) {
10765   SDValue Elt = N->getOperand(i);
10766   if (Elt.getOpcode() != ISD::MERGE_VALUES)
10767     return Elt.getNode();
10768   return Elt.getOperand(Elt.getResNo()).getNode();
10769 }
10770 
10771 /// build_pair (load, load) -> load
10772 /// if load locations are consecutive.
10773 SDValue DAGCombiner::CombineConsecutiveLoads(SDNode *N, EVT VT) {
10774   assert(N->getOpcode() == ISD::BUILD_PAIR);
10775 
10776   LoadSDNode *LD1 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 0));
10777   LoadSDNode *LD2 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 1));
10778 
10779   // A BUILD_PAIR is always having the least significant part in elt 0 and the
10780   // most significant part in elt 1. So when combining into one large load, we
10781   // need to consider the endianness.
10782   if (DAG.getDataLayout().isBigEndian())
10783     std::swap(LD1, LD2);
10784 
10785   if (!LD1 || !LD2 || !ISD::isNON_EXTLoad(LD1) || !LD1->hasOneUse() ||
10786       LD1->getAddressSpace() != LD2->getAddressSpace())
10787     return SDValue();
10788   EVT LD1VT = LD1->getValueType(0);
10789   unsigned LD1Bytes = LD1VT.getStoreSize();
10790   if (ISD::isNON_EXTLoad(LD2) && LD2->hasOneUse() &&
10791       DAG.areNonVolatileConsecutiveLoads(LD2, LD1, LD1Bytes, 1)) {
10792     unsigned Align = LD1->getAlignment();
10793     unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
10794         VT.getTypeForEVT(*DAG.getContext()));
10795 
10796     if (NewAlign <= Align &&
10797         (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT)))
10798       return DAG.getLoad(VT, SDLoc(N), LD1->getChain(), LD1->getBasePtr(),
10799                          LD1->getPointerInfo(), Align);
10800   }
10801 
10802   return SDValue();
10803 }
10804 
10805 static unsigned getPPCf128HiElementSelector(const SelectionDAG &DAG) {
10806   // On little-endian machines, bitcasting from ppcf128 to i128 does swap the Hi
10807   // and Lo parts; on big-endian machines it doesn't.
10808   return DAG.getDataLayout().isBigEndian() ? 1 : 0;
10809 }
10810 
10811 static SDValue foldBitcastedFPLogic(SDNode *N, SelectionDAG &DAG,
10812                                     const TargetLowering &TLI) {
10813   // If this is not a bitcast to an FP type or if the target doesn't have
10814   // IEEE754-compliant FP logic, we're done.
10815   EVT VT = N->getValueType(0);
10816   if (!VT.isFloatingPoint() || !TLI.hasBitPreservingFPLogic(VT))
10817     return SDValue();
10818 
10819   // TODO: Handle cases where the integer constant is a different scalar
10820   // bitwidth to the FP.
10821   SDValue N0 = N->getOperand(0);
10822   EVT SourceVT = N0.getValueType();
10823   if (VT.getScalarSizeInBits() != SourceVT.getScalarSizeInBits())
10824     return SDValue();
10825 
10826   unsigned FPOpcode;
10827   APInt SignMask;
10828   switch (N0.getOpcode()) {
10829   case ISD::AND:
10830     FPOpcode = ISD::FABS;
10831     SignMask = ~APInt::getSignMask(SourceVT.getScalarSizeInBits());
10832     break;
10833   case ISD::XOR:
10834     FPOpcode = ISD::FNEG;
10835     SignMask = APInt::getSignMask(SourceVT.getScalarSizeInBits());
10836     break;
10837   case ISD::OR:
10838     FPOpcode = ISD::FABS;
10839     SignMask = APInt::getSignMask(SourceVT.getScalarSizeInBits());
10840     break;
10841   default:
10842     return SDValue();
10843   }
10844 
10845   // Fold (bitcast int (and (bitcast fp X to int), 0x7fff...) to fp) -> fabs X
10846   // Fold (bitcast int (xor (bitcast fp X to int), 0x8000...) to fp) -> fneg X
10847   // Fold (bitcast int (or (bitcast fp X to int), 0x8000...) to fp) ->
10848   //   fneg (fabs X)
10849   SDValue LogicOp0 = N0.getOperand(0);
10850   ConstantSDNode *LogicOp1 = isConstOrConstSplat(N0.getOperand(1), true);
10851   if (LogicOp1 && LogicOp1->getAPIntValue() == SignMask &&
10852       LogicOp0.getOpcode() == ISD::BITCAST &&
10853       LogicOp0.getOperand(0).getValueType() == VT) {
10854     SDValue FPOp = DAG.getNode(FPOpcode, SDLoc(N), VT, LogicOp0.getOperand(0));
10855     NumFPLogicOpsConv++;
10856     if (N0.getOpcode() == ISD::OR)
10857       return DAG.getNode(ISD::FNEG, SDLoc(N), VT, FPOp);
10858     return FPOp;
10859   }
10860 
10861   return SDValue();
10862 }
10863 
10864 SDValue DAGCombiner::visitBITCAST(SDNode *N) {
10865   SDValue N0 = N->getOperand(0);
10866   EVT VT = N->getValueType(0);
10867 
10868   if (N0.isUndef())
10869     return DAG.getUNDEF(VT);
10870 
10871   // If the input is a BUILD_VECTOR with all constant elements, fold this now.
10872   // Only do this before legalize types, unless both types are integer and the
10873   // scalar type is legal. Only do this before legalize ops, since the target
10874   // maybe depending on the bitcast.
10875   // First check to see if this is all constant.
10876   // TODO: Support FP bitcasts after legalize types.
10877   if (VT.isVector() &&
10878       (!LegalTypes ||
10879        (!LegalOperations && VT.isInteger() && N0.getValueType().isInteger() &&
10880         TLI.isTypeLegal(VT.getVectorElementType()))) &&
10881       N0.getOpcode() == ISD::BUILD_VECTOR && N0.getNode()->hasOneUse() &&
10882       cast<BuildVectorSDNode>(N0)->isConstant())
10883     return ConstantFoldBITCASTofBUILD_VECTOR(N0.getNode(),
10884                                              VT.getVectorElementType());
10885 
10886   // If the input is a constant, let getNode fold it.
10887   if (isa<ConstantSDNode>(N0) || isa<ConstantFPSDNode>(N0)) {
10888     // If we can't allow illegal operations, we need to check that this is just
10889     // a fp -> int or int -> conversion and that the resulting operation will
10890     // be legal.
10891     if (!LegalOperations ||
10892         (isa<ConstantSDNode>(N0) && VT.isFloatingPoint() && !VT.isVector() &&
10893          TLI.isOperationLegal(ISD::ConstantFP, VT)) ||
10894         (isa<ConstantFPSDNode>(N0) && VT.isInteger() && !VT.isVector() &&
10895          TLI.isOperationLegal(ISD::Constant, VT))) {
10896       SDValue C = DAG.getBitcast(VT, N0);
10897       if (C.getNode() != N)
10898         return C;
10899     }
10900   }
10901 
10902   // (conv (conv x, t1), t2) -> (conv x, t2)
10903   if (N0.getOpcode() == ISD::BITCAST)
10904     return DAG.getBitcast(VT, N0.getOperand(0));
10905 
10906   // fold (conv (load x)) -> (load (conv*)x)
10907   // If the resultant load doesn't need a higher alignment than the original!
10908   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
10909       // Do not remove the cast if the types differ in endian layout.
10910       TLI.hasBigEndianPartOrdering(N0.getValueType(), DAG.getDataLayout()) ==
10911           TLI.hasBigEndianPartOrdering(VT, DAG.getDataLayout()) &&
10912       // If the load is volatile, we only want to change the load type if the
10913       // resulting load is legal. Otherwise we might increase the number of
10914       // memory accesses. We don't care if the original type was legal or not
10915       // as we assume software couldn't rely on the number of accesses of an
10916       // illegal type.
10917       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
10918        TLI.isOperationLegal(ISD::LOAD, VT)) &&
10919       TLI.isLoadBitCastBeneficial(N0.getValueType(), VT)) {
10920     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
10921 
10922     bool Fast = false;
10923     if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT,
10924                                *LN0->getMemOperand(), &Fast) &&
10925         Fast) {
10926       SDValue Load =
10927           DAG.getLoad(VT, SDLoc(N), LN0->getChain(), LN0->getBasePtr(),
10928                       LN0->getPointerInfo(), LN0->getAlignment(),
10929                       LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
10930       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
10931       return Load;
10932     }
10933   }
10934 
10935   if (SDValue V = foldBitcastedFPLogic(N, DAG, TLI))
10936     return V;
10937 
10938   // fold (bitconvert (fneg x)) -> (xor (bitconvert x), signbit)
10939   // fold (bitconvert (fabs x)) -> (and (bitconvert x), (not signbit))
10940   //
10941   // For ppc_fp128:
10942   // fold (bitcast (fneg x)) ->
10943   //     flipbit = signbit
10944   //     (xor (bitcast x) (build_pair flipbit, flipbit))
10945   //
10946   // fold (bitcast (fabs x)) ->
10947   //     flipbit = (and (extract_element (bitcast x), 0), signbit)
10948   //     (xor (bitcast x) (build_pair flipbit, flipbit))
10949   // This often reduces constant pool loads.
10950   if (((N0.getOpcode() == ISD::FNEG && !TLI.isFNegFree(N0.getValueType())) ||
10951        (N0.getOpcode() == ISD::FABS && !TLI.isFAbsFree(N0.getValueType()))) &&
10952       N0.getNode()->hasOneUse() && VT.isInteger() &&
10953       !VT.isVector() && !N0.getValueType().isVector()) {
10954     SDValue NewConv = DAG.getBitcast(VT, N0.getOperand(0));
10955     AddToWorklist(NewConv.getNode());
10956 
10957     SDLoc DL(N);
10958     if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
10959       assert(VT.getSizeInBits() == 128);
10960       SDValue SignBit = DAG.getConstant(
10961           APInt::getSignMask(VT.getSizeInBits() / 2), SDLoc(N0), MVT::i64);
10962       SDValue FlipBit;
10963       if (N0.getOpcode() == ISD::FNEG) {
10964         FlipBit = SignBit;
10965         AddToWorklist(FlipBit.getNode());
10966       } else {
10967         assert(N0.getOpcode() == ISD::FABS);
10968         SDValue Hi =
10969             DAG.getNode(ISD::EXTRACT_ELEMENT, SDLoc(NewConv), MVT::i64, NewConv,
10970                         DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
10971                                               SDLoc(NewConv)));
10972         AddToWorklist(Hi.getNode());
10973         FlipBit = DAG.getNode(ISD::AND, SDLoc(N0), MVT::i64, Hi, SignBit);
10974         AddToWorklist(FlipBit.getNode());
10975       }
10976       SDValue FlipBits =
10977           DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
10978       AddToWorklist(FlipBits.getNode());
10979       return DAG.getNode(ISD::XOR, DL, VT, NewConv, FlipBits);
10980     }
10981     APInt SignBit = APInt::getSignMask(VT.getSizeInBits());
10982     if (N0.getOpcode() == ISD::FNEG)
10983       return DAG.getNode(ISD::XOR, DL, VT,
10984                          NewConv, DAG.getConstant(SignBit, DL, VT));
10985     assert(N0.getOpcode() == ISD::FABS);
10986     return DAG.getNode(ISD::AND, DL, VT,
10987                        NewConv, DAG.getConstant(~SignBit, DL, VT));
10988   }
10989 
10990   // fold (bitconvert (fcopysign cst, x)) ->
10991   //         (or (and (bitconvert x), sign), (and cst, (not sign)))
10992   // Note that we don't handle (copysign x, cst) because this can always be
10993   // folded to an fneg or fabs.
10994   //
10995   // For ppc_fp128:
10996   // fold (bitcast (fcopysign cst, x)) ->
10997   //     flipbit = (and (extract_element
10998   //                     (xor (bitcast cst), (bitcast x)), 0),
10999   //                    signbit)
11000   //     (xor (bitcast cst) (build_pair flipbit, flipbit))
11001   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse() &&
11002       isa<ConstantFPSDNode>(N0.getOperand(0)) &&
11003       VT.isInteger() && !VT.isVector()) {
11004     unsigned OrigXWidth = N0.getOperand(1).getValueSizeInBits();
11005     EVT IntXVT = EVT::getIntegerVT(*DAG.getContext(), OrigXWidth);
11006     if (isTypeLegal(IntXVT)) {
11007       SDValue X = DAG.getBitcast(IntXVT, N0.getOperand(1));
11008       AddToWorklist(X.getNode());
11009 
11010       // If X has a different width than the result/lhs, sext it or truncate it.
11011       unsigned VTWidth = VT.getSizeInBits();
11012       if (OrigXWidth < VTWidth) {
11013         X = DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, X);
11014         AddToWorklist(X.getNode());
11015       } else if (OrigXWidth > VTWidth) {
11016         // To get the sign bit in the right place, we have to shift it right
11017         // before truncating.
11018         SDLoc DL(X);
11019         X = DAG.getNode(ISD::SRL, DL,
11020                         X.getValueType(), X,
11021                         DAG.getConstant(OrigXWidth-VTWidth, DL,
11022                                         X.getValueType()));
11023         AddToWorklist(X.getNode());
11024         X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
11025         AddToWorklist(X.getNode());
11026       }
11027 
11028       if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
11029         APInt SignBit = APInt::getSignMask(VT.getSizeInBits() / 2);
11030         SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
11031         AddToWorklist(Cst.getNode());
11032         SDValue X = DAG.getBitcast(VT, N0.getOperand(1));
11033         AddToWorklist(X.getNode());
11034         SDValue XorResult = DAG.getNode(ISD::XOR, SDLoc(N0), VT, Cst, X);
11035         AddToWorklist(XorResult.getNode());
11036         SDValue XorResult64 = DAG.getNode(
11037             ISD::EXTRACT_ELEMENT, SDLoc(XorResult), MVT::i64, XorResult,
11038             DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
11039                                   SDLoc(XorResult)));
11040         AddToWorklist(XorResult64.getNode());
11041         SDValue FlipBit =
11042             DAG.getNode(ISD::AND, SDLoc(XorResult64), MVT::i64, XorResult64,
11043                         DAG.getConstant(SignBit, SDLoc(XorResult64), MVT::i64));
11044         AddToWorklist(FlipBit.getNode());
11045         SDValue FlipBits =
11046             DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
11047         AddToWorklist(FlipBits.getNode());
11048         return DAG.getNode(ISD::XOR, SDLoc(N), VT, Cst, FlipBits);
11049       }
11050       APInt SignBit = APInt::getSignMask(VT.getSizeInBits());
11051       X = DAG.getNode(ISD::AND, SDLoc(X), VT,
11052                       X, DAG.getConstant(SignBit, SDLoc(X), VT));
11053       AddToWorklist(X.getNode());
11054 
11055       SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
11056       Cst = DAG.getNode(ISD::AND, SDLoc(Cst), VT,
11057                         Cst, DAG.getConstant(~SignBit, SDLoc(Cst), VT));
11058       AddToWorklist(Cst.getNode());
11059 
11060       return DAG.getNode(ISD::OR, SDLoc(N), VT, X, Cst);
11061     }
11062   }
11063 
11064   // bitconvert(build_pair(ld, ld)) -> ld iff load locations are consecutive.
11065   if (N0.getOpcode() == ISD::BUILD_PAIR)
11066     if (SDValue CombineLD = CombineConsecutiveLoads(N0.getNode(), VT))
11067       return CombineLD;
11068 
11069   // Remove double bitcasts from shuffles - this is often a legacy of
11070   // XformToShuffleWithZero being used to combine bitmaskings (of
11071   // float vectors bitcast to integer vectors) into shuffles.
11072   // bitcast(shuffle(bitcast(s0),bitcast(s1))) -> shuffle(s0,s1)
11073   if (Level < AfterLegalizeDAG && TLI.isTypeLegal(VT) && VT.isVector() &&
11074       N0->getOpcode() == ISD::VECTOR_SHUFFLE && N0.hasOneUse() &&
11075       VT.getVectorNumElements() >= N0.getValueType().getVectorNumElements() &&
11076       !(VT.getVectorNumElements() % N0.getValueType().getVectorNumElements())) {
11077     ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N0);
11078 
11079     // If operands are a bitcast, peek through if it casts the original VT.
11080     // If operands are a constant, just bitcast back to original VT.
11081     auto PeekThroughBitcast = [&](SDValue Op) {
11082       if (Op.getOpcode() == ISD::BITCAST &&
11083           Op.getOperand(0).getValueType() == VT)
11084         return SDValue(Op.getOperand(0));
11085       if (Op.isUndef() || ISD::isBuildVectorOfConstantSDNodes(Op.getNode()) ||
11086           ISD::isBuildVectorOfConstantFPSDNodes(Op.getNode()))
11087         return DAG.getBitcast(VT, Op);
11088       return SDValue();
11089     };
11090 
11091     // FIXME: If either input vector is bitcast, try to convert the shuffle to
11092     // the result type of this bitcast. This would eliminate at least one
11093     // bitcast. See the transform in InstCombine.
11094     SDValue SV0 = PeekThroughBitcast(N0->getOperand(0));
11095     SDValue SV1 = PeekThroughBitcast(N0->getOperand(1));
11096     if (!(SV0 && SV1))
11097       return SDValue();
11098 
11099     int MaskScale =
11100         VT.getVectorNumElements() / N0.getValueType().getVectorNumElements();
11101     SmallVector<int, 8> NewMask;
11102     for (int M : SVN->getMask())
11103       for (int i = 0; i != MaskScale; ++i)
11104         NewMask.push_back(M < 0 ? -1 : M * MaskScale + i);
11105 
11106     bool LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
11107     if (!LegalMask) {
11108       std::swap(SV0, SV1);
11109       ShuffleVectorSDNode::commuteMask(NewMask);
11110       LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
11111     }
11112 
11113     if (LegalMask)
11114       return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, NewMask);
11115   }
11116 
11117   return SDValue();
11118 }
11119 
11120 SDValue DAGCombiner::visitBUILD_PAIR(SDNode *N) {
11121   EVT VT = N->getValueType(0);
11122   return CombineConsecutiveLoads(N, VT);
11123 }
11124 
11125 /// We know that BV is a build_vector node with Constant, ConstantFP or Undef
11126 /// operands. DstEltVT indicates the destination element value type.
11127 SDValue DAGCombiner::
11128 ConstantFoldBITCASTofBUILD_VECTOR(SDNode *BV, EVT DstEltVT) {
11129   EVT SrcEltVT = BV->getValueType(0).getVectorElementType();
11130 
11131   // If this is already the right type, we're done.
11132   if (SrcEltVT == DstEltVT) return SDValue(BV, 0);
11133 
11134   unsigned SrcBitSize = SrcEltVT.getSizeInBits();
11135   unsigned DstBitSize = DstEltVT.getSizeInBits();
11136 
11137   // If this is a conversion of N elements of one type to N elements of another
11138   // type, convert each element.  This handles FP<->INT cases.
11139   if (SrcBitSize == DstBitSize) {
11140     SmallVector<SDValue, 8> Ops;
11141     for (SDValue Op : BV->op_values()) {
11142       // If the vector element type is not legal, the BUILD_VECTOR operands
11143       // are promoted and implicitly truncated.  Make that explicit here.
11144       if (Op.getValueType() != SrcEltVT)
11145         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(BV), SrcEltVT, Op);
11146       Ops.push_back(DAG.getBitcast(DstEltVT, Op));
11147       AddToWorklist(Ops.back().getNode());
11148     }
11149     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
11150                               BV->getValueType(0).getVectorNumElements());
11151     return DAG.getBuildVector(VT, SDLoc(BV), Ops);
11152   }
11153 
11154   // Otherwise, we're growing or shrinking the elements.  To avoid having to
11155   // handle annoying details of growing/shrinking FP values, we convert them to
11156   // int first.
11157   if (SrcEltVT.isFloatingPoint()) {
11158     // Convert the input float vector to a int vector where the elements are the
11159     // same sizes.
11160     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), SrcEltVT.getSizeInBits());
11161     BV = ConstantFoldBITCASTofBUILD_VECTOR(BV, IntVT).getNode();
11162     SrcEltVT = IntVT;
11163   }
11164 
11165   // Now we know the input is an integer vector.  If the output is a FP type,
11166   // convert to integer first, then to FP of the right size.
11167   if (DstEltVT.isFloatingPoint()) {
11168     EVT TmpVT = EVT::getIntegerVT(*DAG.getContext(), DstEltVT.getSizeInBits());
11169     SDNode *Tmp = ConstantFoldBITCASTofBUILD_VECTOR(BV, TmpVT).getNode();
11170 
11171     // Next, convert to FP elements of the same size.
11172     return ConstantFoldBITCASTofBUILD_VECTOR(Tmp, DstEltVT);
11173   }
11174 
11175   SDLoc DL(BV);
11176 
11177   // Okay, we know the src/dst types are both integers of differing types.
11178   // Handling growing first.
11179   assert(SrcEltVT.isInteger() && DstEltVT.isInteger());
11180   if (SrcBitSize < DstBitSize) {
11181     unsigned NumInputsPerOutput = DstBitSize/SrcBitSize;
11182 
11183     SmallVector<SDValue, 8> Ops;
11184     for (unsigned i = 0, e = BV->getNumOperands(); i != e;
11185          i += NumInputsPerOutput) {
11186       bool isLE = DAG.getDataLayout().isLittleEndian();
11187       APInt NewBits = APInt(DstBitSize, 0);
11188       bool EltIsUndef = true;
11189       for (unsigned j = 0; j != NumInputsPerOutput; ++j) {
11190         // Shift the previously computed bits over.
11191         NewBits <<= SrcBitSize;
11192         SDValue Op = BV->getOperand(i+ (isLE ? (NumInputsPerOutput-j-1) : j));
11193         if (Op.isUndef()) continue;
11194         EltIsUndef = false;
11195 
11196         NewBits |= cast<ConstantSDNode>(Op)->getAPIntValue().
11197                    zextOrTrunc(SrcBitSize).zext(DstBitSize);
11198       }
11199 
11200       if (EltIsUndef)
11201         Ops.push_back(DAG.getUNDEF(DstEltVT));
11202       else
11203         Ops.push_back(DAG.getConstant(NewBits, DL, DstEltVT));
11204     }
11205 
11206     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT, Ops.size());
11207     return DAG.getBuildVector(VT, DL, Ops);
11208   }
11209 
11210   // Finally, this must be the case where we are shrinking elements: each input
11211   // turns into multiple outputs.
11212   unsigned NumOutputsPerInput = SrcBitSize/DstBitSize;
11213   EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
11214                             NumOutputsPerInput*BV->getNumOperands());
11215   SmallVector<SDValue, 8> Ops;
11216 
11217   for (const SDValue &Op : BV->op_values()) {
11218     if (Op.isUndef()) {
11219       Ops.append(NumOutputsPerInput, DAG.getUNDEF(DstEltVT));
11220       continue;
11221     }
11222 
11223     APInt OpVal = cast<ConstantSDNode>(Op)->
11224                   getAPIntValue().zextOrTrunc(SrcBitSize);
11225 
11226     for (unsigned j = 0; j != NumOutputsPerInput; ++j) {
11227       APInt ThisVal = OpVal.trunc(DstBitSize);
11228       Ops.push_back(DAG.getConstant(ThisVal, DL, DstEltVT));
11229       OpVal.lshrInPlace(DstBitSize);
11230     }
11231 
11232     // For big endian targets, swap the order of the pieces of each element.
11233     if (DAG.getDataLayout().isBigEndian())
11234       std::reverse(Ops.end()-NumOutputsPerInput, Ops.end());
11235   }
11236 
11237   return DAG.getBuildVector(VT, DL, Ops);
11238 }
11239 
11240 static bool isContractable(SDNode *N) {
11241   SDNodeFlags F = N->getFlags();
11242   return F.hasAllowContract() || F.hasAllowReassociation();
11243 }
11244 
11245 /// Try to perform FMA combining on a given FADD node.
11246 SDValue DAGCombiner::visitFADDForFMACombine(SDNode *N) {
11247   SDValue N0 = N->getOperand(0);
11248   SDValue N1 = N->getOperand(1);
11249   EVT VT = N->getValueType(0);
11250   SDLoc SL(N);
11251 
11252   const TargetOptions &Options = DAG.getTarget().Options;
11253 
11254   // Floating-point multiply-add with intermediate rounding.
11255   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
11256 
11257   // Floating-point multiply-add without intermediate rounding.
11258   bool HasFMA =
11259       TLI.isFMAFasterThanFMulAndFAdd(VT) &&
11260       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
11261 
11262   // No valid opcode, do not combine.
11263   if (!HasFMAD && !HasFMA)
11264     return SDValue();
11265 
11266   SDNodeFlags Flags = N->getFlags();
11267   bool CanFuse = Options.UnsafeFPMath || isContractable(N);
11268   bool AllowFusionGlobally = (Options.AllowFPOpFusion == FPOpFusion::Fast ||
11269                               CanFuse || HasFMAD);
11270   // If the addition is not contractable, do not combine.
11271   if (!AllowFusionGlobally && !isContractable(N))
11272     return SDValue();
11273 
11274   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
11275   if (STI && STI->generateFMAsInMachineCombiner(OptLevel))
11276     return SDValue();
11277 
11278   // Always prefer FMAD to FMA for precision.
11279   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
11280   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
11281 
11282   // Is the node an FMUL and contractable either due to global flags or
11283   // SDNodeFlags.
11284   auto isContractableFMUL = [AllowFusionGlobally](SDValue N) {
11285     if (N.getOpcode() != ISD::FMUL)
11286       return false;
11287     return AllowFusionGlobally || isContractable(N.getNode());
11288   };
11289   // If we have two choices trying to fold (fadd (fmul u, v), (fmul x, y)),
11290   // prefer to fold the multiply with fewer uses.
11291   if (Aggressive && isContractableFMUL(N0) && isContractableFMUL(N1)) {
11292     if (N0.getNode()->use_size() > N1.getNode()->use_size())
11293       std::swap(N0, N1);
11294   }
11295 
11296   // fold (fadd (fmul x, y), z) -> (fma x, y, z)
11297   if (isContractableFMUL(N0) && (Aggressive || N0->hasOneUse())) {
11298     return DAG.getNode(PreferredFusedOpcode, SL, VT,
11299                        N0.getOperand(0), N0.getOperand(1), N1, Flags);
11300   }
11301 
11302   // fold (fadd x, (fmul y, z)) -> (fma y, z, x)
11303   // Note: Commutes FADD operands.
11304   if (isContractableFMUL(N1) && (Aggressive || N1->hasOneUse())) {
11305     return DAG.getNode(PreferredFusedOpcode, SL, VT,
11306                        N1.getOperand(0), N1.getOperand(1), N0, Flags);
11307   }
11308 
11309   // Look through FP_EXTEND nodes to do more combining.
11310 
11311   // fold (fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext y), z)
11312   if (N0.getOpcode() == ISD::FP_EXTEND) {
11313     SDValue N00 = N0.getOperand(0);
11314     if (isContractableFMUL(N00) &&
11315         TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) {
11316       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11317                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11318                                      N00.getOperand(0)),
11319                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11320                                      N00.getOperand(1)), N1, Flags);
11321     }
11322   }
11323 
11324   // fold (fadd x, (fpext (fmul y, z))) -> (fma (fpext y), (fpext z), x)
11325   // Note: Commutes FADD operands.
11326   if (N1.getOpcode() == ISD::FP_EXTEND) {
11327     SDValue N10 = N1.getOperand(0);
11328     if (isContractableFMUL(N10) &&
11329         TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N10.getValueType())) {
11330       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11331                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11332                                      N10.getOperand(0)),
11333                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11334                                      N10.getOperand(1)), N0, Flags);
11335     }
11336   }
11337 
11338   // More folding opportunities when target permits.
11339   if (Aggressive) {
11340     // fold (fadd (fma x, y, (fmul u, v)), z) -> (fma x, y (fma u, v, z))
11341     if (CanFuse &&
11342         N0.getOpcode() == PreferredFusedOpcode &&
11343         N0.getOperand(2).getOpcode() == ISD::FMUL &&
11344         N0->hasOneUse() && N0.getOperand(2)->hasOneUse()) {
11345       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11346                          N0.getOperand(0), N0.getOperand(1),
11347                          DAG.getNode(PreferredFusedOpcode, SL, VT,
11348                                      N0.getOperand(2).getOperand(0),
11349                                      N0.getOperand(2).getOperand(1),
11350                                      N1, Flags), Flags);
11351     }
11352 
11353     // fold (fadd x, (fma y, z, (fmul u, v)) -> (fma y, z (fma u, v, x))
11354     if (CanFuse &&
11355         N1->getOpcode() == PreferredFusedOpcode &&
11356         N1.getOperand(2).getOpcode() == ISD::FMUL &&
11357         N1->hasOneUse() && N1.getOperand(2)->hasOneUse()) {
11358       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11359                          N1.getOperand(0), N1.getOperand(1),
11360                          DAG.getNode(PreferredFusedOpcode, SL, VT,
11361                                      N1.getOperand(2).getOperand(0),
11362                                      N1.getOperand(2).getOperand(1),
11363                                      N0, Flags), Flags);
11364     }
11365 
11366 
11367     // fold (fadd (fma x, y, (fpext (fmul u, v))), z)
11368     //   -> (fma x, y, (fma (fpext u), (fpext v), z))
11369     auto FoldFAddFMAFPExtFMul = [&] (
11370       SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z,
11371       SDNodeFlags Flags) {
11372       return DAG.getNode(PreferredFusedOpcode, SL, VT, X, Y,
11373                          DAG.getNode(PreferredFusedOpcode, SL, VT,
11374                                      DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
11375                                      DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
11376                                      Z, Flags), Flags);
11377     };
11378     if (N0.getOpcode() == PreferredFusedOpcode) {
11379       SDValue N02 = N0.getOperand(2);
11380       if (N02.getOpcode() == ISD::FP_EXTEND) {
11381         SDValue N020 = N02.getOperand(0);
11382         if (isContractableFMUL(N020) &&
11383             TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N020.getValueType())) {
11384           return FoldFAddFMAFPExtFMul(N0.getOperand(0), N0.getOperand(1),
11385                                       N020.getOperand(0), N020.getOperand(1),
11386                                       N1, Flags);
11387         }
11388       }
11389     }
11390 
11391     // fold (fadd (fpext (fma x, y, (fmul u, v))), z)
11392     //   -> (fma (fpext x), (fpext y), (fma (fpext u), (fpext v), z))
11393     // FIXME: This turns two single-precision and one double-precision
11394     // operation into two double-precision operations, which might not be
11395     // interesting for all targets, especially GPUs.
11396     auto FoldFAddFPExtFMAFMul = [&] (
11397       SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z,
11398       SDNodeFlags Flags) {
11399       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11400                          DAG.getNode(ISD::FP_EXTEND, SL, VT, X),
11401                          DAG.getNode(ISD::FP_EXTEND, SL, VT, Y),
11402                          DAG.getNode(PreferredFusedOpcode, SL, VT,
11403                                      DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
11404                                      DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
11405                                      Z, Flags), Flags);
11406     };
11407     if (N0.getOpcode() == ISD::FP_EXTEND) {
11408       SDValue N00 = N0.getOperand(0);
11409       if (N00.getOpcode() == PreferredFusedOpcode) {
11410         SDValue N002 = N00.getOperand(2);
11411         if (isContractableFMUL(N002) &&
11412             TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) {
11413           return FoldFAddFPExtFMAFMul(N00.getOperand(0), N00.getOperand(1),
11414                                       N002.getOperand(0), N002.getOperand(1),
11415                                       N1, Flags);
11416         }
11417       }
11418     }
11419 
11420     // fold (fadd x, (fma y, z, (fpext (fmul u, v)))
11421     //   -> (fma y, z, (fma (fpext u), (fpext v), x))
11422     if (N1.getOpcode() == PreferredFusedOpcode) {
11423       SDValue N12 = N1.getOperand(2);
11424       if (N12.getOpcode() == ISD::FP_EXTEND) {
11425         SDValue N120 = N12.getOperand(0);
11426         if (isContractableFMUL(N120) &&
11427             TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N120.getValueType())) {
11428           return FoldFAddFMAFPExtFMul(N1.getOperand(0), N1.getOperand(1),
11429                                       N120.getOperand(0), N120.getOperand(1),
11430                                       N0, Flags);
11431         }
11432       }
11433     }
11434 
11435     // fold (fadd x, (fpext (fma y, z, (fmul u, v)))
11436     //   -> (fma (fpext y), (fpext z), (fma (fpext u), (fpext v), x))
11437     // FIXME: This turns two single-precision and one double-precision
11438     // operation into two double-precision operations, which might not be
11439     // interesting for all targets, especially GPUs.
11440     if (N1.getOpcode() == ISD::FP_EXTEND) {
11441       SDValue N10 = N1.getOperand(0);
11442       if (N10.getOpcode() == PreferredFusedOpcode) {
11443         SDValue N102 = N10.getOperand(2);
11444         if (isContractableFMUL(N102) &&
11445             TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N10.getValueType())) {
11446           return FoldFAddFPExtFMAFMul(N10.getOperand(0), N10.getOperand(1),
11447                                       N102.getOperand(0), N102.getOperand(1),
11448                                       N0, Flags);
11449         }
11450       }
11451     }
11452   }
11453 
11454   return SDValue();
11455 }
11456 
11457 /// Try to perform FMA combining on a given FSUB node.
11458 SDValue DAGCombiner::visitFSUBForFMACombine(SDNode *N) {
11459   SDValue N0 = N->getOperand(0);
11460   SDValue N1 = N->getOperand(1);
11461   EVT VT = N->getValueType(0);
11462   SDLoc SL(N);
11463 
11464   const TargetOptions &Options = DAG.getTarget().Options;
11465   // Floating-point multiply-add with intermediate rounding.
11466   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
11467 
11468   // Floating-point multiply-add without intermediate rounding.
11469   bool HasFMA =
11470       TLI.isFMAFasterThanFMulAndFAdd(VT) &&
11471       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
11472 
11473   // No valid opcode, do not combine.
11474   if (!HasFMAD && !HasFMA)
11475     return SDValue();
11476 
11477   const SDNodeFlags Flags = N->getFlags();
11478   bool CanFuse = Options.UnsafeFPMath || isContractable(N);
11479   bool AllowFusionGlobally = (Options.AllowFPOpFusion == FPOpFusion::Fast ||
11480                               CanFuse || HasFMAD);
11481 
11482   // If the subtraction is not contractable, do not combine.
11483   if (!AllowFusionGlobally && !isContractable(N))
11484     return SDValue();
11485 
11486   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
11487   if (STI && STI->generateFMAsInMachineCombiner(OptLevel))
11488     return SDValue();
11489 
11490   // Always prefer FMAD to FMA for precision.
11491   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
11492   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
11493 
11494   // Is the node an FMUL and contractable either due to global flags or
11495   // SDNodeFlags.
11496   auto isContractableFMUL = [AllowFusionGlobally](SDValue N) {
11497     if (N.getOpcode() != ISD::FMUL)
11498       return false;
11499     return AllowFusionGlobally || isContractable(N.getNode());
11500   };
11501 
11502   // fold (fsub (fmul x, y), z) -> (fma x, y, (fneg z))
11503   if (isContractableFMUL(N0) && (Aggressive || N0->hasOneUse())) {
11504     return DAG.getNode(PreferredFusedOpcode, SL, VT,
11505                        N0.getOperand(0), N0.getOperand(1),
11506                        DAG.getNode(ISD::FNEG, SL, VT, N1), Flags);
11507   }
11508 
11509   // fold (fsub x, (fmul y, z)) -> (fma (fneg y), z, x)
11510   // Note: Commutes FSUB operands.
11511   if (isContractableFMUL(N1) && (Aggressive || N1->hasOneUse())) {
11512     return DAG.getNode(PreferredFusedOpcode, SL, VT,
11513                        DAG.getNode(ISD::FNEG, SL, VT,
11514                                    N1.getOperand(0)),
11515                        N1.getOperand(1), N0, Flags);
11516   }
11517 
11518   // fold (fsub (fneg (fmul, x, y)), z) -> (fma (fneg x), y, (fneg z))
11519   if (N0.getOpcode() == ISD::FNEG && isContractableFMUL(N0.getOperand(0)) &&
11520       (Aggressive || (N0->hasOneUse() && N0.getOperand(0).hasOneUse()))) {
11521     SDValue N00 = N0.getOperand(0).getOperand(0);
11522     SDValue N01 = N0.getOperand(0).getOperand(1);
11523     return DAG.getNode(PreferredFusedOpcode, SL, VT,
11524                        DAG.getNode(ISD::FNEG, SL, VT, N00), N01,
11525                        DAG.getNode(ISD::FNEG, SL, VT, N1), Flags);
11526   }
11527 
11528   // Look through FP_EXTEND nodes to do more combining.
11529 
11530   // fold (fsub (fpext (fmul x, y)), z)
11531   //   -> (fma (fpext x), (fpext y), (fneg z))
11532   if (N0.getOpcode() == ISD::FP_EXTEND) {
11533     SDValue N00 = N0.getOperand(0);
11534     if (isContractableFMUL(N00) &&
11535         TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) {
11536       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11537                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11538                                      N00.getOperand(0)),
11539                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11540                                      N00.getOperand(1)),
11541                          DAG.getNode(ISD::FNEG, SL, VT, N1), Flags);
11542     }
11543   }
11544 
11545   // fold (fsub x, (fpext (fmul y, z)))
11546   //   -> (fma (fneg (fpext y)), (fpext z), x)
11547   // Note: Commutes FSUB operands.
11548   if (N1.getOpcode() == ISD::FP_EXTEND) {
11549     SDValue N10 = N1.getOperand(0);
11550     if (isContractableFMUL(N10) &&
11551         TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N10.getValueType())) {
11552       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11553                          DAG.getNode(ISD::FNEG, SL, VT,
11554                                      DAG.getNode(ISD::FP_EXTEND, SL, VT,
11555                                                  N10.getOperand(0))),
11556                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11557                                      N10.getOperand(1)),
11558                          N0, Flags);
11559     }
11560   }
11561 
11562   // fold (fsub (fpext (fneg (fmul, x, y))), z)
11563   //   -> (fneg (fma (fpext x), (fpext y), z))
11564   // Note: This could be removed with appropriate canonicalization of the
11565   // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
11566   // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
11567   // from implementing the canonicalization in visitFSUB.
11568   if (N0.getOpcode() == ISD::FP_EXTEND) {
11569     SDValue N00 = N0.getOperand(0);
11570     if (N00.getOpcode() == ISD::FNEG) {
11571       SDValue N000 = N00.getOperand(0);
11572       if (isContractableFMUL(N000) &&
11573           TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) {
11574         return DAG.getNode(ISD::FNEG, SL, VT,
11575                            DAG.getNode(PreferredFusedOpcode, SL, VT,
11576                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
11577                                                    N000.getOperand(0)),
11578                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
11579                                                    N000.getOperand(1)),
11580                                        N1, Flags));
11581       }
11582     }
11583   }
11584 
11585   // fold (fsub (fneg (fpext (fmul, x, y))), z)
11586   //   -> (fneg (fma (fpext x)), (fpext y), z)
11587   // Note: This could be removed with appropriate canonicalization of the
11588   // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
11589   // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
11590   // from implementing the canonicalization in visitFSUB.
11591   if (N0.getOpcode() == ISD::FNEG) {
11592     SDValue N00 = N0.getOperand(0);
11593     if (N00.getOpcode() == ISD::FP_EXTEND) {
11594       SDValue N000 = N00.getOperand(0);
11595       if (isContractableFMUL(N000) &&
11596           TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N000.getValueType())) {
11597         return DAG.getNode(ISD::FNEG, SL, VT,
11598                            DAG.getNode(PreferredFusedOpcode, SL, VT,
11599                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
11600                                                    N000.getOperand(0)),
11601                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
11602                                                    N000.getOperand(1)),
11603                                        N1, Flags));
11604       }
11605     }
11606   }
11607 
11608   // More folding opportunities when target permits.
11609   if (Aggressive) {
11610     // fold (fsub (fma x, y, (fmul u, v)), z)
11611     //   -> (fma x, y (fma u, v, (fneg z)))
11612     if (CanFuse && N0.getOpcode() == PreferredFusedOpcode &&
11613         isContractableFMUL(N0.getOperand(2)) && N0->hasOneUse() &&
11614         N0.getOperand(2)->hasOneUse()) {
11615       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11616                          N0.getOperand(0), N0.getOperand(1),
11617                          DAG.getNode(PreferredFusedOpcode, SL, VT,
11618                                      N0.getOperand(2).getOperand(0),
11619                                      N0.getOperand(2).getOperand(1),
11620                                      DAG.getNode(ISD::FNEG, SL, VT,
11621                                                  N1), Flags), Flags);
11622     }
11623 
11624     // fold (fsub x, (fma y, z, (fmul u, v)))
11625     //   -> (fma (fneg y), z, (fma (fneg u), v, x))
11626     if (CanFuse && N1.getOpcode() == PreferredFusedOpcode &&
11627         isContractableFMUL(N1.getOperand(2))) {
11628       SDValue N20 = N1.getOperand(2).getOperand(0);
11629       SDValue N21 = N1.getOperand(2).getOperand(1);
11630       return DAG.getNode(PreferredFusedOpcode, SL, VT,
11631                          DAG.getNode(ISD::FNEG, SL, VT,
11632                                      N1.getOperand(0)),
11633                          N1.getOperand(1),
11634                          DAG.getNode(PreferredFusedOpcode, SL, VT,
11635                                      DAG.getNode(ISD::FNEG, SL, VT, N20),
11636                                      N21, N0, Flags), Flags);
11637     }
11638 
11639 
11640     // fold (fsub (fma x, y, (fpext (fmul u, v))), z)
11641     //   -> (fma x, y (fma (fpext u), (fpext v), (fneg z)))
11642     if (N0.getOpcode() == PreferredFusedOpcode) {
11643       SDValue N02 = N0.getOperand(2);
11644       if (N02.getOpcode() == ISD::FP_EXTEND) {
11645         SDValue N020 = N02.getOperand(0);
11646         if (isContractableFMUL(N020) &&
11647             TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N020.getValueType())) {
11648           return DAG.getNode(PreferredFusedOpcode, SL, VT,
11649                              N0.getOperand(0), N0.getOperand(1),
11650                              DAG.getNode(PreferredFusedOpcode, SL, VT,
11651                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11652                                                      N020.getOperand(0)),
11653                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11654                                                      N020.getOperand(1)),
11655                                          DAG.getNode(ISD::FNEG, SL, VT,
11656                                                      N1), Flags), Flags);
11657         }
11658       }
11659     }
11660 
11661     // fold (fsub (fpext (fma x, y, (fmul u, v))), z)
11662     //   -> (fma (fpext x), (fpext y),
11663     //           (fma (fpext u), (fpext v), (fneg z)))
11664     // FIXME: This turns two single-precision and one double-precision
11665     // operation into two double-precision operations, which might not be
11666     // interesting for all targets, especially GPUs.
11667     if (N0.getOpcode() == ISD::FP_EXTEND) {
11668       SDValue N00 = N0.getOperand(0);
11669       if (N00.getOpcode() == PreferredFusedOpcode) {
11670         SDValue N002 = N00.getOperand(2);
11671         if (isContractableFMUL(N002) &&
11672             TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) {
11673           return DAG.getNode(PreferredFusedOpcode, SL, VT,
11674                              DAG.getNode(ISD::FP_EXTEND, SL, VT,
11675                                          N00.getOperand(0)),
11676                              DAG.getNode(ISD::FP_EXTEND, SL, VT,
11677                                          N00.getOperand(1)),
11678                              DAG.getNode(PreferredFusedOpcode, SL, VT,
11679                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11680                                                      N002.getOperand(0)),
11681                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
11682                                                      N002.getOperand(1)),
11683                                          DAG.getNode(ISD::FNEG, SL, VT,
11684                                                      N1), Flags), Flags);
11685         }
11686       }
11687     }
11688 
11689     // fold (fsub x, (fma y, z, (fpext (fmul u, v))))
11690     //   -> (fma (fneg y), z, (fma (fneg (fpext u)), (fpext v), x))
11691     if (N1.getOpcode() == PreferredFusedOpcode &&
11692         N1.getOperand(2).getOpcode() == ISD::FP_EXTEND) {
11693       SDValue N120 = N1.getOperand(2).getOperand(0);
11694       if (isContractableFMUL(N120) &&
11695           TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N120.getValueType())) {
11696         SDValue N1200 = N120.getOperand(0);
11697         SDValue N1201 = N120.getOperand(1);
11698         return DAG.getNode(PreferredFusedOpcode, SL, VT,
11699                            DAG.getNode(ISD::FNEG, SL, VT, N1.getOperand(0)),
11700                            N1.getOperand(1),
11701                            DAG.getNode(PreferredFusedOpcode, SL, VT,
11702                                        DAG.getNode(ISD::FNEG, SL, VT,
11703                                                    DAG.getNode(ISD::FP_EXTEND, SL,
11704                                                                VT, N1200)),
11705                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
11706                                                    N1201),
11707                                        N0, Flags), Flags);
11708       }
11709     }
11710 
11711     // fold (fsub x, (fpext (fma y, z, (fmul u, v))))
11712     //   -> (fma (fneg (fpext y)), (fpext z),
11713     //           (fma (fneg (fpext u)), (fpext v), x))
11714     // FIXME: This turns two single-precision and one double-precision
11715     // operation into two double-precision operations, which might not be
11716     // interesting for all targets, especially GPUs.
11717     if (N1.getOpcode() == ISD::FP_EXTEND &&
11718         N1.getOperand(0).getOpcode() == PreferredFusedOpcode) {
11719       SDValue CvtSrc = N1.getOperand(0);
11720       SDValue N100 = CvtSrc.getOperand(0);
11721       SDValue N101 = CvtSrc.getOperand(1);
11722       SDValue N102 = CvtSrc.getOperand(2);
11723       if (isContractableFMUL(N102) &&
11724           TLI.isFPExtFoldable(PreferredFusedOpcode, VT, CvtSrc.getValueType())) {
11725         SDValue N1020 = N102.getOperand(0);
11726         SDValue N1021 = N102.getOperand(1);
11727         return DAG.getNode(PreferredFusedOpcode, SL, VT,
11728                            DAG.getNode(ISD::FNEG, SL, VT,
11729                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
11730                                                    N100)),
11731                            DAG.getNode(ISD::FP_EXTEND, SL, VT, N101),
11732                            DAG.getNode(PreferredFusedOpcode, SL, VT,
11733                                        DAG.getNode(ISD::FNEG, SL, VT,
11734                                                    DAG.getNode(ISD::FP_EXTEND, SL,
11735                                                                VT, N1020)),
11736                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
11737                                                    N1021),
11738                                        N0, Flags), Flags);
11739       }
11740     }
11741   }
11742 
11743   return SDValue();
11744 }
11745 
11746 /// Try to perform FMA combining on a given FMUL node based on the distributive
11747 /// law x * (y + 1) = x * y + x and variants thereof (commuted versions,
11748 /// subtraction instead of addition).
11749 SDValue DAGCombiner::visitFMULForFMADistributiveCombine(SDNode *N) {
11750   SDValue N0 = N->getOperand(0);
11751   SDValue N1 = N->getOperand(1);
11752   EVT VT = N->getValueType(0);
11753   SDLoc SL(N);
11754   const SDNodeFlags Flags = N->getFlags();
11755 
11756   assert(N->getOpcode() == ISD::FMUL && "Expected FMUL Operation");
11757 
11758   const TargetOptions &Options = DAG.getTarget().Options;
11759 
11760   // The transforms below are incorrect when x == 0 and y == inf, because the
11761   // intermediate multiplication produces a nan.
11762   if (!Options.NoInfsFPMath)
11763     return SDValue();
11764 
11765   // Floating-point multiply-add without intermediate rounding.
11766   bool HasFMA =
11767       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath) &&
11768       TLI.isFMAFasterThanFMulAndFAdd(VT) &&
11769       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
11770 
11771   // Floating-point multiply-add with intermediate rounding. This can result
11772   // in a less precise result due to the changed rounding order.
11773   bool HasFMAD = Options.UnsafeFPMath &&
11774                  (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
11775 
11776   // No valid opcode, do not combine.
11777   if (!HasFMAD && !HasFMA)
11778     return SDValue();
11779 
11780   // Always prefer FMAD to FMA for precision.
11781   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
11782   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
11783 
11784   // fold (fmul (fadd x0, +1.0), y) -> (fma x0, y, y)
11785   // fold (fmul (fadd x0, -1.0), y) -> (fma x0, y, (fneg y))
11786   auto FuseFADD = [&](SDValue X, SDValue Y, const SDNodeFlags Flags) {
11787     if (X.getOpcode() == ISD::FADD && (Aggressive || X->hasOneUse())) {
11788       if (auto *C = isConstOrConstSplatFP(X.getOperand(1), true)) {
11789         if (C->isExactlyValue(+1.0))
11790           return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
11791                              Y, Flags);
11792         if (C->isExactlyValue(-1.0))
11793           return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
11794                              DAG.getNode(ISD::FNEG, SL, VT, Y), Flags);
11795       }
11796     }
11797     return SDValue();
11798   };
11799 
11800   if (SDValue FMA = FuseFADD(N0, N1, Flags))
11801     return FMA;
11802   if (SDValue FMA = FuseFADD(N1, N0, Flags))
11803     return FMA;
11804 
11805   // fold (fmul (fsub +1.0, x1), y) -> (fma (fneg x1), y, y)
11806   // fold (fmul (fsub -1.0, x1), y) -> (fma (fneg x1), y, (fneg y))
11807   // fold (fmul (fsub x0, +1.0), y) -> (fma x0, y, (fneg y))
11808   // fold (fmul (fsub x0, -1.0), y) -> (fma x0, y, y)
11809   auto FuseFSUB = [&](SDValue X, SDValue Y, const SDNodeFlags Flags) {
11810     if (X.getOpcode() == ISD::FSUB && (Aggressive || X->hasOneUse())) {
11811       if (auto *C0 = isConstOrConstSplatFP(X.getOperand(0), true)) {
11812         if (C0->isExactlyValue(+1.0))
11813           return DAG.getNode(PreferredFusedOpcode, SL, VT,
11814                              DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
11815                              Y, Flags);
11816         if (C0->isExactlyValue(-1.0))
11817           return DAG.getNode(PreferredFusedOpcode, SL, VT,
11818                              DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
11819                              DAG.getNode(ISD::FNEG, SL, VT, Y), Flags);
11820       }
11821       if (auto *C1 = isConstOrConstSplatFP(X.getOperand(1), true)) {
11822         if (C1->isExactlyValue(+1.0))
11823           return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
11824                              DAG.getNode(ISD::FNEG, SL, VT, Y), Flags);
11825         if (C1->isExactlyValue(-1.0))
11826           return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
11827                              Y, Flags);
11828       }
11829     }
11830     return SDValue();
11831   };
11832 
11833   if (SDValue FMA = FuseFSUB(N0, N1, Flags))
11834     return FMA;
11835   if (SDValue FMA = FuseFSUB(N1, N0, Flags))
11836     return FMA;
11837 
11838   return SDValue();
11839 }
11840 
11841 SDValue DAGCombiner::visitFADD(SDNode *N) {
11842   SDValue N0 = N->getOperand(0);
11843   SDValue N1 = N->getOperand(1);
11844   bool N0CFP = isConstantFPBuildVectorOrConstantFP(N0);
11845   bool N1CFP = isConstantFPBuildVectorOrConstantFP(N1);
11846   EVT VT = N->getValueType(0);
11847   SDLoc DL(N);
11848   const TargetOptions &Options = DAG.getTarget().Options;
11849   const SDNodeFlags Flags = N->getFlags();
11850 
11851   // fold vector ops
11852   if (VT.isVector())
11853     if (SDValue FoldedVOp = SimplifyVBinOp(N))
11854       return FoldedVOp;
11855 
11856   // fold (fadd c1, c2) -> c1 + c2
11857   if (N0CFP && N1CFP)
11858     return DAG.getNode(ISD::FADD, DL, VT, N0, N1, Flags);
11859 
11860   // canonicalize constant to RHS
11861   if (N0CFP && !N1CFP)
11862     return DAG.getNode(ISD::FADD, DL, VT, N1, N0, Flags);
11863 
11864   // N0 + -0.0 --> N0 (also allowed with +0.0 and fast-math)
11865   ConstantFPSDNode *N1C = isConstOrConstSplatFP(N1, true);
11866   if (N1C && N1C->isZero())
11867     if (N1C->isNegative() || Options.UnsafeFPMath || Flags.hasNoSignedZeros())
11868       return N0;
11869 
11870   if (SDValue NewSel = foldBinOpIntoSelect(N))
11871     return NewSel;
11872 
11873   // fold (fadd A, (fneg B)) -> (fsub A, B)
11874   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
11875       isNegatibleForFree(N1, LegalOperations, TLI, &Options, ForCodeSize) == 2)
11876     return DAG.getNode(ISD::FSUB, DL, VT, N0,
11877                        GetNegatedExpression(N1, DAG, LegalOperations,
11878                                             ForCodeSize), Flags);
11879 
11880   // fold (fadd (fneg A), B) -> (fsub B, A)
11881   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
11882       isNegatibleForFree(N0, LegalOperations, TLI, &Options, ForCodeSize) == 2)
11883     return DAG.getNode(ISD::FSUB, DL, VT, N1,
11884                        GetNegatedExpression(N0, DAG, LegalOperations,
11885                                             ForCodeSize), Flags);
11886 
11887   auto isFMulNegTwo = [](SDValue FMul) {
11888     if (!FMul.hasOneUse() || FMul.getOpcode() != ISD::FMUL)
11889       return false;
11890     auto *C = isConstOrConstSplatFP(FMul.getOperand(1), true);
11891     return C && C->isExactlyValue(-2.0);
11892   };
11893 
11894   // fadd (fmul B, -2.0), A --> fsub A, (fadd B, B)
11895   if (isFMulNegTwo(N0)) {
11896     SDValue B = N0.getOperand(0);
11897     SDValue Add = DAG.getNode(ISD::FADD, DL, VT, B, B, Flags);
11898     return DAG.getNode(ISD::FSUB, DL, VT, N1, Add, Flags);
11899   }
11900   // fadd A, (fmul B, -2.0) --> fsub A, (fadd B, B)
11901   if (isFMulNegTwo(N1)) {
11902     SDValue B = N1.getOperand(0);
11903     SDValue Add = DAG.getNode(ISD::FADD, DL, VT, B, B, Flags);
11904     return DAG.getNode(ISD::FSUB, DL, VT, N0, Add, Flags);
11905   }
11906 
11907   // No FP constant should be created after legalization as Instruction
11908   // Selection pass has a hard time dealing with FP constants.
11909   bool AllowNewConst = (Level < AfterLegalizeDAG);
11910 
11911   // If 'unsafe math' or nnan is enabled, fold lots of things.
11912   if ((Options.UnsafeFPMath || Flags.hasNoNaNs()) && AllowNewConst) {
11913     // If allowed, fold (fadd (fneg x), x) -> 0.0
11914     if (N0.getOpcode() == ISD::FNEG && N0.getOperand(0) == N1)
11915       return DAG.getConstantFP(0.0, DL, VT);
11916 
11917     // If allowed, fold (fadd x, (fneg x)) -> 0.0
11918     if (N1.getOpcode() == ISD::FNEG && N1.getOperand(0) == N0)
11919       return DAG.getConstantFP(0.0, DL, VT);
11920   }
11921 
11922   // If 'unsafe math' or reassoc and nsz, fold lots of things.
11923   // TODO: break out portions of the transformations below for which Unsafe is
11924   //       considered and which do not require both nsz and reassoc
11925   if ((Options.UnsafeFPMath ||
11926        (Flags.hasAllowReassociation() && Flags.hasNoSignedZeros())) &&
11927       AllowNewConst) {
11928     // fadd (fadd x, c1), c2 -> fadd x, c1 + c2
11929     if (N1CFP && N0.getOpcode() == ISD::FADD &&
11930         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1))) {
11931       SDValue NewC = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1), N1, Flags);
11932       return DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(0), NewC, Flags);
11933     }
11934 
11935     // We can fold chains of FADD's of the same value into multiplications.
11936     // This transform is not safe in general because we are reducing the number
11937     // of rounding steps.
11938     if (TLI.isOperationLegalOrCustom(ISD::FMUL, VT) && !N0CFP && !N1CFP) {
11939       if (N0.getOpcode() == ISD::FMUL) {
11940         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
11941         bool CFP01 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(1));
11942 
11943         // (fadd (fmul x, c), x) -> (fmul x, c+1)
11944         if (CFP01 && !CFP00 && N0.getOperand(0) == N1) {
11945           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
11946                                        DAG.getConstantFP(1.0, DL, VT), Flags);
11947           return DAG.getNode(ISD::FMUL, DL, VT, N1, NewCFP, Flags);
11948         }
11949 
11950         // (fadd (fmul x, c), (fadd x, x)) -> (fmul x, c+2)
11951         if (CFP01 && !CFP00 && N1.getOpcode() == ISD::FADD &&
11952             N1.getOperand(0) == N1.getOperand(1) &&
11953             N0.getOperand(0) == N1.getOperand(0)) {
11954           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
11955                                        DAG.getConstantFP(2.0, DL, VT), Flags);
11956           return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), NewCFP, Flags);
11957         }
11958       }
11959 
11960       if (N1.getOpcode() == ISD::FMUL) {
11961         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
11962         bool CFP11 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(1));
11963 
11964         // (fadd x, (fmul x, c)) -> (fmul x, c+1)
11965         if (CFP11 && !CFP10 && N1.getOperand(0) == N0) {
11966           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
11967                                        DAG.getConstantFP(1.0, DL, VT), Flags);
11968           return DAG.getNode(ISD::FMUL, DL, VT, N0, NewCFP, Flags);
11969         }
11970 
11971         // (fadd (fadd x, x), (fmul x, c)) -> (fmul x, c+2)
11972         if (CFP11 && !CFP10 && N0.getOpcode() == ISD::FADD &&
11973             N0.getOperand(0) == N0.getOperand(1) &&
11974             N1.getOperand(0) == N0.getOperand(0)) {
11975           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
11976                                        DAG.getConstantFP(2.0, DL, VT), Flags);
11977           return DAG.getNode(ISD::FMUL, DL, VT, N1.getOperand(0), NewCFP, Flags);
11978         }
11979       }
11980 
11981       if (N0.getOpcode() == ISD::FADD) {
11982         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
11983         // (fadd (fadd x, x), x) -> (fmul x, 3.0)
11984         if (!CFP00 && N0.getOperand(0) == N0.getOperand(1) &&
11985             (N0.getOperand(0) == N1)) {
11986           return DAG.getNode(ISD::FMUL, DL, VT,
11987                              N1, DAG.getConstantFP(3.0, DL, VT), Flags);
11988         }
11989       }
11990 
11991       if (N1.getOpcode() == ISD::FADD) {
11992         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
11993         // (fadd x, (fadd x, x)) -> (fmul x, 3.0)
11994         if (!CFP10 && N1.getOperand(0) == N1.getOperand(1) &&
11995             N1.getOperand(0) == N0) {
11996           return DAG.getNode(ISD::FMUL, DL, VT,
11997                              N0, DAG.getConstantFP(3.0, DL, VT), Flags);
11998         }
11999       }
12000 
12001       // (fadd (fadd x, x), (fadd x, x)) -> (fmul x, 4.0)
12002       if (N0.getOpcode() == ISD::FADD && N1.getOpcode() == ISD::FADD &&
12003           N0.getOperand(0) == N0.getOperand(1) &&
12004           N1.getOperand(0) == N1.getOperand(1) &&
12005           N0.getOperand(0) == N1.getOperand(0)) {
12006         return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0),
12007                            DAG.getConstantFP(4.0, DL, VT), Flags);
12008       }
12009     }
12010   } // enable-unsafe-fp-math
12011 
12012   // FADD -> FMA combines:
12013   if (SDValue Fused = visitFADDForFMACombine(N)) {
12014     AddToWorklist(Fused.getNode());
12015     return Fused;
12016   }
12017   return SDValue();
12018 }
12019 
12020 SDValue DAGCombiner::visitFSUB(SDNode *N) {
12021   SDValue N0 = N->getOperand(0);
12022   SDValue N1 = N->getOperand(1);
12023   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0, true);
12024   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1, true);
12025   EVT VT = N->getValueType(0);
12026   SDLoc DL(N);
12027   const TargetOptions &Options = DAG.getTarget().Options;
12028   const SDNodeFlags Flags = N->getFlags();
12029 
12030   // fold vector ops
12031   if (VT.isVector())
12032     if (SDValue FoldedVOp = SimplifyVBinOp(N))
12033       return FoldedVOp;
12034 
12035   // fold (fsub c1, c2) -> c1-c2
12036   if (N0CFP && N1CFP)
12037     return DAG.getNode(ISD::FSUB, DL, VT, N0, N1, Flags);
12038 
12039   if (SDValue NewSel = foldBinOpIntoSelect(N))
12040     return NewSel;
12041 
12042   // (fsub A, 0) -> A
12043   if (N1CFP && N1CFP->isZero()) {
12044     if (!N1CFP->isNegative() || Options.UnsafeFPMath ||
12045         Flags.hasNoSignedZeros()) {
12046       return N0;
12047     }
12048   }
12049 
12050   if (N0 == N1) {
12051     // (fsub x, x) -> 0.0
12052     if (Options.UnsafeFPMath || Flags.hasNoNaNs())
12053       return DAG.getConstantFP(0.0f, DL, VT);
12054   }
12055 
12056   // (fsub -0.0, N1) -> -N1
12057   // NOTE: It is safe to transform an FSUB(-0.0,X) into an FNEG(X), since the
12058   //       FSUB does not specify the sign bit of a NaN. Also note that for
12059   //       the same reason, the inverse transform is not safe, unless fast math
12060   //       flags are in play.
12061   if (N0CFP && N0CFP->isZero()) {
12062     if (N0CFP->isNegative() ||
12063         (Options.NoSignedZerosFPMath || Flags.hasNoSignedZeros())) {
12064       if (isNegatibleForFree(N1, LegalOperations, TLI, &Options, ForCodeSize))
12065         return GetNegatedExpression(N1, DAG, LegalOperations, ForCodeSize);
12066       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
12067         return DAG.getNode(ISD::FNEG, DL, VT, N1, Flags);
12068     }
12069   }
12070 
12071   if ((Options.UnsafeFPMath ||
12072       (Flags.hasAllowReassociation() && Flags.hasNoSignedZeros()))
12073       && N1.getOpcode() == ISD::FADD) {
12074     // X - (X + Y) -> -Y
12075     if (N0 == N1->getOperand(0))
12076       return DAG.getNode(ISD::FNEG, DL, VT, N1->getOperand(1), Flags);
12077     // X - (Y + X) -> -Y
12078     if (N0 == N1->getOperand(1))
12079       return DAG.getNode(ISD::FNEG, DL, VT, N1->getOperand(0), Flags);
12080   }
12081 
12082   // fold (fsub A, (fneg B)) -> (fadd A, B)
12083   if (isNegatibleForFree(N1, LegalOperations, TLI, &Options, ForCodeSize))
12084     return DAG.getNode(ISD::FADD, DL, VT, N0,
12085                        GetNegatedExpression(N1, DAG, LegalOperations,
12086                                             ForCodeSize), Flags);
12087 
12088   // FSUB -> FMA combines:
12089   if (SDValue Fused = visitFSUBForFMACombine(N)) {
12090     AddToWorklist(Fused.getNode());
12091     return Fused;
12092   }
12093 
12094   return SDValue();
12095 }
12096 
12097 SDValue DAGCombiner::visitFMUL(SDNode *N) {
12098   SDValue N0 = N->getOperand(0);
12099   SDValue N1 = N->getOperand(1);
12100   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0, true);
12101   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1, true);
12102   EVT VT = N->getValueType(0);
12103   SDLoc DL(N);
12104   const TargetOptions &Options = DAG.getTarget().Options;
12105   const SDNodeFlags Flags = N->getFlags();
12106 
12107   // fold vector ops
12108   if (VT.isVector()) {
12109     // This just handles C1 * C2 for vectors. Other vector folds are below.
12110     if (SDValue FoldedVOp = SimplifyVBinOp(N))
12111       return FoldedVOp;
12112   }
12113 
12114   // fold (fmul c1, c2) -> c1*c2
12115   if (N0CFP && N1CFP)
12116     return DAG.getNode(ISD::FMUL, DL, VT, N0, N1, Flags);
12117 
12118   // canonicalize constant to RHS
12119   if (isConstantFPBuildVectorOrConstantFP(N0) &&
12120      !isConstantFPBuildVectorOrConstantFP(N1))
12121     return DAG.getNode(ISD::FMUL, DL, VT, N1, N0, Flags);
12122 
12123   // fold (fmul A, 1.0) -> A
12124   if (N1CFP && N1CFP->isExactlyValue(1.0))
12125     return N0;
12126 
12127   if (SDValue NewSel = foldBinOpIntoSelect(N))
12128     return NewSel;
12129 
12130   if (Options.UnsafeFPMath ||
12131       (Flags.hasNoNaNs() && Flags.hasNoSignedZeros())) {
12132     // fold (fmul A, 0) -> 0
12133     if (N1CFP && N1CFP->isZero())
12134       return N1;
12135   }
12136 
12137   if (Options.UnsafeFPMath || Flags.hasAllowReassociation()) {
12138     // fmul (fmul X, C1), C2 -> fmul X, C1 * C2
12139     if (isConstantFPBuildVectorOrConstantFP(N1) &&
12140         N0.getOpcode() == ISD::FMUL) {
12141       SDValue N00 = N0.getOperand(0);
12142       SDValue N01 = N0.getOperand(1);
12143       // Avoid an infinite loop by making sure that N00 is not a constant
12144       // (the inner multiply has not been constant folded yet).
12145       if (isConstantFPBuildVectorOrConstantFP(N01) &&
12146           !isConstantFPBuildVectorOrConstantFP(N00)) {
12147         SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, N01, N1, Flags);
12148         return DAG.getNode(ISD::FMUL, DL, VT, N00, MulConsts, Flags);
12149       }
12150     }
12151 
12152     // Match a special-case: we convert X * 2.0 into fadd.
12153     // fmul (fadd X, X), C -> fmul X, 2.0 * C
12154     if (N0.getOpcode() == ISD::FADD && N0.hasOneUse() &&
12155         N0.getOperand(0) == N0.getOperand(1)) {
12156       const SDValue Two = DAG.getConstantFP(2.0, DL, VT);
12157       SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, Two, N1, Flags);
12158       return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), MulConsts, Flags);
12159     }
12160   }
12161 
12162   // fold (fmul X, 2.0) -> (fadd X, X)
12163   if (N1CFP && N1CFP->isExactlyValue(+2.0))
12164     return DAG.getNode(ISD::FADD, DL, VT, N0, N0, Flags);
12165 
12166   // fold (fmul X, -1.0) -> (fneg X)
12167   if (N1CFP && N1CFP->isExactlyValue(-1.0))
12168     if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
12169       return DAG.getNode(ISD::FNEG, DL, VT, N0);
12170 
12171   // fold (fmul (fneg X), (fneg Y)) -> (fmul X, Y)
12172   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options,
12173                                        ForCodeSize)) {
12174     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options,
12175                                          ForCodeSize)) {
12176       // Both can be negated for free, check to see if at least one is cheaper
12177       // negated.
12178       if (LHSNeg == 2 || RHSNeg == 2)
12179         return DAG.getNode(ISD::FMUL, DL, VT,
12180                            GetNegatedExpression(N0, DAG, LegalOperations,
12181                                                 ForCodeSize),
12182                            GetNegatedExpression(N1, DAG, LegalOperations,
12183                                                 ForCodeSize),
12184                            Flags);
12185     }
12186   }
12187 
12188   // fold (fmul X, (select (fcmp X > 0.0), -1.0, 1.0)) -> (fneg (fabs X))
12189   // fold (fmul X, (select (fcmp X > 0.0), 1.0, -1.0)) -> (fabs X)
12190   if (Flags.hasNoNaNs() && Flags.hasNoSignedZeros() &&
12191       (N0.getOpcode() == ISD::SELECT || N1.getOpcode() == ISD::SELECT) &&
12192       TLI.isOperationLegal(ISD::FABS, VT)) {
12193     SDValue Select = N0, X = N1;
12194     if (Select.getOpcode() != ISD::SELECT)
12195       std::swap(Select, X);
12196 
12197     SDValue Cond = Select.getOperand(0);
12198     auto TrueOpnd  = dyn_cast<ConstantFPSDNode>(Select.getOperand(1));
12199     auto FalseOpnd = dyn_cast<ConstantFPSDNode>(Select.getOperand(2));
12200 
12201     if (TrueOpnd && FalseOpnd &&
12202         Cond.getOpcode() == ISD::SETCC && Cond.getOperand(0) == X &&
12203         isa<ConstantFPSDNode>(Cond.getOperand(1)) &&
12204         cast<ConstantFPSDNode>(Cond.getOperand(1))->isExactlyValue(0.0)) {
12205       ISD::CondCode CC = cast<CondCodeSDNode>(Cond.getOperand(2))->get();
12206       switch (CC) {
12207       default: break;
12208       case ISD::SETOLT:
12209       case ISD::SETULT:
12210       case ISD::SETOLE:
12211       case ISD::SETULE:
12212       case ISD::SETLT:
12213       case ISD::SETLE:
12214         std::swap(TrueOpnd, FalseOpnd);
12215         LLVM_FALLTHROUGH;
12216       case ISD::SETOGT:
12217       case ISD::SETUGT:
12218       case ISD::SETOGE:
12219       case ISD::SETUGE:
12220       case ISD::SETGT:
12221       case ISD::SETGE:
12222         if (TrueOpnd->isExactlyValue(-1.0) && FalseOpnd->isExactlyValue(1.0) &&
12223             TLI.isOperationLegal(ISD::FNEG, VT))
12224           return DAG.getNode(ISD::FNEG, DL, VT,
12225                    DAG.getNode(ISD::FABS, DL, VT, X));
12226         if (TrueOpnd->isExactlyValue(1.0) && FalseOpnd->isExactlyValue(-1.0))
12227           return DAG.getNode(ISD::FABS, DL, VT, X);
12228 
12229         break;
12230       }
12231     }
12232   }
12233 
12234   // FMUL -> FMA combines:
12235   if (SDValue Fused = visitFMULForFMADistributiveCombine(N)) {
12236     AddToWorklist(Fused.getNode());
12237     return Fused;
12238   }
12239 
12240   return SDValue();
12241 }
12242 
12243 SDValue DAGCombiner::visitFMA(SDNode *N) {
12244   SDValue N0 = N->getOperand(0);
12245   SDValue N1 = N->getOperand(1);
12246   SDValue N2 = N->getOperand(2);
12247   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
12248   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
12249   EVT VT = N->getValueType(0);
12250   SDLoc DL(N);
12251   const TargetOptions &Options = DAG.getTarget().Options;
12252 
12253   // FMA nodes have flags that propagate to the created nodes.
12254   const SDNodeFlags Flags = N->getFlags();
12255   bool UnsafeFPMath = Options.UnsafeFPMath || isContractable(N);
12256 
12257   // Constant fold FMA.
12258   if (isa<ConstantFPSDNode>(N0) &&
12259       isa<ConstantFPSDNode>(N1) &&
12260       isa<ConstantFPSDNode>(N2)) {
12261     return DAG.getNode(ISD::FMA, DL, VT, N0, N1, N2);
12262   }
12263 
12264   if (UnsafeFPMath) {
12265     if (N0CFP && N0CFP->isZero())
12266       return N2;
12267     if (N1CFP && N1CFP->isZero())
12268       return N2;
12269   }
12270   // TODO: The FMA node should have flags that propagate to these nodes.
12271   if (N0CFP && N0CFP->isExactlyValue(1.0))
12272     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N1, N2);
12273   if (N1CFP && N1CFP->isExactlyValue(1.0))
12274     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N0, N2);
12275 
12276   // Canonicalize (fma c, x, y) -> (fma x, c, y)
12277   if (isConstantFPBuildVectorOrConstantFP(N0) &&
12278      !isConstantFPBuildVectorOrConstantFP(N1))
12279     return DAG.getNode(ISD::FMA, SDLoc(N), VT, N1, N0, N2);
12280 
12281   if (UnsafeFPMath) {
12282     // (fma x, c1, (fmul x, c2)) -> (fmul x, c1+c2)
12283     if (N2.getOpcode() == ISD::FMUL && N0 == N2.getOperand(0) &&
12284         isConstantFPBuildVectorOrConstantFP(N1) &&
12285         isConstantFPBuildVectorOrConstantFP(N2.getOperand(1))) {
12286       return DAG.getNode(ISD::FMUL, DL, VT, N0,
12287                          DAG.getNode(ISD::FADD, DL, VT, N1, N2.getOperand(1),
12288                                      Flags), Flags);
12289     }
12290 
12291     // (fma (fmul x, c1), c2, y) -> (fma x, c1*c2, y)
12292     if (N0.getOpcode() == ISD::FMUL &&
12293         isConstantFPBuildVectorOrConstantFP(N1) &&
12294         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1))) {
12295       return DAG.getNode(ISD::FMA, DL, VT,
12296                          N0.getOperand(0),
12297                          DAG.getNode(ISD::FMUL, DL, VT, N1, N0.getOperand(1),
12298                                      Flags),
12299                          N2);
12300     }
12301   }
12302 
12303   // (fma x, 1, y) -> (fadd x, y)
12304   // (fma x, -1, y) -> (fadd (fneg x), y)
12305   if (N1CFP) {
12306     if (N1CFP->isExactlyValue(1.0))
12307       // TODO: The FMA node should have flags that propagate to this node.
12308       return DAG.getNode(ISD::FADD, DL, VT, N0, N2);
12309 
12310     if (N1CFP->isExactlyValue(-1.0) &&
12311         (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))) {
12312       SDValue RHSNeg = DAG.getNode(ISD::FNEG, DL, VT, N0);
12313       AddToWorklist(RHSNeg.getNode());
12314       // TODO: The FMA node should have flags that propagate to this node.
12315       return DAG.getNode(ISD::FADD, DL, VT, N2, RHSNeg);
12316     }
12317 
12318     // fma (fneg x), K, y -> fma x -K, y
12319     if (N0.getOpcode() == ISD::FNEG &&
12320         (TLI.isOperationLegal(ISD::ConstantFP, VT) ||
12321          (N1.hasOneUse() && !TLI.isFPImmLegal(N1CFP->getValueAPF(), VT,
12322                                               ForCodeSize)))) {
12323       return DAG.getNode(ISD::FMA, DL, VT, N0.getOperand(0),
12324                          DAG.getNode(ISD::FNEG, DL, VT, N1, Flags), N2);
12325     }
12326   }
12327 
12328   if (UnsafeFPMath) {
12329     // (fma x, c, x) -> (fmul x, (c+1))
12330     if (N1CFP && N0 == N2) {
12331       return DAG.getNode(ISD::FMUL, DL, VT, N0,
12332                          DAG.getNode(ISD::FADD, DL, VT, N1,
12333                                      DAG.getConstantFP(1.0, DL, VT), Flags),
12334                          Flags);
12335     }
12336 
12337     // (fma x, c, (fneg x)) -> (fmul x, (c-1))
12338     if (N1CFP && N2.getOpcode() == ISD::FNEG && N2.getOperand(0) == N0) {
12339       return DAG.getNode(ISD::FMUL, DL, VT, N0,
12340                          DAG.getNode(ISD::FADD, DL, VT, N1,
12341                                      DAG.getConstantFP(-1.0, DL, VT), Flags),
12342                          Flags);
12343     }
12344   }
12345 
12346   return SDValue();
12347 }
12348 
12349 // Combine multiple FDIVs with the same divisor into multiple FMULs by the
12350 // reciprocal.
12351 // E.g., (a / D; b / D;) -> (recip = 1.0 / D; a * recip; b * recip)
12352 // Notice that this is not always beneficial. One reason is different targets
12353 // may have different costs for FDIV and FMUL, so sometimes the cost of two
12354 // FDIVs may be lower than the cost of one FDIV and two FMULs. Another reason
12355 // is the critical path is increased from "one FDIV" to "one FDIV + one FMUL".
12356 SDValue DAGCombiner::combineRepeatedFPDivisors(SDNode *N) {
12357   // TODO: Limit this transform based on optsize/minsize - it always creates at
12358   //       least 1 extra instruction. But the perf win may be substantial enough
12359   //       that only minsize should restrict this.
12360   bool UnsafeMath = DAG.getTarget().Options.UnsafeFPMath;
12361   const SDNodeFlags Flags = N->getFlags();
12362   if (!UnsafeMath && !Flags.hasAllowReciprocal())
12363     return SDValue();
12364 
12365   // Skip if current node is a reciprocal.
12366   SDValue N0 = N->getOperand(0);
12367   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0, /* AllowUndefs */ true);
12368   if (N0CFP && N0CFP->isExactlyValue(1.0))
12369     return SDValue();
12370 
12371   // Exit early if the target does not want this transform or if there can't
12372   // possibly be enough uses of the divisor to make the transform worthwhile.
12373   SDValue N1 = N->getOperand(1);
12374   unsigned MinUses = TLI.combineRepeatedFPDivisors();
12375 
12376   // For splat vectors, scale the number of uses by the splat factor. If we can
12377   // convert the division into a scalar op, that will likely be much faster.
12378   unsigned NumElts = 1;
12379   EVT VT = N->getValueType(0);
12380   if (VT.isVector() && DAG.isSplatValue(N1))
12381     NumElts = VT.getVectorNumElements();
12382 
12383   if (!MinUses || (N1->use_size() * NumElts) < MinUses)
12384     return SDValue();
12385 
12386   // Find all FDIV users of the same divisor.
12387   // Use a set because duplicates may be present in the user list.
12388   SetVector<SDNode *> Users;
12389   for (auto *U : N1->uses()) {
12390     if (U->getOpcode() == ISD::FDIV && U->getOperand(1) == N1) {
12391       // This division is eligible for optimization only if global unsafe math
12392       // is enabled or if this division allows reciprocal formation.
12393       if (UnsafeMath || U->getFlags().hasAllowReciprocal())
12394         Users.insert(U);
12395     }
12396   }
12397 
12398   // Now that we have the actual number of divisor uses, make sure it meets
12399   // the minimum threshold specified by the target.
12400   if ((Users.size() * NumElts) < MinUses)
12401     return SDValue();
12402 
12403   SDLoc DL(N);
12404   SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
12405   SDValue Reciprocal = DAG.getNode(ISD::FDIV, DL, VT, FPOne, N1, Flags);
12406 
12407   // Dividend / Divisor -> Dividend * Reciprocal
12408   for (auto *U : Users) {
12409     SDValue Dividend = U->getOperand(0);
12410     if (Dividend != FPOne) {
12411       SDValue NewNode = DAG.getNode(ISD::FMUL, SDLoc(U), VT, Dividend,
12412                                     Reciprocal, Flags);
12413       CombineTo(U, NewNode);
12414     } else if (U != Reciprocal.getNode()) {
12415       // In the absence of fast-math-flags, this user node is always the
12416       // same node as Reciprocal, but with FMF they may be different nodes.
12417       CombineTo(U, Reciprocal);
12418     }
12419   }
12420   return SDValue(N, 0);  // N was replaced.
12421 }
12422 
12423 SDValue DAGCombiner::visitFDIV(SDNode *N) {
12424   SDValue N0 = N->getOperand(0);
12425   SDValue N1 = N->getOperand(1);
12426   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
12427   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
12428   EVT VT = N->getValueType(0);
12429   SDLoc DL(N);
12430   const TargetOptions &Options = DAG.getTarget().Options;
12431   SDNodeFlags Flags = N->getFlags();
12432 
12433   // fold vector ops
12434   if (VT.isVector())
12435     if (SDValue FoldedVOp = SimplifyVBinOp(N))
12436       return FoldedVOp;
12437 
12438   // fold (fdiv c1, c2) -> c1/c2
12439   if (N0CFP && N1CFP)
12440     return DAG.getNode(ISD::FDIV, SDLoc(N), VT, N0, N1, Flags);
12441 
12442   if (SDValue NewSel = foldBinOpIntoSelect(N))
12443     return NewSel;
12444 
12445   if (SDValue V = combineRepeatedFPDivisors(N))
12446     return V;
12447 
12448   if (Options.UnsafeFPMath || Flags.hasAllowReciprocal()) {
12449     // fold (fdiv X, c2) -> fmul X, 1/c2 if losing precision is acceptable.
12450     if (N1CFP) {
12451       // Compute the reciprocal 1.0 / c2.
12452       const APFloat &N1APF = N1CFP->getValueAPF();
12453       APFloat Recip(N1APF.getSemantics(), 1); // 1.0
12454       APFloat::opStatus st = Recip.divide(N1APF, APFloat::rmNearestTiesToEven);
12455       // Only do the transform if the reciprocal is a legal fp immediate that
12456       // isn't too nasty (eg NaN, denormal, ...).
12457       if ((st == APFloat::opOK || st == APFloat::opInexact) && // Not too nasty
12458           (!LegalOperations ||
12459            // FIXME: custom lowering of ConstantFP might fail (see e.g. ARM
12460            // backend)... we should handle this gracefully after Legalize.
12461            // TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT) ||
12462            TLI.isOperationLegal(ISD::ConstantFP, VT) ||
12463            TLI.isFPImmLegal(Recip, VT, ForCodeSize)))
12464         return DAG.getNode(ISD::FMUL, DL, VT, N0,
12465                            DAG.getConstantFP(Recip, DL, VT), Flags);
12466     }
12467 
12468     // If this FDIV is part of a reciprocal square root, it may be folded
12469     // into a target-specific square root estimate instruction.
12470     if (N1.getOpcode() == ISD::FSQRT) {
12471       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0), Flags)) {
12472         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
12473       }
12474     } else if (N1.getOpcode() == ISD::FP_EXTEND &&
12475                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
12476       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
12477                                           Flags)) {
12478         RV = DAG.getNode(ISD::FP_EXTEND, SDLoc(N1), VT, RV);
12479         AddToWorklist(RV.getNode());
12480         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
12481       }
12482     } else if (N1.getOpcode() == ISD::FP_ROUND &&
12483                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
12484       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
12485                                           Flags)) {
12486         RV = DAG.getNode(ISD::FP_ROUND, SDLoc(N1), VT, RV, N1.getOperand(1));
12487         AddToWorklist(RV.getNode());
12488         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
12489       }
12490     } else if (N1.getOpcode() == ISD::FMUL) {
12491       // Look through an FMUL. Even though this won't remove the FDIV directly,
12492       // it's still worthwhile to get rid of the FSQRT if possible.
12493       SDValue SqrtOp;
12494       SDValue OtherOp;
12495       if (N1.getOperand(0).getOpcode() == ISD::FSQRT) {
12496         SqrtOp = N1.getOperand(0);
12497         OtherOp = N1.getOperand(1);
12498       } else if (N1.getOperand(1).getOpcode() == ISD::FSQRT) {
12499         SqrtOp = N1.getOperand(1);
12500         OtherOp = N1.getOperand(0);
12501       }
12502       if (SqrtOp.getNode()) {
12503         // We found a FSQRT, so try to make this fold:
12504         // x / (y * sqrt(z)) -> x * (rsqrt(z) / y)
12505         if (SDValue RV = buildRsqrtEstimate(SqrtOp.getOperand(0), Flags)) {
12506           RV = DAG.getNode(ISD::FDIV, SDLoc(N1), VT, RV, OtherOp, Flags);
12507           AddToWorklist(RV.getNode());
12508           return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
12509         }
12510       }
12511     }
12512 
12513     // Fold into a reciprocal estimate and multiply instead of a real divide.
12514     if (SDValue RV = BuildReciprocalEstimate(N1, Flags)) {
12515       AddToWorklist(RV.getNode());
12516       return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
12517     }
12518   }
12519 
12520   // (fdiv (fneg X), (fneg Y)) -> (fdiv X, Y)
12521   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options,
12522                                        ForCodeSize)) {
12523     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options,
12524                                          ForCodeSize)) {
12525       // Both can be negated for free, check to see if at least one is cheaper
12526       // negated.
12527       if (LHSNeg == 2 || RHSNeg == 2)
12528         return DAG.getNode(ISD::FDIV, SDLoc(N), VT,
12529                            GetNegatedExpression(N0, DAG, LegalOperations,
12530                                                 ForCodeSize),
12531                            GetNegatedExpression(N1, DAG, LegalOperations,
12532                                                 ForCodeSize),
12533                            Flags);
12534     }
12535   }
12536 
12537   return SDValue();
12538 }
12539 
12540 SDValue DAGCombiner::visitFREM(SDNode *N) {
12541   SDValue N0 = N->getOperand(0);
12542   SDValue N1 = N->getOperand(1);
12543   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
12544   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
12545   EVT VT = N->getValueType(0);
12546 
12547   // fold (frem c1, c2) -> fmod(c1,c2)
12548   if (N0CFP && N1CFP)
12549     return DAG.getNode(ISD::FREM, SDLoc(N), VT, N0, N1, N->getFlags());
12550 
12551   if (SDValue NewSel = foldBinOpIntoSelect(N))
12552     return NewSel;
12553 
12554   return SDValue();
12555 }
12556 
12557 SDValue DAGCombiner::visitFSQRT(SDNode *N) {
12558   SDNodeFlags Flags = N->getFlags();
12559   if (!DAG.getTarget().Options.UnsafeFPMath &&
12560       !Flags.hasApproximateFuncs())
12561     return SDValue();
12562 
12563   SDValue N0 = N->getOperand(0);
12564   if (TLI.isFsqrtCheap(N0, DAG))
12565     return SDValue();
12566 
12567   // FSQRT nodes have flags that propagate to the created nodes.
12568   return buildSqrtEstimate(N0, Flags);
12569 }
12570 
12571 /// copysign(x, fp_extend(y)) -> copysign(x, y)
12572 /// copysign(x, fp_round(y)) -> copysign(x, y)
12573 static inline bool CanCombineFCOPYSIGN_EXTEND_ROUND(SDNode *N) {
12574   SDValue N1 = N->getOperand(1);
12575   if ((N1.getOpcode() == ISD::FP_EXTEND ||
12576        N1.getOpcode() == ISD::FP_ROUND)) {
12577     // Do not optimize out type conversion of f128 type yet.
12578     // For some targets like x86_64, configuration is changed to keep one f128
12579     // value in one SSE register, but instruction selection cannot handle
12580     // FCOPYSIGN on SSE registers yet.
12581     EVT N1VT = N1->getValueType(0);
12582     EVT N1Op0VT = N1->getOperand(0).getValueType();
12583     return (N1VT == N1Op0VT || N1Op0VT != MVT::f128);
12584   }
12585   return false;
12586 }
12587 
12588 SDValue DAGCombiner::visitFCOPYSIGN(SDNode *N) {
12589   SDValue N0 = N->getOperand(0);
12590   SDValue N1 = N->getOperand(1);
12591   bool N0CFP = isConstantFPBuildVectorOrConstantFP(N0);
12592   bool N1CFP = isConstantFPBuildVectorOrConstantFP(N1);
12593   EVT VT = N->getValueType(0);
12594 
12595   if (N0CFP && N1CFP) // Constant fold
12596     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1);
12597 
12598   if (ConstantFPSDNode *N1C = isConstOrConstSplatFP(N->getOperand(1))) {
12599     const APFloat &V = N1C->getValueAPF();
12600     // copysign(x, c1) -> fabs(x)       iff ispos(c1)
12601     // copysign(x, c1) -> fneg(fabs(x)) iff isneg(c1)
12602     if (!V.isNegative()) {
12603       if (!LegalOperations || TLI.isOperationLegal(ISD::FABS, VT))
12604         return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
12605     } else {
12606       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
12607         return DAG.getNode(ISD::FNEG, SDLoc(N), VT,
12608                            DAG.getNode(ISD::FABS, SDLoc(N0), VT, N0));
12609     }
12610   }
12611 
12612   // copysign(fabs(x), y) -> copysign(x, y)
12613   // copysign(fneg(x), y) -> copysign(x, y)
12614   // copysign(copysign(x,z), y) -> copysign(x, y)
12615   if (N0.getOpcode() == ISD::FABS || N0.getOpcode() == ISD::FNEG ||
12616       N0.getOpcode() == ISD::FCOPYSIGN)
12617     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0.getOperand(0), N1);
12618 
12619   // copysign(x, abs(y)) -> abs(x)
12620   if (N1.getOpcode() == ISD::FABS)
12621     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
12622 
12623   // copysign(x, copysign(y,z)) -> copysign(x, z)
12624   if (N1.getOpcode() == ISD::FCOPYSIGN)
12625     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(1));
12626 
12627   // copysign(x, fp_extend(y)) -> copysign(x, y)
12628   // copysign(x, fp_round(y)) -> copysign(x, y)
12629   if (CanCombineFCOPYSIGN_EXTEND_ROUND(N))
12630     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(0));
12631 
12632   return SDValue();
12633 }
12634 
12635 SDValue DAGCombiner::visitFPOW(SDNode *N) {
12636   ConstantFPSDNode *ExponentC = isConstOrConstSplatFP(N->getOperand(1));
12637   if (!ExponentC)
12638     return SDValue();
12639 
12640   // Try to convert x ** (1/3) into cube root.
12641   // TODO: Handle the various flavors of long double.
12642   // TODO: Since we're approximating, we don't need an exact 1/3 exponent.
12643   //       Some range near 1/3 should be fine.
12644   EVT VT = N->getValueType(0);
12645   if ((VT == MVT::f32 && ExponentC->getValueAPF().isExactlyValue(1.0f/3.0f)) ||
12646       (VT == MVT::f64 && ExponentC->getValueAPF().isExactlyValue(1.0/3.0))) {
12647     // pow(-0.0, 1/3) = +0.0; cbrt(-0.0) = -0.0.
12648     // pow(-inf, 1/3) = +inf; cbrt(-inf) = -inf.
12649     // pow(-val, 1/3) =  nan; cbrt(-val) = -num.
12650     // For regular numbers, rounding may cause the results to differ.
12651     // Therefore, we require { nsz ninf nnan afn } for this transform.
12652     // TODO: We could select out the special cases if we don't have nsz/ninf.
12653     SDNodeFlags Flags = N->getFlags();
12654     if (!Flags.hasNoSignedZeros() || !Flags.hasNoInfs() || !Flags.hasNoNaNs() ||
12655         !Flags.hasApproximateFuncs())
12656       return SDValue();
12657 
12658     // Do not create a cbrt() libcall if the target does not have it, and do not
12659     // turn a pow that has lowering support into a cbrt() libcall.
12660     if (!DAG.getLibInfo().has(LibFunc_cbrt) ||
12661         (!DAG.getTargetLoweringInfo().isOperationExpand(ISD::FPOW, VT) &&
12662          DAG.getTargetLoweringInfo().isOperationExpand(ISD::FCBRT, VT)))
12663       return SDValue();
12664 
12665     return DAG.getNode(ISD::FCBRT, SDLoc(N), VT, N->getOperand(0), Flags);
12666   }
12667 
12668   // Try to convert x ** (1/4) and x ** (3/4) into square roots.
12669   // x ** (1/2) is canonicalized to sqrt, so we do not bother with that case.
12670   // TODO: This could be extended (using a target hook) to handle smaller
12671   // power-of-2 fractional exponents.
12672   bool ExponentIs025 = ExponentC->getValueAPF().isExactlyValue(0.25);
12673   bool ExponentIs075 = ExponentC->getValueAPF().isExactlyValue(0.75);
12674   if (ExponentIs025 || ExponentIs075) {
12675     // pow(-0.0, 0.25) = +0.0; sqrt(sqrt(-0.0)) = -0.0.
12676     // pow(-inf, 0.25) = +inf; sqrt(sqrt(-inf)) =  NaN.
12677     // pow(-0.0, 0.75) = +0.0; sqrt(-0.0) * sqrt(sqrt(-0.0)) = +0.0.
12678     // pow(-inf, 0.75) = +inf; sqrt(-inf) * sqrt(sqrt(-inf)) =  NaN.
12679     // For regular numbers, rounding may cause the results to differ.
12680     // Therefore, we require { nsz ninf afn } for this transform.
12681     // TODO: We could select out the special cases if we don't have nsz/ninf.
12682     SDNodeFlags Flags = N->getFlags();
12683 
12684     // We only need no signed zeros for the 0.25 case.
12685     if ((!Flags.hasNoSignedZeros() && ExponentIs025) || !Flags.hasNoInfs() ||
12686         !Flags.hasApproximateFuncs())
12687       return SDValue();
12688 
12689     // Don't double the number of libcalls. We are trying to inline fast code.
12690     if (!DAG.getTargetLoweringInfo().isOperationLegalOrCustom(ISD::FSQRT, VT))
12691       return SDValue();
12692 
12693     // Assume that libcalls are the smallest code.
12694     // TODO: This restriction should probably be lifted for vectors.
12695     if (DAG.getMachineFunction().getFunction().hasOptSize())
12696       return SDValue();
12697 
12698     // pow(X, 0.25) --> sqrt(sqrt(X))
12699     SDLoc DL(N);
12700     SDValue Sqrt = DAG.getNode(ISD::FSQRT, DL, VT, N->getOperand(0), Flags);
12701     SDValue SqrtSqrt = DAG.getNode(ISD::FSQRT, DL, VT, Sqrt, Flags);
12702     if (ExponentIs025)
12703       return SqrtSqrt;
12704     // pow(X, 0.75) --> sqrt(X) * sqrt(sqrt(X))
12705     return DAG.getNode(ISD::FMUL, DL, VT, Sqrt, SqrtSqrt, Flags);
12706   }
12707 
12708   return SDValue();
12709 }
12710 
12711 static SDValue foldFPToIntToFP(SDNode *N, SelectionDAG &DAG,
12712                                const TargetLowering &TLI) {
12713   // This optimization is guarded by a function attribute because it may produce
12714   // unexpected results. Ie, programs may be relying on the platform-specific
12715   // undefined behavior when the float-to-int conversion overflows.
12716   const Function &F = DAG.getMachineFunction().getFunction();
12717   Attribute StrictOverflow = F.getFnAttribute("strict-float-cast-overflow");
12718   if (StrictOverflow.getValueAsString().equals("false"))
12719     return SDValue();
12720 
12721   // We only do this if the target has legal ftrunc. Otherwise, we'd likely be
12722   // replacing casts with a libcall. We also must be allowed to ignore -0.0
12723   // because FTRUNC will return -0.0 for (-1.0, -0.0), but using integer
12724   // conversions would return +0.0.
12725   // FIXME: We should be able to use node-level FMF here.
12726   // TODO: If strict math, should we use FABS (+ range check for signed cast)?
12727   EVT VT = N->getValueType(0);
12728   if (!TLI.isOperationLegal(ISD::FTRUNC, VT) ||
12729       !DAG.getTarget().Options.NoSignedZerosFPMath)
12730     return SDValue();
12731 
12732   // fptosi/fptoui round towards zero, so converting from FP to integer and
12733   // back is the same as an 'ftrunc': [us]itofp (fpto[us]i X) --> ftrunc X
12734   SDValue N0 = N->getOperand(0);
12735   if (N->getOpcode() == ISD::SINT_TO_FP && N0.getOpcode() == ISD::FP_TO_SINT &&
12736       N0.getOperand(0).getValueType() == VT)
12737     return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0.getOperand(0));
12738 
12739   if (N->getOpcode() == ISD::UINT_TO_FP && N0.getOpcode() == ISD::FP_TO_UINT &&
12740       N0.getOperand(0).getValueType() == VT)
12741     return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0.getOperand(0));
12742 
12743   return SDValue();
12744 }
12745 
12746 SDValue DAGCombiner::visitSINT_TO_FP(SDNode *N) {
12747   SDValue N0 = N->getOperand(0);
12748   EVT VT = N->getValueType(0);
12749   EVT OpVT = N0.getValueType();
12750 
12751   // [us]itofp(undef) = 0, because the result value is bounded.
12752   if (N0.isUndef())
12753     return DAG.getConstantFP(0.0, SDLoc(N), VT);
12754 
12755   // fold (sint_to_fp c1) -> c1fp
12756   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
12757       // ...but only if the target supports immediate floating-point values
12758       (!LegalOperations ||
12759        TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT)))
12760     return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
12761 
12762   // If the input is a legal type, and SINT_TO_FP is not legal on this target,
12763   // but UINT_TO_FP is legal on this target, try to convert.
12764   if (!hasOperation(ISD::SINT_TO_FP, OpVT) &&
12765       hasOperation(ISD::UINT_TO_FP, OpVT)) {
12766     // If the sign bit is known to be zero, we can change this to UINT_TO_FP.
12767     if (DAG.SignBitIsZero(N0))
12768       return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
12769   }
12770 
12771   // The next optimizations are desirable only if SELECT_CC can be lowered.
12772   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
12773     // fold (sint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
12774     if (N0.getOpcode() == ISD::SETCC && N0.getValueType() == MVT::i1 &&
12775         !VT.isVector() &&
12776         (!LegalOperations ||
12777          TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) {
12778       SDLoc DL(N);
12779       SDValue Ops[] =
12780         { N0.getOperand(0), N0.getOperand(1),
12781           DAG.getConstantFP(-1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
12782           N0.getOperand(2) };
12783       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
12784     }
12785 
12786     // fold (sint_to_fp (zext (setcc x, y, cc))) ->
12787     //      (select_cc x, y, 1.0, 0.0,, cc)
12788     if (N0.getOpcode() == ISD::ZERO_EXTEND &&
12789         N0.getOperand(0).getOpcode() == ISD::SETCC &&!VT.isVector() &&
12790         (!LegalOperations ||
12791          TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) {
12792       SDLoc DL(N);
12793       SDValue Ops[] =
12794         { N0.getOperand(0).getOperand(0), N0.getOperand(0).getOperand(1),
12795           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
12796           N0.getOperand(0).getOperand(2) };
12797       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
12798     }
12799   }
12800 
12801   if (SDValue FTrunc = foldFPToIntToFP(N, DAG, TLI))
12802     return FTrunc;
12803 
12804   return SDValue();
12805 }
12806 
12807 SDValue DAGCombiner::visitUINT_TO_FP(SDNode *N) {
12808   SDValue N0 = N->getOperand(0);
12809   EVT VT = N->getValueType(0);
12810   EVT OpVT = N0.getValueType();
12811 
12812   // [us]itofp(undef) = 0, because the result value is bounded.
12813   if (N0.isUndef())
12814     return DAG.getConstantFP(0.0, SDLoc(N), VT);
12815 
12816   // fold (uint_to_fp c1) -> c1fp
12817   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
12818       // ...but only if the target supports immediate floating-point values
12819       (!LegalOperations ||
12820        TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT)))
12821     return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
12822 
12823   // If the input is a legal type, and UINT_TO_FP is not legal on this target,
12824   // but SINT_TO_FP is legal on this target, try to convert.
12825   if (!hasOperation(ISD::UINT_TO_FP, OpVT) &&
12826       hasOperation(ISD::SINT_TO_FP, OpVT)) {
12827     // If the sign bit is known to be zero, we can change this to SINT_TO_FP.
12828     if (DAG.SignBitIsZero(N0))
12829       return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
12830   }
12831 
12832   // The next optimizations are desirable only if SELECT_CC can be lowered.
12833   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
12834     // fold (uint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
12835     if (N0.getOpcode() == ISD::SETCC && !VT.isVector() &&
12836         (!LegalOperations ||
12837          TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) {
12838       SDLoc DL(N);
12839       SDValue Ops[] =
12840         { N0.getOperand(0), N0.getOperand(1),
12841           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
12842           N0.getOperand(2) };
12843       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
12844     }
12845   }
12846 
12847   if (SDValue FTrunc = foldFPToIntToFP(N, DAG, TLI))
12848     return FTrunc;
12849 
12850   return SDValue();
12851 }
12852 
12853 // Fold (fp_to_{s/u}int ({s/u}int_to_fpx)) -> zext x, sext x, trunc x, or x
12854 static SDValue FoldIntToFPToInt(SDNode *N, SelectionDAG &DAG) {
12855   SDValue N0 = N->getOperand(0);
12856   EVT VT = N->getValueType(0);
12857 
12858   if (N0.getOpcode() != ISD::UINT_TO_FP && N0.getOpcode() != ISD::SINT_TO_FP)
12859     return SDValue();
12860 
12861   SDValue Src = N0.getOperand(0);
12862   EVT SrcVT = Src.getValueType();
12863   bool IsInputSigned = N0.getOpcode() == ISD::SINT_TO_FP;
12864   bool IsOutputSigned = N->getOpcode() == ISD::FP_TO_SINT;
12865 
12866   // We can safely assume the conversion won't overflow the output range,
12867   // because (for example) (uint8_t)18293.f is undefined behavior.
12868 
12869   // Since we can assume the conversion won't overflow, our decision as to
12870   // whether the input will fit in the float should depend on the minimum
12871   // of the input range and output range.
12872 
12873   // This means this is also safe for a signed input and unsigned output, since
12874   // a negative input would lead to undefined behavior.
12875   unsigned InputSize = (int)SrcVT.getScalarSizeInBits() - IsInputSigned;
12876   unsigned OutputSize = (int)VT.getScalarSizeInBits() - IsOutputSigned;
12877   unsigned ActualSize = std::min(InputSize, OutputSize);
12878   const fltSemantics &sem = DAG.EVTToAPFloatSemantics(N0.getValueType());
12879 
12880   // We can only fold away the float conversion if the input range can be
12881   // represented exactly in the float range.
12882   if (APFloat::semanticsPrecision(sem) >= ActualSize) {
12883     if (VT.getScalarSizeInBits() > SrcVT.getScalarSizeInBits()) {
12884       unsigned ExtOp = IsInputSigned && IsOutputSigned ? ISD::SIGN_EXTEND
12885                                                        : ISD::ZERO_EXTEND;
12886       return DAG.getNode(ExtOp, SDLoc(N), VT, Src);
12887     }
12888     if (VT.getScalarSizeInBits() < SrcVT.getScalarSizeInBits())
12889       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Src);
12890     return DAG.getBitcast(VT, Src);
12891   }
12892   return SDValue();
12893 }
12894 
12895 SDValue DAGCombiner::visitFP_TO_SINT(SDNode *N) {
12896   SDValue N0 = N->getOperand(0);
12897   EVT VT = N->getValueType(0);
12898 
12899   // fold (fp_to_sint undef) -> undef
12900   if (N0.isUndef())
12901     return DAG.getUNDEF(VT);
12902 
12903   // fold (fp_to_sint c1fp) -> c1
12904   if (isConstantFPBuildVectorOrConstantFP(N0))
12905     return DAG.getNode(ISD::FP_TO_SINT, SDLoc(N), VT, N0);
12906 
12907   return FoldIntToFPToInt(N, DAG);
12908 }
12909 
12910 SDValue DAGCombiner::visitFP_TO_UINT(SDNode *N) {
12911   SDValue N0 = N->getOperand(0);
12912   EVT VT = N->getValueType(0);
12913 
12914   // fold (fp_to_uint undef) -> undef
12915   if (N0.isUndef())
12916     return DAG.getUNDEF(VT);
12917 
12918   // fold (fp_to_uint c1fp) -> c1
12919   if (isConstantFPBuildVectorOrConstantFP(N0))
12920     return DAG.getNode(ISD::FP_TO_UINT, SDLoc(N), VT, N0);
12921 
12922   return FoldIntToFPToInt(N, DAG);
12923 }
12924 
12925 SDValue DAGCombiner::visitFP_ROUND(SDNode *N) {
12926   SDValue N0 = N->getOperand(0);
12927   SDValue N1 = N->getOperand(1);
12928   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
12929   EVT VT = N->getValueType(0);
12930 
12931   // fold (fp_round c1fp) -> c1fp
12932   if (N0CFP)
12933     return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT, N0, N1);
12934 
12935   // fold (fp_round (fp_extend x)) -> x
12936   if (N0.getOpcode() == ISD::FP_EXTEND && VT == N0.getOperand(0).getValueType())
12937     return N0.getOperand(0);
12938 
12939   // fold (fp_round (fp_round x)) -> (fp_round x)
12940   if (N0.getOpcode() == ISD::FP_ROUND) {
12941     const bool NIsTrunc = N->getConstantOperandVal(1) == 1;
12942     const bool N0IsTrunc = N0.getConstantOperandVal(1) == 1;
12943 
12944     // Skip this folding if it results in an fp_round from f80 to f16.
12945     //
12946     // f80 to f16 always generates an expensive (and as yet, unimplemented)
12947     // libcall to __truncxfhf2 instead of selecting native f16 conversion
12948     // instructions from f32 or f64.  Moreover, the first (value-preserving)
12949     // fp_round from f80 to either f32 or f64 may become a NOP in platforms like
12950     // x86.
12951     if (N0.getOperand(0).getValueType() == MVT::f80 && VT == MVT::f16)
12952       return SDValue();
12953 
12954     // If the first fp_round isn't a value preserving truncation, it might
12955     // introduce a tie in the second fp_round, that wouldn't occur in the
12956     // single-step fp_round we want to fold to.
12957     // In other words, double rounding isn't the same as rounding.
12958     // Also, this is a value preserving truncation iff both fp_round's are.
12959     if (DAG.getTarget().Options.UnsafeFPMath || N0IsTrunc) {
12960       SDLoc DL(N);
12961       return DAG.getNode(ISD::FP_ROUND, DL, VT, N0.getOperand(0),
12962                          DAG.getIntPtrConstant(NIsTrunc && N0IsTrunc, DL));
12963     }
12964   }
12965 
12966   // fold (fp_round (copysign X, Y)) -> (copysign (fp_round X), Y)
12967   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse()) {
12968     SDValue Tmp = DAG.getNode(ISD::FP_ROUND, SDLoc(N0), VT,
12969                               N0.getOperand(0), N1);
12970     AddToWorklist(Tmp.getNode());
12971     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT,
12972                        Tmp, N0.getOperand(1));
12973   }
12974 
12975   if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N))
12976     return NewVSel;
12977 
12978   return SDValue();
12979 }
12980 
12981 SDValue DAGCombiner::visitFP_ROUND_INREG(SDNode *N) {
12982   SDValue N0 = N->getOperand(0);
12983   EVT VT = N->getValueType(0);
12984   EVT EVT = cast<VTSDNode>(N->getOperand(1))->getVT();
12985   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
12986 
12987   // fold (fp_round_inreg c1fp) -> c1fp
12988   if (N0CFP && isTypeLegal(EVT)) {
12989     SDLoc DL(N);
12990     SDValue Round = DAG.getConstantFP(*N0CFP->getConstantFPValue(), DL, EVT);
12991     return DAG.getNode(ISD::FP_EXTEND, DL, VT, Round);
12992   }
12993 
12994   return SDValue();
12995 }
12996 
12997 SDValue DAGCombiner::visitFP_EXTEND(SDNode *N) {
12998   SDValue N0 = N->getOperand(0);
12999   EVT VT = N->getValueType(0);
13000 
13001   // If this is fp_round(fpextend), don't fold it, allow ourselves to be folded.
13002   if (N->hasOneUse() &&
13003       N->use_begin()->getOpcode() == ISD::FP_ROUND)
13004     return SDValue();
13005 
13006   // fold (fp_extend c1fp) -> c1fp
13007   if (isConstantFPBuildVectorOrConstantFP(N0))
13008     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, N0);
13009 
13010   // fold (fp_extend (fp16_to_fp op)) -> (fp16_to_fp op)
13011   if (N0.getOpcode() == ISD::FP16_TO_FP &&
13012       TLI.getOperationAction(ISD::FP16_TO_FP, VT) == TargetLowering::Legal)
13013     return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), VT, N0.getOperand(0));
13014 
13015   // Turn fp_extend(fp_round(X, 1)) -> x since the fp_round doesn't affect the
13016   // value of X.
13017   if (N0.getOpcode() == ISD::FP_ROUND
13018       && N0.getConstantOperandVal(1) == 1) {
13019     SDValue In = N0.getOperand(0);
13020     if (In.getValueType() == VT) return In;
13021     if (VT.bitsLT(In.getValueType()))
13022       return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT,
13023                          In, N0.getOperand(1));
13024     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, In);
13025   }
13026 
13027   // fold (fpext (load x)) -> (fpext (fptrunc (extload x)))
13028   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
13029        TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
13030     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
13031     SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
13032                                      LN0->getChain(),
13033                                      LN0->getBasePtr(), N0.getValueType(),
13034                                      LN0->getMemOperand());
13035     CombineTo(N, ExtLoad);
13036     CombineTo(N0.getNode(),
13037               DAG.getNode(ISD::FP_ROUND, SDLoc(N0),
13038                           N0.getValueType(), ExtLoad,
13039                           DAG.getIntPtrConstant(1, SDLoc(N0))),
13040               ExtLoad.getValue(1));
13041     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
13042   }
13043 
13044   if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N))
13045     return NewVSel;
13046 
13047   return SDValue();
13048 }
13049 
13050 SDValue DAGCombiner::visitFCEIL(SDNode *N) {
13051   SDValue N0 = N->getOperand(0);
13052   EVT VT = N->getValueType(0);
13053 
13054   // fold (fceil c1) -> fceil(c1)
13055   if (isConstantFPBuildVectorOrConstantFP(N0))
13056     return DAG.getNode(ISD::FCEIL, SDLoc(N), VT, N0);
13057 
13058   return SDValue();
13059 }
13060 
13061 SDValue DAGCombiner::visitFTRUNC(SDNode *N) {
13062   SDValue N0 = N->getOperand(0);
13063   EVT VT = N->getValueType(0);
13064 
13065   // fold (ftrunc c1) -> ftrunc(c1)
13066   if (isConstantFPBuildVectorOrConstantFP(N0))
13067     return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0);
13068 
13069   // fold ftrunc (known rounded int x) -> x
13070   // ftrunc is a part of fptosi/fptoui expansion on some targets, so this is
13071   // likely to be generated to extract integer from a rounded floating value.
13072   switch (N0.getOpcode()) {
13073   default: break;
13074   case ISD::FRINT:
13075   case ISD::FTRUNC:
13076   case ISD::FNEARBYINT:
13077   case ISD::FFLOOR:
13078   case ISD::FCEIL:
13079     return N0;
13080   }
13081 
13082   return SDValue();
13083 }
13084 
13085 SDValue DAGCombiner::visitFFLOOR(SDNode *N) {
13086   SDValue N0 = N->getOperand(0);
13087   EVT VT = N->getValueType(0);
13088 
13089   // fold (ffloor c1) -> ffloor(c1)
13090   if (isConstantFPBuildVectorOrConstantFP(N0))
13091     return DAG.getNode(ISD::FFLOOR, SDLoc(N), VT, N0);
13092 
13093   return SDValue();
13094 }
13095 
13096 // FIXME: FNEG and FABS have a lot in common; refactor.
13097 SDValue DAGCombiner::visitFNEG(SDNode *N) {
13098   SDValue N0 = N->getOperand(0);
13099   EVT VT = N->getValueType(0);
13100 
13101   // Constant fold FNEG.
13102   if (isConstantFPBuildVectorOrConstantFP(N0))
13103     return DAG.getNode(ISD::FNEG, SDLoc(N), VT, N0);
13104 
13105   if (isNegatibleForFree(N0, LegalOperations, DAG.getTargetLoweringInfo(),
13106                          &DAG.getTarget().Options, ForCodeSize))
13107     return GetNegatedExpression(N0, DAG, LegalOperations, ForCodeSize);
13108 
13109   // Transform fneg(bitconvert(x)) -> bitconvert(x ^ sign) to avoid loading
13110   // constant pool values.
13111   if (!TLI.isFNegFree(VT) &&
13112       N0.getOpcode() == ISD::BITCAST &&
13113       N0.getNode()->hasOneUse()) {
13114     SDValue Int = N0.getOperand(0);
13115     EVT IntVT = Int.getValueType();
13116     if (IntVT.isInteger() && !IntVT.isVector()) {
13117       APInt SignMask;
13118       if (N0.getValueType().isVector()) {
13119         // For a vector, get a mask such as 0x80... per scalar element
13120         // and splat it.
13121         SignMask = APInt::getSignMask(N0.getScalarValueSizeInBits());
13122         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
13123       } else {
13124         // For a scalar, just generate 0x80...
13125         SignMask = APInt::getSignMask(IntVT.getSizeInBits());
13126       }
13127       SDLoc DL0(N0);
13128       Int = DAG.getNode(ISD::XOR, DL0, IntVT, Int,
13129                         DAG.getConstant(SignMask, DL0, IntVT));
13130       AddToWorklist(Int.getNode());
13131       return DAG.getBitcast(VT, Int);
13132     }
13133   }
13134 
13135   // (fneg (fmul c, x)) -> (fmul -c, x)
13136   if (N0.getOpcode() == ISD::FMUL &&
13137       (N0.getNode()->hasOneUse() || !TLI.isFNegFree(VT))) {
13138     ConstantFPSDNode *CFP1 = dyn_cast<ConstantFPSDNode>(N0.getOperand(1));
13139     if (CFP1) {
13140       APFloat CVal = CFP1->getValueAPF();
13141       CVal.changeSign();
13142       if (Level >= AfterLegalizeDAG &&
13143           (TLI.isFPImmLegal(CVal, VT, ForCodeSize) ||
13144            TLI.isOperationLegal(ISD::ConstantFP, VT)))
13145         return DAG.getNode(
13146             ISD::FMUL, SDLoc(N), VT, N0.getOperand(0),
13147             DAG.getNode(ISD::FNEG, SDLoc(N), VT, N0.getOperand(1)),
13148             N0->getFlags());
13149     }
13150   }
13151 
13152   return SDValue();
13153 }
13154 
13155 static SDValue visitFMinMax(SelectionDAG &DAG, SDNode *N,
13156                             APFloat (*Op)(const APFloat &, const APFloat &)) {
13157   SDValue N0 = N->getOperand(0);
13158   SDValue N1 = N->getOperand(1);
13159   EVT VT = N->getValueType(0);
13160   const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
13161   const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
13162 
13163   if (N0CFP && N1CFP) {
13164     const APFloat &C0 = N0CFP->getValueAPF();
13165     const APFloat &C1 = N1CFP->getValueAPF();
13166     return DAG.getConstantFP(Op(C0, C1), SDLoc(N), VT);
13167   }
13168 
13169   // Canonicalize to constant on RHS.
13170   if (isConstantFPBuildVectorOrConstantFP(N0) &&
13171       !isConstantFPBuildVectorOrConstantFP(N1))
13172     return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N1, N0);
13173 
13174   return SDValue();
13175 }
13176 
13177 SDValue DAGCombiner::visitFMINNUM(SDNode *N) {
13178   return visitFMinMax(DAG, N, minnum);
13179 }
13180 
13181 SDValue DAGCombiner::visitFMAXNUM(SDNode *N) {
13182   return visitFMinMax(DAG, N, maxnum);
13183 }
13184 
13185 SDValue DAGCombiner::visitFMINIMUM(SDNode *N) {
13186   return visitFMinMax(DAG, N, minimum);
13187 }
13188 
13189 SDValue DAGCombiner::visitFMAXIMUM(SDNode *N) {
13190   return visitFMinMax(DAG, N, maximum);
13191 }
13192 
13193 SDValue DAGCombiner::visitFABS(SDNode *N) {
13194   SDValue N0 = N->getOperand(0);
13195   EVT VT = N->getValueType(0);
13196 
13197   // fold (fabs c1) -> fabs(c1)
13198   if (isConstantFPBuildVectorOrConstantFP(N0))
13199     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
13200 
13201   // fold (fabs (fabs x)) -> (fabs x)
13202   if (N0.getOpcode() == ISD::FABS)
13203     return N->getOperand(0);
13204 
13205   // fold (fabs (fneg x)) -> (fabs x)
13206   // fold (fabs (fcopysign x, y)) -> (fabs x)
13207   if (N0.getOpcode() == ISD::FNEG || N0.getOpcode() == ISD::FCOPYSIGN)
13208     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0.getOperand(0));
13209 
13210   // fabs(bitcast(x)) -> bitcast(x & ~sign) to avoid constant pool loads.
13211   if (!TLI.isFAbsFree(VT) && N0.getOpcode() == ISD::BITCAST && N0.hasOneUse()) {
13212     SDValue Int = N0.getOperand(0);
13213     EVT IntVT = Int.getValueType();
13214     if (IntVT.isInteger() && !IntVT.isVector()) {
13215       APInt SignMask;
13216       if (N0.getValueType().isVector()) {
13217         // For a vector, get a mask such as 0x7f... per scalar element
13218         // and splat it.
13219         SignMask = ~APInt::getSignMask(N0.getScalarValueSizeInBits());
13220         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
13221       } else {
13222         // For a scalar, just generate 0x7f...
13223         SignMask = ~APInt::getSignMask(IntVT.getSizeInBits());
13224       }
13225       SDLoc DL(N0);
13226       Int = DAG.getNode(ISD::AND, DL, IntVT, Int,
13227                         DAG.getConstant(SignMask, DL, IntVT));
13228       AddToWorklist(Int.getNode());
13229       return DAG.getBitcast(N->getValueType(0), Int);
13230     }
13231   }
13232 
13233   return SDValue();
13234 }
13235 
13236 SDValue DAGCombiner::visitBRCOND(SDNode *N) {
13237   SDValue Chain = N->getOperand(0);
13238   SDValue N1 = N->getOperand(1);
13239   SDValue N2 = N->getOperand(2);
13240 
13241   // If N is a constant we could fold this into a fallthrough or unconditional
13242   // branch. However that doesn't happen very often in normal code, because
13243   // Instcombine/SimplifyCFG should have handled the available opportunities.
13244   // If we did this folding here, it would be necessary to update the
13245   // MachineBasicBlock CFG, which is awkward.
13246 
13247   // fold a brcond with a setcc condition into a BR_CC node if BR_CC is legal
13248   // on the target.
13249   if (N1.getOpcode() == ISD::SETCC &&
13250       TLI.isOperationLegalOrCustom(ISD::BR_CC,
13251                                    N1.getOperand(0).getValueType())) {
13252     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
13253                        Chain, N1.getOperand(2),
13254                        N1.getOperand(0), N1.getOperand(1), N2);
13255   }
13256 
13257   if (N1.hasOneUse()) {
13258     if (SDValue NewN1 = rebuildSetCC(N1))
13259       return DAG.getNode(ISD::BRCOND, SDLoc(N), MVT::Other, Chain, NewN1, N2);
13260   }
13261 
13262   return SDValue();
13263 }
13264 
13265 SDValue DAGCombiner::rebuildSetCC(SDValue N) {
13266   if (N.getOpcode() == ISD::SRL ||
13267       (N.getOpcode() == ISD::TRUNCATE &&
13268        (N.getOperand(0).hasOneUse() &&
13269         N.getOperand(0).getOpcode() == ISD::SRL))) {
13270     // Look pass the truncate.
13271     if (N.getOpcode() == ISD::TRUNCATE)
13272       N = N.getOperand(0);
13273 
13274     // Match this pattern so that we can generate simpler code:
13275     //
13276     //   %a = ...
13277     //   %b = and i32 %a, 2
13278     //   %c = srl i32 %b, 1
13279     //   brcond i32 %c ...
13280     //
13281     // into
13282     //
13283     //   %a = ...
13284     //   %b = and i32 %a, 2
13285     //   %c = setcc eq %b, 0
13286     //   brcond %c ...
13287     //
13288     // This applies only when the AND constant value has one bit set and the
13289     // SRL constant is equal to the log2 of the AND constant. The back-end is
13290     // smart enough to convert the result into a TEST/JMP sequence.
13291     SDValue Op0 = N.getOperand(0);
13292     SDValue Op1 = N.getOperand(1);
13293 
13294     if (Op0.getOpcode() == ISD::AND && Op1.getOpcode() == ISD::Constant) {
13295       SDValue AndOp1 = Op0.getOperand(1);
13296 
13297       if (AndOp1.getOpcode() == ISD::Constant) {
13298         const APInt &AndConst = cast<ConstantSDNode>(AndOp1)->getAPIntValue();
13299 
13300         if (AndConst.isPowerOf2() &&
13301             cast<ConstantSDNode>(Op1)->getAPIntValue() == AndConst.logBase2()) {
13302           SDLoc DL(N);
13303           return DAG.getSetCC(DL, getSetCCResultType(Op0.getValueType()),
13304                               Op0, DAG.getConstant(0, DL, Op0.getValueType()),
13305                               ISD::SETNE);
13306         }
13307       }
13308     }
13309   }
13310 
13311   // Transform br(xor(x, y)) -> br(x != y)
13312   // Transform br(xor(xor(x,y), 1)) -> br (x == y)
13313   if (N.getOpcode() == ISD::XOR) {
13314     // Because we may call this on a speculatively constructed
13315     // SimplifiedSetCC Node, we need to simplify this node first.
13316     // Ideally this should be folded into SimplifySetCC and not
13317     // here. For now, grab a handle to N so we don't lose it from
13318     // replacements interal to the visit.
13319     HandleSDNode XORHandle(N);
13320     while (N.getOpcode() == ISD::XOR) {
13321       SDValue Tmp = visitXOR(N.getNode());
13322       // No simplification done.
13323       if (!Tmp.getNode())
13324         break;
13325       // Returning N is form in-visit replacement that may invalidated
13326       // N. Grab value from Handle.
13327       if (Tmp.getNode() == N.getNode())
13328         N = XORHandle.getValue();
13329       else // Node simplified. Try simplifying again.
13330         N = Tmp;
13331     }
13332 
13333     if (N.getOpcode() != ISD::XOR)
13334       return N;
13335 
13336     SDNode *TheXor = N.getNode();
13337 
13338     SDValue Op0 = TheXor->getOperand(0);
13339     SDValue Op1 = TheXor->getOperand(1);
13340 
13341     if (Op0.getOpcode() != ISD::SETCC && Op1.getOpcode() != ISD::SETCC) {
13342       bool Equal = false;
13343       if (isOneConstant(Op0) && Op0.hasOneUse() &&
13344           Op0.getOpcode() == ISD::XOR) {
13345         TheXor = Op0.getNode();
13346         Equal = true;
13347       }
13348 
13349       EVT SetCCVT = N.getValueType();
13350       if (LegalTypes)
13351         SetCCVT = getSetCCResultType(SetCCVT);
13352       // Replace the uses of XOR with SETCC
13353       return DAG.getSetCC(SDLoc(TheXor), SetCCVT, Op0, Op1,
13354                           Equal ? ISD::SETEQ : ISD::SETNE);
13355     }
13356   }
13357 
13358   return SDValue();
13359 }
13360 
13361 // Operand List for BR_CC: Chain, CondCC, CondLHS, CondRHS, DestBB.
13362 //
13363 SDValue DAGCombiner::visitBR_CC(SDNode *N) {
13364   CondCodeSDNode *CC = cast<CondCodeSDNode>(N->getOperand(1));
13365   SDValue CondLHS = N->getOperand(2), CondRHS = N->getOperand(3);
13366 
13367   // If N is a constant we could fold this into a fallthrough or unconditional
13368   // branch. However that doesn't happen very often in normal code, because
13369   // Instcombine/SimplifyCFG should have handled the available opportunities.
13370   // If we did this folding here, it would be necessary to update the
13371   // MachineBasicBlock CFG, which is awkward.
13372 
13373   // Use SimplifySetCC to simplify SETCC's.
13374   SDValue Simp = SimplifySetCC(getSetCCResultType(CondLHS.getValueType()),
13375                                CondLHS, CondRHS, CC->get(), SDLoc(N),
13376                                false);
13377   if (Simp.getNode()) AddToWorklist(Simp.getNode());
13378 
13379   // fold to a simpler setcc
13380   if (Simp.getNode() && Simp.getOpcode() == ISD::SETCC)
13381     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
13382                        N->getOperand(0), Simp.getOperand(2),
13383                        Simp.getOperand(0), Simp.getOperand(1),
13384                        N->getOperand(4));
13385 
13386   return SDValue();
13387 }
13388 
13389 /// Return true if 'Use' is a load or a store that uses N as its base pointer
13390 /// and that N may be folded in the load / store addressing mode.
13391 static bool canFoldInAddressingMode(SDNode *N, SDNode *Use,
13392                                     SelectionDAG &DAG,
13393                                     const TargetLowering &TLI) {
13394   EVT VT;
13395   unsigned AS;
13396 
13397   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(Use)) {
13398     if (LD->isIndexed() || LD->getBasePtr().getNode() != N)
13399       return false;
13400     VT = LD->getMemoryVT();
13401     AS = LD->getAddressSpace();
13402   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(Use)) {
13403     if (ST->isIndexed() || ST->getBasePtr().getNode() != N)
13404       return false;
13405     VT = ST->getMemoryVT();
13406     AS = ST->getAddressSpace();
13407   } else
13408     return false;
13409 
13410   TargetLowering::AddrMode AM;
13411   if (N->getOpcode() == ISD::ADD) {
13412     AM.HasBaseReg = true;
13413     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
13414     if (Offset)
13415       // [reg +/- imm]
13416       AM.BaseOffs = Offset->getSExtValue();
13417     else
13418       // [reg +/- reg]
13419       AM.Scale = 1;
13420   } else if (N->getOpcode() == ISD::SUB) {
13421     AM.HasBaseReg = true;
13422     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
13423     if (Offset)
13424       // [reg +/- imm]
13425       AM.BaseOffs = -Offset->getSExtValue();
13426     else
13427       // [reg +/- reg]
13428       AM.Scale = 1;
13429   } else
13430     return false;
13431 
13432   return TLI.isLegalAddressingMode(DAG.getDataLayout(), AM,
13433                                    VT.getTypeForEVT(*DAG.getContext()), AS);
13434 }
13435 
13436 /// Try turning a load/store into a pre-indexed load/store when the base
13437 /// pointer is an add or subtract and it has other uses besides the load/store.
13438 /// After the transformation, the new indexed load/store has effectively folded
13439 /// the add/subtract in and all of its other uses are redirected to the
13440 /// new load/store.
13441 bool DAGCombiner::CombineToPreIndexedLoadStore(SDNode *N) {
13442   if (Level < AfterLegalizeDAG)
13443     return false;
13444 
13445   bool isLoad = true;
13446   SDValue Ptr;
13447   EVT VT;
13448   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
13449     if (LD->isIndexed())
13450       return false;
13451     VT = LD->getMemoryVT();
13452     if (!TLI.isIndexedLoadLegal(ISD::PRE_INC, VT) &&
13453         !TLI.isIndexedLoadLegal(ISD::PRE_DEC, VT))
13454       return false;
13455     Ptr = LD->getBasePtr();
13456   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
13457     if (ST->isIndexed())
13458       return false;
13459     VT = ST->getMemoryVT();
13460     if (!TLI.isIndexedStoreLegal(ISD::PRE_INC, VT) &&
13461         !TLI.isIndexedStoreLegal(ISD::PRE_DEC, VT))
13462       return false;
13463     Ptr = ST->getBasePtr();
13464     isLoad = false;
13465   } else {
13466     return false;
13467   }
13468 
13469   // If the pointer is not an add/sub, or if it doesn't have multiple uses, bail
13470   // out.  There is no reason to make this a preinc/predec.
13471   if ((Ptr.getOpcode() != ISD::ADD && Ptr.getOpcode() != ISD::SUB) ||
13472       Ptr.getNode()->hasOneUse())
13473     return false;
13474 
13475   // Ask the target to do addressing mode selection.
13476   SDValue BasePtr;
13477   SDValue Offset;
13478   ISD::MemIndexedMode AM = ISD::UNINDEXED;
13479   if (!TLI.getPreIndexedAddressParts(N, BasePtr, Offset, AM, DAG))
13480     return false;
13481 
13482   // Backends without true r+i pre-indexed forms may need to pass a
13483   // constant base with a variable offset so that constant coercion
13484   // will work with the patterns in canonical form.
13485   bool Swapped = false;
13486   if (isa<ConstantSDNode>(BasePtr)) {
13487     std::swap(BasePtr, Offset);
13488     Swapped = true;
13489   }
13490 
13491   // Don't create a indexed load / store with zero offset.
13492   if (isNullConstant(Offset))
13493     return false;
13494 
13495   // Try turning it into a pre-indexed load / store except when:
13496   // 1) The new base ptr is a frame index.
13497   // 2) If N is a store and the new base ptr is either the same as or is a
13498   //    predecessor of the value being stored.
13499   // 3) Another use of old base ptr is a predecessor of N. If ptr is folded
13500   //    that would create a cycle.
13501   // 4) All uses are load / store ops that use it as old base ptr.
13502 
13503   // Check #1.  Preinc'ing a frame index would require copying the stack pointer
13504   // (plus the implicit offset) to a register to preinc anyway.
13505   if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
13506     return false;
13507 
13508   // Check #2.
13509   if (!isLoad) {
13510     SDValue Val = cast<StoreSDNode>(N)->getValue();
13511 
13512     // Would require a copy.
13513     if (Val == BasePtr)
13514       return false;
13515 
13516     // Would create a cycle.
13517     if (Val == Ptr || Ptr->isPredecessorOf(Val.getNode()))
13518       return false;
13519   }
13520 
13521   // Caches for hasPredecessorHelper.
13522   SmallPtrSet<const SDNode *, 32> Visited;
13523   SmallVector<const SDNode *, 16> Worklist;
13524   Worklist.push_back(N);
13525 
13526   // If the offset is a constant, there may be other adds of constants that
13527   // can be folded with this one. We should do this to avoid having to keep
13528   // a copy of the original base pointer.
13529   SmallVector<SDNode *, 16> OtherUses;
13530   if (isa<ConstantSDNode>(Offset))
13531     for (SDNode::use_iterator UI = BasePtr.getNode()->use_begin(),
13532                               UE = BasePtr.getNode()->use_end();
13533          UI != UE; ++UI) {
13534       SDUse &Use = UI.getUse();
13535       // Skip the use that is Ptr and uses of other results from BasePtr's
13536       // node (important for nodes that return multiple results).
13537       if (Use.getUser() == Ptr.getNode() || Use != BasePtr)
13538         continue;
13539 
13540       if (SDNode::hasPredecessorHelper(Use.getUser(), Visited, Worklist))
13541         continue;
13542 
13543       if (Use.getUser()->getOpcode() != ISD::ADD &&
13544           Use.getUser()->getOpcode() != ISD::SUB) {
13545         OtherUses.clear();
13546         break;
13547       }
13548 
13549       SDValue Op1 = Use.getUser()->getOperand((UI.getOperandNo() + 1) & 1);
13550       if (!isa<ConstantSDNode>(Op1)) {
13551         OtherUses.clear();
13552         break;
13553       }
13554 
13555       // FIXME: In some cases, we can be smarter about this.
13556       if (Op1.getValueType() != Offset.getValueType()) {
13557         OtherUses.clear();
13558         break;
13559       }
13560 
13561       OtherUses.push_back(Use.getUser());
13562     }
13563 
13564   if (Swapped)
13565     std::swap(BasePtr, Offset);
13566 
13567   // Now check for #3 and #4.
13568   bool RealUse = false;
13569 
13570   for (SDNode *Use : Ptr.getNode()->uses()) {
13571     if (Use == N)
13572       continue;
13573     if (SDNode::hasPredecessorHelper(Use, Visited, Worklist))
13574       return false;
13575 
13576     // If Ptr may be folded in addressing mode of other use, then it's
13577     // not profitable to do this transformation.
13578     if (!canFoldInAddressingMode(Ptr.getNode(), Use, DAG, TLI))
13579       RealUse = true;
13580   }
13581 
13582   if (!RealUse)
13583     return false;
13584 
13585   SDValue Result;
13586   if (isLoad)
13587     Result = DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
13588                                 BasePtr, Offset, AM);
13589   else
13590     Result = DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
13591                                  BasePtr, Offset, AM);
13592   ++PreIndexedNodes;
13593   ++NodesCombined;
13594   LLVM_DEBUG(dbgs() << "\nReplacing.4 "; N->dump(&DAG); dbgs() << "\nWith: ";
13595              Result.getNode()->dump(&DAG); dbgs() << '\n');
13596   WorklistRemover DeadNodes(*this);
13597   if (isLoad) {
13598     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
13599     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
13600   } else {
13601     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
13602   }
13603 
13604   // Finally, since the node is now dead, remove it from the graph.
13605   deleteAndRecombine(N);
13606 
13607   if (Swapped)
13608     std::swap(BasePtr, Offset);
13609 
13610   // Replace other uses of BasePtr that can be updated to use Ptr
13611   for (unsigned i = 0, e = OtherUses.size(); i != e; ++i) {
13612     unsigned OffsetIdx = 1;
13613     if (OtherUses[i]->getOperand(OffsetIdx).getNode() == BasePtr.getNode())
13614       OffsetIdx = 0;
13615     assert(OtherUses[i]->getOperand(!OffsetIdx).getNode() ==
13616            BasePtr.getNode() && "Expected BasePtr operand");
13617 
13618     // We need to replace ptr0 in the following expression:
13619     //   x0 * offset0 + y0 * ptr0 = t0
13620     // knowing that
13621     //   x1 * offset1 + y1 * ptr0 = t1 (the indexed load/store)
13622     //
13623     // where x0, x1, y0 and y1 in {-1, 1} are given by the types of the
13624     // indexed load/store and the expression that needs to be re-written.
13625     //
13626     // Therefore, we have:
13627     //   t0 = (x0 * offset0 - x1 * y0 * y1 *offset1) + (y0 * y1) * t1
13628 
13629     ConstantSDNode *CN =
13630       cast<ConstantSDNode>(OtherUses[i]->getOperand(OffsetIdx));
13631     int X0, X1, Y0, Y1;
13632     const APInt &Offset0 = CN->getAPIntValue();
13633     APInt Offset1 = cast<ConstantSDNode>(Offset)->getAPIntValue();
13634 
13635     X0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 1) ? -1 : 1;
13636     Y0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 0) ? -1 : 1;
13637     X1 = (AM == ISD::PRE_DEC && !Swapped) ? -1 : 1;
13638     Y1 = (AM == ISD::PRE_DEC && Swapped) ? -1 : 1;
13639 
13640     unsigned Opcode = (Y0 * Y1 < 0) ? ISD::SUB : ISD::ADD;
13641 
13642     APInt CNV = Offset0;
13643     if (X0 < 0) CNV = -CNV;
13644     if (X1 * Y0 * Y1 < 0) CNV = CNV + Offset1;
13645     else CNV = CNV - Offset1;
13646 
13647     SDLoc DL(OtherUses[i]);
13648 
13649     // We can now generate the new expression.
13650     SDValue NewOp1 = DAG.getConstant(CNV, DL, CN->getValueType(0));
13651     SDValue NewOp2 = Result.getValue(isLoad ? 1 : 0);
13652 
13653     SDValue NewUse = DAG.getNode(Opcode,
13654                                  DL,
13655                                  OtherUses[i]->getValueType(0), NewOp1, NewOp2);
13656     DAG.ReplaceAllUsesOfValueWith(SDValue(OtherUses[i], 0), NewUse);
13657     deleteAndRecombine(OtherUses[i]);
13658   }
13659 
13660   // Replace the uses of Ptr with uses of the updated base value.
13661   DAG.ReplaceAllUsesOfValueWith(Ptr, Result.getValue(isLoad ? 1 : 0));
13662   deleteAndRecombine(Ptr.getNode());
13663   AddToWorklist(Result.getNode());
13664 
13665   return true;
13666 }
13667 
13668 /// Try to combine a load/store with a add/sub of the base pointer node into a
13669 /// post-indexed load/store. The transformation folded the add/subtract into the
13670 /// new indexed load/store effectively and all of its uses are redirected to the
13671 /// new load/store.
13672 bool DAGCombiner::CombineToPostIndexedLoadStore(SDNode *N) {
13673   if (Level < AfterLegalizeDAG)
13674     return false;
13675 
13676   bool isLoad = true;
13677   SDValue Ptr;
13678   EVT VT;
13679   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
13680     if (LD->isIndexed())
13681       return false;
13682     VT = LD->getMemoryVT();
13683     if (!TLI.isIndexedLoadLegal(ISD::POST_INC, VT) &&
13684         !TLI.isIndexedLoadLegal(ISD::POST_DEC, VT))
13685       return false;
13686     Ptr = LD->getBasePtr();
13687   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
13688     if (ST->isIndexed())
13689       return false;
13690     VT = ST->getMemoryVT();
13691     if (!TLI.isIndexedStoreLegal(ISD::POST_INC, VT) &&
13692         !TLI.isIndexedStoreLegal(ISD::POST_DEC, VT))
13693       return false;
13694     Ptr = ST->getBasePtr();
13695     isLoad = false;
13696   } else {
13697     return false;
13698   }
13699 
13700   if (Ptr.getNode()->hasOneUse())
13701     return false;
13702 
13703   for (SDNode *Op : Ptr.getNode()->uses()) {
13704     if (Op == N ||
13705         (Op->getOpcode() != ISD::ADD && Op->getOpcode() != ISD::SUB))
13706       continue;
13707 
13708     SDValue BasePtr;
13709     SDValue Offset;
13710     ISD::MemIndexedMode AM = ISD::UNINDEXED;
13711     if (TLI.getPostIndexedAddressParts(N, Op, BasePtr, Offset, AM, DAG)) {
13712       // Don't create a indexed load / store with zero offset.
13713       if (isNullConstant(Offset))
13714         continue;
13715 
13716       // Try turning it into a post-indexed load / store except when
13717       // 1) All uses are load / store ops that use it as base ptr (and
13718       //    it may be folded as addressing mmode).
13719       // 2) Op must be independent of N, i.e. Op is neither a predecessor
13720       //    nor a successor of N. Otherwise, if Op is folded that would
13721       //    create a cycle.
13722 
13723       if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
13724         continue;
13725 
13726       // Check for #1.
13727       bool TryNext = false;
13728       for (SDNode *Use : BasePtr.getNode()->uses()) {
13729         if (Use == Ptr.getNode())
13730           continue;
13731 
13732         // If all the uses are load / store addresses, then don't do the
13733         // transformation.
13734         if (Use->getOpcode() == ISD::ADD || Use->getOpcode() == ISD::SUB){
13735           bool RealUse = false;
13736           for (SDNode *UseUse : Use->uses()) {
13737             if (!canFoldInAddressingMode(Use, UseUse, DAG, TLI))
13738               RealUse = true;
13739           }
13740 
13741           if (!RealUse) {
13742             TryNext = true;
13743             break;
13744           }
13745         }
13746       }
13747 
13748       if (TryNext)
13749         continue;
13750 
13751       // Check for #2.
13752       SmallPtrSet<const SDNode *, 32> Visited;
13753       SmallVector<const SDNode *, 8> Worklist;
13754       // Ptr is predecessor to both N and Op.
13755       Visited.insert(Ptr.getNode());
13756       Worklist.push_back(N);
13757       Worklist.push_back(Op);
13758       if (!SDNode::hasPredecessorHelper(N, Visited, Worklist) &&
13759           !SDNode::hasPredecessorHelper(Op, Visited, Worklist)) {
13760         SDValue Result = isLoad
13761           ? DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
13762                                BasePtr, Offset, AM)
13763           : DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
13764                                 BasePtr, Offset, AM);
13765         ++PostIndexedNodes;
13766         ++NodesCombined;
13767         LLVM_DEBUG(dbgs() << "\nReplacing.5 "; N->dump(&DAG);
13768                    dbgs() << "\nWith: "; Result.getNode()->dump(&DAG);
13769                    dbgs() << '\n');
13770         WorklistRemover DeadNodes(*this);
13771         if (isLoad) {
13772           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
13773           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
13774         } else {
13775           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
13776         }
13777 
13778         // Finally, since the node is now dead, remove it from the graph.
13779         deleteAndRecombine(N);
13780 
13781         // Replace the uses of Use with uses of the updated base value.
13782         DAG.ReplaceAllUsesOfValueWith(SDValue(Op, 0),
13783                                       Result.getValue(isLoad ? 1 : 0));
13784         deleteAndRecombine(Op);
13785         return true;
13786       }
13787     }
13788   }
13789 
13790   return false;
13791 }
13792 
13793 /// Return the base-pointer arithmetic from an indexed \p LD.
13794 SDValue DAGCombiner::SplitIndexingFromLoad(LoadSDNode *LD) {
13795   ISD::MemIndexedMode AM = LD->getAddressingMode();
13796   assert(AM != ISD::UNINDEXED);
13797   SDValue BP = LD->getOperand(1);
13798   SDValue Inc = LD->getOperand(2);
13799 
13800   // Some backends use TargetConstants for load offsets, but don't expect
13801   // TargetConstants in general ADD nodes. We can convert these constants into
13802   // regular Constants (if the constant is not opaque).
13803   assert((Inc.getOpcode() != ISD::TargetConstant ||
13804           !cast<ConstantSDNode>(Inc)->isOpaque()) &&
13805          "Cannot split out indexing using opaque target constants");
13806   if (Inc.getOpcode() == ISD::TargetConstant) {
13807     ConstantSDNode *ConstInc = cast<ConstantSDNode>(Inc);
13808     Inc = DAG.getConstant(*ConstInc->getConstantIntValue(), SDLoc(Inc),
13809                           ConstInc->getValueType(0));
13810   }
13811 
13812   unsigned Opc =
13813       (AM == ISD::PRE_INC || AM == ISD::POST_INC ? ISD::ADD : ISD::SUB);
13814   return DAG.getNode(Opc, SDLoc(LD), BP.getSimpleValueType(), BP, Inc);
13815 }
13816 
13817 static inline int numVectorEltsOrZero(EVT T) {
13818   return T.isVector() ? T.getVectorNumElements() : 0;
13819 }
13820 
13821 bool DAGCombiner::getTruncatedStoreValue(StoreSDNode *ST, SDValue &Val) {
13822   Val = ST->getValue();
13823   EVT STType = Val.getValueType();
13824   EVT STMemType = ST->getMemoryVT();
13825   if (STType == STMemType)
13826     return true;
13827   if (isTypeLegal(STMemType))
13828     return false; // fail.
13829   if (STType.isFloatingPoint() && STMemType.isFloatingPoint() &&
13830       TLI.isOperationLegal(ISD::FTRUNC, STMemType)) {
13831     Val = DAG.getNode(ISD::FTRUNC, SDLoc(ST), STMemType, Val);
13832     return true;
13833   }
13834   if (numVectorEltsOrZero(STType) == numVectorEltsOrZero(STMemType) &&
13835       STType.isInteger() && STMemType.isInteger()) {
13836     Val = DAG.getNode(ISD::TRUNCATE, SDLoc(ST), STMemType, Val);
13837     return true;
13838   }
13839   if (STType.getSizeInBits() == STMemType.getSizeInBits()) {
13840     Val = DAG.getBitcast(STMemType, Val);
13841     return true;
13842   }
13843   return false; // fail.
13844 }
13845 
13846 bool DAGCombiner::extendLoadedValueToExtension(LoadSDNode *LD, SDValue &Val) {
13847   EVT LDMemType = LD->getMemoryVT();
13848   EVT LDType = LD->getValueType(0);
13849   assert(Val.getValueType() == LDMemType &&
13850          "Attempting to extend value of non-matching type");
13851   if (LDType == LDMemType)
13852     return true;
13853   if (LDMemType.isInteger() && LDType.isInteger()) {
13854     switch (LD->getExtensionType()) {
13855     case ISD::NON_EXTLOAD:
13856       Val = DAG.getBitcast(LDType, Val);
13857       return true;
13858     case ISD::EXTLOAD:
13859       Val = DAG.getNode(ISD::ANY_EXTEND, SDLoc(LD), LDType, Val);
13860       return true;
13861     case ISD::SEXTLOAD:
13862       Val = DAG.getNode(ISD::SIGN_EXTEND, SDLoc(LD), LDType, Val);
13863       return true;
13864     case ISD::ZEXTLOAD:
13865       Val = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(LD), LDType, Val);
13866       return true;
13867     }
13868   }
13869   return false;
13870 }
13871 
13872 SDValue DAGCombiner::ForwardStoreValueToDirectLoad(LoadSDNode *LD) {
13873   if (OptLevel == CodeGenOpt::None || LD->isVolatile())
13874     return SDValue();
13875   SDValue Chain = LD->getOperand(0);
13876   StoreSDNode *ST = dyn_cast<StoreSDNode>(Chain.getNode());
13877   if (!ST || ST->isVolatile())
13878     return SDValue();
13879 
13880   EVT LDType = LD->getValueType(0);
13881   EVT LDMemType = LD->getMemoryVT();
13882   EVT STMemType = ST->getMemoryVT();
13883   EVT STType = ST->getValue().getValueType();
13884 
13885   BaseIndexOffset BasePtrLD = BaseIndexOffset::match(LD, DAG);
13886   BaseIndexOffset BasePtrST = BaseIndexOffset::match(ST, DAG);
13887   int64_t Offset;
13888   if (!BasePtrST.equalBaseIndex(BasePtrLD, DAG, Offset))
13889     return SDValue();
13890 
13891   // Normalize for Endianness. After this Offset=0 will denote that the least
13892   // significant bit in the loaded value maps to the least significant bit in
13893   // the stored value). With Offset=n (for n > 0) the loaded value starts at the
13894   // n:th least significant byte of the stored value.
13895   if (DAG.getDataLayout().isBigEndian())
13896     Offset = (STMemType.getStoreSizeInBits() -
13897               LDMemType.getStoreSizeInBits()) / 8 - Offset;
13898 
13899   // Check that the stored value cover all bits that are loaded.
13900   bool STCoversLD =
13901       (Offset >= 0) &&
13902       (Offset * 8 + LDMemType.getSizeInBits() <= STMemType.getSizeInBits());
13903 
13904   auto ReplaceLd = [&](LoadSDNode *LD, SDValue Val, SDValue Chain) -> SDValue {
13905     if (LD->isIndexed()) {
13906       bool IsSub = (LD->getAddressingMode() == ISD::PRE_DEC ||
13907                     LD->getAddressingMode() == ISD::POST_DEC);
13908       unsigned Opc = IsSub ? ISD::SUB : ISD::ADD;
13909       SDValue Idx = DAG.getNode(Opc, SDLoc(LD), LD->getOperand(1).getValueType(),
13910                              LD->getOperand(1), LD->getOperand(2));
13911       SDValue Ops[] = {Val, Idx, Chain};
13912       return CombineTo(LD, Ops, 3);
13913     }
13914     return CombineTo(LD, Val, Chain);
13915   };
13916 
13917   if (!STCoversLD)
13918     return SDValue();
13919 
13920   // Memory as copy space (potentially masked).
13921   if (Offset == 0 && LDType == STType && STMemType == LDMemType) {
13922     // Simple case: Direct non-truncating forwarding
13923     if (LDType.getSizeInBits() == LDMemType.getSizeInBits())
13924       return ReplaceLd(LD, ST->getValue(), Chain);
13925     // Can we model the truncate and extension with an and mask?
13926     if (STType.isInteger() && LDMemType.isInteger() && !STType.isVector() &&
13927         !LDMemType.isVector() && LD->getExtensionType() != ISD::SEXTLOAD) {
13928       // Mask to size of LDMemType
13929       auto Mask =
13930           DAG.getConstant(APInt::getLowBitsSet(STType.getSizeInBits(),
13931                                                STMemType.getSizeInBits()),
13932                           SDLoc(ST), STType);
13933       auto Val = DAG.getNode(ISD::AND, SDLoc(LD), LDType, ST->getValue(), Mask);
13934       return ReplaceLd(LD, Val, Chain);
13935     }
13936   }
13937 
13938   // TODO: Deal with nonzero offset.
13939   if (LD->getBasePtr().isUndef() || Offset != 0)
13940     return SDValue();
13941   // Model necessary truncations / extenstions.
13942   SDValue Val;
13943   // Truncate Value To Stored Memory Size.
13944   do {
13945     if (!getTruncatedStoreValue(ST, Val))
13946       continue;
13947     if (!isTypeLegal(LDMemType))
13948       continue;
13949     if (STMemType != LDMemType) {
13950       // TODO: Support vectors? This requires extract_subvector/bitcast.
13951       if (!STMemType.isVector() && !LDMemType.isVector() &&
13952           STMemType.isInteger() && LDMemType.isInteger())
13953         Val = DAG.getNode(ISD::TRUNCATE, SDLoc(LD), LDMemType, Val);
13954       else
13955         continue;
13956     }
13957     if (!extendLoadedValueToExtension(LD, Val))
13958       continue;
13959     return ReplaceLd(LD, Val, Chain);
13960   } while (false);
13961 
13962   // On failure, cleanup dead nodes we may have created.
13963   if (Val->use_empty())
13964     deleteAndRecombine(Val.getNode());
13965   return SDValue();
13966 }
13967 
13968 SDValue DAGCombiner::visitLOAD(SDNode *N) {
13969   LoadSDNode *LD  = cast<LoadSDNode>(N);
13970   SDValue Chain = LD->getChain();
13971   SDValue Ptr   = LD->getBasePtr();
13972 
13973   // If load is not volatile and there are no uses of the loaded value (and
13974   // the updated indexed value in case of indexed loads), change uses of the
13975   // chain value into uses of the chain input (i.e. delete the dead load).
13976   if (!LD->isVolatile()) {
13977     if (N->getValueType(1) == MVT::Other) {
13978       // Unindexed loads.
13979       if (!N->hasAnyUseOfValue(0)) {
13980         // It's not safe to use the two value CombineTo variant here. e.g.
13981         // v1, chain2 = load chain1, loc
13982         // v2, chain3 = load chain2, loc
13983         // v3         = add v2, c
13984         // Now we replace use of chain2 with chain1.  This makes the second load
13985         // isomorphic to the one we are deleting, and thus makes this load live.
13986         LLVM_DEBUG(dbgs() << "\nReplacing.6 "; N->dump(&DAG);
13987                    dbgs() << "\nWith chain: "; Chain.getNode()->dump(&DAG);
13988                    dbgs() << "\n");
13989         WorklistRemover DeadNodes(*this);
13990         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
13991         AddUsersToWorklist(Chain.getNode());
13992         if (N->use_empty())
13993           deleteAndRecombine(N);
13994 
13995         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
13996       }
13997     } else {
13998       // Indexed loads.
13999       assert(N->getValueType(2) == MVT::Other && "Malformed indexed loads?");
14000 
14001       // If this load has an opaque TargetConstant offset, then we cannot split
14002       // the indexing into an add/sub directly (that TargetConstant may not be
14003       // valid for a different type of node, and we cannot convert an opaque
14004       // target constant into a regular constant).
14005       bool HasOTCInc = LD->getOperand(2).getOpcode() == ISD::TargetConstant &&
14006                        cast<ConstantSDNode>(LD->getOperand(2))->isOpaque();
14007 
14008       if (!N->hasAnyUseOfValue(0) &&
14009           ((MaySplitLoadIndex && !HasOTCInc) || !N->hasAnyUseOfValue(1))) {
14010         SDValue Undef = DAG.getUNDEF(N->getValueType(0));
14011         SDValue Index;
14012         if (N->hasAnyUseOfValue(1) && MaySplitLoadIndex && !HasOTCInc) {
14013           Index = SplitIndexingFromLoad(LD);
14014           // Try to fold the base pointer arithmetic into subsequent loads and
14015           // stores.
14016           AddUsersToWorklist(N);
14017         } else
14018           Index = DAG.getUNDEF(N->getValueType(1));
14019         LLVM_DEBUG(dbgs() << "\nReplacing.7 "; N->dump(&DAG);
14020                    dbgs() << "\nWith: "; Undef.getNode()->dump(&DAG);
14021                    dbgs() << " and 2 other values\n");
14022         WorklistRemover DeadNodes(*this);
14023         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Undef);
14024         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Index);
14025         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 2), Chain);
14026         deleteAndRecombine(N);
14027         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
14028       }
14029     }
14030   }
14031 
14032   // If this load is directly stored, replace the load value with the stored
14033   // value.
14034   if (auto V = ForwardStoreValueToDirectLoad(LD))
14035     return V;
14036 
14037   // Try to infer better alignment information than the load already has.
14038   if (OptLevel != CodeGenOpt::None && LD->isUnindexed()) {
14039     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
14040       if (Align > LD->getAlignment() && LD->getSrcValueOffset() % Align == 0) {
14041         SDValue NewLoad = DAG.getExtLoad(
14042             LD->getExtensionType(), SDLoc(N), LD->getValueType(0), Chain, Ptr,
14043             LD->getPointerInfo(), LD->getMemoryVT(), Align,
14044             LD->getMemOperand()->getFlags(), LD->getAAInfo());
14045         // NewLoad will always be N as we are only refining the alignment
14046         assert(NewLoad.getNode() == N);
14047         (void)NewLoad;
14048       }
14049     }
14050   }
14051 
14052   if (LD->isUnindexed()) {
14053     // Walk up chain skipping non-aliasing memory nodes.
14054     SDValue BetterChain = FindBetterChain(LD, Chain);
14055 
14056     // If there is a better chain.
14057     if (Chain != BetterChain) {
14058       SDValue ReplLoad;
14059 
14060       // Replace the chain to void dependency.
14061       if (LD->getExtensionType() == ISD::NON_EXTLOAD) {
14062         ReplLoad = DAG.getLoad(N->getValueType(0), SDLoc(LD),
14063                                BetterChain, Ptr, LD->getMemOperand());
14064       } else {
14065         ReplLoad = DAG.getExtLoad(LD->getExtensionType(), SDLoc(LD),
14066                                   LD->getValueType(0),
14067                                   BetterChain, Ptr, LD->getMemoryVT(),
14068                                   LD->getMemOperand());
14069       }
14070 
14071       // Create token factor to keep old chain connected.
14072       SDValue Token = DAG.getNode(ISD::TokenFactor, SDLoc(N),
14073                                   MVT::Other, Chain, ReplLoad.getValue(1));
14074 
14075       // Replace uses with load result and token factor
14076       return CombineTo(N, ReplLoad.getValue(0), Token);
14077     }
14078   }
14079 
14080   // Try transforming N to an indexed load.
14081   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
14082     return SDValue(N, 0);
14083 
14084   // Try to slice up N to more direct loads if the slices are mapped to
14085   // different register banks or pairing can take place.
14086   if (SliceUpLoad(N))
14087     return SDValue(N, 0);
14088 
14089   return SDValue();
14090 }
14091 
14092 namespace {
14093 
14094 /// Helper structure used to slice a load in smaller loads.
14095 /// Basically a slice is obtained from the following sequence:
14096 /// Origin = load Ty1, Base
14097 /// Shift = srl Ty1 Origin, CstTy Amount
14098 /// Inst = trunc Shift to Ty2
14099 ///
14100 /// Then, it will be rewritten into:
14101 /// Slice = load SliceTy, Base + SliceOffset
14102 /// [Inst = zext Slice to Ty2], only if SliceTy <> Ty2
14103 ///
14104 /// SliceTy is deduced from the number of bits that are actually used to
14105 /// build Inst.
14106 struct LoadedSlice {
14107   /// Helper structure used to compute the cost of a slice.
14108   struct Cost {
14109     /// Are we optimizing for code size.
14110     bool ForCodeSize;
14111 
14112     /// Various cost.
14113     unsigned Loads = 0;
14114     unsigned Truncates = 0;
14115     unsigned CrossRegisterBanksCopies = 0;
14116     unsigned ZExts = 0;
14117     unsigned Shift = 0;
14118 
14119     Cost(bool ForCodeSize = false) : ForCodeSize(ForCodeSize) {}
14120 
14121     /// Get the cost of one isolated slice.
14122     Cost(const LoadedSlice &LS, bool ForCodeSize = false)
14123         : ForCodeSize(ForCodeSize), Loads(1) {
14124       EVT TruncType = LS.Inst->getValueType(0);
14125       EVT LoadedType = LS.getLoadedType();
14126       if (TruncType != LoadedType &&
14127           !LS.DAG->getTargetLoweringInfo().isZExtFree(LoadedType, TruncType))
14128         ZExts = 1;
14129     }
14130 
14131     /// Account for slicing gain in the current cost.
14132     /// Slicing provide a few gains like removing a shift or a
14133     /// truncate. This method allows to grow the cost of the original
14134     /// load with the gain from this slice.
14135     void addSliceGain(const LoadedSlice &LS) {
14136       // Each slice saves a truncate.
14137       const TargetLowering &TLI = LS.DAG->getTargetLoweringInfo();
14138       if (!TLI.isTruncateFree(LS.Inst->getOperand(0).getValueType(),
14139                               LS.Inst->getValueType(0)))
14140         ++Truncates;
14141       // If there is a shift amount, this slice gets rid of it.
14142       if (LS.Shift)
14143         ++Shift;
14144       // If this slice can merge a cross register bank copy, account for it.
14145       if (LS.canMergeExpensiveCrossRegisterBankCopy())
14146         ++CrossRegisterBanksCopies;
14147     }
14148 
14149     Cost &operator+=(const Cost &RHS) {
14150       Loads += RHS.Loads;
14151       Truncates += RHS.Truncates;
14152       CrossRegisterBanksCopies += RHS.CrossRegisterBanksCopies;
14153       ZExts += RHS.ZExts;
14154       Shift += RHS.Shift;
14155       return *this;
14156     }
14157 
14158     bool operator==(const Cost &RHS) const {
14159       return Loads == RHS.Loads && Truncates == RHS.Truncates &&
14160              CrossRegisterBanksCopies == RHS.CrossRegisterBanksCopies &&
14161              ZExts == RHS.ZExts && Shift == RHS.Shift;
14162     }
14163 
14164     bool operator!=(const Cost &RHS) const { return !(*this == RHS); }
14165 
14166     bool operator<(const Cost &RHS) const {
14167       // Assume cross register banks copies are as expensive as loads.
14168       // FIXME: Do we want some more target hooks?
14169       unsigned ExpensiveOpsLHS = Loads + CrossRegisterBanksCopies;
14170       unsigned ExpensiveOpsRHS = RHS.Loads + RHS.CrossRegisterBanksCopies;
14171       // Unless we are optimizing for code size, consider the
14172       // expensive operation first.
14173       if (!ForCodeSize && ExpensiveOpsLHS != ExpensiveOpsRHS)
14174         return ExpensiveOpsLHS < ExpensiveOpsRHS;
14175       return (Truncates + ZExts + Shift + ExpensiveOpsLHS) <
14176              (RHS.Truncates + RHS.ZExts + RHS.Shift + ExpensiveOpsRHS);
14177     }
14178 
14179     bool operator>(const Cost &RHS) const { return RHS < *this; }
14180 
14181     bool operator<=(const Cost &RHS) const { return !(RHS < *this); }
14182 
14183     bool operator>=(const Cost &RHS) const { return !(*this < RHS); }
14184   };
14185 
14186   // The last instruction that represent the slice. This should be a
14187   // truncate instruction.
14188   SDNode *Inst;
14189 
14190   // The original load instruction.
14191   LoadSDNode *Origin;
14192 
14193   // The right shift amount in bits from the original load.
14194   unsigned Shift;
14195 
14196   // The DAG from which Origin came from.
14197   // This is used to get some contextual information about legal types, etc.
14198   SelectionDAG *DAG;
14199 
14200   LoadedSlice(SDNode *Inst = nullptr, LoadSDNode *Origin = nullptr,
14201               unsigned Shift = 0, SelectionDAG *DAG = nullptr)
14202       : Inst(Inst), Origin(Origin), Shift(Shift), DAG(DAG) {}
14203 
14204   /// Get the bits used in a chunk of bits \p BitWidth large.
14205   /// \return Result is \p BitWidth and has used bits set to 1 and
14206   ///         not used bits set to 0.
14207   APInt getUsedBits() const {
14208     // Reproduce the trunc(lshr) sequence:
14209     // - Start from the truncated value.
14210     // - Zero extend to the desired bit width.
14211     // - Shift left.
14212     assert(Origin && "No original load to compare against.");
14213     unsigned BitWidth = Origin->getValueSizeInBits(0);
14214     assert(Inst && "This slice is not bound to an instruction");
14215     assert(Inst->getValueSizeInBits(0) <= BitWidth &&
14216            "Extracted slice is bigger than the whole type!");
14217     APInt UsedBits(Inst->getValueSizeInBits(0), 0);
14218     UsedBits.setAllBits();
14219     UsedBits = UsedBits.zext(BitWidth);
14220     UsedBits <<= Shift;
14221     return UsedBits;
14222   }
14223 
14224   /// Get the size of the slice to be loaded in bytes.
14225   unsigned getLoadedSize() const {
14226     unsigned SliceSize = getUsedBits().countPopulation();
14227     assert(!(SliceSize & 0x7) && "Size is not a multiple of a byte.");
14228     return SliceSize / 8;
14229   }
14230 
14231   /// Get the type that will be loaded for this slice.
14232   /// Note: This may not be the final type for the slice.
14233   EVT getLoadedType() const {
14234     assert(DAG && "Missing context");
14235     LLVMContext &Ctxt = *DAG->getContext();
14236     return EVT::getIntegerVT(Ctxt, getLoadedSize() * 8);
14237   }
14238 
14239   /// Get the alignment of the load used for this slice.
14240   unsigned getAlignment() const {
14241     unsigned Alignment = Origin->getAlignment();
14242     unsigned Offset = getOffsetFromBase();
14243     if (Offset != 0)
14244       Alignment = MinAlign(Alignment, Alignment + Offset);
14245     return Alignment;
14246   }
14247 
14248   /// Check if this slice can be rewritten with legal operations.
14249   bool isLegal() const {
14250     // An invalid slice is not legal.
14251     if (!Origin || !Inst || !DAG)
14252       return false;
14253 
14254     // Offsets are for indexed load only, we do not handle that.
14255     if (!Origin->getOffset().isUndef())
14256       return false;
14257 
14258     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
14259 
14260     // Check that the type is legal.
14261     EVT SliceType = getLoadedType();
14262     if (!TLI.isTypeLegal(SliceType))
14263       return false;
14264 
14265     // Check that the load is legal for this type.
14266     if (!TLI.isOperationLegal(ISD::LOAD, SliceType))
14267       return false;
14268 
14269     // Check that the offset can be computed.
14270     // 1. Check its type.
14271     EVT PtrType = Origin->getBasePtr().getValueType();
14272     if (PtrType == MVT::Untyped || PtrType.isExtended())
14273       return false;
14274 
14275     // 2. Check that it fits in the immediate.
14276     if (!TLI.isLegalAddImmediate(getOffsetFromBase()))
14277       return false;
14278 
14279     // 3. Check that the computation is legal.
14280     if (!TLI.isOperationLegal(ISD::ADD, PtrType))
14281       return false;
14282 
14283     // Check that the zext is legal if it needs one.
14284     EVT TruncateType = Inst->getValueType(0);
14285     if (TruncateType != SliceType &&
14286         !TLI.isOperationLegal(ISD::ZERO_EXTEND, TruncateType))
14287       return false;
14288 
14289     return true;
14290   }
14291 
14292   /// Get the offset in bytes of this slice in the original chunk of
14293   /// bits.
14294   /// \pre DAG != nullptr.
14295   uint64_t getOffsetFromBase() const {
14296     assert(DAG && "Missing context.");
14297     bool IsBigEndian = DAG->getDataLayout().isBigEndian();
14298     assert(!(Shift & 0x7) && "Shifts not aligned on Bytes are not supported.");
14299     uint64_t Offset = Shift / 8;
14300     unsigned TySizeInBytes = Origin->getValueSizeInBits(0) / 8;
14301     assert(!(Origin->getValueSizeInBits(0) & 0x7) &&
14302            "The size of the original loaded type is not a multiple of a"
14303            " byte.");
14304     // If Offset is bigger than TySizeInBytes, it means we are loading all
14305     // zeros. This should have been optimized before in the process.
14306     assert(TySizeInBytes > Offset &&
14307            "Invalid shift amount for given loaded size");
14308     if (IsBigEndian)
14309       Offset = TySizeInBytes - Offset - getLoadedSize();
14310     return Offset;
14311   }
14312 
14313   /// Generate the sequence of instructions to load the slice
14314   /// represented by this object and redirect the uses of this slice to
14315   /// this new sequence of instructions.
14316   /// \pre this->Inst && this->Origin are valid Instructions and this
14317   /// object passed the legal check: LoadedSlice::isLegal returned true.
14318   /// \return The last instruction of the sequence used to load the slice.
14319   SDValue loadSlice() const {
14320     assert(Inst && Origin && "Unable to replace a non-existing slice.");
14321     const SDValue &OldBaseAddr = Origin->getBasePtr();
14322     SDValue BaseAddr = OldBaseAddr;
14323     // Get the offset in that chunk of bytes w.r.t. the endianness.
14324     int64_t Offset = static_cast<int64_t>(getOffsetFromBase());
14325     assert(Offset >= 0 && "Offset too big to fit in int64_t!");
14326     if (Offset) {
14327       // BaseAddr = BaseAddr + Offset.
14328       EVT ArithType = BaseAddr.getValueType();
14329       SDLoc DL(Origin);
14330       BaseAddr = DAG->getNode(ISD::ADD, DL, ArithType, BaseAddr,
14331                               DAG->getConstant(Offset, DL, ArithType));
14332     }
14333 
14334     // Create the type of the loaded slice according to its size.
14335     EVT SliceType = getLoadedType();
14336 
14337     // Create the load for the slice.
14338     SDValue LastInst =
14339         DAG->getLoad(SliceType, SDLoc(Origin), Origin->getChain(), BaseAddr,
14340                      Origin->getPointerInfo().getWithOffset(Offset),
14341                      getAlignment(), Origin->getMemOperand()->getFlags());
14342     // If the final type is not the same as the loaded type, this means that
14343     // we have to pad with zero. Create a zero extend for that.
14344     EVT FinalType = Inst->getValueType(0);
14345     if (SliceType != FinalType)
14346       LastInst =
14347           DAG->getNode(ISD::ZERO_EXTEND, SDLoc(LastInst), FinalType, LastInst);
14348     return LastInst;
14349   }
14350 
14351   /// Check if this slice can be merged with an expensive cross register
14352   /// bank copy. E.g.,
14353   /// i = load i32
14354   /// f = bitcast i32 i to float
14355   bool canMergeExpensiveCrossRegisterBankCopy() const {
14356     if (!Inst || !Inst->hasOneUse())
14357       return false;
14358     SDNode *Use = *Inst->use_begin();
14359     if (Use->getOpcode() != ISD::BITCAST)
14360       return false;
14361     assert(DAG && "Missing context");
14362     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
14363     EVT ResVT = Use->getValueType(0);
14364     const TargetRegisterClass *ResRC =
14365         TLI.getRegClassFor(ResVT.getSimpleVT(), Use->isDivergent());
14366     const TargetRegisterClass *ArgRC =
14367         TLI.getRegClassFor(Use->getOperand(0).getValueType().getSimpleVT(),
14368                            Use->getOperand(0)->isDivergent());
14369     if (ArgRC == ResRC || !TLI.isOperationLegal(ISD::LOAD, ResVT))
14370       return false;
14371 
14372     // At this point, we know that we perform a cross-register-bank copy.
14373     // Check if it is expensive.
14374     const TargetRegisterInfo *TRI = DAG->getSubtarget().getRegisterInfo();
14375     // Assume bitcasts are cheap, unless both register classes do not
14376     // explicitly share a common sub class.
14377     if (!TRI || TRI->getCommonSubClass(ArgRC, ResRC))
14378       return false;
14379 
14380     // Check if it will be merged with the load.
14381     // 1. Check the alignment constraint.
14382     unsigned RequiredAlignment = DAG->getDataLayout().getABITypeAlignment(
14383         ResVT.getTypeForEVT(*DAG->getContext()));
14384 
14385     if (RequiredAlignment > getAlignment())
14386       return false;
14387 
14388     // 2. Check that the load is a legal operation for that type.
14389     if (!TLI.isOperationLegal(ISD::LOAD, ResVT))
14390       return false;
14391 
14392     // 3. Check that we do not have a zext in the way.
14393     if (Inst->getValueType(0) != getLoadedType())
14394       return false;
14395 
14396     return true;
14397   }
14398 };
14399 
14400 } // end anonymous namespace
14401 
14402 /// Check that all bits set in \p UsedBits form a dense region, i.e.,
14403 /// \p UsedBits looks like 0..0 1..1 0..0.
14404 static bool areUsedBitsDense(const APInt &UsedBits) {
14405   // If all the bits are one, this is dense!
14406   if (UsedBits.isAllOnesValue())
14407     return true;
14408 
14409   // Get rid of the unused bits on the right.
14410   APInt NarrowedUsedBits = UsedBits.lshr(UsedBits.countTrailingZeros());
14411   // Get rid of the unused bits on the left.
14412   if (NarrowedUsedBits.countLeadingZeros())
14413     NarrowedUsedBits = NarrowedUsedBits.trunc(NarrowedUsedBits.getActiveBits());
14414   // Check that the chunk of bits is completely used.
14415   return NarrowedUsedBits.isAllOnesValue();
14416 }
14417 
14418 /// Check whether or not \p First and \p Second are next to each other
14419 /// in memory. This means that there is no hole between the bits loaded
14420 /// by \p First and the bits loaded by \p Second.
14421 static bool areSlicesNextToEachOther(const LoadedSlice &First,
14422                                      const LoadedSlice &Second) {
14423   assert(First.Origin == Second.Origin && First.Origin &&
14424          "Unable to match different memory origins.");
14425   APInt UsedBits = First.getUsedBits();
14426   assert((UsedBits & Second.getUsedBits()) == 0 &&
14427          "Slices are not supposed to overlap.");
14428   UsedBits |= Second.getUsedBits();
14429   return areUsedBitsDense(UsedBits);
14430 }
14431 
14432 /// Adjust the \p GlobalLSCost according to the target
14433 /// paring capabilities and the layout of the slices.
14434 /// \pre \p GlobalLSCost should account for at least as many loads as
14435 /// there is in the slices in \p LoadedSlices.
14436 static void adjustCostForPairing(SmallVectorImpl<LoadedSlice> &LoadedSlices,
14437                                  LoadedSlice::Cost &GlobalLSCost) {
14438   unsigned NumberOfSlices = LoadedSlices.size();
14439   // If there is less than 2 elements, no pairing is possible.
14440   if (NumberOfSlices < 2)
14441     return;
14442 
14443   // Sort the slices so that elements that are likely to be next to each
14444   // other in memory are next to each other in the list.
14445   llvm::sort(LoadedSlices, [](const LoadedSlice &LHS, const LoadedSlice &RHS) {
14446     assert(LHS.Origin == RHS.Origin && "Different bases not implemented.");
14447     return LHS.getOffsetFromBase() < RHS.getOffsetFromBase();
14448   });
14449   const TargetLowering &TLI = LoadedSlices[0].DAG->getTargetLoweringInfo();
14450   // First (resp. Second) is the first (resp. Second) potentially candidate
14451   // to be placed in a paired load.
14452   const LoadedSlice *First = nullptr;
14453   const LoadedSlice *Second = nullptr;
14454   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice,
14455                 // Set the beginning of the pair.
14456                                                            First = Second) {
14457     Second = &LoadedSlices[CurrSlice];
14458 
14459     // If First is NULL, it means we start a new pair.
14460     // Get to the next slice.
14461     if (!First)
14462       continue;
14463 
14464     EVT LoadedType = First->getLoadedType();
14465 
14466     // If the types of the slices are different, we cannot pair them.
14467     if (LoadedType != Second->getLoadedType())
14468       continue;
14469 
14470     // Check if the target supplies paired loads for this type.
14471     unsigned RequiredAlignment = 0;
14472     if (!TLI.hasPairedLoad(LoadedType, RequiredAlignment)) {
14473       // move to the next pair, this type is hopeless.
14474       Second = nullptr;
14475       continue;
14476     }
14477     // Check if we meet the alignment requirement.
14478     if (RequiredAlignment > First->getAlignment())
14479       continue;
14480 
14481     // Check that both loads are next to each other in memory.
14482     if (!areSlicesNextToEachOther(*First, *Second))
14483       continue;
14484 
14485     assert(GlobalLSCost.Loads > 0 && "We save more loads than we created!");
14486     --GlobalLSCost.Loads;
14487     // Move to the next pair.
14488     Second = nullptr;
14489   }
14490 }
14491 
14492 /// Check the profitability of all involved LoadedSlice.
14493 /// Currently, it is considered profitable if there is exactly two
14494 /// involved slices (1) which are (2) next to each other in memory, and
14495 /// whose cost (\see LoadedSlice::Cost) is smaller than the original load (3).
14496 ///
14497 /// Note: The order of the elements in \p LoadedSlices may be modified, but not
14498 /// the elements themselves.
14499 ///
14500 /// FIXME: When the cost model will be mature enough, we can relax
14501 /// constraints (1) and (2).
14502 static bool isSlicingProfitable(SmallVectorImpl<LoadedSlice> &LoadedSlices,
14503                                 const APInt &UsedBits, bool ForCodeSize) {
14504   unsigned NumberOfSlices = LoadedSlices.size();
14505   if (StressLoadSlicing)
14506     return NumberOfSlices > 1;
14507 
14508   // Check (1).
14509   if (NumberOfSlices != 2)
14510     return false;
14511 
14512   // Check (2).
14513   if (!areUsedBitsDense(UsedBits))
14514     return false;
14515 
14516   // Check (3).
14517   LoadedSlice::Cost OrigCost(ForCodeSize), GlobalSlicingCost(ForCodeSize);
14518   // The original code has one big load.
14519   OrigCost.Loads = 1;
14520   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice) {
14521     const LoadedSlice &LS = LoadedSlices[CurrSlice];
14522     // Accumulate the cost of all the slices.
14523     LoadedSlice::Cost SliceCost(LS, ForCodeSize);
14524     GlobalSlicingCost += SliceCost;
14525 
14526     // Account as cost in the original configuration the gain obtained
14527     // with the current slices.
14528     OrigCost.addSliceGain(LS);
14529   }
14530 
14531   // If the target supports paired load, adjust the cost accordingly.
14532   adjustCostForPairing(LoadedSlices, GlobalSlicingCost);
14533   return OrigCost > GlobalSlicingCost;
14534 }
14535 
14536 /// If the given load, \p LI, is used only by trunc or trunc(lshr)
14537 /// operations, split it in the various pieces being extracted.
14538 ///
14539 /// This sort of thing is introduced by SROA.
14540 /// This slicing takes care not to insert overlapping loads.
14541 /// \pre LI is a simple load (i.e., not an atomic or volatile load).
14542 bool DAGCombiner::SliceUpLoad(SDNode *N) {
14543   if (Level < AfterLegalizeDAG)
14544     return false;
14545 
14546   LoadSDNode *LD = cast<LoadSDNode>(N);
14547   if (LD->isVolatile() || !ISD::isNormalLoad(LD) ||
14548       !LD->getValueType(0).isInteger())
14549     return false;
14550 
14551   // Keep track of already used bits to detect overlapping values.
14552   // In that case, we will just abort the transformation.
14553   APInt UsedBits(LD->getValueSizeInBits(0), 0);
14554 
14555   SmallVector<LoadedSlice, 4> LoadedSlices;
14556 
14557   // Check if this load is used as several smaller chunks of bits.
14558   // Basically, look for uses in trunc or trunc(lshr) and record a new chain
14559   // of computation for each trunc.
14560   for (SDNode::use_iterator UI = LD->use_begin(), UIEnd = LD->use_end();
14561        UI != UIEnd; ++UI) {
14562     // Skip the uses of the chain.
14563     if (UI.getUse().getResNo() != 0)
14564       continue;
14565 
14566     SDNode *User = *UI;
14567     unsigned Shift = 0;
14568 
14569     // Check if this is a trunc(lshr).
14570     if (User->getOpcode() == ISD::SRL && User->hasOneUse() &&
14571         isa<ConstantSDNode>(User->getOperand(1))) {
14572       Shift = User->getConstantOperandVal(1);
14573       User = *User->use_begin();
14574     }
14575 
14576     // At this point, User is a Truncate, iff we encountered, trunc or
14577     // trunc(lshr).
14578     if (User->getOpcode() != ISD::TRUNCATE)
14579       return false;
14580 
14581     // The width of the type must be a power of 2 and greater than 8-bits.
14582     // Otherwise the load cannot be represented in LLVM IR.
14583     // Moreover, if we shifted with a non-8-bits multiple, the slice
14584     // will be across several bytes. We do not support that.
14585     unsigned Width = User->getValueSizeInBits(0);
14586     if (Width < 8 || !isPowerOf2_32(Width) || (Shift & 0x7))
14587       return false;
14588 
14589     // Build the slice for this chain of computations.
14590     LoadedSlice LS(User, LD, Shift, &DAG);
14591     APInt CurrentUsedBits = LS.getUsedBits();
14592 
14593     // Check if this slice overlaps with another.
14594     if ((CurrentUsedBits & UsedBits) != 0)
14595       return false;
14596     // Update the bits used globally.
14597     UsedBits |= CurrentUsedBits;
14598 
14599     // Check if the new slice would be legal.
14600     if (!LS.isLegal())
14601       return false;
14602 
14603     // Record the slice.
14604     LoadedSlices.push_back(LS);
14605   }
14606 
14607   // Abort slicing if it does not seem to be profitable.
14608   if (!isSlicingProfitable(LoadedSlices, UsedBits, ForCodeSize))
14609     return false;
14610 
14611   ++SlicedLoads;
14612 
14613   // Rewrite each chain to use an independent load.
14614   // By construction, each chain can be represented by a unique load.
14615 
14616   // Prepare the argument for the new token factor for all the slices.
14617   SmallVector<SDValue, 8> ArgChains;
14618   for (SmallVectorImpl<LoadedSlice>::const_iterator
14619            LSIt = LoadedSlices.begin(),
14620            LSItEnd = LoadedSlices.end();
14621        LSIt != LSItEnd; ++LSIt) {
14622     SDValue SliceInst = LSIt->loadSlice();
14623     CombineTo(LSIt->Inst, SliceInst, true);
14624     if (SliceInst.getOpcode() != ISD::LOAD)
14625       SliceInst = SliceInst.getOperand(0);
14626     assert(SliceInst->getOpcode() == ISD::LOAD &&
14627            "It takes more than a zext to get to the loaded slice!!");
14628     ArgChains.push_back(SliceInst.getValue(1));
14629   }
14630 
14631   SDValue Chain = DAG.getNode(ISD::TokenFactor, SDLoc(LD), MVT::Other,
14632                               ArgChains);
14633   DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
14634   AddToWorklist(Chain.getNode());
14635   return true;
14636 }
14637 
14638 /// Check to see if V is (and load (ptr), imm), where the load is having
14639 /// specific bytes cleared out.  If so, return the byte size being masked out
14640 /// and the shift amount.
14641 static std::pair<unsigned, unsigned>
14642 CheckForMaskedLoad(SDValue V, SDValue Ptr, SDValue Chain) {
14643   std::pair<unsigned, unsigned> Result(0, 0);
14644 
14645   // Check for the structure we're looking for.
14646   if (V->getOpcode() != ISD::AND ||
14647       !isa<ConstantSDNode>(V->getOperand(1)) ||
14648       !ISD::isNormalLoad(V->getOperand(0).getNode()))
14649     return Result;
14650 
14651   // Check the chain and pointer.
14652   LoadSDNode *LD = cast<LoadSDNode>(V->getOperand(0));
14653   if (LD->getBasePtr() != Ptr) return Result;  // Not from same pointer.
14654 
14655   // This only handles simple types.
14656   if (V.getValueType() != MVT::i16 &&
14657       V.getValueType() != MVT::i32 &&
14658       V.getValueType() != MVT::i64)
14659     return Result;
14660 
14661   // Check the constant mask.  Invert it so that the bits being masked out are
14662   // 0 and the bits being kept are 1.  Use getSExtValue so that leading bits
14663   // follow the sign bit for uniformity.
14664   uint64_t NotMask = ~cast<ConstantSDNode>(V->getOperand(1))->getSExtValue();
14665   unsigned NotMaskLZ = countLeadingZeros(NotMask);
14666   if (NotMaskLZ & 7) return Result;  // Must be multiple of a byte.
14667   unsigned NotMaskTZ = countTrailingZeros(NotMask);
14668   if (NotMaskTZ & 7) return Result;  // Must be multiple of a byte.
14669   if (NotMaskLZ == 64) return Result;  // All zero mask.
14670 
14671   // See if we have a continuous run of bits.  If so, we have 0*1+0*
14672   if (countTrailingOnes(NotMask >> NotMaskTZ) + NotMaskTZ + NotMaskLZ != 64)
14673     return Result;
14674 
14675   // Adjust NotMaskLZ down to be from the actual size of the int instead of i64.
14676   if (V.getValueType() != MVT::i64 && NotMaskLZ)
14677     NotMaskLZ -= 64-V.getValueSizeInBits();
14678 
14679   unsigned MaskedBytes = (V.getValueSizeInBits()-NotMaskLZ-NotMaskTZ)/8;
14680   switch (MaskedBytes) {
14681   case 1:
14682   case 2:
14683   case 4: break;
14684   default: return Result; // All one mask, or 5-byte mask.
14685   }
14686 
14687   // Verify that the first bit starts at a multiple of mask so that the access
14688   // is aligned the same as the access width.
14689   if (NotMaskTZ && NotMaskTZ/8 % MaskedBytes) return Result;
14690 
14691   // For narrowing to be valid, it must be the case that the load the
14692   // immediately preceding memory operation before the store.
14693   if (LD == Chain.getNode())
14694     ; // ok.
14695   else if (Chain->getOpcode() == ISD::TokenFactor &&
14696            SDValue(LD, 1).hasOneUse()) {
14697     // LD has only 1 chain use so they are no indirect dependencies.
14698     bool isOk = false;
14699     for (const SDValue &ChainOp : Chain->op_values())
14700       if (ChainOp.getNode() == LD) {
14701         isOk = true;
14702         break;
14703       }
14704     if (!isOk)
14705       return Result;
14706   } else
14707     return Result; // Fail.
14708 
14709   Result.first = MaskedBytes;
14710   Result.second = NotMaskTZ/8;
14711   return Result;
14712 }
14713 
14714 /// Check to see if IVal is something that provides a value as specified by
14715 /// MaskInfo. If so, replace the specified store with a narrower store of
14716 /// truncated IVal.
14717 static SDNode *
14718 ShrinkLoadReplaceStoreWithStore(const std::pair<unsigned, unsigned> &MaskInfo,
14719                                 SDValue IVal, StoreSDNode *St,
14720                                 DAGCombiner *DC) {
14721   unsigned NumBytes = MaskInfo.first;
14722   unsigned ByteShift = MaskInfo.second;
14723   SelectionDAG &DAG = DC->getDAG();
14724 
14725   // Check to see if IVal is all zeros in the part being masked in by the 'or'
14726   // that uses this.  If not, this is not a replacement.
14727   APInt Mask = ~APInt::getBitsSet(IVal.getValueSizeInBits(),
14728                                   ByteShift*8, (ByteShift+NumBytes)*8);
14729   if (!DAG.MaskedValueIsZero(IVal, Mask)) return nullptr;
14730 
14731   // Check that it is legal on the target to do this.  It is legal if the new
14732   // VT we're shrinking to (i8/i16/i32) is legal or we're still before type
14733   // legalization.
14734   MVT VT = MVT::getIntegerVT(NumBytes*8);
14735   if (!DC->isTypeLegal(VT))
14736     return nullptr;
14737 
14738   // Okay, we can do this!  Replace the 'St' store with a store of IVal that is
14739   // shifted by ByteShift and truncated down to NumBytes.
14740   if (ByteShift) {
14741     SDLoc DL(IVal);
14742     IVal = DAG.getNode(ISD::SRL, DL, IVal.getValueType(), IVal,
14743                        DAG.getConstant(ByteShift*8, DL,
14744                                     DC->getShiftAmountTy(IVal.getValueType())));
14745   }
14746 
14747   // Figure out the offset for the store and the alignment of the access.
14748   unsigned StOffset;
14749   unsigned NewAlign = St->getAlignment();
14750 
14751   if (DAG.getDataLayout().isLittleEndian())
14752     StOffset = ByteShift;
14753   else
14754     StOffset = IVal.getValueType().getStoreSize() - ByteShift - NumBytes;
14755 
14756   SDValue Ptr = St->getBasePtr();
14757   if (StOffset) {
14758     SDLoc DL(IVal);
14759     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(),
14760                       Ptr, DAG.getConstant(StOffset, DL, Ptr.getValueType()));
14761     NewAlign = MinAlign(NewAlign, StOffset);
14762   }
14763 
14764   // Truncate down to the new size.
14765   IVal = DAG.getNode(ISD::TRUNCATE, SDLoc(IVal), VT, IVal);
14766 
14767   ++OpsNarrowed;
14768   return DAG
14769       .getStore(St->getChain(), SDLoc(St), IVal, Ptr,
14770                 St->getPointerInfo().getWithOffset(StOffset), NewAlign)
14771       .getNode();
14772 }
14773 
14774 /// Look for sequence of load / op / store where op is one of 'or', 'xor', and
14775 /// 'and' of immediates. If 'op' is only touching some of the loaded bits, try
14776 /// narrowing the load and store if it would end up being a win for performance
14777 /// or code size.
14778 SDValue DAGCombiner::ReduceLoadOpStoreWidth(SDNode *N) {
14779   StoreSDNode *ST  = cast<StoreSDNode>(N);
14780   if (ST->isVolatile())
14781     return SDValue();
14782 
14783   SDValue Chain = ST->getChain();
14784   SDValue Value = ST->getValue();
14785   SDValue Ptr   = ST->getBasePtr();
14786   EVT VT = Value.getValueType();
14787 
14788   if (ST->isTruncatingStore() || VT.isVector() || !Value.hasOneUse())
14789     return SDValue();
14790 
14791   unsigned Opc = Value.getOpcode();
14792 
14793   // If this is "store (or X, Y), P" and X is "(and (load P), cst)", where cst
14794   // is a byte mask indicating a consecutive number of bytes, check to see if
14795   // Y is known to provide just those bytes.  If so, we try to replace the
14796   // load + replace + store sequence with a single (narrower) store, which makes
14797   // the load dead.
14798   if (Opc == ISD::OR) {
14799     std::pair<unsigned, unsigned> MaskedLoad;
14800     MaskedLoad = CheckForMaskedLoad(Value.getOperand(0), Ptr, Chain);
14801     if (MaskedLoad.first)
14802       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
14803                                                   Value.getOperand(1), ST,this))
14804         return SDValue(NewST, 0);
14805 
14806     // Or is commutative, so try swapping X and Y.
14807     MaskedLoad = CheckForMaskedLoad(Value.getOperand(1), Ptr, Chain);
14808     if (MaskedLoad.first)
14809       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
14810                                                   Value.getOperand(0), ST,this))
14811         return SDValue(NewST, 0);
14812   }
14813 
14814   if ((Opc != ISD::OR && Opc != ISD::XOR && Opc != ISD::AND) ||
14815       Value.getOperand(1).getOpcode() != ISD::Constant)
14816     return SDValue();
14817 
14818   SDValue N0 = Value.getOperand(0);
14819   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
14820       Chain == SDValue(N0.getNode(), 1)) {
14821     LoadSDNode *LD = cast<LoadSDNode>(N0);
14822     if (LD->getBasePtr() != Ptr ||
14823         LD->getPointerInfo().getAddrSpace() !=
14824         ST->getPointerInfo().getAddrSpace())
14825       return SDValue();
14826 
14827     // Find the type to narrow it the load / op / store to.
14828     SDValue N1 = Value.getOperand(1);
14829     unsigned BitWidth = N1.getValueSizeInBits();
14830     APInt Imm = cast<ConstantSDNode>(N1)->getAPIntValue();
14831     if (Opc == ISD::AND)
14832       Imm ^= APInt::getAllOnesValue(BitWidth);
14833     if (Imm == 0 || Imm.isAllOnesValue())
14834       return SDValue();
14835     unsigned ShAmt = Imm.countTrailingZeros();
14836     unsigned MSB = BitWidth - Imm.countLeadingZeros() - 1;
14837     unsigned NewBW = NextPowerOf2(MSB - ShAmt);
14838     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
14839     // The narrowing should be profitable, the load/store operation should be
14840     // legal (or custom) and the store size should be equal to the NewVT width.
14841     while (NewBW < BitWidth &&
14842            (NewVT.getStoreSizeInBits() != NewBW ||
14843             !TLI.isOperationLegalOrCustom(Opc, NewVT) ||
14844             !TLI.isNarrowingProfitable(VT, NewVT))) {
14845       NewBW = NextPowerOf2(NewBW);
14846       NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
14847     }
14848     if (NewBW >= BitWidth)
14849       return SDValue();
14850 
14851     // If the lsb changed does not start at the type bitwidth boundary,
14852     // start at the previous one.
14853     if (ShAmt % NewBW)
14854       ShAmt = (((ShAmt + NewBW - 1) / NewBW) * NewBW) - NewBW;
14855     APInt Mask = APInt::getBitsSet(BitWidth, ShAmt,
14856                                    std::min(BitWidth, ShAmt + NewBW));
14857     if ((Imm & Mask) == Imm) {
14858       APInt NewImm = (Imm & Mask).lshr(ShAmt).trunc(NewBW);
14859       if (Opc == ISD::AND)
14860         NewImm ^= APInt::getAllOnesValue(NewBW);
14861       uint64_t PtrOff = ShAmt / 8;
14862       // For big endian targets, we need to adjust the offset to the pointer to
14863       // load the correct bytes.
14864       if (DAG.getDataLayout().isBigEndian())
14865         PtrOff = (BitWidth + 7 - NewBW) / 8 - PtrOff;
14866 
14867       unsigned NewAlign = MinAlign(LD->getAlignment(), PtrOff);
14868       Type *NewVTTy = NewVT.getTypeForEVT(*DAG.getContext());
14869       if (NewAlign < DAG.getDataLayout().getABITypeAlignment(NewVTTy))
14870         return SDValue();
14871 
14872       SDValue NewPtr = DAG.getNode(ISD::ADD, SDLoc(LD),
14873                                    Ptr.getValueType(), Ptr,
14874                                    DAG.getConstant(PtrOff, SDLoc(LD),
14875                                                    Ptr.getValueType()));
14876       SDValue NewLD =
14877           DAG.getLoad(NewVT, SDLoc(N0), LD->getChain(), NewPtr,
14878                       LD->getPointerInfo().getWithOffset(PtrOff), NewAlign,
14879                       LD->getMemOperand()->getFlags(), LD->getAAInfo());
14880       SDValue NewVal = DAG.getNode(Opc, SDLoc(Value), NewVT, NewLD,
14881                                    DAG.getConstant(NewImm, SDLoc(Value),
14882                                                    NewVT));
14883       SDValue NewST =
14884           DAG.getStore(Chain, SDLoc(N), NewVal, NewPtr,
14885                        ST->getPointerInfo().getWithOffset(PtrOff), NewAlign);
14886 
14887       AddToWorklist(NewPtr.getNode());
14888       AddToWorklist(NewLD.getNode());
14889       AddToWorklist(NewVal.getNode());
14890       WorklistRemover DeadNodes(*this);
14891       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLD.getValue(1));
14892       ++OpsNarrowed;
14893       return NewST;
14894     }
14895   }
14896 
14897   return SDValue();
14898 }
14899 
14900 /// For a given floating point load / store pair, if the load value isn't used
14901 /// by any other operations, then consider transforming the pair to integer
14902 /// load / store operations if the target deems the transformation profitable.
14903 SDValue DAGCombiner::TransformFPLoadStorePair(SDNode *N) {
14904   StoreSDNode *ST  = cast<StoreSDNode>(N);
14905   SDValue Chain = ST->getChain();
14906   SDValue Value = ST->getValue();
14907   if (ISD::isNormalStore(ST) && ISD::isNormalLoad(Value.getNode()) &&
14908       Value.hasOneUse() &&
14909       Chain == SDValue(Value.getNode(), 1)) {
14910     LoadSDNode *LD = cast<LoadSDNode>(Value);
14911     EVT VT = LD->getMemoryVT();
14912     if (!VT.isFloatingPoint() ||
14913         VT != ST->getMemoryVT() ||
14914         LD->isNonTemporal() ||
14915         ST->isNonTemporal() ||
14916         LD->getPointerInfo().getAddrSpace() != 0 ||
14917         ST->getPointerInfo().getAddrSpace() != 0)
14918       return SDValue();
14919 
14920     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits());
14921     if (!TLI.isOperationLegal(ISD::LOAD, IntVT) ||
14922         !TLI.isOperationLegal(ISD::STORE, IntVT) ||
14923         !TLI.isDesirableToTransformToIntegerOp(ISD::LOAD, VT) ||
14924         !TLI.isDesirableToTransformToIntegerOp(ISD::STORE, VT))
14925       return SDValue();
14926 
14927     unsigned LDAlign = LD->getAlignment();
14928     unsigned STAlign = ST->getAlignment();
14929     Type *IntVTTy = IntVT.getTypeForEVT(*DAG.getContext());
14930     unsigned ABIAlign = DAG.getDataLayout().getABITypeAlignment(IntVTTy);
14931     if (LDAlign < ABIAlign || STAlign < ABIAlign)
14932       return SDValue();
14933 
14934     SDValue NewLD =
14935         DAG.getLoad(IntVT, SDLoc(Value), LD->getChain(), LD->getBasePtr(),
14936                     LD->getPointerInfo(), LDAlign);
14937 
14938     SDValue NewST =
14939         DAG.getStore(NewLD.getValue(1), SDLoc(N), NewLD, ST->getBasePtr(),
14940                      ST->getPointerInfo(), STAlign);
14941 
14942     AddToWorklist(NewLD.getNode());
14943     AddToWorklist(NewST.getNode());
14944     WorklistRemover DeadNodes(*this);
14945     DAG.ReplaceAllUsesOfValueWith(Value.getValue(1), NewLD.getValue(1));
14946     ++LdStFP2Int;
14947     return NewST;
14948   }
14949 
14950   return SDValue();
14951 }
14952 
14953 // This is a helper function for visitMUL to check the profitability
14954 // of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
14955 // MulNode is the original multiply, AddNode is (add x, c1),
14956 // and ConstNode is c2.
14957 //
14958 // If the (add x, c1) has multiple uses, we could increase
14959 // the number of adds if we make this transformation.
14960 // It would only be worth doing this if we can remove a
14961 // multiply in the process. Check for that here.
14962 // To illustrate:
14963 //     (A + c1) * c3
14964 //     (A + c2) * c3
14965 // We're checking for cases where we have common "c3 * A" expressions.
14966 bool DAGCombiner::isMulAddWithConstProfitable(SDNode *MulNode,
14967                                               SDValue &AddNode,
14968                                               SDValue &ConstNode) {
14969   APInt Val;
14970 
14971   // If the add only has one use, this would be OK to do.
14972   if (AddNode.getNode()->hasOneUse())
14973     return true;
14974 
14975   // Walk all the users of the constant with which we're multiplying.
14976   for (SDNode *Use : ConstNode->uses()) {
14977     if (Use == MulNode) // This use is the one we're on right now. Skip it.
14978       continue;
14979 
14980     if (Use->getOpcode() == ISD::MUL) { // We have another multiply use.
14981       SDNode *OtherOp;
14982       SDNode *MulVar = AddNode.getOperand(0).getNode();
14983 
14984       // OtherOp is what we're multiplying against the constant.
14985       if (Use->getOperand(0) == ConstNode)
14986         OtherOp = Use->getOperand(1).getNode();
14987       else
14988         OtherOp = Use->getOperand(0).getNode();
14989 
14990       // Check to see if multiply is with the same operand of our "add".
14991       //
14992       //     ConstNode  = CONST
14993       //     Use = ConstNode * A  <-- visiting Use. OtherOp is A.
14994       //     ...
14995       //     AddNode  = (A + c1)  <-- MulVar is A.
14996       //         = AddNode * ConstNode   <-- current visiting instruction.
14997       //
14998       // If we make this transformation, we will have a common
14999       // multiply (ConstNode * A) that we can save.
15000       if (OtherOp == MulVar)
15001         return true;
15002 
15003       // Now check to see if a future expansion will give us a common
15004       // multiply.
15005       //
15006       //     ConstNode  = CONST
15007       //     AddNode    = (A + c1)
15008       //     ...   = AddNode * ConstNode <-- current visiting instruction.
15009       //     ...
15010       //     OtherOp = (A + c2)
15011       //     Use     = OtherOp * ConstNode <-- visiting Use.
15012       //
15013       // If we make this transformation, we will have a common
15014       // multiply (CONST * A) after we also do the same transformation
15015       // to the "t2" instruction.
15016       if (OtherOp->getOpcode() == ISD::ADD &&
15017           DAG.isConstantIntBuildVectorOrConstantInt(OtherOp->getOperand(1)) &&
15018           OtherOp->getOperand(0).getNode() == MulVar)
15019         return true;
15020     }
15021   }
15022 
15023   // Didn't find a case where this would be profitable.
15024   return false;
15025 }
15026 
15027 SDValue DAGCombiner::getMergeStoreChains(SmallVectorImpl<MemOpLink> &StoreNodes,
15028                                          unsigned NumStores) {
15029   SmallVector<SDValue, 8> Chains;
15030   SmallPtrSet<const SDNode *, 8> Visited;
15031   SDLoc StoreDL(StoreNodes[0].MemNode);
15032 
15033   for (unsigned i = 0; i < NumStores; ++i) {
15034     Visited.insert(StoreNodes[i].MemNode);
15035   }
15036 
15037   // don't include nodes that are children or repeated nodes.
15038   for (unsigned i = 0; i < NumStores; ++i) {
15039     if (Visited.insert(StoreNodes[i].MemNode->getChain().getNode()).second)
15040       Chains.push_back(StoreNodes[i].MemNode->getChain());
15041   }
15042 
15043   assert(Chains.size() > 0 && "Chain should have generated a chain");
15044   return DAG.getTokenFactor(StoreDL, Chains);
15045 }
15046 
15047 bool DAGCombiner::MergeStoresOfConstantsOrVecElts(
15048     SmallVectorImpl<MemOpLink> &StoreNodes, EVT MemVT, unsigned NumStores,
15049     bool IsConstantSrc, bool UseVector, bool UseTrunc) {
15050   // Make sure we have something to merge.
15051   if (NumStores < 2)
15052     return false;
15053 
15054   // The latest Node in the DAG.
15055   SDLoc DL(StoreNodes[0].MemNode);
15056 
15057   int64_t ElementSizeBits = MemVT.getStoreSizeInBits();
15058   unsigned SizeInBits = NumStores * ElementSizeBits;
15059   unsigned NumMemElts = MemVT.isVector() ? MemVT.getVectorNumElements() : 1;
15060 
15061   EVT StoreTy;
15062   if (UseVector) {
15063     unsigned Elts = NumStores * NumMemElts;
15064     // Get the type for the merged vector store.
15065     StoreTy = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
15066   } else
15067     StoreTy = EVT::getIntegerVT(*DAG.getContext(), SizeInBits);
15068 
15069   SDValue StoredVal;
15070   if (UseVector) {
15071     if (IsConstantSrc) {
15072       SmallVector<SDValue, 8> BuildVector;
15073       for (unsigned I = 0; I != NumStores; ++I) {
15074         StoreSDNode *St = cast<StoreSDNode>(StoreNodes[I].MemNode);
15075         SDValue Val = St->getValue();
15076         // If constant is of the wrong type, convert it now.
15077         if (MemVT != Val.getValueType()) {
15078           Val = peekThroughBitcasts(Val);
15079           // Deal with constants of wrong size.
15080           if (ElementSizeBits != Val.getValueSizeInBits()) {
15081             EVT IntMemVT =
15082                 EVT::getIntegerVT(*DAG.getContext(), MemVT.getSizeInBits());
15083             if (isa<ConstantFPSDNode>(Val)) {
15084               // Not clear how to truncate FP values.
15085               return false;
15086             } else if (auto *C = dyn_cast<ConstantSDNode>(Val))
15087               Val = DAG.getConstant(C->getAPIntValue()
15088                                         .zextOrTrunc(Val.getValueSizeInBits())
15089                                         .zextOrTrunc(ElementSizeBits),
15090                                     SDLoc(C), IntMemVT);
15091           }
15092           // Make sure correctly size type is the correct type.
15093           Val = DAG.getBitcast(MemVT, Val);
15094         }
15095         BuildVector.push_back(Val);
15096       }
15097       StoredVal = DAG.getNode(MemVT.isVector() ? ISD::CONCAT_VECTORS
15098                                                : ISD::BUILD_VECTOR,
15099                               DL, StoreTy, BuildVector);
15100     } else {
15101       SmallVector<SDValue, 8> Ops;
15102       for (unsigned i = 0; i < NumStores; ++i) {
15103         StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
15104         SDValue Val = peekThroughBitcasts(St->getValue());
15105         // All operands of BUILD_VECTOR / CONCAT_VECTOR must be of
15106         // type MemVT. If the underlying value is not the correct
15107         // type, but it is an extraction of an appropriate vector we
15108         // can recast Val to be of the correct type. This may require
15109         // converting between EXTRACT_VECTOR_ELT and
15110         // EXTRACT_SUBVECTOR.
15111         if ((MemVT != Val.getValueType()) &&
15112             (Val.getOpcode() == ISD::EXTRACT_VECTOR_ELT ||
15113              Val.getOpcode() == ISD::EXTRACT_SUBVECTOR)) {
15114           EVT MemVTScalarTy = MemVT.getScalarType();
15115           // We may need to add a bitcast here to get types to line up.
15116           if (MemVTScalarTy != Val.getValueType().getScalarType()) {
15117             Val = DAG.getBitcast(MemVT, Val);
15118           } else {
15119             unsigned OpC = MemVT.isVector() ? ISD::EXTRACT_SUBVECTOR
15120                                             : ISD::EXTRACT_VECTOR_ELT;
15121             SDValue Vec = Val.getOperand(0);
15122             SDValue Idx = Val.getOperand(1);
15123             Val = DAG.getNode(OpC, SDLoc(Val), MemVT, Vec, Idx);
15124           }
15125         }
15126         Ops.push_back(Val);
15127       }
15128 
15129       // Build the extracted vector elements back into a vector.
15130       StoredVal = DAG.getNode(MemVT.isVector() ? ISD::CONCAT_VECTORS
15131                                                : ISD::BUILD_VECTOR,
15132                               DL, StoreTy, Ops);
15133     }
15134   } else {
15135     // We should always use a vector store when merging extracted vector
15136     // elements, so this path implies a store of constants.
15137     assert(IsConstantSrc && "Merged vector elements should use vector store");
15138 
15139     APInt StoreInt(SizeInBits, 0);
15140 
15141     // Construct a single integer constant which is made of the smaller
15142     // constant inputs.
15143     bool IsLE = DAG.getDataLayout().isLittleEndian();
15144     for (unsigned i = 0; i < NumStores; ++i) {
15145       unsigned Idx = IsLE ? (NumStores - 1 - i) : i;
15146       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[Idx].MemNode);
15147 
15148       SDValue Val = St->getValue();
15149       Val = peekThroughBitcasts(Val);
15150       StoreInt <<= ElementSizeBits;
15151       if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Val)) {
15152         StoreInt |= C->getAPIntValue()
15153                         .zextOrTrunc(ElementSizeBits)
15154                         .zextOrTrunc(SizeInBits);
15155       } else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(Val)) {
15156         StoreInt |= C->getValueAPF()
15157                         .bitcastToAPInt()
15158                         .zextOrTrunc(ElementSizeBits)
15159                         .zextOrTrunc(SizeInBits);
15160         // If fp truncation is necessary give up for now.
15161         if (MemVT.getSizeInBits() != ElementSizeBits)
15162           return false;
15163       } else {
15164         llvm_unreachable("Invalid constant element type");
15165       }
15166     }
15167 
15168     // Create the new Load and Store operations.
15169     StoredVal = DAG.getConstant(StoreInt, DL, StoreTy);
15170   }
15171 
15172   LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
15173   SDValue NewChain = getMergeStoreChains(StoreNodes, NumStores);
15174 
15175   // make sure we use trunc store if it's necessary to be legal.
15176   SDValue NewStore;
15177   if (!UseTrunc) {
15178     NewStore = DAG.getStore(NewChain, DL, StoredVal, FirstInChain->getBasePtr(),
15179                             FirstInChain->getPointerInfo(),
15180                             FirstInChain->getAlignment());
15181   } else { // Must be realized as a trunc store
15182     EVT LegalizedStoredValTy =
15183         TLI.getTypeToTransformTo(*DAG.getContext(), StoredVal.getValueType());
15184     unsigned LegalizedStoreSize = LegalizedStoredValTy.getSizeInBits();
15185     ConstantSDNode *C = cast<ConstantSDNode>(StoredVal);
15186     SDValue ExtendedStoreVal =
15187         DAG.getConstant(C->getAPIntValue().zextOrTrunc(LegalizedStoreSize), DL,
15188                         LegalizedStoredValTy);
15189     NewStore = DAG.getTruncStore(
15190         NewChain, DL, ExtendedStoreVal, FirstInChain->getBasePtr(),
15191         FirstInChain->getPointerInfo(), StoredVal.getValueType() /*TVT*/,
15192         FirstInChain->getAlignment(),
15193         FirstInChain->getMemOperand()->getFlags());
15194   }
15195 
15196   // Replace all merged stores with the new store.
15197   for (unsigned i = 0; i < NumStores; ++i)
15198     CombineTo(StoreNodes[i].MemNode, NewStore);
15199 
15200   AddToWorklist(NewChain.getNode());
15201   return true;
15202 }
15203 
15204 void DAGCombiner::getStoreMergeCandidates(
15205     StoreSDNode *St, SmallVectorImpl<MemOpLink> &StoreNodes,
15206     SDNode *&RootNode) {
15207   // This holds the base pointer, index, and the offset in bytes from the base
15208   // pointer.
15209   BaseIndexOffset BasePtr = BaseIndexOffset::match(St, DAG);
15210   EVT MemVT = St->getMemoryVT();
15211 
15212   SDValue Val = peekThroughBitcasts(St->getValue());
15213   // We must have a base and an offset.
15214   if (!BasePtr.getBase().getNode())
15215     return;
15216 
15217   // Do not handle stores to undef base pointers.
15218   if (BasePtr.getBase().isUndef())
15219     return;
15220 
15221   bool IsConstantSrc = isa<ConstantSDNode>(Val) || isa<ConstantFPSDNode>(Val);
15222   bool IsExtractVecSrc = (Val.getOpcode() == ISD::EXTRACT_VECTOR_ELT ||
15223                           Val.getOpcode() == ISD::EXTRACT_SUBVECTOR);
15224   bool IsLoadSrc = isa<LoadSDNode>(Val);
15225   BaseIndexOffset LBasePtr;
15226   // Match on loadbaseptr if relevant.
15227   EVT LoadVT;
15228   if (IsLoadSrc) {
15229     auto *Ld = cast<LoadSDNode>(Val);
15230     LBasePtr = BaseIndexOffset::match(Ld, DAG);
15231     LoadVT = Ld->getMemoryVT();
15232     // Load and store should be the same type.
15233     if (MemVT != LoadVT)
15234       return;
15235     // Loads must only have one use.
15236     if (!Ld->hasNUsesOfValue(1, 0))
15237       return;
15238     // The memory operands must not be volatile/indexed.
15239     if (Ld->isVolatile() || Ld->isIndexed())
15240       return;
15241   }
15242   auto CandidateMatch = [&](StoreSDNode *Other, BaseIndexOffset &Ptr,
15243                             int64_t &Offset) -> bool {
15244     // The memory operands must not be volatile/indexed.
15245     if (Other->isVolatile() || Other->isIndexed())
15246       return false;
15247     // Don't mix temporal stores with non-temporal stores.
15248     if (St->isNonTemporal() != Other->isNonTemporal())
15249       return false;
15250     SDValue OtherBC = peekThroughBitcasts(Other->getValue());
15251     // Allow merging constants of different types as integers.
15252     bool NoTypeMatch = (MemVT.isInteger()) ? !MemVT.bitsEq(Other->getMemoryVT())
15253                                            : Other->getMemoryVT() != MemVT;
15254     if (IsLoadSrc) {
15255       if (NoTypeMatch)
15256         return false;
15257       // The Load's Base Ptr must also match
15258       if (LoadSDNode *OtherLd = dyn_cast<LoadSDNode>(OtherBC)) {
15259         BaseIndexOffset LPtr = BaseIndexOffset::match(OtherLd, DAG);
15260         if (LoadVT != OtherLd->getMemoryVT())
15261           return false;
15262         // Loads must only have one use.
15263         if (!OtherLd->hasNUsesOfValue(1, 0))
15264           return false;
15265         // The memory operands must not be volatile/indexed.
15266         if (OtherLd->isVolatile() || OtherLd->isIndexed())
15267           return false;
15268         // Don't mix temporal loads with non-temporal loads.
15269         if (cast<LoadSDNode>(Val)->isNonTemporal() != OtherLd->isNonTemporal())
15270           return false;
15271         if (!(LBasePtr.equalBaseIndex(LPtr, DAG)))
15272           return false;
15273       } else
15274         return false;
15275     }
15276     if (IsConstantSrc) {
15277       if (NoTypeMatch)
15278         return false;
15279       if (!(isa<ConstantSDNode>(OtherBC) || isa<ConstantFPSDNode>(OtherBC)))
15280         return false;
15281     }
15282     if (IsExtractVecSrc) {
15283       // Do not merge truncated stores here.
15284       if (Other->isTruncatingStore())
15285         return false;
15286       if (!MemVT.bitsEq(OtherBC.getValueType()))
15287         return false;
15288       if (OtherBC.getOpcode() != ISD::EXTRACT_VECTOR_ELT &&
15289           OtherBC.getOpcode() != ISD::EXTRACT_SUBVECTOR)
15290         return false;
15291     }
15292     Ptr = BaseIndexOffset::match(Other, DAG);
15293     return (BasePtr.equalBaseIndex(Ptr, DAG, Offset));
15294   };
15295 
15296   // We looking for a root node which is an ancestor to all mergable
15297   // stores. We search up through a load, to our root and then down
15298   // through all children. For instance we will find Store{1,2,3} if
15299   // St is Store1, Store2. or Store3 where the root is not a load
15300   // which always true for nonvolatile ops. TODO: Expand
15301   // the search to find all valid candidates through multiple layers of loads.
15302   //
15303   // Root
15304   // |-------|-------|
15305   // Load    Load    Store3
15306   // |       |
15307   // Store1   Store2
15308   //
15309   // FIXME: We should be able to climb and
15310   // descend TokenFactors to find candidates as well.
15311 
15312   RootNode = St->getChain().getNode();
15313 
15314   unsigned NumNodesExplored = 0;
15315   if (LoadSDNode *Ldn = dyn_cast<LoadSDNode>(RootNode)) {
15316     RootNode = Ldn->getChain().getNode();
15317     for (auto I = RootNode->use_begin(), E = RootNode->use_end();
15318          I != E && NumNodesExplored < 1024; ++I, ++NumNodesExplored)
15319       if (I.getOperandNo() == 0 && isa<LoadSDNode>(*I)) // walk down chain
15320         for (auto I2 = (*I)->use_begin(), E2 = (*I)->use_end(); I2 != E2; ++I2)
15321           if (I2.getOperandNo() == 0)
15322             if (StoreSDNode *OtherST = dyn_cast<StoreSDNode>(*I2)) {
15323               BaseIndexOffset Ptr;
15324               int64_t PtrDiff;
15325               if (CandidateMatch(OtherST, Ptr, PtrDiff))
15326                 StoreNodes.push_back(MemOpLink(OtherST, PtrDiff));
15327             }
15328   } else
15329     for (auto I = RootNode->use_begin(), E = RootNode->use_end();
15330          I != E && NumNodesExplored < 1024; ++I, ++NumNodesExplored)
15331       if (I.getOperandNo() == 0)
15332         if (StoreSDNode *OtherST = dyn_cast<StoreSDNode>(*I)) {
15333           BaseIndexOffset Ptr;
15334           int64_t PtrDiff;
15335           if (CandidateMatch(OtherST, Ptr, PtrDiff))
15336             StoreNodes.push_back(MemOpLink(OtherST, PtrDiff));
15337         }
15338 }
15339 
15340 // We need to check that merging these stores does not cause a loop in
15341 // the DAG. Any store candidate may depend on another candidate
15342 // indirectly through its operand (we already consider dependencies
15343 // through the chain). Check in parallel by searching up from
15344 // non-chain operands of candidates.
15345 bool DAGCombiner::checkMergeStoreCandidatesForDependencies(
15346     SmallVectorImpl<MemOpLink> &StoreNodes, unsigned NumStores,
15347     SDNode *RootNode) {
15348   // FIXME: We should be able to truncate a full search of
15349   // predecessors by doing a BFS and keeping tabs the originating
15350   // stores from which worklist nodes come from in a similar way to
15351   // TokenFactor simplfication.
15352 
15353   SmallPtrSet<const SDNode *, 32> Visited;
15354   SmallVector<const SDNode *, 8> Worklist;
15355 
15356   // RootNode is a predecessor to all candidates so we need not search
15357   // past it. Add RootNode (peeking through TokenFactors). Do not count
15358   // these towards size check.
15359 
15360   Worklist.push_back(RootNode);
15361   while (!Worklist.empty()) {
15362     auto N = Worklist.pop_back_val();
15363     if (!Visited.insert(N).second)
15364       continue; // Already present in Visited.
15365     if (N->getOpcode() == ISD::TokenFactor) {
15366       for (SDValue Op : N->ops())
15367         Worklist.push_back(Op.getNode());
15368     }
15369   }
15370 
15371   // Don't count pruning nodes towards max.
15372   unsigned int Max = 1024 + Visited.size();
15373   // Search Ops of store candidates.
15374   for (unsigned i = 0; i < NumStores; ++i) {
15375     SDNode *N = StoreNodes[i].MemNode;
15376     // Of the 4 Store Operands:
15377     //   * Chain (Op 0) -> We have already considered these
15378     //                    in candidate selection and can be
15379     //                    safely ignored
15380     //   * Value (Op 1) -> Cycles may happen (e.g. through load chains)
15381     //   * Address (Op 2) -> Merged addresses may only vary by a fixed constant,
15382     //                       but aren't necessarily fromt the same base node, so
15383     //                       cycles possible (e.g. via indexed store).
15384     //   * (Op 3) -> Represents the pre or post-indexing offset (or undef for
15385     //               non-indexed stores). Not constant on all targets (e.g. ARM)
15386     //               and so can participate in a cycle.
15387     for (unsigned j = 1; j < N->getNumOperands(); ++j)
15388       Worklist.push_back(N->getOperand(j).getNode());
15389   }
15390   // Search through DAG. We can stop early if we find a store node.
15391   for (unsigned i = 0; i < NumStores; ++i)
15392     if (SDNode::hasPredecessorHelper(StoreNodes[i].MemNode, Visited, Worklist,
15393                                      Max))
15394       return false;
15395   return true;
15396 }
15397 
15398 bool DAGCombiner::MergeConsecutiveStores(StoreSDNode *St) {
15399   if (OptLevel == CodeGenOpt::None)
15400     return false;
15401 
15402   EVT MemVT = St->getMemoryVT();
15403   int64_t ElementSizeBytes = MemVT.getStoreSize();
15404   unsigned NumMemElts = MemVT.isVector() ? MemVT.getVectorNumElements() : 1;
15405 
15406   if (MemVT.getSizeInBits() * 2 > MaximumLegalStoreInBits)
15407     return false;
15408 
15409   bool NoVectors = DAG.getMachineFunction().getFunction().hasFnAttribute(
15410       Attribute::NoImplicitFloat);
15411 
15412   // This function cannot currently deal with non-byte-sized memory sizes.
15413   if (ElementSizeBytes * 8 != MemVT.getSizeInBits())
15414     return false;
15415 
15416   if (!MemVT.isSimple())
15417     return false;
15418 
15419   // Perform an early exit check. Do not bother looking at stored values that
15420   // are not constants, loads, or extracted vector elements.
15421   SDValue StoredVal = peekThroughBitcasts(St->getValue());
15422   bool IsLoadSrc = isa<LoadSDNode>(StoredVal);
15423   bool IsConstantSrc = isa<ConstantSDNode>(StoredVal) ||
15424                        isa<ConstantFPSDNode>(StoredVal);
15425   bool IsExtractVecSrc = (StoredVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT ||
15426                           StoredVal.getOpcode() == ISD::EXTRACT_SUBVECTOR);
15427   bool IsNonTemporalStore = St->isNonTemporal();
15428   bool IsNonTemporalLoad =
15429       IsLoadSrc && cast<LoadSDNode>(StoredVal)->isNonTemporal();
15430 
15431   if (!IsConstantSrc && !IsLoadSrc && !IsExtractVecSrc)
15432     return false;
15433 
15434   SmallVector<MemOpLink, 8> StoreNodes;
15435   SDNode *RootNode;
15436   // Find potential store merge candidates by searching through chain sub-DAG
15437   getStoreMergeCandidates(St, StoreNodes, RootNode);
15438 
15439   // Check if there is anything to merge.
15440   if (StoreNodes.size() < 2)
15441     return false;
15442 
15443   // Sort the memory operands according to their distance from the
15444   // base pointer.
15445   llvm::sort(StoreNodes, [](MemOpLink LHS, MemOpLink RHS) {
15446     return LHS.OffsetFromBase < RHS.OffsetFromBase;
15447   });
15448 
15449   // Store Merge attempts to merge the lowest stores. This generally
15450   // works out as if successful, as the remaining stores are checked
15451   // after the first collection of stores is merged. However, in the
15452   // case that a non-mergeable store is found first, e.g., {p[-2],
15453   // p[0], p[1], p[2], p[3]}, we would fail and miss the subsequent
15454   // mergeable cases. To prevent this, we prune such stores from the
15455   // front of StoreNodes here.
15456 
15457   bool RV = false;
15458   while (StoreNodes.size() > 1) {
15459     unsigned StartIdx = 0;
15460     while ((StartIdx + 1 < StoreNodes.size()) &&
15461            StoreNodes[StartIdx].OffsetFromBase + ElementSizeBytes !=
15462                StoreNodes[StartIdx + 1].OffsetFromBase)
15463       ++StartIdx;
15464 
15465     // Bail if we don't have enough candidates to merge.
15466     if (StartIdx + 1 >= StoreNodes.size())
15467       return RV;
15468 
15469     if (StartIdx)
15470       StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + StartIdx);
15471 
15472     // Scan the memory operations on the chain and find the first
15473     // non-consecutive store memory address.
15474     unsigned NumConsecutiveStores = 1;
15475     int64_t StartAddress = StoreNodes[0].OffsetFromBase;
15476     // Check that the addresses are consecutive starting from the second
15477     // element in the list of stores.
15478     for (unsigned i = 1, e = StoreNodes.size(); i < e; ++i) {
15479       int64_t CurrAddress = StoreNodes[i].OffsetFromBase;
15480       if (CurrAddress - StartAddress != (ElementSizeBytes * i))
15481         break;
15482       NumConsecutiveStores = i + 1;
15483     }
15484 
15485     if (NumConsecutiveStores < 2) {
15486       StoreNodes.erase(StoreNodes.begin(),
15487                        StoreNodes.begin() + NumConsecutiveStores);
15488       continue;
15489     }
15490 
15491     // The node with the lowest store address.
15492     LLVMContext &Context = *DAG.getContext();
15493     const DataLayout &DL = DAG.getDataLayout();
15494 
15495     // Store the constants into memory as one consecutive store.
15496     if (IsConstantSrc) {
15497       while (NumConsecutiveStores >= 2) {
15498         LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
15499         unsigned FirstStoreAS = FirstInChain->getAddressSpace();
15500         unsigned FirstStoreAlign = FirstInChain->getAlignment();
15501         unsigned LastLegalType = 1;
15502         unsigned LastLegalVectorType = 1;
15503         bool LastIntegerTrunc = false;
15504         bool NonZero = false;
15505         unsigned FirstZeroAfterNonZero = NumConsecutiveStores;
15506         for (unsigned i = 0; i < NumConsecutiveStores; ++i) {
15507           StoreSDNode *ST = cast<StoreSDNode>(StoreNodes[i].MemNode);
15508           SDValue StoredVal = ST->getValue();
15509           bool IsElementZero = false;
15510           if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(StoredVal))
15511             IsElementZero = C->isNullValue();
15512           else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(StoredVal))
15513             IsElementZero = C->getConstantFPValue()->isNullValue();
15514           if (IsElementZero) {
15515             if (NonZero && FirstZeroAfterNonZero == NumConsecutiveStores)
15516               FirstZeroAfterNonZero = i;
15517           }
15518           NonZero |= !IsElementZero;
15519 
15520           // Find a legal type for the constant store.
15521           unsigned SizeInBits = (i + 1) * ElementSizeBytes * 8;
15522           EVT StoreTy = EVT::getIntegerVT(Context, SizeInBits);
15523           bool IsFast = false;
15524 
15525           // Break early when size is too large to be legal.
15526           if (StoreTy.getSizeInBits() > MaximumLegalStoreInBits)
15527             break;
15528 
15529           if (TLI.isTypeLegal(StoreTy) &&
15530               TLI.canMergeStoresTo(FirstStoreAS, StoreTy, DAG) &&
15531               TLI.allowsMemoryAccess(Context, DL, StoreTy,
15532                                      *FirstInChain->getMemOperand(), &IsFast) &&
15533               IsFast) {
15534             LastIntegerTrunc = false;
15535             LastLegalType = i + 1;
15536             // Or check whether a truncstore is legal.
15537           } else if (TLI.getTypeAction(Context, StoreTy) ==
15538                      TargetLowering::TypePromoteInteger) {
15539             EVT LegalizedStoredValTy =
15540                 TLI.getTypeToTransformTo(Context, StoredVal.getValueType());
15541             if (TLI.isTruncStoreLegal(LegalizedStoredValTy, StoreTy) &&
15542                 TLI.canMergeStoresTo(FirstStoreAS, LegalizedStoredValTy, DAG) &&
15543                 TLI.allowsMemoryAccess(Context, DL, StoreTy,
15544                                        *FirstInChain->getMemOperand(),
15545                                        &IsFast) &&
15546                 IsFast) {
15547               LastIntegerTrunc = true;
15548               LastLegalType = i + 1;
15549             }
15550           }
15551 
15552           // We only use vectors if the constant is known to be zero or the
15553           // target allows it and the function is not marked with the
15554           // noimplicitfloat attribute.
15555           if ((!NonZero ||
15556                TLI.storeOfVectorConstantIsCheap(MemVT, i + 1, FirstStoreAS)) &&
15557               !NoVectors) {
15558             // Find a legal type for the vector store.
15559             unsigned Elts = (i + 1) * NumMemElts;
15560             EVT Ty = EVT::getVectorVT(Context, MemVT.getScalarType(), Elts);
15561             if (TLI.isTypeLegal(Ty) && TLI.isTypeLegal(MemVT) &&
15562                 TLI.canMergeStoresTo(FirstStoreAS, Ty, DAG) &&
15563                 TLI.allowsMemoryAccess(
15564                     Context, DL, Ty, *FirstInChain->getMemOperand(), &IsFast) &&
15565                 IsFast)
15566               LastLegalVectorType = i + 1;
15567           }
15568         }
15569 
15570         bool UseVector = (LastLegalVectorType > LastLegalType) && !NoVectors;
15571         unsigned NumElem = (UseVector) ? LastLegalVectorType : LastLegalType;
15572 
15573         // Check if we found a legal integer type that creates a meaningful
15574         // merge.
15575         if (NumElem < 2) {
15576           // We know that candidate stores are in order and of correct
15577           // shape. While there is no mergeable sequence from the
15578           // beginning one may start later in the sequence. The only
15579           // reason a merge of size N could have failed where another of
15580           // the same size would not have, is if the alignment has
15581           // improved or we've dropped a non-zero value. Drop as many
15582           // candidates as we can here.
15583           unsigned NumSkip = 1;
15584           while (
15585               (NumSkip < NumConsecutiveStores) &&
15586               (NumSkip < FirstZeroAfterNonZero) &&
15587               (StoreNodes[NumSkip].MemNode->getAlignment() <= FirstStoreAlign))
15588             NumSkip++;
15589 
15590           StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumSkip);
15591           NumConsecutiveStores -= NumSkip;
15592           continue;
15593         }
15594 
15595         // Check that we can merge these candidates without causing a cycle.
15596         if (!checkMergeStoreCandidatesForDependencies(StoreNodes, NumElem,
15597                                                       RootNode)) {
15598           StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem);
15599           NumConsecutiveStores -= NumElem;
15600           continue;
15601         }
15602 
15603         RV |= MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumElem, true,
15604                                               UseVector, LastIntegerTrunc);
15605 
15606         // Remove merged stores for next iteration.
15607         StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem);
15608         NumConsecutiveStores -= NumElem;
15609       }
15610       continue;
15611     }
15612 
15613     // When extracting multiple vector elements, try to store them
15614     // in one vector store rather than a sequence of scalar stores.
15615     if (IsExtractVecSrc) {
15616       // Loop on Consecutive Stores on success.
15617       while (NumConsecutiveStores >= 2) {
15618         LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
15619         unsigned FirstStoreAS = FirstInChain->getAddressSpace();
15620         unsigned FirstStoreAlign = FirstInChain->getAlignment();
15621         unsigned NumStoresToMerge = 1;
15622         for (unsigned i = 0; i < NumConsecutiveStores; ++i) {
15623           // Find a legal type for the vector store.
15624           unsigned Elts = (i + 1) * NumMemElts;
15625           EVT Ty =
15626               EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
15627           bool IsFast;
15628 
15629           // Break early when size is too large to be legal.
15630           if (Ty.getSizeInBits() > MaximumLegalStoreInBits)
15631             break;
15632 
15633           if (TLI.isTypeLegal(Ty) &&
15634               TLI.canMergeStoresTo(FirstStoreAS, Ty, DAG) &&
15635               TLI.allowsMemoryAccess(Context, DL, Ty,
15636                                      *FirstInChain->getMemOperand(), &IsFast) &&
15637               IsFast)
15638             NumStoresToMerge = i + 1;
15639         }
15640 
15641         // Check if we found a legal integer type creating a meaningful
15642         // merge.
15643         if (NumStoresToMerge < 2) {
15644           // We know that candidate stores are in order and of correct
15645           // shape. While there is no mergeable sequence from the
15646           // beginning one may start later in the sequence. The only
15647           // reason a merge of size N could have failed where another of
15648           // the same size would not have, is if the alignment has
15649           // improved. Drop as many candidates as we can here.
15650           unsigned NumSkip = 1;
15651           while (
15652               (NumSkip < NumConsecutiveStores) &&
15653               (StoreNodes[NumSkip].MemNode->getAlignment() <= FirstStoreAlign))
15654             NumSkip++;
15655 
15656           StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumSkip);
15657           NumConsecutiveStores -= NumSkip;
15658           continue;
15659         }
15660 
15661         // Check that we can merge these candidates without causing a cycle.
15662         if (!checkMergeStoreCandidatesForDependencies(
15663                 StoreNodes, NumStoresToMerge, RootNode)) {
15664           StoreNodes.erase(StoreNodes.begin(),
15665                            StoreNodes.begin() + NumStoresToMerge);
15666           NumConsecutiveStores -= NumStoresToMerge;
15667           continue;
15668         }
15669 
15670         RV |= MergeStoresOfConstantsOrVecElts(
15671             StoreNodes, MemVT, NumStoresToMerge, false, true, false);
15672 
15673         StoreNodes.erase(StoreNodes.begin(),
15674                          StoreNodes.begin() + NumStoresToMerge);
15675         NumConsecutiveStores -= NumStoresToMerge;
15676       }
15677       continue;
15678     }
15679 
15680     // Below we handle the case of multiple consecutive stores that
15681     // come from multiple consecutive loads. We merge them into a single
15682     // wide load and a single wide store.
15683 
15684     // Look for load nodes which are used by the stored values.
15685     SmallVector<MemOpLink, 8> LoadNodes;
15686 
15687     // Find acceptable loads. Loads need to have the same chain (token factor),
15688     // must not be zext, volatile, indexed, and they must be consecutive.
15689     BaseIndexOffset LdBasePtr;
15690 
15691     for (unsigned i = 0; i < NumConsecutiveStores; ++i) {
15692       StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
15693       SDValue Val = peekThroughBitcasts(St->getValue());
15694       LoadSDNode *Ld = cast<LoadSDNode>(Val);
15695 
15696       BaseIndexOffset LdPtr = BaseIndexOffset::match(Ld, DAG);
15697       // If this is not the first ptr that we check.
15698       int64_t LdOffset = 0;
15699       if (LdBasePtr.getBase().getNode()) {
15700         // The base ptr must be the same.
15701         if (!LdBasePtr.equalBaseIndex(LdPtr, DAG, LdOffset))
15702           break;
15703       } else {
15704         // Check that all other base pointers are the same as this one.
15705         LdBasePtr = LdPtr;
15706       }
15707 
15708       // We found a potential memory operand to merge.
15709       LoadNodes.push_back(MemOpLink(Ld, LdOffset));
15710     }
15711 
15712     while (NumConsecutiveStores >= 2 && LoadNodes.size() >= 2) {
15713       // If we have load/store pair instructions and we only have two values,
15714       // don't bother merging.
15715       unsigned RequiredAlignment;
15716       if (LoadNodes.size() == 2 &&
15717           TLI.hasPairedLoad(MemVT, RequiredAlignment) &&
15718           StoreNodes[0].MemNode->getAlignment() >= RequiredAlignment) {
15719         StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + 2);
15720         LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + 2);
15721         break;
15722       }
15723       LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
15724       unsigned FirstStoreAS = FirstInChain->getAddressSpace();
15725       unsigned FirstStoreAlign = FirstInChain->getAlignment();
15726       LoadSDNode *FirstLoad = cast<LoadSDNode>(LoadNodes[0].MemNode);
15727       unsigned FirstLoadAlign = FirstLoad->getAlignment();
15728 
15729       // Scan the memory operations on the chain and find the first
15730       // non-consecutive load memory address. These variables hold the index in
15731       // the store node array.
15732 
15733       unsigned LastConsecutiveLoad = 1;
15734 
15735       // This variable refers to the size and not index in the array.
15736       unsigned LastLegalVectorType = 1;
15737       unsigned LastLegalIntegerType = 1;
15738       bool isDereferenceable = true;
15739       bool DoIntegerTruncate = false;
15740       StartAddress = LoadNodes[0].OffsetFromBase;
15741       SDValue FirstChain = FirstLoad->getChain();
15742       for (unsigned i = 1; i < LoadNodes.size(); ++i) {
15743         // All loads must share the same chain.
15744         if (LoadNodes[i].MemNode->getChain() != FirstChain)
15745           break;
15746 
15747         int64_t CurrAddress = LoadNodes[i].OffsetFromBase;
15748         if (CurrAddress - StartAddress != (ElementSizeBytes * i))
15749           break;
15750         LastConsecutiveLoad = i;
15751 
15752         if (isDereferenceable && !LoadNodes[i].MemNode->isDereferenceable())
15753           isDereferenceable = false;
15754 
15755         // Find a legal type for the vector store.
15756         unsigned Elts = (i + 1) * NumMemElts;
15757         EVT StoreTy = EVT::getVectorVT(Context, MemVT.getScalarType(), Elts);
15758 
15759         // Break early when size is too large to be legal.
15760         if (StoreTy.getSizeInBits() > MaximumLegalStoreInBits)
15761           break;
15762 
15763         bool IsFastSt, IsFastLd;
15764         if (TLI.isTypeLegal(StoreTy) &&
15765             TLI.canMergeStoresTo(FirstStoreAS, StoreTy, DAG) &&
15766             TLI.allowsMemoryAccess(Context, DL, StoreTy,
15767                                    *FirstInChain->getMemOperand(), &IsFastSt) &&
15768             IsFastSt &&
15769             TLI.allowsMemoryAccess(Context, DL, StoreTy,
15770                                    *FirstLoad->getMemOperand(), &IsFastLd) &&
15771             IsFastLd) {
15772           LastLegalVectorType = i + 1;
15773         }
15774 
15775         // Find a legal type for the integer store.
15776         unsigned SizeInBits = (i + 1) * ElementSizeBytes * 8;
15777         StoreTy = EVT::getIntegerVT(Context, SizeInBits);
15778         if (TLI.isTypeLegal(StoreTy) &&
15779             TLI.canMergeStoresTo(FirstStoreAS, StoreTy, DAG) &&
15780             TLI.allowsMemoryAccess(Context, DL, StoreTy,
15781                                    *FirstInChain->getMemOperand(), &IsFastSt) &&
15782             IsFastSt &&
15783             TLI.allowsMemoryAccess(Context, DL, StoreTy,
15784                                    *FirstLoad->getMemOperand(), &IsFastLd) &&
15785             IsFastLd) {
15786           LastLegalIntegerType = i + 1;
15787           DoIntegerTruncate = false;
15788           // Or check whether a truncstore and extload is legal.
15789         } else if (TLI.getTypeAction(Context, StoreTy) ==
15790                    TargetLowering::TypePromoteInteger) {
15791           EVT LegalizedStoredValTy = TLI.getTypeToTransformTo(Context, StoreTy);
15792           if (TLI.isTruncStoreLegal(LegalizedStoredValTy, StoreTy) &&
15793               TLI.canMergeStoresTo(FirstStoreAS, LegalizedStoredValTy, DAG) &&
15794               TLI.isLoadExtLegal(ISD::ZEXTLOAD, LegalizedStoredValTy,
15795                                  StoreTy) &&
15796               TLI.isLoadExtLegal(ISD::SEXTLOAD, LegalizedStoredValTy,
15797                                  StoreTy) &&
15798               TLI.isLoadExtLegal(ISD::EXTLOAD, LegalizedStoredValTy, StoreTy) &&
15799               TLI.allowsMemoryAccess(Context, DL, StoreTy,
15800                                      *FirstInChain->getMemOperand(),
15801                                      &IsFastSt) &&
15802               IsFastSt &&
15803               TLI.allowsMemoryAccess(Context, DL, StoreTy,
15804                                      *FirstLoad->getMemOperand(), &IsFastLd) &&
15805               IsFastLd) {
15806             LastLegalIntegerType = i + 1;
15807             DoIntegerTruncate = true;
15808           }
15809         }
15810       }
15811 
15812       // Only use vector types if the vector type is larger than the integer
15813       // type. If they are the same, use integers.
15814       bool UseVectorTy =
15815           LastLegalVectorType > LastLegalIntegerType && !NoVectors;
15816       unsigned LastLegalType =
15817           std::max(LastLegalVectorType, LastLegalIntegerType);
15818 
15819       // We add +1 here because the LastXXX variables refer to location while
15820       // the NumElem refers to array/index size.
15821       unsigned NumElem =
15822           std::min(NumConsecutiveStores, LastConsecutiveLoad + 1);
15823       NumElem = std::min(LastLegalType, NumElem);
15824 
15825       if (NumElem < 2) {
15826         // We know that candidate stores are in order and of correct
15827         // shape. While there is no mergeable sequence from the
15828         // beginning one may start later in the sequence. The only
15829         // reason a merge of size N could have failed where another of
15830         // the same size would not have is if the alignment or either
15831         // the load or store has improved. Drop as many candidates as we
15832         // can here.
15833         unsigned NumSkip = 1;
15834         while ((NumSkip < LoadNodes.size()) &&
15835                (LoadNodes[NumSkip].MemNode->getAlignment() <= FirstLoadAlign) &&
15836                (StoreNodes[NumSkip].MemNode->getAlignment() <= FirstStoreAlign))
15837           NumSkip++;
15838         StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumSkip);
15839         LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + NumSkip);
15840         NumConsecutiveStores -= NumSkip;
15841         continue;
15842       }
15843 
15844       // Check that we can merge these candidates without causing a cycle.
15845       if (!checkMergeStoreCandidatesForDependencies(StoreNodes, NumElem,
15846                                                     RootNode)) {
15847         StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem);
15848         LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + NumElem);
15849         NumConsecutiveStores -= NumElem;
15850         continue;
15851       }
15852 
15853       // Find if it is better to use vectors or integers to load and store
15854       // to memory.
15855       EVT JointMemOpVT;
15856       if (UseVectorTy) {
15857         // Find a legal type for the vector store.
15858         unsigned Elts = NumElem * NumMemElts;
15859         JointMemOpVT = EVT::getVectorVT(Context, MemVT.getScalarType(), Elts);
15860       } else {
15861         unsigned SizeInBits = NumElem * ElementSizeBytes * 8;
15862         JointMemOpVT = EVT::getIntegerVT(Context, SizeInBits);
15863       }
15864 
15865       SDLoc LoadDL(LoadNodes[0].MemNode);
15866       SDLoc StoreDL(StoreNodes[0].MemNode);
15867 
15868       // The merged loads are required to have the same incoming chain, so
15869       // using the first's chain is acceptable.
15870 
15871       SDValue NewStoreChain = getMergeStoreChains(StoreNodes, NumElem);
15872       AddToWorklist(NewStoreChain.getNode());
15873 
15874       MachineMemOperand::Flags LdMMOFlags =
15875           isDereferenceable ? MachineMemOperand::MODereferenceable
15876                             : MachineMemOperand::MONone;
15877       if (IsNonTemporalLoad)
15878         LdMMOFlags |= MachineMemOperand::MONonTemporal;
15879 
15880       MachineMemOperand::Flags StMMOFlags =
15881           IsNonTemporalStore ? MachineMemOperand::MONonTemporal
15882                              : MachineMemOperand::MONone;
15883 
15884       SDValue NewLoad, NewStore;
15885       if (UseVectorTy || !DoIntegerTruncate) {
15886         NewLoad =
15887             DAG.getLoad(JointMemOpVT, LoadDL, FirstLoad->getChain(),
15888                         FirstLoad->getBasePtr(), FirstLoad->getPointerInfo(),
15889                         FirstLoadAlign, LdMMOFlags);
15890         NewStore = DAG.getStore(
15891             NewStoreChain, StoreDL, NewLoad, FirstInChain->getBasePtr(),
15892             FirstInChain->getPointerInfo(), FirstStoreAlign, StMMOFlags);
15893       } else { // This must be the truncstore/extload case
15894         EVT ExtendedTy =
15895             TLI.getTypeToTransformTo(*DAG.getContext(), JointMemOpVT);
15896         NewLoad = DAG.getExtLoad(ISD::EXTLOAD, LoadDL, ExtendedTy,
15897                                  FirstLoad->getChain(), FirstLoad->getBasePtr(),
15898                                  FirstLoad->getPointerInfo(), JointMemOpVT,
15899                                  FirstLoadAlign, LdMMOFlags);
15900         NewStore = DAG.getTruncStore(NewStoreChain, StoreDL, NewLoad,
15901                                      FirstInChain->getBasePtr(),
15902                                      FirstInChain->getPointerInfo(),
15903                                      JointMemOpVT, FirstInChain->getAlignment(),
15904                                      FirstInChain->getMemOperand()->getFlags());
15905       }
15906 
15907       // Transfer chain users from old loads to the new load.
15908       for (unsigned i = 0; i < NumElem; ++i) {
15909         LoadSDNode *Ld = cast<LoadSDNode>(LoadNodes[i].MemNode);
15910         DAG.ReplaceAllUsesOfValueWith(SDValue(Ld, 1),
15911                                       SDValue(NewLoad.getNode(), 1));
15912       }
15913 
15914       // Replace the all stores with the new store. Recursively remove
15915       // corresponding value if its no longer used.
15916       for (unsigned i = 0; i < NumElem; ++i) {
15917         SDValue Val = StoreNodes[i].MemNode->getOperand(1);
15918         CombineTo(StoreNodes[i].MemNode, NewStore);
15919         if (Val.getNode()->use_empty())
15920           recursivelyDeleteUnusedNodes(Val.getNode());
15921       }
15922 
15923       RV = true;
15924       StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem);
15925       LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + NumElem);
15926       NumConsecutiveStores -= NumElem;
15927     }
15928   }
15929   return RV;
15930 }
15931 
15932 SDValue DAGCombiner::replaceStoreChain(StoreSDNode *ST, SDValue BetterChain) {
15933   SDLoc SL(ST);
15934   SDValue ReplStore;
15935 
15936   // Replace the chain to avoid dependency.
15937   if (ST->isTruncatingStore()) {
15938     ReplStore = DAG.getTruncStore(BetterChain, SL, ST->getValue(),
15939                                   ST->getBasePtr(), ST->getMemoryVT(),
15940                                   ST->getMemOperand());
15941   } else {
15942     ReplStore = DAG.getStore(BetterChain, SL, ST->getValue(), ST->getBasePtr(),
15943                              ST->getMemOperand());
15944   }
15945 
15946   // Create token to keep both nodes around.
15947   SDValue Token = DAG.getNode(ISD::TokenFactor, SL,
15948                               MVT::Other, ST->getChain(), ReplStore);
15949 
15950   // Make sure the new and old chains are cleaned up.
15951   AddToWorklist(Token.getNode());
15952 
15953   // Don't add users to work list.
15954   return CombineTo(ST, Token, false);
15955 }
15956 
15957 SDValue DAGCombiner::replaceStoreOfFPConstant(StoreSDNode *ST) {
15958   SDValue Value = ST->getValue();
15959   if (Value.getOpcode() == ISD::TargetConstantFP)
15960     return SDValue();
15961 
15962   SDLoc DL(ST);
15963 
15964   SDValue Chain = ST->getChain();
15965   SDValue Ptr = ST->getBasePtr();
15966 
15967   const ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Value);
15968 
15969   // NOTE: If the original store is volatile, this transform must not increase
15970   // the number of stores.  For example, on x86-32 an f64 can be stored in one
15971   // processor operation but an i64 (which is not legal) requires two.  So the
15972   // transform should not be done in this case.
15973 
15974   SDValue Tmp;
15975   switch (CFP->getSimpleValueType(0).SimpleTy) {
15976   default:
15977     llvm_unreachable("Unknown FP type");
15978   case MVT::f16:    // We don't do this for these yet.
15979   case MVT::f80:
15980   case MVT::f128:
15981   case MVT::ppcf128:
15982     return SDValue();
15983   case MVT::f32:
15984     if ((isTypeLegal(MVT::i32) && !LegalOperations && !ST->isVolatile()) ||
15985         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
15986       ;
15987       Tmp = DAG.getConstant((uint32_t)CFP->getValueAPF().
15988                             bitcastToAPInt().getZExtValue(), SDLoc(CFP),
15989                             MVT::i32);
15990       return DAG.getStore(Chain, DL, Tmp, Ptr, ST->getMemOperand());
15991     }
15992 
15993     return SDValue();
15994   case MVT::f64:
15995     if ((TLI.isTypeLegal(MVT::i64) && !LegalOperations &&
15996          !ST->isVolatile()) ||
15997         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i64)) {
15998       ;
15999       Tmp = DAG.getConstant(CFP->getValueAPF().bitcastToAPInt().
16000                             getZExtValue(), SDLoc(CFP), MVT::i64);
16001       return DAG.getStore(Chain, DL, Tmp,
16002                           Ptr, ST->getMemOperand());
16003     }
16004 
16005     if (!ST->isVolatile() &&
16006         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
16007       // Many FP stores are not made apparent until after legalize, e.g. for
16008       // argument passing.  Since this is so common, custom legalize the
16009       // 64-bit integer store into two 32-bit stores.
16010       uint64_t Val = CFP->getValueAPF().bitcastToAPInt().getZExtValue();
16011       SDValue Lo = DAG.getConstant(Val & 0xFFFFFFFF, SDLoc(CFP), MVT::i32);
16012       SDValue Hi = DAG.getConstant(Val >> 32, SDLoc(CFP), MVT::i32);
16013       if (DAG.getDataLayout().isBigEndian())
16014         std::swap(Lo, Hi);
16015 
16016       unsigned Alignment = ST->getAlignment();
16017       MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
16018       AAMDNodes AAInfo = ST->getAAInfo();
16019 
16020       SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
16021                                  ST->getAlignment(), MMOFlags, AAInfo);
16022       Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
16023                         DAG.getConstant(4, DL, Ptr.getValueType()));
16024       Alignment = MinAlign(Alignment, 4U);
16025       SDValue St1 = DAG.getStore(Chain, DL, Hi, Ptr,
16026                                  ST->getPointerInfo().getWithOffset(4),
16027                                  Alignment, MMOFlags, AAInfo);
16028       return DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
16029                          St0, St1);
16030     }
16031 
16032     return SDValue();
16033   }
16034 }
16035 
16036 SDValue DAGCombiner::visitSTORE(SDNode *N) {
16037   StoreSDNode *ST  = cast<StoreSDNode>(N);
16038   SDValue Chain = ST->getChain();
16039   SDValue Value = ST->getValue();
16040   SDValue Ptr   = ST->getBasePtr();
16041 
16042   // If this is a store of a bit convert, store the input value if the
16043   // resultant store does not need a higher alignment than the original.
16044   if (Value.getOpcode() == ISD::BITCAST && !ST->isTruncatingStore() &&
16045       ST->isUnindexed()) {
16046     EVT SVT = Value.getOperand(0).getValueType();
16047     // If the store is volatile, we only want to change the store type if the
16048     // resulting store is legal. Otherwise we might increase the number of
16049     // memory accesses. We don't care if the original type was legal or not
16050     // as we assume software couldn't rely on the number of accesses of an
16051     // illegal type.
16052     if (((!LegalOperations && !ST->isVolatile()) ||
16053          TLI.isOperationLegal(ISD::STORE, SVT)) &&
16054         TLI.isStoreBitCastBeneficial(Value.getValueType(), SVT)) {
16055       bool Fast = false;
16056       if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), SVT,
16057                                  *ST->getMemOperand(), &Fast) &&
16058           Fast) {
16059         return DAG.getStore(Chain, SDLoc(N), Value.getOperand(0), Ptr,
16060                             ST->getPointerInfo(), ST->getAlignment(),
16061                             ST->getMemOperand()->getFlags(), ST->getAAInfo());
16062       }
16063     }
16064   }
16065 
16066   // Turn 'store undef, Ptr' -> nothing.
16067   if (Value.isUndef() && ST->isUnindexed())
16068     return Chain;
16069 
16070   // Try to infer better alignment information than the store already has.
16071   if (OptLevel != CodeGenOpt::None && ST->isUnindexed()) {
16072     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
16073       if (Align > ST->getAlignment() && ST->getSrcValueOffset() % Align == 0) {
16074         SDValue NewStore =
16075             DAG.getTruncStore(Chain, SDLoc(N), Value, Ptr, ST->getPointerInfo(),
16076                               ST->getMemoryVT(), Align,
16077                               ST->getMemOperand()->getFlags(), ST->getAAInfo());
16078         // NewStore will always be N as we are only refining the alignment
16079         assert(NewStore.getNode() == N);
16080         (void)NewStore;
16081       }
16082     }
16083   }
16084 
16085   // Try transforming a pair floating point load / store ops to integer
16086   // load / store ops.
16087   if (SDValue NewST = TransformFPLoadStorePair(N))
16088     return NewST;
16089 
16090   // Try transforming several stores into STORE (BSWAP).
16091   if (SDValue Store = MatchStoreCombine(ST))
16092     return Store;
16093 
16094   if (ST->isUnindexed()) {
16095     // Walk up chain skipping non-aliasing memory nodes, on this store and any
16096     // adjacent stores.
16097     if (findBetterNeighborChains(ST)) {
16098       // replaceStoreChain uses CombineTo, which handled all of the worklist
16099       // manipulation. Return the original node to not do anything else.
16100       return SDValue(ST, 0);
16101     }
16102     Chain = ST->getChain();
16103   }
16104 
16105   // FIXME: is there such a thing as a truncating indexed store?
16106   if (ST->isTruncatingStore() && ST->isUnindexed() &&
16107       Value.getValueType().isInteger() &&
16108       (!isa<ConstantSDNode>(Value) ||
16109        !cast<ConstantSDNode>(Value)->isOpaque())) {
16110     APInt TruncDemandedBits =
16111         APInt::getLowBitsSet(Value.getScalarValueSizeInBits(),
16112                              ST->getMemoryVT().getScalarSizeInBits());
16113 
16114     // See if we can simplify the input to this truncstore with knowledge that
16115     // only the low bits are being used.  For example:
16116     // "truncstore (or (shl x, 8), y), i8"  -> "truncstore y, i8"
16117     SDValue Shorter = DAG.GetDemandedBits(Value, TruncDemandedBits);
16118     AddToWorklist(Value.getNode());
16119     if (Shorter)
16120       return DAG.getTruncStore(Chain, SDLoc(N), Shorter, Ptr, ST->getMemoryVT(),
16121                                ST->getMemOperand());
16122 
16123     // Otherwise, see if we can simplify the operation with
16124     // SimplifyDemandedBits, which only works if the value has a single use.
16125     if (SimplifyDemandedBits(Value, TruncDemandedBits)) {
16126       // Re-visit the store if anything changed and the store hasn't been merged
16127       // with another node (N is deleted) SimplifyDemandedBits will add Value's
16128       // node back to the worklist if necessary, but we also need to re-visit
16129       // the Store node itself.
16130       if (N->getOpcode() != ISD::DELETED_NODE)
16131         AddToWorklist(N);
16132       return SDValue(N, 0);
16133     }
16134   }
16135 
16136   // If this is a load followed by a store to the same location, then the store
16137   // is dead/noop.
16138   if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Value)) {
16139     if (Ld->getBasePtr() == Ptr && ST->getMemoryVT() == Ld->getMemoryVT() &&
16140         ST->isUnindexed() && !ST->isVolatile() &&
16141         // There can't be any side effects between the load and store, such as
16142         // a call or store.
16143         Chain.reachesChainWithoutSideEffects(SDValue(Ld, 1))) {
16144       // The store is dead, remove it.
16145       return Chain;
16146     }
16147   }
16148 
16149   if (StoreSDNode *ST1 = dyn_cast<StoreSDNode>(Chain)) {
16150     if (ST->isUnindexed() && !ST->isVolatile() && ST1->isUnindexed() &&
16151         !ST1->isVolatile()) {
16152       if (ST1->getBasePtr() == Ptr && ST1->getValue() == Value &&
16153           ST->getMemoryVT() == ST1->getMemoryVT()) {
16154         // If this is a store followed by a store with the same value to the
16155         // same location, then the store is dead/noop.
16156         return Chain;
16157       }
16158 
16159       if (OptLevel != CodeGenOpt::None && ST1->hasOneUse() &&
16160           !ST1->getBasePtr().isUndef()) {
16161         const BaseIndexOffset STBase = BaseIndexOffset::match(ST, DAG);
16162         const BaseIndexOffset ChainBase = BaseIndexOffset::match(ST1, DAG);
16163         unsigned STBitSize = ST->getMemoryVT().getSizeInBits();
16164         unsigned ChainBitSize = ST1->getMemoryVT().getSizeInBits();
16165         // If this is a store who's preceding store to a subset of the current
16166         // location and no one other node is chained to that store we can
16167         // effectively drop the store. Do not remove stores to undef as they may
16168         // be used as data sinks.
16169         if (STBase.contains(DAG, STBitSize, ChainBase, ChainBitSize)) {
16170           CombineTo(ST1, ST1->getChain());
16171           return SDValue();
16172         }
16173 
16174         // If ST stores to a subset of preceding store's write set, we may be
16175         // able to fold ST's value into the preceding stored value. As we know
16176         // the other uses of ST1's chain are unconcerned with ST, this folding
16177         // will not affect those nodes.
16178         int64_t BitOffset;
16179         if (ChainBase.contains(DAG, ChainBitSize, STBase, STBitSize,
16180                                BitOffset)) {
16181           SDValue ChainValue = ST1->getValue();
16182           if (auto *C1 = dyn_cast<ConstantSDNode>(ChainValue)) {
16183             if (auto *C = dyn_cast<ConstantSDNode>(Value)) {
16184               APInt Val = C1->getAPIntValue();
16185               APInt InsertVal = C->getAPIntValue().zextOrTrunc(STBitSize);
16186               // FIXME: Handle Big-endian mode.
16187               if (!DAG.getDataLayout().isBigEndian()) {
16188                 Val.insertBits(InsertVal, BitOffset);
16189                 SDValue NewSDVal =
16190                     DAG.getConstant(Val, SDLoc(C), ChainValue.getValueType(),
16191                                     C1->isTargetOpcode(), C1->isOpaque());
16192                 SDNode *NewST1 = DAG.UpdateNodeOperands(
16193                     ST1, ST1->getChain(), NewSDVal, ST1->getOperand(2),
16194                     ST1->getOperand(3));
16195                 return CombineTo(ST, SDValue(NewST1, 0));
16196               }
16197             }
16198           }
16199         } // End ST subset of ST1 case.
16200       }
16201     }
16202   }
16203 
16204   // If this is an FP_ROUND or TRUNC followed by a store, fold this into a
16205   // truncating store.  We can do this even if this is already a truncstore.
16206   if ((Value.getOpcode() == ISD::FP_ROUND || Value.getOpcode() == ISD::TRUNCATE)
16207       && Value.getNode()->hasOneUse() && ST->isUnindexed() &&
16208       TLI.isTruncStoreLegal(Value.getOperand(0).getValueType(),
16209                             ST->getMemoryVT())) {
16210     return DAG.getTruncStore(Chain, SDLoc(N), Value.getOperand(0),
16211                              Ptr, ST->getMemoryVT(), ST->getMemOperand());
16212   }
16213 
16214   // Always perform this optimization before types are legal. If the target
16215   // prefers, also try this after legalization to catch stores that were created
16216   // by intrinsics or other nodes.
16217   if (!LegalTypes || (TLI.mergeStoresAfterLegalization(ST->getMemoryVT()))) {
16218     while (true) {
16219       // There can be multiple store sequences on the same chain.
16220       // Keep trying to merge store sequences until we are unable to do so
16221       // or until we merge the last store on the chain.
16222       bool Changed = MergeConsecutiveStores(ST);
16223       if (!Changed) break;
16224       // Return N as merge only uses CombineTo and no worklist clean
16225       // up is necessary.
16226       if (N->getOpcode() == ISD::DELETED_NODE || !isa<StoreSDNode>(N))
16227         return SDValue(N, 0);
16228     }
16229   }
16230 
16231   // Try transforming N to an indexed store.
16232   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
16233     return SDValue(N, 0);
16234 
16235   // Turn 'store float 1.0, Ptr' -> 'store int 0x12345678, Ptr'
16236   //
16237   // Make sure to do this only after attempting to merge stores in order to
16238   //  avoid changing the types of some subset of stores due to visit order,
16239   //  preventing their merging.
16240   if (isa<ConstantFPSDNode>(ST->getValue())) {
16241     if (SDValue NewSt = replaceStoreOfFPConstant(ST))
16242       return NewSt;
16243   }
16244 
16245   if (SDValue NewSt = splitMergedValStore(ST))
16246     return NewSt;
16247 
16248   return ReduceLoadOpStoreWidth(N);
16249 }
16250 
16251 SDValue DAGCombiner::visitLIFETIME_END(SDNode *N) {
16252   const auto *LifetimeEnd = cast<LifetimeSDNode>(N);
16253   if (!LifetimeEnd->hasOffset())
16254     return SDValue();
16255 
16256   const BaseIndexOffset LifetimeEndBase(N->getOperand(1), SDValue(),
16257                                         LifetimeEnd->getOffset(), false);
16258 
16259   // We walk up the chains to find stores.
16260   SmallVector<SDValue, 8> Chains = {N->getOperand(0)};
16261   while (!Chains.empty()) {
16262     SDValue Chain = Chains.back();
16263     Chains.pop_back();
16264     if (!Chain.hasOneUse())
16265       continue;
16266     switch (Chain.getOpcode()) {
16267     case ISD::TokenFactor:
16268       for (unsigned Nops = Chain.getNumOperands(); Nops;)
16269         Chains.push_back(Chain.getOperand(--Nops));
16270       break;
16271     case ISD::LIFETIME_START:
16272     case ISD::LIFETIME_END:
16273       // We can forward past any lifetime start/end that can be proven not to
16274       // alias the node.
16275       if (!isAlias(Chain.getNode(), N))
16276         Chains.push_back(Chain.getOperand(0));
16277       break;
16278     case ISD::STORE: {
16279       StoreSDNode *ST = dyn_cast<StoreSDNode>(Chain);
16280       if (ST->isVolatile() || ST->isIndexed())
16281         continue;
16282       const BaseIndexOffset StoreBase = BaseIndexOffset::match(ST, DAG);
16283       // If we store purely within object bounds just before its lifetime ends,
16284       // we can remove the store.
16285       if (LifetimeEndBase.contains(DAG, LifetimeEnd->getSize() * 8, StoreBase,
16286                                    ST->getMemoryVT().getStoreSizeInBits())) {
16287         LLVM_DEBUG(dbgs() << "\nRemoving store:"; StoreBase.dump();
16288                    dbgs() << "\nwithin LIFETIME_END of : ";
16289                    LifetimeEndBase.dump(); dbgs() << "\n");
16290         CombineTo(ST, ST->getChain());
16291         return SDValue(N, 0);
16292       }
16293     }
16294     }
16295   }
16296   return SDValue();
16297 }
16298 
16299 /// For the instruction sequence of store below, F and I values
16300 /// are bundled together as an i64 value before being stored into memory.
16301 /// Sometimes it is more efficent to generate separate stores for F and I,
16302 /// which can remove the bitwise instructions or sink them to colder places.
16303 ///
16304 ///   (store (or (zext (bitcast F to i32) to i64),
16305 ///              (shl (zext I to i64), 32)), addr)  -->
16306 ///   (store F, addr) and (store I, addr+4)
16307 ///
16308 /// Similarly, splitting for other merged store can also be beneficial, like:
16309 /// For pair of {i32, i32}, i64 store --> two i32 stores.
16310 /// For pair of {i32, i16}, i64 store --> two i32 stores.
16311 /// For pair of {i16, i16}, i32 store --> two i16 stores.
16312 /// For pair of {i16, i8},  i32 store --> two i16 stores.
16313 /// For pair of {i8, i8},   i16 store --> two i8 stores.
16314 ///
16315 /// We allow each target to determine specifically which kind of splitting is
16316 /// supported.
16317 ///
16318 /// The store patterns are commonly seen from the simple code snippet below
16319 /// if only std::make_pair(...) is sroa transformed before inlined into hoo.
16320 ///   void goo(const std::pair<int, float> &);
16321 ///   hoo() {
16322 ///     ...
16323 ///     goo(std::make_pair(tmp, ftmp));
16324 ///     ...
16325 ///   }
16326 ///
16327 SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
16328   if (OptLevel == CodeGenOpt::None)
16329     return SDValue();
16330 
16331   SDValue Val = ST->getValue();
16332   SDLoc DL(ST);
16333 
16334   // Match OR operand.
16335   if (!Val.getValueType().isScalarInteger() || Val.getOpcode() != ISD::OR)
16336     return SDValue();
16337 
16338   // Match SHL operand and get Lower and Higher parts of Val.
16339   SDValue Op1 = Val.getOperand(0);
16340   SDValue Op2 = Val.getOperand(1);
16341   SDValue Lo, Hi;
16342   if (Op1.getOpcode() != ISD::SHL) {
16343     std::swap(Op1, Op2);
16344     if (Op1.getOpcode() != ISD::SHL)
16345       return SDValue();
16346   }
16347   Lo = Op2;
16348   Hi = Op1.getOperand(0);
16349   if (!Op1.hasOneUse())
16350     return SDValue();
16351 
16352   // Match shift amount to HalfValBitSize.
16353   unsigned HalfValBitSize = Val.getValueSizeInBits() / 2;
16354   ConstantSDNode *ShAmt = dyn_cast<ConstantSDNode>(Op1.getOperand(1));
16355   if (!ShAmt || ShAmt->getAPIntValue() != HalfValBitSize)
16356     return SDValue();
16357 
16358   // Lo and Hi are zero-extended from int with size less equal than 32
16359   // to i64.
16360   if (Lo.getOpcode() != ISD::ZERO_EXTEND || !Lo.hasOneUse() ||
16361       !Lo.getOperand(0).getValueType().isScalarInteger() ||
16362       Lo.getOperand(0).getValueSizeInBits() > HalfValBitSize ||
16363       Hi.getOpcode() != ISD::ZERO_EXTEND || !Hi.hasOneUse() ||
16364       !Hi.getOperand(0).getValueType().isScalarInteger() ||
16365       Hi.getOperand(0).getValueSizeInBits() > HalfValBitSize)
16366     return SDValue();
16367 
16368   // Use the EVT of low and high parts before bitcast as the input
16369   // of target query.
16370   EVT LowTy = (Lo.getOperand(0).getOpcode() == ISD::BITCAST)
16371                   ? Lo.getOperand(0).getValueType()
16372                   : Lo.getValueType();
16373   EVT HighTy = (Hi.getOperand(0).getOpcode() == ISD::BITCAST)
16374                    ? Hi.getOperand(0).getValueType()
16375                    : Hi.getValueType();
16376   if (!TLI.isMultiStoresCheaperThanBitsMerge(LowTy, HighTy))
16377     return SDValue();
16378 
16379   // Start to split store.
16380   unsigned Alignment = ST->getAlignment();
16381   MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
16382   AAMDNodes AAInfo = ST->getAAInfo();
16383 
16384   // Change the sizes of Lo and Hi's value types to HalfValBitSize.
16385   EVT VT = EVT::getIntegerVT(*DAG.getContext(), HalfValBitSize);
16386   Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Lo.getOperand(0));
16387   Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Hi.getOperand(0));
16388 
16389   SDValue Chain = ST->getChain();
16390   SDValue Ptr = ST->getBasePtr();
16391   // Lower value store.
16392   SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
16393                              ST->getAlignment(), MMOFlags, AAInfo);
16394   Ptr =
16395       DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
16396                   DAG.getConstant(HalfValBitSize / 8, DL, Ptr.getValueType()));
16397   // Higher value store.
16398   SDValue St1 =
16399       DAG.getStore(St0, DL, Hi, Ptr,
16400                    ST->getPointerInfo().getWithOffset(HalfValBitSize / 8),
16401                    Alignment / 2, MMOFlags, AAInfo);
16402   return St1;
16403 }
16404 
16405 /// Convert a disguised subvector insertion into a shuffle:
16406 /// insert_vector_elt V, (bitcast X from vector type), IdxC -->
16407 /// bitcast(shuffle (bitcast V), (extended X), Mask)
16408 /// Note: We do not use an insert_subvector node because that requires a legal
16409 /// subvector type.
16410 SDValue DAGCombiner::combineInsertEltToShuffle(SDNode *N, unsigned InsIndex) {
16411   SDValue InsertVal = N->getOperand(1);
16412   if (InsertVal.getOpcode() != ISD::BITCAST || !InsertVal.hasOneUse() ||
16413       !InsertVal.getOperand(0).getValueType().isVector())
16414     return SDValue();
16415 
16416   SDValue SubVec = InsertVal.getOperand(0);
16417   SDValue DestVec = N->getOperand(0);
16418   EVT SubVecVT = SubVec.getValueType();
16419   EVT VT = DestVec.getValueType();
16420   unsigned NumSrcElts = SubVecVT.getVectorNumElements();
16421   unsigned ExtendRatio = VT.getSizeInBits() / SubVecVT.getSizeInBits();
16422   unsigned NumMaskVals = ExtendRatio * NumSrcElts;
16423 
16424   // Step 1: Create a shuffle mask that implements this insert operation. The
16425   // vector that we are inserting into will be operand 0 of the shuffle, so
16426   // those elements are just 'i'. The inserted subvector is in the first
16427   // positions of operand 1 of the shuffle. Example:
16428   // insert v4i32 V, (v2i16 X), 2 --> shuffle v8i16 V', X', {0,1,2,3,8,9,6,7}
16429   SmallVector<int, 16> Mask(NumMaskVals);
16430   for (unsigned i = 0; i != NumMaskVals; ++i) {
16431     if (i / NumSrcElts == InsIndex)
16432       Mask[i] = (i % NumSrcElts) + NumMaskVals;
16433     else
16434       Mask[i] = i;
16435   }
16436 
16437   // Bail out if the target can not handle the shuffle we want to create.
16438   EVT SubVecEltVT = SubVecVT.getVectorElementType();
16439   EVT ShufVT = EVT::getVectorVT(*DAG.getContext(), SubVecEltVT, NumMaskVals);
16440   if (!TLI.isShuffleMaskLegal(Mask, ShufVT))
16441     return SDValue();
16442 
16443   // Step 2: Create a wide vector from the inserted source vector by appending
16444   // undefined elements. This is the same size as our destination vector.
16445   SDLoc DL(N);
16446   SmallVector<SDValue, 8> ConcatOps(ExtendRatio, DAG.getUNDEF(SubVecVT));
16447   ConcatOps[0] = SubVec;
16448   SDValue PaddedSubV = DAG.getNode(ISD::CONCAT_VECTORS, DL, ShufVT, ConcatOps);
16449 
16450   // Step 3: Shuffle in the padded subvector.
16451   SDValue DestVecBC = DAG.getBitcast(ShufVT, DestVec);
16452   SDValue Shuf = DAG.getVectorShuffle(ShufVT, DL, DestVecBC, PaddedSubV, Mask);
16453   AddToWorklist(PaddedSubV.getNode());
16454   AddToWorklist(DestVecBC.getNode());
16455   AddToWorklist(Shuf.getNode());
16456   return DAG.getBitcast(VT, Shuf);
16457 }
16458 
16459 SDValue DAGCombiner::visitINSERT_VECTOR_ELT(SDNode *N) {
16460   SDValue InVec = N->getOperand(0);
16461   SDValue InVal = N->getOperand(1);
16462   SDValue EltNo = N->getOperand(2);
16463   SDLoc DL(N);
16464 
16465   // If the inserted element is an UNDEF, just use the input vector.
16466   if (InVal.isUndef())
16467     return InVec;
16468 
16469   EVT VT = InVec.getValueType();
16470   unsigned NumElts = VT.getVectorNumElements();
16471 
16472   // Remove redundant insertions:
16473   // (insert_vector_elt x (extract_vector_elt x idx) idx) -> x
16474   if (InVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
16475       InVec == InVal.getOperand(0) && EltNo == InVal.getOperand(1))
16476     return InVec;
16477 
16478   auto *IndexC = dyn_cast<ConstantSDNode>(EltNo);
16479   if (!IndexC) {
16480     // If this is variable insert to undef vector, it might be better to splat:
16481     // inselt undef, InVal, EltNo --> build_vector < InVal, InVal, ... >
16482     if (InVec.isUndef() && TLI.shouldSplatInsEltVarIndex(VT)) {
16483       SmallVector<SDValue, 8> Ops(NumElts, InVal);
16484       return DAG.getBuildVector(VT, DL, Ops);
16485     }
16486     return SDValue();
16487   }
16488 
16489   // We must know which element is being inserted for folds below here.
16490   unsigned Elt = IndexC->getZExtValue();
16491   if (SDValue Shuf = combineInsertEltToShuffle(N, Elt))
16492     return Shuf;
16493 
16494   // Canonicalize insert_vector_elt dag nodes.
16495   // Example:
16496   // (insert_vector_elt (insert_vector_elt A, Idx0), Idx1)
16497   // -> (insert_vector_elt (insert_vector_elt A, Idx1), Idx0)
16498   //
16499   // Do this only if the child insert_vector node has one use; also
16500   // do this only if indices are both constants and Idx1 < Idx0.
16501   if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT && InVec.hasOneUse()
16502       && isa<ConstantSDNode>(InVec.getOperand(2))) {
16503     unsigned OtherElt = InVec.getConstantOperandVal(2);
16504     if (Elt < OtherElt) {
16505       // Swap nodes.
16506       SDValue NewOp = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT,
16507                                   InVec.getOperand(0), InVal, EltNo);
16508       AddToWorklist(NewOp.getNode());
16509       return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(InVec.getNode()),
16510                          VT, NewOp, InVec.getOperand(1), InVec.getOperand(2));
16511     }
16512   }
16513 
16514   // If we can't generate a legal BUILD_VECTOR, exit
16515   if (LegalOperations && !TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
16516     return SDValue();
16517 
16518   // Check that the operand is a BUILD_VECTOR (or UNDEF, which can essentially
16519   // be converted to a BUILD_VECTOR).  Fill in the Ops vector with the
16520   // vector elements.
16521   SmallVector<SDValue, 8> Ops;
16522   // Do not combine these two vectors if the output vector will not replace
16523   // the input vector.
16524   if (InVec.getOpcode() == ISD::BUILD_VECTOR && InVec.hasOneUse()) {
16525     Ops.append(InVec.getNode()->op_begin(),
16526                InVec.getNode()->op_end());
16527   } else if (InVec.isUndef()) {
16528     Ops.append(NumElts, DAG.getUNDEF(InVal.getValueType()));
16529   } else {
16530     return SDValue();
16531   }
16532   assert(Ops.size() == NumElts && "Unexpected vector size");
16533 
16534   // Insert the element
16535   if (Elt < Ops.size()) {
16536     // All the operands of BUILD_VECTOR must have the same type;
16537     // we enforce that here.
16538     EVT OpVT = Ops[0].getValueType();
16539     Ops[Elt] = OpVT.isInteger() ? DAG.getAnyExtOrTrunc(InVal, DL, OpVT) : InVal;
16540   }
16541 
16542   // Return the new vector
16543   return DAG.getBuildVector(VT, DL, Ops);
16544 }
16545 
16546 SDValue DAGCombiner::scalarizeExtractedVectorLoad(SDNode *EVE, EVT InVecVT,
16547                                                   SDValue EltNo,
16548                                                   LoadSDNode *OriginalLoad) {
16549   assert(!OriginalLoad->isVolatile());
16550 
16551   EVT ResultVT = EVE->getValueType(0);
16552   EVT VecEltVT = InVecVT.getVectorElementType();
16553   unsigned Align = OriginalLoad->getAlignment();
16554   unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
16555       VecEltVT.getTypeForEVT(*DAG.getContext()));
16556 
16557   if (NewAlign > Align || !TLI.isOperationLegalOrCustom(ISD::LOAD, VecEltVT))
16558     return SDValue();
16559 
16560   ISD::LoadExtType ExtTy = ResultVT.bitsGT(VecEltVT) ?
16561     ISD::NON_EXTLOAD : ISD::EXTLOAD;
16562   if (!TLI.shouldReduceLoadWidth(OriginalLoad, ExtTy, VecEltVT))
16563     return SDValue();
16564 
16565   Align = NewAlign;
16566 
16567   SDValue NewPtr = OriginalLoad->getBasePtr();
16568   SDValue Offset;
16569   EVT PtrType = NewPtr.getValueType();
16570   MachinePointerInfo MPI;
16571   SDLoc DL(EVE);
16572   if (auto *ConstEltNo = dyn_cast<ConstantSDNode>(EltNo)) {
16573     int Elt = ConstEltNo->getZExtValue();
16574     unsigned PtrOff = VecEltVT.getSizeInBits() * Elt / 8;
16575     Offset = DAG.getConstant(PtrOff, DL, PtrType);
16576     MPI = OriginalLoad->getPointerInfo().getWithOffset(PtrOff);
16577   } else {
16578     Offset = DAG.getZExtOrTrunc(EltNo, DL, PtrType);
16579     Offset = DAG.getNode(
16580         ISD::MUL, DL, PtrType, Offset,
16581         DAG.getConstant(VecEltVT.getStoreSize(), DL, PtrType));
16582     // Discard the pointer info except the address space because the memory
16583     // operand can't represent this new access since the offset is variable.
16584     MPI = MachinePointerInfo(OriginalLoad->getPointerInfo().getAddrSpace());
16585   }
16586   NewPtr = DAG.getNode(ISD::ADD, DL, PtrType, NewPtr, Offset);
16587 
16588   // The replacement we need to do here is a little tricky: we need to
16589   // replace an extractelement of a load with a load.
16590   // Use ReplaceAllUsesOfValuesWith to do the replacement.
16591   // Note that this replacement assumes that the extractvalue is the only
16592   // use of the load; that's okay because we don't want to perform this
16593   // transformation in other cases anyway.
16594   SDValue Load;
16595   SDValue Chain;
16596   if (ResultVT.bitsGT(VecEltVT)) {
16597     // If the result type of vextract is wider than the load, then issue an
16598     // extending load instead.
16599     ISD::LoadExtType ExtType = TLI.isLoadExtLegal(ISD::ZEXTLOAD, ResultVT,
16600                                                   VecEltVT)
16601                                    ? ISD::ZEXTLOAD
16602                                    : ISD::EXTLOAD;
16603     Load = DAG.getExtLoad(ExtType, SDLoc(EVE), ResultVT,
16604                           OriginalLoad->getChain(), NewPtr, MPI, VecEltVT,
16605                           Align, OriginalLoad->getMemOperand()->getFlags(),
16606                           OriginalLoad->getAAInfo());
16607     Chain = Load.getValue(1);
16608   } else {
16609     Load = DAG.getLoad(VecEltVT, SDLoc(EVE), OriginalLoad->getChain(), NewPtr,
16610                        MPI, Align, OriginalLoad->getMemOperand()->getFlags(),
16611                        OriginalLoad->getAAInfo());
16612     Chain = Load.getValue(1);
16613     if (ResultVT.bitsLT(VecEltVT))
16614       Load = DAG.getNode(ISD::TRUNCATE, SDLoc(EVE), ResultVT, Load);
16615     else
16616       Load = DAG.getBitcast(ResultVT, Load);
16617   }
16618   WorklistRemover DeadNodes(*this);
16619   SDValue From[] = { SDValue(EVE, 0), SDValue(OriginalLoad, 1) };
16620   SDValue To[] = { Load, Chain };
16621   DAG.ReplaceAllUsesOfValuesWith(From, To, 2);
16622   // Since we're explicitly calling ReplaceAllUses, add the new node to the
16623   // worklist explicitly as well.
16624   AddToWorklist(Load.getNode());
16625   AddUsersToWorklist(Load.getNode()); // Add users too
16626   // Make sure to revisit this node to clean it up; it will usually be dead.
16627   AddToWorklist(EVE);
16628   ++OpsNarrowed;
16629   return SDValue(EVE, 0);
16630 }
16631 
16632 /// Transform a vector binary operation into a scalar binary operation by moving
16633 /// the math/logic after an extract element of a vector.
16634 static SDValue scalarizeExtractedBinop(SDNode *ExtElt, SelectionDAG &DAG,
16635                                        bool LegalOperations) {
16636   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
16637   SDValue Vec = ExtElt->getOperand(0);
16638   SDValue Index = ExtElt->getOperand(1);
16639   auto *IndexC = dyn_cast<ConstantSDNode>(Index);
16640   if (!IndexC || !TLI.isBinOp(Vec.getOpcode()) || !Vec.hasOneUse() ||
16641       Vec.getNode()->getNumValues() != 1)
16642     return SDValue();
16643 
16644   // Targets may want to avoid this to prevent an expensive register transfer.
16645   if (!TLI.shouldScalarizeBinop(Vec))
16646     return SDValue();
16647 
16648   // Extracting an element of a vector constant is constant-folded, so this
16649   // transform is just replacing a vector op with a scalar op while moving the
16650   // extract.
16651   SDValue Op0 = Vec.getOperand(0);
16652   SDValue Op1 = Vec.getOperand(1);
16653   if (isAnyConstantBuildVector(Op0, true) ||
16654       isAnyConstantBuildVector(Op1, true)) {
16655     // extractelt (binop X, C), IndexC --> binop (extractelt X, IndexC), C'
16656     // extractelt (binop C, X), IndexC --> binop C', (extractelt X, IndexC)
16657     SDLoc DL(ExtElt);
16658     EVT VT = ExtElt->getValueType(0);
16659     SDValue Ext0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Op0, Index);
16660     SDValue Ext1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Op1, Index);
16661     return DAG.getNode(Vec.getOpcode(), DL, VT, Ext0, Ext1);
16662   }
16663 
16664   return SDValue();
16665 }
16666 
16667 SDValue DAGCombiner::visitEXTRACT_VECTOR_ELT(SDNode *N) {
16668   SDValue VecOp = N->getOperand(0);
16669   SDValue Index = N->getOperand(1);
16670   EVT ScalarVT = N->getValueType(0);
16671   EVT VecVT = VecOp.getValueType();
16672   if (VecOp.isUndef())
16673     return DAG.getUNDEF(ScalarVT);
16674 
16675   // extract_vector_elt (insert_vector_elt vec, val, idx), idx) -> val
16676   //
16677   // This only really matters if the index is non-constant since other combines
16678   // on the constant elements already work.
16679   SDLoc DL(N);
16680   if (VecOp.getOpcode() == ISD::INSERT_VECTOR_ELT &&
16681       Index == VecOp.getOperand(2)) {
16682     SDValue Elt = VecOp.getOperand(1);
16683     return VecVT.isInteger() ? DAG.getAnyExtOrTrunc(Elt, DL, ScalarVT) : Elt;
16684   }
16685 
16686   // (vextract (scalar_to_vector val, 0) -> val
16687   if (VecOp.getOpcode() == ISD::SCALAR_TO_VECTOR) {
16688     // Check if the result type doesn't match the inserted element type. A
16689     // SCALAR_TO_VECTOR may truncate the inserted element and the
16690     // EXTRACT_VECTOR_ELT may widen the extracted vector.
16691     SDValue InOp = VecOp.getOperand(0);
16692     if (InOp.getValueType() != ScalarVT) {
16693       assert(InOp.getValueType().isInteger() && ScalarVT.isInteger());
16694       return DAG.getSExtOrTrunc(InOp, DL, ScalarVT);
16695     }
16696     return InOp;
16697   }
16698 
16699   // extract_vector_elt of out-of-bounds element -> UNDEF
16700   auto *IndexC = dyn_cast<ConstantSDNode>(Index);
16701   unsigned NumElts = VecVT.getVectorNumElements();
16702   if (IndexC && IndexC->getAPIntValue().uge(NumElts))
16703     return DAG.getUNDEF(ScalarVT);
16704 
16705   // extract_vector_elt (build_vector x, y), 1 -> y
16706   if (IndexC && VecOp.getOpcode() == ISD::BUILD_VECTOR &&
16707       TLI.isTypeLegal(VecVT) &&
16708       (VecOp.hasOneUse() || TLI.aggressivelyPreferBuildVectorSources(VecVT))) {
16709     SDValue Elt = VecOp.getOperand(IndexC->getZExtValue());
16710     EVT InEltVT = Elt.getValueType();
16711 
16712     // Sometimes build_vector's scalar input types do not match result type.
16713     if (ScalarVT == InEltVT)
16714       return Elt;
16715 
16716     // TODO: It may be useful to truncate if free if the build_vector implicitly
16717     // converts.
16718   }
16719 
16720   // TODO: These transforms should not require the 'hasOneUse' restriction, but
16721   // there are regressions on multiple targets without it. We can end up with a
16722   // mess of scalar and vector code if we reduce only part of the DAG to scalar.
16723   if (IndexC && VecOp.getOpcode() == ISD::BITCAST && VecVT.isInteger() &&
16724       VecOp.hasOneUse()) {
16725     // The vector index of the LSBs of the source depend on the endian-ness.
16726     bool IsLE = DAG.getDataLayout().isLittleEndian();
16727     unsigned ExtractIndex = IndexC->getZExtValue();
16728     // extract_elt (v2i32 (bitcast i64:x)), BCTruncElt -> i32 (trunc i64:x)
16729     unsigned BCTruncElt = IsLE ? 0 : NumElts - 1;
16730     SDValue BCSrc = VecOp.getOperand(0);
16731     if (ExtractIndex == BCTruncElt && BCSrc.getValueType().isScalarInteger())
16732       return DAG.getNode(ISD::TRUNCATE, DL, ScalarVT, BCSrc);
16733 
16734     if (LegalTypes && BCSrc.getValueType().isInteger() &&
16735         BCSrc.getOpcode() == ISD::SCALAR_TO_VECTOR) {
16736       // ext_elt (bitcast (scalar_to_vec i64 X to v2i64) to v4i32), TruncElt -->
16737       // trunc i64 X to i32
16738       SDValue X = BCSrc.getOperand(0);
16739       assert(X.getValueType().isScalarInteger() && ScalarVT.isScalarInteger() &&
16740              "Extract element and scalar to vector can't change element type "
16741              "from FP to integer.");
16742       unsigned XBitWidth = X.getValueSizeInBits();
16743       unsigned VecEltBitWidth = VecVT.getScalarSizeInBits();
16744       BCTruncElt = IsLE ? 0 : XBitWidth / VecEltBitWidth - 1;
16745 
16746       // An extract element return value type can be wider than its vector
16747       // operand element type. In that case, the high bits are undefined, so
16748       // it's possible that we may need to extend rather than truncate.
16749       if (ExtractIndex == BCTruncElt && XBitWidth > VecEltBitWidth) {
16750         assert(XBitWidth % VecEltBitWidth == 0 &&
16751                "Scalar bitwidth must be a multiple of vector element bitwidth");
16752         return DAG.getAnyExtOrTrunc(X, DL, ScalarVT);
16753       }
16754     }
16755   }
16756 
16757   if (SDValue BO = scalarizeExtractedBinop(N, DAG, LegalOperations))
16758     return BO;
16759 
16760   // Transform: (EXTRACT_VECTOR_ELT( VECTOR_SHUFFLE )) -> EXTRACT_VECTOR_ELT.
16761   // We only perform this optimization before the op legalization phase because
16762   // we may introduce new vector instructions which are not backed by TD
16763   // patterns. For example on AVX, extracting elements from a wide vector
16764   // without using extract_subvector. However, if we can find an underlying
16765   // scalar value, then we can always use that.
16766   if (IndexC && VecOp.getOpcode() == ISD::VECTOR_SHUFFLE) {
16767     auto *Shuf = cast<ShuffleVectorSDNode>(VecOp);
16768     // Find the new index to extract from.
16769     int OrigElt = Shuf->getMaskElt(IndexC->getZExtValue());
16770 
16771     // Extracting an undef index is undef.
16772     if (OrigElt == -1)
16773       return DAG.getUNDEF(ScalarVT);
16774 
16775     // Select the right vector half to extract from.
16776     SDValue SVInVec;
16777     if (OrigElt < (int)NumElts) {
16778       SVInVec = VecOp.getOperand(0);
16779     } else {
16780       SVInVec = VecOp.getOperand(1);
16781       OrigElt -= NumElts;
16782     }
16783 
16784     if (SVInVec.getOpcode() == ISD::BUILD_VECTOR) {
16785       SDValue InOp = SVInVec.getOperand(OrigElt);
16786       if (InOp.getValueType() != ScalarVT) {
16787         assert(InOp.getValueType().isInteger() && ScalarVT.isInteger());
16788         InOp = DAG.getSExtOrTrunc(InOp, DL, ScalarVT);
16789       }
16790 
16791       return InOp;
16792     }
16793 
16794     // FIXME: We should handle recursing on other vector shuffles and
16795     // scalar_to_vector here as well.
16796 
16797     if (!LegalOperations ||
16798         // FIXME: Should really be just isOperationLegalOrCustom.
16799         TLI.isOperationLegal(ISD::EXTRACT_VECTOR_ELT, VecVT) ||
16800         TLI.isOperationExpand(ISD::VECTOR_SHUFFLE, VecVT)) {
16801       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
16802       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ScalarVT, SVInVec,
16803                          DAG.getConstant(OrigElt, DL, IndexTy));
16804     }
16805   }
16806 
16807   // If only EXTRACT_VECTOR_ELT nodes use the source vector we can
16808   // simplify it based on the (valid) extraction indices.
16809   if (llvm::all_of(VecOp->uses(), [&](SDNode *Use) {
16810         return Use->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
16811                Use->getOperand(0) == VecOp &&
16812                isa<ConstantSDNode>(Use->getOperand(1));
16813       })) {
16814     APInt DemandedElts = APInt::getNullValue(NumElts);
16815     for (SDNode *Use : VecOp->uses()) {
16816       auto *CstElt = cast<ConstantSDNode>(Use->getOperand(1));
16817       if (CstElt->getAPIntValue().ult(NumElts))
16818         DemandedElts.setBit(CstElt->getZExtValue());
16819     }
16820     if (SimplifyDemandedVectorElts(VecOp, DemandedElts, true)) {
16821       // We simplified the vector operand of this extract element. If this
16822       // extract is not dead, visit it again so it is folded properly.
16823       if (N->getOpcode() != ISD::DELETED_NODE)
16824         AddToWorklist(N);
16825       return SDValue(N, 0);
16826     }
16827   }
16828 
16829   // Everything under here is trying to match an extract of a loaded value.
16830   // If the result of load has to be truncated, then it's not necessarily
16831   // profitable.
16832   bool BCNumEltsChanged = false;
16833   EVT ExtVT = VecVT.getVectorElementType();
16834   EVT LVT = ExtVT;
16835   if (ScalarVT.bitsLT(LVT) && !TLI.isTruncateFree(LVT, ScalarVT))
16836     return SDValue();
16837 
16838   if (VecOp.getOpcode() == ISD::BITCAST) {
16839     // Don't duplicate a load with other uses.
16840     if (!VecOp.hasOneUse())
16841       return SDValue();
16842 
16843     EVT BCVT = VecOp.getOperand(0).getValueType();
16844     if (!BCVT.isVector() || ExtVT.bitsGT(BCVT.getVectorElementType()))
16845       return SDValue();
16846     if (NumElts != BCVT.getVectorNumElements())
16847       BCNumEltsChanged = true;
16848     VecOp = VecOp.getOperand(0);
16849     ExtVT = BCVT.getVectorElementType();
16850   }
16851 
16852   // extract (vector load $addr), i --> load $addr + i * size
16853   if (!LegalOperations && !IndexC && VecOp.hasOneUse() &&
16854       ISD::isNormalLoad(VecOp.getNode()) &&
16855       !Index->hasPredecessor(VecOp.getNode())) {
16856     auto *VecLoad = dyn_cast<LoadSDNode>(VecOp);
16857     if (VecLoad && !VecLoad->isVolatile())
16858       return scalarizeExtractedVectorLoad(N, VecVT, Index, VecLoad);
16859   }
16860 
16861   // Perform only after legalization to ensure build_vector / vector_shuffle
16862   // optimizations have already been done.
16863   if (!LegalOperations || !IndexC)
16864     return SDValue();
16865 
16866   // (vextract (v4f32 load $addr), c) -> (f32 load $addr+c*size)
16867   // (vextract (v4f32 s2v (f32 load $addr)), c) -> (f32 load $addr+c*size)
16868   // (vextract (v4f32 shuffle (load $addr), <1,u,u,u>), 0) -> (f32 load $addr)
16869   int Elt = IndexC->getZExtValue();
16870   LoadSDNode *LN0 = nullptr;
16871   if (ISD::isNormalLoad(VecOp.getNode())) {
16872     LN0 = cast<LoadSDNode>(VecOp);
16873   } else if (VecOp.getOpcode() == ISD::SCALAR_TO_VECTOR &&
16874              VecOp.getOperand(0).getValueType() == ExtVT &&
16875              ISD::isNormalLoad(VecOp.getOperand(0).getNode())) {
16876     // Don't duplicate a load with other uses.
16877     if (!VecOp.hasOneUse())
16878       return SDValue();
16879 
16880     LN0 = cast<LoadSDNode>(VecOp.getOperand(0));
16881   }
16882   if (auto *Shuf = dyn_cast<ShuffleVectorSDNode>(VecOp)) {
16883     // (vextract (vector_shuffle (load $addr), v2, <1, u, u, u>), 1)
16884     // =>
16885     // (load $addr+1*size)
16886 
16887     // Don't duplicate a load with other uses.
16888     if (!VecOp.hasOneUse())
16889       return SDValue();
16890 
16891     // If the bit convert changed the number of elements, it is unsafe
16892     // to examine the mask.
16893     if (BCNumEltsChanged)
16894       return SDValue();
16895 
16896     // Select the input vector, guarding against out of range extract vector.
16897     int Idx = (Elt > (int)NumElts) ? -1 : Shuf->getMaskElt(Elt);
16898     VecOp = (Idx < (int)NumElts) ? VecOp.getOperand(0) : VecOp.getOperand(1);
16899 
16900     if (VecOp.getOpcode() == ISD::BITCAST) {
16901       // Don't duplicate a load with other uses.
16902       if (!VecOp.hasOneUse())
16903         return SDValue();
16904 
16905       VecOp = VecOp.getOperand(0);
16906     }
16907     if (ISD::isNormalLoad(VecOp.getNode())) {
16908       LN0 = cast<LoadSDNode>(VecOp);
16909       Elt = (Idx < (int)NumElts) ? Idx : Idx - (int)NumElts;
16910       Index = DAG.getConstant(Elt, DL, Index.getValueType());
16911     }
16912   }
16913 
16914   // Make sure we found a non-volatile load and the extractelement is
16915   // the only use.
16916   if (!LN0 || !LN0->hasNUsesOfValue(1,0) || LN0->isVolatile())
16917     return SDValue();
16918 
16919   // If Idx was -1 above, Elt is going to be -1, so just return undef.
16920   if (Elt == -1)
16921     return DAG.getUNDEF(LVT);
16922 
16923   return scalarizeExtractedVectorLoad(N, VecVT, Index, LN0);
16924 }
16925 
16926 // Simplify (build_vec (ext )) to (bitcast (build_vec ))
16927 SDValue DAGCombiner::reduceBuildVecExtToExtBuildVec(SDNode *N) {
16928   // We perform this optimization post type-legalization because
16929   // the type-legalizer often scalarizes integer-promoted vectors.
16930   // Performing this optimization before may create bit-casts which
16931   // will be type-legalized to complex code sequences.
16932   // We perform this optimization only before the operation legalizer because we
16933   // may introduce illegal operations.
16934   if (Level != AfterLegalizeVectorOps && Level != AfterLegalizeTypes)
16935     return SDValue();
16936 
16937   unsigned NumInScalars = N->getNumOperands();
16938   SDLoc DL(N);
16939   EVT VT = N->getValueType(0);
16940 
16941   // Check to see if this is a BUILD_VECTOR of a bunch of values
16942   // which come from any_extend or zero_extend nodes. If so, we can create
16943   // a new BUILD_VECTOR using bit-casts which may enable other BUILD_VECTOR
16944   // optimizations. We do not handle sign-extend because we can't fill the sign
16945   // using shuffles.
16946   EVT SourceType = MVT::Other;
16947   bool AllAnyExt = true;
16948 
16949   for (unsigned i = 0; i != NumInScalars; ++i) {
16950     SDValue In = N->getOperand(i);
16951     // Ignore undef inputs.
16952     if (In.isUndef()) continue;
16953 
16954     bool AnyExt  = In.getOpcode() == ISD::ANY_EXTEND;
16955     bool ZeroExt = In.getOpcode() == ISD::ZERO_EXTEND;
16956 
16957     // Abort if the element is not an extension.
16958     if (!ZeroExt && !AnyExt) {
16959       SourceType = MVT::Other;
16960       break;
16961     }
16962 
16963     // The input is a ZeroExt or AnyExt. Check the original type.
16964     EVT InTy = In.getOperand(0).getValueType();
16965 
16966     // Check that all of the widened source types are the same.
16967     if (SourceType == MVT::Other)
16968       // First time.
16969       SourceType = InTy;
16970     else if (InTy != SourceType) {
16971       // Multiple income types. Abort.
16972       SourceType = MVT::Other;
16973       break;
16974     }
16975 
16976     // Check if all of the extends are ANY_EXTENDs.
16977     AllAnyExt &= AnyExt;
16978   }
16979 
16980   // In order to have valid types, all of the inputs must be extended from the
16981   // same source type and all of the inputs must be any or zero extend.
16982   // Scalar sizes must be a power of two.
16983   EVT OutScalarTy = VT.getScalarType();
16984   bool ValidTypes = SourceType != MVT::Other &&
16985                  isPowerOf2_32(OutScalarTy.getSizeInBits()) &&
16986                  isPowerOf2_32(SourceType.getSizeInBits());
16987 
16988   // Create a new simpler BUILD_VECTOR sequence which other optimizations can
16989   // turn into a single shuffle instruction.
16990   if (!ValidTypes)
16991     return SDValue();
16992 
16993   bool isLE = DAG.getDataLayout().isLittleEndian();
16994   unsigned ElemRatio = OutScalarTy.getSizeInBits()/SourceType.getSizeInBits();
16995   assert(ElemRatio > 1 && "Invalid element size ratio");
16996   SDValue Filler = AllAnyExt ? DAG.getUNDEF(SourceType):
16997                                DAG.getConstant(0, DL, SourceType);
16998 
16999   unsigned NewBVElems = ElemRatio * VT.getVectorNumElements();
17000   SmallVector<SDValue, 8> Ops(NewBVElems, Filler);
17001 
17002   // Populate the new build_vector
17003   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
17004     SDValue Cast = N->getOperand(i);
17005     assert((Cast.getOpcode() == ISD::ANY_EXTEND ||
17006             Cast.getOpcode() == ISD::ZERO_EXTEND ||
17007             Cast.isUndef()) && "Invalid cast opcode");
17008     SDValue In;
17009     if (Cast.isUndef())
17010       In = DAG.getUNDEF(SourceType);
17011     else
17012       In = Cast->getOperand(0);
17013     unsigned Index = isLE ? (i * ElemRatio) :
17014                             (i * ElemRatio + (ElemRatio - 1));
17015 
17016     assert(Index < Ops.size() && "Invalid index");
17017     Ops[Index] = In;
17018   }
17019 
17020   // The type of the new BUILD_VECTOR node.
17021   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SourceType, NewBVElems);
17022   assert(VecVT.getSizeInBits() == VT.getSizeInBits() &&
17023          "Invalid vector size");
17024   // Check if the new vector type is legal.
17025   if (!isTypeLegal(VecVT) ||
17026       (!TLI.isOperationLegal(ISD::BUILD_VECTOR, VecVT) &&
17027        TLI.isOperationLegal(ISD::BUILD_VECTOR, VT)))
17028     return SDValue();
17029 
17030   // Make the new BUILD_VECTOR.
17031   SDValue BV = DAG.getBuildVector(VecVT, DL, Ops);
17032 
17033   // The new BUILD_VECTOR node has the potential to be further optimized.
17034   AddToWorklist(BV.getNode());
17035   // Bitcast to the desired type.
17036   return DAG.getBitcast(VT, BV);
17037 }
17038 
17039 SDValue DAGCombiner::createBuildVecShuffle(const SDLoc &DL, SDNode *N,
17040                                            ArrayRef<int> VectorMask,
17041                                            SDValue VecIn1, SDValue VecIn2,
17042                                            unsigned LeftIdx, bool DidSplitVec) {
17043   MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout());
17044   SDValue ZeroIdx = DAG.getConstant(0, DL, IdxTy);
17045 
17046   EVT VT = N->getValueType(0);
17047   EVT InVT1 = VecIn1.getValueType();
17048   EVT InVT2 = VecIn2.getNode() ? VecIn2.getValueType() : InVT1;
17049 
17050   unsigned NumElems = VT.getVectorNumElements();
17051   unsigned ShuffleNumElems = NumElems;
17052 
17053   // If we artificially split a vector in two already, then the offsets in the
17054   // operands will all be based off of VecIn1, even those in VecIn2.
17055   unsigned Vec2Offset = DidSplitVec ? 0 : InVT1.getVectorNumElements();
17056 
17057   // We can't generate a shuffle node with mismatched input and output types.
17058   // Try to make the types match the type of the output.
17059   if (InVT1 != VT || InVT2 != VT) {
17060     if ((VT.getSizeInBits() % InVT1.getSizeInBits() == 0) && InVT1 == InVT2) {
17061       // If the output vector length is a multiple of both input lengths,
17062       // we can concatenate them and pad the rest with undefs.
17063       unsigned NumConcats = VT.getSizeInBits() / InVT1.getSizeInBits();
17064       assert(NumConcats >= 2 && "Concat needs at least two inputs!");
17065       SmallVector<SDValue, 2> ConcatOps(NumConcats, DAG.getUNDEF(InVT1));
17066       ConcatOps[0] = VecIn1;
17067       ConcatOps[1] = VecIn2 ? VecIn2 : DAG.getUNDEF(InVT1);
17068       VecIn1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, ConcatOps);
17069       VecIn2 = SDValue();
17070     } else if (InVT1.getSizeInBits() == VT.getSizeInBits() * 2) {
17071       if (!TLI.isExtractSubvectorCheap(VT, InVT1, NumElems))
17072         return SDValue();
17073 
17074       if (!VecIn2.getNode()) {
17075         // If we only have one input vector, and it's twice the size of the
17076         // output, split it in two.
17077         VecIn2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1,
17078                              DAG.getConstant(NumElems, DL, IdxTy));
17079         VecIn1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1, ZeroIdx);
17080         // Since we now have shorter input vectors, adjust the offset of the
17081         // second vector's start.
17082         Vec2Offset = NumElems;
17083       } else if (InVT2.getSizeInBits() <= InVT1.getSizeInBits()) {
17084         // VecIn1 is wider than the output, and we have another, possibly
17085         // smaller input. Pad the smaller input with undefs, shuffle at the
17086         // input vector width, and extract the output.
17087         // The shuffle type is different than VT, so check legality again.
17088         if (LegalOperations &&
17089             !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, InVT1))
17090           return SDValue();
17091 
17092         // Legalizing INSERT_SUBVECTOR is tricky - you basically have to
17093         // lower it back into a BUILD_VECTOR. So if the inserted type is
17094         // illegal, don't even try.
17095         if (InVT1 != InVT2) {
17096           if (!TLI.isTypeLegal(InVT2))
17097             return SDValue();
17098           VecIn2 = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, InVT1,
17099                                DAG.getUNDEF(InVT1), VecIn2, ZeroIdx);
17100         }
17101         ShuffleNumElems = NumElems * 2;
17102       } else {
17103         // Both VecIn1 and VecIn2 are wider than the output, and VecIn2 is wider
17104         // than VecIn1. We can't handle this for now - this case will disappear
17105         // when we start sorting the vectors by type.
17106         return SDValue();
17107       }
17108     } else if (InVT2.getSizeInBits() * 2 == VT.getSizeInBits() &&
17109                InVT1.getSizeInBits() == VT.getSizeInBits()) {
17110       SmallVector<SDValue, 2> ConcatOps(2, DAG.getUNDEF(InVT2));
17111       ConcatOps[0] = VecIn2;
17112       VecIn2 = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, ConcatOps);
17113     } else {
17114       // TODO: Support cases where the length mismatch isn't exactly by a
17115       // factor of 2.
17116       // TODO: Move this check upwards, so that if we have bad type
17117       // mismatches, we don't create any DAG nodes.
17118       return SDValue();
17119     }
17120   }
17121 
17122   // Initialize mask to undef.
17123   SmallVector<int, 8> Mask(ShuffleNumElems, -1);
17124 
17125   // Only need to run up to the number of elements actually used, not the
17126   // total number of elements in the shuffle - if we are shuffling a wider
17127   // vector, the high lanes should be set to undef.
17128   for (unsigned i = 0; i != NumElems; ++i) {
17129     if (VectorMask[i] <= 0)
17130       continue;
17131 
17132     unsigned ExtIndex = N->getOperand(i).getConstantOperandVal(1);
17133     if (VectorMask[i] == (int)LeftIdx) {
17134       Mask[i] = ExtIndex;
17135     } else if (VectorMask[i] == (int)LeftIdx + 1) {
17136       Mask[i] = Vec2Offset + ExtIndex;
17137     }
17138   }
17139 
17140   // The type the input vectors may have changed above.
17141   InVT1 = VecIn1.getValueType();
17142 
17143   // If we already have a VecIn2, it should have the same type as VecIn1.
17144   // If we don't, get an undef/zero vector of the appropriate type.
17145   VecIn2 = VecIn2.getNode() ? VecIn2 : DAG.getUNDEF(InVT1);
17146   assert(InVT1 == VecIn2.getValueType() && "Unexpected second input type.");
17147 
17148   SDValue Shuffle = DAG.getVectorShuffle(InVT1, DL, VecIn1, VecIn2, Mask);
17149   if (ShuffleNumElems > NumElems)
17150     Shuffle = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Shuffle, ZeroIdx);
17151 
17152   return Shuffle;
17153 }
17154 
17155 static SDValue reduceBuildVecToShuffleWithZero(SDNode *BV, SelectionDAG &DAG) {
17156   assert(BV->getOpcode() == ISD::BUILD_VECTOR && "Expected build vector");
17157 
17158   // First, determine where the build vector is not undef.
17159   // TODO: We could extend this to handle zero elements as well as undefs.
17160   int NumBVOps = BV->getNumOperands();
17161   int ZextElt = -1;
17162   for (int i = 0; i != NumBVOps; ++i) {
17163     SDValue Op = BV->getOperand(i);
17164     if (Op.isUndef())
17165       continue;
17166     if (ZextElt == -1)
17167       ZextElt = i;
17168     else
17169       return SDValue();
17170   }
17171   // Bail out if there's no non-undef element.
17172   if (ZextElt == -1)
17173     return SDValue();
17174 
17175   // The build vector contains some number of undef elements and exactly
17176   // one other element. That other element must be a zero-extended scalar
17177   // extracted from a vector at a constant index to turn this into a shuffle.
17178   // Also, require that the build vector does not implicitly truncate/extend
17179   // its elements.
17180   // TODO: This could be enhanced to allow ANY_EXTEND as well as ZERO_EXTEND.
17181   EVT VT = BV->getValueType(0);
17182   SDValue Zext = BV->getOperand(ZextElt);
17183   if (Zext.getOpcode() != ISD::ZERO_EXTEND || !Zext.hasOneUse() ||
17184       Zext.getOperand(0).getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
17185       !isa<ConstantSDNode>(Zext.getOperand(0).getOperand(1)) ||
17186       Zext.getValueSizeInBits() != VT.getScalarSizeInBits())
17187     return SDValue();
17188 
17189   // The zero-extend must be a multiple of the source size, and we must be
17190   // building a vector of the same size as the source of the extract element.
17191   SDValue Extract = Zext.getOperand(0);
17192   unsigned DestSize = Zext.getValueSizeInBits();
17193   unsigned SrcSize = Extract.getValueSizeInBits();
17194   if (DestSize % SrcSize != 0 ||
17195       Extract.getOperand(0).getValueSizeInBits() != VT.getSizeInBits())
17196     return SDValue();
17197 
17198   // Create a shuffle mask that will combine the extracted element with zeros
17199   // and undefs.
17200   int ZextRatio = DestSize / SrcSize;
17201   int NumMaskElts = NumBVOps * ZextRatio;
17202   SmallVector<int, 32> ShufMask(NumMaskElts, -1);
17203   for (int i = 0; i != NumMaskElts; ++i) {
17204     if (i / ZextRatio == ZextElt) {
17205       // The low bits of the (potentially translated) extracted element map to
17206       // the source vector. The high bits map to zero. We will use a zero vector
17207       // as the 2nd source operand of the shuffle, so use the 1st element of
17208       // that vector (mask value is number-of-elements) for the high bits.
17209       if (i % ZextRatio == 0)
17210         ShufMask[i] = Extract.getConstantOperandVal(1);
17211       else
17212         ShufMask[i] = NumMaskElts;
17213     }
17214 
17215     // Undef elements of the build vector remain undef because we initialize
17216     // the shuffle mask with -1.
17217   }
17218 
17219   // Turn this into a shuffle with zero if that's legal.
17220   EVT VecVT = Extract.getOperand(0).getValueType();
17221   if (!DAG.getTargetLoweringInfo().isShuffleMaskLegal(ShufMask, VecVT))
17222     return SDValue();
17223 
17224   // buildvec undef, ..., (zext (extractelt V, IndexC)), undef... -->
17225   // bitcast (shuffle V, ZeroVec, VectorMask)
17226   SDLoc DL(BV);
17227   SDValue ZeroVec = DAG.getConstant(0, DL, VecVT);
17228   SDValue Shuf = DAG.getVectorShuffle(VecVT, DL, Extract.getOperand(0), ZeroVec,
17229                                       ShufMask);
17230   return DAG.getBitcast(VT, Shuf);
17231 }
17232 
17233 // Check to see if this is a BUILD_VECTOR of a bunch of EXTRACT_VECTOR_ELT
17234 // operations. If the types of the vectors we're extracting from allow it,
17235 // turn this into a vector_shuffle node.
17236 SDValue DAGCombiner::reduceBuildVecToShuffle(SDNode *N) {
17237   SDLoc DL(N);
17238   EVT VT = N->getValueType(0);
17239 
17240   // Only type-legal BUILD_VECTOR nodes are converted to shuffle nodes.
17241   if (!isTypeLegal(VT))
17242     return SDValue();
17243 
17244   if (SDValue V = reduceBuildVecToShuffleWithZero(N, DAG))
17245     return V;
17246 
17247   // May only combine to shuffle after legalize if shuffle is legal.
17248   if (LegalOperations && !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, VT))
17249     return SDValue();
17250 
17251   bool UsesZeroVector = false;
17252   unsigned NumElems = N->getNumOperands();
17253 
17254   // Record, for each element of the newly built vector, which input vector
17255   // that element comes from. -1 stands for undef, 0 for the zero vector,
17256   // and positive values for the input vectors.
17257   // VectorMask maps each element to its vector number, and VecIn maps vector
17258   // numbers to their initial SDValues.
17259 
17260   SmallVector<int, 8> VectorMask(NumElems, -1);
17261   SmallVector<SDValue, 8> VecIn;
17262   VecIn.push_back(SDValue());
17263 
17264   for (unsigned i = 0; i != NumElems; ++i) {
17265     SDValue Op = N->getOperand(i);
17266 
17267     if (Op.isUndef())
17268       continue;
17269 
17270     // See if we can use a blend with a zero vector.
17271     // TODO: Should we generalize this to a blend with an arbitrary constant
17272     // vector?
17273     if (isNullConstant(Op) || isNullFPConstant(Op)) {
17274       UsesZeroVector = true;
17275       VectorMask[i] = 0;
17276       continue;
17277     }
17278 
17279     // Not an undef or zero. If the input is something other than an
17280     // EXTRACT_VECTOR_ELT with an in-range constant index, bail out.
17281     if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
17282         !isa<ConstantSDNode>(Op.getOperand(1)))
17283       return SDValue();
17284     SDValue ExtractedFromVec = Op.getOperand(0);
17285 
17286     const APInt &ExtractIdx = Op.getConstantOperandAPInt(1);
17287     if (ExtractIdx.uge(ExtractedFromVec.getValueType().getVectorNumElements()))
17288       return SDValue();
17289 
17290     // All inputs must have the same element type as the output.
17291     if (VT.getVectorElementType() !=
17292         ExtractedFromVec.getValueType().getVectorElementType())
17293       return SDValue();
17294 
17295     // Have we seen this input vector before?
17296     // The vectors are expected to be tiny (usually 1 or 2 elements), so using
17297     // a map back from SDValues to numbers isn't worth it.
17298     unsigned Idx = std::distance(
17299         VecIn.begin(), std::find(VecIn.begin(), VecIn.end(), ExtractedFromVec));
17300     if (Idx == VecIn.size())
17301       VecIn.push_back(ExtractedFromVec);
17302 
17303     VectorMask[i] = Idx;
17304   }
17305 
17306   // If we didn't find at least one input vector, bail out.
17307   if (VecIn.size() < 2)
17308     return SDValue();
17309 
17310   // If all the Operands of BUILD_VECTOR extract from same
17311   // vector, then split the vector efficiently based on the maximum
17312   // vector access index and adjust the VectorMask and
17313   // VecIn accordingly.
17314   bool DidSplitVec = false;
17315   if (VecIn.size() == 2) {
17316     unsigned MaxIndex = 0;
17317     unsigned NearestPow2 = 0;
17318     SDValue Vec = VecIn.back();
17319     EVT InVT = Vec.getValueType();
17320     MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout());
17321     SmallVector<unsigned, 8> IndexVec(NumElems, 0);
17322 
17323     for (unsigned i = 0; i < NumElems; i++) {
17324       if (VectorMask[i] <= 0)
17325         continue;
17326       unsigned Index = N->getOperand(i).getConstantOperandVal(1);
17327       IndexVec[i] = Index;
17328       MaxIndex = std::max(MaxIndex, Index);
17329     }
17330 
17331     NearestPow2 = PowerOf2Ceil(MaxIndex);
17332     if (InVT.isSimple() && NearestPow2 > 2 && MaxIndex < NearestPow2 &&
17333         NumElems * 2 < NearestPow2) {
17334       unsigned SplitSize = NearestPow2 / 2;
17335       EVT SplitVT = EVT::getVectorVT(*DAG.getContext(),
17336                                      InVT.getVectorElementType(), SplitSize);
17337       if (TLI.isTypeLegal(SplitVT)) {
17338         SDValue VecIn2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, SplitVT, Vec,
17339                                      DAG.getConstant(SplitSize, DL, IdxTy));
17340         SDValue VecIn1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, SplitVT, Vec,
17341                                      DAG.getConstant(0, DL, IdxTy));
17342         VecIn.pop_back();
17343         VecIn.push_back(VecIn1);
17344         VecIn.push_back(VecIn2);
17345         DidSplitVec = true;
17346 
17347         for (unsigned i = 0; i < NumElems; i++) {
17348           if (VectorMask[i] <= 0)
17349             continue;
17350           VectorMask[i] = (IndexVec[i] < SplitSize) ? 1 : 2;
17351         }
17352       }
17353     }
17354   }
17355 
17356   // TODO: We want to sort the vectors by descending length, so that adjacent
17357   // pairs have similar length, and the longer vector is always first in the
17358   // pair.
17359 
17360   // TODO: Should this fire if some of the input vectors has illegal type (like
17361   // it does now), or should we let legalization run its course first?
17362 
17363   // Shuffle phase:
17364   // Take pairs of vectors, and shuffle them so that the result has elements
17365   // from these vectors in the correct places.
17366   // For example, given:
17367   // t10: i32 = extract_vector_elt t1, Constant:i64<0>
17368   // t11: i32 = extract_vector_elt t2, Constant:i64<0>
17369   // t12: i32 = extract_vector_elt t3, Constant:i64<0>
17370   // t13: i32 = extract_vector_elt t1, Constant:i64<1>
17371   // t14: v4i32 = BUILD_VECTOR t10, t11, t12, t13
17372   // We will generate:
17373   // t20: v4i32 = vector_shuffle<0,4,u,1> t1, t2
17374   // t21: v4i32 = vector_shuffle<u,u,0,u> t3, undef
17375   SmallVector<SDValue, 4> Shuffles;
17376   for (unsigned In = 0, Len = (VecIn.size() / 2); In < Len; ++In) {
17377     unsigned LeftIdx = 2 * In + 1;
17378     SDValue VecLeft = VecIn[LeftIdx];
17379     SDValue VecRight =
17380         (LeftIdx + 1) < VecIn.size() ? VecIn[LeftIdx + 1] : SDValue();
17381 
17382     if (SDValue Shuffle = createBuildVecShuffle(DL, N, VectorMask, VecLeft,
17383                                                 VecRight, LeftIdx, DidSplitVec))
17384       Shuffles.push_back(Shuffle);
17385     else
17386       return SDValue();
17387   }
17388 
17389   // If we need the zero vector as an "ingredient" in the blend tree, add it
17390   // to the list of shuffles.
17391   if (UsesZeroVector)
17392     Shuffles.push_back(VT.isInteger() ? DAG.getConstant(0, DL, VT)
17393                                       : DAG.getConstantFP(0.0, DL, VT));
17394 
17395   // If we only have one shuffle, we're done.
17396   if (Shuffles.size() == 1)
17397     return Shuffles[0];
17398 
17399   // Update the vector mask to point to the post-shuffle vectors.
17400   for (int &Vec : VectorMask)
17401     if (Vec == 0)
17402       Vec = Shuffles.size() - 1;
17403     else
17404       Vec = (Vec - 1) / 2;
17405 
17406   // More than one shuffle. Generate a binary tree of blends, e.g. if from
17407   // the previous step we got the set of shuffles t10, t11, t12, t13, we will
17408   // generate:
17409   // t10: v8i32 = vector_shuffle<0,8,u,u,u,u,u,u> t1, t2
17410   // t11: v8i32 = vector_shuffle<u,u,0,8,u,u,u,u> t3, t4
17411   // t12: v8i32 = vector_shuffle<u,u,u,u,0,8,u,u> t5, t6
17412   // t13: v8i32 = vector_shuffle<u,u,u,u,u,u,0,8> t7, t8
17413   // t20: v8i32 = vector_shuffle<0,1,10,11,u,u,u,u> t10, t11
17414   // t21: v8i32 = vector_shuffle<u,u,u,u,4,5,14,15> t12, t13
17415   // t30: v8i32 = vector_shuffle<0,1,2,3,12,13,14,15> t20, t21
17416 
17417   // Make sure the initial size of the shuffle list is even.
17418   if (Shuffles.size() % 2)
17419     Shuffles.push_back(DAG.getUNDEF(VT));
17420 
17421   for (unsigned CurSize = Shuffles.size(); CurSize > 1; CurSize /= 2) {
17422     if (CurSize % 2) {
17423       Shuffles[CurSize] = DAG.getUNDEF(VT);
17424       CurSize++;
17425     }
17426     for (unsigned In = 0, Len = CurSize / 2; In < Len; ++In) {
17427       int Left = 2 * In;
17428       int Right = 2 * In + 1;
17429       SmallVector<int, 8> Mask(NumElems, -1);
17430       for (unsigned i = 0; i != NumElems; ++i) {
17431         if (VectorMask[i] == Left) {
17432           Mask[i] = i;
17433           VectorMask[i] = In;
17434         } else if (VectorMask[i] == Right) {
17435           Mask[i] = i + NumElems;
17436           VectorMask[i] = In;
17437         }
17438       }
17439 
17440       Shuffles[In] =
17441           DAG.getVectorShuffle(VT, DL, Shuffles[Left], Shuffles[Right], Mask);
17442     }
17443   }
17444   return Shuffles[0];
17445 }
17446 
17447 // Try to turn a build vector of zero extends of extract vector elts into a
17448 // a vector zero extend and possibly an extract subvector.
17449 // TODO: Support sign extend?
17450 // TODO: Allow undef elements?
17451 SDValue DAGCombiner::convertBuildVecZextToZext(SDNode *N) {
17452   if (LegalOperations)
17453     return SDValue();
17454 
17455   EVT VT = N->getValueType(0);
17456 
17457   bool FoundZeroExtend = false;
17458   SDValue Op0 = N->getOperand(0);
17459   auto checkElem = [&](SDValue Op) -> int64_t {
17460     unsigned Opc = Op.getOpcode();
17461     FoundZeroExtend |= (Opc == ISD::ZERO_EXTEND);
17462     if ((Op.getOpcode() == ISD::ZERO_EXTEND || Opc == ISD::ANY_EXTEND) &&
17463         Op.getOperand(0).getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
17464         Op0.getOperand(0).getOperand(0) == Op.getOperand(0).getOperand(0))
17465       if (auto *C = dyn_cast<ConstantSDNode>(Op.getOperand(0).getOperand(1)))
17466         return C->getZExtValue();
17467     return -1;
17468   };
17469 
17470   // Make sure the first element matches
17471   // (zext (extract_vector_elt X, C))
17472   int64_t Offset = checkElem(Op0);
17473   if (Offset < 0)
17474     return SDValue();
17475 
17476   unsigned NumElems = N->getNumOperands();
17477   SDValue In = Op0.getOperand(0).getOperand(0);
17478   EVT InSVT = In.getValueType().getScalarType();
17479   EVT InVT = EVT::getVectorVT(*DAG.getContext(), InSVT, NumElems);
17480 
17481   // Don't create an illegal input type after type legalization.
17482   if (LegalTypes && !TLI.isTypeLegal(InVT))
17483     return SDValue();
17484 
17485   // Ensure all the elements come from the same vector and are adjacent.
17486   for (unsigned i = 1; i != NumElems; ++i) {
17487     if ((Offset + i) != checkElem(N->getOperand(i)))
17488       return SDValue();
17489   }
17490 
17491   SDLoc DL(N);
17492   In = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, InVT, In,
17493                    Op0.getOperand(0).getOperand(1));
17494   return DAG.getNode(FoundZeroExtend ? ISD::ZERO_EXTEND : ISD::ANY_EXTEND, DL,
17495                      VT, In);
17496 }
17497 
17498 SDValue DAGCombiner::visitBUILD_VECTOR(SDNode *N) {
17499   EVT VT = N->getValueType(0);
17500 
17501   // A vector built entirely of undefs is undef.
17502   if (ISD::allOperandsUndef(N))
17503     return DAG.getUNDEF(VT);
17504 
17505   // If this is a splat of a bitcast from another vector, change to a
17506   // concat_vector.
17507   // For example:
17508   //   (build_vector (i64 (bitcast (v2i32 X))), (i64 (bitcast (v2i32 X)))) ->
17509   //     (v2i64 (bitcast (concat_vectors (v2i32 X), (v2i32 X))))
17510   //
17511   // If X is a build_vector itself, the concat can become a larger build_vector.
17512   // TODO: Maybe this is useful for non-splat too?
17513   if (!LegalOperations) {
17514     if (SDValue Splat = cast<BuildVectorSDNode>(N)->getSplatValue()) {
17515       Splat = peekThroughBitcasts(Splat);
17516       EVT SrcVT = Splat.getValueType();
17517       if (SrcVT.isVector()) {
17518         unsigned NumElts = N->getNumOperands() * SrcVT.getVectorNumElements();
17519         EVT NewVT = EVT::getVectorVT(*DAG.getContext(),
17520                                      SrcVT.getVectorElementType(), NumElts);
17521         if (!LegalTypes || TLI.isTypeLegal(NewVT)) {
17522           SmallVector<SDValue, 8> Ops(N->getNumOperands(), Splat);
17523           SDValue Concat = DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N),
17524                                        NewVT, Ops);
17525           return DAG.getBitcast(VT, Concat);
17526         }
17527       }
17528     }
17529   }
17530 
17531   // Check if we can express BUILD VECTOR via subvector extract.
17532   if (!LegalTypes && (N->getNumOperands() > 1)) {
17533     SDValue Op0 = N->getOperand(0);
17534     auto checkElem = [&](SDValue Op) -> uint64_t {
17535       if ((Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT) &&
17536           (Op0.getOperand(0) == Op.getOperand(0)))
17537         if (auto CNode = dyn_cast<ConstantSDNode>(Op.getOperand(1)))
17538           return CNode->getZExtValue();
17539       return -1;
17540     };
17541 
17542     int Offset = checkElem(Op0);
17543     for (unsigned i = 0; i < N->getNumOperands(); ++i) {
17544       if (Offset + i != checkElem(N->getOperand(i))) {
17545         Offset = -1;
17546         break;
17547       }
17548     }
17549 
17550     if ((Offset == 0) &&
17551         (Op0.getOperand(0).getValueType() == N->getValueType(0)))
17552       return Op0.getOperand(0);
17553     if ((Offset != -1) &&
17554         ((Offset % N->getValueType(0).getVectorNumElements()) ==
17555          0)) // IDX must be multiple of output size.
17556       return DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(N), N->getValueType(0),
17557                          Op0.getOperand(0), Op0.getOperand(1));
17558   }
17559 
17560   if (SDValue V = convertBuildVecZextToZext(N))
17561     return V;
17562 
17563   if (SDValue V = reduceBuildVecExtToExtBuildVec(N))
17564     return V;
17565 
17566   if (SDValue V = reduceBuildVecToShuffle(N))
17567     return V;
17568 
17569   return SDValue();
17570 }
17571 
17572 static SDValue combineConcatVectorOfScalars(SDNode *N, SelectionDAG &DAG) {
17573   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
17574   EVT OpVT = N->getOperand(0).getValueType();
17575 
17576   // If the operands are legal vectors, leave them alone.
17577   if (TLI.isTypeLegal(OpVT))
17578     return SDValue();
17579 
17580   SDLoc DL(N);
17581   EVT VT = N->getValueType(0);
17582   SmallVector<SDValue, 8> Ops;
17583 
17584   EVT SVT = EVT::getIntegerVT(*DAG.getContext(), OpVT.getSizeInBits());
17585   SDValue ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
17586 
17587   // Keep track of what we encounter.
17588   bool AnyInteger = false;
17589   bool AnyFP = false;
17590   for (const SDValue &Op : N->ops()) {
17591     if (ISD::BITCAST == Op.getOpcode() &&
17592         !Op.getOperand(0).getValueType().isVector())
17593       Ops.push_back(Op.getOperand(0));
17594     else if (ISD::UNDEF == Op.getOpcode())
17595       Ops.push_back(ScalarUndef);
17596     else
17597       return SDValue();
17598 
17599     // Note whether we encounter an integer or floating point scalar.
17600     // If it's neither, bail out, it could be something weird like x86mmx.
17601     EVT LastOpVT = Ops.back().getValueType();
17602     if (LastOpVT.isFloatingPoint())
17603       AnyFP = true;
17604     else if (LastOpVT.isInteger())
17605       AnyInteger = true;
17606     else
17607       return SDValue();
17608   }
17609 
17610   // If any of the operands is a floating point scalar bitcast to a vector,
17611   // use floating point types throughout, and bitcast everything.
17612   // Replace UNDEFs by another scalar UNDEF node, of the final desired type.
17613   if (AnyFP) {
17614     SVT = EVT::getFloatingPointVT(OpVT.getSizeInBits());
17615     ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
17616     if (AnyInteger) {
17617       for (SDValue &Op : Ops) {
17618         if (Op.getValueType() == SVT)
17619           continue;
17620         if (Op.isUndef())
17621           Op = ScalarUndef;
17622         else
17623           Op = DAG.getBitcast(SVT, Op);
17624       }
17625     }
17626   }
17627 
17628   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SVT,
17629                                VT.getSizeInBits() / SVT.getSizeInBits());
17630   return DAG.getBitcast(VT, DAG.getBuildVector(VecVT, DL, Ops));
17631 }
17632 
17633 // Check to see if this is a CONCAT_VECTORS of a bunch of EXTRACT_SUBVECTOR
17634 // operations. If so, and if the EXTRACT_SUBVECTOR vector inputs come from at
17635 // most two distinct vectors the same size as the result, attempt to turn this
17636 // into a legal shuffle.
17637 static SDValue combineConcatVectorOfExtracts(SDNode *N, SelectionDAG &DAG) {
17638   EVT VT = N->getValueType(0);
17639   EVT OpVT = N->getOperand(0).getValueType();
17640   int NumElts = VT.getVectorNumElements();
17641   int NumOpElts = OpVT.getVectorNumElements();
17642 
17643   SDValue SV0 = DAG.getUNDEF(VT), SV1 = DAG.getUNDEF(VT);
17644   SmallVector<int, 8> Mask;
17645 
17646   for (SDValue Op : N->ops()) {
17647     Op = peekThroughBitcasts(Op);
17648 
17649     // UNDEF nodes convert to UNDEF shuffle mask values.
17650     if (Op.isUndef()) {
17651       Mask.append((unsigned)NumOpElts, -1);
17652       continue;
17653     }
17654 
17655     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
17656       return SDValue();
17657 
17658     // What vector are we extracting the subvector from and at what index?
17659     SDValue ExtVec = Op.getOperand(0);
17660 
17661     // We want the EVT of the original extraction to correctly scale the
17662     // extraction index.
17663     EVT ExtVT = ExtVec.getValueType();
17664     ExtVec = peekThroughBitcasts(ExtVec);
17665 
17666     // UNDEF nodes convert to UNDEF shuffle mask values.
17667     if (ExtVec.isUndef()) {
17668       Mask.append((unsigned)NumOpElts, -1);
17669       continue;
17670     }
17671 
17672     if (!isa<ConstantSDNode>(Op.getOperand(1)))
17673       return SDValue();
17674     int ExtIdx = Op.getConstantOperandVal(1);
17675 
17676     // Ensure that we are extracting a subvector from a vector the same
17677     // size as the result.
17678     if (ExtVT.getSizeInBits() != VT.getSizeInBits())
17679       return SDValue();
17680 
17681     // Scale the subvector index to account for any bitcast.
17682     int NumExtElts = ExtVT.getVectorNumElements();
17683     if (0 == (NumExtElts % NumElts))
17684       ExtIdx /= (NumExtElts / NumElts);
17685     else if (0 == (NumElts % NumExtElts))
17686       ExtIdx *= (NumElts / NumExtElts);
17687     else
17688       return SDValue();
17689 
17690     // At most we can reference 2 inputs in the final shuffle.
17691     if (SV0.isUndef() || SV0 == ExtVec) {
17692       SV0 = ExtVec;
17693       for (int i = 0; i != NumOpElts; ++i)
17694         Mask.push_back(i + ExtIdx);
17695     } else if (SV1.isUndef() || SV1 == ExtVec) {
17696       SV1 = ExtVec;
17697       for (int i = 0; i != NumOpElts; ++i)
17698         Mask.push_back(i + ExtIdx + NumElts);
17699     } else {
17700       return SDValue();
17701     }
17702   }
17703 
17704   if (!DAG.getTargetLoweringInfo().isShuffleMaskLegal(Mask, VT))
17705     return SDValue();
17706 
17707   return DAG.getVectorShuffle(VT, SDLoc(N), DAG.getBitcast(VT, SV0),
17708                               DAG.getBitcast(VT, SV1), Mask);
17709 }
17710 
17711 SDValue DAGCombiner::visitCONCAT_VECTORS(SDNode *N) {
17712   // If we only have one input vector, we don't need to do any concatenation.
17713   if (N->getNumOperands() == 1)
17714     return N->getOperand(0);
17715 
17716   // Check if all of the operands are undefs.
17717   EVT VT = N->getValueType(0);
17718   if (ISD::allOperandsUndef(N))
17719     return DAG.getUNDEF(VT);
17720 
17721   // Optimize concat_vectors where all but the first of the vectors are undef.
17722   if (std::all_of(std::next(N->op_begin()), N->op_end(), [](const SDValue &Op) {
17723         return Op.isUndef();
17724       })) {
17725     SDValue In = N->getOperand(0);
17726     assert(In.getValueType().isVector() && "Must concat vectors");
17727 
17728     SDValue Scalar = peekThroughOneUseBitcasts(In);
17729 
17730     // concat_vectors(scalar_to_vector(scalar), undef) ->
17731     //     scalar_to_vector(scalar)
17732     if (!LegalOperations && Scalar.getOpcode() == ISD::SCALAR_TO_VECTOR &&
17733          Scalar.hasOneUse()) {
17734       EVT SVT = Scalar.getValueType().getVectorElementType();
17735       if (SVT == Scalar.getOperand(0).getValueType())
17736         Scalar = Scalar.getOperand(0);
17737     }
17738 
17739     // concat_vectors(scalar, undef) -> scalar_to_vector(scalar)
17740     if (!Scalar.getValueType().isVector()) {
17741       // If the bitcast type isn't legal, it might be a trunc of a legal type;
17742       // look through the trunc so we can still do the transform:
17743       //   concat_vectors(trunc(scalar), undef) -> scalar_to_vector(scalar)
17744       if (Scalar->getOpcode() == ISD::TRUNCATE &&
17745           !TLI.isTypeLegal(Scalar.getValueType()) &&
17746           TLI.isTypeLegal(Scalar->getOperand(0).getValueType()))
17747         Scalar = Scalar->getOperand(0);
17748 
17749       EVT SclTy = Scalar.getValueType();
17750 
17751       if (!SclTy.isFloatingPoint() && !SclTy.isInteger())
17752         return SDValue();
17753 
17754       // Bail out if the vector size is not a multiple of the scalar size.
17755       if (VT.getSizeInBits() % SclTy.getSizeInBits())
17756         return SDValue();
17757 
17758       unsigned VNTNumElms = VT.getSizeInBits() / SclTy.getSizeInBits();
17759       if (VNTNumElms < 2)
17760         return SDValue();
17761 
17762       EVT NVT = EVT::getVectorVT(*DAG.getContext(), SclTy, VNTNumElms);
17763       if (!TLI.isTypeLegal(NVT) || !TLI.isTypeLegal(Scalar.getValueType()))
17764         return SDValue();
17765 
17766       SDValue Res = DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), NVT, Scalar);
17767       return DAG.getBitcast(VT, Res);
17768     }
17769   }
17770 
17771   // Fold any combination of BUILD_VECTOR or UNDEF nodes into one BUILD_VECTOR.
17772   // We have already tested above for an UNDEF only concatenation.
17773   // fold (concat_vectors (BUILD_VECTOR A, B, ...), (BUILD_VECTOR C, D, ...))
17774   // -> (BUILD_VECTOR A, B, ..., C, D, ...)
17775   auto IsBuildVectorOrUndef = [](const SDValue &Op) {
17776     return ISD::UNDEF == Op.getOpcode() || ISD::BUILD_VECTOR == Op.getOpcode();
17777   };
17778   if (llvm::all_of(N->ops(), IsBuildVectorOrUndef)) {
17779     SmallVector<SDValue, 8> Opnds;
17780     EVT SVT = VT.getScalarType();
17781 
17782     EVT MinVT = SVT;
17783     if (!SVT.isFloatingPoint()) {
17784       // If BUILD_VECTOR are from built from integer, they may have different
17785       // operand types. Get the smallest type and truncate all operands to it.
17786       bool FoundMinVT = false;
17787       for (const SDValue &Op : N->ops())
17788         if (ISD::BUILD_VECTOR == Op.getOpcode()) {
17789           EVT OpSVT = Op.getOperand(0).getValueType();
17790           MinVT = (!FoundMinVT || OpSVT.bitsLE(MinVT)) ? OpSVT : MinVT;
17791           FoundMinVT = true;
17792         }
17793       assert(FoundMinVT && "Concat vector type mismatch");
17794     }
17795 
17796     for (const SDValue &Op : N->ops()) {
17797       EVT OpVT = Op.getValueType();
17798       unsigned NumElts = OpVT.getVectorNumElements();
17799 
17800       if (ISD::UNDEF == Op.getOpcode())
17801         Opnds.append(NumElts, DAG.getUNDEF(MinVT));
17802 
17803       if (ISD::BUILD_VECTOR == Op.getOpcode()) {
17804         if (SVT.isFloatingPoint()) {
17805           assert(SVT == OpVT.getScalarType() && "Concat vector type mismatch");
17806           Opnds.append(Op->op_begin(), Op->op_begin() + NumElts);
17807         } else {
17808           for (unsigned i = 0; i != NumElts; ++i)
17809             Opnds.push_back(
17810                 DAG.getNode(ISD::TRUNCATE, SDLoc(N), MinVT, Op.getOperand(i)));
17811         }
17812       }
17813     }
17814 
17815     assert(VT.getVectorNumElements() == Opnds.size() &&
17816            "Concat vector type mismatch");
17817     return DAG.getBuildVector(VT, SDLoc(N), Opnds);
17818   }
17819 
17820   // Fold CONCAT_VECTORS of only bitcast scalars (or undef) to BUILD_VECTOR.
17821   if (SDValue V = combineConcatVectorOfScalars(N, DAG))
17822     return V;
17823 
17824   // Fold CONCAT_VECTORS of EXTRACT_SUBVECTOR (or undef) to VECTOR_SHUFFLE.
17825   if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT))
17826     if (SDValue V = combineConcatVectorOfExtracts(N, DAG))
17827       return V;
17828 
17829   // Type legalization of vectors and DAG canonicalization of SHUFFLE_VECTOR
17830   // nodes often generate nop CONCAT_VECTOR nodes.
17831   // Scan the CONCAT_VECTOR operands and look for a CONCAT operations that
17832   // place the incoming vectors at the exact same location.
17833   SDValue SingleSource = SDValue();
17834   unsigned PartNumElem = N->getOperand(0).getValueType().getVectorNumElements();
17835 
17836   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
17837     SDValue Op = N->getOperand(i);
17838 
17839     if (Op.isUndef())
17840       continue;
17841 
17842     // Check if this is the identity extract:
17843     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
17844       return SDValue();
17845 
17846     // Find the single incoming vector for the extract_subvector.
17847     if (SingleSource.getNode()) {
17848       if (Op.getOperand(0) != SingleSource)
17849         return SDValue();
17850     } else {
17851       SingleSource = Op.getOperand(0);
17852 
17853       // Check the source type is the same as the type of the result.
17854       // If not, this concat may extend the vector, so we can not
17855       // optimize it away.
17856       if (SingleSource.getValueType() != N->getValueType(0))
17857         return SDValue();
17858     }
17859 
17860     auto *CS = dyn_cast<ConstantSDNode>(Op.getOperand(1));
17861     // The extract index must be constant.
17862     if (!CS)
17863       return SDValue();
17864 
17865     // Check that we are reading from the identity index.
17866     unsigned IdentityIndex = i * PartNumElem;
17867     if (CS->getAPIntValue() != IdentityIndex)
17868       return SDValue();
17869   }
17870 
17871   if (SingleSource.getNode())
17872     return SingleSource;
17873 
17874   return SDValue();
17875 }
17876 
17877 static SDValue narrowInsertExtractVectorBinOp(SDNode *Extract,
17878                                               SelectionDAG &DAG) {
17879   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
17880   SDValue BinOp = Extract->getOperand(0);
17881   unsigned BinOpcode = BinOp.getOpcode();
17882   if (!TLI.isBinOp(BinOpcode) || BinOp.getNode()->getNumValues() != 1)
17883     return SDValue();
17884 
17885   SDValue Bop0 = BinOp.getOperand(0), Bop1 = BinOp.getOperand(1);
17886   SDValue Index = Extract->getOperand(1);
17887   EVT VT = Extract->getValueType(0);
17888 
17889   auto GetSubVector = [VT, Index](SDValue V) {
17890     if (V.getOpcode() != ISD::INSERT_SUBVECTOR ||
17891         V.getOperand(1).getValueType() != VT || V.getOperand(2) != Index)
17892       return SDValue();
17893     return V.getOperand(1);
17894   };
17895   SDValue Sub0 = GetSubVector(Bop0);
17896   SDValue Sub1 = GetSubVector(Bop1);
17897 
17898   // TODO: We could handle the case where only 1 operand is being inserted by
17899   //       creating an extract of the other operand, but that requires checking
17900   //       number of uses and/or costs.
17901   if (!Sub0 || !Sub1 || !TLI.isOperationLegalOrCustom(BinOpcode, VT))
17902     return SDValue();
17903 
17904   // We are inserting both operands of the wide binop only to extract back
17905   // to the narrow vector size. Eliminate all of the insert/extract:
17906   // ext (binop (ins ?, X, Index), (ins ?, Y, Index)), Index --> binop X, Y
17907   return DAG.getNode(BinOpcode, SDLoc(Extract), VT, Sub0, Sub1,
17908                      BinOp->getFlags());
17909 }
17910 
17911 /// If we are extracting a subvector produced by a wide binary operator try
17912 /// to use a narrow binary operator and/or avoid concatenation and extraction.
17913 static SDValue narrowExtractedVectorBinOp(SDNode *Extract, SelectionDAG &DAG) {
17914   // TODO: Refactor with the caller (visitEXTRACT_SUBVECTOR), so we can share
17915   // some of these bailouts with other transforms.
17916 
17917   if (SDValue V = narrowInsertExtractVectorBinOp(Extract, DAG))
17918     return V;
17919 
17920   // The extract index must be a constant, so we can map it to a concat operand.
17921   auto *ExtractIndexC = dyn_cast<ConstantSDNode>(Extract->getOperand(1));
17922   if (!ExtractIndexC)
17923     return SDValue();
17924 
17925   // We are looking for an optionally bitcasted wide vector binary operator
17926   // feeding an extract subvector.
17927   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
17928   SDValue BinOp = peekThroughBitcasts(Extract->getOperand(0));
17929   unsigned BOpcode = BinOp.getOpcode();
17930   if (!TLI.isBinOp(BOpcode) || BinOp.getNode()->getNumValues() != 1)
17931     return SDValue();
17932 
17933   // The binop must be a vector type, so we can extract some fraction of it.
17934   EVT WideBVT = BinOp.getValueType();
17935   if (!WideBVT.isVector())
17936     return SDValue();
17937 
17938   EVT VT = Extract->getValueType(0);
17939   unsigned ExtractIndex = ExtractIndexC->getZExtValue();
17940   assert(ExtractIndex % VT.getVectorNumElements() == 0 &&
17941          "Extract index is not a multiple of the vector length.");
17942 
17943   // Bail out if this is not a proper multiple width extraction.
17944   unsigned WideWidth = WideBVT.getSizeInBits();
17945   unsigned NarrowWidth = VT.getSizeInBits();
17946   if (WideWidth % NarrowWidth != 0)
17947     return SDValue();
17948 
17949   // Bail out if we are extracting a fraction of a single operation. This can
17950   // occur because we potentially looked through a bitcast of the binop.
17951   unsigned NarrowingRatio = WideWidth / NarrowWidth;
17952   unsigned WideNumElts = WideBVT.getVectorNumElements();
17953   if (WideNumElts % NarrowingRatio != 0)
17954     return SDValue();
17955 
17956   // Bail out if the target does not support a narrower version of the binop.
17957   EVT NarrowBVT = EVT::getVectorVT(*DAG.getContext(), WideBVT.getScalarType(),
17958                                    WideNumElts / NarrowingRatio);
17959   if (!TLI.isOperationLegalOrCustomOrPromote(BOpcode, NarrowBVT))
17960     return SDValue();
17961 
17962   // If extraction is cheap, we don't need to look at the binop operands
17963   // for concat ops. The narrow binop alone makes this transform profitable.
17964   // We can't just reuse the original extract index operand because we may have
17965   // bitcasted.
17966   unsigned ConcatOpNum = ExtractIndex / VT.getVectorNumElements();
17967   unsigned ExtBOIdx = ConcatOpNum * NarrowBVT.getVectorNumElements();
17968   EVT ExtBOIdxVT = Extract->getOperand(1).getValueType();
17969   if (TLI.isExtractSubvectorCheap(NarrowBVT, WideBVT, ExtBOIdx) &&
17970       BinOp.hasOneUse() && Extract->getOperand(0)->hasOneUse()) {
17971     // extract (binop B0, B1), N --> binop (extract B0, N), (extract B1, N)
17972     SDLoc DL(Extract);
17973     SDValue NewExtIndex = DAG.getConstant(ExtBOIdx, DL, ExtBOIdxVT);
17974     SDValue X = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT,
17975                             BinOp.getOperand(0), NewExtIndex);
17976     SDValue Y = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT,
17977                             BinOp.getOperand(1), NewExtIndex);
17978     SDValue NarrowBinOp = DAG.getNode(BOpcode, DL, NarrowBVT, X, Y,
17979                                       BinOp.getNode()->getFlags());
17980     return DAG.getBitcast(VT, NarrowBinOp);
17981   }
17982 
17983   // Only handle the case where we are doubling and then halving. A larger ratio
17984   // may require more than two narrow binops to replace the wide binop.
17985   if (NarrowingRatio != 2)
17986     return SDValue();
17987 
17988   // TODO: The motivating case for this transform is an x86 AVX1 target. That
17989   // target has temptingly almost legal versions of bitwise logic ops in 256-bit
17990   // flavors, but no other 256-bit integer support. This could be extended to
17991   // handle any binop, but that may require fixing/adding other folds to avoid
17992   // codegen regressions.
17993   if (BOpcode != ISD::AND && BOpcode != ISD::OR && BOpcode != ISD::XOR)
17994     return SDValue();
17995 
17996   // We need at least one concatenation operation of a binop operand to make
17997   // this transform worthwhile. The concat must double the input vector sizes.
17998   SDValue LHS = peekThroughBitcasts(BinOp.getOperand(0));
17999   SDValue RHS = peekThroughBitcasts(BinOp.getOperand(1));
18000   bool ConcatL =
18001       LHS.getOpcode() == ISD::CONCAT_VECTORS && LHS.getNumOperands() == 2;
18002   bool ConcatR =
18003       RHS.getOpcode() == ISD::CONCAT_VECTORS && RHS.getNumOperands() == 2;
18004   if (ConcatL || ConcatR) {
18005     // If a binop operand was not the result of a concat, we must extract a
18006     // half-sized operand for our new narrow binop:
18007     // extract (binop (concat X1, X2), (concat Y1, Y2)), N --> binop XN, YN
18008     // extract (binop (concat X1, X2), Y), N --> binop XN, (extract Y, IndexC)
18009     // extract (binop X, (concat Y1, Y2)), N --> binop (extract X, IndexC), YN
18010     SDLoc DL(Extract);
18011     SDValue IndexC = DAG.getConstant(ExtBOIdx, DL, ExtBOIdxVT);
18012     SDValue X = ConcatL ? DAG.getBitcast(NarrowBVT, LHS.getOperand(ConcatOpNum))
18013                         : DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT,
18014                                       BinOp.getOperand(0), IndexC);
18015 
18016     SDValue Y = ConcatR ? DAG.getBitcast(NarrowBVT, RHS.getOperand(ConcatOpNum))
18017                         : DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT,
18018                                       BinOp.getOperand(1), IndexC);
18019 
18020     SDValue NarrowBinOp = DAG.getNode(BOpcode, DL, NarrowBVT, X, Y);
18021     return DAG.getBitcast(VT, NarrowBinOp);
18022   }
18023 
18024   return SDValue();
18025 }
18026 
18027 /// If we are extracting a subvector from a wide vector load, convert to a
18028 /// narrow load to eliminate the extraction:
18029 /// (extract_subvector (load wide vector)) --> (load narrow vector)
18030 static SDValue narrowExtractedVectorLoad(SDNode *Extract, SelectionDAG &DAG) {
18031   // TODO: Add support for big-endian. The offset calculation must be adjusted.
18032   if (DAG.getDataLayout().isBigEndian())
18033     return SDValue();
18034 
18035   auto *Ld = dyn_cast<LoadSDNode>(Extract->getOperand(0));
18036   auto *ExtIdx = dyn_cast<ConstantSDNode>(Extract->getOperand(1));
18037   if (!Ld || Ld->getExtensionType() || Ld->isVolatile() || !ExtIdx)
18038     return SDValue();
18039 
18040   // Allow targets to opt-out.
18041   EVT VT = Extract->getValueType(0);
18042   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
18043   if (!TLI.shouldReduceLoadWidth(Ld, Ld->getExtensionType(), VT))
18044     return SDValue();
18045 
18046   // The narrow load will be offset from the base address of the old load if
18047   // we are extracting from something besides index 0 (little-endian).
18048   SDLoc DL(Extract);
18049   SDValue BaseAddr = Ld->getOperand(1);
18050   unsigned Offset = ExtIdx->getZExtValue() * VT.getScalarType().getStoreSize();
18051 
18052   // TODO: Use "BaseIndexOffset" to make this more effective.
18053   SDValue NewAddr = DAG.getMemBasePlusOffset(BaseAddr, Offset, DL);
18054   MachineFunction &MF = DAG.getMachineFunction();
18055   MachineMemOperand *MMO = MF.getMachineMemOperand(Ld->getMemOperand(), Offset,
18056                                                    VT.getStoreSize());
18057   SDValue NewLd = DAG.getLoad(VT, DL, Ld->getChain(), NewAddr, MMO);
18058   DAG.makeEquivalentMemoryOrdering(Ld, NewLd);
18059   return NewLd;
18060 }
18061 
18062 SDValue DAGCombiner::visitEXTRACT_SUBVECTOR(SDNode *N) {
18063   EVT NVT = N->getValueType(0);
18064   SDValue V = N->getOperand(0);
18065 
18066   // Extract from UNDEF is UNDEF.
18067   if (V.isUndef())
18068     return DAG.getUNDEF(NVT);
18069 
18070   if (TLI.isOperationLegalOrCustomOrPromote(ISD::LOAD, NVT))
18071     if (SDValue NarrowLoad = narrowExtractedVectorLoad(N, DAG))
18072       return NarrowLoad;
18073 
18074   // Combine an extract of an extract into a single extract_subvector.
18075   // ext (ext X, C), 0 --> ext X, C
18076   SDValue Index = N->getOperand(1);
18077   if (isNullConstant(Index) && V.getOpcode() == ISD::EXTRACT_SUBVECTOR &&
18078       V.hasOneUse() && isa<ConstantSDNode>(V.getOperand(1))) {
18079     if (TLI.isExtractSubvectorCheap(NVT, V.getOperand(0).getValueType(),
18080                                     V.getConstantOperandVal(1)) &&
18081         TLI.isOperationLegalOrCustom(ISD::EXTRACT_SUBVECTOR, NVT)) {
18082       return DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(N), NVT, V.getOperand(0),
18083                          V.getOperand(1));
18084     }
18085   }
18086 
18087   // Try to move vector bitcast after extract_subv by scaling extraction index:
18088   // extract_subv (bitcast X), Index --> bitcast (extract_subv X, Index')
18089   if (isa<ConstantSDNode>(Index) && V.getOpcode() == ISD::BITCAST &&
18090       V.getOperand(0).getValueType().isVector()) {
18091     SDValue SrcOp = V.getOperand(0);
18092     EVT SrcVT = SrcOp.getValueType();
18093     unsigned SrcNumElts = SrcVT.getVectorNumElements();
18094     unsigned DestNumElts = V.getValueType().getVectorNumElements();
18095     if ((SrcNumElts % DestNumElts) == 0) {
18096       unsigned SrcDestRatio = SrcNumElts / DestNumElts;
18097       unsigned NewExtNumElts = NVT.getVectorNumElements() * SrcDestRatio;
18098       EVT NewExtVT = EVT::getVectorVT(*DAG.getContext(), SrcVT.getScalarType(),
18099                                       NewExtNumElts);
18100       if (TLI.isOperationLegalOrCustom(ISD::EXTRACT_SUBVECTOR, NewExtVT)) {
18101         unsigned IndexValScaled = N->getConstantOperandVal(1) * SrcDestRatio;
18102         SDLoc DL(N);
18103         SDValue NewIndex = DAG.getIntPtrConstant(IndexValScaled, DL);
18104         SDValue NewExtract = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NewExtVT,
18105                                          V.getOperand(0), NewIndex);
18106         return DAG.getBitcast(NVT, NewExtract);
18107       }
18108     }
18109   }
18110 
18111   // Combine:
18112   //    (extract_subvec (concat V1, V2, ...), i)
18113   // Into:
18114   //    Vi if possible
18115   // Only operand 0 is checked as 'concat' assumes all inputs of the same
18116   // type.
18117   if (V.getOpcode() == ISD::CONCAT_VECTORS && isa<ConstantSDNode>(Index) &&
18118       V.getOperand(0).getValueType() == NVT) {
18119     unsigned Idx = N->getConstantOperandVal(1);
18120     unsigned NumElems = NVT.getVectorNumElements();
18121     assert((Idx % NumElems) == 0 &&
18122            "IDX in concat is not a multiple of the result vector length.");
18123     return V->getOperand(Idx / NumElems);
18124   }
18125 
18126   V = peekThroughBitcasts(V);
18127 
18128   // If the input is a build vector. Try to make a smaller build vector.
18129   if (V.getOpcode() == ISD::BUILD_VECTOR) {
18130     if (auto *IdxC = dyn_cast<ConstantSDNode>(Index)) {
18131       EVT InVT = V.getValueType();
18132       unsigned ExtractSize = NVT.getSizeInBits();
18133       unsigned EltSize = InVT.getScalarSizeInBits();
18134       // Only do this if we won't split any elements.
18135       if (ExtractSize % EltSize == 0) {
18136         unsigned NumElems = ExtractSize / EltSize;
18137         EVT EltVT = InVT.getVectorElementType();
18138         EVT ExtractVT = NumElems == 1 ? EltVT
18139                                       : EVT::getVectorVT(*DAG.getContext(),
18140                                                          EltVT, NumElems);
18141         if ((Level < AfterLegalizeDAG ||
18142              (NumElems == 1 ||
18143               TLI.isOperationLegal(ISD::BUILD_VECTOR, ExtractVT))) &&
18144             (!LegalTypes || TLI.isTypeLegal(ExtractVT))) {
18145           unsigned IdxVal = IdxC->getZExtValue();
18146           IdxVal *= NVT.getScalarSizeInBits();
18147           IdxVal /= EltSize;
18148 
18149           if (NumElems == 1) {
18150             SDValue Src = V->getOperand(IdxVal);
18151             if (EltVT != Src.getValueType())
18152               Src = DAG.getNode(ISD::TRUNCATE, SDLoc(N), InVT, Src);
18153             return DAG.getBitcast(NVT, Src);
18154           }
18155 
18156           // Extract the pieces from the original build_vector.
18157           SDValue BuildVec = DAG.getBuildVector(
18158               ExtractVT, SDLoc(N), V->ops().slice(IdxVal, NumElems));
18159           return DAG.getBitcast(NVT, BuildVec);
18160         }
18161       }
18162     }
18163   }
18164 
18165   if (V.getOpcode() == ISD::INSERT_SUBVECTOR) {
18166     // Handle only simple case where vector being inserted and vector
18167     // being extracted are of same size.
18168     EVT SmallVT = V.getOperand(1).getValueType();
18169     if (!NVT.bitsEq(SmallVT))
18170       return SDValue();
18171 
18172     // Only handle cases where both indexes are constants.
18173     auto *ExtIdx = dyn_cast<ConstantSDNode>(Index);
18174     auto *InsIdx = dyn_cast<ConstantSDNode>(V.getOperand(2));
18175     if (InsIdx && ExtIdx) {
18176       // Combine:
18177       //    (extract_subvec (insert_subvec V1, V2, InsIdx), ExtIdx)
18178       // Into:
18179       //    indices are equal or bit offsets are equal => V1
18180       //    otherwise => (extract_subvec V1, ExtIdx)
18181       if (InsIdx->getZExtValue() * SmallVT.getScalarSizeInBits() ==
18182           ExtIdx->getZExtValue() * NVT.getScalarSizeInBits())
18183         return DAG.getBitcast(NVT, V.getOperand(1));
18184       return DAG.getNode(
18185           ISD::EXTRACT_SUBVECTOR, SDLoc(N), NVT,
18186           DAG.getBitcast(N->getOperand(0).getValueType(), V.getOperand(0)),
18187           Index);
18188     }
18189   }
18190 
18191   if (SDValue NarrowBOp = narrowExtractedVectorBinOp(N, DAG))
18192     return NarrowBOp;
18193 
18194   if (SimplifyDemandedVectorElts(SDValue(N, 0)))
18195     return SDValue(N, 0);
18196 
18197   return SDValue();
18198 }
18199 
18200 /// Try to convert a wide shuffle of concatenated vectors into 2 narrow shuffles
18201 /// followed by concatenation. Narrow vector ops may have better performance
18202 /// than wide ops, and this can unlock further narrowing of other vector ops.
18203 /// Targets can invert this transform later if it is not profitable.
18204 static SDValue foldShuffleOfConcatUndefs(ShuffleVectorSDNode *Shuf,
18205                                          SelectionDAG &DAG) {
18206   SDValue N0 = Shuf->getOperand(0), N1 = Shuf->getOperand(1);
18207   if (N0.getOpcode() != ISD::CONCAT_VECTORS || N0.getNumOperands() != 2 ||
18208       N1.getOpcode() != ISD::CONCAT_VECTORS || N1.getNumOperands() != 2 ||
18209       !N0.getOperand(1).isUndef() || !N1.getOperand(1).isUndef())
18210     return SDValue();
18211 
18212   // Split the wide shuffle mask into halves. Any mask element that is accessing
18213   // operand 1 is offset down to account for narrowing of the vectors.
18214   ArrayRef<int> Mask = Shuf->getMask();
18215   EVT VT = Shuf->getValueType(0);
18216   unsigned NumElts = VT.getVectorNumElements();
18217   unsigned HalfNumElts = NumElts / 2;
18218   SmallVector<int, 16> Mask0(HalfNumElts, -1);
18219   SmallVector<int, 16> Mask1(HalfNumElts, -1);
18220   for (unsigned i = 0; i != NumElts; ++i) {
18221     if (Mask[i] == -1)
18222       continue;
18223     int M = Mask[i] < (int)NumElts ? Mask[i] : Mask[i] - (int)HalfNumElts;
18224     if (i < HalfNumElts)
18225       Mask0[i] = M;
18226     else
18227       Mask1[i - HalfNumElts] = M;
18228   }
18229 
18230   // Ask the target if this is a valid transform.
18231   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
18232   EVT HalfVT = EVT::getVectorVT(*DAG.getContext(), VT.getScalarType(),
18233                                 HalfNumElts);
18234   if (!TLI.isShuffleMaskLegal(Mask0, HalfVT) ||
18235       !TLI.isShuffleMaskLegal(Mask1, HalfVT))
18236     return SDValue();
18237 
18238   // shuffle (concat X, undef), (concat Y, undef), Mask -->
18239   // concat (shuffle X, Y, Mask0), (shuffle X, Y, Mask1)
18240   SDValue X = N0.getOperand(0), Y = N1.getOperand(0);
18241   SDLoc DL(Shuf);
18242   SDValue Shuf0 = DAG.getVectorShuffle(HalfVT, DL, X, Y, Mask0);
18243   SDValue Shuf1 = DAG.getVectorShuffle(HalfVT, DL, X, Y, Mask1);
18244   return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Shuf0, Shuf1);
18245 }
18246 
18247 // Tries to turn a shuffle of two CONCAT_VECTORS into a single concat,
18248 // or turn a shuffle of a single concat into simpler shuffle then concat.
18249 static SDValue partitionShuffleOfConcats(SDNode *N, SelectionDAG &DAG) {
18250   EVT VT = N->getValueType(0);
18251   unsigned NumElts = VT.getVectorNumElements();
18252 
18253   SDValue N0 = N->getOperand(0);
18254   SDValue N1 = N->getOperand(1);
18255   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
18256   ArrayRef<int> Mask = SVN->getMask();
18257 
18258   SmallVector<SDValue, 4> Ops;
18259   EVT ConcatVT = N0.getOperand(0).getValueType();
18260   unsigned NumElemsPerConcat = ConcatVT.getVectorNumElements();
18261   unsigned NumConcats = NumElts / NumElemsPerConcat;
18262 
18263   auto IsUndefMaskElt = [](int i) { return i == -1; };
18264 
18265   // Special case: shuffle(concat(A,B)) can be more efficiently represented
18266   // as concat(shuffle(A,B),UNDEF) if the shuffle doesn't set any of the high
18267   // half vector elements.
18268   if (NumElemsPerConcat * 2 == NumElts && N1.isUndef() &&
18269       llvm::all_of(Mask.slice(NumElemsPerConcat, NumElemsPerConcat),
18270                    IsUndefMaskElt)) {
18271     N0 = DAG.getVectorShuffle(ConcatVT, SDLoc(N), N0.getOperand(0),
18272                               N0.getOperand(1),
18273                               Mask.slice(0, NumElemsPerConcat));
18274     N1 = DAG.getUNDEF(ConcatVT);
18275     return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N0, N1);
18276   }
18277 
18278   // Look at every vector that's inserted. We're looking for exact
18279   // subvector-sized copies from a concatenated vector
18280   for (unsigned I = 0; I != NumConcats; ++I) {
18281     unsigned Begin = I * NumElemsPerConcat;
18282     ArrayRef<int> SubMask = Mask.slice(Begin, NumElemsPerConcat);
18283 
18284     // Make sure we're dealing with a copy.
18285     if (llvm::all_of(SubMask, IsUndefMaskElt)) {
18286       Ops.push_back(DAG.getUNDEF(ConcatVT));
18287       continue;
18288     }
18289 
18290     int OpIdx = -1;
18291     for (int i = 0; i != (int)NumElemsPerConcat; ++i) {
18292       if (IsUndefMaskElt(SubMask[i]))
18293         continue;
18294       if ((SubMask[i] % (int)NumElemsPerConcat) != i)
18295         return SDValue();
18296       int EltOpIdx = SubMask[i] / NumElemsPerConcat;
18297       if (0 <= OpIdx && EltOpIdx != OpIdx)
18298         return SDValue();
18299       OpIdx = EltOpIdx;
18300     }
18301     assert(0 <= OpIdx && "Unknown concat_vectors op");
18302 
18303     if (OpIdx < (int)N0.getNumOperands())
18304       Ops.push_back(N0.getOperand(OpIdx));
18305     else
18306       Ops.push_back(N1.getOperand(OpIdx - N0.getNumOperands()));
18307   }
18308 
18309   return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops);
18310 }
18311 
18312 // Attempt to combine a shuffle of 2 inputs of 'scalar sources' -
18313 // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR.
18314 //
18315 // SHUFFLE(BUILD_VECTOR(), BUILD_VECTOR()) -> BUILD_VECTOR() is always
18316 // a simplification in some sense, but it isn't appropriate in general: some
18317 // BUILD_VECTORs are substantially cheaper than others. The general case
18318 // of a BUILD_VECTOR requires inserting each element individually (or
18319 // performing the equivalent in a temporary stack variable). A BUILD_VECTOR of
18320 // all constants is a single constant pool load.  A BUILD_VECTOR where each
18321 // element is identical is a splat.  A BUILD_VECTOR where most of the operands
18322 // are undef lowers to a small number of element insertions.
18323 //
18324 // To deal with this, we currently use a bunch of mostly arbitrary heuristics.
18325 // We don't fold shuffles where one side is a non-zero constant, and we don't
18326 // fold shuffles if the resulting (non-splat) BUILD_VECTOR would have duplicate
18327 // non-constant operands. This seems to work out reasonably well in practice.
18328 static SDValue combineShuffleOfScalars(ShuffleVectorSDNode *SVN,
18329                                        SelectionDAG &DAG,
18330                                        const TargetLowering &TLI) {
18331   EVT VT = SVN->getValueType(0);
18332   unsigned NumElts = VT.getVectorNumElements();
18333   SDValue N0 = SVN->getOperand(0);
18334   SDValue N1 = SVN->getOperand(1);
18335 
18336   if (!N0->hasOneUse())
18337     return SDValue();
18338 
18339   // If only one of N1,N2 is constant, bail out if it is not ALL_ZEROS as
18340   // discussed above.
18341   if (!N1.isUndef()) {
18342     if (!N1->hasOneUse())
18343       return SDValue();
18344 
18345     bool N0AnyConst = isAnyConstantBuildVector(N0);
18346     bool N1AnyConst = isAnyConstantBuildVector(N1);
18347     if (N0AnyConst && !N1AnyConst && !ISD::isBuildVectorAllZeros(N0.getNode()))
18348       return SDValue();
18349     if (!N0AnyConst && N1AnyConst && !ISD::isBuildVectorAllZeros(N1.getNode()))
18350       return SDValue();
18351   }
18352 
18353   // If both inputs are splats of the same value then we can safely merge this
18354   // to a single BUILD_VECTOR with undef elements based on the shuffle mask.
18355   bool IsSplat = false;
18356   auto *BV0 = dyn_cast<BuildVectorSDNode>(N0);
18357   auto *BV1 = dyn_cast<BuildVectorSDNode>(N1);
18358   if (BV0 && BV1)
18359     if (SDValue Splat0 = BV0->getSplatValue())
18360       IsSplat = (Splat0 == BV1->getSplatValue());
18361 
18362   SmallVector<SDValue, 8> Ops;
18363   SmallSet<SDValue, 16> DuplicateOps;
18364   for (int M : SVN->getMask()) {
18365     SDValue Op = DAG.getUNDEF(VT.getScalarType());
18366     if (M >= 0) {
18367       int Idx = M < (int)NumElts ? M : M - NumElts;
18368       SDValue &S = (M < (int)NumElts ? N0 : N1);
18369       if (S.getOpcode() == ISD::BUILD_VECTOR) {
18370         Op = S.getOperand(Idx);
18371       } else if (S.getOpcode() == ISD::SCALAR_TO_VECTOR) {
18372         SDValue Op0 = S.getOperand(0);
18373         Op = Idx == 0 ? Op0 : DAG.getUNDEF(Op0.getValueType());
18374       } else {
18375         // Operand can't be combined - bail out.
18376         return SDValue();
18377       }
18378     }
18379 
18380     // Don't duplicate a non-constant BUILD_VECTOR operand unless we're
18381     // generating a splat; semantically, this is fine, but it's likely to
18382     // generate low-quality code if the target can't reconstruct an appropriate
18383     // shuffle.
18384     if (!Op.isUndef() && !isa<ConstantSDNode>(Op) && !isa<ConstantFPSDNode>(Op))
18385       if (!IsSplat && !DuplicateOps.insert(Op).second)
18386         return SDValue();
18387 
18388     Ops.push_back(Op);
18389   }
18390 
18391   // BUILD_VECTOR requires all inputs to be of the same type, find the
18392   // maximum type and extend them all.
18393   EVT SVT = VT.getScalarType();
18394   if (SVT.isInteger())
18395     for (SDValue &Op : Ops)
18396       SVT = (SVT.bitsLT(Op.getValueType()) ? Op.getValueType() : SVT);
18397   if (SVT != VT.getScalarType())
18398     for (SDValue &Op : Ops)
18399       Op = TLI.isZExtFree(Op.getValueType(), SVT)
18400                ? DAG.getZExtOrTrunc(Op, SDLoc(SVN), SVT)
18401                : DAG.getSExtOrTrunc(Op, SDLoc(SVN), SVT);
18402   return DAG.getBuildVector(VT, SDLoc(SVN), Ops);
18403 }
18404 
18405 // Match shuffles that can be converted to any_vector_extend_in_reg.
18406 // This is often generated during legalization.
18407 // e.g. v4i32 <0,u,1,u> -> (v2i64 any_vector_extend_in_reg(v4i32 src))
18408 // TODO Add support for ZERO_EXTEND_VECTOR_INREG when we have a test case.
18409 static SDValue combineShuffleToVectorExtend(ShuffleVectorSDNode *SVN,
18410                                             SelectionDAG &DAG,
18411                                             const TargetLowering &TLI,
18412                                             bool LegalOperations) {
18413   EVT VT = SVN->getValueType(0);
18414   bool IsBigEndian = DAG.getDataLayout().isBigEndian();
18415 
18416   // TODO Add support for big-endian when we have a test case.
18417   if (!VT.isInteger() || IsBigEndian)
18418     return SDValue();
18419 
18420   unsigned NumElts = VT.getVectorNumElements();
18421   unsigned EltSizeInBits = VT.getScalarSizeInBits();
18422   ArrayRef<int> Mask = SVN->getMask();
18423   SDValue N0 = SVN->getOperand(0);
18424 
18425   // shuffle<0,-1,1,-1> == (v2i64 anyextend_vector_inreg(v4i32))
18426   auto isAnyExtend = [&Mask, &NumElts](unsigned Scale) {
18427     for (unsigned i = 0; i != NumElts; ++i) {
18428       if (Mask[i] < 0)
18429         continue;
18430       if ((i % Scale) == 0 && Mask[i] == (int)(i / Scale))
18431         continue;
18432       return false;
18433     }
18434     return true;
18435   };
18436 
18437   // Attempt to match a '*_extend_vector_inreg' shuffle, we just search for
18438   // power-of-2 extensions as they are the most likely.
18439   for (unsigned Scale = 2; Scale < NumElts; Scale *= 2) {
18440     // Check for non power of 2 vector sizes
18441     if (NumElts % Scale != 0)
18442       continue;
18443     if (!isAnyExtend(Scale))
18444       continue;
18445 
18446     EVT OutSVT = EVT::getIntegerVT(*DAG.getContext(), EltSizeInBits * Scale);
18447     EVT OutVT = EVT::getVectorVT(*DAG.getContext(), OutSVT, NumElts / Scale);
18448     // Never create an illegal type. Only create unsupported operations if we
18449     // are pre-legalization.
18450     if (TLI.isTypeLegal(OutVT))
18451       if (!LegalOperations ||
18452           TLI.isOperationLegalOrCustom(ISD::ANY_EXTEND_VECTOR_INREG, OutVT))
18453         return DAG.getBitcast(VT,
18454                               DAG.getNode(ISD::ANY_EXTEND_VECTOR_INREG,
18455                                           SDLoc(SVN), OutVT, N0));
18456   }
18457 
18458   return SDValue();
18459 }
18460 
18461 // Detect 'truncate_vector_inreg' style shuffles that pack the lower parts of
18462 // each source element of a large type into the lowest elements of a smaller
18463 // destination type. This is often generated during legalization.
18464 // If the source node itself was a '*_extend_vector_inreg' node then we should
18465 // then be able to remove it.
18466 static SDValue combineTruncationShuffle(ShuffleVectorSDNode *SVN,
18467                                         SelectionDAG &DAG) {
18468   EVT VT = SVN->getValueType(0);
18469   bool IsBigEndian = DAG.getDataLayout().isBigEndian();
18470 
18471   // TODO Add support for big-endian when we have a test case.
18472   if (!VT.isInteger() || IsBigEndian)
18473     return SDValue();
18474 
18475   SDValue N0 = peekThroughBitcasts(SVN->getOperand(0));
18476 
18477   unsigned Opcode = N0.getOpcode();
18478   if (Opcode != ISD::ANY_EXTEND_VECTOR_INREG &&
18479       Opcode != ISD::SIGN_EXTEND_VECTOR_INREG &&
18480       Opcode != ISD::ZERO_EXTEND_VECTOR_INREG)
18481     return SDValue();
18482 
18483   SDValue N00 = N0.getOperand(0);
18484   ArrayRef<int> Mask = SVN->getMask();
18485   unsigned NumElts = VT.getVectorNumElements();
18486   unsigned EltSizeInBits = VT.getScalarSizeInBits();
18487   unsigned ExtSrcSizeInBits = N00.getScalarValueSizeInBits();
18488   unsigned ExtDstSizeInBits = N0.getScalarValueSizeInBits();
18489 
18490   if (ExtDstSizeInBits % ExtSrcSizeInBits != 0)
18491     return SDValue();
18492   unsigned ExtScale = ExtDstSizeInBits / ExtSrcSizeInBits;
18493 
18494   // (v4i32 truncate_vector_inreg(v2i64)) == shuffle<0,2-1,-1>
18495   // (v8i16 truncate_vector_inreg(v4i32)) == shuffle<0,2,4,6,-1,-1,-1,-1>
18496   // (v8i16 truncate_vector_inreg(v2i64)) == shuffle<0,4,-1,-1,-1,-1,-1,-1>
18497   auto isTruncate = [&Mask, &NumElts](unsigned Scale) {
18498     for (unsigned i = 0; i != NumElts; ++i) {
18499       if (Mask[i] < 0)
18500         continue;
18501       if ((i * Scale) < NumElts && Mask[i] == (int)(i * Scale))
18502         continue;
18503       return false;
18504     }
18505     return true;
18506   };
18507 
18508   // At the moment we just handle the case where we've truncated back to the
18509   // same size as before the extension.
18510   // TODO: handle more extension/truncation cases as cases arise.
18511   if (EltSizeInBits != ExtSrcSizeInBits)
18512     return SDValue();
18513 
18514   // We can remove *extend_vector_inreg only if the truncation happens at
18515   // the same scale as the extension.
18516   if (isTruncate(ExtScale))
18517     return DAG.getBitcast(VT, N00);
18518 
18519   return SDValue();
18520 }
18521 
18522 // Combine shuffles of splat-shuffles of the form:
18523 // shuffle (shuffle V, undef, splat-mask), undef, M
18524 // If splat-mask contains undef elements, we need to be careful about
18525 // introducing undef's in the folded mask which are not the result of composing
18526 // the masks of the shuffles.
18527 static SDValue combineShuffleOfSplatVal(ShuffleVectorSDNode *Shuf,
18528                                         SelectionDAG &DAG) {
18529   if (!Shuf->getOperand(1).isUndef())
18530     return SDValue();
18531   auto *Splat = dyn_cast<ShuffleVectorSDNode>(Shuf->getOperand(0));
18532   if (!Splat || !Splat->isSplat())
18533     return SDValue();
18534 
18535   ArrayRef<int> ShufMask = Shuf->getMask();
18536   ArrayRef<int> SplatMask = Splat->getMask();
18537   assert(ShufMask.size() == SplatMask.size() && "Mask length mismatch");
18538 
18539   // Prefer simplifying to the splat-shuffle, if possible. This is legal if
18540   // every undef mask element in the splat-shuffle has a corresponding undef
18541   // element in the user-shuffle's mask or if the composition of mask elements
18542   // would result in undef.
18543   // Examples for (shuffle (shuffle v, undef, SplatMask), undef, UserMask):
18544   // * UserMask=[0,2,u,u], SplatMask=[2,u,2,u] -> [2,2,u,u]
18545   //   In this case it is not legal to simplify to the splat-shuffle because we
18546   //   may be exposing the users of the shuffle an undef element at index 1
18547   //   which was not there before the combine.
18548   // * UserMask=[0,u,2,u], SplatMask=[2,u,2,u] -> [2,u,2,u]
18549   //   In this case the composition of masks yields SplatMask, so it's ok to
18550   //   simplify to the splat-shuffle.
18551   // * UserMask=[3,u,2,u], SplatMask=[2,u,2,u] -> [u,u,2,u]
18552   //   In this case the composed mask includes all undef elements of SplatMask
18553   //   and in addition sets element zero to undef. It is safe to simplify to
18554   //   the splat-shuffle.
18555   auto CanSimplifyToExistingSplat = [](ArrayRef<int> UserMask,
18556                                        ArrayRef<int> SplatMask) {
18557     for (unsigned i = 0, e = UserMask.size(); i != e; ++i)
18558       if (UserMask[i] != -1 && SplatMask[i] == -1 &&
18559           SplatMask[UserMask[i]] != -1)
18560         return false;
18561     return true;
18562   };
18563   if (CanSimplifyToExistingSplat(ShufMask, SplatMask))
18564     return Shuf->getOperand(0);
18565 
18566   // Create a new shuffle with a mask that is composed of the two shuffles'
18567   // masks.
18568   SmallVector<int, 32> NewMask;
18569   for (int Idx : ShufMask)
18570     NewMask.push_back(Idx == -1 ? -1 : SplatMask[Idx]);
18571 
18572   return DAG.getVectorShuffle(Splat->getValueType(0), SDLoc(Splat),
18573                               Splat->getOperand(0), Splat->getOperand(1),
18574                               NewMask);
18575 }
18576 
18577 /// If the shuffle mask is taking exactly one element from the first vector
18578 /// operand and passing through all other elements from the second vector
18579 /// operand, return the index of the mask element that is choosing an element
18580 /// from the first operand. Otherwise, return -1.
18581 static int getShuffleMaskIndexOfOneElementFromOp0IntoOp1(ArrayRef<int> Mask) {
18582   int MaskSize = Mask.size();
18583   int EltFromOp0 = -1;
18584   // TODO: This does not match if there are undef elements in the shuffle mask.
18585   // Should we ignore undefs in the shuffle mask instead? The trade-off is
18586   // removing an instruction (a shuffle), but losing the knowledge that some
18587   // vector lanes are not needed.
18588   for (int i = 0; i != MaskSize; ++i) {
18589     if (Mask[i] >= 0 && Mask[i] < MaskSize) {
18590       // We're looking for a shuffle of exactly one element from operand 0.
18591       if (EltFromOp0 != -1)
18592         return -1;
18593       EltFromOp0 = i;
18594     } else if (Mask[i] != i + MaskSize) {
18595       // Nothing from operand 1 can change lanes.
18596       return -1;
18597     }
18598   }
18599   return EltFromOp0;
18600 }
18601 
18602 /// If a shuffle inserts exactly one element from a source vector operand into
18603 /// another vector operand and we can access the specified element as a scalar,
18604 /// then we can eliminate the shuffle.
18605 static SDValue replaceShuffleOfInsert(ShuffleVectorSDNode *Shuf,
18606                                       SelectionDAG &DAG) {
18607   // First, check if we are taking one element of a vector and shuffling that
18608   // element into another vector.
18609   ArrayRef<int> Mask = Shuf->getMask();
18610   SmallVector<int, 16> CommutedMask(Mask.begin(), Mask.end());
18611   SDValue Op0 = Shuf->getOperand(0);
18612   SDValue Op1 = Shuf->getOperand(1);
18613   int ShufOp0Index = getShuffleMaskIndexOfOneElementFromOp0IntoOp1(Mask);
18614   if (ShufOp0Index == -1) {
18615     // Commute mask and check again.
18616     ShuffleVectorSDNode::commuteMask(CommutedMask);
18617     ShufOp0Index = getShuffleMaskIndexOfOneElementFromOp0IntoOp1(CommutedMask);
18618     if (ShufOp0Index == -1)
18619       return SDValue();
18620     // Commute operands to match the commuted shuffle mask.
18621     std::swap(Op0, Op1);
18622     Mask = CommutedMask;
18623   }
18624 
18625   // The shuffle inserts exactly one element from operand 0 into operand 1.
18626   // Now see if we can access that element as a scalar via a real insert element
18627   // instruction.
18628   // TODO: We can try harder to locate the element as a scalar. Examples: it
18629   // could be an operand of SCALAR_TO_VECTOR, BUILD_VECTOR, or a constant.
18630   assert(Mask[ShufOp0Index] >= 0 && Mask[ShufOp0Index] < (int)Mask.size() &&
18631          "Shuffle mask value must be from operand 0");
18632   if (Op0.getOpcode() != ISD::INSERT_VECTOR_ELT)
18633     return SDValue();
18634 
18635   auto *InsIndexC = dyn_cast<ConstantSDNode>(Op0.getOperand(2));
18636   if (!InsIndexC || InsIndexC->getSExtValue() != Mask[ShufOp0Index])
18637     return SDValue();
18638 
18639   // There's an existing insertelement with constant insertion index, so we
18640   // don't need to check the legality/profitability of a replacement operation
18641   // that differs at most in the constant value. The target should be able to
18642   // lower any of those in a similar way. If not, legalization will expand this
18643   // to a scalar-to-vector plus shuffle.
18644   //
18645   // Note that the shuffle may move the scalar from the position that the insert
18646   // element used. Therefore, our new insert element occurs at the shuffle's
18647   // mask index value, not the insert's index value.
18648   // shuffle (insertelt v1, x, C), v2, mask --> insertelt v2, x, C'
18649   SDValue NewInsIndex = DAG.getConstant(ShufOp0Index, SDLoc(Shuf),
18650                                         Op0.getOperand(2).getValueType());
18651   return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(Shuf), Op0.getValueType(),
18652                      Op1, Op0.getOperand(1), NewInsIndex);
18653 }
18654 
18655 /// If we have a unary shuffle of a shuffle, see if it can be folded away
18656 /// completely. This has the potential to lose undef knowledge because the first
18657 /// shuffle may not have an undef mask element where the second one does. So
18658 /// only call this after doing simplifications based on demanded elements.
18659 static SDValue simplifyShuffleOfShuffle(ShuffleVectorSDNode *Shuf) {
18660   // shuf (shuf0 X, Y, Mask0), undef, Mask
18661   auto *Shuf0 = dyn_cast<ShuffleVectorSDNode>(Shuf->getOperand(0));
18662   if (!Shuf0 || !Shuf->getOperand(1).isUndef())
18663     return SDValue();
18664 
18665   ArrayRef<int> Mask = Shuf->getMask();
18666   ArrayRef<int> Mask0 = Shuf0->getMask();
18667   for (int i = 0, e = (int)Mask.size(); i != e; ++i) {
18668     // Ignore undef elements.
18669     if (Mask[i] == -1)
18670       continue;
18671     assert(Mask[i] >= 0 && Mask[i] < e && "Unexpected shuffle mask value");
18672 
18673     // Is the element of the shuffle operand chosen by this shuffle the same as
18674     // the element chosen by the shuffle operand itself?
18675     if (Mask0[Mask[i]] != Mask0[i])
18676       return SDValue();
18677   }
18678   // Every element of this shuffle is identical to the result of the previous
18679   // shuffle, so we can replace this value.
18680   return Shuf->getOperand(0);
18681 }
18682 
18683 SDValue DAGCombiner::visitVECTOR_SHUFFLE(SDNode *N) {
18684   EVT VT = N->getValueType(0);
18685   unsigned NumElts = VT.getVectorNumElements();
18686 
18687   SDValue N0 = N->getOperand(0);
18688   SDValue N1 = N->getOperand(1);
18689 
18690   assert(N0.getValueType() == VT && "Vector shuffle must be normalized in DAG");
18691 
18692   // Canonicalize shuffle undef, undef -> undef
18693   if (N0.isUndef() && N1.isUndef())
18694     return DAG.getUNDEF(VT);
18695 
18696   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
18697 
18698   // Canonicalize shuffle v, v -> v, undef
18699   if (N0 == N1) {
18700     SmallVector<int, 8> NewMask;
18701     for (unsigned i = 0; i != NumElts; ++i) {
18702       int Idx = SVN->getMaskElt(i);
18703       if (Idx >= (int)NumElts) Idx -= NumElts;
18704       NewMask.push_back(Idx);
18705     }
18706     return DAG.getVectorShuffle(VT, SDLoc(N), N0, DAG.getUNDEF(VT), NewMask);
18707   }
18708 
18709   // Canonicalize shuffle undef, v -> v, undef.  Commute the shuffle mask.
18710   if (N0.isUndef())
18711     return DAG.getCommutedVectorShuffle(*SVN);
18712 
18713   // Remove references to rhs if it is undef
18714   if (N1.isUndef()) {
18715     bool Changed = false;
18716     SmallVector<int, 8> NewMask;
18717     for (unsigned i = 0; i != NumElts; ++i) {
18718       int Idx = SVN->getMaskElt(i);
18719       if (Idx >= (int)NumElts) {
18720         Idx = -1;
18721         Changed = true;
18722       }
18723       NewMask.push_back(Idx);
18724     }
18725     if (Changed)
18726       return DAG.getVectorShuffle(VT, SDLoc(N), N0, N1, NewMask);
18727   }
18728 
18729   if (SDValue InsElt = replaceShuffleOfInsert(SVN, DAG))
18730     return InsElt;
18731 
18732   // A shuffle of a single vector that is a splatted value can always be folded.
18733   if (SDValue V = combineShuffleOfSplatVal(SVN, DAG))
18734     return V;
18735 
18736   // If it is a splat, check if the argument vector is another splat or a
18737   // build_vector.
18738   if (SVN->isSplat() && SVN->getSplatIndex() < (int)NumElts) {
18739     int SplatIndex = SVN->getSplatIndex();
18740     if (TLI.isExtractVecEltCheap(VT, SplatIndex) &&
18741         TLI.isBinOp(N0.getOpcode()) && N0.getNode()->getNumValues() == 1) {
18742       // splat (vector_bo L, R), Index -->
18743       // splat (scalar_bo (extelt L, Index), (extelt R, Index))
18744       SDValue L = N0.getOperand(0), R = N0.getOperand(1);
18745       SDLoc DL(N);
18746       EVT EltVT = VT.getScalarType();
18747       SDValue Index = DAG.getIntPtrConstant(SplatIndex, DL);
18748       SDValue ExtL = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, L, Index);
18749       SDValue ExtR = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, R, Index);
18750       SDValue NewBO = DAG.getNode(N0.getOpcode(), DL, EltVT, ExtL, ExtR,
18751                                   N0.getNode()->getFlags());
18752       SDValue Insert = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VT, NewBO);
18753       SmallVector<int, 16> ZeroMask(VT.getVectorNumElements(), 0);
18754       return DAG.getVectorShuffle(VT, DL, Insert, DAG.getUNDEF(VT), ZeroMask);
18755     }
18756 
18757     // If this is a bit convert that changes the element type of the vector but
18758     // not the number of vector elements, look through it.  Be careful not to
18759     // look though conversions that change things like v4f32 to v2f64.
18760     SDNode *V = N0.getNode();
18761     if (V->getOpcode() == ISD::BITCAST) {
18762       SDValue ConvInput = V->getOperand(0);
18763       if (ConvInput.getValueType().isVector() &&
18764           ConvInput.getValueType().getVectorNumElements() == NumElts)
18765         V = ConvInput.getNode();
18766     }
18767 
18768     if (V->getOpcode() == ISD::BUILD_VECTOR) {
18769       assert(V->getNumOperands() == NumElts &&
18770              "BUILD_VECTOR has wrong number of operands");
18771       SDValue Base;
18772       bool AllSame = true;
18773       for (unsigned i = 0; i != NumElts; ++i) {
18774         if (!V->getOperand(i).isUndef()) {
18775           Base = V->getOperand(i);
18776           break;
18777         }
18778       }
18779       // Splat of <u, u, u, u>, return <u, u, u, u>
18780       if (!Base.getNode())
18781         return N0;
18782       for (unsigned i = 0; i != NumElts; ++i) {
18783         if (V->getOperand(i) != Base) {
18784           AllSame = false;
18785           break;
18786         }
18787       }
18788       // Splat of <x, x, x, x>, return <x, x, x, x>
18789       if (AllSame)
18790         return N0;
18791 
18792       // Canonicalize any other splat as a build_vector.
18793       SDValue Splatted = V->getOperand(SplatIndex);
18794       SmallVector<SDValue, 8> Ops(NumElts, Splatted);
18795       SDValue NewBV = DAG.getBuildVector(V->getValueType(0), SDLoc(N), Ops);
18796 
18797       // We may have jumped through bitcasts, so the type of the
18798       // BUILD_VECTOR may not match the type of the shuffle.
18799       if (V->getValueType(0) != VT)
18800         NewBV = DAG.getBitcast(VT, NewBV);
18801       return NewBV;
18802     }
18803   }
18804 
18805   // Simplify source operands based on shuffle mask.
18806   if (SimplifyDemandedVectorElts(SDValue(N, 0)))
18807     return SDValue(N, 0);
18808 
18809   // This is intentionally placed after demanded elements simplification because
18810   // it could eliminate knowledge of undef elements created by this shuffle.
18811   if (SDValue ShufOp = simplifyShuffleOfShuffle(SVN))
18812     return ShufOp;
18813 
18814   // Match shuffles that can be converted to any_vector_extend_in_reg.
18815   if (SDValue V = combineShuffleToVectorExtend(SVN, DAG, TLI, LegalOperations))
18816     return V;
18817 
18818   // Combine "truncate_vector_in_reg" style shuffles.
18819   if (SDValue V = combineTruncationShuffle(SVN, DAG))
18820     return V;
18821 
18822   if (N0.getOpcode() == ISD::CONCAT_VECTORS &&
18823       Level < AfterLegalizeVectorOps &&
18824       (N1.isUndef() ||
18825       (N1.getOpcode() == ISD::CONCAT_VECTORS &&
18826        N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType()))) {
18827     if (SDValue V = partitionShuffleOfConcats(N, DAG))
18828       return V;
18829   }
18830 
18831   // Attempt to combine a shuffle of 2 inputs of 'scalar sources' -
18832   // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR.
18833   if (Level < AfterLegalizeDAG && TLI.isTypeLegal(VT))
18834     if (SDValue Res = combineShuffleOfScalars(SVN, DAG, TLI))
18835       return Res;
18836 
18837   // If this shuffle only has a single input that is a bitcasted shuffle,
18838   // attempt to merge the 2 shuffles and suitably bitcast the inputs/output
18839   // back to their original types.
18840   if (N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
18841       N1.isUndef() && Level < AfterLegalizeVectorOps &&
18842       TLI.isTypeLegal(VT)) {
18843     auto ScaleShuffleMask = [](ArrayRef<int> Mask, int Scale) {
18844       if (Scale == 1)
18845         return SmallVector<int, 8>(Mask.begin(), Mask.end());
18846 
18847       SmallVector<int, 8> NewMask;
18848       for (int M : Mask)
18849         for (int s = 0; s != Scale; ++s)
18850           NewMask.push_back(M < 0 ? -1 : Scale * M + s);
18851       return NewMask;
18852     };
18853 
18854     SDValue BC0 = peekThroughOneUseBitcasts(N0);
18855     if (BC0.getOpcode() == ISD::VECTOR_SHUFFLE && BC0.hasOneUse()) {
18856       EVT SVT = VT.getScalarType();
18857       EVT InnerVT = BC0->getValueType(0);
18858       EVT InnerSVT = InnerVT.getScalarType();
18859 
18860       // Determine which shuffle works with the smaller scalar type.
18861       EVT ScaleVT = SVT.bitsLT(InnerSVT) ? VT : InnerVT;
18862       EVT ScaleSVT = ScaleVT.getScalarType();
18863 
18864       if (TLI.isTypeLegal(ScaleVT) &&
18865           0 == (InnerSVT.getSizeInBits() % ScaleSVT.getSizeInBits()) &&
18866           0 == (SVT.getSizeInBits() % ScaleSVT.getSizeInBits())) {
18867         int InnerScale = InnerSVT.getSizeInBits() / ScaleSVT.getSizeInBits();
18868         int OuterScale = SVT.getSizeInBits() / ScaleSVT.getSizeInBits();
18869 
18870         // Scale the shuffle masks to the smaller scalar type.
18871         ShuffleVectorSDNode *InnerSVN = cast<ShuffleVectorSDNode>(BC0);
18872         SmallVector<int, 8> InnerMask =
18873             ScaleShuffleMask(InnerSVN->getMask(), InnerScale);
18874         SmallVector<int, 8> OuterMask =
18875             ScaleShuffleMask(SVN->getMask(), OuterScale);
18876 
18877         // Merge the shuffle masks.
18878         SmallVector<int, 8> NewMask;
18879         for (int M : OuterMask)
18880           NewMask.push_back(M < 0 ? -1 : InnerMask[M]);
18881 
18882         // Test for shuffle mask legality over both commutations.
18883         SDValue SV0 = BC0->getOperand(0);
18884         SDValue SV1 = BC0->getOperand(1);
18885         bool LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
18886         if (!LegalMask) {
18887           std::swap(SV0, SV1);
18888           ShuffleVectorSDNode::commuteMask(NewMask);
18889           LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
18890         }
18891 
18892         if (LegalMask) {
18893           SV0 = DAG.getBitcast(ScaleVT, SV0);
18894           SV1 = DAG.getBitcast(ScaleVT, SV1);
18895           return DAG.getBitcast(
18896               VT, DAG.getVectorShuffle(ScaleVT, SDLoc(N), SV0, SV1, NewMask));
18897         }
18898       }
18899     }
18900   }
18901 
18902   // Canonicalize shuffles according to rules:
18903   //  shuffle(A, shuffle(A, B)) -> shuffle(shuffle(A,B), A)
18904   //  shuffle(B, shuffle(A, B)) -> shuffle(shuffle(A,B), B)
18905   //  shuffle(B, shuffle(A, Undef)) -> shuffle(shuffle(A, Undef), B)
18906   if (N1.getOpcode() == ISD::VECTOR_SHUFFLE &&
18907       N0.getOpcode() != ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG &&
18908       TLI.isTypeLegal(VT)) {
18909     // The incoming shuffle must be of the same type as the result of the
18910     // current shuffle.
18911     assert(N1->getOperand(0).getValueType() == VT &&
18912            "Shuffle types don't match");
18913 
18914     SDValue SV0 = N1->getOperand(0);
18915     SDValue SV1 = N1->getOperand(1);
18916     bool HasSameOp0 = N0 == SV0;
18917     bool IsSV1Undef = SV1.isUndef();
18918     if (HasSameOp0 || IsSV1Undef || N0 == SV1)
18919       // Commute the operands of this shuffle so that next rule
18920       // will trigger.
18921       return DAG.getCommutedVectorShuffle(*SVN);
18922   }
18923 
18924   // Try to fold according to rules:
18925   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
18926   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
18927   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
18928   // Don't try to fold shuffles with illegal type.
18929   // Only fold if this shuffle is the only user of the other shuffle.
18930   if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && N->isOnlyUserOf(N0.getNode()) &&
18931       Level < AfterLegalizeDAG && TLI.isTypeLegal(VT)) {
18932     ShuffleVectorSDNode *OtherSV = cast<ShuffleVectorSDNode>(N0);
18933 
18934     // Don't try to fold splats; they're likely to simplify somehow, or they
18935     // might be free.
18936     if (OtherSV->isSplat())
18937       return SDValue();
18938 
18939     // The incoming shuffle must be of the same type as the result of the
18940     // current shuffle.
18941     assert(OtherSV->getOperand(0).getValueType() == VT &&
18942            "Shuffle types don't match");
18943 
18944     SDValue SV0, SV1;
18945     SmallVector<int, 4> Mask;
18946     // Compute the combined shuffle mask for a shuffle with SV0 as the first
18947     // operand, and SV1 as the second operand.
18948     for (unsigned i = 0; i != NumElts; ++i) {
18949       int Idx = SVN->getMaskElt(i);
18950       if (Idx < 0) {
18951         // Propagate Undef.
18952         Mask.push_back(Idx);
18953         continue;
18954       }
18955 
18956       SDValue CurrentVec;
18957       if (Idx < (int)NumElts) {
18958         // This shuffle index refers to the inner shuffle N0. Lookup the inner
18959         // shuffle mask to identify which vector is actually referenced.
18960         Idx = OtherSV->getMaskElt(Idx);
18961         if (Idx < 0) {
18962           // Propagate Undef.
18963           Mask.push_back(Idx);
18964           continue;
18965         }
18966 
18967         CurrentVec = (Idx < (int) NumElts) ? OtherSV->getOperand(0)
18968                                            : OtherSV->getOperand(1);
18969       } else {
18970         // This shuffle index references an element within N1.
18971         CurrentVec = N1;
18972       }
18973 
18974       // Simple case where 'CurrentVec' is UNDEF.
18975       if (CurrentVec.isUndef()) {
18976         Mask.push_back(-1);
18977         continue;
18978       }
18979 
18980       // Canonicalize the shuffle index. We don't know yet if CurrentVec
18981       // will be the first or second operand of the combined shuffle.
18982       Idx = Idx % NumElts;
18983       if (!SV0.getNode() || SV0 == CurrentVec) {
18984         // Ok. CurrentVec is the left hand side.
18985         // Update the mask accordingly.
18986         SV0 = CurrentVec;
18987         Mask.push_back(Idx);
18988         continue;
18989       }
18990 
18991       // Bail out if we cannot convert the shuffle pair into a single shuffle.
18992       if (SV1.getNode() && SV1 != CurrentVec)
18993         return SDValue();
18994 
18995       // Ok. CurrentVec is the right hand side.
18996       // Update the mask accordingly.
18997       SV1 = CurrentVec;
18998       Mask.push_back(Idx + NumElts);
18999     }
19000 
19001     // Check if all indices in Mask are Undef. In case, propagate Undef.
19002     bool isUndefMask = true;
19003     for (unsigned i = 0; i != NumElts && isUndefMask; ++i)
19004       isUndefMask &= Mask[i] < 0;
19005 
19006     if (isUndefMask)
19007       return DAG.getUNDEF(VT);
19008 
19009     if (!SV0.getNode())
19010       SV0 = DAG.getUNDEF(VT);
19011     if (!SV1.getNode())
19012       SV1 = DAG.getUNDEF(VT);
19013 
19014     // Avoid introducing shuffles with illegal mask.
19015     if (!TLI.isShuffleMaskLegal(Mask, VT)) {
19016       ShuffleVectorSDNode::commuteMask(Mask);
19017 
19018       if (!TLI.isShuffleMaskLegal(Mask, VT))
19019         return SDValue();
19020 
19021       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, A, M2)
19022       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, A, M2)
19023       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, B, M2)
19024       std::swap(SV0, SV1);
19025     }
19026 
19027     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
19028     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
19029     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
19030     return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, Mask);
19031   }
19032 
19033   if (SDValue V = foldShuffleOfConcatUndefs(SVN, DAG))
19034     return V;
19035 
19036   return SDValue();
19037 }
19038 
19039 SDValue DAGCombiner::visitSCALAR_TO_VECTOR(SDNode *N) {
19040   SDValue InVal = N->getOperand(0);
19041   EVT VT = N->getValueType(0);
19042 
19043   // Replace a SCALAR_TO_VECTOR(EXTRACT_VECTOR_ELT(V,C0)) pattern
19044   // with a VECTOR_SHUFFLE and possible truncate.
19045   if (InVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
19046     SDValue InVec = InVal->getOperand(0);
19047     SDValue EltNo = InVal->getOperand(1);
19048     auto InVecT = InVec.getValueType();
19049     if (ConstantSDNode *C0 = dyn_cast<ConstantSDNode>(EltNo)) {
19050       SmallVector<int, 8> NewMask(InVecT.getVectorNumElements(), -1);
19051       int Elt = C0->getZExtValue();
19052       NewMask[0] = Elt;
19053       SDValue Val;
19054       // If we have an implict truncate do truncate here as long as it's legal.
19055       // if it's not legal, this should
19056       if (VT.getScalarType() != InVal.getValueType() &&
19057           InVal.getValueType().isScalarInteger() &&
19058           isTypeLegal(VT.getScalarType())) {
19059         Val =
19060             DAG.getNode(ISD::TRUNCATE, SDLoc(InVal), VT.getScalarType(), InVal);
19061         return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), VT, Val);
19062       }
19063       if (VT.getScalarType() == InVecT.getScalarType() &&
19064           VT.getVectorNumElements() <= InVecT.getVectorNumElements() &&
19065           TLI.isShuffleMaskLegal(NewMask, VT)) {
19066         Val = DAG.getVectorShuffle(InVecT, SDLoc(N), InVec,
19067                                    DAG.getUNDEF(InVecT), NewMask);
19068         // If the initial vector is the correct size this shuffle is a
19069         // valid result.
19070         if (VT == InVecT)
19071           return Val;
19072         // If not we must truncate the vector.
19073         if (VT.getVectorNumElements() != InVecT.getVectorNumElements()) {
19074           MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout());
19075           SDValue ZeroIdx = DAG.getConstant(0, SDLoc(N), IdxTy);
19076           EVT SubVT =
19077               EVT::getVectorVT(*DAG.getContext(), InVecT.getVectorElementType(),
19078                                VT.getVectorNumElements());
19079           Val = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(N), SubVT, Val,
19080                             ZeroIdx);
19081           return Val;
19082         }
19083       }
19084     }
19085   }
19086 
19087   return SDValue();
19088 }
19089 
19090 SDValue DAGCombiner::visitINSERT_SUBVECTOR(SDNode *N) {
19091   EVT VT = N->getValueType(0);
19092   SDValue N0 = N->getOperand(0);
19093   SDValue N1 = N->getOperand(1);
19094   SDValue N2 = N->getOperand(2);
19095 
19096   // If inserting an UNDEF, just return the original vector.
19097   if (N1.isUndef())
19098     return N0;
19099 
19100   // If this is an insert of an extracted vector into an undef vector, we can
19101   // just use the input to the extract.
19102   if (N0.isUndef() && N1.getOpcode() == ISD::EXTRACT_SUBVECTOR &&
19103       N1.getOperand(1) == N2 && N1.getOperand(0).getValueType() == VT)
19104     return N1.getOperand(0);
19105 
19106   // If we are inserting a bitcast value into an undef, with the same
19107   // number of elements, just use the bitcast input of the extract.
19108   // i.e. INSERT_SUBVECTOR UNDEF (BITCAST N1) N2 ->
19109   //        BITCAST (INSERT_SUBVECTOR UNDEF N1 N2)
19110   if (N0.isUndef() && N1.getOpcode() == ISD::BITCAST &&
19111       N1.getOperand(0).getOpcode() == ISD::EXTRACT_SUBVECTOR &&
19112       N1.getOperand(0).getOperand(1) == N2 &&
19113       N1.getOperand(0).getOperand(0).getValueType().getVectorNumElements() ==
19114           VT.getVectorNumElements() &&
19115       N1.getOperand(0).getOperand(0).getValueType().getSizeInBits() ==
19116           VT.getSizeInBits()) {
19117     return DAG.getBitcast(VT, N1.getOperand(0).getOperand(0));
19118   }
19119 
19120   // If both N1 and N2 are bitcast values on which insert_subvector
19121   // would makes sense, pull the bitcast through.
19122   // i.e. INSERT_SUBVECTOR (BITCAST N0) (BITCAST N1) N2 ->
19123   //        BITCAST (INSERT_SUBVECTOR N0 N1 N2)
19124   if (N0.getOpcode() == ISD::BITCAST && N1.getOpcode() == ISD::BITCAST) {
19125     SDValue CN0 = N0.getOperand(0);
19126     SDValue CN1 = N1.getOperand(0);
19127     EVT CN0VT = CN0.getValueType();
19128     EVT CN1VT = CN1.getValueType();
19129     if (CN0VT.isVector() && CN1VT.isVector() &&
19130         CN0VT.getVectorElementType() == CN1VT.getVectorElementType() &&
19131         CN0VT.getVectorNumElements() == VT.getVectorNumElements()) {
19132       SDValue NewINSERT = DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N),
19133                                       CN0.getValueType(), CN0, CN1, N2);
19134       return DAG.getBitcast(VT, NewINSERT);
19135     }
19136   }
19137 
19138   // Combine INSERT_SUBVECTORs where we are inserting to the same index.
19139   // INSERT_SUBVECTOR( INSERT_SUBVECTOR( Vec, SubOld, Idx ), SubNew, Idx )
19140   // --> INSERT_SUBVECTOR( Vec, SubNew, Idx )
19141   if (N0.getOpcode() == ISD::INSERT_SUBVECTOR &&
19142       N0.getOperand(1).getValueType() == N1.getValueType() &&
19143       N0.getOperand(2) == N2)
19144     return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0.getOperand(0),
19145                        N1, N2);
19146 
19147   // Eliminate an intermediate insert into an undef vector:
19148   // insert_subvector undef, (insert_subvector undef, X, 0), N2 -->
19149   // insert_subvector undef, X, N2
19150   if (N0.isUndef() && N1.getOpcode() == ISD::INSERT_SUBVECTOR &&
19151       N1.getOperand(0).isUndef() && isNullConstant(N1.getOperand(2)))
19152     return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0,
19153                        N1.getOperand(1), N2);
19154 
19155   if (!isa<ConstantSDNode>(N2))
19156     return SDValue();
19157 
19158   unsigned InsIdx = cast<ConstantSDNode>(N2)->getZExtValue();
19159 
19160   // Push subvector bitcasts to the output, adjusting the index as we go.
19161   // insert_subvector(bitcast(v), bitcast(s), c1)
19162   // -> bitcast(insert_subvector(v, s, c2))
19163   if ((N0.isUndef() || N0.getOpcode() == ISD::BITCAST) &&
19164       N1.getOpcode() == ISD::BITCAST) {
19165     SDValue N0Src = peekThroughBitcasts(N0);
19166     SDValue N1Src = peekThroughBitcasts(N1);
19167     EVT N0SrcSVT = N0Src.getValueType().getScalarType();
19168     EVT N1SrcSVT = N1Src.getValueType().getScalarType();
19169     if ((N0.isUndef() || N0SrcSVT == N1SrcSVT) &&
19170         N0Src.getValueType().isVector() && N1Src.getValueType().isVector()) {
19171       EVT NewVT;
19172       SDLoc DL(N);
19173       SDValue NewIdx;
19174       MVT IdxVT = TLI.getVectorIdxTy(DAG.getDataLayout());
19175       LLVMContext &Ctx = *DAG.getContext();
19176       unsigned NumElts = VT.getVectorNumElements();
19177       unsigned EltSizeInBits = VT.getScalarSizeInBits();
19178       if ((EltSizeInBits % N1SrcSVT.getSizeInBits()) == 0) {
19179         unsigned Scale = EltSizeInBits / N1SrcSVT.getSizeInBits();
19180         NewVT = EVT::getVectorVT(Ctx, N1SrcSVT, NumElts * Scale);
19181         NewIdx = DAG.getConstant(InsIdx * Scale, DL, IdxVT);
19182       } else if ((N1SrcSVT.getSizeInBits() % EltSizeInBits) == 0) {
19183         unsigned Scale = N1SrcSVT.getSizeInBits() / EltSizeInBits;
19184         if ((NumElts % Scale) == 0 && (InsIdx % Scale) == 0) {
19185           NewVT = EVT::getVectorVT(Ctx, N1SrcSVT, NumElts / Scale);
19186           NewIdx = DAG.getConstant(InsIdx / Scale, DL, IdxVT);
19187         }
19188       }
19189       if (NewIdx && hasOperation(ISD::INSERT_SUBVECTOR, NewVT)) {
19190         SDValue Res = DAG.getBitcast(NewVT, N0Src);
19191         Res = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, NewVT, Res, N1Src, NewIdx);
19192         return DAG.getBitcast(VT, Res);
19193       }
19194     }
19195   }
19196 
19197   // Canonicalize insert_subvector dag nodes.
19198   // Example:
19199   // (insert_subvector (insert_subvector A, Idx0), Idx1)
19200   // -> (insert_subvector (insert_subvector A, Idx1), Idx0)
19201   if (N0.getOpcode() == ISD::INSERT_SUBVECTOR && N0.hasOneUse() &&
19202       N1.getValueType() == N0.getOperand(1).getValueType() &&
19203       isa<ConstantSDNode>(N0.getOperand(2))) {
19204     unsigned OtherIdx = N0.getConstantOperandVal(2);
19205     if (InsIdx < OtherIdx) {
19206       // Swap nodes.
19207       SDValue NewOp = DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT,
19208                                   N0.getOperand(0), N1, N2);
19209       AddToWorklist(NewOp.getNode());
19210       return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N0.getNode()),
19211                          VT, NewOp, N0.getOperand(1), N0.getOperand(2));
19212     }
19213   }
19214 
19215   // If the input vector is a concatenation, and the insert replaces
19216   // one of the pieces, we can optimize into a single concat_vectors.
19217   if (N0.getOpcode() == ISD::CONCAT_VECTORS && N0.hasOneUse() &&
19218       N0.getOperand(0).getValueType() == N1.getValueType()) {
19219     unsigned Factor = N1.getValueType().getVectorNumElements();
19220 
19221     SmallVector<SDValue, 8> Ops(N0->op_begin(), N0->op_end());
19222     Ops[cast<ConstantSDNode>(N2)->getZExtValue() / Factor] = N1;
19223 
19224     return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops);
19225   }
19226 
19227   // Simplify source operands based on insertion.
19228   if (SimplifyDemandedVectorElts(SDValue(N, 0)))
19229     return SDValue(N, 0);
19230 
19231   return SDValue();
19232 }
19233 
19234 SDValue DAGCombiner::visitFP_TO_FP16(SDNode *N) {
19235   SDValue N0 = N->getOperand(0);
19236 
19237   // fold (fp_to_fp16 (fp16_to_fp op)) -> op
19238   if (N0->getOpcode() == ISD::FP16_TO_FP)
19239     return N0->getOperand(0);
19240 
19241   return SDValue();
19242 }
19243 
19244 SDValue DAGCombiner::visitFP16_TO_FP(SDNode *N) {
19245   SDValue N0 = N->getOperand(0);
19246 
19247   // fold fp16_to_fp(op & 0xffff) -> fp16_to_fp(op)
19248   if (N0->getOpcode() == ISD::AND) {
19249     ConstantSDNode *AndConst = getAsNonOpaqueConstant(N0.getOperand(1));
19250     if (AndConst && AndConst->getAPIntValue() == 0xffff) {
19251       return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), N->getValueType(0),
19252                          N0.getOperand(0));
19253     }
19254   }
19255 
19256   return SDValue();
19257 }
19258 
19259 SDValue DAGCombiner::visitVECREDUCE(SDNode *N) {
19260   SDValue N0 = N->getOperand(0);
19261   EVT VT = N0.getValueType();
19262   unsigned Opcode = N->getOpcode();
19263 
19264   // VECREDUCE over 1-element vector is just an extract.
19265   if (VT.getVectorNumElements() == 1) {
19266     SDLoc dl(N);
19267     SDValue Res = DAG.getNode(
19268         ISD::EXTRACT_VECTOR_ELT, dl, VT.getVectorElementType(), N0,
19269         DAG.getConstant(0, dl, TLI.getVectorIdxTy(DAG.getDataLayout())));
19270     if (Res.getValueType() != N->getValueType(0))
19271       Res = DAG.getNode(ISD::ANY_EXTEND, dl, N->getValueType(0), Res);
19272     return Res;
19273   }
19274 
19275   // On an boolean vector an and/or reduction is the same as a umin/umax
19276   // reduction. Convert them if the latter is legal while the former isn't.
19277   if (Opcode == ISD::VECREDUCE_AND || Opcode == ISD::VECREDUCE_OR) {
19278     unsigned NewOpcode = Opcode == ISD::VECREDUCE_AND
19279         ? ISD::VECREDUCE_UMIN : ISD::VECREDUCE_UMAX;
19280     if (!TLI.isOperationLegalOrCustom(Opcode, VT) &&
19281         TLI.isOperationLegalOrCustom(NewOpcode, VT) &&
19282         DAG.ComputeNumSignBits(N0) == VT.getScalarSizeInBits())
19283       return DAG.getNode(NewOpcode, SDLoc(N), N->getValueType(0), N0);
19284   }
19285 
19286   return SDValue();
19287 }
19288 
19289 /// Returns a vector_shuffle if it able to transform an AND to a vector_shuffle
19290 /// with the destination vector and a zero vector.
19291 /// e.g. AND V, <0xffffffff, 0, 0xffffffff, 0>. ==>
19292 ///      vector_shuffle V, Zero, <0, 4, 2, 4>
19293 SDValue DAGCombiner::XformToShuffleWithZero(SDNode *N) {
19294   assert(N->getOpcode() == ISD::AND && "Unexpected opcode!");
19295 
19296   EVT VT = N->getValueType(0);
19297   SDValue LHS = N->getOperand(0);
19298   SDValue RHS = peekThroughBitcasts(N->getOperand(1));
19299   SDLoc DL(N);
19300 
19301   // Make sure we're not running after operation legalization where it
19302   // may have custom lowered the vector shuffles.
19303   if (LegalOperations)
19304     return SDValue();
19305 
19306   if (RHS.getOpcode() != ISD::BUILD_VECTOR)
19307     return SDValue();
19308 
19309   EVT RVT = RHS.getValueType();
19310   unsigned NumElts = RHS.getNumOperands();
19311 
19312   // Attempt to create a valid clear mask, splitting the mask into
19313   // sub elements and checking to see if each is
19314   // all zeros or all ones - suitable for shuffle masking.
19315   auto BuildClearMask = [&](int Split) {
19316     int NumSubElts = NumElts * Split;
19317     int NumSubBits = RVT.getScalarSizeInBits() / Split;
19318 
19319     SmallVector<int, 8> Indices;
19320     for (int i = 0; i != NumSubElts; ++i) {
19321       int EltIdx = i / Split;
19322       int SubIdx = i % Split;
19323       SDValue Elt = RHS.getOperand(EltIdx);
19324       if (Elt.isUndef()) {
19325         Indices.push_back(-1);
19326         continue;
19327       }
19328 
19329       APInt Bits;
19330       if (isa<ConstantSDNode>(Elt))
19331         Bits = cast<ConstantSDNode>(Elt)->getAPIntValue();
19332       else if (isa<ConstantFPSDNode>(Elt))
19333         Bits = cast<ConstantFPSDNode>(Elt)->getValueAPF().bitcastToAPInt();
19334       else
19335         return SDValue();
19336 
19337       // Extract the sub element from the constant bit mask.
19338       if (DAG.getDataLayout().isBigEndian()) {
19339         Bits.lshrInPlace((Split - SubIdx - 1) * NumSubBits);
19340       } else {
19341         Bits.lshrInPlace(SubIdx * NumSubBits);
19342       }
19343 
19344       if (Split > 1)
19345         Bits = Bits.trunc(NumSubBits);
19346 
19347       if (Bits.isAllOnesValue())
19348         Indices.push_back(i);
19349       else if (Bits == 0)
19350         Indices.push_back(i + NumSubElts);
19351       else
19352         return SDValue();
19353     }
19354 
19355     // Let's see if the target supports this vector_shuffle.
19356     EVT ClearSVT = EVT::getIntegerVT(*DAG.getContext(), NumSubBits);
19357     EVT ClearVT = EVT::getVectorVT(*DAG.getContext(), ClearSVT, NumSubElts);
19358     if (!TLI.isVectorClearMaskLegal(Indices, ClearVT))
19359       return SDValue();
19360 
19361     SDValue Zero = DAG.getConstant(0, DL, ClearVT);
19362     return DAG.getBitcast(VT, DAG.getVectorShuffle(ClearVT, DL,
19363                                                    DAG.getBitcast(ClearVT, LHS),
19364                                                    Zero, Indices));
19365   };
19366 
19367   // Determine maximum split level (byte level masking).
19368   int MaxSplit = 1;
19369   if (RVT.getScalarSizeInBits() % 8 == 0)
19370     MaxSplit = RVT.getScalarSizeInBits() / 8;
19371 
19372   for (int Split = 1; Split <= MaxSplit; ++Split)
19373     if (RVT.getScalarSizeInBits() % Split == 0)
19374       if (SDValue S = BuildClearMask(Split))
19375         return S;
19376 
19377   return SDValue();
19378 }
19379 
19380 /// If a vector binop is performed on splat values, it may be profitable to
19381 /// extract, scalarize, and insert/splat.
19382 static SDValue scalarizeBinOpOfSplats(SDNode *N, SelectionDAG &DAG) {
19383   SDValue N0 = N->getOperand(0);
19384   SDValue N1 = N->getOperand(1);
19385   unsigned Opcode = N->getOpcode();
19386   EVT VT = N->getValueType(0);
19387   EVT EltVT = VT.getVectorElementType();
19388   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
19389 
19390   // TODO: Remove/replace the extract cost check? If the elements are available
19391   //       as scalars, then there may be no extract cost. Should we ask if
19392   //       inserting a scalar back into a vector is cheap instead?
19393   int Index0, Index1;
19394   SDValue Src0 = DAG.getSplatSourceVector(N0, Index0);
19395   SDValue Src1 = DAG.getSplatSourceVector(N1, Index1);
19396   if (!Src0 || !Src1 || Index0 != Index1 ||
19397       Src0.getValueType().getVectorElementType() != EltVT ||
19398       Src1.getValueType().getVectorElementType() != EltVT ||
19399       !TLI.isExtractVecEltCheap(VT, Index0) ||
19400       !TLI.isOperationLegalOrCustom(Opcode, EltVT))
19401     return SDValue();
19402 
19403   SDLoc DL(N);
19404   SDValue IndexC =
19405       DAG.getConstant(Index0, DL, TLI.getVectorIdxTy(DAG.getDataLayout()));
19406   SDValue X = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, N0, IndexC);
19407   SDValue Y = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, N1, IndexC);
19408   SDValue ScalarBO = DAG.getNode(Opcode, DL, EltVT, X, Y, N->getFlags());
19409 
19410   // If all lanes but 1 are undefined, no need to splat the scalar result.
19411   // TODO: Keep track of undefs and use that info in the general case.
19412   if (N0.getOpcode() == ISD::BUILD_VECTOR && N0.getOpcode() == N1.getOpcode() &&
19413       count_if(N0->ops(), [](SDValue V) { return !V.isUndef(); }) == 1 &&
19414       count_if(N1->ops(), [](SDValue V) { return !V.isUndef(); }) == 1) {
19415     // bo (build_vec ..undef, X, undef...), (build_vec ..undef, Y, undef...) -->
19416     // build_vec ..undef, (bo X, Y), undef...
19417     SmallVector<SDValue, 8> Ops(VT.getVectorNumElements(), DAG.getUNDEF(EltVT));
19418     Ops[Index0] = ScalarBO;
19419     return DAG.getBuildVector(VT, DL, Ops);
19420   }
19421 
19422   // bo (splat X, Index), (splat Y, Index) --> splat (bo X, Y), Index
19423   SmallVector<SDValue, 8> Ops(VT.getVectorNumElements(), ScalarBO);
19424   return DAG.getBuildVector(VT, DL, Ops);
19425 }
19426 
19427 /// Visit a binary vector operation, like ADD.
19428 SDValue DAGCombiner::SimplifyVBinOp(SDNode *N) {
19429   assert(N->getValueType(0).isVector() &&
19430          "SimplifyVBinOp only works on vectors!");
19431 
19432   SDValue LHS = N->getOperand(0);
19433   SDValue RHS = N->getOperand(1);
19434   SDValue Ops[] = {LHS, RHS};
19435   EVT VT = N->getValueType(0);
19436   unsigned Opcode = N->getOpcode();
19437 
19438   // See if we can constant fold the vector operation.
19439   if (SDValue Fold = DAG.FoldConstantVectorArithmetic(
19440           Opcode, SDLoc(LHS), LHS.getValueType(), Ops, N->getFlags()))
19441     return Fold;
19442 
19443   // Move unary shuffles with identical masks after a vector binop:
19444   // VBinOp (shuffle A, Undef, Mask), (shuffle B, Undef, Mask))
19445   //   --> shuffle (VBinOp A, B), Undef, Mask
19446   // This does not require type legality checks because we are creating the
19447   // same types of operations that are in the original sequence. We do have to
19448   // restrict ops like integer div that have immediate UB (eg, div-by-zero)
19449   // though. This code is adapted from the identical transform in instcombine.
19450   if (Opcode != ISD::UDIV && Opcode != ISD::SDIV &&
19451       Opcode != ISD::UREM && Opcode != ISD::SREM &&
19452       Opcode != ISD::UDIVREM && Opcode != ISD::SDIVREM) {
19453     auto *Shuf0 = dyn_cast<ShuffleVectorSDNode>(LHS);
19454     auto *Shuf1 = dyn_cast<ShuffleVectorSDNode>(RHS);
19455     if (Shuf0 && Shuf1 && Shuf0->getMask().equals(Shuf1->getMask()) &&
19456         LHS.getOperand(1).isUndef() && RHS.getOperand(1).isUndef() &&
19457         (LHS.hasOneUse() || RHS.hasOneUse() || LHS == RHS)) {
19458       SDLoc DL(N);
19459       SDValue NewBinOp = DAG.getNode(Opcode, DL, VT, LHS.getOperand(0),
19460                                      RHS.getOperand(0), N->getFlags());
19461       SDValue UndefV = LHS.getOperand(1);
19462       return DAG.getVectorShuffle(VT, DL, NewBinOp, UndefV, Shuf0->getMask());
19463     }
19464   }
19465 
19466   // The following pattern is likely to emerge with vector reduction ops. Moving
19467   // the binary operation ahead of insertion may allow using a narrower vector
19468   // instruction that has better performance than the wide version of the op:
19469   // VBinOp (ins undef, X, Z), (ins undef, Y, Z) --> ins VecC, (VBinOp X, Y), Z
19470   if (LHS.getOpcode() == ISD::INSERT_SUBVECTOR && LHS.getOperand(0).isUndef() &&
19471       RHS.getOpcode() == ISD::INSERT_SUBVECTOR && RHS.getOperand(0).isUndef() &&
19472       LHS.getOperand(2) == RHS.getOperand(2) &&
19473       (LHS.hasOneUse() || RHS.hasOneUse())) {
19474     SDValue X = LHS.getOperand(1);
19475     SDValue Y = RHS.getOperand(1);
19476     SDValue Z = LHS.getOperand(2);
19477     EVT NarrowVT = X.getValueType();
19478     if (NarrowVT == Y.getValueType() &&
19479         TLI.isOperationLegalOrCustomOrPromote(Opcode, NarrowVT)) {
19480       // (binop undef, undef) may not return undef, so compute that result.
19481       SDLoc DL(N);
19482       SDValue VecC =
19483           DAG.getNode(Opcode, DL, VT, DAG.getUNDEF(VT), DAG.getUNDEF(VT));
19484       SDValue NarrowBO = DAG.getNode(Opcode, DL, NarrowVT, X, Y);
19485       return DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VT, VecC, NarrowBO, Z);
19486     }
19487   }
19488 
19489   if (SDValue V = scalarizeBinOpOfSplats(N, DAG))
19490     return V;
19491 
19492   return SDValue();
19493 }
19494 
19495 SDValue DAGCombiner::SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1,
19496                                     SDValue N2) {
19497   assert(N0.getOpcode() ==ISD::SETCC && "First argument must be a SetCC node!");
19498 
19499   SDValue SCC = SimplifySelectCC(DL, N0.getOperand(0), N0.getOperand(1), N1, N2,
19500                                  cast<CondCodeSDNode>(N0.getOperand(2))->get());
19501 
19502   // If we got a simplified select_cc node back from SimplifySelectCC, then
19503   // break it down into a new SETCC node, and a new SELECT node, and then return
19504   // the SELECT node, since we were called with a SELECT node.
19505   if (SCC.getNode()) {
19506     // Check to see if we got a select_cc back (to turn into setcc/select).
19507     // Otherwise, just return whatever node we got back, like fabs.
19508     if (SCC.getOpcode() == ISD::SELECT_CC) {
19509       const SDNodeFlags Flags = N0.getNode()->getFlags();
19510       SDValue SETCC = DAG.getNode(ISD::SETCC, SDLoc(N0),
19511                                   N0.getValueType(),
19512                                   SCC.getOperand(0), SCC.getOperand(1),
19513                                   SCC.getOperand(4), Flags);
19514       AddToWorklist(SETCC.getNode());
19515       SDValue SelectNode = DAG.getSelect(SDLoc(SCC), SCC.getValueType(), SETCC,
19516                                          SCC.getOperand(2), SCC.getOperand(3));
19517       SelectNode->setFlags(Flags);
19518       return SelectNode;
19519     }
19520 
19521     return SCC;
19522   }
19523   return SDValue();
19524 }
19525 
19526 /// Given a SELECT or a SELECT_CC node, where LHS and RHS are the two values
19527 /// being selected between, see if we can simplify the select.  Callers of this
19528 /// should assume that TheSelect is deleted if this returns true.  As such, they
19529 /// should return the appropriate thing (e.g. the node) back to the top-level of
19530 /// the DAG combiner loop to avoid it being looked at.
19531 bool DAGCombiner::SimplifySelectOps(SDNode *TheSelect, SDValue LHS,
19532                                     SDValue RHS) {
19533   // fold (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
19534   // The select + setcc is redundant, because fsqrt returns NaN for X < 0.
19535   if (const ConstantFPSDNode *NaN = isConstOrConstSplatFP(LHS)) {
19536     if (NaN->isNaN() && RHS.getOpcode() == ISD::FSQRT) {
19537       // We have: (select (setcc ?, ?, ?), NaN, (fsqrt ?))
19538       SDValue Sqrt = RHS;
19539       ISD::CondCode CC;
19540       SDValue CmpLHS;
19541       const ConstantFPSDNode *Zero = nullptr;
19542 
19543       if (TheSelect->getOpcode() == ISD::SELECT_CC) {
19544         CC = cast<CondCodeSDNode>(TheSelect->getOperand(4))->get();
19545         CmpLHS = TheSelect->getOperand(0);
19546         Zero = isConstOrConstSplatFP(TheSelect->getOperand(1));
19547       } else {
19548         // SELECT or VSELECT
19549         SDValue Cmp = TheSelect->getOperand(0);
19550         if (Cmp.getOpcode() == ISD::SETCC) {
19551           CC = cast<CondCodeSDNode>(Cmp.getOperand(2))->get();
19552           CmpLHS = Cmp.getOperand(0);
19553           Zero = isConstOrConstSplatFP(Cmp.getOperand(1));
19554         }
19555       }
19556       if (Zero && Zero->isZero() &&
19557           Sqrt.getOperand(0) == CmpLHS && (CC == ISD::SETOLT ||
19558           CC == ISD::SETULT || CC == ISD::SETLT)) {
19559         // We have: (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
19560         CombineTo(TheSelect, Sqrt);
19561         return true;
19562       }
19563     }
19564   }
19565   // Cannot simplify select with vector condition
19566   if (TheSelect->getOperand(0).getValueType().isVector()) return false;
19567 
19568   // If this is a select from two identical things, try to pull the operation
19569   // through the select.
19570   if (LHS.getOpcode() != RHS.getOpcode() ||
19571       !LHS.hasOneUse() || !RHS.hasOneUse())
19572     return false;
19573 
19574   // If this is a load and the token chain is identical, replace the select
19575   // of two loads with a load through a select of the address to load from.
19576   // This triggers in things like "select bool X, 10.0, 123.0" after the FP
19577   // constants have been dropped into the constant pool.
19578   if (LHS.getOpcode() == ISD::LOAD) {
19579     LoadSDNode *LLD = cast<LoadSDNode>(LHS);
19580     LoadSDNode *RLD = cast<LoadSDNode>(RHS);
19581 
19582     // Token chains must be identical.
19583     if (LHS.getOperand(0) != RHS.getOperand(0) ||
19584         // Do not let this transformation reduce the number of volatile loads.
19585         LLD->isVolatile() || RLD->isVolatile() ||
19586         // FIXME: If either is a pre/post inc/dec load,
19587         // we'd need to split out the address adjustment.
19588         LLD->isIndexed() || RLD->isIndexed() ||
19589         // If this is an EXTLOAD, the VT's must match.
19590         LLD->getMemoryVT() != RLD->getMemoryVT() ||
19591         // If this is an EXTLOAD, the kind of extension must match.
19592         (LLD->getExtensionType() != RLD->getExtensionType() &&
19593          // The only exception is if one of the extensions is anyext.
19594          LLD->getExtensionType() != ISD::EXTLOAD &&
19595          RLD->getExtensionType() != ISD::EXTLOAD) ||
19596         // FIXME: this discards src value information.  This is
19597         // over-conservative. It would be beneficial to be able to remember
19598         // both potential memory locations.  Since we are discarding
19599         // src value info, don't do the transformation if the memory
19600         // locations are not in the default address space.
19601         LLD->getPointerInfo().getAddrSpace() != 0 ||
19602         RLD->getPointerInfo().getAddrSpace() != 0 ||
19603         // We can't produce a CMOV of a TargetFrameIndex since we won't
19604         // generate the address generation required.
19605         LLD->getBasePtr().getOpcode() == ISD::TargetFrameIndex ||
19606         RLD->getBasePtr().getOpcode() == ISD::TargetFrameIndex ||
19607         !TLI.isOperationLegalOrCustom(TheSelect->getOpcode(),
19608                                       LLD->getBasePtr().getValueType()))
19609       return false;
19610 
19611     // The loads must not depend on one another.
19612     if (LLD->isPredecessorOf(RLD) || RLD->isPredecessorOf(LLD))
19613       return false;
19614 
19615     // Check that the select condition doesn't reach either load.  If so,
19616     // folding this will induce a cycle into the DAG.  If not, this is safe to
19617     // xform, so create a select of the addresses.
19618 
19619     SmallPtrSet<const SDNode *, 32> Visited;
19620     SmallVector<const SDNode *, 16> Worklist;
19621 
19622     // Always fail if LLD and RLD are not independent. TheSelect is a
19623     // predecessor to all Nodes in question so we need not search past it.
19624 
19625     Visited.insert(TheSelect);
19626     Worklist.push_back(LLD);
19627     Worklist.push_back(RLD);
19628 
19629     if (SDNode::hasPredecessorHelper(LLD, Visited, Worklist) ||
19630         SDNode::hasPredecessorHelper(RLD, Visited, Worklist))
19631       return false;
19632 
19633     SDValue Addr;
19634     if (TheSelect->getOpcode() == ISD::SELECT) {
19635       // We cannot do this optimization if any pair of {RLD, LLD} is a
19636       // predecessor to {RLD, LLD, CondNode}. As we've already compared the
19637       // Loads, we only need to check if CondNode is a successor to one of the
19638       // loads. We can further avoid this if there's no use of their chain
19639       // value.
19640       SDNode *CondNode = TheSelect->getOperand(0).getNode();
19641       Worklist.push_back(CondNode);
19642 
19643       if ((LLD->hasAnyUseOfValue(1) &&
19644            SDNode::hasPredecessorHelper(LLD, Visited, Worklist)) ||
19645           (RLD->hasAnyUseOfValue(1) &&
19646            SDNode::hasPredecessorHelper(RLD, Visited, Worklist)))
19647         return false;
19648 
19649       Addr = DAG.getSelect(SDLoc(TheSelect),
19650                            LLD->getBasePtr().getValueType(),
19651                            TheSelect->getOperand(0), LLD->getBasePtr(),
19652                            RLD->getBasePtr());
19653     } else {  // Otherwise SELECT_CC
19654       // We cannot do this optimization if any pair of {RLD, LLD} is a
19655       // predecessor to {RLD, LLD, CondLHS, CondRHS}. As we've already compared
19656       // the Loads, we only need to check if CondLHS/CondRHS is a successor to
19657       // one of the loads. We can further avoid this if there's no use of their
19658       // chain value.
19659 
19660       SDNode *CondLHS = TheSelect->getOperand(0).getNode();
19661       SDNode *CondRHS = TheSelect->getOperand(1).getNode();
19662       Worklist.push_back(CondLHS);
19663       Worklist.push_back(CondRHS);
19664 
19665       if ((LLD->hasAnyUseOfValue(1) &&
19666            SDNode::hasPredecessorHelper(LLD, Visited, Worklist)) ||
19667           (RLD->hasAnyUseOfValue(1) &&
19668            SDNode::hasPredecessorHelper(RLD, Visited, Worklist)))
19669         return false;
19670 
19671       Addr = DAG.getNode(ISD::SELECT_CC, SDLoc(TheSelect),
19672                          LLD->getBasePtr().getValueType(),
19673                          TheSelect->getOperand(0),
19674                          TheSelect->getOperand(1),
19675                          LLD->getBasePtr(), RLD->getBasePtr(),
19676                          TheSelect->getOperand(4));
19677     }
19678 
19679     SDValue Load;
19680     // It is safe to replace the two loads if they have different alignments,
19681     // but the new load must be the minimum (most restrictive) alignment of the
19682     // inputs.
19683     unsigned Alignment = std::min(LLD->getAlignment(), RLD->getAlignment());
19684     MachineMemOperand::Flags MMOFlags = LLD->getMemOperand()->getFlags();
19685     if (!RLD->isInvariant())
19686       MMOFlags &= ~MachineMemOperand::MOInvariant;
19687     if (!RLD->isDereferenceable())
19688       MMOFlags &= ~MachineMemOperand::MODereferenceable;
19689     if (LLD->getExtensionType() == ISD::NON_EXTLOAD) {
19690       // FIXME: Discards pointer and AA info.
19691       Load = DAG.getLoad(TheSelect->getValueType(0), SDLoc(TheSelect),
19692                          LLD->getChain(), Addr, MachinePointerInfo(), Alignment,
19693                          MMOFlags);
19694     } else {
19695       // FIXME: Discards pointer and AA info.
19696       Load = DAG.getExtLoad(
19697           LLD->getExtensionType() == ISD::EXTLOAD ? RLD->getExtensionType()
19698                                                   : LLD->getExtensionType(),
19699           SDLoc(TheSelect), TheSelect->getValueType(0), LLD->getChain(), Addr,
19700           MachinePointerInfo(), LLD->getMemoryVT(), Alignment, MMOFlags);
19701     }
19702 
19703     // Users of the select now use the result of the load.
19704     CombineTo(TheSelect, Load);
19705 
19706     // Users of the old loads now use the new load's chain.  We know the
19707     // old-load value is dead now.
19708     CombineTo(LHS.getNode(), Load.getValue(0), Load.getValue(1));
19709     CombineTo(RHS.getNode(), Load.getValue(0), Load.getValue(1));
19710     return true;
19711   }
19712 
19713   return false;
19714 }
19715 
19716 /// Try to fold an expression of the form (N0 cond N1) ? N2 : N3 to a shift and
19717 /// bitwise 'and'.
19718 SDValue DAGCombiner::foldSelectCCToShiftAnd(const SDLoc &DL, SDValue N0,
19719                                             SDValue N1, SDValue N2, SDValue N3,
19720                                             ISD::CondCode CC) {
19721   // If this is a select where the false operand is zero and the compare is a
19722   // check of the sign bit, see if we can perform the "gzip trick":
19723   // select_cc setlt X, 0, A, 0 -> and (sra X, size(X)-1), A
19724   // select_cc setgt X, 0, A, 0 -> and (not (sra X, size(X)-1)), A
19725   EVT XType = N0.getValueType();
19726   EVT AType = N2.getValueType();
19727   if (!isNullConstant(N3) || !XType.bitsGE(AType))
19728     return SDValue();
19729 
19730   // If the comparison is testing for a positive value, we have to invert
19731   // the sign bit mask, so only do that transform if the target has a bitwise
19732   // 'and not' instruction (the invert is free).
19733   if (CC == ISD::SETGT && TLI.hasAndNot(N2)) {
19734     // (X > -1) ? A : 0
19735     // (X >  0) ? X : 0 <-- This is canonical signed max.
19736     if (!(isAllOnesConstant(N1) || (isNullConstant(N1) && N0 == N2)))
19737       return SDValue();
19738   } else if (CC == ISD::SETLT) {
19739     // (X <  0) ? A : 0
19740     // (X <  1) ? X : 0 <-- This is un-canonicalized signed min.
19741     if (!(isNullConstant(N1) || (isOneConstant(N1) && N0 == N2)))
19742       return SDValue();
19743   } else {
19744     return SDValue();
19745   }
19746 
19747   // and (sra X, size(X)-1), A -> "and (srl X, C2), A" iff A is a single-bit
19748   // constant.
19749   EVT ShiftAmtTy = getShiftAmountTy(N0.getValueType());
19750   auto *N2C = dyn_cast<ConstantSDNode>(N2.getNode());
19751   if (N2C && ((N2C->getAPIntValue() & (N2C->getAPIntValue() - 1)) == 0)) {
19752     unsigned ShCt = XType.getSizeInBits() - N2C->getAPIntValue().logBase2() - 1;
19753     SDValue ShiftAmt = DAG.getConstant(ShCt, DL, ShiftAmtTy);
19754     SDValue Shift = DAG.getNode(ISD::SRL, DL, XType, N0, ShiftAmt);
19755     AddToWorklist(Shift.getNode());
19756 
19757     if (XType.bitsGT(AType)) {
19758       Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
19759       AddToWorklist(Shift.getNode());
19760     }
19761 
19762     if (CC == ISD::SETGT)
19763       Shift = DAG.getNOT(DL, Shift, AType);
19764 
19765     return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
19766   }
19767 
19768   SDValue ShiftAmt = DAG.getConstant(XType.getSizeInBits() - 1, DL, ShiftAmtTy);
19769   SDValue Shift = DAG.getNode(ISD::SRA, DL, XType, N0, ShiftAmt);
19770   AddToWorklist(Shift.getNode());
19771 
19772   if (XType.bitsGT(AType)) {
19773     Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
19774     AddToWorklist(Shift.getNode());
19775   }
19776 
19777   if (CC == ISD::SETGT)
19778     Shift = DAG.getNOT(DL, Shift, AType);
19779 
19780   return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
19781 }
19782 
19783 /// Turn "(a cond b) ? 1.0f : 2.0f" into "load (tmp + ((a cond b) ? 0 : 4)"
19784 /// where "tmp" is a constant pool entry containing an array with 1.0 and 2.0
19785 /// in it. This may be a win when the constant is not otherwise available
19786 /// because it replaces two constant pool loads with one.
19787 SDValue DAGCombiner::convertSelectOfFPConstantsToLoadOffset(
19788     const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3,
19789     ISD::CondCode CC) {
19790   if (!TLI.reduceSelectOfFPConstantLoads(N0.getValueType().isFloatingPoint()))
19791     return SDValue();
19792 
19793   // If we are before legalize types, we want the other legalization to happen
19794   // first (for example, to avoid messing with soft float).
19795   auto *TV = dyn_cast<ConstantFPSDNode>(N2);
19796   auto *FV = dyn_cast<ConstantFPSDNode>(N3);
19797   EVT VT = N2.getValueType();
19798   if (!TV || !FV || !TLI.isTypeLegal(VT))
19799     return SDValue();
19800 
19801   // If a constant can be materialized without loads, this does not make sense.
19802   if (TLI.getOperationAction(ISD::ConstantFP, VT) == TargetLowering::Legal ||
19803       TLI.isFPImmLegal(TV->getValueAPF(), TV->getValueType(0), ForCodeSize) ||
19804       TLI.isFPImmLegal(FV->getValueAPF(), FV->getValueType(0), ForCodeSize))
19805     return SDValue();
19806 
19807   // If both constants have multiple uses, then we won't need to do an extra
19808   // load. The values are likely around in registers for other users.
19809   if (!TV->hasOneUse() && !FV->hasOneUse())
19810     return SDValue();
19811 
19812   Constant *Elts[] = { const_cast<ConstantFP*>(FV->getConstantFPValue()),
19813                        const_cast<ConstantFP*>(TV->getConstantFPValue()) };
19814   Type *FPTy = Elts[0]->getType();
19815   const DataLayout &TD = DAG.getDataLayout();
19816 
19817   // Create a ConstantArray of the two constants.
19818   Constant *CA = ConstantArray::get(ArrayType::get(FPTy, 2), Elts);
19819   SDValue CPIdx = DAG.getConstantPool(CA, TLI.getPointerTy(DAG.getDataLayout()),
19820                                       TD.getPrefTypeAlignment(FPTy));
19821   unsigned Alignment = cast<ConstantPoolSDNode>(CPIdx)->getAlignment();
19822 
19823   // Get offsets to the 0 and 1 elements of the array, so we can select between
19824   // them.
19825   SDValue Zero = DAG.getIntPtrConstant(0, DL);
19826   unsigned EltSize = (unsigned)TD.getTypeAllocSize(Elts[0]->getType());
19827   SDValue One = DAG.getIntPtrConstant(EltSize, SDLoc(FV));
19828   SDValue Cond =
19829       DAG.getSetCC(DL, getSetCCResultType(N0.getValueType()), N0, N1, CC);
19830   AddToWorklist(Cond.getNode());
19831   SDValue CstOffset = DAG.getSelect(DL, Zero.getValueType(), Cond, One, Zero);
19832   AddToWorklist(CstOffset.getNode());
19833   CPIdx = DAG.getNode(ISD::ADD, DL, CPIdx.getValueType(), CPIdx, CstOffset);
19834   AddToWorklist(CPIdx.getNode());
19835   return DAG.getLoad(TV->getValueType(0), DL, DAG.getEntryNode(), CPIdx,
19836                      MachinePointerInfo::getConstantPool(
19837                          DAG.getMachineFunction()), Alignment);
19838 }
19839 
19840 /// Simplify an expression of the form (N0 cond N1) ? N2 : N3
19841 /// where 'cond' is the comparison specified by CC.
19842 SDValue DAGCombiner::SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
19843                                       SDValue N2, SDValue N3, ISD::CondCode CC,
19844                                       bool NotExtCompare) {
19845   // (x ? y : y) -> y.
19846   if (N2 == N3) return N2;
19847 
19848   EVT CmpOpVT = N0.getValueType();
19849   EVT CmpResVT = getSetCCResultType(CmpOpVT);
19850   EVT VT = N2.getValueType();
19851   auto *N1C = dyn_cast<ConstantSDNode>(N1.getNode());
19852   auto *N2C = dyn_cast<ConstantSDNode>(N2.getNode());
19853   auto *N3C = dyn_cast<ConstantSDNode>(N3.getNode());
19854 
19855   // Determine if the condition we're dealing with is constant.
19856   if (SDValue SCC = DAG.FoldSetCC(CmpResVT, N0, N1, CC, DL)) {
19857     AddToWorklist(SCC.getNode());
19858     if (auto *SCCC = dyn_cast<ConstantSDNode>(SCC)) {
19859       // fold select_cc true, x, y -> x
19860       // fold select_cc false, x, y -> y
19861       return !(SCCC->isNullValue()) ? N2 : N3;
19862     }
19863   }
19864 
19865   if (SDValue V =
19866           convertSelectOfFPConstantsToLoadOffset(DL, N0, N1, N2, N3, CC))
19867     return V;
19868 
19869   if (SDValue V = foldSelectCCToShiftAnd(DL, N0, N1, N2, N3, CC))
19870     return V;
19871 
19872   // fold (select_cc seteq (and x, y), 0, 0, A) -> (and (shr (shl x)) A)
19873   // where y is has a single bit set.
19874   // A plaintext description would be, we can turn the SELECT_CC into an AND
19875   // when the condition can be materialized as an all-ones register.  Any
19876   // single bit-test can be materialized as an all-ones register with
19877   // shift-left and shift-right-arith.
19878   if (CC == ISD::SETEQ && N0->getOpcode() == ISD::AND &&
19879       N0->getValueType(0) == VT && isNullConstant(N1) && isNullConstant(N2)) {
19880     SDValue AndLHS = N0->getOperand(0);
19881     auto *ConstAndRHS = dyn_cast<ConstantSDNode>(N0->getOperand(1));
19882     if (ConstAndRHS && ConstAndRHS->getAPIntValue().countPopulation() == 1) {
19883       // Shift the tested bit over the sign bit.
19884       const APInt &AndMask = ConstAndRHS->getAPIntValue();
19885       SDValue ShlAmt =
19886         DAG.getConstant(AndMask.countLeadingZeros(), SDLoc(AndLHS),
19887                         getShiftAmountTy(AndLHS.getValueType()));
19888       SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N0), VT, AndLHS, ShlAmt);
19889 
19890       // Now arithmetic right shift it all the way over, so the result is either
19891       // all-ones, or zero.
19892       SDValue ShrAmt =
19893         DAG.getConstant(AndMask.getBitWidth() - 1, SDLoc(Shl),
19894                         getShiftAmountTy(Shl.getValueType()));
19895       SDValue Shr = DAG.getNode(ISD::SRA, SDLoc(N0), VT, Shl, ShrAmt);
19896 
19897       return DAG.getNode(ISD::AND, DL, VT, Shr, N3);
19898     }
19899   }
19900 
19901   // fold select C, 16, 0 -> shl C, 4
19902   bool Fold = N2C && isNullConstant(N3) && N2C->getAPIntValue().isPowerOf2();
19903   bool Swap = N3C && isNullConstant(N2) && N3C->getAPIntValue().isPowerOf2();
19904 
19905   if ((Fold || Swap) &&
19906       TLI.getBooleanContents(CmpOpVT) ==
19907           TargetLowering::ZeroOrOneBooleanContent &&
19908       (!LegalOperations || TLI.isOperationLegal(ISD::SETCC, CmpOpVT))) {
19909 
19910     if (Swap) {
19911       CC = ISD::getSetCCInverse(CC, CmpOpVT.isInteger());
19912       std::swap(N2C, N3C);
19913     }
19914 
19915     // If the caller doesn't want us to simplify this into a zext of a compare,
19916     // don't do it.
19917     if (NotExtCompare && N2C->isOne())
19918       return SDValue();
19919 
19920     SDValue Temp, SCC;
19921     // zext (setcc n0, n1)
19922     if (LegalTypes) {
19923       SCC = DAG.getSetCC(DL, CmpResVT, N0, N1, CC);
19924       if (VT.bitsLT(SCC.getValueType()))
19925         Temp = DAG.getZeroExtendInReg(SCC, SDLoc(N2), VT);
19926       else
19927         Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2), VT, SCC);
19928     } else {
19929       SCC = DAG.getSetCC(SDLoc(N0), MVT::i1, N0, N1, CC);
19930       Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2), VT, SCC);
19931     }
19932 
19933     AddToWorklist(SCC.getNode());
19934     AddToWorklist(Temp.getNode());
19935 
19936     if (N2C->isOne())
19937       return Temp;
19938 
19939     // shl setcc result by log2 n2c
19940     return DAG.getNode(ISD::SHL, DL, N2.getValueType(), Temp,
19941                        DAG.getConstant(N2C->getAPIntValue().logBase2(),
19942                                        SDLoc(Temp),
19943                                        getShiftAmountTy(Temp.getValueType())));
19944   }
19945 
19946   // select_cc seteq X, 0, sizeof(X), ctlz(X) -> ctlz(X)
19947   // select_cc seteq X, 0, sizeof(X), ctlz_zero_undef(X) -> ctlz(X)
19948   // select_cc seteq X, 0, sizeof(X), cttz(X) -> cttz(X)
19949   // select_cc seteq X, 0, sizeof(X), cttz_zero_undef(X) -> cttz(X)
19950   // select_cc setne X, 0, ctlz(X), sizeof(X) -> ctlz(X)
19951   // select_cc setne X, 0, ctlz_zero_undef(X), sizeof(X) -> ctlz(X)
19952   // select_cc setne X, 0, cttz(X), sizeof(X) -> cttz(X)
19953   // select_cc setne X, 0, cttz_zero_undef(X), sizeof(X) -> cttz(X)
19954   if (N1C && N1C->isNullValue() && (CC == ISD::SETEQ || CC == ISD::SETNE)) {
19955     SDValue ValueOnZero = N2;
19956     SDValue Count = N3;
19957     // If the condition is NE instead of E, swap the operands.
19958     if (CC == ISD::SETNE)
19959       std::swap(ValueOnZero, Count);
19960     // Check if the value on zero is a constant equal to the bits in the type.
19961     if (auto *ValueOnZeroC = dyn_cast<ConstantSDNode>(ValueOnZero)) {
19962       if (ValueOnZeroC->getAPIntValue() == VT.getSizeInBits()) {
19963         // If the other operand is cttz/cttz_zero_undef of N0, and cttz is
19964         // legal, combine to just cttz.
19965         if ((Count.getOpcode() == ISD::CTTZ ||
19966              Count.getOpcode() == ISD::CTTZ_ZERO_UNDEF) &&
19967             N0 == Count.getOperand(0) &&
19968             (!LegalOperations || TLI.isOperationLegal(ISD::CTTZ, VT)))
19969           return DAG.getNode(ISD::CTTZ, DL, VT, N0);
19970         // If the other operand is ctlz/ctlz_zero_undef of N0, and ctlz is
19971         // legal, combine to just ctlz.
19972         if ((Count.getOpcode() == ISD::CTLZ ||
19973              Count.getOpcode() == ISD::CTLZ_ZERO_UNDEF) &&
19974             N0 == Count.getOperand(0) &&
19975             (!LegalOperations || TLI.isOperationLegal(ISD::CTLZ, VT)))
19976           return DAG.getNode(ISD::CTLZ, DL, VT, N0);
19977       }
19978     }
19979   }
19980 
19981   return SDValue();
19982 }
19983 
19984 /// This is a stub for TargetLowering::SimplifySetCC.
19985 SDValue DAGCombiner::SimplifySetCC(EVT VT, SDValue N0, SDValue N1,
19986                                    ISD::CondCode Cond, const SDLoc &DL,
19987                                    bool foldBooleans) {
19988   TargetLowering::DAGCombinerInfo
19989     DagCombineInfo(DAG, Level, false, this);
19990   return TLI.SimplifySetCC(VT, N0, N1, Cond, foldBooleans, DagCombineInfo, DL);
19991 }
19992 
19993 /// Given an ISD::SDIV node expressing a divide by constant, return
19994 /// a DAG expression to select that will generate the same value by multiplying
19995 /// by a magic number.
19996 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
19997 SDValue DAGCombiner::BuildSDIV(SDNode *N) {
19998   // when optimising for minimum size, we don't want to expand a div to a mul
19999   // and a shift.
20000   if (DAG.getMachineFunction().getFunction().hasMinSize())
20001     return SDValue();
20002 
20003   SmallVector<SDNode *, 8> Built;
20004   if (SDValue S = TLI.BuildSDIV(N, DAG, LegalOperations, Built)) {
20005     for (SDNode *N : Built)
20006       AddToWorklist(N);
20007     return S;
20008   }
20009 
20010   return SDValue();
20011 }
20012 
20013 /// Given an ISD::SDIV node expressing a divide by constant power of 2, return a
20014 /// DAG expression that will generate the same value by right shifting.
20015 SDValue DAGCombiner::BuildSDIVPow2(SDNode *N) {
20016   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
20017   if (!C)
20018     return SDValue();
20019 
20020   // Avoid division by zero.
20021   if (C->isNullValue())
20022     return SDValue();
20023 
20024   SmallVector<SDNode *, 8> Built;
20025   if (SDValue S = TLI.BuildSDIVPow2(N, C->getAPIntValue(), DAG, Built)) {
20026     for (SDNode *N : Built)
20027       AddToWorklist(N);
20028     return S;
20029   }
20030 
20031   return SDValue();
20032 }
20033 
20034 /// Given an ISD::UDIV node expressing a divide by constant, return a DAG
20035 /// expression that will generate the same value by multiplying by a magic
20036 /// number.
20037 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
20038 SDValue DAGCombiner::BuildUDIV(SDNode *N) {
20039   // when optimising for minimum size, we don't want to expand a div to a mul
20040   // and a shift.
20041   if (DAG.getMachineFunction().getFunction().hasMinSize())
20042     return SDValue();
20043 
20044   SmallVector<SDNode *, 8> Built;
20045   if (SDValue S = TLI.BuildUDIV(N, DAG, LegalOperations, Built)) {
20046     for (SDNode *N : Built)
20047       AddToWorklist(N);
20048     return S;
20049   }
20050 
20051   return SDValue();
20052 }
20053 
20054 /// Determines the LogBase2 value for a non-null input value using the
20055 /// transform: LogBase2(V) = (EltBits - 1) - ctlz(V).
20056 SDValue DAGCombiner::BuildLogBase2(SDValue V, const SDLoc &DL) {
20057   EVT VT = V.getValueType();
20058   unsigned EltBits = VT.getScalarSizeInBits();
20059   SDValue Ctlz = DAG.getNode(ISD::CTLZ, DL, VT, V);
20060   SDValue Base = DAG.getConstant(EltBits - 1, DL, VT);
20061   SDValue LogBase2 = DAG.getNode(ISD::SUB, DL, VT, Base, Ctlz);
20062   return LogBase2;
20063 }
20064 
20065 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
20066 /// For the reciprocal, we need to find the zero of the function:
20067 ///   F(X) = A X - 1 [which has a zero at X = 1/A]
20068 ///     =>
20069 ///   X_{i+1} = X_i (2 - A X_i) = X_i + X_i (1 - A X_i) [this second form
20070 ///     does not require additional intermediate precision]
20071 SDValue DAGCombiner::BuildReciprocalEstimate(SDValue Op, SDNodeFlags Flags) {
20072   if (Level >= AfterLegalizeDAG)
20073     return SDValue();
20074 
20075   // TODO: Handle half and/or extended types?
20076   EVT VT = Op.getValueType();
20077   if (VT.getScalarType() != MVT::f32 && VT.getScalarType() != MVT::f64)
20078     return SDValue();
20079 
20080   // If estimates are explicitly disabled for this function, we're done.
20081   MachineFunction &MF = DAG.getMachineFunction();
20082   int Enabled = TLI.getRecipEstimateDivEnabled(VT, MF);
20083   if (Enabled == TLI.ReciprocalEstimate::Disabled)
20084     return SDValue();
20085 
20086   // Estimates may be explicitly enabled for this type with a custom number of
20087   // refinement steps.
20088   int Iterations = TLI.getDivRefinementSteps(VT, MF);
20089   if (SDValue Est = TLI.getRecipEstimate(Op, DAG, Enabled, Iterations)) {
20090     AddToWorklist(Est.getNode());
20091 
20092     if (Iterations) {
20093       SDLoc DL(Op);
20094       SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
20095 
20096       // Newton iterations: Est = Est + Est (1 - Arg * Est)
20097       for (int i = 0; i < Iterations; ++i) {
20098         SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Op, Est, Flags);
20099         AddToWorklist(NewEst.getNode());
20100 
20101         NewEst = DAG.getNode(ISD::FSUB, DL, VT, FPOne, NewEst, Flags);
20102         AddToWorklist(NewEst.getNode());
20103 
20104         NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
20105         AddToWorklist(NewEst.getNode());
20106 
20107         Est = DAG.getNode(ISD::FADD, DL, VT, Est, NewEst, Flags);
20108         AddToWorklist(Est.getNode());
20109       }
20110     }
20111     return Est;
20112   }
20113 
20114   return SDValue();
20115 }
20116 
20117 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
20118 /// For the reciprocal sqrt, we need to find the zero of the function:
20119 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
20120 ///     =>
20121 ///   X_{i+1} = X_i (1.5 - A X_i^2 / 2)
20122 /// As a result, we precompute A/2 prior to the iteration loop.
20123 SDValue DAGCombiner::buildSqrtNROneConst(SDValue Arg, SDValue Est,
20124                                          unsigned Iterations,
20125                                          SDNodeFlags Flags, bool Reciprocal) {
20126   EVT VT = Arg.getValueType();
20127   SDLoc DL(Arg);
20128   SDValue ThreeHalves = DAG.getConstantFP(1.5, DL, VT);
20129 
20130   // We now need 0.5 * Arg which we can write as (1.5 * Arg - Arg) so that
20131   // this entire sequence requires only one FP constant.
20132   SDValue HalfArg = DAG.getNode(ISD::FMUL, DL, VT, ThreeHalves, Arg, Flags);
20133   AddToWorklist(HalfArg.getNode());
20134 
20135   HalfArg = DAG.getNode(ISD::FSUB, DL, VT, HalfArg, Arg, Flags);
20136   AddToWorklist(HalfArg.getNode());
20137 
20138   // Newton iterations: Est = Est * (1.5 - HalfArg * Est * Est)
20139   for (unsigned i = 0; i < Iterations; ++i) {
20140     SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, Est, Flags);
20141     AddToWorklist(NewEst.getNode());
20142 
20143     NewEst = DAG.getNode(ISD::FMUL, DL, VT, HalfArg, NewEst, Flags);
20144     AddToWorklist(NewEst.getNode());
20145 
20146     NewEst = DAG.getNode(ISD::FSUB, DL, VT, ThreeHalves, NewEst, Flags);
20147     AddToWorklist(NewEst.getNode());
20148 
20149     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
20150     AddToWorklist(Est.getNode());
20151   }
20152 
20153   // If non-reciprocal square root is requested, multiply the result by Arg.
20154   if (!Reciprocal) {
20155     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, Arg, Flags);
20156     AddToWorklist(Est.getNode());
20157   }
20158 
20159   return Est;
20160 }
20161 
20162 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
20163 /// For the reciprocal sqrt, we need to find the zero of the function:
20164 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
20165 ///     =>
20166 ///   X_{i+1} = (-0.5 * X_i) * (A * X_i * X_i + (-3.0))
20167 SDValue DAGCombiner::buildSqrtNRTwoConst(SDValue Arg, SDValue Est,
20168                                          unsigned Iterations,
20169                                          SDNodeFlags Flags, bool Reciprocal) {
20170   EVT VT = Arg.getValueType();
20171   SDLoc DL(Arg);
20172   SDValue MinusThree = DAG.getConstantFP(-3.0, DL, VT);
20173   SDValue MinusHalf = DAG.getConstantFP(-0.5, DL, VT);
20174 
20175   // This routine must enter the loop below to work correctly
20176   // when (Reciprocal == false).
20177   assert(Iterations > 0);
20178 
20179   // Newton iterations for reciprocal square root:
20180   // E = (E * -0.5) * ((A * E) * E + -3.0)
20181   for (unsigned i = 0; i < Iterations; ++i) {
20182     SDValue AE = DAG.getNode(ISD::FMUL, DL, VT, Arg, Est, Flags);
20183     AddToWorklist(AE.getNode());
20184 
20185     SDValue AEE = DAG.getNode(ISD::FMUL, DL, VT, AE, Est, Flags);
20186     AddToWorklist(AEE.getNode());
20187 
20188     SDValue RHS = DAG.getNode(ISD::FADD, DL, VT, AEE, MinusThree, Flags);
20189     AddToWorklist(RHS.getNode());
20190 
20191     // When calculating a square root at the last iteration build:
20192     // S = ((A * E) * -0.5) * ((A * E) * E + -3.0)
20193     // (notice a common subexpression)
20194     SDValue LHS;
20195     if (Reciprocal || (i + 1) < Iterations) {
20196       // RSQRT: LHS = (E * -0.5)
20197       LHS = DAG.getNode(ISD::FMUL, DL, VT, Est, MinusHalf, Flags);
20198     } else {
20199       // SQRT: LHS = (A * E) * -0.5
20200       LHS = DAG.getNode(ISD::FMUL, DL, VT, AE, MinusHalf, Flags);
20201     }
20202     AddToWorklist(LHS.getNode());
20203 
20204     Est = DAG.getNode(ISD::FMUL, DL, VT, LHS, RHS, Flags);
20205     AddToWorklist(Est.getNode());
20206   }
20207 
20208   return Est;
20209 }
20210 
20211 /// Build code to calculate either rsqrt(Op) or sqrt(Op). In the latter case
20212 /// Op*rsqrt(Op) is actually computed, so additional postprocessing is needed if
20213 /// Op can be zero.
20214 SDValue DAGCombiner::buildSqrtEstimateImpl(SDValue Op, SDNodeFlags Flags,
20215                                            bool Reciprocal) {
20216   if (Level >= AfterLegalizeDAG)
20217     return SDValue();
20218 
20219   // TODO: Handle half and/or extended types?
20220   EVT VT = Op.getValueType();
20221   if (VT.getScalarType() != MVT::f32 && VT.getScalarType() != MVT::f64)
20222     return SDValue();
20223 
20224   // If estimates are explicitly disabled for this function, we're done.
20225   MachineFunction &MF = DAG.getMachineFunction();
20226   int Enabled = TLI.getRecipEstimateSqrtEnabled(VT, MF);
20227   if (Enabled == TLI.ReciprocalEstimate::Disabled)
20228     return SDValue();
20229 
20230   // Estimates may be explicitly enabled for this type with a custom number of
20231   // refinement steps.
20232   int Iterations = TLI.getSqrtRefinementSteps(VT, MF);
20233 
20234   bool UseOneConstNR = false;
20235   if (SDValue Est =
20236       TLI.getSqrtEstimate(Op, DAG, Enabled, Iterations, UseOneConstNR,
20237                           Reciprocal)) {
20238     AddToWorklist(Est.getNode());
20239 
20240     if (Iterations) {
20241       Est = UseOneConstNR
20242             ? buildSqrtNROneConst(Op, Est, Iterations, Flags, Reciprocal)
20243             : buildSqrtNRTwoConst(Op, Est, Iterations, Flags, Reciprocal);
20244 
20245       if (!Reciprocal) {
20246         // The estimate is now completely wrong if the input was exactly 0.0 or
20247         // possibly a denormal. Force the answer to 0.0 for those cases.
20248         SDLoc DL(Op);
20249         EVT CCVT = getSetCCResultType(VT);
20250         ISD::NodeType SelOpcode = VT.isVector() ? ISD::VSELECT : ISD::SELECT;
20251         const Function &F = DAG.getMachineFunction().getFunction();
20252         Attribute Denorms = F.getFnAttribute("denormal-fp-math");
20253         if (Denorms.getValueAsString().equals("ieee")) {
20254           // fabs(X) < SmallestNormal ? 0.0 : Est
20255           const fltSemantics &FltSem = DAG.EVTToAPFloatSemantics(VT);
20256           APFloat SmallestNorm = APFloat::getSmallestNormalized(FltSem);
20257           SDValue NormC = DAG.getConstantFP(SmallestNorm, DL, VT);
20258           SDValue FPZero = DAG.getConstantFP(0.0, DL, VT);
20259           SDValue Fabs = DAG.getNode(ISD::FABS, DL, VT, Op);
20260           SDValue IsDenorm = DAG.getSetCC(DL, CCVT, Fabs, NormC, ISD::SETLT);
20261           Est = DAG.getNode(SelOpcode, DL, VT, IsDenorm, FPZero, Est);
20262           AddToWorklist(Fabs.getNode());
20263           AddToWorklist(IsDenorm.getNode());
20264           AddToWorklist(Est.getNode());
20265         } else {
20266           // X == 0.0 ? 0.0 : Est
20267           SDValue FPZero = DAG.getConstantFP(0.0, DL, VT);
20268           SDValue IsZero = DAG.getSetCC(DL, CCVT, Op, FPZero, ISD::SETEQ);
20269           Est = DAG.getNode(SelOpcode, DL, VT, IsZero, FPZero, Est);
20270           AddToWorklist(IsZero.getNode());
20271           AddToWorklist(Est.getNode());
20272         }
20273       }
20274     }
20275     return Est;
20276   }
20277 
20278   return SDValue();
20279 }
20280 
20281 SDValue DAGCombiner::buildRsqrtEstimate(SDValue Op, SDNodeFlags Flags) {
20282   return buildSqrtEstimateImpl(Op, Flags, true);
20283 }
20284 
20285 SDValue DAGCombiner::buildSqrtEstimate(SDValue Op, SDNodeFlags Flags) {
20286   return buildSqrtEstimateImpl(Op, Flags, false);
20287 }
20288 
20289 /// Return true if there is any possibility that the two addresses overlap.
20290 bool DAGCombiner::isAlias(SDNode *Op0, SDNode *Op1) const {
20291 
20292   struct MemUseCharacteristics {
20293     bool IsVolatile;
20294     SDValue BasePtr;
20295     int64_t Offset;
20296     Optional<int64_t> NumBytes;
20297     MachineMemOperand *MMO;
20298   };
20299 
20300   auto getCharacteristics = [](SDNode *N) -> MemUseCharacteristics {
20301     if (const auto *LSN = dyn_cast<LSBaseSDNode>(N)) {
20302       int64_t Offset = 0;
20303       if (auto *C = dyn_cast<ConstantSDNode>(LSN->getOffset()))
20304         Offset = (LSN->getAddressingMode() == ISD::PRE_INC)
20305                      ? C->getSExtValue()
20306                      : (LSN->getAddressingMode() == ISD::PRE_DEC)
20307                            ? -1 * C->getSExtValue()
20308                            : 0;
20309       return {LSN->isVolatile(), LSN->getBasePtr(), Offset /*base offset*/,
20310               Optional<int64_t>(LSN->getMemoryVT().getStoreSize()),
20311               LSN->getMemOperand()};
20312     }
20313     if (const auto *LN = cast<LifetimeSDNode>(N))
20314       return {false /*isVolatile*/, LN->getOperand(1),
20315               (LN->hasOffset()) ? LN->getOffset() : 0,
20316               (LN->hasOffset()) ? Optional<int64_t>(LN->getSize())
20317                                 : Optional<int64_t>(),
20318               (MachineMemOperand *)nullptr};
20319     // Default.
20320     return {false /*isvolatile*/, SDValue(), (int64_t)0 /*offset*/,
20321             Optional<int64_t>() /*size*/, (MachineMemOperand *)nullptr};
20322   };
20323 
20324   MemUseCharacteristics MUC0 = getCharacteristics(Op0),
20325                         MUC1 = getCharacteristics(Op1);
20326 
20327   // If they are to the same address, then they must be aliases.
20328   if (MUC0.BasePtr.getNode() && MUC0.BasePtr == MUC1.BasePtr &&
20329       MUC0.Offset == MUC1.Offset)
20330     return true;
20331 
20332   // If they are both volatile then they cannot be reordered.
20333   if (MUC0.IsVolatile && MUC1.IsVolatile)
20334     return true;
20335 
20336   if (MUC0.MMO && MUC1.MMO) {
20337     if ((MUC0.MMO->isInvariant() && MUC1.MMO->isStore()) ||
20338         (MUC1.MMO->isInvariant() && MUC0.MMO->isStore()))
20339       return false;
20340   }
20341 
20342   // Try to prove that there is aliasing, or that there is no aliasing. Either
20343   // way, we can return now. If nothing can be proved, proceed with more tests.
20344   bool IsAlias;
20345   if (BaseIndexOffset::computeAliasing(Op0, MUC0.NumBytes, Op1, MUC1.NumBytes,
20346                                        DAG, IsAlias))
20347     return IsAlias;
20348 
20349   // The following all rely on MMO0 and MMO1 being valid. Fail conservatively if
20350   // either are not known.
20351   if (!MUC0.MMO || !MUC1.MMO)
20352     return true;
20353 
20354   // If one operation reads from invariant memory, and the other may store, they
20355   // cannot alias. These should really be checking the equivalent of mayWrite,
20356   // but it only matters for memory nodes other than load /store.
20357   if ((MUC0.MMO->isInvariant() && MUC1.MMO->isStore()) ||
20358       (MUC1.MMO->isInvariant() && MUC0.MMO->isStore()))
20359     return false;
20360 
20361   // If we know required SrcValue1 and SrcValue2 have relatively large
20362   // alignment compared to the size and offset of the access, we may be able
20363   // to prove they do not alias. This check is conservative for now to catch
20364   // cases created by splitting vector types.
20365   int64_t SrcValOffset0 = MUC0.MMO->getOffset();
20366   int64_t SrcValOffset1 = MUC1.MMO->getOffset();
20367   unsigned OrigAlignment0 = MUC0.MMO->getBaseAlignment();
20368   unsigned OrigAlignment1 = MUC1.MMO->getBaseAlignment();
20369   if (OrigAlignment0 == OrigAlignment1 && SrcValOffset0 != SrcValOffset1 &&
20370       MUC0.NumBytes.hasValue() && MUC1.NumBytes.hasValue() &&
20371       *MUC0.NumBytes == *MUC1.NumBytes && OrigAlignment0 > *MUC0.NumBytes) {
20372     int64_t OffAlign0 = SrcValOffset0 % OrigAlignment0;
20373     int64_t OffAlign1 = SrcValOffset1 % OrigAlignment1;
20374 
20375     // There is no overlap between these relatively aligned accesses of
20376     // similar size. Return no alias.
20377     if ((OffAlign0 + *MUC0.NumBytes) <= OffAlign1 ||
20378         (OffAlign1 + *MUC1.NumBytes) <= OffAlign0)
20379       return false;
20380   }
20381 
20382   bool UseAA = CombinerGlobalAA.getNumOccurrences() > 0
20383                    ? CombinerGlobalAA
20384                    : DAG.getSubtarget().useAA();
20385 #ifndef NDEBUG
20386   if (CombinerAAOnlyFunc.getNumOccurrences() &&
20387       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
20388     UseAA = false;
20389 #endif
20390 
20391   if (UseAA && AA && MUC0.MMO->getValue() && MUC1.MMO->getValue()) {
20392     // Use alias analysis information.
20393     int64_t MinOffset = std::min(SrcValOffset0, SrcValOffset1);
20394     int64_t Overlap0 = *MUC0.NumBytes + SrcValOffset0 - MinOffset;
20395     int64_t Overlap1 = *MUC1.NumBytes + SrcValOffset1 - MinOffset;
20396     AliasResult AAResult = AA->alias(
20397         MemoryLocation(MUC0.MMO->getValue(), Overlap0,
20398                        UseTBAA ? MUC0.MMO->getAAInfo() : AAMDNodes()),
20399         MemoryLocation(MUC1.MMO->getValue(), Overlap1,
20400                        UseTBAA ? MUC1.MMO->getAAInfo() : AAMDNodes()));
20401     if (AAResult == NoAlias)
20402       return false;
20403   }
20404 
20405   // Otherwise we have to assume they alias.
20406   return true;
20407 }
20408 
20409 /// Walk up chain skipping non-aliasing memory nodes,
20410 /// looking for aliasing nodes and adding them to the Aliases vector.
20411 void DAGCombiner::GatherAllAliases(SDNode *N, SDValue OriginalChain,
20412                                    SmallVectorImpl<SDValue> &Aliases) {
20413   SmallVector<SDValue, 8> Chains;     // List of chains to visit.
20414   SmallPtrSet<SDNode *, 16> Visited;  // Visited node set.
20415 
20416   // Get alias information for node.
20417   const bool IsLoad = isa<LoadSDNode>(N) && !cast<LoadSDNode>(N)->isVolatile();
20418 
20419   // Starting off.
20420   Chains.push_back(OriginalChain);
20421   unsigned Depth = 0;
20422 
20423   // Attempt to improve chain by a single step
20424   std::function<bool(SDValue &)> ImproveChain = [&](SDValue &C) -> bool {
20425     switch (C.getOpcode()) {
20426     case ISD::EntryToken:
20427       // No need to mark EntryToken.
20428       C = SDValue();
20429       return true;
20430     case ISD::LOAD:
20431     case ISD::STORE: {
20432       // Get alias information for C.
20433       bool IsOpLoad = isa<LoadSDNode>(C.getNode()) &&
20434                       !cast<LSBaseSDNode>(C.getNode())->isVolatile();
20435       if ((IsLoad && IsOpLoad) || !isAlias(N, C.getNode())) {
20436         // Look further up the chain.
20437         C = C.getOperand(0);
20438         return true;
20439       }
20440       // Alias, so stop here.
20441       return false;
20442     }
20443 
20444     case ISD::CopyFromReg:
20445       // Always forward past past CopyFromReg.
20446       C = C.getOperand(0);
20447       return true;
20448 
20449     case ISD::LIFETIME_START:
20450     case ISD::LIFETIME_END: {
20451       // We can forward past any lifetime start/end that can be proven not to
20452       // alias the memory access.
20453       if (!isAlias(N, C.getNode())) {
20454         // Look further up the chain.
20455         C = C.getOperand(0);
20456         return true;
20457       }
20458       return false;
20459     }
20460     default:
20461       return false;
20462     }
20463   };
20464 
20465   // Look at each chain and determine if it is an alias.  If so, add it to the
20466   // aliases list.  If not, then continue up the chain looking for the next
20467   // candidate.
20468   while (!Chains.empty()) {
20469     SDValue Chain = Chains.pop_back_val();
20470 
20471     // Don't bother if we've seen Chain before.
20472     if (!Visited.insert(Chain.getNode()).second)
20473       continue;
20474 
20475     // For TokenFactor nodes, look at each operand and only continue up the
20476     // chain until we reach the depth limit.
20477     //
20478     // FIXME: The depth check could be made to return the last non-aliasing
20479     // chain we found before we hit a tokenfactor rather than the original
20480     // chain.
20481     if (Depth > TLI.getGatherAllAliasesMaxDepth()) {
20482       Aliases.clear();
20483       Aliases.push_back(OriginalChain);
20484       return;
20485     }
20486 
20487     if (Chain.getOpcode() == ISD::TokenFactor) {
20488       // We have to check each of the operands of the token factor for "small"
20489       // token factors, so we queue them up.  Adding the operands to the queue
20490       // (stack) in reverse order maintains the original order and increases the
20491       // likelihood that getNode will find a matching token factor (CSE.)
20492       if (Chain.getNumOperands() > 16) {
20493         Aliases.push_back(Chain);
20494         continue;
20495       }
20496       for (unsigned n = Chain.getNumOperands(); n;)
20497         Chains.push_back(Chain.getOperand(--n));
20498       ++Depth;
20499       continue;
20500     }
20501     // Everything else
20502     if (ImproveChain(Chain)) {
20503       // Updated Chain Found, Consider new chain if one exists.
20504       if (Chain.getNode())
20505         Chains.push_back(Chain);
20506       ++Depth;
20507       continue;
20508     }
20509     // No Improved Chain Possible, treat as Alias.
20510     Aliases.push_back(Chain);
20511   }
20512 }
20513 
20514 /// Walk up chain skipping non-aliasing memory nodes, looking for a better chain
20515 /// (aliasing node.)
20516 SDValue DAGCombiner::FindBetterChain(SDNode *N, SDValue OldChain) {
20517   if (OptLevel == CodeGenOpt::None)
20518     return OldChain;
20519 
20520   // Ops for replacing token factor.
20521   SmallVector<SDValue, 8> Aliases;
20522 
20523   // Accumulate all the aliases to this node.
20524   GatherAllAliases(N, OldChain, Aliases);
20525 
20526   // If no operands then chain to entry token.
20527   if (Aliases.size() == 0)
20528     return DAG.getEntryNode();
20529 
20530   // If a single operand then chain to it.  We don't need to revisit it.
20531   if (Aliases.size() == 1)
20532     return Aliases[0];
20533 
20534   // Construct a custom tailored token factor.
20535   return DAG.getTokenFactor(SDLoc(N), Aliases);
20536 }
20537 
20538 namespace {
20539 // TODO: Replace with with std::monostate when we move to C++17.
20540 struct UnitT { } Unit;
20541 bool operator==(const UnitT &, const UnitT &) { return true; }
20542 bool operator!=(const UnitT &, const UnitT &) { return false; }
20543 } // namespace
20544 
20545 // This function tries to collect a bunch of potentially interesting
20546 // nodes to improve the chains of, all at once. This might seem
20547 // redundant, as this function gets called when visiting every store
20548 // node, so why not let the work be done on each store as it's visited?
20549 //
20550 // I believe this is mainly important because MergeConsecutiveStores
20551 // is unable to deal with merging stores of different sizes, so unless
20552 // we improve the chains of all the potential candidates up-front
20553 // before running MergeConsecutiveStores, it might only see some of
20554 // the nodes that will eventually be candidates, and then not be able
20555 // to go from a partially-merged state to the desired final
20556 // fully-merged state.
20557 
20558 bool DAGCombiner::parallelizeChainedStores(StoreSDNode *St) {
20559   SmallVector<StoreSDNode *, 8> ChainedStores;
20560   StoreSDNode *STChain = St;
20561   // Intervals records which offsets from BaseIndex have been covered. In
20562   // the common case, every store writes to the immediately previous address
20563   // space and thus merged with the previous interval at insertion time.
20564 
20565   using IMap =
20566       llvm::IntervalMap<int64_t, UnitT, 8, IntervalMapHalfOpenInfo<int64_t>>;
20567   IMap::Allocator A;
20568   IMap Intervals(A);
20569 
20570   // This holds the base pointer, index, and the offset in bytes from the base
20571   // pointer.
20572   const BaseIndexOffset BasePtr = BaseIndexOffset::match(St, DAG);
20573 
20574   // We must have a base and an offset.
20575   if (!BasePtr.getBase().getNode())
20576     return false;
20577 
20578   // Do not handle stores to undef base pointers.
20579   if (BasePtr.getBase().isUndef())
20580     return false;
20581 
20582   // Add ST's interval.
20583   Intervals.insert(0, (St->getMemoryVT().getSizeInBits() + 7) / 8, Unit);
20584 
20585   while (StoreSDNode *Chain = dyn_cast<StoreSDNode>(STChain->getChain())) {
20586     // If the chain has more than one use, then we can't reorder the mem ops.
20587     if (!SDValue(Chain, 0)->hasOneUse())
20588       break;
20589     if (Chain->isVolatile() || Chain->isIndexed())
20590       break;
20591 
20592     // Find the base pointer and offset for this memory node.
20593     const BaseIndexOffset Ptr = BaseIndexOffset::match(Chain, DAG);
20594     // Check that the base pointer is the same as the original one.
20595     int64_t Offset;
20596     if (!BasePtr.equalBaseIndex(Ptr, DAG, Offset))
20597       break;
20598     int64_t Length = (Chain->getMemoryVT().getSizeInBits() + 7) / 8;
20599     // Make sure we don't overlap with other intervals by checking the ones to
20600     // the left or right before inserting.
20601     auto I = Intervals.find(Offset);
20602     // If there's a next interval, we should end before it.
20603     if (I != Intervals.end() && I.start() < (Offset + Length))
20604       break;
20605     // If there's a previous interval, we should start after it.
20606     if (I != Intervals.begin() && (--I).stop() <= Offset)
20607       break;
20608     Intervals.insert(Offset, Offset + Length, Unit);
20609 
20610     ChainedStores.push_back(Chain);
20611     STChain = Chain;
20612   }
20613 
20614   // If we didn't find a chained store, exit.
20615   if (ChainedStores.size() == 0)
20616     return false;
20617 
20618   // Improve all chained stores (St and ChainedStores members) starting from
20619   // where the store chain ended and return single TokenFactor.
20620   SDValue NewChain = STChain->getChain();
20621   SmallVector<SDValue, 8> TFOps;
20622   for (unsigned I = ChainedStores.size(); I;) {
20623     StoreSDNode *S = ChainedStores[--I];
20624     SDValue BetterChain = FindBetterChain(S, NewChain);
20625     S = cast<StoreSDNode>(DAG.UpdateNodeOperands(
20626         S, BetterChain, S->getOperand(1), S->getOperand(2), S->getOperand(3)));
20627     TFOps.push_back(SDValue(S, 0));
20628     ChainedStores[I] = S;
20629   }
20630 
20631   // Improve St's chain. Use a new node to avoid creating a loop from CombineTo.
20632   SDValue BetterChain = FindBetterChain(St, NewChain);
20633   SDValue NewST;
20634   if (St->isTruncatingStore())
20635     NewST = DAG.getTruncStore(BetterChain, SDLoc(St), St->getValue(),
20636                               St->getBasePtr(), St->getMemoryVT(),
20637                               St->getMemOperand());
20638   else
20639     NewST = DAG.getStore(BetterChain, SDLoc(St), St->getValue(),
20640                          St->getBasePtr(), St->getMemOperand());
20641 
20642   TFOps.push_back(NewST);
20643 
20644   // If we improved every element of TFOps, then we've lost the dependence on
20645   // NewChain to successors of St and we need to add it back to TFOps. Do so at
20646   // the beginning to keep relative order consistent with FindBetterChains.
20647   auto hasImprovedChain = [&](SDValue ST) -> bool {
20648     return ST->getOperand(0) != NewChain;
20649   };
20650   bool AddNewChain = llvm::all_of(TFOps, hasImprovedChain);
20651   if (AddNewChain)
20652     TFOps.insert(TFOps.begin(), NewChain);
20653 
20654   SDValue TF = DAG.getTokenFactor(SDLoc(STChain), TFOps);
20655   CombineTo(St, TF);
20656 
20657   AddToWorklist(STChain);
20658   // Add TF operands worklist in reverse order.
20659   for (auto I = TF->getNumOperands(); I;)
20660     AddToWorklist(TF->getOperand(--I).getNode());
20661   AddToWorklist(TF.getNode());
20662   return true;
20663 }
20664 
20665 bool DAGCombiner::findBetterNeighborChains(StoreSDNode *St) {
20666   if (OptLevel == CodeGenOpt::None)
20667     return false;
20668 
20669   const BaseIndexOffset BasePtr = BaseIndexOffset::match(St, DAG);
20670 
20671   // We must have a base and an offset.
20672   if (!BasePtr.getBase().getNode())
20673     return false;
20674 
20675   // Do not handle stores to undef base pointers.
20676   if (BasePtr.getBase().isUndef())
20677     return false;
20678 
20679   // Directly improve a chain of disjoint stores starting at St.
20680   if (parallelizeChainedStores(St))
20681     return true;
20682 
20683   // Improve St's Chain..
20684   SDValue BetterChain = FindBetterChain(St, St->getChain());
20685   if (St->getChain() != BetterChain) {
20686     replaceStoreChain(St, BetterChain);
20687     return true;
20688   }
20689   return false;
20690 }
20691 
20692 /// This is the entry point for the file.
20693 void SelectionDAG::Combine(CombineLevel Level, AliasAnalysis *AA,
20694                            CodeGenOpt::Level OptLevel) {
20695   /// This is the main entry point to this class.
20696   DAGCombiner(*this, AA, OptLevel).Run(Level);
20697 }
20698