1 //===-- DAGCombiner.cpp - Implement a DAG node combiner -------------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 // This pass combines dag nodes to form fewer, simpler DAG nodes.  It can be run
11 // both before and after the DAG is legalized.
12 //
13 // This pass is not a substitute for the LLVM IR instcombine pass. This pass is
14 // primarily intended to handle simplification opportunities that are implicit
15 // in the LLVM IR and exposed by the various codegen lowering phases.
16 //
17 //===----------------------------------------------------------------------===//
18 
19 #include "llvm/CodeGen/SelectionDAG.h"
20 #include "llvm/ADT/SetVector.h"
21 #include "llvm/ADT/SmallBitVector.h"
22 #include "llvm/ADT/SmallPtrSet.h"
23 #include "llvm/ADT/Statistic.h"
24 #include "llvm/Analysis/AliasAnalysis.h"
25 #include "llvm/CodeGen/MachineFrameInfo.h"
26 #include "llvm/CodeGen/MachineFunction.h"
27 #include "llvm/CodeGen/SelectionDAGTargetInfo.h"
28 #include "llvm/IR/DataLayout.h"
29 #include "llvm/IR/DerivedTypes.h"
30 #include "llvm/IR/Function.h"
31 #include "llvm/IR/LLVMContext.h"
32 #include "llvm/Support/CommandLine.h"
33 #include "llvm/Support/Debug.h"
34 #include "llvm/Support/ErrorHandling.h"
35 #include "llvm/Support/MathExtras.h"
36 #include "llvm/Support/raw_ostream.h"
37 #include "llvm/Target/TargetLowering.h"
38 #include "llvm/Target/TargetOptions.h"
39 #include "llvm/Target/TargetRegisterInfo.h"
40 #include "llvm/Target/TargetSubtargetInfo.h"
41 #include <algorithm>
42 using namespace llvm;
43 
44 #define DEBUG_TYPE "dagcombine"
45 
46 STATISTIC(NodesCombined   , "Number of dag nodes combined");
47 STATISTIC(PreIndexedNodes , "Number of pre-indexed nodes created");
48 STATISTIC(PostIndexedNodes, "Number of post-indexed nodes created");
49 STATISTIC(OpsNarrowed     , "Number of load/op/store narrowed");
50 STATISTIC(LdStFP2Int      , "Number of fp load/store pairs transformed to int");
51 STATISTIC(SlicedLoads, "Number of load sliced");
52 
53 namespace {
54   static cl::opt<bool>
55     CombinerAA("combiner-alias-analysis", cl::Hidden,
56                cl::desc("Enable DAG combiner alias-analysis heuristics"));
57 
58   static cl::opt<bool>
59     CombinerGlobalAA("combiner-global-alias-analysis", cl::Hidden,
60                cl::desc("Enable DAG combiner's use of IR alias analysis"));
61 
62   static cl::opt<bool>
63     UseTBAA("combiner-use-tbaa", cl::Hidden, cl::init(true),
64                cl::desc("Enable DAG combiner's use of TBAA"));
65 
66 #ifndef NDEBUG
67   static cl::opt<std::string>
68     CombinerAAOnlyFunc("combiner-aa-only-func", cl::Hidden,
69                cl::desc("Only use DAG-combiner alias analysis in this"
70                         " function"));
71 #endif
72 
73   /// Hidden option to stress test load slicing, i.e., when this option
74   /// is enabled, load slicing bypasses most of its profitability guards.
75   static cl::opt<bool>
76   StressLoadSlicing("combiner-stress-load-slicing", cl::Hidden,
77                     cl::desc("Bypass the profitability model of load "
78                              "slicing"),
79                     cl::init(false));
80 
81   static cl::opt<bool>
82     MaySplitLoadIndex("combiner-split-load-index", cl::Hidden, cl::init(true),
83                       cl::desc("DAG combiner may split indexing from loads"));
84 
85 //------------------------------ DAGCombiner ---------------------------------//
86 
87   class DAGCombiner {
88     SelectionDAG &DAG;
89     const TargetLowering &TLI;
90     CombineLevel Level;
91     CodeGenOpt::Level OptLevel;
92     bool LegalOperations;
93     bool LegalTypes;
94     bool ForCodeSize;
95 
96     /// \brief Worklist of all of the nodes that need to be simplified.
97     ///
98     /// This must behave as a stack -- new nodes to process are pushed onto the
99     /// back and when processing we pop off of the back.
100     ///
101     /// The worklist will not contain duplicates but may contain null entries
102     /// due to nodes being deleted from the underlying DAG.
103     SmallVector<SDNode *, 64> Worklist;
104 
105     /// \brief Mapping from an SDNode to its position on the worklist.
106     ///
107     /// This is used to find and remove nodes from the worklist (by nulling
108     /// them) when they are deleted from the underlying DAG. It relies on
109     /// stable indices of nodes within the worklist.
110     DenseMap<SDNode *, unsigned> WorklistMap;
111 
112     /// \brief Set of nodes which have been combined (at least once).
113     ///
114     /// This is used to allow us to reliably add any operands of a DAG node
115     /// which have not yet been combined to the worklist.
116     SmallPtrSet<SDNode *, 32> CombinedNodes;
117 
118     // AA - Used for DAG load/store alias analysis.
119     AliasAnalysis &AA;
120 
121     /// When an instruction is simplified, add all users of the instruction to
122     /// the work lists because they might get more simplified now.
123     void AddUsersToWorklist(SDNode *N) {
124       for (SDNode *Node : N->uses())
125         AddToWorklist(Node);
126     }
127 
128     /// Call the node-specific routine that folds each particular type of node.
129     SDValue visit(SDNode *N);
130 
131   public:
132     /// Add to the worklist making sure its instance is at the back (next to be
133     /// processed.)
134     void AddToWorklist(SDNode *N) {
135       // Skip handle nodes as they can't usefully be combined and confuse the
136       // zero-use deletion strategy.
137       if (N->getOpcode() == ISD::HANDLENODE)
138         return;
139 
140       if (WorklistMap.insert(std::make_pair(N, Worklist.size())).second)
141         Worklist.push_back(N);
142     }
143 
144     /// Remove all instances of N from the worklist.
145     void removeFromWorklist(SDNode *N) {
146       CombinedNodes.erase(N);
147 
148       auto It = WorklistMap.find(N);
149       if (It == WorklistMap.end())
150         return; // Not in the worklist.
151 
152       // Null out the entry rather than erasing it to avoid a linear operation.
153       Worklist[It->second] = nullptr;
154       WorklistMap.erase(It);
155     }
156 
157     void deleteAndRecombine(SDNode *N);
158     bool recursivelyDeleteUnusedNodes(SDNode *N);
159 
160     /// Replaces all uses of the results of one DAG node with new values.
161     SDValue CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
162                       bool AddTo = true);
163 
164     /// Replaces all uses of the results of one DAG node with new values.
165     SDValue CombineTo(SDNode *N, SDValue Res, bool AddTo = true) {
166       return CombineTo(N, &Res, 1, AddTo);
167     }
168 
169     /// Replaces all uses of the results of one DAG node with new values.
170     SDValue CombineTo(SDNode *N, SDValue Res0, SDValue Res1,
171                       bool AddTo = true) {
172       SDValue To[] = { Res0, Res1 };
173       return CombineTo(N, To, 2, AddTo);
174     }
175 
176     void CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO);
177 
178   private:
179 
180     /// Check the specified integer node value to see if it can be simplified or
181     /// if things it uses can be simplified by bit propagation.
182     /// If so, return true.
183     bool SimplifyDemandedBits(SDValue Op) {
184       unsigned BitWidth = Op.getScalarValueSizeInBits();
185       APInt Demanded = APInt::getAllOnesValue(BitWidth);
186       return SimplifyDemandedBits(Op, Demanded);
187     }
188 
189     bool SimplifyDemandedBits(SDValue Op, const APInt &Demanded);
190 
191     bool CombineToPreIndexedLoadStore(SDNode *N);
192     bool CombineToPostIndexedLoadStore(SDNode *N);
193     SDValue SplitIndexingFromLoad(LoadSDNode *LD);
194     bool SliceUpLoad(SDNode *N);
195 
196     /// \brief Replace an ISD::EXTRACT_VECTOR_ELT of a load with a narrowed
197     ///   load.
198     ///
199     /// \param EVE ISD::EXTRACT_VECTOR_ELT to be replaced.
200     /// \param InVecVT type of the input vector to EVE with bitcasts resolved.
201     /// \param EltNo index of the vector element to load.
202     /// \param OriginalLoad load that EVE came from to be replaced.
203     /// \returns EVE on success SDValue() on failure.
204     SDValue ReplaceExtractVectorEltOfLoadWithNarrowedLoad(
205         SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad);
206     void ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad);
207     SDValue PromoteOperand(SDValue Op, EVT PVT, bool &Replace);
208     SDValue SExtPromoteOperand(SDValue Op, EVT PVT);
209     SDValue ZExtPromoteOperand(SDValue Op, EVT PVT);
210     SDValue PromoteIntBinOp(SDValue Op);
211     SDValue PromoteIntShiftOp(SDValue Op);
212     SDValue PromoteExtend(SDValue Op);
213     bool PromoteLoad(SDValue Op);
214 
215     void ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs, SDValue Trunc,
216                          SDValue ExtLoad, const SDLoc &DL,
217                          ISD::NodeType ExtType);
218 
219     /// Call the node-specific routine that knows how to fold each
220     /// particular type of node. If that doesn't do anything, try the
221     /// target-specific DAG combines.
222     SDValue combine(SDNode *N);
223 
224     // Visitation implementation - Implement dag node combining for different
225     // node types.  The semantics are as follows:
226     // Return Value:
227     //   SDValue.getNode() == 0 - No change was made
228     //   SDValue.getNode() == N - N was replaced, is dead and has been handled.
229     //   otherwise              - N should be replaced by the returned Operand.
230     //
231     SDValue visitTokenFactor(SDNode *N);
232     SDValue visitMERGE_VALUES(SDNode *N);
233     SDValue visitADD(SDNode *N);
234     SDValue visitSUB(SDNode *N);
235     SDValue visitADDC(SDNode *N);
236     SDValue visitSUBC(SDNode *N);
237     SDValue visitADDE(SDNode *N);
238     SDValue visitSUBE(SDNode *N);
239     SDValue visitMUL(SDNode *N);
240     SDValue useDivRem(SDNode *N);
241     SDValue visitSDIV(SDNode *N);
242     SDValue visitUDIV(SDNode *N);
243     SDValue visitREM(SDNode *N);
244     SDValue visitMULHU(SDNode *N);
245     SDValue visitMULHS(SDNode *N);
246     SDValue visitSMUL_LOHI(SDNode *N);
247     SDValue visitUMUL_LOHI(SDNode *N);
248     SDValue visitSMULO(SDNode *N);
249     SDValue visitUMULO(SDNode *N);
250     SDValue visitIMINMAX(SDNode *N);
251     SDValue visitAND(SDNode *N);
252     SDValue visitANDLike(SDValue N0, SDValue N1, SDNode *LocReference);
253     SDValue visitOR(SDNode *N);
254     SDValue visitORLike(SDValue N0, SDValue N1, SDNode *LocReference);
255     SDValue visitXOR(SDNode *N);
256     SDValue SimplifyVBinOp(SDNode *N);
257     SDValue visitSHL(SDNode *N);
258     SDValue visitSRA(SDNode *N);
259     SDValue visitSRL(SDNode *N);
260     SDValue visitRotate(SDNode *N);
261     SDValue visitBSWAP(SDNode *N);
262     SDValue visitBITREVERSE(SDNode *N);
263     SDValue visitCTLZ(SDNode *N);
264     SDValue visitCTLZ_ZERO_UNDEF(SDNode *N);
265     SDValue visitCTTZ(SDNode *N);
266     SDValue visitCTTZ_ZERO_UNDEF(SDNode *N);
267     SDValue visitCTPOP(SDNode *N);
268     SDValue visitSELECT(SDNode *N);
269     SDValue visitVSELECT(SDNode *N);
270     SDValue visitSELECT_CC(SDNode *N);
271     SDValue visitSETCC(SDNode *N);
272     SDValue visitSETCCE(SDNode *N);
273     SDValue visitSIGN_EXTEND(SDNode *N);
274     SDValue visitZERO_EXTEND(SDNode *N);
275     SDValue visitANY_EXTEND(SDNode *N);
276     SDValue visitSIGN_EXTEND_INREG(SDNode *N);
277     SDValue visitSIGN_EXTEND_VECTOR_INREG(SDNode *N);
278     SDValue visitZERO_EXTEND_VECTOR_INREG(SDNode *N);
279     SDValue visitTRUNCATE(SDNode *N);
280     SDValue visitBITCAST(SDNode *N);
281     SDValue visitBUILD_PAIR(SDNode *N);
282     SDValue visitFADD(SDNode *N);
283     SDValue visitFSUB(SDNode *N);
284     SDValue visitFMUL(SDNode *N);
285     SDValue visitFMA(SDNode *N);
286     SDValue visitFDIV(SDNode *N);
287     SDValue visitFREM(SDNode *N);
288     SDValue visitFSQRT(SDNode *N);
289     SDValue visitFCOPYSIGN(SDNode *N);
290     SDValue visitSINT_TO_FP(SDNode *N);
291     SDValue visitUINT_TO_FP(SDNode *N);
292     SDValue visitFP_TO_SINT(SDNode *N);
293     SDValue visitFP_TO_UINT(SDNode *N);
294     SDValue visitFP_ROUND(SDNode *N);
295     SDValue visitFP_ROUND_INREG(SDNode *N);
296     SDValue visitFP_EXTEND(SDNode *N);
297     SDValue visitFNEG(SDNode *N);
298     SDValue visitFABS(SDNode *N);
299     SDValue visitFCEIL(SDNode *N);
300     SDValue visitFTRUNC(SDNode *N);
301     SDValue visitFFLOOR(SDNode *N);
302     SDValue visitFMINNUM(SDNode *N);
303     SDValue visitFMAXNUM(SDNode *N);
304     SDValue visitBRCOND(SDNode *N);
305     SDValue visitBR_CC(SDNode *N);
306     SDValue visitLOAD(SDNode *N);
307 
308     SDValue replaceStoreChain(StoreSDNode *ST, SDValue BetterChain);
309     SDValue replaceStoreOfFPConstant(StoreSDNode *ST);
310 
311     SDValue visitSTORE(SDNode *N);
312     SDValue visitINSERT_VECTOR_ELT(SDNode *N);
313     SDValue visitEXTRACT_VECTOR_ELT(SDNode *N);
314     SDValue visitBUILD_VECTOR(SDNode *N);
315     SDValue visitCONCAT_VECTORS(SDNode *N);
316     SDValue visitEXTRACT_SUBVECTOR(SDNode *N);
317     SDValue visitVECTOR_SHUFFLE(SDNode *N);
318     SDValue visitSCALAR_TO_VECTOR(SDNode *N);
319     SDValue visitINSERT_SUBVECTOR(SDNode *N);
320     SDValue visitMLOAD(SDNode *N);
321     SDValue visitMSTORE(SDNode *N);
322     SDValue visitMGATHER(SDNode *N);
323     SDValue visitMSCATTER(SDNode *N);
324     SDValue visitFP_TO_FP16(SDNode *N);
325     SDValue visitFP16_TO_FP(SDNode *N);
326 
327     SDValue visitFADDForFMACombine(SDNode *N);
328     SDValue visitFSUBForFMACombine(SDNode *N);
329     SDValue visitFMULForFMACombine(SDNode *N);
330 
331     SDValue XformToShuffleWithZero(SDNode *N);
332     SDValue ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue LHS,
333                            SDValue RHS);
334 
335     SDValue visitShiftByConstant(SDNode *N, ConstantSDNode *Amt);
336 
337     SDValue foldSelectOfConstants(SDNode *N);
338     bool SimplifySelectOps(SDNode *SELECT, SDValue LHS, SDValue RHS);
339     SDValue SimplifyBinOpWithSameOpcodeHands(SDNode *N);
340     SDValue SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2);
341     SDValue SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
342                              SDValue N2, SDValue N3, ISD::CondCode CC,
343                              bool NotExtCompare = false);
344     SDValue SimplifySetCC(EVT VT, SDValue N0, SDValue N1, ISD::CondCode Cond,
345                           const SDLoc &DL, bool foldBooleans = true);
346 
347     bool isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
348                            SDValue &CC) const;
349     bool isOneUseSetCC(SDValue N) const;
350 
351     SDValue SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
352                                          unsigned HiOp);
353     SDValue CombineConsecutiveLoads(SDNode *N, EVT VT);
354     SDValue CombineExtLoad(SDNode *N);
355     SDValue combineRepeatedFPDivisors(SDNode *N);
356     SDValue ConstantFoldBITCASTofBUILD_VECTOR(SDNode *, EVT);
357     SDValue BuildSDIV(SDNode *N);
358     SDValue BuildSDIVPow2(SDNode *N);
359     SDValue BuildUDIV(SDNode *N);
360     SDValue BuildReciprocalEstimate(SDValue Op, SDNodeFlags *Flags);
361     SDValue buildRsqrtEstimate(SDValue Op, SDNodeFlags *Flags);
362     SDValue buildSqrtEstimate(SDValue Op, SDNodeFlags *Flags);
363     SDValue buildSqrtEstimateImpl(SDValue Op, SDNodeFlags *Flags, bool Recip);
364     SDValue buildSqrtNROneConst(SDValue Op, SDValue Est, unsigned Iterations,
365                                 SDNodeFlags *Flags, bool Reciprocal);
366     SDValue buildSqrtNRTwoConst(SDValue Op, SDValue Est, unsigned Iterations,
367                                 SDNodeFlags *Flags, bool Reciprocal);
368     SDValue MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
369                                bool DemandHighBits = true);
370     SDValue MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1);
371     SDNode *MatchRotatePosNeg(SDValue Shifted, SDValue Pos, SDValue Neg,
372                               SDValue InnerPos, SDValue InnerNeg,
373                               unsigned PosOpcode, unsigned NegOpcode,
374                               const SDLoc &DL);
375     SDNode *MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL);
376     SDValue ReduceLoadWidth(SDNode *N);
377     SDValue ReduceLoadOpStoreWidth(SDNode *N);
378     SDValue splitMergedValStore(StoreSDNode *ST);
379     SDValue TransformFPLoadStorePair(SDNode *N);
380     SDValue reduceBuildVecExtToExtBuildVec(SDNode *N);
381     SDValue reduceBuildVecConvertToConvertBuildVec(SDNode *N);
382     SDValue reduceBuildVecToShuffle(SDNode *N);
383     SDValue createBuildVecShuffle(SDLoc DL, SDNode *N, ArrayRef<int> VectorMask,
384                                   SDValue VecIn1, SDValue VecIn2,
385                                   unsigned LeftIdx);
386 
387     SDValue GetDemandedBits(SDValue V, const APInt &Mask);
388 
389     /// Walk up chain skipping non-aliasing memory nodes,
390     /// looking for aliasing nodes and adding them to the Aliases vector.
391     void GatherAllAliases(SDNode *N, SDValue OriginalChain,
392                           SmallVectorImpl<SDValue> &Aliases);
393 
394     /// Return true if there is any possibility that the two addresses overlap.
395     bool isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const;
396 
397     /// Walk up chain skipping non-aliasing memory nodes, looking for a better
398     /// chain (aliasing node.)
399     SDValue FindBetterChain(SDNode *N, SDValue Chain);
400 
401     /// Try to replace a store and any possibly adjacent stores on
402     /// consecutive chains with better chains. Return true only if St is
403     /// replaced.
404     ///
405     /// Notice that other chains may still be replaced even if the function
406     /// returns false.
407     bool findBetterNeighborChains(StoreSDNode *St);
408 
409     /// Match "(X shl/srl V1) & V2" where V2 may not be present.
410     bool MatchRotateHalf(SDValue Op, SDValue &Shift, SDValue &Mask);
411 
412     /// Holds a pointer to an LSBaseSDNode as well as information on where it
413     /// is located in a sequence of memory operations connected by a chain.
414     struct MemOpLink {
415       MemOpLink (LSBaseSDNode *N, int64_t Offset, unsigned Seq):
416       MemNode(N), OffsetFromBase(Offset), SequenceNum(Seq) { }
417       // Ptr to the mem node.
418       LSBaseSDNode *MemNode;
419       // Offset from the base ptr.
420       int64_t OffsetFromBase;
421       // What is the sequence number of this mem node.
422       // Lowest mem operand in the DAG starts at zero.
423       unsigned SequenceNum;
424     };
425 
426     /// This is a helper function for visitMUL to check the profitability
427     /// of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
428     /// MulNode is the original multiply, AddNode is (add x, c1),
429     /// and ConstNode is c2.
430     bool isMulAddWithConstProfitable(SDNode *MulNode,
431                                      SDValue &AddNode,
432                                      SDValue &ConstNode);
433 
434     /// This is a helper function for MergeStoresOfConstantsOrVecElts. Returns a
435     /// constant build_vector of the stored constant values in Stores.
436     SDValue getMergedConstantVectorStore(SelectionDAG &DAG, const SDLoc &SL,
437                                          ArrayRef<MemOpLink> Stores,
438                                          SmallVectorImpl<SDValue> &Chains,
439                                          EVT Ty) const;
440 
441     /// This is a helper function for visitAND and visitZERO_EXTEND.  Returns
442     /// true if the (and (load x) c) pattern matches an extload.  ExtVT returns
443     /// the type of the loaded value to be extended.  LoadedVT returns the type
444     /// of the original loaded value.  NarrowLoad returns whether the load would
445     /// need to be narrowed in order to match.
446     bool isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
447                           EVT LoadResultTy, EVT &ExtVT, EVT &LoadedVT,
448                           bool &NarrowLoad);
449 
450     /// This is a helper function for MergeConsecutiveStores. When the source
451     /// elements of the consecutive stores are all constants or all extracted
452     /// vector elements, try to merge them into one larger store.
453     /// \return True if a merged store was created.
454     bool MergeStoresOfConstantsOrVecElts(SmallVectorImpl<MemOpLink> &StoreNodes,
455                                          EVT MemVT, unsigned NumStores,
456                                          bool IsConstantSrc, bool UseVector);
457 
458     /// This is a helper function for MergeConsecutiveStores.
459     /// Stores that may be merged are placed in StoreNodes.
460     /// Loads that may alias with those stores are placed in AliasLoadNodes.
461     void getStoreMergeAndAliasCandidates(
462         StoreSDNode* St, SmallVectorImpl<MemOpLink> &StoreNodes,
463         SmallVectorImpl<LSBaseSDNode*> &AliasLoadNodes);
464 
465     /// Helper function for MergeConsecutiveStores. Checks if
466     /// Candidate stores have indirect dependency through their
467     /// operands. \return True if safe to merge
468     bool checkMergeStoreCandidatesForDependencies(
469         SmallVectorImpl<MemOpLink> &StoreNodes);
470 
471     /// Merge consecutive store operations into a wide store.
472     /// This optimization uses wide integers or vectors when possible.
473     /// \return True if some memory operations were changed.
474     bool MergeConsecutiveStores(StoreSDNode *N);
475 
476     /// \brief Try to transform a truncation where C is a constant:
477     ///     (trunc (and X, C)) -> (and (trunc X), (trunc C))
478     ///
479     /// \p N needs to be a truncation and its first operand an AND. Other
480     /// requirements are checked by the function (e.g. that trunc is
481     /// single-use) and if missed an empty SDValue is returned.
482     SDValue distributeTruncateThroughAnd(SDNode *N);
483 
484   public:
485     DAGCombiner(SelectionDAG &D, AliasAnalysis &A, CodeGenOpt::Level OL)
486         : DAG(D), TLI(D.getTargetLoweringInfo()), Level(BeforeLegalizeTypes),
487           OptLevel(OL), LegalOperations(false), LegalTypes(false), AA(A) {
488       ForCodeSize = DAG.getMachineFunction().getFunction()->optForSize();
489     }
490 
491     /// Runs the dag combiner on all nodes in the work list
492     void Run(CombineLevel AtLevel);
493 
494     SelectionDAG &getDAG() const { return DAG; }
495 
496     /// Returns a type large enough to hold any valid shift amount - before type
497     /// legalization these can be huge.
498     EVT getShiftAmountTy(EVT LHSTy) {
499       assert(LHSTy.isInteger() && "Shift amount is not an integer type!");
500       if (LHSTy.isVector())
501         return LHSTy;
502       auto &DL = DAG.getDataLayout();
503       return LegalTypes ? TLI.getScalarShiftAmountTy(DL, LHSTy)
504                         : TLI.getPointerTy(DL);
505     }
506 
507     /// This method returns true if we are running before type legalization or
508     /// if the specified VT is legal.
509     bool isTypeLegal(const EVT &VT) {
510       if (!LegalTypes) return true;
511       return TLI.isTypeLegal(VT);
512     }
513 
514     /// Convenience wrapper around TargetLowering::getSetCCResultType
515     EVT getSetCCResultType(EVT VT) const {
516       return TLI.getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
517     }
518   };
519 }
520 
521 
522 namespace {
523 /// This class is a DAGUpdateListener that removes any deleted
524 /// nodes from the worklist.
525 class WorklistRemover : public SelectionDAG::DAGUpdateListener {
526   DAGCombiner &DC;
527 public:
528   explicit WorklistRemover(DAGCombiner &dc)
529     : SelectionDAG::DAGUpdateListener(dc.getDAG()), DC(dc) {}
530 
531   void NodeDeleted(SDNode *N, SDNode *E) override {
532     DC.removeFromWorklist(N);
533   }
534 };
535 }
536 
537 //===----------------------------------------------------------------------===//
538 //  TargetLowering::DAGCombinerInfo implementation
539 //===----------------------------------------------------------------------===//
540 
541 void TargetLowering::DAGCombinerInfo::AddToWorklist(SDNode *N) {
542   ((DAGCombiner*)DC)->AddToWorklist(N);
543 }
544 
545 SDValue TargetLowering::DAGCombinerInfo::
546 CombineTo(SDNode *N, ArrayRef<SDValue> To, bool AddTo) {
547   return ((DAGCombiner*)DC)->CombineTo(N, &To[0], To.size(), AddTo);
548 }
549 
550 SDValue TargetLowering::DAGCombinerInfo::
551 CombineTo(SDNode *N, SDValue Res, bool AddTo) {
552   return ((DAGCombiner*)DC)->CombineTo(N, Res, AddTo);
553 }
554 
555 
556 SDValue TargetLowering::DAGCombinerInfo::
557 CombineTo(SDNode *N, SDValue Res0, SDValue Res1, bool AddTo) {
558   return ((DAGCombiner*)DC)->CombineTo(N, Res0, Res1, AddTo);
559 }
560 
561 void TargetLowering::DAGCombinerInfo::
562 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
563   return ((DAGCombiner*)DC)->CommitTargetLoweringOpt(TLO);
564 }
565 
566 //===----------------------------------------------------------------------===//
567 // Helper Functions
568 //===----------------------------------------------------------------------===//
569 
570 void DAGCombiner::deleteAndRecombine(SDNode *N) {
571   removeFromWorklist(N);
572 
573   // If the operands of this node are only used by the node, they will now be
574   // dead. Make sure to re-visit them and recursively delete dead nodes.
575   for (const SDValue &Op : N->ops())
576     // For an operand generating multiple values, one of the values may
577     // become dead allowing further simplification (e.g. split index
578     // arithmetic from an indexed load).
579     if (Op->hasOneUse() || Op->getNumValues() > 1)
580       AddToWorklist(Op.getNode());
581 
582   DAG.DeleteNode(N);
583 }
584 
585 /// Return 1 if we can compute the negated form of the specified expression for
586 /// the same cost as the expression itself, or 2 if we can compute the negated
587 /// form more cheaply than the expression itself.
588 static char isNegatibleForFree(SDValue Op, bool LegalOperations,
589                                const TargetLowering &TLI,
590                                const TargetOptions *Options,
591                                unsigned Depth = 0) {
592   // fneg is removable even if it has multiple uses.
593   if (Op.getOpcode() == ISD::FNEG) return 2;
594 
595   // Don't allow anything with multiple uses.
596   if (!Op.hasOneUse()) return 0;
597 
598   // Don't recurse exponentially.
599   if (Depth > 6) return 0;
600 
601   switch (Op.getOpcode()) {
602   default: return false;
603   case ISD::ConstantFP:
604     // Don't invert constant FP values after legalize.  The negated constant
605     // isn't necessarily legal.
606     return LegalOperations ? 0 : 1;
607   case ISD::FADD:
608     // FIXME: determine better conditions for this xform.
609     if (!Options->UnsafeFPMath) return 0;
610 
611     // After operation legalization, it might not be legal to create new FSUBs.
612     if (LegalOperations &&
613         !TLI.isOperationLegalOrCustom(ISD::FSUB,  Op.getValueType()))
614       return 0;
615 
616     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
617     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
618                                     Options, Depth + 1))
619       return V;
620     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
621     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
622                               Depth + 1);
623   case ISD::FSUB:
624     // We can't turn -(A-B) into B-A when we honor signed zeros.
625     if (!Options->UnsafeFPMath) return 0;
626 
627     // fold (fneg (fsub A, B)) -> (fsub B, A)
628     return 1;
629 
630   case ISD::FMUL:
631   case ISD::FDIV:
632     if (Options->HonorSignDependentRoundingFPMath()) return 0;
633 
634     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y) or (fmul X, (fneg Y))
635     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
636                                     Options, Depth + 1))
637       return V;
638 
639     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
640                               Depth + 1);
641 
642   case ISD::FP_EXTEND:
643   case ISD::FP_ROUND:
644   case ISD::FSIN:
645     return isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI, Options,
646                               Depth + 1);
647   }
648 }
649 
650 /// If isNegatibleForFree returns true, return the newly negated expression.
651 static SDValue GetNegatedExpression(SDValue Op, SelectionDAG &DAG,
652                                     bool LegalOperations, unsigned Depth = 0) {
653   const TargetOptions &Options = DAG.getTarget().Options;
654   // fneg is removable even if it has multiple uses.
655   if (Op.getOpcode() == ISD::FNEG) return Op.getOperand(0);
656 
657   // Don't allow anything with multiple uses.
658   assert(Op.hasOneUse() && "Unknown reuse!");
659 
660   assert(Depth <= 6 && "GetNegatedExpression doesn't match isNegatibleForFree");
661 
662   const SDNodeFlags *Flags = Op.getNode()->getFlags();
663 
664   switch (Op.getOpcode()) {
665   default: llvm_unreachable("Unknown code");
666   case ISD::ConstantFP: {
667     APFloat V = cast<ConstantFPSDNode>(Op)->getValueAPF();
668     V.changeSign();
669     return DAG.getConstantFP(V, SDLoc(Op), Op.getValueType());
670   }
671   case ISD::FADD:
672     // FIXME: determine better conditions for this xform.
673     assert(Options.UnsafeFPMath);
674 
675     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
676     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
677                            DAG.getTargetLoweringInfo(), &Options, Depth+1))
678       return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
679                          GetNegatedExpression(Op.getOperand(0), DAG,
680                                               LegalOperations, Depth+1),
681                          Op.getOperand(1), Flags);
682     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
683     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
684                        GetNegatedExpression(Op.getOperand(1), DAG,
685                                             LegalOperations, Depth+1),
686                        Op.getOperand(0), Flags);
687   case ISD::FSUB:
688     // We can't turn -(A-B) into B-A when we honor signed zeros.
689     assert(Options.UnsafeFPMath);
690 
691     // fold (fneg (fsub 0, B)) -> B
692     if (ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(Op.getOperand(0)))
693       if (N0CFP->isZero())
694         return Op.getOperand(1);
695 
696     // fold (fneg (fsub A, B)) -> (fsub B, A)
697     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
698                        Op.getOperand(1), Op.getOperand(0), Flags);
699 
700   case ISD::FMUL:
701   case ISD::FDIV:
702     assert(!Options.HonorSignDependentRoundingFPMath());
703 
704     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y)
705     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
706                            DAG.getTargetLoweringInfo(), &Options, Depth+1))
707       return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
708                          GetNegatedExpression(Op.getOperand(0), DAG,
709                                               LegalOperations, Depth+1),
710                          Op.getOperand(1), Flags);
711 
712     // fold (fneg (fmul X, Y)) -> (fmul X, (fneg Y))
713     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
714                        Op.getOperand(0),
715                        GetNegatedExpression(Op.getOperand(1), DAG,
716                                             LegalOperations, Depth+1), Flags);
717 
718   case ISD::FP_EXTEND:
719   case ISD::FSIN:
720     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
721                        GetNegatedExpression(Op.getOperand(0), DAG,
722                                             LegalOperations, Depth+1));
723   case ISD::FP_ROUND:
724       return DAG.getNode(ISD::FP_ROUND, SDLoc(Op), Op.getValueType(),
725                          GetNegatedExpression(Op.getOperand(0), DAG,
726                                               LegalOperations, Depth+1),
727                          Op.getOperand(1));
728   }
729 }
730 
731 // APInts must be the same size for most operations, this helper
732 // function zero extends the shorter of the pair so that they match.
733 // We provide an Offset so that we can create bitwidths that won't overflow.
734 static void zeroExtendToMatch(APInt &LHS, APInt &RHS, unsigned Offset = 0) {
735   unsigned Bits = Offset + std::max(LHS.getBitWidth(), RHS.getBitWidth());
736   LHS = LHS.zextOrSelf(Bits);
737   RHS = RHS.zextOrSelf(Bits);
738 }
739 
740 // Return true if this node is a setcc, or is a select_cc
741 // that selects between the target values used for true and false, making it
742 // equivalent to a setcc. Also, set the incoming LHS, RHS, and CC references to
743 // the appropriate nodes based on the type of node we are checking. This
744 // simplifies life a bit for the callers.
745 bool DAGCombiner::isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
746                                     SDValue &CC) const {
747   if (N.getOpcode() == ISD::SETCC) {
748     LHS = N.getOperand(0);
749     RHS = N.getOperand(1);
750     CC  = N.getOperand(2);
751     return true;
752   }
753 
754   if (N.getOpcode() != ISD::SELECT_CC ||
755       !TLI.isConstTrueVal(N.getOperand(2).getNode()) ||
756       !TLI.isConstFalseVal(N.getOperand(3).getNode()))
757     return false;
758 
759   if (TLI.getBooleanContents(N.getValueType()) ==
760       TargetLowering::UndefinedBooleanContent)
761     return false;
762 
763   LHS = N.getOperand(0);
764   RHS = N.getOperand(1);
765   CC  = N.getOperand(4);
766   return true;
767 }
768 
769 /// Return true if this is a SetCC-equivalent operation with only one use.
770 /// If this is true, it allows the users to invert the operation for free when
771 /// it is profitable to do so.
772 bool DAGCombiner::isOneUseSetCC(SDValue N) const {
773   SDValue N0, N1, N2;
774   if (isSetCCEquivalent(N, N0, N1, N2) && N.getNode()->hasOneUse())
775     return true;
776   return false;
777 }
778 
779 // \brief Returns the SDNode if it is a constant float BuildVector
780 // or constant float.
781 static SDNode *isConstantFPBuildVectorOrConstantFP(SDValue N) {
782   if (isa<ConstantFPSDNode>(N))
783     return N.getNode();
784   if (ISD::isBuildVectorOfConstantFPSDNodes(N.getNode()))
785     return N.getNode();
786   return nullptr;
787 }
788 
789 // \brief Returns the SDNode if it is a constant splat BuildVector or constant
790 // int.
791 static ConstantSDNode *isConstOrConstSplat(SDValue N) {
792   if (ConstantSDNode *CN = dyn_cast<ConstantSDNode>(N))
793     return CN;
794 
795   if (BuildVectorSDNode *BV = dyn_cast<BuildVectorSDNode>(N)) {
796     BitVector UndefElements;
797     ConstantSDNode *CN = BV->getConstantSplatNode(&UndefElements);
798 
799     // BuildVectors can truncate their operands. Ignore that case here.
800     // FIXME: We blindly ignore splats which include undef which is overly
801     // pessimistic.
802     if (CN && UndefElements.none() &&
803         CN->getValueType(0) == N.getValueType().getScalarType())
804       return CN;
805   }
806 
807   return nullptr;
808 }
809 
810 // \brief Returns the SDNode if it is a constant splat BuildVector or constant
811 // float.
812 static ConstantFPSDNode *isConstOrConstSplatFP(SDValue N) {
813   if (ConstantFPSDNode *CN = dyn_cast<ConstantFPSDNode>(N))
814     return CN;
815 
816   if (BuildVectorSDNode *BV = dyn_cast<BuildVectorSDNode>(N)) {
817     BitVector UndefElements;
818     ConstantFPSDNode *CN = BV->getConstantFPSplatNode(&UndefElements);
819 
820     if (CN && UndefElements.none())
821       return CN;
822   }
823 
824   return nullptr;
825 }
826 
827 // Determines if it is a constant integer or a build vector of constant
828 // integers (and undefs).
829 // Do not permit build vector implicit truncation.
830 static bool isConstantOrConstantVector(SDValue N, bool NoOpaques = false) {
831   if (ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N))
832     return !(Const->isOpaque() && NoOpaques);
833   if (N.getOpcode() != ISD::BUILD_VECTOR)
834     return false;
835   unsigned BitWidth = N.getScalarValueSizeInBits();
836   for (const SDValue &Op : N->op_values()) {
837     if (Op.isUndef())
838       continue;
839     ConstantSDNode *Const = dyn_cast<ConstantSDNode>(Op);
840     if (!Const || Const->getAPIntValue().getBitWidth() != BitWidth ||
841         (Const->isOpaque() && NoOpaques))
842       return false;
843   }
844   return true;
845 }
846 
847 // Determines if it is a constant null integer or a splatted vector of a
848 // constant null integer (with no undefs).
849 // Build vector implicit truncation is not an issue for null values.
850 static bool isNullConstantOrNullSplatConstant(SDValue N) {
851   if (ConstantSDNode *Splat = isConstOrConstSplat(N))
852     return Splat->isNullValue();
853   return false;
854 }
855 
856 // Determines if it is a constant integer of one or a splatted vector of a
857 // constant integer of one (with no undefs).
858 // Do not permit build vector implicit truncation.
859 static bool isOneConstantOrOneSplatConstant(SDValue N) {
860   unsigned BitWidth = N.getScalarValueSizeInBits();
861   if (ConstantSDNode *Splat = isConstOrConstSplat(N))
862     return Splat->isOne() && Splat->getAPIntValue().getBitWidth() == BitWidth;
863   return false;
864 }
865 
866 SDValue DAGCombiner::ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue N0,
867                                     SDValue N1) {
868   EVT VT = N0.getValueType();
869   if (N0.getOpcode() == Opc) {
870     if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1))) {
871       if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
872         // reassoc. (op (op x, c1), c2) -> (op x, (op c1, c2))
873         if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, L, R))
874           return DAG.getNode(Opc, DL, VT, N0.getOperand(0), OpNode);
875         return SDValue();
876       }
877       if (N0.hasOneUse()) {
878         // reassoc. (op (op x, c1), y) -> (op (op x, y), c1) iff x+c1 has one
879         // use
880         SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0.getOperand(0), N1);
881         if (!OpNode.getNode())
882           return SDValue();
883         AddToWorklist(OpNode.getNode());
884         return DAG.getNode(Opc, DL, VT, OpNode, N0.getOperand(1));
885       }
886     }
887   }
888 
889   if (N1.getOpcode() == Opc) {
890     if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1.getOperand(1))) {
891       if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
892         // reassoc. (op c2, (op x, c1)) -> (op x, (op c1, c2))
893         if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, R, L))
894           return DAG.getNode(Opc, DL, VT, N1.getOperand(0), OpNode);
895         return SDValue();
896       }
897       if (N1.hasOneUse()) {
898         // reassoc. (op x, (op y, c1)) -> (op (op x, y), c1) iff x+c1 has one
899         // use
900         SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0, N1.getOperand(0));
901         if (!OpNode.getNode())
902           return SDValue();
903         AddToWorklist(OpNode.getNode());
904         return DAG.getNode(Opc, DL, VT, OpNode, N1.getOperand(1));
905       }
906     }
907   }
908 
909   return SDValue();
910 }
911 
912 SDValue DAGCombiner::CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
913                                bool AddTo) {
914   assert(N->getNumValues() == NumTo && "Broken CombineTo call!");
915   ++NodesCombined;
916   DEBUG(dbgs() << "\nReplacing.1 ";
917         N->dump(&DAG);
918         dbgs() << "\nWith: ";
919         To[0].getNode()->dump(&DAG);
920         dbgs() << " and " << NumTo-1 << " other values\n");
921   for (unsigned i = 0, e = NumTo; i != e; ++i)
922     assert((!To[i].getNode() ||
923             N->getValueType(i) == To[i].getValueType()) &&
924            "Cannot combine value to value of different type!");
925 
926   WorklistRemover DeadNodes(*this);
927   DAG.ReplaceAllUsesWith(N, To);
928   if (AddTo) {
929     // Push the new nodes and any users onto the worklist
930     for (unsigned i = 0, e = NumTo; i != e; ++i) {
931       if (To[i].getNode()) {
932         AddToWorklist(To[i].getNode());
933         AddUsersToWorklist(To[i].getNode());
934       }
935     }
936   }
937 
938   // Finally, if the node is now dead, remove it from the graph.  The node
939   // may not be dead if the replacement process recursively simplified to
940   // something else needing this node.
941   if (N->use_empty())
942     deleteAndRecombine(N);
943   return SDValue(N, 0);
944 }
945 
946 void DAGCombiner::
947 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
948   // Replace all uses.  If any nodes become isomorphic to other nodes and
949   // are deleted, make sure to remove them from our worklist.
950   WorklistRemover DeadNodes(*this);
951   DAG.ReplaceAllUsesOfValueWith(TLO.Old, TLO.New);
952 
953   // Push the new node and any (possibly new) users onto the worklist.
954   AddToWorklist(TLO.New.getNode());
955   AddUsersToWorklist(TLO.New.getNode());
956 
957   // Finally, if the node is now dead, remove it from the graph.  The node
958   // may not be dead if the replacement process recursively simplified to
959   // something else needing this node.
960   if (TLO.Old.getNode()->use_empty())
961     deleteAndRecombine(TLO.Old.getNode());
962 }
963 
964 /// Check the specified integer node value to see if it can be simplified or if
965 /// things it uses can be simplified by bit propagation. If so, return true.
966 bool DAGCombiner::SimplifyDemandedBits(SDValue Op, const APInt &Demanded) {
967   TargetLowering::TargetLoweringOpt TLO(DAG, LegalTypes, LegalOperations);
968   APInt KnownZero, KnownOne;
969   if (!TLI.SimplifyDemandedBits(Op, Demanded, KnownZero, KnownOne, TLO))
970     return false;
971 
972   // Revisit the node.
973   AddToWorklist(Op.getNode());
974 
975   // Replace the old value with the new one.
976   ++NodesCombined;
977   DEBUG(dbgs() << "\nReplacing.2 ";
978         TLO.Old.getNode()->dump(&DAG);
979         dbgs() << "\nWith: ";
980         TLO.New.getNode()->dump(&DAG);
981         dbgs() << '\n');
982 
983   CommitTargetLoweringOpt(TLO);
984   return true;
985 }
986 
987 void DAGCombiner::ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad) {
988   SDLoc DL(Load);
989   EVT VT = Load->getValueType(0);
990   SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, VT, SDValue(ExtLoad, 0));
991 
992   DEBUG(dbgs() << "\nReplacing.9 ";
993         Load->dump(&DAG);
994         dbgs() << "\nWith: ";
995         Trunc.getNode()->dump(&DAG);
996         dbgs() << '\n');
997   WorklistRemover DeadNodes(*this);
998   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 0), Trunc);
999   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), SDValue(ExtLoad, 1));
1000   deleteAndRecombine(Load);
1001   AddToWorklist(Trunc.getNode());
1002 }
1003 
1004 SDValue DAGCombiner::PromoteOperand(SDValue Op, EVT PVT, bool &Replace) {
1005   Replace = false;
1006   SDLoc DL(Op);
1007   if (ISD::isUNINDEXEDLoad(Op.getNode())) {
1008     LoadSDNode *LD = cast<LoadSDNode>(Op);
1009     EVT MemVT = LD->getMemoryVT();
1010     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD)
1011       ? (TLI.isLoadExtLegal(ISD::ZEXTLOAD, PVT, MemVT) ? ISD::ZEXTLOAD
1012                                                        : ISD::EXTLOAD)
1013       : LD->getExtensionType();
1014     Replace = true;
1015     return DAG.getExtLoad(ExtType, DL, PVT,
1016                           LD->getChain(), LD->getBasePtr(),
1017                           MemVT, LD->getMemOperand());
1018   }
1019 
1020   unsigned Opc = Op.getOpcode();
1021   switch (Opc) {
1022   default: break;
1023   case ISD::AssertSext:
1024     return DAG.getNode(ISD::AssertSext, DL, PVT,
1025                        SExtPromoteOperand(Op.getOperand(0), PVT),
1026                        Op.getOperand(1));
1027   case ISD::AssertZext:
1028     return DAG.getNode(ISD::AssertZext, DL, PVT,
1029                        ZExtPromoteOperand(Op.getOperand(0), PVT),
1030                        Op.getOperand(1));
1031   case ISD::Constant: {
1032     unsigned ExtOpc =
1033       Op.getValueType().isByteSized() ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
1034     return DAG.getNode(ExtOpc, DL, PVT, Op);
1035   }
1036   }
1037 
1038   if (!TLI.isOperationLegal(ISD::ANY_EXTEND, PVT))
1039     return SDValue();
1040   return DAG.getNode(ISD::ANY_EXTEND, DL, PVT, Op);
1041 }
1042 
1043 SDValue DAGCombiner::SExtPromoteOperand(SDValue Op, EVT PVT) {
1044   if (!TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, PVT))
1045     return SDValue();
1046   EVT OldVT = Op.getValueType();
1047   SDLoc DL(Op);
1048   bool Replace = false;
1049   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1050   if (!NewOp.getNode())
1051     return SDValue();
1052   AddToWorklist(NewOp.getNode());
1053 
1054   if (Replace)
1055     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1056   return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, NewOp.getValueType(), NewOp,
1057                      DAG.getValueType(OldVT));
1058 }
1059 
1060 SDValue DAGCombiner::ZExtPromoteOperand(SDValue Op, EVT PVT) {
1061   EVT OldVT = Op.getValueType();
1062   SDLoc DL(Op);
1063   bool Replace = false;
1064   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1065   if (!NewOp.getNode())
1066     return SDValue();
1067   AddToWorklist(NewOp.getNode());
1068 
1069   if (Replace)
1070     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1071   return DAG.getZeroExtendInReg(NewOp, DL, OldVT);
1072 }
1073 
1074 /// Promote the specified integer binary operation if the target indicates it is
1075 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1076 /// i32 since i16 instructions are longer.
1077 SDValue DAGCombiner::PromoteIntBinOp(SDValue Op) {
1078   if (!LegalOperations)
1079     return SDValue();
1080 
1081   EVT VT = Op.getValueType();
1082   if (VT.isVector() || !VT.isInteger())
1083     return SDValue();
1084 
1085   // If operation type is 'undesirable', e.g. i16 on x86, consider
1086   // promoting it.
1087   unsigned Opc = Op.getOpcode();
1088   if (TLI.isTypeDesirableForOp(Opc, VT))
1089     return SDValue();
1090 
1091   EVT PVT = VT;
1092   // Consult target whether it is a good idea to promote this operation and
1093   // what's the right type to promote it to.
1094   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1095     assert(PVT != VT && "Don't know what type to promote to!");
1096 
1097     bool Replace0 = false;
1098     SDValue N0 = Op.getOperand(0);
1099     SDValue NN0 = PromoteOperand(N0, PVT, Replace0);
1100     if (!NN0.getNode())
1101       return SDValue();
1102 
1103     bool Replace1 = false;
1104     SDValue N1 = Op.getOperand(1);
1105     SDValue NN1;
1106     if (N0 == N1)
1107       NN1 = NN0;
1108     else {
1109       NN1 = PromoteOperand(N1, PVT, Replace1);
1110       if (!NN1.getNode())
1111         return SDValue();
1112     }
1113 
1114     AddToWorklist(NN0.getNode());
1115     if (NN1.getNode())
1116       AddToWorklist(NN1.getNode());
1117 
1118     if (Replace0)
1119       ReplaceLoadWithPromotedLoad(N0.getNode(), NN0.getNode());
1120     if (Replace1)
1121       ReplaceLoadWithPromotedLoad(N1.getNode(), NN1.getNode());
1122 
1123     DEBUG(dbgs() << "\nPromoting ";
1124           Op.getNode()->dump(&DAG));
1125     SDLoc DL(Op);
1126     return DAG.getNode(ISD::TRUNCATE, DL, VT,
1127                        DAG.getNode(Opc, DL, PVT, NN0, NN1));
1128   }
1129   return SDValue();
1130 }
1131 
1132 /// Promote the specified integer shift operation if the target indicates it is
1133 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1134 /// i32 since i16 instructions are longer.
1135 SDValue DAGCombiner::PromoteIntShiftOp(SDValue Op) {
1136   if (!LegalOperations)
1137     return SDValue();
1138 
1139   EVT VT = Op.getValueType();
1140   if (VT.isVector() || !VT.isInteger())
1141     return SDValue();
1142 
1143   // If operation type is 'undesirable', e.g. i16 on x86, consider
1144   // promoting it.
1145   unsigned Opc = Op.getOpcode();
1146   if (TLI.isTypeDesirableForOp(Opc, VT))
1147     return SDValue();
1148 
1149   EVT PVT = VT;
1150   // Consult target whether it is a good idea to promote this operation and
1151   // what's the right type to promote it to.
1152   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1153     assert(PVT != VT && "Don't know what type to promote to!");
1154 
1155     bool Replace = false;
1156     SDValue N0 = Op.getOperand(0);
1157     if (Opc == ISD::SRA)
1158       N0 = SExtPromoteOperand(Op.getOperand(0), PVT);
1159     else if (Opc == ISD::SRL)
1160       N0 = ZExtPromoteOperand(Op.getOperand(0), PVT);
1161     else
1162       N0 = PromoteOperand(N0, PVT, Replace);
1163     if (!N0.getNode())
1164       return SDValue();
1165 
1166     AddToWorklist(N0.getNode());
1167     if (Replace)
1168       ReplaceLoadWithPromotedLoad(Op.getOperand(0).getNode(), N0.getNode());
1169 
1170     DEBUG(dbgs() << "\nPromoting ";
1171           Op.getNode()->dump(&DAG));
1172     SDLoc DL(Op);
1173     return DAG.getNode(ISD::TRUNCATE, DL, VT,
1174                        DAG.getNode(Opc, DL, PVT, N0, Op.getOperand(1)));
1175   }
1176   return SDValue();
1177 }
1178 
1179 SDValue DAGCombiner::PromoteExtend(SDValue Op) {
1180   if (!LegalOperations)
1181     return SDValue();
1182 
1183   EVT VT = Op.getValueType();
1184   if (VT.isVector() || !VT.isInteger())
1185     return SDValue();
1186 
1187   // If operation type is 'undesirable', e.g. i16 on x86, consider
1188   // promoting it.
1189   unsigned Opc = Op.getOpcode();
1190   if (TLI.isTypeDesirableForOp(Opc, VT))
1191     return SDValue();
1192 
1193   EVT PVT = VT;
1194   // Consult target whether it is a good idea to promote this operation and
1195   // what's the right type to promote it to.
1196   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1197     assert(PVT != VT && "Don't know what type to promote to!");
1198     // fold (aext (aext x)) -> (aext x)
1199     // fold (aext (zext x)) -> (zext x)
1200     // fold (aext (sext x)) -> (sext x)
1201     DEBUG(dbgs() << "\nPromoting ";
1202           Op.getNode()->dump(&DAG));
1203     return DAG.getNode(Op.getOpcode(), SDLoc(Op), VT, Op.getOperand(0));
1204   }
1205   return SDValue();
1206 }
1207 
1208 bool DAGCombiner::PromoteLoad(SDValue Op) {
1209   if (!LegalOperations)
1210     return false;
1211 
1212   if (!ISD::isUNINDEXEDLoad(Op.getNode()))
1213     return false;
1214 
1215   EVT VT = Op.getValueType();
1216   if (VT.isVector() || !VT.isInteger())
1217     return false;
1218 
1219   // If operation type is 'undesirable', e.g. i16 on x86, consider
1220   // promoting it.
1221   unsigned Opc = Op.getOpcode();
1222   if (TLI.isTypeDesirableForOp(Opc, VT))
1223     return false;
1224 
1225   EVT PVT = VT;
1226   // Consult target whether it is a good idea to promote this operation and
1227   // what's the right type to promote it to.
1228   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1229     assert(PVT != VT && "Don't know what type to promote to!");
1230 
1231     SDLoc DL(Op);
1232     SDNode *N = Op.getNode();
1233     LoadSDNode *LD = cast<LoadSDNode>(N);
1234     EVT MemVT = LD->getMemoryVT();
1235     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD)
1236       ? (TLI.isLoadExtLegal(ISD::ZEXTLOAD, PVT, MemVT) ? ISD::ZEXTLOAD
1237                                                        : ISD::EXTLOAD)
1238       : LD->getExtensionType();
1239     SDValue NewLD = DAG.getExtLoad(ExtType, DL, PVT,
1240                                    LD->getChain(), LD->getBasePtr(),
1241                                    MemVT, LD->getMemOperand());
1242     SDValue Result = DAG.getNode(ISD::TRUNCATE, DL, VT, NewLD);
1243 
1244     DEBUG(dbgs() << "\nPromoting ";
1245           N->dump(&DAG);
1246           dbgs() << "\nTo: ";
1247           Result.getNode()->dump(&DAG);
1248           dbgs() << '\n');
1249     WorklistRemover DeadNodes(*this);
1250     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result);
1251     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), NewLD.getValue(1));
1252     deleteAndRecombine(N);
1253     AddToWorklist(Result.getNode());
1254     return true;
1255   }
1256   return false;
1257 }
1258 
1259 /// \brief Recursively delete a node which has no uses and any operands for
1260 /// which it is the only use.
1261 ///
1262 /// Note that this both deletes the nodes and removes them from the worklist.
1263 /// It also adds any nodes who have had a user deleted to the worklist as they
1264 /// may now have only one use and subject to other combines.
1265 bool DAGCombiner::recursivelyDeleteUnusedNodes(SDNode *N) {
1266   if (!N->use_empty())
1267     return false;
1268 
1269   SmallSetVector<SDNode *, 16> Nodes;
1270   Nodes.insert(N);
1271   do {
1272     N = Nodes.pop_back_val();
1273     if (!N)
1274       continue;
1275 
1276     if (N->use_empty()) {
1277       for (const SDValue &ChildN : N->op_values())
1278         Nodes.insert(ChildN.getNode());
1279 
1280       removeFromWorklist(N);
1281       DAG.DeleteNode(N);
1282     } else {
1283       AddToWorklist(N);
1284     }
1285   } while (!Nodes.empty());
1286   return true;
1287 }
1288 
1289 //===----------------------------------------------------------------------===//
1290 //  Main DAG Combiner implementation
1291 //===----------------------------------------------------------------------===//
1292 
1293 void DAGCombiner::Run(CombineLevel AtLevel) {
1294   // set the instance variables, so that the various visit routines may use it.
1295   Level = AtLevel;
1296   LegalOperations = Level >= AfterLegalizeVectorOps;
1297   LegalTypes = Level >= AfterLegalizeTypes;
1298 
1299   // Add all the dag nodes to the worklist.
1300   for (SDNode &Node : DAG.allnodes())
1301     AddToWorklist(&Node);
1302 
1303   // Create a dummy node (which is not added to allnodes), that adds a reference
1304   // to the root node, preventing it from being deleted, and tracking any
1305   // changes of the root.
1306   HandleSDNode Dummy(DAG.getRoot());
1307 
1308   // While the worklist isn't empty, find a node and try to combine it.
1309   while (!WorklistMap.empty()) {
1310     SDNode *N;
1311     // The Worklist holds the SDNodes in order, but it may contain null entries.
1312     do {
1313       N = Worklist.pop_back_val();
1314     } while (!N);
1315 
1316     bool GoodWorklistEntry = WorklistMap.erase(N);
1317     (void)GoodWorklistEntry;
1318     assert(GoodWorklistEntry &&
1319            "Found a worklist entry without a corresponding map entry!");
1320 
1321     // If N has no uses, it is dead.  Make sure to revisit all N's operands once
1322     // N is deleted from the DAG, since they too may now be dead or may have a
1323     // reduced number of uses, allowing other xforms.
1324     if (recursivelyDeleteUnusedNodes(N))
1325       continue;
1326 
1327     WorklistRemover DeadNodes(*this);
1328 
1329     // If this combine is running after legalizing the DAG, re-legalize any
1330     // nodes pulled off the worklist.
1331     if (Level == AfterLegalizeDAG) {
1332       SmallSetVector<SDNode *, 16> UpdatedNodes;
1333       bool NIsValid = DAG.LegalizeOp(N, UpdatedNodes);
1334 
1335       for (SDNode *LN : UpdatedNodes) {
1336         AddToWorklist(LN);
1337         AddUsersToWorklist(LN);
1338       }
1339       if (!NIsValid)
1340         continue;
1341     }
1342 
1343     DEBUG(dbgs() << "\nCombining: "; N->dump(&DAG));
1344 
1345     // Add any operands of the new node which have not yet been combined to the
1346     // worklist as well. Because the worklist uniques things already, this
1347     // won't repeatedly process the same operand.
1348     CombinedNodes.insert(N);
1349     for (const SDValue &ChildN : N->op_values())
1350       if (!CombinedNodes.count(ChildN.getNode()))
1351         AddToWorklist(ChildN.getNode());
1352 
1353     SDValue RV = combine(N);
1354 
1355     if (!RV.getNode())
1356       continue;
1357 
1358     ++NodesCombined;
1359 
1360     // If we get back the same node we passed in, rather than a new node or
1361     // zero, we know that the node must have defined multiple values and
1362     // CombineTo was used.  Since CombineTo takes care of the worklist
1363     // mechanics for us, we have no work to do in this case.
1364     if (RV.getNode() == N)
1365       continue;
1366 
1367     assert(N->getOpcode() != ISD::DELETED_NODE &&
1368            RV.getOpcode() != ISD::DELETED_NODE &&
1369            "Node was deleted but visit returned new node!");
1370 
1371     DEBUG(dbgs() << " ... into: ";
1372           RV.getNode()->dump(&DAG));
1373 
1374     if (N->getNumValues() == RV.getNode()->getNumValues())
1375       DAG.ReplaceAllUsesWith(N, RV.getNode());
1376     else {
1377       assert(N->getValueType(0) == RV.getValueType() &&
1378              N->getNumValues() == 1 && "Type mismatch");
1379       SDValue OpV = RV;
1380       DAG.ReplaceAllUsesWith(N, &OpV);
1381     }
1382 
1383     // Push the new node and any users onto the worklist
1384     AddToWorklist(RV.getNode());
1385     AddUsersToWorklist(RV.getNode());
1386 
1387     // Finally, if the node is now dead, remove it from the graph.  The node
1388     // may not be dead if the replacement process recursively simplified to
1389     // something else needing this node. This will also take care of adding any
1390     // operands which have lost a user to the worklist.
1391     recursivelyDeleteUnusedNodes(N);
1392   }
1393 
1394   // If the root changed (e.g. it was a dead load, update the root).
1395   DAG.setRoot(Dummy.getValue());
1396   DAG.RemoveDeadNodes();
1397 }
1398 
1399 SDValue DAGCombiner::visit(SDNode *N) {
1400   switch (N->getOpcode()) {
1401   default: break;
1402   case ISD::TokenFactor:        return visitTokenFactor(N);
1403   case ISD::MERGE_VALUES:       return visitMERGE_VALUES(N);
1404   case ISD::ADD:                return visitADD(N);
1405   case ISD::SUB:                return visitSUB(N);
1406   case ISD::ADDC:               return visitADDC(N);
1407   case ISD::SUBC:               return visitSUBC(N);
1408   case ISD::ADDE:               return visitADDE(N);
1409   case ISD::SUBE:               return visitSUBE(N);
1410   case ISD::MUL:                return visitMUL(N);
1411   case ISD::SDIV:               return visitSDIV(N);
1412   case ISD::UDIV:               return visitUDIV(N);
1413   case ISD::SREM:
1414   case ISD::UREM:               return visitREM(N);
1415   case ISD::MULHU:              return visitMULHU(N);
1416   case ISD::MULHS:              return visitMULHS(N);
1417   case ISD::SMUL_LOHI:          return visitSMUL_LOHI(N);
1418   case ISD::UMUL_LOHI:          return visitUMUL_LOHI(N);
1419   case ISD::SMULO:              return visitSMULO(N);
1420   case ISD::UMULO:              return visitUMULO(N);
1421   case ISD::SMIN:
1422   case ISD::SMAX:
1423   case ISD::UMIN:
1424   case ISD::UMAX:               return visitIMINMAX(N);
1425   case ISD::AND:                return visitAND(N);
1426   case ISD::OR:                 return visitOR(N);
1427   case ISD::XOR:                return visitXOR(N);
1428   case ISD::SHL:                return visitSHL(N);
1429   case ISD::SRA:                return visitSRA(N);
1430   case ISD::SRL:                return visitSRL(N);
1431   case ISD::ROTR:
1432   case ISD::ROTL:               return visitRotate(N);
1433   case ISD::BSWAP:              return visitBSWAP(N);
1434   case ISD::BITREVERSE:         return visitBITREVERSE(N);
1435   case ISD::CTLZ:               return visitCTLZ(N);
1436   case ISD::CTLZ_ZERO_UNDEF:    return visitCTLZ_ZERO_UNDEF(N);
1437   case ISD::CTTZ:               return visitCTTZ(N);
1438   case ISD::CTTZ_ZERO_UNDEF:    return visitCTTZ_ZERO_UNDEF(N);
1439   case ISD::CTPOP:              return visitCTPOP(N);
1440   case ISD::SELECT:             return visitSELECT(N);
1441   case ISD::VSELECT:            return visitVSELECT(N);
1442   case ISD::SELECT_CC:          return visitSELECT_CC(N);
1443   case ISD::SETCC:              return visitSETCC(N);
1444   case ISD::SETCCE:             return visitSETCCE(N);
1445   case ISD::SIGN_EXTEND:        return visitSIGN_EXTEND(N);
1446   case ISD::ZERO_EXTEND:        return visitZERO_EXTEND(N);
1447   case ISD::ANY_EXTEND:         return visitANY_EXTEND(N);
1448   case ISD::SIGN_EXTEND_INREG:  return visitSIGN_EXTEND_INREG(N);
1449   case ISD::SIGN_EXTEND_VECTOR_INREG: return visitSIGN_EXTEND_VECTOR_INREG(N);
1450   case ISD::ZERO_EXTEND_VECTOR_INREG: return visitZERO_EXTEND_VECTOR_INREG(N);
1451   case ISD::TRUNCATE:           return visitTRUNCATE(N);
1452   case ISD::BITCAST:            return visitBITCAST(N);
1453   case ISD::BUILD_PAIR:         return visitBUILD_PAIR(N);
1454   case ISD::FADD:               return visitFADD(N);
1455   case ISD::FSUB:               return visitFSUB(N);
1456   case ISD::FMUL:               return visitFMUL(N);
1457   case ISD::FMA:                return visitFMA(N);
1458   case ISD::FDIV:               return visitFDIV(N);
1459   case ISD::FREM:               return visitFREM(N);
1460   case ISD::FSQRT:              return visitFSQRT(N);
1461   case ISD::FCOPYSIGN:          return visitFCOPYSIGN(N);
1462   case ISD::SINT_TO_FP:         return visitSINT_TO_FP(N);
1463   case ISD::UINT_TO_FP:         return visitUINT_TO_FP(N);
1464   case ISD::FP_TO_SINT:         return visitFP_TO_SINT(N);
1465   case ISD::FP_TO_UINT:         return visitFP_TO_UINT(N);
1466   case ISD::FP_ROUND:           return visitFP_ROUND(N);
1467   case ISD::FP_ROUND_INREG:     return visitFP_ROUND_INREG(N);
1468   case ISD::FP_EXTEND:          return visitFP_EXTEND(N);
1469   case ISD::FNEG:               return visitFNEG(N);
1470   case ISD::FABS:               return visitFABS(N);
1471   case ISD::FFLOOR:             return visitFFLOOR(N);
1472   case ISD::FMINNUM:            return visitFMINNUM(N);
1473   case ISD::FMAXNUM:            return visitFMAXNUM(N);
1474   case ISD::FCEIL:              return visitFCEIL(N);
1475   case ISD::FTRUNC:             return visitFTRUNC(N);
1476   case ISD::BRCOND:             return visitBRCOND(N);
1477   case ISD::BR_CC:              return visitBR_CC(N);
1478   case ISD::LOAD:               return visitLOAD(N);
1479   case ISD::STORE:              return visitSTORE(N);
1480   case ISD::INSERT_VECTOR_ELT:  return visitINSERT_VECTOR_ELT(N);
1481   case ISD::EXTRACT_VECTOR_ELT: return visitEXTRACT_VECTOR_ELT(N);
1482   case ISD::BUILD_VECTOR:       return visitBUILD_VECTOR(N);
1483   case ISD::CONCAT_VECTORS:     return visitCONCAT_VECTORS(N);
1484   case ISD::EXTRACT_SUBVECTOR:  return visitEXTRACT_SUBVECTOR(N);
1485   case ISD::VECTOR_SHUFFLE:     return visitVECTOR_SHUFFLE(N);
1486   case ISD::SCALAR_TO_VECTOR:   return visitSCALAR_TO_VECTOR(N);
1487   case ISD::INSERT_SUBVECTOR:   return visitINSERT_SUBVECTOR(N);
1488   case ISD::MGATHER:            return visitMGATHER(N);
1489   case ISD::MLOAD:              return visitMLOAD(N);
1490   case ISD::MSCATTER:           return visitMSCATTER(N);
1491   case ISD::MSTORE:             return visitMSTORE(N);
1492   case ISD::FP_TO_FP16:         return visitFP_TO_FP16(N);
1493   case ISD::FP16_TO_FP:         return visitFP16_TO_FP(N);
1494   }
1495   return SDValue();
1496 }
1497 
1498 SDValue DAGCombiner::combine(SDNode *N) {
1499   SDValue RV = visit(N);
1500 
1501   // If nothing happened, try a target-specific DAG combine.
1502   if (!RV.getNode()) {
1503     assert(N->getOpcode() != ISD::DELETED_NODE &&
1504            "Node was deleted but visit returned NULL!");
1505 
1506     if (N->getOpcode() >= ISD::BUILTIN_OP_END ||
1507         TLI.hasTargetDAGCombine((ISD::NodeType)N->getOpcode())) {
1508 
1509       // Expose the DAG combiner to the target combiner impls.
1510       TargetLowering::DAGCombinerInfo
1511         DagCombineInfo(DAG, Level, false, this);
1512 
1513       RV = TLI.PerformDAGCombine(N, DagCombineInfo);
1514     }
1515   }
1516 
1517   // If nothing happened still, try promoting the operation.
1518   if (!RV.getNode()) {
1519     switch (N->getOpcode()) {
1520     default: break;
1521     case ISD::ADD:
1522     case ISD::SUB:
1523     case ISD::MUL:
1524     case ISD::AND:
1525     case ISD::OR:
1526     case ISD::XOR:
1527       RV = PromoteIntBinOp(SDValue(N, 0));
1528       break;
1529     case ISD::SHL:
1530     case ISD::SRA:
1531     case ISD::SRL:
1532       RV = PromoteIntShiftOp(SDValue(N, 0));
1533       break;
1534     case ISD::SIGN_EXTEND:
1535     case ISD::ZERO_EXTEND:
1536     case ISD::ANY_EXTEND:
1537       RV = PromoteExtend(SDValue(N, 0));
1538       break;
1539     case ISD::LOAD:
1540       if (PromoteLoad(SDValue(N, 0)))
1541         RV = SDValue(N, 0);
1542       break;
1543     }
1544   }
1545 
1546   // If N is a commutative binary node, try commuting it to enable more
1547   // sdisel CSE.
1548   if (!RV.getNode() && SelectionDAG::isCommutativeBinOp(N->getOpcode()) &&
1549       N->getNumValues() == 1) {
1550     SDValue N0 = N->getOperand(0);
1551     SDValue N1 = N->getOperand(1);
1552 
1553     // Constant operands are canonicalized to RHS.
1554     if (isa<ConstantSDNode>(N0) || !isa<ConstantSDNode>(N1)) {
1555       SDValue Ops[] = {N1, N0};
1556       SDNode *CSENode = DAG.getNodeIfExists(N->getOpcode(), N->getVTList(), Ops,
1557                                             N->getFlags());
1558       if (CSENode)
1559         return SDValue(CSENode, 0);
1560     }
1561   }
1562 
1563   return RV;
1564 }
1565 
1566 /// Given a node, return its input chain if it has one, otherwise return a null
1567 /// sd operand.
1568 static SDValue getInputChainForNode(SDNode *N) {
1569   if (unsigned NumOps = N->getNumOperands()) {
1570     if (N->getOperand(0).getValueType() == MVT::Other)
1571       return N->getOperand(0);
1572     if (N->getOperand(NumOps-1).getValueType() == MVT::Other)
1573       return N->getOperand(NumOps-1);
1574     for (unsigned i = 1; i < NumOps-1; ++i)
1575       if (N->getOperand(i).getValueType() == MVT::Other)
1576         return N->getOperand(i);
1577   }
1578   return SDValue();
1579 }
1580 
1581 SDValue DAGCombiner::visitTokenFactor(SDNode *N) {
1582   // If N has two operands, where one has an input chain equal to the other,
1583   // the 'other' chain is redundant.
1584   if (N->getNumOperands() == 2) {
1585     if (getInputChainForNode(N->getOperand(0).getNode()) == N->getOperand(1))
1586       return N->getOperand(0);
1587     if (getInputChainForNode(N->getOperand(1).getNode()) == N->getOperand(0))
1588       return N->getOperand(1);
1589   }
1590 
1591   SmallVector<SDNode *, 8> TFs;     // List of token factors to visit.
1592   SmallVector<SDValue, 8> Ops;    // Ops for replacing token factor.
1593   SmallPtrSet<SDNode*, 16> SeenOps;
1594   bool Changed = false;             // If we should replace this token factor.
1595 
1596   // Start out with this token factor.
1597   TFs.push_back(N);
1598 
1599   // Iterate through token factors.  The TFs grows when new token factors are
1600   // encountered.
1601   for (unsigned i = 0; i < TFs.size(); ++i) {
1602     SDNode *TF = TFs[i];
1603 
1604     // Check each of the operands.
1605     for (const SDValue &Op : TF->op_values()) {
1606 
1607       switch (Op.getOpcode()) {
1608       case ISD::EntryToken:
1609         // Entry tokens don't need to be added to the list. They are
1610         // redundant.
1611         Changed = true;
1612         break;
1613 
1614       case ISD::TokenFactor:
1615         if (Op.hasOneUse() && !is_contained(TFs, Op.getNode())) {
1616           // Queue up for processing.
1617           TFs.push_back(Op.getNode());
1618           // Clean up in case the token factor is removed.
1619           AddToWorklist(Op.getNode());
1620           Changed = true;
1621           break;
1622         }
1623         LLVM_FALLTHROUGH;
1624 
1625       default:
1626         // Only add if it isn't already in the list.
1627         if (SeenOps.insert(Op.getNode()).second)
1628           Ops.push_back(Op);
1629         else
1630           Changed = true;
1631         break;
1632       }
1633     }
1634   }
1635 
1636   SDValue Result;
1637 
1638   // If we've changed things around then replace token factor.
1639   if (Changed) {
1640     if (Ops.empty()) {
1641       // The entry token is the only possible outcome.
1642       Result = DAG.getEntryNode();
1643     } else {
1644       // New and improved token factor.
1645       Result = DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Ops);
1646     }
1647 
1648     // Add users to worklist if AA is enabled, since it may introduce
1649     // a lot of new chained token factors while removing memory deps.
1650     bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
1651       : DAG.getSubtarget().useAA();
1652     return CombineTo(N, Result, UseAA /*add to worklist*/);
1653   }
1654 
1655   return Result;
1656 }
1657 
1658 /// MERGE_VALUES can always be eliminated.
1659 SDValue DAGCombiner::visitMERGE_VALUES(SDNode *N) {
1660   WorklistRemover DeadNodes(*this);
1661   // Replacing results may cause a different MERGE_VALUES to suddenly
1662   // be CSE'd with N, and carry its uses with it. Iterate until no
1663   // uses remain, to ensure that the node can be safely deleted.
1664   // First add the users of this node to the work list so that they
1665   // can be tried again once they have new operands.
1666   AddUsersToWorklist(N);
1667   do {
1668     for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i)
1669       DAG.ReplaceAllUsesOfValueWith(SDValue(N, i), N->getOperand(i));
1670   } while (!N->use_empty());
1671   deleteAndRecombine(N);
1672   return SDValue(N, 0);   // Return N so it doesn't get rechecked!
1673 }
1674 
1675 /// If \p N is a ConstantSDNode with isOpaque() == false return it casted to a
1676 /// ConstantSDNode pointer else nullptr.
1677 static ConstantSDNode *getAsNonOpaqueConstant(SDValue N) {
1678   ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N);
1679   return Const != nullptr && !Const->isOpaque() ? Const : nullptr;
1680 }
1681 
1682 SDValue DAGCombiner::visitADD(SDNode *N) {
1683   SDValue N0 = N->getOperand(0);
1684   SDValue N1 = N->getOperand(1);
1685   EVT VT = N0.getValueType();
1686 
1687   // fold vector ops
1688   if (VT.isVector()) {
1689     if (SDValue FoldedVOp = SimplifyVBinOp(N))
1690       return FoldedVOp;
1691 
1692     // fold (add x, 0) -> x, vector edition
1693     if (ISD::isBuildVectorAllZeros(N1.getNode()))
1694       return N0;
1695     if (ISD::isBuildVectorAllZeros(N0.getNode()))
1696       return N1;
1697   }
1698 
1699   // fold (add x, undef) -> undef
1700   if (N0.isUndef())
1701     return N0;
1702   if (N1.isUndef())
1703     return N1;
1704   if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
1705     // canonicalize constant to RHS
1706     if (!DAG.isConstantIntBuildVectorOrConstantInt(N1))
1707       return DAG.getNode(ISD::ADD, SDLoc(N), VT, N1, N0);
1708     // fold (add c1, c2) -> c1+c2
1709     return DAG.FoldConstantArithmetic(ISD::ADD, SDLoc(N), VT,
1710                                       N0.getNode(), N1.getNode());
1711   }
1712   // fold (add x, 0) -> x
1713   if (isNullConstant(N1))
1714     return N0;
1715   // fold ((c1-A)+c2) -> (c1+c2)-A
1716   if (isConstantOrConstantVector(N1, /* NoOpaque */ true)) {
1717     if (N0.getOpcode() == ISD::SUB)
1718       if (isConstantOrConstantVector(N0.getOperand(0), /* NoOpaque */ true)) {
1719         SDLoc DL(N);
1720         return DAG.getNode(ISD::SUB, DL, VT,
1721                            DAG.getNode(ISD::ADD, DL, VT, N1, N0.getOperand(0)),
1722                            N0.getOperand(1));
1723       }
1724   }
1725   // reassociate add
1726   if (SDValue RADD = ReassociateOps(ISD::ADD, SDLoc(N), N0, N1))
1727     return RADD;
1728   // fold ((0-A) + B) -> B-A
1729   if (N0.getOpcode() == ISD::SUB &&
1730       isNullConstantOrNullSplatConstant(N0.getOperand(0)))
1731     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1, N0.getOperand(1));
1732   // fold (A + (0-B)) -> A-B
1733   if (N1.getOpcode() == ISD::SUB &&
1734       isNullConstantOrNullSplatConstant(N1.getOperand(0)))
1735     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N0, N1.getOperand(1));
1736   // fold (A+(B-A)) -> B
1737   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(1))
1738     return N1.getOperand(0);
1739   // fold ((B-A)+A) -> B
1740   if (N0.getOpcode() == ISD::SUB && N1 == N0.getOperand(1))
1741     return N0.getOperand(0);
1742   // fold (A+(B-(A+C))) to (B-C)
1743   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
1744       N0 == N1.getOperand(1).getOperand(0))
1745     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1.getOperand(0),
1746                        N1.getOperand(1).getOperand(1));
1747   // fold (A+(B-(C+A))) to (B-C)
1748   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
1749       N0 == N1.getOperand(1).getOperand(1))
1750     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1.getOperand(0),
1751                        N1.getOperand(1).getOperand(0));
1752   // fold (A+((B-A)+or-C)) to (B+or-C)
1753   if ((N1.getOpcode() == ISD::SUB || N1.getOpcode() == ISD::ADD) &&
1754       N1.getOperand(0).getOpcode() == ISD::SUB &&
1755       N0 == N1.getOperand(0).getOperand(1))
1756     return DAG.getNode(N1.getOpcode(), SDLoc(N), VT,
1757                        N1.getOperand(0).getOperand(0), N1.getOperand(1));
1758 
1759   // fold (A-B)+(C-D) to (A+C)-(B+D) when A or C is constant
1760   if (N0.getOpcode() == ISD::SUB && N1.getOpcode() == ISD::SUB) {
1761     SDValue N00 = N0.getOperand(0);
1762     SDValue N01 = N0.getOperand(1);
1763     SDValue N10 = N1.getOperand(0);
1764     SDValue N11 = N1.getOperand(1);
1765 
1766     if (isConstantOrConstantVector(N00) ||
1767         isConstantOrConstantVector(N10))
1768       return DAG.getNode(ISD::SUB, SDLoc(N), VT,
1769                          DAG.getNode(ISD::ADD, SDLoc(N0), VT, N00, N10),
1770                          DAG.getNode(ISD::ADD, SDLoc(N1), VT, N01, N11));
1771   }
1772 
1773   if (SimplifyDemandedBits(SDValue(N, 0)))
1774     return SDValue(N, 0);
1775 
1776   // fold (a+b) -> (a|b) iff a and b share no bits.
1777   if ((!LegalOperations || TLI.isOperationLegal(ISD::OR, VT)) &&
1778       VT.isInteger() && DAG.haveNoCommonBitsSet(N0, N1))
1779     return DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N1);
1780 
1781   // fold (add x, shl(0 - y, n)) -> sub(x, shl(y, n))
1782   if (N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::SUB &&
1783       isNullConstantOrNullSplatConstant(N1.getOperand(0).getOperand(0)))
1784     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N0,
1785                        DAG.getNode(ISD::SHL, SDLoc(N), VT,
1786                                    N1.getOperand(0).getOperand(1),
1787                                    N1.getOperand(1)));
1788   if (N0.getOpcode() == ISD::SHL && N0.getOperand(0).getOpcode() == ISD::SUB &&
1789       isNullConstantOrNullSplatConstant(N0.getOperand(0).getOperand(0)))
1790     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1,
1791                        DAG.getNode(ISD::SHL, SDLoc(N), VT,
1792                                    N0.getOperand(0).getOperand(1),
1793                                    N0.getOperand(1)));
1794 
1795   if (N1.getOpcode() == ISD::AND) {
1796     SDValue AndOp0 = N1.getOperand(0);
1797     unsigned NumSignBits = DAG.ComputeNumSignBits(AndOp0);
1798     unsigned DestBits = VT.getScalarSizeInBits();
1799 
1800     // (add z, (and (sbbl x, x), 1)) -> (sub z, (sbbl x, x))
1801     // and similar xforms where the inner op is either ~0 or 0.
1802     if (NumSignBits == DestBits &&
1803         isOneConstantOrOneSplatConstant(N1->getOperand(1))) {
1804       SDLoc DL(N);
1805       return DAG.getNode(ISD::SUB, DL, VT, N->getOperand(0), AndOp0);
1806     }
1807   }
1808 
1809   // add (sext i1), X -> sub X, (zext i1)
1810   if (N0.getOpcode() == ISD::SIGN_EXTEND &&
1811       N0.getOperand(0).getValueType() == MVT::i1 &&
1812       !TLI.isOperationLegal(ISD::SIGN_EXTEND, MVT::i1)) {
1813     SDLoc DL(N);
1814     SDValue ZExt = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0));
1815     return DAG.getNode(ISD::SUB, DL, VT, N1, ZExt);
1816   }
1817 
1818   // add X, (sextinreg Y i1) -> sub X, (and Y 1)
1819   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
1820     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
1821     if (TN->getVT() == MVT::i1) {
1822       SDLoc DL(N);
1823       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
1824                                  DAG.getConstant(1, DL, VT));
1825       return DAG.getNode(ISD::SUB, DL, VT, N0, ZExt);
1826     }
1827   }
1828 
1829   return SDValue();
1830 }
1831 
1832 SDValue DAGCombiner::visitADDC(SDNode *N) {
1833   SDValue N0 = N->getOperand(0);
1834   SDValue N1 = N->getOperand(1);
1835   EVT VT = N0.getValueType();
1836 
1837   // If the flag result is dead, turn this into an ADD.
1838   if (!N->hasAnyUseOfValue(1))
1839     return CombineTo(N, DAG.getNode(ISD::ADD, SDLoc(N), VT, N0, N1),
1840                      DAG.getNode(ISD::CARRY_FALSE,
1841                                  SDLoc(N), MVT::Glue));
1842 
1843   // canonicalize constant to RHS.
1844   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
1845   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
1846   if (N0C && !N1C)
1847     return DAG.getNode(ISD::ADDC, SDLoc(N), N->getVTList(), N1, N0);
1848 
1849   // fold (addc x, 0) -> x + no carry out
1850   if (isNullConstant(N1))
1851     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE,
1852                                         SDLoc(N), MVT::Glue));
1853 
1854   // fold (addc a, b) -> (or a, b), CARRY_FALSE iff a and b share no bits.
1855   APInt LHSZero, LHSOne;
1856   APInt RHSZero, RHSOne;
1857   DAG.computeKnownBits(N0, LHSZero, LHSOne);
1858 
1859   if (LHSZero.getBoolValue()) {
1860     DAG.computeKnownBits(N1, RHSZero, RHSOne);
1861 
1862     // If all possibly-set bits on the LHS are clear on the RHS, return an OR.
1863     // If all possibly-set bits on the RHS are clear on the LHS, return an OR.
1864     if ((RHSZero & ~LHSZero) == ~LHSZero || (LHSZero & ~RHSZero) == ~RHSZero)
1865       return CombineTo(N, DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N1),
1866                        DAG.getNode(ISD::CARRY_FALSE,
1867                                    SDLoc(N), MVT::Glue));
1868   }
1869 
1870   return SDValue();
1871 }
1872 
1873 SDValue DAGCombiner::visitADDE(SDNode *N) {
1874   SDValue N0 = N->getOperand(0);
1875   SDValue N1 = N->getOperand(1);
1876   SDValue CarryIn = N->getOperand(2);
1877 
1878   // canonicalize constant to RHS
1879   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
1880   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
1881   if (N0C && !N1C)
1882     return DAG.getNode(ISD::ADDE, SDLoc(N), N->getVTList(),
1883                        N1, N0, CarryIn);
1884 
1885   // fold (adde x, y, false) -> (addc x, y)
1886   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
1887     return DAG.getNode(ISD::ADDC, SDLoc(N), N->getVTList(), N0, N1);
1888 
1889   return SDValue();
1890 }
1891 
1892 // Since it may not be valid to emit a fold to zero for vector initializers
1893 // check if we can before folding.
1894 static SDValue tryFoldToZero(const SDLoc &DL, const TargetLowering &TLI, EVT VT,
1895                              SelectionDAG &DAG, bool LegalOperations,
1896                              bool LegalTypes) {
1897   if (!VT.isVector())
1898     return DAG.getConstant(0, DL, VT);
1899   if (!LegalOperations || TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
1900     return DAG.getConstant(0, DL, VT);
1901   return SDValue();
1902 }
1903 
1904 SDValue DAGCombiner::visitSUB(SDNode *N) {
1905   SDValue N0 = N->getOperand(0);
1906   SDValue N1 = N->getOperand(1);
1907   EVT VT = N0.getValueType();
1908   SDLoc DL(N);
1909 
1910   // fold vector ops
1911   if (VT.isVector()) {
1912     if (SDValue FoldedVOp = SimplifyVBinOp(N))
1913       return FoldedVOp;
1914 
1915     // fold (sub x, 0) -> x, vector edition
1916     if (ISD::isBuildVectorAllZeros(N1.getNode()))
1917       return N0;
1918   }
1919 
1920   // fold (sub x, x) -> 0
1921   // FIXME: Refactor this and xor and other similar operations together.
1922   if (N0 == N1)
1923     return tryFoldToZero(DL, TLI, VT, DAG, LegalOperations, LegalTypes);
1924   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
1925       DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
1926     // fold (sub c1, c2) -> c1-c2
1927     return DAG.FoldConstantArithmetic(ISD::SUB, DL, VT, N0.getNode(),
1928                                       N1.getNode());
1929   }
1930 
1931   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
1932   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
1933 
1934   // fold (sub x, c) -> (add x, -c)
1935   if (N1C) {
1936     return DAG.getNode(ISD::ADD, DL, VT, N0,
1937                        DAG.getConstant(-N1C->getAPIntValue(), DL, VT));
1938   }
1939 
1940   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1)
1941   if (isAllOnesConstant(N0))
1942     return DAG.getNode(ISD::XOR, DL, VT, N1, N0);
1943 
1944   // fold A-(A-B) -> B
1945   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(0))
1946     return N1.getOperand(1);
1947 
1948   // fold (A+B)-A -> B
1949   if (N0.getOpcode() == ISD::ADD && N0.getOperand(0) == N1)
1950     return N0.getOperand(1);
1951 
1952   // fold (A+B)-B -> A
1953   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1) == N1)
1954     return N0.getOperand(0);
1955 
1956   // fold C2-(A+C1) -> (C2-C1)-A
1957   if (N1.getOpcode() == ISD::ADD && N0C) {
1958     if (auto *N1C1 = dyn_cast<ConstantSDNode>(N1.getOperand(1).getNode())) {
1959       SDValue NewC =
1960           DAG.getConstant(N0C->getAPIntValue() - N1C1->getAPIntValue(), DL, VT);
1961       return DAG.getNode(ISD::SUB, DL, VT, NewC, N1.getOperand(0));
1962     }
1963   }
1964 
1965   // fold ((A+(B+or-C))-B) -> A+or-C
1966   if (N0.getOpcode() == ISD::ADD &&
1967       (N0.getOperand(1).getOpcode() == ISD::SUB ||
1968        N0.getOperand(1).getOpcode() == ISD::ADD) &&
1969       N0.getOperand(1).getOperand(0) == N1)
1970     return DAG.getNode(N0.getOperand(1).getOpcode(), DL, VT, N0.getOperand(0),
1971                        N0.getOperand(1).getOperand(1));
1972 
1973   // fold ((A+(C+B))-B) -> A+C
1974   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1).getOpcode() == ISD::ADD &&
1975       N0.getOperand(1).getOperand(1) == N1)
1976     return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0),
1977                        N0.getOperand(1).getOperand(0));
1978 
1979   // fold ((A-(B-C))-C) -> A-B
1980   if (N0.getOpcode() == ISD::SUB && N0.getOperand(1).getOpcode() == ISD::SUB &&
1981       N0.getOperand(1).getOperand(1) == N1)
1982     return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0),
1983                        N0.getOperand(1).getOperand(0));
1984 
1985   // If either operand of a sub is undef, the result is undef
1986   if (N0.isUndef())
1987     return N0;
1988   if (N1.isUndef())
1989     return N1;
1990 
1991   // If the relocation model supports it, consider symbol offsets.
1992   if (GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(N0))
1993     if (!LegalOperations && TLI.isOffsetFoldingLegal(GA)) {
1994       // fold (sub Sym, c) -> Sym-c
1995       if (N1C && GA->getOpcode() == ISD::GlobalAddress)
1996         return DAG.getGlobalAddress(GA->getGlobal(), SDLoc(N1C), VT,
1997                                     GA->getOffset() -
1998                                         (uint64_t)N1C->getSExtValue());
1999       // fold (sub Sym+c1, Sym+c2) -> c1-c2
2000       if (GlobalAddressSDNode *GB = dyn_cast<GlobalAddressSDNode>(N1))
2001         if (GA->getGlobal() == GB->getGlobal())
2002           return DAG.getConstant((uint64_t)GA->getOffset() - GB->getOffset(),
2003                                  DL, VT);
2004     }
2005 
2006   // sub X, (sextinreg Y i1) -> add X, (and Y 1)
2007   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
2008     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
2009     if (TN->getVT() == MVT::i1) {
2010       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
2011                                  DAG.getConstant(1, DL, VT));
2012       return DAG.getNode(ISD::ADD, DL, VT, N0, ZExt);
2013     }
2014   }
2015 
2016   return SDValue();
2017 }
2018 
2019 SDValue DAGCombiner::visitSUBC(SDNode *N) {
2020   SDValue N0 = N->getOperand(0);
2021   SDValue N1 = N->getOperand(1);
2022   EVT VT = N0.getValueType();
2023   SDLoc DL(N);
2024 
2025   // If the flag result is dead, turn this into an SUB.
2026   if (!N->hasAnyUseOfValue(1))
2027     return CombineTo(N, DAG.getNode(ISD::SUB, DL, VT, N0, N1),
2028                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2029 
2030   // fold (subc x, x) -> 0 + no borrow
2031   if (N0 == N1)
2032     return CombineTo(N, DAG.getConstant(0, DL, VT),
2033                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2034 
2035   // fold (subc x, 0) -> x + no borrow
2036   if (isNullConstant(N1))
2037     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2038 
2039   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1) + no borrow
2040   if (isAllOnesConstant(N0))
2041     return CombineTo(N, DAG.getNode(ISD::XOR, DL, VT, N1, N0),
2042                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2043 
2044   return SDValue();
2045 }
2046 
2047 SDValue DAGCombiner::visitSUBE(SDNode *N) {
2048   SDValue N0 = N->getOperand(0);
2049   SDValue N1 = N->getOperand(1);
2050   SDValue CarryIn = N->getOperand(2);
2051 
2052   // fold (sube x, y, false) -> (subc x, y)
2053   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
2054     return DAG.getNode(ISD::SUBC, SDLoc(N), N->getVTList(), N0, N1);
2055 
2056   return SDValue();
2057 }
2058 
2059 SDValue DAGCombiner::visitMUL(SDNode *N) {
2060   SDValue N0 = N->getOperand(0);
2061   SDValue N1 = N->getOperand(1);
2062   EVT VT = N0.getValueType();
2063 
2064   // fold (mul x, undef) -> 0
2065   if (N0.isUndef() || N1.isUndef())
2066     return DAG.getConstant(0, SDLoc(N), VT);
2067 
2068   bool N0IsConst = false;
2069   bool N1IsConst = false;
2070   bool N1IsOpaqueConst = false;
2071   bool N0IsOpaqueConst = false;
2072   APInt ConstValue0, ConstValue1;
2073   // fold vector ops
2074   if (VT.isVector()) {
2075     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2076       return FoldedVOp;
2077 
2078     N0IsConst = ISD::isConstantSplatVector(N0.getNode(), ConstValue0);
2079     N1IsConst = ISD::isConstantSplatVector(N1.getNode(), ConstValue1);
2080   } else {
2081     N0IsConst = isa<ConstantSDNode>(N0);
2082     if (N0IsConst) {
2083       ConstValue0 = cast<ConstantSDNode>(N0)->getAPIntValue();
2084       N0IsOpaqueConst = cast<ConstantSDNode>(N0)->isOpaque();
2085     }
2086     N1IsConst = isa<ConstantSDNode>(N1);
2087     if (N1IsConst) {
2088       ConstValue1 = cast<ConstantSDNode>(N1)->getAPIntValue();
2089       N1IsOpaqueConst = cast<ConstantSDNode>(N1)->isOpaque();
2090     }
2091   }
2092 
2093   // fold (mul c1, c2) -> c1*c2
2094   if (N0IsConst && N1IsConst && !N0IsOpaqueConst && !N1IsOpaqueConst)
2095     return DAG.FoldConstantArithmetic(ISD::MUL, SDLoc(N), VT,
2096                                       N0.getNode(), N1.getNode());
2097 
2098   // canonicalize constant to RHS (vector doesn't have to splat)
2099   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2100      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
2101     return DAG.getNode(ISD::MUL, SDLoc(N), VT, N1, N0);
2102   // fold (mul x, 0) -> 0
2103   if (N1IsConst && ConstValue1 == 0)
2104     return N1;
2105   // We require a splat of the entire scalar bit width for non-contiguous
2106   // bit patterns.
2107   bool IsFullSplat =
2108     ConstValue1.getBitWidth() == VT.getScalarSizeInBits();
2109   // fold (mul x, 1) -> x
2110   if (N1IsConst && ConstValue1 == 1 && IsFullSplat)
2111     return N0;
2112   // fold (mul x, -1) -> 0-x
2113   if (N1IsConst && ConstValue1.isAllOnesValue()) {
2114     SDLoc DL(N);
2115     return DAG.getNode(ISD::SUB, DL, VT,
2116                        DAG.getConstant(0, DL, VT), N0);
2117   }
2118   // fold (mul x, (1 << c)) -> x << c
2119   if (N1IsConst && !N1IsOpaqueConst && ConstValue1.isPowerOf2() &&
2120       IsFullSplat) {
2121     SDLoc DL(N);
2122     return DAG.getNode(ISD::SHL, DL, VT, N0,
2123                        DAG.getConstant(ConstValue1.logBase2(), DL,
2124                                        getShiftAmountTy(N0.getValueType())));
2125   }
2126   // fold (mul x, -(1 << c)) -> -(x << c) or (-x) << c
2127   if (N1IsConst && !N1IsOpaqueConst && (-ConstValue1).isPowerOf2() &&
2128       IsFullSplat) {
2129     unsigned Log2Val = (-ConstValue1).logBase2();
2130     SDLoc DL(N);
2131     // FIXME: If the input is something that is easily negated (e.g. a
2132     // single-use add), we should put the negate there.
2133     return DAG.getNode(ISD::SUB, DL, VT,
2134                        DAG.getConstant(0, DL, VT),
2135                        DAG.getNode(ISD::SHL, DL, VT, N0,
2136                             DAG.getConstant(Log2Val, DL,
2137                                       getShiftAmountTy(N0.getValueType()))));
2138   }
2139 
2140   APInt Val;
2141   // (mul (shl X, c1), c2) -> (mul X, c2 << c1)
2142   if (N1IsConst && N0.getOpcode() == ISD::SHL &&
2143       (ISD::isConstantSplatVector(N0.getOperand(1).getNode(), Val) ||
2144        isa<ConstantSDNode>(N0.getOperand(1)))) {
2145     SDValue C3 = DAG.getNode(ISD::SHL, SDLoc(N), VT, N1, N0.getOperand(1));
2146     AddToWorklist(C3.getNode());
2147     return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), C3);
2148   }
2149 
2150   // Change (mul (shl X, C), Y) -> (shl (mul X, Y), C) when the shift has one
2151   // use.
2152   {
2153     SDValue Sh(nullptr, 0), Y(nullptr, 0);
2154     // Check for both (mul (shl X, C), Y)  and  (mul Y, (shl X, C)).
2155     if (N0.getOpcode() == ISD::SHL &&
2156         (ISD::isConstantSplatVector(N0.getOperand(1).getNode(), Val) ||
2157          isa<ConstantSDNode>(N0.getOperand(1))) &&
2158         N0.getNode()->hasOneUse()) {
2159       Sh = N0; Y = N1;
2160     } else if (N1.getOpcode() == ISD::SHL &&
2161                isa<ConstantSDNode>(N1.getOperand(1)) &&
2162                N1.getNode()->hasOneUse()) {
2163       Sh = N1; Y = N0;
2164     }
2165 
2166     if (Sh.getNode()) {
2167       SDValue Mul = DAG.getNode(ISD::MUL, SDLoc(N), VT, Sh.getOperand(0), Y);
2168       return DAG.getNode(ISD::SHL, SDLoc(N), VT, Mul, Sh.getOperand(1));
2169     }
2170   }
2171 
2172   // fold (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2)
2173   if (DAG.isConstantIntBuildVectorOrConstantInt(N1) &&
2174       N0.getOpcode() == ISD::ADD &&
2175       DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1)) &&
2176       isMulAddWithConstProfitable(N, N0, N1))
2177       return DAG.getNode(ISD::ADD, SDLoc(N), VT,
2178                          DAG.getNode(ISD::MUL, SDLoc(N0), VT,
2179                                      N0.getOperand(0), N1),
2180                          DAG.getNode(ISD::MUL, SDLoc(N1), VT,
2181                                      N0.getOperand(1), N1));
2182 
2183   // reassociate mul
2184   if (SDValue RMUL = ReassociateOps(ISD::MUL, SDLoc(N), N0, N1))
2185     return RMUL;
2186 
2187   return SDValue();
2188 }
2189 
2190 /// Return true if divmod libcall is available.
2191 static bool isDivRemLibcallAvailable(SDNode *Node, bool isSigned,
2192                                      const TargetLowering &TLI) {
2193   RTLIB::Libcall LC;
2194   EVT NodeType = Node->getValueType(0);
2195   if (!NodeType.isSimple())
2196     return false;
2197   switch (NodeType.getSimpleVT().SimpleTy) {
2198   default: return false; // No libcall for vector types.
2199   case MVT::i8:   LC= isSigned ? RTLIB::SDIVREM_I8  : RTLIB::UDIVREM_I8;  break;
2200   case MVT::i16:  LC= isSigned ? RTLIB::SDIVREM_I16 : RTLIB::UDIVREM_I16; break;
2201   case MVT::i32:  LC= isSigned ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32; break;
2202   case MVT::i64:  LC= isSigned ? RTLIB::SDIVREM_I64 : RTLIB::UDIVREM_I64; break;
2203   case MVT::i128: LC= isSigned ? RTLIB::SDIVREM_I128:RTLIB::UDIVREM_I128; break;
2204   }
2205 
2206   return TLI.getLibcallName(LC) != nullptr;
2207 }
2208 
2209 /// Issue divrem if both quotient and remainder are needed.
2210 SDValue DAGCombiner::useDivRem(SDNode *Node) {
2211   if (Node->use_empty())
2212     return SDValue(); // This is a dead node, leave it alone.
2213 
2214   unsigned Opcode = Node->getOpcode();
2215   bool isSigned = (Opcode == ISD::SDIV) || (Opcode == ISD::SREM);
2216   unsigned DivRemOpc = isSigned ? ISD::SDIVREM : ISD::UDIVREM;
2217 
2218   // DivMod lib calls can still work on non-legal types if using lib-calls.
2219   EVT VT = Node->getValueType(0);
2220   if (VT.isVector() || !VT.isInteger())
2221     return SDValue();
2222 
2223   if (!TLI.isTypeLegal(VT) && !TLI.isOperationCustom(DivRemOpc, VT))
2224     return SDValue();
2225 
2226   // If DIVREM is going to get expanded into a libcall,
2227   // but there is no libcall available, then don't combine.
2228   if (!TLI.isOperationLegalOrCustom(DivRemOpc, VT) &&
2229       !isDivRemLibcallAvailable(Node, isSigned, TLI))
2230     return SDValue();
2231 
2232   // If div is legal, it's better to do the normal expansion
2233   unsigned OtherOpcode = 0;
2234   if ((Opcode == ISD::SDIV) || (Opcode == ISD::UDIV)) {
2235     OtherOpcode = isSigned ? ISD::SREM : ISD::UREM;
2236     if (TLI.isOperationLegalOrCustom(Opcode, VT))
2237       return SDValue();
2238   } else {
2239     OtherOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
2240     if (TLI.isOperationLegalOrCustom(OtherOpcode, VT))
2241       return SDValue();
2242   }
2243 
2244   SDValue Op0 = Node->getOperand(0);
2245   SDValue Op1 = Node->getOperand(1);
2246   SDValue combined;
2247   for (SDNode::use_iterator UI = Op0.getNode()->use_begin(),
2248          UE = Op0.getNode()->use_end(); UI != UE; ++UI) {
2249     SDNode *User = *UI;
2250     if (User == Node || User->use_empty())
2251       continue;
2252     // Convert the other matching node(s), too;
2253     // otherwise, the DIVREM may get target-legalized into something
2254     // target-specific that we won't be able to recognize.
2255     unsigned UserOpc = User->getOpcode();
2256     if ((UserOpc == Opcode || UserOpc == OtherOpcode || UserOpc == DivRemOpc) &&
2257         User->getOperand(0) == Op0 &&
2258         User->getOperand(1) == Op1) {
2259       if (!combined) {
2260         if (UserOpc == OtherOpcode) {
2261           SDVTList VTs = DAG.getVTList(VT, VT);
2262           combined = DAG.getNode(DivRemOpc, SDLoc(Node), VTs, Op0, Op1);
2263         } else if (UserOpc == DivRemOpc) {
2264           combined = SDValue(User, 0);
2265         } else {
2266           assert(UserOpc == Opcode);
2267           continue;
2268         }
2269       }
2270       if (UserOpc == ISD::SDIV || UserOpc == ISD::UDIV)
2271         CombineTo(User, combined);
2272       else if (UserOpc == ISD::SREM || UserOpc == ISD::UREM)
2273         CombineTo(User, combined.getValue(1));
2274     }
2275   }
2276   return combined;
2277 }
2278 
2279 SDValue DAGCombiner::visitSDIV(SDNode *N) {
2280   SDValue N0 = N->getOperand(0);
2281   SDValue N1 = N->getOperand(1);
2282   EVT VT = N->getValueType(0);
2283 
2284   // fold vector ops
2285   if (VT.isVector())
2286     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2287       return FoldedVOp;
2288 
2289   SDLoc DL(N);
2290 
2291   // fold (sdiv c1, c2) -> c1/c2
2292   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2293   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2294   if (N0C && N1C && !N0C->isOpaque() && !N1C->isOpaque())
2295     return DAG.FoldConstantArithmetic(ISD::SDIV, DL, VT, N0C, N1C);
2296   // fold (sdiv X, 1) -> X
2297   if (N1C && N1C->isOne())
2298     return N0;
2299   // fold (sdiv X, -1) -> 0-X
2300   if (N1C && N1C->isAllOnesValue())
2301     return DAG.getNode(ISD::SUB, DL, VT,
2302                        DAG.getConstant(0, DL, VT), N0);
2303 
2304   // If we know the sign bits of both operands are zero, strength reduce to a
2305   // udiv instead.  Handles (X&15) /s 4 -> X&15 >> 2
2306   if (!VT.isVector()) {
2307     if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
2308       return DAG.getNode(ISD::UDIV, DL, N1.getValueType(), N0, N1);
2309   }
2310 
2311   // fold (sdiv X, pow2) -> simple ops after legalize
2312   // FIXME: We check for the exact bit here because the generic lowering gives
2313   // better results in that case. The target-specific lowering should learn how
2314   // to handle exact sdivs efficiently.
2315   if (N1C && !N1C->isNullValue() && !N1C->isOpaque() &&
2316       !cast<BinaryWithFlagsSDNode>(N)->Flags.hasExact() &&
2317       (N1C->getAPIntValue().isPowerOf2() ||
2318        (-N1C->getAPIntValue()).isPowerOf2())) {
2319     // Target-specific implementation of sdiv x, pow2.
2320     if (SDValue Res = BuildSDIVPow2(N))
2321       return Res;
2322 
2323     unsigned lg2 = N1C->getAPIntValue().countTrailingZeros();
2324 
2325     // Splat the sign bit into the register
2326     SDValue SGN =
2327         DAG.getNode(ISD::SRA, DL, VT, N0,
2328                     DAG.getConstant(VT.getScalarSizeInBits() - 1, DL,
2329                                     getShiftAmountTy(N0.getValueType())));
2330     AddToWorklist(SGN.getNode());
2331 
2332     // Add (N0 < 0) ? abs2 - 1 : 0;
2333     SDValue SRL =
2334         DAG.getNode(ISD::SRL, DL, VT, SGN,
2335                     DAG.getConstant(VT.getScalarSizeInBits() - lg2, DL,
2336                                     getShiftAmountTy(SGN.getValueType())));
2337     SDValue ADD = DAG.getNode(ISD::ADD, DL, VT, N0, SRL);
2338     AddToWorklist(SRL.getNode());
2339     AddToWorklist(ADD.getNode());    // Divide by pow2
2340     SDValue SRA = DAG.getNode(ISD::SRA, DL, VT, ADD,
2341                   DAG.getConstant(lg2, DL,
2342                                   getShiftAmountTy(ADD.getValueType())));
2343 
2344     // If we're dividing by a positive value, we're done.  Otherwise, we must
2345     // negate the result.
2346     if (N1C->getAPIntValue().isNonNegative())
2347       return SRA;
2348 
2349     AddToWorklist(SRA.getNode());
2350     return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), SRA);
2351   }
2352 
2353   // If integer divide is expensive and we satisfy the requirements, emit an
2354   // alternate sequence.  Targets may check function attributes for size/speed
2355   // trade-offs.
2356   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2357   if (N1C && !TLI.isIntDivCheap(N->getValueType(0), Attr))
2358     if (SDValue Op = BuildSDIV(N))
2359       return Op;
2360 
2361   // sdiv, srem -> sdivrem
2362   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is true.
2363   // Otherwise, we break the simplification logic in visitREM().
2364   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
2365     if (SDValue DivRem = useDivRem(N))
2366         return DivRem;
2367 
2368   // undef / X -> 0
2369   if (N0.isUndef())
2370     return DAG.getConstant(0, DL, VT);
2371   // X / undef -> undef
2372   if (N1.isUndef())
2373     return N1;
2374 
2375   return SDValue();
2376 }
2377 
2378 SDValue DAGCombiner::visitUDIV(SDNode *N) {
2379   SDValue N0 = N->getOperand(0);
2380   SDValue N1 = N->getOperand(1);
2381   EVT VT = N->getValueType(0);
2382 
2383   // fold vector ops
2384   if (VT.isVector())
2385     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2386       return FoldedVOp;
2387 
2388   SDLoc DL(N);
2389 
2390   // fold (udiv c1, c2) -> c1/c2
2391   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2392   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2393   if (N0C && N1C)
2394     if (SDValue Folded = DAG.FoldConstantArithmetic(ISD::UDIV, DL, VT,
2395                                                     N0C, N1C))
2396       return Folded;
2397   // fold (udiv x, (1 << c)) -> x >>u c
2398   if (N1C && !N1C->isOpaque() && N1C->getAPIntValue().isPowerOf2())
2399     return DAG.getNode(ISD::SRL, DL, VT, N0,
2400                        DAG.getConstant(N1C->getAPIntValue().logBase2(), DL,
2401                                        getShiftAmountTy(N0.getValueType())));
2402 
2403   // fold (udiv x, (shl c, y)) -> x >>u (log2(c)+y) iff c is power of 2
2404   if (N1.getOpcode() == ISD::SHL) {
2405     if (ConstantSDNode *SHC = getAsNonOpaqueConstant(N1.getOperand(0))) {
2406       if (SHC->getAPIntValue().isPowerOf2()) {
2407         EVT ADDVT = N1.getOperand(1).getValueType();
2408         SDValue Add = DAG.getNode(ISD::ADD, DL, ADDVT,
2409                                   N1.getOperand(1),
2410                                   DAG.getConstant(SHC->getAPIntValue()
2411                                                                   .logBase2(),
2412                                                   DL, ADDVT));
2413         AddToWorklist(Add.getNode());
2414         return DAG.getNode(ISD::SRL, DL, VT, N0, Add);
2415       }
2416     }
2417   }
2418 
2419   // fold (udiv x, c) -> alternate
2420   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2421   if (N1C && !TLI.isIntDivCheap(N->getValueType(0), Attr))
2422     if (SDValue Op = BuildUDIV(N))
2423       return Op;
2424 
2425   // sdiv, srem -> sdivrem
2426   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is true.
2427   // Otherwise, we break the simplification logic in visitREM().
2428   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
2429     if (SDValue DivRem = useDivRem(N))
2430         return DivRem;
2431 
2432   // undef / X -> 0
2433   if (N0.isUndef())
2434     return DAG.getConstant(0, DL, VT);
2435   // X / undef -> undef
2436   if (N1.isUndef())
2437     return N1;
2438 
2439   return SDValue();
2440 }
2441 
2442 // handles ISD::SREM and ISD::UREM
2443 SDValue DAGCombiner::visitREM(SDNode *N) {
2444   unsigned Opcode = N->getOpcode();
2445   SDValue N0 = N->getOperand(0);
2446   SDValue N1 = N->getOperand(1);
2447   EVT VT = N->getValueType(0);
2448   bool isSigned = (Opcode == ISD::SREM);
2449   SDLoc DL(N);
2450 
2451   // fold (rem c1, c2) -> c1%c2
2452   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2453   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2454   if (N0C && N1C)
2455     if (SDValue Folded = DAG.FoldConstantArithmetic(Opcode, DL, VT, N0C, N1C))
2456       return Folded;
2457 
2458   if (isSigned) {
2459     // If we know the sign bits of both operands are zero, strength reduce to a
2460     // urem instead.  Handles (X & 0x0FFFFFFF) %s 16 -> X&15
2461     if (!VT.isVector()) {
2462       if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
2463         return DAG.getNode(ISD::UREM, DL, VT, N0, N1);
2464     }
2465   } else {
2466     // fold (urem x, pow2) -> (and x, pow2-1)
2467     if (N1C && !N1C->isNullValue() && !N1C->isOpaque() &&
2468         N1C->getAPIntValue().isPowerOf2()) {
2469       return DAG.getNode(ISD::AND, DL, VT, N0,
2470                          DAG.getConstant(N1C->getAPIntValue() - 1, DL, VT));
2471     }
2472     // fold (urem x, (shl pow2, y)) -> (and x, (add (shl pow2, y), -1))
2473     if (N1.getOpcode() == ISD::SHL) {
2474       ConstantSDNode *SHC = getAsNonOpaqueConstant(N1.getOperand(0));
2475       if (SHC && SHC->getAPIntValue().isPowerOf2()) {
2476         APInt NegOne = APInt::getAllOnesValue(VT.getSizeInBits());
2477         SDValue Add =
2478             DAG.getNode(ISD::ADD, DL, VT, N1, DAG.getConstant(NegOne, DL, VT));
2479         AddToWorklist(Add.getNode());
2480         return DAG.getNode(ISD::AND, DL, VT, N0, Add);
2481       }
2482     }
2483   }
2484 
2485   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2486 
2487   // If X/C can be simplified by the division-by-constant logic, lower
2488   // X%C to the equivalent of X-X/C*C.
2489   // To avoid mangling nodes, this simplification requires that the combine()
2490   // call for the speculative DIV must not cause a DIVREM conversion.  We guard
2491   // against this by skipping the simplification if isIntDivCheap().  When
2492   // div is not cheap, combine will not return a DIVREM.  Regardless,
2493   // checking cheapness here makes sense since the simplification results in
2494   // fatter code.
2495   if (N1C && !N1C->isNullValue() && !TLI.isIntDivCheap(VT, Attr)) {
2496     unsigned DivOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
2497     SDValue Div = DAG.getNode(DivOpcode, DL, VT, N0, N1);
2498     AddToWorklist(Div.getNode());
2499     SDValue OptimizedDiv = combine(Div.getNode());
2500     if (OptimizedDiv.getNode() && OptimizedDiv.getNode() != Div.getNode()) {
2501       assert((OptimizedDiv.getOpcode() != ISD::UDIVREM) &&
2502              (OptimizedDiv.getOpcode() != ISD::SDIVREM));
2503       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, OptimizedDiv, N1);
2504       SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul);
2505       AddToWorklist(Mul.getNode());
2506       return Sub;
2507     }
2508   }
2509 
2510   // sdiv, srem -> sdivrem
2511   if (SDValue DivRem = useDivRem(N))
2512     return DivRem.getValue(1);
2513 
2514   // undef % X -> 0
2515   if (N0.isUndef())
2516     return DAG.getConstant(0, DL, VT);
2517   // X % undef -> undef
2518   if (N1.isUndef())
2519     return N1;
2520 
2521   return SDValue();
2522 }
2523 
2524 SDValue DAGCombiner::visitMULHS(SDNode *N) {
2525   SDValue N0 = N->getOperand(0);
2526   SDValue N1 = N->getOperand(1);
2527   EVT VT = N->getValueType(0);
2528   SDLoc DL(N);
2529 
2530   // fold (mulhs x, 0) -> 0
2531   if (isNullConstant(N1))
2532     return N1;
2533   // fold (mulhs x, 1) -> (sra x, size(x)-1)
2534   if (isOneConstant(N1)) {
2535     SDLoc DL(N);
2536     return DAG.getNode(ISD::SRA, DL, N0.getValueType(), N0,
2537                        DAG.getConstant(N0.getValueSizeInBits() - 1, DL,
2538                                        getShiftAmountTy(N0.getValueType())));
2539   }
2540   // fold (mulhs x, undef) -> 0
2541   if (N0.isUndef() || N1.isUndef())
2542     return DAG.getConstant(0, SDLoc(N), VT);
2543 
2544   // If the type twice as wide is legal, transform the mulhs to a wider multiply
2545   // plus a shift.
2546   if (VT.isSimple() && !VT.isVector()) {
2547     MVT Simple = VT.getSimpleVT();
2548     unsigned SimpleSize = Simple.getSizeInBits();
2549     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2550     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2551       N0 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N0);
2552       N1 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N1);
2553       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
2554       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
2555             DAG.getConstant(SimpleSize, DL,
2556                             getShiftAmountTy(N1.getValueType())));
2557       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
2558     }
2559   }
2560 
2561   return SDValue();
2562 }
2563 
2564 SDValue DAGCombiner::visitMULHU(SDNode *N) {
2565   SDValue N0 = N->getOperand(0);
2566   SDValue N1 = N->getOperand(1);
2567   EVT VT = N->getValueType(0);
2568   SDLoc DL(N);
2569 
2570   // fold (mulhu x, 0) -> 0
2571   if (isNullConstant(N1))
2572     return N1;
2573   // fold (mulhu x, 1) -> 0
2574   if (isOneConstant(N1))
2575     return DAG.getConstant(0, DL, N0.getValueType());
2576   // fold (mulhu x, undef) -> 0
2577   if (N0.isUndef() || N1.isUndef())
2578     return DAG.getConstant(0, DL, VT);
2579 
2580   // If the type twice as wide is legal, transform the mulhu to a wider multiply
2581   // plus a shift.
2582   if (VT.isSimple() && !VT.isVector()) {
2583     MVT Simple = VT.getSimpleVT();
2584     unsigned SimpleSize = Simple.getSizeInBits();
2585     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2586     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2587       N0 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N0);
2588       N1 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N1);
2589       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
2590       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
2591             DAG.getConstant(SimpleSize, DL,
2592                             getShiftAmountTy(N1.getValueType())));
2593       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
2594     }
2595   }
2596 
2597   return SDValue();
2598 }
2599 
2600 /// Perform optimizations common to nodes that compute two values. LoOp and HiOp
2601 /// give the opcodes for the two computations that are being performed. Return
2602 /// true if a simplification was made.
2603 SDValue DAGCombiner::SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
2604                                                 unsigned HiOp) {
2605   // If the high half is not needed, just compute the low half.
2606   bool HiExists = N->hasAnyUseOfValue(1);
2607   if (!HiExists &&
2608       (!LegalOperations ||
2609        TLI.isOperationLegalOrCustom(LoOp, N->getValueType(0)))) {
2610     SDValue Res = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
2611     return CombineTo(N, Res, Res);
2612   }
2613 
2614   // If the low half is not needed, just compute the high half.
2615   bool LoExists = N->hasAnyUseOfValue(0);
2616   if (!LoExists &&
2617       (!LegalOperations ||
2618        TLI.isOperationLegal(HiOp, N->getValueType(1)))) {
2619     SDValue Res = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
2620     return CombineTo(N, Res, Res);
2621   }
2622 
2623   // If both halves are used, return as it is.
2624   if (LoExists && HiExists)
2625     return SDValue();
2626 
2627   // If the two computed results can be simplified separately, separate them.
2628   if (LoExists) {
2629     SDValue Lo = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
2630     AddToWorklist(Lo.getNode());
2631     SDValue LoOpt = combine(Lo.getNode());
2632     if (LoOpt.getNode() && LoOpt.getNode() != Lo.getNode() &&
2633         (!LegalOperations ||
2634          TLI.isOperationLegal(LoOpt.getOpcode(), LoOpt.getValueType())))
2635       return CombineTo(N, LoOpt, LoOpt);
2636   }
2637 
2638   if (HiExists) {
2639     SDValue Hi = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
2640     AddToWorklist(Hi.getNode());
2641     SDValue HiOpt = combine(Hi.getNode());
2642     if (HiOpt.getNode() && HiOpt != Hi &&
2643         (!LegalOperations ||
2644          TLI.isOperationLegal(HiOpt.getOpcode(), HiOpt.getValueType())))
2645       return CombineTo(N, HiOpt, HiOpt);
2646   }
2647 
2648   return SDValue();
2649 }
2650 
2651 SDValue DAGCombiner::visitSMUL_LOHI(SDNode *N) {
2652   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHS))
2653     return Res;
2654 
2655   EVT VT = N->getValueType(0);
2656   SDLoc DL(N);
2657 
2658   // If the type is twice as wide is legal, transform the mulhu to a wider
2659   // multiply plus a shift.
2660   if (VT.isSimple() && !VT.isVector()) {
2661     MVT Simple = VT.getSimpleVT();
2662     unsigned SimpleSize = Simple.getSizeInBits();
2663     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2664     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2665       SDValue Lo = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(0));
2666       SDValue Hi = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(1));
2667       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
2668       // Compute the high part as N1.
2669       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
2670             DAG.getConstant(SimpleSize, DL,
2671                             getShiftAmountTy(Lo.getValueType())));
2672       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
2673       // Compute the low part as N0.
2674       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
2675       return CombineTo(N, Lo, Hi);
2676     }
2677   }
2678 
2679   return SDValue();
2680 }
2681 
2682 SDValue DAGCombiner::visitUMUL_LOHI(SDNode *N) {
2683   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHU))
2684     return Res;
2685 
2686   EVT VT = N->getValueType(0);
2687   SDLoc DL(N);
2688 
2689   // If the type is twice as wide is legal, transform the mulhu to a wider
2690   // multiply plus a shift.
2691   if (VT.isSimple() && !VT.isVector()) {
2692     MVT Simple = VT.getSimpleVT();
2693     unsigned SimpleSize = Simple.getSizeInBits();
2694     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2695     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2696       SDValue Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(0));
2697       SDValue Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(1));
2698       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
2699       // Compute the high part as N1.
2700       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
2701             DAG.getConstant(SimpleSize, DL,
2702                             getShiftAmountTy(Lo.getValueType())));
2703       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
2704       // Compute the low part as N0.
2705       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
2706       return CombineTo(N, Lo, Hi);
2707     }
2708   }
2709 
2710   return SDValue();
2711 }
2712 
2713 SDValue DAGCombiner::visitSMULO(SDNode *N) {
2714   // (smulo x, 2) -> (saddo x, x)
2715   if (ConstantSDNode *C2 = dyn_cast<ConstantSDNode>(N->getOperand(1)))
2716     if (C2->getAPIntValue() == 2)
2717       return DAG.getNode(ISD::SADDO, SDLoc(N), N->getVTList(),
2718                          N->getOperand(0), N->getOperand(0));
2719 
2720   return SDValue();
2721 }
2722 
2723 SDValue DAGCombiner::visitUMULO(SDNode *N) {
2724   // (umulo x, 2) -> (uaddo x, x)
2725   if (ConstantSDNode *C2 = dyn_cast<ConstantSDNode>(N->getOperand(1)))
2726     if (C2->getAPIntValue() == 2)
2727       return DAG.getNode(ISD::UADDO, SDLoc(N), N->getVTList(),
2728                          N->getOperand(0), N->getOperand(0));
2729 
2730   return SDValue();
2731 }
2732 
2733 SDValue DAGCombiner::visitIMINMAX(SDNode *N) {
2734   SDValue N0 = N->getOperand(0);
2735   SDValue N1 = N->getOperand(1);
2736   EVT VT = N0.getValueType();
2737 
2738   // fold vector ops
2739   if (VT.isVector())
2740     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2741       return FoldedVOp;
2742 
2743   // fold (add c1, c2) -> c1+c2
2744   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
2745   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
2746   if (N0C && N1C)
2747     return DAG.FoldConstantArithmetic(N->getOpcode(), SDLoc(N), VT, N0C, N1C);
2748 
2749   // canonicalize constant to RHS
2750   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2751      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
2752     return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N1, N0);
2753 
2754   return SDValue();
2755 }
2756 
2757 /// If this is a binary operator with two operands of the same opcode, try to
2758 /// simplify it.
2759 SDValue DAGCombiner::SimplifyBinOpWithSameOpcodeHands(SDNode *N) {
2760   SDValue N0 = N->getOperand(0), N1 = N->getOperand(1);
2761   EVT VT = N0.getValueType();
2762   assert(N0.getOpcode() == N1.getOpcode() && "Bad input!");
2763 
2764   // Bail early if none of these transforms apply.
2765   if (N0.getNode()->getNumOperands() == 0) return SDValue();
2766 
2767   // For each of OP in AND/OR/XOR:
2768   // fold (OP (zext x), (zext y)) -> (zext (OP x, y))
2769   // fold (OP (sext x), (sext y)) -> (sext (OP x, y))
2770   // fold (OP (aext x), (aext y)) -> (aext (OP x, y))
2771   // fold (OP (bswap x), (bswap y)) -> (bswap (OP x, y))
2772   // fold (OP (trunc x), (trunc y)) -> (trunc (OP x, y)) (if trunc isn't free)
2773   //
2774   // do not sink logical op inside of a vector extend, since it may combine
2775   // into a vsetcc.
2776   EVT Op0VT = N0.getOperand(0).getValueType();
2777   if ((N0.getOpcode() == ISD::ZERO_EXTEND ||
2778        N0.getOpcode() == ISD::SIGN_EXTEND ||
2779        N0.getOpcode() == ISD::BSWAP ||
2780        // Avoid infinite looping with PromoteIntBinOp.
2781        (N0.getOpcode() == ISD::ANY_EXTEND &&
2782         (!LegalTypes || TLI.isTypeDesirableForOp(N->getOpcode(), Op0VT))) ||
2783        (N0.getOpcode() == ISD::TRUNCATE &&
2784         (!TLI.isZExtFree(VT, Op0VT) ||
2785          !TLI.isTruncateFree(Op0VT, VT)) &&
2786         TLI.isTypeLegal(Op0VT))) &&
2787       !VT.isVector() &&
2788       Op0VT == N1.getOperand(0).getValueType() &&
2789       (!LegalOperations || TLI.isOperationLegal(N->getOpcode(), Op0VT))) {
2790     SDValue ORNode = DAG.getNode(N->getOpcode(), SDLoc(N0),
2791                                  N0.getOperand(0).getValueType(),
2792                                  N0.getOperand(0), N1.getOperand(0));
2793     AddToWorklist(ORNode.getNode());
2794     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, ORNode);
2795   }
2796 
2797   // For each of OP in SHL/SRL/SRA/AND...
2798   //   fold (and (OP x, z), (OP y, z)) -> (OP (and x, y), z)
2799   //   fold (or  (OP x, z), (OP y, z)) -> (OP (or  x, y), z)
2800   //   fold (xor (OP x, z), (OP y, z)) -> (OP (xor x, y), z)
2801   if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL ||
2802        N0.getOpcode() == ISD::SRA || N0.getOpcode() == ISD::AND) &&
2803       N0.getOperand(1) == N1.getOperand(1)) {
2804     SDValue ORNode = DAG.getNode(N->getOpcode(), SDLoc(N0),
2805                                  N0.getOperand(0).getValueType(),
2806                                  N0.getOperand(0), N1.getOperand(0));
2807     AddToWorklist(ORNode.getNode());
2808     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT,
2809                        ORNode, N0.getOperand(1));
2810   }
2811 
2812   // Simplify xor/and/or (bitcast(A), bitcast(B)) -> bitcast(op (A,B))
2813   // Only perform this optimization up until type legalization, before
2814   // LegalizeVectorOprs. LegalizeVectorOprs promotes vector operations by
2815   // adding bitcasts. For example (xor v4i32) is promoted to (v2i64), and
2816   // we don't want to undo this promotion.
2817   // We also handle SCALAR_TO_VECTOR because xor/or/and operations are cheaper
2818   // on scalars.
2819   if ((N0.getOpcode() == ISD::BITCAST ||
2820        N0.getOpcode() == ISD::SCALAR_TO_VECTOR) &&
2821        Level <= AfterLegalizeTypes) {
2822     SDValue In0 = N0.getOperand(0);
2823     SDValue In1 = N1.getOperand(0);
2824     EVT In0Ty = In0.getValueType();
2825     EVT In1Ty = In1.getValueType();
2826     SDLoc DL(N);
2827     // If both incoming values are integers, and the original types are the
2828     // same.
2829     if (In0Ty.isInteger() && In1Ty.isInteger() && In0Ty == In1Ty) {
2830       SDValue Op = DAG.getNode(N->getOpcode(), DL, In0Ty, In0, In1);
2831       SDValue BC = DAG.getNode(N0.getOpcode(), DL, VT, Op);
2832       AddToWorklist(Op.getNode());
2833       return BC;
2834     }
2835   }
2836 
2837   // Xor/and/or are indifferent to the swizzle operation (shuffle of one value).
2838   // Simplify xor/and/or (shuff(A), shuff(B)) -> shuff(op (A,B))
2839   // If both shuffles use the same mask, and both shuffle within a single
2840   // vector, then it is worthwhile to move the swizzle after the operation.
2841   // The type-legalizer generates this pattern when loading illegal
2842   // vector types from memory. In many cases this allows additional shuffle
2843   // optimizations.
2844   // There are other cases where moving the shuffle after the xor/and/or
2845   // is profitable even if shuffles don't perform a swizzle.
2846   // If both shuffles use the same mask, and both shuffles have the same first
2847   // or second operand, then it might still be profitable to move the shuffle
2848   // after the xor/and/or operation.
2849   if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG) {
2850     ShuffleVectorSDNode *SVN0 = cast<ShuffleVectorSDNode>(N0);
2851     ShuffleVectorSDNode *SVN1 = cast<ShuffleVectorSDNode>(N1);
2852 
2853     assert(N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType() &&
2854            "Inputs to shuffles are not the same type");
2855 
2856     // Check that both shuffles use the same mask. The masks are known to be of
2857     // the same length because the result vector type is the same.
2858     // Check also that shuffles have only one use to avoid introducing extra
2859     // instructions.
2860     if (SVN0->hasOneUse() && SVN1->hasOneUse() &&
2861         SVN0->getMask().equals(SVN1->getMask())) {
2862       SDValue ShOp = N0->getOperand(1);
2863 
2864       // Don't try to fold this node if it requires introducing a
2865       // build vector of all zeros that might be illegal at this stage.
2866       if (N->getOpcode() == ISD::XOR && !ShOp.isUndef()) {
2867         if (!LegalTypes)
2868           ShOp = DAG.getConstant(0, SDLoc(N), VT);
2869         else
2870           ShOp = SDValue();
2871       }
2872 
2873       // (AND (shuf (A, C), shuf (B, C)) -> shuf (AND (A, B), C)
2874       // (OR  (shuf (A, C), shuf (B, C)) -> shuf (OR  (A, B), C)
2875       // (XOR (shuf (A, C), shuf (B, C)) -> shuf (XOR (A, B), V_0)
2876       if (N0.getOperand(1) == N1.getOperand(1) && ShOp.getNode()) {
2877         SDValue NewNode = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
2878                                       N0->getOperand(0), N1->getOperand(0));
2879         AddToWorklist(NewNode.getNode());
2880         return DAG.getVectorShuffle(VT, SDLoc(N), NewNode, ShOp,
2881                                     SVN0->getMask());
2882       }
2883 
2884       // Don't try to fold this node if it requires introducing a
2885       // build vector of all zeros that might be illegal at this stage.
2886       ShOp = N0->getOperand(0);
2887       if (N->getOpcode() == ISD::XOR && !ShOp.isUndef()) {
2888         if (!LegalTypes)
2889           ShOp = DAG.getConstant(0, SDLoc(N), VT);
2890         else
2891           ShOp = SDValue();
2892       }
2893 
2894       // (AND (shuf (C, A), shuf (C, B)) -> shuf (C, AND (A, B))
2895       // (OR  (shuf (C, A), shuf (C, B)) -> shuf (C, OR  (A, B))
2896       // (XOR (shuf (C, A), shuf (C, B)) -> shuf (V_0, XOR (A, B))
2897       if (N0->getOperand(0) == N1->getOperand(0) && ShOp.getNode()) {
2898         SDValue NewNode = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
2899                                       N0->getOperand(1), N1->getOperand(1));
2900         AddToWorklist(NewNode.getNode());
2901         return DAG.getVectorShuffle(VT, SDLoc(N), ShOp, NewNode,
2902                                     SVN0->getMask());
2903       }
2904     }
2905   }
2906 
2907   return SDValue();
2908 }
2909 
2910 /// This contains all DAGCombine rules which reduce two values combined by
2911 /// an And operation to a single value. This makes them reusable in the context
2912 /// of visitSELECT(). Rules involving constants are not included as
2913 /// visitSELECT() already handles those cases.
2914 SDValue DAGCombiner::visitANDLike(SDValue N0, SDValue N1,
2915                                   SDNode *LocReference) {
2916   EVT VT = N1.getValueType();
2917 
2918   // fold (and x, undef) -> 0
2919   if (N0.isUndef() || N1.isUndef())
2920     return DAG.getConstant(0, SDLoc(LocReference), VT);
2921   // fold (and (setcc x), (setcc y)) -> (setcc (and x, y))
2922   SDValue LL, LR, RL, RR, CC0, CC1;
2923   if (isSetCCEquivalent(N0, LL, LR, CC0) && isSetCCEquivalent(N1, RL, RR, CC1)){
2924     ISD::CondCode Op0 = cast<CondCodeSDNode>(CC0)->get();
2925     ISD::CondCode Op1 = cast<CondCodeSDNode>(CC1)->get();
2926 
2927     if (LR == RR && isa<ConstantSDNode>(LR) && Op0 == Op1 &&
2928         LL.getValueType().isInteger()) {
2929       // fold (and (seteq X, 0), (seteq Y, 0)) -> (seteq (or X, Y), 0)
2930       if (isNullConstant(LR) && Op1 == ISD::SETEQ) {
2931         EVT CCVT = getSetCCResultType(LR.getValueType());
2932         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2933           SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(N0),
2934                                        LR.getValueType(), LL, RL);
2935           AddToWorklist(ORNode.getNode());
2936           return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
2937         }
2938       }
2939       if (isAllOnesConstant(LR)) {
2940         // fold (and (seteq X, -1), (seteq Y, -1)) -> (seteq (and X, Y), -1)
2941         if (Op1 == ISD::SETEQ) {
2942           EVT CCVT = getSetCCResultType(LR.getValueType());
2943           if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2944             SDValue ANDNode = DAG.getNode(ISD::AND, SDLoc(N0),
2945                                           LR.getValueType(), LL, RL);
2946             AddToWorklist(ANDNode.getNode());
2947             return DAG.getSetCC(SDLoc(LocReference), VT, ANDNode, LR, Op1);
2948           }
2949         }
2950         // fold (and (setgt X, -1), (setgt Y, -1)) -> (setgt (or X, Y), -1)
2951         if (Op1 == ISD::SETGT) {
2952           EVT CCVT = getSetCCResultType(LR.getValueType());
2953           if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2954             SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(N0),
2955                                          LR.getValueType(), LL, RL);
2956             AddToWorklist(ORNode.getNode());
2957             return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
2958           }
2959         }
2960       }
2961     }
2962     // Simplify (and (setne X, 0), (setne X, -1)) -> (setuge (add X, 1), 2)
2963     if (LL == RL && isa<ConstantSDNode>(LR) && isa<ConstantSDNode>(RR) &&
2964         Op0 == Op1 && LL.getValueType().isInteger() &&
2965       Op0 == ISD::SETNE && ((isNullConstant(LR) && isAllOnesConstant(RR)) ||
2966                             (isAllOnesConstant(LR) && isNullConstant(RR)))) {
2967       EVT CCVT = getSetCCResultType(LL.getValueType());
2968       if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2969         SDLoc DL(N0);
2970         SDValue ADDNode = DAG.getNode(ISD::ADD, DL, LL.getValueType(),
2971                                       LL, DAG.getConstant(1, DL,
2972                                                           LL.getValueType()));
2973         AddToWorklist(ADDNode.getNode());
2974         return DAG.getSetCC(SDLoc(LocReference), VT, ADDNode,
2975                             DAG.getConstant(2, DL, LL.getValueType()),
2976                             ISD::SETUGE);
2977       }
2978     }
2979     // canonicalize equivalent to ll == rl
2980     if (LL == RR && LR == RL) {
2981       Op1 = ISD::getSetCCSwappedOperands(Op1);
2982       std::swap(RL, RR);
2983     }
2984     if (LL == RL && LR == RR) {
2985       bool isInteger = LL.getValueType().isInteger();
2986       ISD::CondCode Result = ISD::getSetCCAndOperation(Op0, Op1, isInteger);
2987       if (Result != ISD::SETCC_INVALID &&
2988           (!LegalOperations ||
2989            (TLI.isCondCodeLegal(Result, LL.getSimpleValueType()) &&
2990             TLI.isOperationLegal(ISD::SETCC, LL.getValueType())))) {
2991         EVT CCVT = getSetCCResultType(LL.getValueType());
2992         if (N0.getValueType() == CCVT ||
2993             (!LegalOperations && N0.getValueType() == MVT::i1))
2994           return DAG.getSetCC(SDLoc(LocReference), N0.getValueType(),
2995                               LL, LR, Result);
2996       }
2997     }
2998   }
2999 
3000   if (N0.getOpcode() == ISD::ADD && N1.getOpcode() == ISD::SRL &&
3001       VT.getSizeInBits() <= 64) {
3002     if (ConstantSDNode *ADDI = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
3003       APInt ADDC = ADDI->getAPIntValue();
3004       if (!TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
3005         // Look for (and (add x, c1), (lshr y, c2)). If C1 wasn't a legal
3006         // immediate for an add, but it is legal if its top c2 bits are set,
3007         // transform the ADD so the immediate doesn't need to be materialized
3008         // in a register.
3009         if (ConstantSDNode *SRLI = dyn_cast<ConstantSDNode>(N1.getOperand(1))) {
3010           APInt Mask = APInt::getHighBitsSet(VT.getSizeInBits(),
3011                                              SRLI->getZExtValue());
3012           if (DAG.MaskedValueIsZero(N0.getOperand(1), Mask)) {
3013             ADDC |= Mask;
3014             if (TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
3015               SDLoc DL(N0);
3016               SDValue NewAdd =
3017                 DAG.getNode(ISD::ADD, DL, VT,
3018                             N0.getOperand(0), DAG.getConstant(ADDC, DL, VT));
3019               CombineTo(N0.getNode(), NewAdd);
3020               // Return N so it doesn't get rechecked!
3021               return SDValue(LocReference, 0);
3022             }
3023           }
3024         }
3025       }
3026     }
3027   }
3028 
3029   // Reduce bit extract of low half of an integer to the narrower type.
3030   // (and (srl i64:x, K), KMask) ->
3031   //   (i64 zero_extend (and (srl (i32 (trunc i64:x)), K)), KMask)
3032   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
3033     if (ConstantSDNode *CAnd = dyn_cast<ConstantSDNode>(N1)) {
3034       if (ConstantSDNode *CShift = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
3035         unsigned Size = VT.getSizeInBits();
3036         const APInt &AndMask = CAnd->getAPIntValue();
3037         unsigned ShiftBits = CShift->getZExtValue();
3038         unsigned MaskBits = AndMask.countTrailingOnes();
3039         EVT HalfVT = EVT::getIntegerVT(*DAG.getContext(), Size / 2);
3040 
3041         if (APIntOps::isMask(AndMask) &&
3042             // Required bits must not span the two halves of the integer and
3043             // must fit in the half size type.
3044             (ShiftBits + MaskBits <= Size / 2) &&
3045             TLI.isNarrowingProfitable(VT, HalfVT) &&
3046             TLI.isTypeDesirableForOp(ISD::AND, HalfVT) &&
3047             TLI.isTypeDesirableForOp(ISD::SRL, HalfVT) &&
3048             TLI.isTruncateFree(VT, HalfVT) &&
3049             TLI.isZExtFree(HalfVT, VT)) {
3050           // The isNarrowingProfitable is to avoid regressions on PPC and
3051           // AArch64 which match a few 64-bit bit insert / bit extract patterns
3052           // on downstream users of this. Those patterns could probably be
3053           // extended to handle extensions mixed in.
3054 
3055           SDValue SL(N0);
3056           assert(ShiftBits != 0 && MaskBits <= Size);
3057 
3058           // Extracting the highest bit of the low half.
3059           EVT ShiftVT = TLI.getShiftAmountTy(HalfVT, DAG.getDataLayout());
3060           SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, HalfVT,
3061                                       N0.getOperand(0));
3062 
3063           SDValue NewMask = DAG.getConstant(AndMask.trunc(Size / 2), SL, HalfVT);
3064           SDValue ShiftK = DAG.getConstant(ShiftBits, SL, ShiftVT);
3065           SDValue Shift = DAG.getNode(ISD::SRL, SL, HalfVT, Trunc, ShiftK);
3066           SDValue And = DAG.getNode(ISD::AND, SL, HalfVT, Shift, NewMask);
3067           return DAG.getNode(ISD::ZERO_EXTEND, SL, VT, And);
3068         }
3069       }
3070     }
3071   }
3072 
3073   return SDValue();
3074 }
3075 
3076 bool DAGCombiner::isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
3077                                    EVT LoadResultTy, EVT &ExtVT, EVT &LoadedVT,
3078                                    bool &NarrowLoad) {
3079   uint32_t ActiveBits = AndC->getAPIntValue().getActiveBits();
3080 
3081   if (ActiveBits == 0 || !APIntOps::isMask(ActiveBits, AndC->getAPIntValue()))
3082     return false;
3083 
3084   ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits);
3085   LoadedVT = LoadN->getMemoryVT();
3086 
3087   if (ExtVT == LoadedVT &&
3088       (!LegalOperations ||
3089        TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))) {
3090     // ZEXTLOAD will match without needing to change the size of the value being
3091     // loaded.
3092     NarrowLoad = false;
3093     return true;
3094   }
3095 
3096   // Do not change the width of a volatile load.
3097   if (LoadN->isVolatile())
3098     return false;
3099 
3100   // Do not generate loads of non-round integer types since these can
3101   // be expensive (and would be wrong if the type is not byte sized).
3102   if (!LoadedVT.bitsGT(ExtVT) || !ExtVT.isRound())
3103     return false;
3104 
3105   if (LegalOperations &&
3106       !TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))
3107     return false;
3108 
3109   if (!TLI.shouldReduceLoadWidth(LoadN, ISD::ZEXTLOAD, ExtVT))
3110     return false;
3111 
3112   NarrowLoad = true;
3113   return true;
3114 }
3115 
3116 SDValue DAGCombiner::visitAND(SDNode *N) {
3117   SDValue N0 = N->getOperand(0);
3118   SDValue N1 = N->getOperand(1);
3119   EVT VT = N1.getValueType();
3120 
3121   // fold vector ops
3122   if (VT.isVector()) {
3123     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3124       return FoldedVOp;
3125 
3126     // fold (and x, 0) -> 0, vector edition
3127     if (ISD::isBuildVectorAllZeros(N0.getNode()))
3128       // do not return N0, because undef node may exist in N0
3129       return DAG.getConstant(APInt::getNullValue(N0.getScalarValueSizeInBits()),
3130                              SDLoc(N), N0.getValueType());
3131     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3132       // do not return N1, because undef node may exist in N1
3133       return DAG.getConstant(APInt::getNullValue(N1.getScalarValueSizeInBits()),
3134                              SDLoc(N), N1.getValueType());
3135 
3136     // fold (and x, -1) -> x, vector edition
3137     if (ISD::isBuildVectorAllOnes(N0.getNode()))
3138       return N1;
3139     if (ISD::isBuildVectorAllOnes(N1.getNode()))
3140       return N0;
3141   }
3142 
3143   // fold (and c1, c2) -> c1&c2
3144   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
3145   ConstantSDNode *N1C = isConstOrConstSplat(N1);
3146   if (N0C && N1C && !N1C->isOpaque())
3147     return DAG.FoldConstantArithmetic(ISD::AND, SDLoc(N), VT, N0C, N1C);
3148   // canonicalize constant to RHS
3149   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3150      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
3151     return DAG.getNode(ISD::AND, SDLoc(N), VT, N1, N0);
3152   // fold (and x, -1) -> x
3153   if (isAllOnesConstant(N1))
3154     return N0;
3155   // if (and x, c) is known to be zero, return 0
3156   unsigned BitWidth = VT.getScalarSizeInBits();
3157   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
3158                                    APInt::getAllOnesValue(BitWidth)))
3159     return DAG.getConstant(0, SDLoc(N), VT);
3160   // reassociate and
3161   if (SDValue RAND = ReassociateOps(ISD::AND, SDLoc(N), N0, N1))
3162     return RAND;
3163   // fold (and (or x, C), D) -> D if (C & D) == D
3164   if (N1C && N0.getOpcode() == ISD::OR)
3165     if (ConstantSDNode *ORI = isConstOrConstSplat(N0.getOperand(1)))
3166       if ((ORI->getAPIntValue() & N1C->getAPIntValue()) == N1C->getAPIntValue())
3167         return N1;
3168   // fold (and (any_ext V), c) -> (zero_ext V) if 'and' only clears top bits.
3169   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
3170     SDValue N0Op0 = N0.getOperand(0);
3171     APInt Mask = ~N1C->getAPIntValue();
3172     Mask = Mask.trunc(N0Op0.getScalarValueSizeInBits());
3173     if (DAG.MaskedValueIsZero(N0Op0, Mask)) {
3174       SDValue Zext = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N),
3175                                  N0.getValueType(), N0Op0);
3176 
3177       // Replace uses of the AND with uses of the Zero extend node.
3178       CombineTo(N, Zext);
3179 
3180       // We actually want to replace all uses of the any_extend with the
3181       // zero_extend, to avoid duplicating things.  This will later cause this
3182       // AND to be folded.
3183       CombineTo(N0.getNode(), Zext);
3184       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3185     }
3186   }
3187   // similarly fold (and (X (load ([non_ext|any_ext|zero_ext] V))), c) ->
3188   // (X (load ([non_ext|zero_ext] V))) if 'and' only clears top bits which must
3189   // already be zero by virtue of the width of the base type of the load.
3190   //
3191   // the 'X' node here can either be nothing or an extract_vector_elt to catch
3192   // more cases.
3193   if ((N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
3194        N0.getValueSizeInBits() == N0.getOperand(0).getScalarValueSizeInBits() &&
3195        N0.getOperand(0).getOpcode() == ISD::LOAD &&
3196        N0.getOperand(0).getResNo() == 0) ||
3197       (N0.getOpcode() == ISD::LOAD && N0.getResNo() == 0)) {
3198     LoadSDNode *Load = cast<LoadSDNode>( (N0.getOpcode() == ISD::LOAD) ?
3199                                          N0 : N0.getOperand(0) );
3200 
3201     // Get the constant (if applicable) the zero'th operand is being ANDed with.
3202     // This can be a pure constant or a vector splat, in which case we treat the
3203     // vector as a scalar and use the splat value.
3204     APInt Constant = APInt::getNullValue(1);
3205     if (const ConstantSDNode *C = dyn_cast<ConstantSDNode>(N1)) {
3206       Constant = C->getAPIntValue();
3207     } else if (BuildVectorSDNode *Vector = dyn_cast<BuildVectorSDNode>(N1)) {
3208       APInt SplatValue, SplatUndef;
3209       unsigned SplatBitSize;
3210       bool HasAnyUndefs;
3211       bool IsSplat = Vector->isConstantSplat(SplatValue, SplatUndef,
3212                                              SplatBitSize, HasAnyUndefs);
3213       if (IsSplat) {
3214         // Undef bits can contribute to a possible optimisation if set, so
3215         // set them.
3216         SplatValue |= SplatUndef;
3217 
3218         // The splat value may be something like "0x00FFFFFF", which means 0 for
3219         // the first vector value and FF for the rest, repeating. We need a mask
3220         // that will apply equally to all members of the vector, so AND all the
3221         // lanes of the constant together.
3222         EVT VT = Vector->getValueType(0);
3223         unsigned BitWidth = VT.getScalarSizeInBits();
3224 
3225         // If the splat value has been compressed to a bitlength lower
3226         // than the size of the vector lane, we need to re-expand it to
3227         // the lane size.
3228         if (BitWidth > SplatBitSize)
3229           for (SplatValue = SplatValue.zextOrTrunc(BitWidth);
3230                SplatBitSize < BitWidth;
3231                SplatBitSize = SplatBitSize * 2)
3232             SplatValue |= SplatValue.shl(SplatBitSize);
3233 
3234         // Make sure that variable 'Constant' is only set if 'SplatBitSize' is a
3235         // multiple of 'BitWidth'. Otherwise, we could propagate a wrong value.
3236         if (SplatBitSize % BitWidth == 0) {
3237           Constant = APInt::getAllOnesValue(BitWidth);
3238           for (unsigned i = 0, n = SplatBitSize/BitWidth; i < n; ++i)
3239             Constant &= SplatValue.lshr(i*BitWidth).zextOrTrunc(BitWidth);
3240         }
3241       }
3242     }
3243 
3244     // If we want to change an EXTLOAD to a ZEXTLOAD, ensure a ZEXTLOAD is
3245     // actually legal and isn't going to get expanded, else this is a false
3246     // optimisation.
3247     bool CanZextLoadProfitably = TLI.isLoadExtLegal(ISD::ZEXTLOAD,
3248                                                     Load->getValueType(0),
3249                                                     Load->getMemoryVT());
3250 
3251     // Resize the constant to the same size as the original memory access before
3252     // extension. If it is still the AllOnesValue then this AND is completely
3253     // unneeded.
3254     Constant = Constant.zextOrTrunc(Load->getMemoryVT().getScalarSizeInBits());
3255 
3256     bool B;
3257     switch (Load->getExtensionType()) {
3258     default: B = false; break;
3259     case ISD::EXTLOAD: B = CanZextLoadProfitably; break;
3260     case ISD::ZEXTLOAD:
3261     case ISD::NON_EXTLOAD: B = true; break;
3262     }
3263 
3264     if (B && Constant.isAllOnesValue()) {
3265       // If the load type was an EXTLOAD, convert to ZEXTLOAD in order to
3266       // preserve semantics once we get rid of the AND.
3267       SDValue NewLoad(Load, 0);
3268       if (Load->getExtensionType() == ISD::EXTLOAD) {
3269         NewLoad = DAG.getLoad(Load->getAddressingMode(), ISD::ZEXTLOAD,
3270                               Load->getValueType(0), SDLoc(Load),
3271                               Load->getChain(), Load->getBasePtr(),
3272                               Load->getOffset(), Load->getMemoryVT(),
3273                               Load->getMemOperand());
3274         // Replace uses of the EXTLOAD with the new ZEXTLOAD.
3275         if (Load->getNumValues() == 3) {
3276           // PRE/POST_INC loads have 3 values.
3277           SDValue To[] = { NewLoad.getValue(0), NewLoad.getValue(1),
3278                            NewLoad.getValue(2) };
3279           CombineTo(Load, To, 3, true);
3280         } else {
3281           CombineTo(Load, NewLoad.getValue(0), NewLoad.getValue(1));
3282         }
3283       }
3284 
3285       // Fold the AND away, taking care not to fold to the old load node if we
3286       // replaced it.
3287       CombineTo(N, (N0.getNode() == Load) ? NewLoad : N0);
3288 
3289       return SDValue(N, 0); // Return N so it doesn't get rechecked!
3290     }
3291   }
3292 
3293   // fold (and (load x), 255) -> (zextload x, i8)
3294   // fold (and (extload x, i16), 255) -> (zextload x, i8)
3295   // fold (and (any_ext (extload x, i16)), 255) -> (zextload x, i8)
3296   if (!VT.isVector() && N1C && (N0.getOpcode() == ISD::LOAD ||
3297                                 (N0.getOpcode() == ISD::ANY_EXTEND &&
3298                                  N0.getOperand(0).getOpcode() == ISD::LOAD))) {
3299     bool HasAnyExt = N0.getOpcode() == ISD::ANY_EXTEND;
3300     LoadSDNode *LN0 = HasAnyExt
3301       ? cast<LoadSDNode>(N0.getOperand(0))
3302       : cast<LoadSDNode>(N0);
3303     if (LN0->getExtensionType() != ISD::SEXTLOAD &&
3304         LN0->isUnindexed() && N0.hasOneUse() && SDValue(LN0, 0).hasOneUse()) {
3305       auto NarrowLoad = false;
3306       EVT LoadResultTy = HasAnyExt ? LN0->getValueType(0) : VT;
3307       EVT ExtVT, LoadedVT;
3308       if (isAndLoadExtLoad(N1C, LN0, LoadResultTy, ExtVT, LoadedVT,
3309                            NarrowLoad)) {
3310         if (!NarrowLoad) {
3311           SDValue NewLoad =
3312             DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN0), LoadResultTy,
3313                            LN0->getChain(), LN0->getBasePtr(), ExtVT,
3314                            LN0->getMemOperand());
3315           AddToWorklist(N);
3316           CombineTo(LN0, NewLoad, NewLoad.getValue(1));
3317           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3318         } else {
3319           EVT PtrType = LN0->getOperand(1).getValueType();
3320 
3321           unsigned Alignment = LN0->getAlignment();
3322           SDValue NewPtr = LN0->getBasePtr();
3323 
3324           // For big endian targets, we need to add an offset to the pointer
3325           // to load the correct bytes.  For little endian systems, we merely
3326           // need to read fewer bytes from the same pointer.
3327           if (DAG.getDataLayout().isBigEndian()) {
3328             unsigned LVTStoreBytes = LoadedVT.getStoreSize();
3329             unsigned EVTStoreBytes = ExtVT.getStoreSize();
3330             unsigned PtrOff = LVTStoreBytes - EVTStoreBytes;
3331             SDLoc DL(LN0);
3332             NewPtr = DAG.getNode(ISD::ADD, DL, PtrType,
3333                                  NewPtr, DAG.getConstant(PtrOff, DL, PtrType));
3334             Alignment = MinAlign(Alignment, PtrOff);
3335           }
3336 
3337           AddToWorklist(NewPtr.getNode());
3338 
3339           SDValue Load = DAG.getExtLoad(
3340               ISD::ZEXTLOAD, SDLoc(LN0), LoadResultTy, LN0->getChain(), NewPtr,
3341               LN0->getPointerInfo(), ExtVT, Alignment,
3342               LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
3343           AddToWorklist(N);
3344           CombineTo(LN0, Load, Load.getValue(1));
3345           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3346         }
3347       }
3348     }
3349   }
3350 
3351   if (SDValue Combined = visitANDLike(N0, N1, N))
3352     return Combined;
3353 
3354   // Simplify: (and (op x...), (op y...))  -> (op (and x, y))
3355   if (N0.getOpcode() == N1.getOpcode())
3356     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
3357       return Tmp;
3358 
3359   // Masking the negated extension of a boolean is just the zero-extended
3360   // boolean:
3361   // and (sub 0, zext(bool X)), 1 --> zext(bool X)
3362   // and (sub 0, sext(bool X)), 1 --> zext(bool X)
3363   //
3364   // Note: the SimplifyDemandedBits fold below can make an information-losing
3365   // transform, and then we have no way to find this better fold.
3366   if (N1C && N1C->isOne() && N0.getOpcode() == ISD::SUB) {
3367     ConstantSDNode *SubLHS = isConstOrConstSplat(N0.getOperand(0));
3368     SDValue SubRHS = N0.getOperand(1);
3369     if (SubLHS && SubLHS->isNullValue()) {
3370       if (SubRHS.getOpcode() == ISD::ZERO_EXTEND &&
3371           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
3372         return SubRHS;
3373       if (SubRHS.getOpcode() == ISD::SIGN_EXTEND &&
3374           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
3375         return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, SubRHS.getOperand(0));
3376     }
3377   }
3378 
3379   // fold (and (sign_extend_inreg x, i16 to i32), 1) -> (and x, 1)
3380   // fold (and (sra)) -> (and (srl)) when possible.
3381   if (!VT.isVector() && SimplifyDemandedBits(SDValue(N, 0)))
3382     return SDValue(N, 0);
3383 
3384   // fold (zext_inreg (extload x)) -> (zextload x)
3385   if (ISD::isEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode())) {
3386     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
3387     EVT MemVT = LN0->getMemoryVT();
3388     // If we zero all the possible extended bits, then we can turn this into
3389     // a zextload if we are running before legalize or the operation is legal.
3390     unsigned BitWidth = N1.getScalarValueSizeInBits();
3391     if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth,
3392                            BitWidth - MemVT.getScalarSizeInBits())) &&
3393         ((!LegalOperations && !LN0->isVolatile()) ||
3394          TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) {
3395       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT,
3396                                        LN0->getChain(), LN0->getBasePtr(),
3397                                        MemVT, LN0->getMemOperand());
3398       AddToWorklist(N);
3399       CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
3400       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3401     }
3402   }
3403   // fold (zext_inreg (sextload x)) -> (zextload x) iff load has one use
3404   if (ISD::isSEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
3405       N0.hasOneUse()) {
3406     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
3407     EVT MemVT = LN0->getMemoryVT();
3408     // If we zero all the possible extended bits, then we can turn this into
3409     // a zextload if we are running before legalize or the operation is legal.
3410     unsigned BitWidth = N1.getScalarValueSizeInBits();
3411     if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth,
3412                            BitWidth - MemVT.getScalarSizeInBits())) &&
3413         ((!LegalOperations && !LN0->isVolatile()) ||
3414          TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) {
3415       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT,
3416                                        LN0->getChain(), LN0->getBasePtr(),
3417                                        MemVT, LN0->getMemOperand());
3418       AddToWorklist(N);
3419       CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
3420       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3421     }
3422   }
3423   // fold (and (or (srl N, 8), (shl N, 8)), 0xffff) -> (srl (bswap N), const)
3424   if (N1C && N1C->getAPIntValue() == 0xffff && N0.getOpcode() == ISD::OR) {
3425     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
3426                                            N0.getOperand(1), false))
3427       return BSwap;
3428   }
3429 
3430   return SDValue();
3431 }
3432 
3433 /// Match (a >> 8) | (a << 8) as (bswap a) >> 16.
3434 SDValue DAGCombiner::MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
3435                                         bool DemandHighBits) {
3436   if (!LegalOperations)
3437     return SDValue();
3438 
3439   EVT VT = N->getValueType(0);
3440   if (VT != MVT::i64 && VT != MVT::i32 && VT != MVT::i16)
3441     return SDValue();
3442   if (!TLI.isOperationLegal(ISD::BSWAP, VT))
3443     return SDValue();
3444 
3445   // Recognize (and (shl a, 8), 0xff), (and (srl a, 8), 0xff00)
3446   bool LookPassAnd0 = false;
3447   bool LookPassAnd1 = false;
3448   if (N0.getOpcode() == ISD::AND && N0.getOperand(0).getOpcode() == ISD::SRL)
3449       std::swap(N0, N1);
3450   if (N1.getOpcode() == ISD::AND && N1.getOperand(0).getOpcode() == ISD::SHL)
3451       std::swap(N0, N1);
3452   if (N0.getOpcode() == ISD::AND) {
3453     if (!N0.getNode()->hasOneUse())
3454       return SDValue();
3455     ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3456     if (!N01C || N01C->getZExtValue() != 0xFF00)
3457       return SDValue();
3458     N0 = N0.getOperand(0);
3459     LookPassAnd0 = true;
3460   }
3461 
3462   if (N1.getOpcode() == ISD::AND) {
3463     if (!N1.getNode()->hasOneUse())
3464       return SDValue();
3465     ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
3466     if (!N11C || N11C->getZExtValue() != 0xFF)
3467       return SDValue();
3468     N1 = N1.getOperand(0);
3469     LookPassAnd1 = true;
3470   }
3471 
3472   if (N0.getOpcode() == ISD::SRL && N1.getOpcode() == ISD::SHL)
3473     std::swap(N0, N1);
3474   if (N0.getOpcode() != ISD::SHL || N1.getOpcode() != ISD::SRL)
3475     return SDValue();
3476   if (!N0.getNode()->hasOneUse() || !N1.getNode()->hasOneUse())
3477     return SDValue();
3478 
3479   ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3480   ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
3481   if (!N01C || !N11C)
3482     return SDValue();
3483   if (N01C->getZExtValue() != 8 || N11C->getZExtValue() != 8)
3484     return SDValue();
3485 
3486   // Look for (shl (and a, 0xff), 8), (srl (and a, 0xff00), 8)
3487   SDValue N00 = N0->getOperand(0);
3488   if (!LookPassAnd0 && N00.getOpcode() == ISD::AND) {
3489     if (!N00.getNode()->hasOneUse())
3490       return SDValue();
3491     ConstantSDNode *N001C = dyn_cast<ConstantSDNode>(N00.getOperand(1));
3492     if (!N001C || N001C->getZExtValue() != 0xFF)
3493       return SDValue();
3494     N00 = N00.getOperand(0);
3495     LookPassAnd0 = true;
3496   }
3497 
3498   SDValue N10 = N1->getOperand(0);
3499   if (!LookPassAnd1 && N10.getOpcode() == ISD::AND) {
3500     if (!N10.getNode()->hasOneUse())
3501       return SDValue();
3502     ConstantSDNode *N101C = dyn_cast<ConstantSDNode>(N10.getOperand(1));
3503     if (!N101C || N101C->getZExtValue() != 0xFF00)
3504       return SDValue();
3505     N10 = N10.getOperand(0);
3506     LookPassAnd1 = true;
3507   }
3508 
3509   if (N00 != N10)
3510     return SDValue();
3511 
3512   // Make sure everything beyond the low halfword gets set to zero since the SRL
3513   // 16 will clear the top bits.
3514   unsigned OpSizeInBits = VT.getSizeInBits();
3515   if (DemandHighBits && OpSizeInBits > 16) {
3516     // If the left-shift isn't masked out then the only way this is a bswap is
3517     // if all bits beyond the low 8 are 0. In that case the entire pattern
3518     // reduces to a left shift anyway: leave it for other parts of the combiner.
3519     if (!LookPassAnd0)
3520       return SDValue();
3521 
3522     // However, if the right shift isn't masked out then it might be because
3523     // it's not needed. See if we can spot that too.
3524     if (!LookPassAnd1 &&
3525         !DAG.MaskedValueIsZero(
3526             N10, APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - 16)))
3527       return SDValue();
3528   }
3529 
3530   SDValue Res = DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N00);
3531   if (OpSizeInBits > 16) {
3532     SDLoc DL(N);
3533     Res = DAG.getNode(ISD::SRL, DL, VT, Res,
3534                       DAG.getConstant(OpSizeInBits - 16, DL,
3535                                       getShiftAmountTy(VT)));
3536   }
3537   return Res;
3538 }
3539 
3540 /// Return true if the specified node is an element that makes up a 32-bit
3541 /// packed halfword byteswap.
3542 /// ((x & 0x000000ff) << 8) |
3543 /// ((x & 0x0000ff00) >> 8) |
3544 /// ((x & 0x00ff0000) << 8) |
3545 /// ((x & 0xff000000) >> 8)
3546 static bool isBSwapHWordElement(SDValue N, MutableArrayRef<SDNode *> Parts) {
3547   if (!N.getNode()->hasOneUse())
3548     return false;
3549 
3550   unsigned Opc = N.getOpcode();
3551   if (Opc != ISD::AND && Opc != ISD::SHL && Opc != ISD::SRL)
3552     return false;
3553 
3554   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3555   if (!N1C)
3556     return false;
3557 
3558   unsigned Num;
3559   switch (N1C->getZExtValue()) {
3560   default:
3561     return false;
3562   case 0xFF:       Num = 0; break;
3563   case 0xFF00:     Num = 1; break;
3564   case 0xFF0000:   Num = 2; break;
3565   case 0xFF000000: Num = 3; break;
3566   }
3567 
3568   // Look for (x & 0xff) << 8 as well as ((x << 8) & 0xff00).
3569   SDValue N0 = N.getOperand(0);
3570   if (Opc == ISD::AND) {
3571     if (Num == 0 || Num == 2) {
3572       // (x >> 8) & 0xff
3573       // (x >> 8) & 0xff0000
3574       if (N0.getOpcode() != ISD::SRL)
3575         return false;
3576       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3577       if (!C || C->getZExtValue() != 8)
3578         return false;
3579     } else {
3580       // (x << 8) & 0xff00
3581       // (x << 8) & 0xff000000
3582       if (N0.getOpcode() != ISD::SHL)
3583         return false;
3584       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3585       if (!C || C->getZExtValue() != 8)
3586         return false;
3587     }
3588   } else if (Opc == ISD::SHL) {
3589     // (x & 0xff) << 8
3590     // (x & 0xff0000) << 8
3591     if (Num != 0 && Num != 2)
3592       return false;
3593     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3594     if (!C || C->getZExtValue() != 8)
3595       return false;
3596   } else { // Opc == ISD::SRL
3597     // (x & 0xff00) >> 8
3598     // (x & 0xff000000) >> 8
3599     if (Num != 1 && Num != 3)
3600       return false;
3601     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3602     if (!C || C->getZExtValue() != 8)
3603       return false;
3604   }
3605 
3606   if (Parts[Num])
3607     return false;
3608 
3609   Parts[Num] = N0.getOperand(0).getNode();
3610   return true;
3611 }
3612 
3613 /// Match a 32-bit packed halfword bswap. That is
3614 /// ((x & 0x000000ff) << 8) |
3615 /// ((x & 0x0000ff00) >> 8) |
3616 /// ((x & 0x00ff0000) << 8) |
3617 /// ((x & 0xff000000) >> 8)
3618 /// => (rotl (bswap x), 16)
3619 SDValue DAGCombiner::MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1) {
3620   if (!LegalOperations)
3621     return SDValue();
3622 
3623   EVT VT = N->getValueType(0);
3624   if (VT != MVT::i32)
3625     return SDValue();
3626   if (!TLI.isOperationLegal(ISD::BSWAP, VT))
3627     return SDValue();
3628 
3629   // Look for either
3630   // (or (or (and), (and)), (or (and), (and)))
3631   // (or (or (or (and), (and)), (and)), (and))
3632   if (N0.getOpcode() != ISD::OR)
3633     return SDValue();
3634   SDValue N00 = N0.getOperand(0);
3635   SDValue N01 = N0.getOperand(1);
3636   SDNode *Parts[4] = {};
3637 
3638   if (N1.getOpcode() == ISD::OR &&
3639       N00.getNumOperands() == 2 && N01.getNumOperands() == 2) {
3640     // (or (or (and), (and)), (or (and), (and)))
3641     SDValue N000 = N00.getOperand(0);
3642     if (!isBSwapHWordElement(N000, Parts))
3643       return SDValue();
3644 
3645     SDValue N001 = N00.getOperand(1);
3646     if (!isBSwapHWordElement(N001, Parts))
3647       return SDValue();
3648     SDValue N010 = N01.getOperand(0);
3649     if (!isBSwapHWordElement(N010, Parts))
3650       return SDValue();
3651     SDValue N011 = N01.getOperand(1);
3652     if (!isBSwapHWordElement(N011, Parts))
3653       return SDValue();
3654   } else {
3655     // (or (or (or (and), (and)), (and)), (and))
3656     if (!isBSwapHWordElement(N1, Parts))
3657       return SDValue();
3658     if (!isBSwapHWordElement(N01, Parts))
3659       return SDValue();
3660     if (N00.getOpcode() != ISD::OR)
3661       return SDValue();
3662     SDValue N000 = N00.getOperand(0);
3663     if (!isBSwapHWordElement(N000, Parts))
3664       return SDValue();
3665     SDValue N001 = N00.getOperand(1);
3666     if (!isBSwapHWordElement(N001, Parts))
3667       return SDValue();
3668   }
3669 
3670   // Make sure the parts are all coming from the same node.
3671   if (Parts[0] != Parts[1] || Parts[0] != Parts[2] || Parts[0] != Parts[3])
3672     return SDValue();
3673 
3674   SDLoc DL(N);
3675   SDValue BSwap = DAG.getNode(ISD::BSWAP, DL, VT,
3676                               SDValue(Parts[0], 0));
3677 
3678   // Result of the bswap should be rotated by 16. If it's not legal, then
3679   // do  (x << 16) | (x >> 16).
3680   SDValue ShAmt = DAG.getConstant(16, DL, getShiftAmountTy(VT));
3681   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT))
3682     return DAG.getNode(ISD::ROTL, DL, VT, BSwap, ShAmt);
3683   if (TLI.isOperationLegalOrCustom(ISD::ROTR, VT))
3684     return DAG.getNode(ISD::ROTR, DL, VT, BSwap, ShAmt);
3685   return DAG.getNode(ISD::OR, DL, VT,
3686                      DAG.getNode(ISD::SHL, DL, VT, BSwap, ShAmt),
3687                      DAG.getNode(ISD::SRL, DL, VT, BSwap, ShAmt));
3688 }
3689 
3690 /// This contains all DAGCombine rules which reduce two values combined by
3691 /// an Or operation to a single value \see visitANDLike().
3692 SDValue DAGCombiner::visitORLike(SDValue N0, SDValue N1, SDNode *LocReference) {
3693   EVT VT = N1.getValueType();
3694   // fold (or x, undef) -> -1
3695   if (!LegalOperations &&
3696       (N0.isUndef() || N1.isUndef())) {
3697     EVT EltVT = VT.isVector() ? VT.getVectorElementType() : VT;
3698     return DAG.getConstant(APInt::getAllOnesValue(EltVT.getSizeInBits()),
3699                            SDLoc(LocReference), VT);
3700   }
3701   // fold (or (setcc x), (setcc y)) -> (setcc (or x, y))
3702   SDValue LL, LR, RL, RR, CC0, CC1;
3703   if (isSetCCEquivalent(N0, LL, LR, CC0) && isSetCCEquivalent(N1, RL, RR, CC1)){
3704     ISD::CondCode Op0 = cast<CondCodeSDNode>(CC0)->get();
3705     ISD::CondCode Op1 = cast<CondCodeSDNode>(CC1)->get();
3706 
3707     if (LR == RR && Op0 == Op1 && LL.getValueType().isInteger()) {
3708       // fold (or (setne X, 0), (setne Y, 0)) -> (setne (or X, Y), 0)
3709       // fold (or (setlt X, 0), (setlt Y, 0)) -> (setne (or X, Y), 0)
3710       if (isNullConstant(LR) && (Op1 == ISD::SETNE || Op1 == ISD::SETLT)) {
3711         EVT CCVT = getSetCCResultType(LR.getValueType());
3712         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
3713           SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(LR),
3714                                        LR.getValueType(), LL, RL);
3715           AddToWorklist(ORNode.getNode());
3716           return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
3717         }
3718       }
3719       // fold (or (setne X, -1), (setne Y, -1)) -> (setne (and X, Y), -1)
3720       // fold (or (setgt X, -1), (setgt Y  -1)) -> (setgt (and X, Y), -1)
3721       if (isAllOnesConstant(LR) && (Op1 == ISD::SETNE || Op1 == ISD::SETGT)) {
3722         EVT CCVT = getSetCCResultType(LR.getValueType());
3723         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
3724           SDValue ANDNode = DAG.getNode(ISD::AND, SDLoc(LR),
3725                                         LR.getValueType(), LL, RL);
3726           AddToWorklist(ANDNode.getNode());
3727           return DAG.getSetCC(SDLoc(LocReference), VT, ANDNode, LR, Op1);
3728         }
3729       }
3730     }
3731     // canonicalize equivalent to ll == rl
3732     if (LL == RR && LR == RL) {
3733       Op1 = ISD::getSetCCSwappedOperands(Op1);
3734       std::swap(RL, RR);
3735     }
3736     if (LL == RL && LR == RR) {
3737       bool isInteger = LL.getValueType().isInteger();
3738       ISD::CondCode Result = ISD::getSetCCOrOperation(Op0, Op1, isInteger);
3739       if (Result != ISD::SETCC_INVALID &&
3740           (!LegalOperations ||
3741            (TLI.isCondCodeLegal(Result, LL.getSimpleValueType()) &&
3742             TLI.isOperationLegal(ISD::SETCC, LL.getValueType())))) {
3743         EVT CCVT = getSetCCResultType(LL.getValueType());
3744         if (N0.getValueType() == CCVT ||
3745             (!LegalOperations && N0.getValueType() == MVT::i1))
3746           return DAG.getSetCC(SDLoc(LocReference), N0.getValueType(),
3747                               LL, LR, Result);
3748       }
3749     }
3750   }
3751 
3752   // (or (and X, C1), (and Y, C2))  -> (and (or X, Y), C3) if possible.
3753   if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND &&
3754       // Don't increase # computations.
3755       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
3756     // We can only do this xform if we know that bits from X that are set in C2
3757     // but not in C1 are already zero.  Likewise for Y.
3758     if (const ConstantSDNode *N0O1C =
3759         getAsNonOpaqueConstant(N0.getOperand(1))) {
3760       if (const ConstantSDNode *N1O1C =
3761           getAsNonOpaqueConstant(N1.getOperand(1))) {
3762         // We can only do this xform if we know that bits from X that are set in
3763         // C2 but not in C1 are already zero.  Likewise for Y.
3764         const APInt &LHSMask = N0O1C->getAPIntValue();
3765         const APInt &RHSMask = N1O1C->getAPIntValue();
3766 
3767         if (DAG.MaskedValueIsZero(N0.getOperand(0), RHSMask&~LHSMask) &&
3768             DAG.MaskedValueIsZero(N1.getOperand(0), LHSMask&~RHSMask)) {
3769           SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
3770                                   N0.getOperand(0), N1.getOperand(0));
3771           SDLoc DL(LocReference);
3772           return DAG.getNode(ISD::AND, DL, VT, X,
3773                              DAG.getConstant(LHSMask | RHSMask, DL, VT));
3774         }
3775       }
3776     }
3777   }
3778 
3779   // (or (and X, M), (and X, N)) -> (and X, (or M, N))
3780   if (N0.getOpcode() == ISD::AND &&
3781       N1.getOpcode() == ISD::AND &&
3782       N0.getOperand(0) == N1.getOperand(0) &&
3783       // Don't increase # computations.
3784       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
3785     SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
3786                             N0.getOperand(1), N1.getOperand(1));
3787     return DAG.getNode(ISD::AND, SDLoc(LocReference), VT, N0.getOperand(0), X);
3788   }
3789 
3790   return SDValue();
3791 }
3792 
3793 SDValue DAGCombiner::visitOR(SDNode *N) {
3794   SDValue N0 = N->getOperand(0);
3795   SDValue N1 = N->getOperand(1);
3796   EVT VT = N1.getValueType();
3797 
3798   // fold vector ops
3799   if (VT.isVector()) {
3800     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3801       return FoldedVOp;
3802 
3803     // fold (or x, 0) -> x, vector edition
3804     if (ISD::isBuildVectorAllZeros(N0.getNode()))
3805       return N1;
3806     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3807       return N0;
3808 
3809     // fold (or x, -1) -> -1, vector edition
3810     if (ISD::isBuildVectorAllOnes(N0.getNode()))
3811       // do not return N0, because undef node may exist in N0
3812       return DAG.getConstant(
3813           APInt::getAllOnesValue(N0.getScalarValueSizeInBits()), SDLoc(N),
3814           N0.getValueType());
3815     if (ISD::isBuildVectorAllOnes(N1.getNode()))
3816       // do not return N1, because undef node may exist in N1
3817       return DAG.getConstant(
3818           APInt::getAllOnesValue(N1.getScalarValueSizeInBits()), SDLoc(N),
3819           N1.getValueType());
3820 
3821     // fold (or (shuf A, V_0, MA), (shuf B, V_0, MB)) -> (shuf A, B, Mask)
3822     // Do this only if the resulting shuffle is legal.
3823     if (isa<ShuffleVectorSDNode>(N0) &&
3824         isa<ShuffleVectorSDNode>(N1) &&
3825         // Avoid folding a node with illegal type.
3826         TLI.isTypeLegal(VT)) {
3827       bool ZeroN00 = ISD::isBuildVectorAllZeros(N0.getOperand(0).getNode());
3828       bool ZeroN01 = ISD::isBuildVectorAllZeros(N0.getOperand(1).getNode());
3829       bool ZeroN10 = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
3830       bool ZeroN11 = ISD::isBuildVectorAllZeros(N1.getOperand(1).getNode());
3831       // Ensure both shuffles have a zero input.
3832       if ((ZeroN00 || ZeroN01) && (ZeroN10 || ZeroN11)) {
3833         assert((!ZeroN00 || !ZeroN01) && "Both inputs zero!");
3834         assert((!ZeroN10 || !ZeroN11) && "Both inputs zero!");
3835         const ShuffleVectorSDNode *SV0 = cast<ShuffleVectorSDNode>(N0);
3836         const ShuffleVectorSDNode *SV1 = cast<ShuffleVectorSDNode>(N1);
3837         bool CanFold = true;
3838         int NumElts = VT.getVectorNumElements();
3839         SmallVector<int, 4> Mask(NumElts);
3840 
3841         for (int i = 0; i != NumElts; ++i) {
3842           int M0 = SV0->getMaskElt(i);
3843           int M1 = SV1->getMaskElt(i);
3844 
3845           // Determine if either index is pointing to a zero vector.
3846           bool M0Zero = M0 < 0 || (ZeroN00 == (M0 < NumElts));
3847           bool M1Zero = M1 < 0 || (ZeroN10 == (M1 < NumElts));
3848 
3849           // If one element is zero and the otherside is undef, keep undef.
3850           // This also handles the case that both are undef.
3851           if ((M0Zero && M1 < 0) || (M1Zero && M0 < 0)) {
3852             Mask[i] = -1;
3853             continue;
3854           }
3855 
3856           // Make sure only one of the elements is zero.
3857           if (M0Zero == M1Zero) {
3858             CanFold = false;
3859             break;
3860           }
3861 
3862           assert((M0 >= 0 || M1 >= 0) && "Undef index!");
3863 
3864           // We have a zero and non-zero element. If the non-zero came from
3865           // SV0 make the index a LHS index. If it came from SV1, make it
3866           // a RHS index. We need to mod by NumElts because we don't care
3867           // which operand it came from in the original shuffles.
3868           Mask[i] = M1Zero ? M0 % NumElts : (M1 % NumElts) + NumElts;
3869         }
3870 
3871         if (CanFold) {
3872           SDValue NewLHS = ZeroN00 ? N0.getOperand(1) : N0.getOperand(0);
3873           SDValue NewRHS = ZeroN10 ? N1.getOperand(1) : N1.getOperand(0);
3874 
3875           bool LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
3876           if (!LegalMask) {
3877             std::swap(NewLHS, NewRHS);
3878             ShuffleVectorSDNode::commuteMask(Mask);
3879             LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
3880           }
3881 
3882           if (LegalMask)
3883             return DAG.getVectorShuffle(VT, SDLoc(N), NewLHS, NewRHS, Mask);
3884         }
3885       }
3886     }
3887   }
3888 
3889   // fold (or c1, c2) -> c1|c2
3890   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
3891   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
3892   if (N0C && N1C && !N1C->isOpaque())
3893     return DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N), VT, N0C, N1C);
3894   // canonicalize constant to RHS
3895   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3896      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
3897     return DAG.getNode(ISD::OR, SDLoc(N), VT, N1, N0);
3898   // fold (or x, 0) -> x
3899   if (isNullConstant(N1))
3900     return N0;
3901   // fold (or x, -1) -> -1
3902   if (isAllOnesConstant(N1))
3903     return N1;
3904   // fold (or x, c) -> c iff (x & ~c) == 0
3905   if (N1C && DAG.MaskedValueIsZero(N0, ~N1C->getAPIntValue()))
3906     return N1;
3907 
3908   if (SDValue Combined = visitORLike(N0, N1, N))
3909     return Combined;
3910 
3911   // Recognize halfword bswaps as (bswap + rotl 16) or (bswap + shl 16)
3912   if (SDValue BSwap = MatchBSwapHWord(N, N0, N1))
3913     return BSwap;
3914   if (SDValue BSwap = MatchBSwapHWordLow(N, N0, N1))
3915     return BSwap;
3916 
3917   // reassociate or
3918   if (SDValue ROR = ReassociateOps(ISD::OR, SDLoc(N), N0, N1))
3919     return ROR;
3920   // Canonicalize (or (and X, c1), c2) -> (and (or X, c2), c1|c2)
3921   // iff (c1 & c2) == 0.
3922   if (N1C && N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
3923              isa<ConstantSDNode>(N0.getOperand(1))) {
3924     ConstantSDNode *C1 = cast<ConstantSDNode>(N0.getOperand(1));
3925     if ((C1->getAPIntValue() & N1C->getAPIntValue()) != 0) {
3926       if (SDValue COR = DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N1), VT,
3927                                                    N1C, C1))
3928         return DAG.getNode(
3929             ISD::AND, SDLoc(N), VT,
3930             DAG.getNode(ISD::OR, SDLoc(N0), VT, N0.getOperand(0), N1), COR);
3931       return SDValue();
3932     }
3933   }
3934   // Simplify: (or (op x...), (op y...))  -> (op (or x, y))
3935   if (N0.getOpcode() == N1.getOpcode())
3936     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
3937       return Tmp;
3938 
3939   // See if this is some rotate idiom.
3940   if (SDNode *Rot = MatchRotate(N0, N1, SDLoc(N)))
3941     return SDValue(Rot, 0);
3942 
3943   // Simplify the operands using demanded-bits information.
3944   if (!VT.isVector() &&
3945       SimplifyDemandedBits(SDValue(N, 0)))
3946     return SDValue(N, 0);
3947 
3948   return SDValue();
3949 }
3950 
3951 /// Match "(X shl/srl V1) & V2" where V2 may not be present.
3952 bool DAGCombiner::MatchRotateHalf(SDValue Op, SDValue &Shift, SDValue &Mask) {
3953   if (Op.getOpcode() == ISD::AND) {
3954     if (DAG.isConstantIntBuildVectorOrConstantInt(Op.getOperand(1))) {
3955       Mask = Op.getOperand(1);
3956       Op = Op.getOperand(0);
3957     } else {
3958       return false;
3959     }
3960   }
3961 
3962   if (Op.getOpcode() == ISD::SRL || Op.getOpcode() == ISD::SHL) {
3963     Shift = Op;
3964     return true;
3965   }
3966 
3967   return false;
3968 }
3969 
3970 // Return true if we can prove that, whenever Neg and Pos are both in the
3971 // range [0, EltSize), Neg == (Pos == 0 ? 0 : EltSize - Pos).  This means that
3972 // for two opposing shifts shift1 and shift2 and a value X with OpBits bits:
3973 //
3974 //     (or (shift1 X, Neg), (shift2 X, Pos))
3975 //
3976 // reduces to a rotate in direction shift2 by Pos or (equivalently) a rotate
3977 // in direction shift1 by Neg.  The range [0, EltSize) means that we only need
3978 // to consider shift amounts with defined behavior.
3979 static bool matchRotateSub(SDValue Pos, SDValue Neg, unsigned EltSize) {
3980   // If EltSize is a power of 2 then:
3981   //
3982   //  (a) (Pos == 0 ? 0 : EltSize - Pos) == (EltSize - Pos) & (EltSize - 1)
3983   //  (b) Neg == Neg & (EltSize - 1) whenever Neg is in [0, EltSize).
3984   //
3985   // So if EltSize is a power of 2 and Neg is (and Neg', EltSize-1), we check
3986   // for the stronger condition:
3987   //
3988   //     Neg & (EltSize - 1) == (EltSize - Pos) & (EltSize - 1)    [A]
3989   //
3990   // for all Neg and Pos.  Since Neg & (EltSize - 1) == Neg' & (EltSize - 1)
3991   // we can just replace Neg with Neg' for the rest of the function.
3992   //
3993   // In other cases we check for the even stronger condition:
3994   //
3995   //     Neg == EltSize - Pos                                    [B]
3996   //
3997   // for all Neg and Pos.  Note that the (or ...) then invokes undefined
3998   // behavior if Pos == 0 (and consequently Neg == EltSize).
3999   //
4000   // We could actually use [A] whenever EltSize is a power of 2, but the
4001   // only extra cases that it would match are those uninteresting ones
4002   // where Neg and Pos are never in range at the same time.  E.g. for
4003   // EltSize == 32, using [A] would allow a Neg of the form (sub 64, Pos)
4004   // as well as (sub 32, Pos), but:
4005   //
4006   //     (or (shift1 X, (sub 64, Pos)), (shift2 X, Pos))
4007   //
4008   // always invokes undefined behavior for 32-bit X.
4009   //
4010   // Below, Mask == EltSize - 1 when using [A] and is all-ones otherwise.
4011   unsigned MaskLoBits = 0;
4012   if (Neg.getOpcode() == ISD::AND && isPowerOf2_64(EltSize)) {
4013     if (ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(1))) {
4014       if (NegC->getAPIntValue() == EltSize - 1) {
4015         Neg = Neg.getOperand(0);
4016         MaskLoBits = Log2_64(EltSize);
4017       }
4018     }
4019   }
4020 
4021   // Check whether Neg has the form (sub NegC, NegOp1) for some NegC and NegOp1.
4022   if (Neg.getOpcode() != ISD::SUB)
4023     return false;
4024   ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(0));
4025   if (!NegC)
4026     return false;
4027   SDValue NegOp1 = Neg.getOperand(1);
4028 
4029   // On the RHS of [A], if Pos is Pos' & (EltSize - 1), just replace Pos with
4030   // Pos'.  The truncation is redundant for the purpose of the equality.
4031   if (MaskLoBits && Pos.getOpcode() == ISD::AND)
4032     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1)))
4033       if (PosC->getAPIntValue() == EltSize - 1)
4034         Pos = Pos.getOperand(0);
4035 
4036   // The condition we need is now:
4037   //
4038   //     (NegC - NegOp1) & Mask == (EltSize - Pos) & Mask
4039   //
4040   // If NegOp1 == Pos then we need:
4041   //
4042   //              EltSize & Mask == NegC & Mask
4043   //
4044   // (because "x & Mask" is a truncation and distributes through subtraction).
4045   APInt Width;
4046   if (Pos == NegOp1)
4047     Width = NegC->getAPIntValue();
4048 
4049   // Check for cases where Pos has the form (add NegOp1, PosC) for some PosC.
4050   // Then the condition we want to prove becomes:
4051   //
4052   //     (NegC - NegOp1) & Mask == (EltSize - (NegOp1 + PosC)) & Mask
4053   //
4054   // which, again because "x & Mask" is a truncation, becomes:
4055   //
4056   //                NegC & Mask == (EltSize - PosC) & Mask
4057   //             EltSize & Mask == (NegC + PosC) & Mask
4058   else if (Pos.getOpcode() == ISD::ADD && Pos.getOperand(0) == NegOp1) {
4059     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1)))
4060       Width = PosC->getAPIntValue() + NegC->getAPIntValue();
4061     else
4062       return false;
4063   } else
4064     return false;
4065 
4066   // Now we just need to check that EltSize & Mask == Width & Mask.
4067   if (MaskLoBits)
4068     // EltSize & Mask is 0 since Mask is EltSize - 1.
4069     return Width.getLoBits(MaskLoBits) == 0;
4070   return Width == EltSize;
4071 }
4072 
4073 // A subroutine of MatchRotate used once we have found an OR of two opposite
4074 // shifts of Shifted.  If Neg == <operand size> - Pos then the OR reduces
4075 // to both (PosOpcode Shifted, Pos) and (NegOpcode Shifted, Neg), with the
4076 // former being preferred if supported.  InnerPos and InnerNeg are Pos and
4077 // Neg with outer conversions stripped away.
4078 SDNode *DAGCombiner::MatchRotatePosNeg(SDValue Shifted, SDValue Pos,
4079                                        SDValue Neg, SDValue InnerPos,
4080                                        SDValue InnerNeg, unsigned PosOpcode,
4081                                        unsigned NegOpcode, const SDLoc &DL) {
4082   // fold (or (shl x, (*ext y)),
4083   //          (srl x, (*ext (sub 32, y)))) ->
4084   //   (rotl x, y) or (rotr x, (sub 32, y))
4085   //
4086   // fold (or (shl x, (*ext (sub 32, y))),
4087   //          (srl x, (*ext y))) ->
4088   //   (rotr x, y) or (rotl x, (sub 32, y))
4089   EVT VT = Shifted.getValueType();
4090   if (matchRotateSub(InnerPos, InnerNeg, VT.getScalarSizeInBits())) {
4091     bool HasPos = TLI.isOperationLegalOrCustom(PosOpcode, VT);
4092     return DAG.getNode(HasPos ? PosOpcode : NegOpcode, DL, VT, Shifted,
4093                        HasPos ? Pos : Neg).getNode();
4094   }
4095 
4096   return nullptr;
4097 }
4098 
4099 // MatchRotate - Handle an 'or' of two operands.  If this is one of the many
4100 // idioms for rotate, and if the target supports rotation instructions, generate
4101 // a rot[lr].
4102 SDNode *DAGCombiner::MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL) {
4103   // Must be a legal type.  Expanded 'n promoted things won't work with rotates.
4104   EVT VT = LHS.getValueType();
4105   if (!TLI.isTypeLegal(VT)) return nullptr;
4106 
4107   // The target must have at least one rotate flavor.
4108   bool HasROTL = TLI.isOperationLegalOrCustom(ISD::ROTL, VT);
4109   bool HasROTR = TLI.isOperationLegalOrCustom(ISD::ROTR, VT);
4110   if (!HasROTL && !HasROTR) return nullptr;
4111 
4112   // Match "(X shl/srl V1) & V2" where V2 may not be present.
4113   SDValue LHSShift;   // The shift.
4114   SDValue LHSMask;    // AND value if any.
4115   if (!MatchRotateHalf(LHS, LHSShift, LHSMask))
4116     return nullptr; // Not part of a rotate.
4117 
4118   SDValue RHSShift;   // The shift.
4119   SDValue RHSMask;    // AND value if any.
4120   if (!MatchRotateHalf(RHS, RHSShift, RHSMask))
4121     return nullptr; // Not part of a rotate.
4122 
4123   if (LHSShift.getOperand(0) != RHSShift.getOperand(0))
4124     return nullptr;   // Not shifting the same value.
4125 
4126   if (LHSShift.getOpcode() == RHSShift.getOpcode())
4127     return nullptr;   // Shifts must disagree.
4128 
4129   // Canonicalize shl to left side in a shl/srl pair.
4130   if (RHSShift.getOpcode() == ISD::SHL) {
4131     std::swap(LHS, RHS);
4132     std::swap(LHSShift, RHSShift);
4133     std::swap(LHSMask, RHSMask);
4134   }
4135 
4136   unsigned EltSizeInBits = VT.getScalarSizeInBits();
4137   SDValue LHSShiftArg = LHSShift.getOperand(0);
4138   SDValue LHSShiftAmt = LHSShift.getOperand(1);
4139   SDValue RHSShiftArg = RHSShift.getOperand(0);
4140   SDValue RHSShiftAmt = RHSShift.getOperand(1);
4141 
4142   // fold (or (shl x, C1), (srl x, C2)) -> (rotl x, C1)
4143   // fold (or (shl x, C1), (srl x, C2)) -> (rotr x, C2)
4144   if (isConstOrConstSplat(LHSShiftAmt) && isConstOrConstSplat(RHSShiftAmt)) {
4145     uint64_t LShVal = isConstOrConstSplat(LHSShiftAmt)->getZExtValue();
4146     uint64_t RShVal = isConstOrConstSplat(RHSShiftAmt)->getZExtValue();
4147     if ((LShVal + RShVal) != EltSizeInBits)
4148       return nullptr;
4149 
4150     SDValue Rot = DAG.getNode(HasROTL ? ISD::ROTL : ISD::ROTR, DL, VT,
4151                               LHSShiftArg, HasROTL ? LHSShiftAmt : RHSShiftAmt);
4152 
4153     // If there is an AND of either shifted operand, apply it to the result.
4154     if (LHSMask.getNode() || RHSMask.getNode()) {
4155       APInt AllBits = APInt::getAllOnesValue(EltSizeInBits);
4156       SDValue Mask = DAG.getConstant(AllBits, DL, VT);
4157 
4158       if (LHSMask.getNode()) {
4159         APInt RHSBits = APInt::getLowBitsSet(EltSizeInBits, LShVal);
4160         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
4161                            DAG.getNode(ISD::OR, DL, VT, LHSMask,
4162                                        DAG.getConstant(RHSBits, DL, VT)));
4163       }
4164       if (RHSMask.getNode()) {
4165         APInt LHSBits = APInt::getHighBitsSet(EltSizeInBits, RShVal);
4166         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
4167                            DAG.getNode(ISD::OR, DL, VT, RHSMask,
4168                                        DAG.getConstant(LHSBits, DL, VT)));
4169       }
4170 
4171       Rot = DAG.getNode(ISD::AND, DL, VT, Rot, Mask);
4172     }
4173 
4174     return Rot.getNode();
4175   }
4176 
4177   // If there is a mask here, and we have a variable shift, we can't be sure
4178   // that we're masking out the right stuff.
4179   if (LHSMask.getNode() || RHSMask.getNode())
4180     return nullptr;
4181 
4182   // If the shift amount is sign/zext/any-extended just peel it off.
4183   SDValue LExtOp0 = LHSShiftAmt;
4184   SDValue RExtOp0 = RHSShiftAmt;
4185   if ((LHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
4186        LHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
4187        LHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
4188        LHSShiftAmt.getOpcode() == ISD::TRUNCATE) &&
4189       (RHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
4190        RHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
4191        RHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
4192        RHSShiftAmt.getOpcode() == ISD::TRUNCATE)) {
4193     LExtOp0 = LHSShiftAmt.getOperand(0);
4194     RExtOp0 = RHSShiftAmt.getOperand(0);
4195   }
4196 
4197   SDNode *TryL = MatchRotatePosNeg(LHSShiftArg, LHSShiftAmt, RHSShiftAmt,
4198                                    LExtOp0, RExtOp0, ISD::ROTL, ISD::ROTR, DL);
4199   if (TryL)
4200     return TryL;
4201 
4202   SDNode *TryR = MatchRotatePosNeg(RHSShiftArg, RHSShiftAmt, LHSShiftAmt,
4203                                    RExtOp0, LExtOp0, ISD::ROTR, ISD::ROTL, DL);
4204   if (TryR)
4205     return TryR;
4206 
4207   return nullptr;
4208 }
4209 
4210 SDValue DAGCombiner::visitXOR(SDNode *N) {
4211   SDValue N0 = N->getOperand(0);
4212   SDValue N1 = N->getOperand(1);
4213   EVT VT = N0.getValueType();
4214 
4215   // fold vector ops
4216   if (VT.isVector()) {
4217     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4218       return FoldedVOp;
4219 
4220     // fold (xor x, 0) -> x, vector edition
4221     if (ISD::isBuildVectorAllZeros(N0.getNode()))
4222       return N1;
4223     if (ISD::isBuildVectorAllZeros(N1.getNode()))
4224       return N0;
4225   }
4226 
4227   // fold (xor undef, undef) -> 0. This is a common idiom (misuse).
4228   if (N0.isUndef() && N1.isUndef())
4229     return DAG.getConstant(0, SDLoc(N), VT);
4230   // fold (xor x, undef) -> undef
4231   if (N0.isUndef())
4232     return N0;
4233   if (N1.isUndef())
4234     return N1;
4235   // fold (xor c1, c2) -> c1^c2
4236   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4237   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
4238   if (N0C && N1C)
4239     return DAG.FoldConstantArithmetic(ISD::XOR, SDLoc(N), VT, N0C, N1C);
4240   // canonicalize constant to RHS
4241   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
4242      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
4243     return DAG.getNode(ISD::XOR, SDLoc(N), VT, N1, N0);
4244   // fold (xor x, 0) -> x
4245   if (isNullConstant(N1))
4246     return N0;
4247   // reassociate xor
4248   if (SDValue RXOR = ReassociateOps(ISD::XOR, SDLoc(N), N0, N1))
4249     return RXOR;
4250 
4251   // fold !(x cc y) -> (x !cc y)
4252   SDValue LHS, RHS, CC;
4253   if (TLI.isConstTrueVal(N1.getNode()) && isSetCCEquivalent(N0, LHS, RHS, CC)) {
4254     bool isInt = LHS.getValueType().isInteger();
4255     ISD::CondCode NotCC = ISD::getSetCCInverse(cast<CondCodeSDNode>(CC)->get(),
4256                                                isInt);
4257 
4258     if (!LegalOperations ||
4259         TLI.isCondCodeLegal(NotCC, LHS.getSimpleValueType())) {
4260       switch (N0.getOpcode()) {
4261       default:
4262         llvm_unreachable("Unhandled SetCC Equivalent!");
4263       case ISD::SETCC:
4264         return DAG.getSetCC(SDLoc(N), VT, LHS, RHS, NotCC);
4265       case ISD::SELECT_CC:
4266         return DAG.getSelectCC(SDLoc(N), LHS, RHS, N0.getOperand(2),
4267                                N0.getOperand(3), NotCC);
4268       }
4269     }
4270   }
4271 
4272   // fold (not (zext (setcc x, y))) -> (zext (not (setcc x, y)))
4273   if (isOneConstant(N1) && N0.getOpcode() == ISD::ZERO_EXTEND &&
4274       N0.getNode()->hasOneUse() &&
4275       isSetCCEquivalent(N0.getOperand(0), LHS, RHS, CC)){
4276     SDValue V = N0.getOperand(0);
4277     SDLoc DL(N0);
4278     V = DAG.getNode(ISD::XOR, DL, V.getValueType(), V,
4279                     DAG.getConstant(1, DL, V.getValueType()));
4280     AddToWorklist(V.getNode());
4281     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, V);
4282   }
4283 
4284   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are setcc
4285   if (isOneConstant(N1) && VT == MVT::i1 &&
4286       (N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::AND)) {
4287     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
4288     if (isOneUseSetCC(RHS) || isOneUseSetCC(LHS)) {
4289       unsigned NewOpcode = N0.getOpcode() == ISD::AND ? ISD::OR : ISD::AND;
4290       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
4291       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
4292       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
4293       return DAG.getNode(NewOpcode, SDLoc(N), VT, LHS, RHS);
4294     }
4295   }
4296   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are constants
4297   if (isAllOnesConstant(N1) &&
4298       (N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::AND)) {
4299     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
4300     if (isa<ConstantSDNode>(RHS) || isa<ConstantSDNode>(LHS)) {
4301       unsigned NewOpcode = N0.getOpcode() == ISD::AND ? ISD::OR : ISD::AND;
4302       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
4303       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
4304       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
4305       return DAG.getNode(NewOpcode, SDLoc(N), VT, LHS, RHS);
4306     }
4307   }
4308   // fold (xor (and x, y), y) -> (and (not x), y)
4309   if (N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
4310       N0->getOperand(1) == N1) {
4311     SDValue X = N0->getOperand(0);
4312     SDValue NotX = DAG.getNOT(SDLoc(X), X, VT);
4313     AddToWorklist(NotX.getNode());
4314     return DAG.getNode(ISD::AND, SDLoc(N), VT, NotX, N1);
4315   }
4316   // fold (xor (xor x, c1), c2) -> (xor x, (xor c1, c2))
4317   if (N1C && N0.getOpcode() == ISD::XOR) {
4318     if (const ConstantSDNode *N00C = getAsNonOpaqueConstant(N0.getOperand(0))) {
4319       SDLoc DL(N);
4320       return DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(1),
4321                          DAG.getConstant(N1C->getAPIntValue() ^
4322                                          N00C->getAPIntValue(), DL, VT));
4323     }
4324     if (const ConstantSDNode *N01C = getAsNonOpaqueConstant(N0.getOperand(1))) {
4325       SDLoc DL(N);
4326       return DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(0),
4327                          DAG.getConstant(N1C->getAPIntValue() ^
4328                                          N01C->getAPIntValue(), DL, VT));
4329     }
4330   }
4331   // fold (xor x, x) -> 0
4332   if (N0 == N1)
4333     return tryFoldToZero(SDLoc(N), TLI, VT, DAG, LegalOperations, LegalTypes);
4334 
4335   // fold (xor (shl 1, x), -1) -> (rotl ~1, x)
4336   // Here is a concrete example of this equivalence:
4337   // i16   x ==  14
4338   // i16 shl ==   1 << 14  == 16384 == 0b0100000000000000
4339   // i16 xor == ~(1 << 14) == 49151 == 0b1011111111111111
4340   //
4341   // =>
4342   //
4343   // i16     ~1      == 0b1111111111111110
4344   // i16 rol(~1, 14) == 0b1011111111111111
4345   //
4346   // Some additional tips to help conceptualize this transform:
4347   // - Try to see the operation as placing a single zero in a value of all ones.
4348   // - There exists no value for x which would allow the result to contain zero.
4349   // - Values of x larger than the bitwidth are undefined and do not require a
4350   //   consistent result.
4351   // - Pushing the zero left requires shifting one bits in from the right.
4352   // A rotate left of ~1 is a nice way of achieving the desired result.
4353   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT) && N0.getOpcode() == ISD::SHL
4354       && isAllOnesConstant(N1) && isOneConstant(N0.getOperand(0))) {
4355     SDLoc DL(N);
4356     return DAG.getNode(ISD::ROTL, DL, VT, DAG.getConstant(~1, DL, VT),
4357                        N0.getOperand(1));
4358   }
4359 
4360   // Simplify: xor (op x...), (op y...)  -> (op (xor x, y))
4361   if (N0.getOpcode() == N1.getOpcode())
4362     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
4363       return Tmp;
4364 
4365   // Simplify the expression using non-local knowledge.
4366   if (!VT.isVector() &&
4367       SimplifyDemandedBits(SDValue(N, 0)))
4368     return SDValue(N, 0);
4369 
4370   return SDValue();
4371 }
4372 
4373 /// Handle transforms common to the three shifts, when the shift amount is a
4374 /// constant.
4375 SDValue DAGCombiner::visitShiftByConstant(SDNode *N, ConstantSDNode *Amt) {
4376   SDNode *LHS = N->getOperand(0).getNode();
4377   if (!LHS->hasOneUse()) return SDValue();
4378 
4379   // We want to pull some binops through shifts, so that we have (and (shift))
4380   // instead of (shift (and)), likewise for add, or, xor, etc.  This sort of
4381   // thing happens with address calculations, so it's important to canonicalize
4382   // it.
4383   bool HighBitSet = false;  // Can we transform this if the high bit is set?
4384 
4385   switch (LHS->getOpcode()) {
4386   default: return SDValue();
4387   case ISD::OR:
4388   case ISD::XOR:
4389     HighBitSet = false; // We can only transform sra if the high bit is clear.
4390     break;
4391   case ISD::AND:
4392     HighBitSet = true;  // We can only transform sra if the high bit is set.
4393     break;
4394   case ISD::ADD:
4395     if (N->getOpcode() != ISD::SHL)
4396       return SDValue(); // only shl(add) not sr[al](add).
4397     HighBitSet = false; // We can only transform sra if the high bit is clear.
4398     break;
4399   }
4400 
4401   // We require the RHS of the binop to be a constant and not opaque as well.
4402   ConstantSDNode *BinOpCst = getAsNonOpaqueConstant(LHS->getOperand(1));
4403   if (!BinOpCst) return SDValue();
4404 
4405   // FIXME: disable this unless the input to the binop is a shift by a constant.
4406   // If it is not a shift, it pessimizes some common cases like:
4407   //
4408   //    void foo(int *X, int i) { X[i & 1235] = 1; }
4409   //    int bar(int *X, int i) { return X[i & 255]; }
4410   SDNode *BinOpLHSVal = LHS->getOperand(0).getNode();
4411   if ((BinOpLHSVal->getOpcode() != ISD::SHL &&
4412        BinOpLHSVal->getOpcode() != ISD::SRA &&
4413        BinOpLHSVal->getOpcode() != ISD::SRL) ||
4414       !isa<ConstantSDNode>(BinOpLHSVal->getOperand(1)))
4415     return SDValue();
4416 
4417   EVT VT = N->getValueType(0);
4418 
4419   // If this is a signed shift right, and the high bit is modified by the
4420   // logical operation, do not perform the transformation. The highBitSet
4421   // boolean indicates the value of the high bit of the constant which would
4422   // cause it to be modified for this operation.
4423   if (N->getOpcode() == ISD::SRA) {
4424     bool BinOpRHSSignSet = BinOpCst->getAPIntValue().isNegative();
4425     if (BinOpRHSSignSet != HighBitSet)
4426       return SDValue();
4427   }
4428 
4429   if (!TLI.isDesirableToCommuteWithShift(LHS))
4430     return SDValue();
4431 
4432   // Fold the constants, shifting the binop RHS by the shift amount.
4433   SDValue NewRHS = DAG.getNode(N->getOpcode(), SDLoc(LHS->getOperand(1)),
4434                                N->getValueType(0),
4435                                LHS->getOperand(1), N->getOperand(1));
4436   assert(isa<ConstantSDNode>(NewRHS) && "Folding was not successful!");
4437 
4438   // Create the new shift.
4439   SDValue NewShift = DAG.getNode(N->getOpcode(),
4440                                  SDLoc(LHS->getOperand(0)),
4441                                  VT, LHS->getOperand(0), N->getOperand(1));
4442 
4443   // Create the new binop.
4444   return DAG.getNode(LHS->getOpcode(), SDLoc(N), VT, NewShift, NewRHS);
4445 }
4446 
4447 SDValue DAGCombiner::distributeTruncateThroughAnd(SDNode *N) {
4448   assert(N->getOpcode() == ISD::TRUNCATE);
4449   assert(N->getOperand(0).getOpcode() == ISD::AND);
4450 
4451   // (truncate:TruncVT (and N00, N01C)) -> (and (truncate:TruncVT N00), TruncC)
4452   if (N->hasOneUse() && N->getOperand(0).hasOneUse()) {
4453     SDValue N01 = N->getOperand(0).getOperand(1);
4454 
4455     if (ConstantSDNode *N01C = isConstOrConstSplat(N01)) {
4456       if (!N01C->isOpaque()) {
4457         EVT TruncVT = N->getValueType(0);
4458         SDValue N00 = N->getOperand(0).getOperand(0);
4459         APInt TruncC = N01C->getAPIntValue();
4460         TruncC = TruncC.trunc(TruncVT.getScalarSizeInBits());
4461         SDLoc DL(N);
4462 
4463         return DAG.getNode(ISD::AND, DL, TruncVT,
4464                            DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N00),
4465                            DAG.getConstant(TruncC, DL, TruncVT));
4466       }
4467     }
4468   }
4469 
4470   return SDValue();
4471 }
4472 
4473 SDValue DAGCombiner::visitRotate(SDNode *N) {
4474   // fold (rot* x, (trunc (and y, c))) -> (rot* x, (and (trunc y), (trunc c))).
4475   if (N->getOperand(1).getOpcode() == ISD::TRUNCATE &&
4476       N->getOperand(1).getOperand(0).getOpcode() == ISD::AND) {
4477     if (SDValue NewOp1 =
4478             distributeTruncateThroughAnd(N->getOperand(1).getNode()))
4479       return DAG.getNode(N->getOpcode(), SDLoc(N), N->getValueType(0),
4480                          N->getOperand(0), NewOp1);
4481   }
4482   return SDValue();
4483 }
4484 
4485 SDValue DAGCombiner::visitSHL(SDNode *N) {
4486   SDValue N0 = N->getOperand(0);
4487   SDValue N1 = N->getOperand(1);
4488   EVT VT = N0.getValueType();
4489   unsigned OpSizeInBits = VT.getScalarSizeInBits();
4490 
4491   // fold vector ops
4492   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
4493   if (VT.isVector()) {
4494     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4495       return FoldedVOp;
4496 
4497     BuildVectorSDNode *N1CV = dyn_cast<BuildVectorSDNode>(N1);
4498     // If setcc produces all-one true value then:
4499     // (shl (and (setcc) N01CV) N1CV) -> (and (setcc) N01CV<<N1CV)
4500     if (N1CV && N1CV->isConstant()) {
4501       if (N0.getOpcode() == ISD::AND) {
4502         SDValue N00 = N0->getOperand(0);
4503         SDValue N01 = N0->getOperand(1);
4504         BuildVectorSDNode *N01CV = dyn_cast<BuildVectorSDNode>(N01);
4505 
4506         if (N01CV && N01CV->isConstant() && N00.getOpcode() == ISD::SETCC &&
4507             TLI.getBooleanContents(N00.getOperand(0).getValueType()) ==
4508                 TargetLowering::ZeroOrNegativeOneBooleanContent) {
4509           if (SDValue C = DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT,
4510                                                      N01CV, N1CV))
4511             return DAG.getNode(ISD::AND, SDLoc(N), VT, N00, C);
4512         }
4513       } else {
4514         N1C = isConstOrConstSplat(N1);
4515       }
4516     }
4517   }
4518 
4519   // fold (shl c1, c2) -> c1<<c2
4520   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4521   if (N0C && N1C && !N1C->isOpaque())
4522     return DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT, N0C, N1C);
4523   // fold (shl 0, x) -> 0
4524   if (isNullConstant(N0))
4525     return N0;
4526   // fold (shl x, c >= size(x)) -> undef
4527   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
4528     return DAG.getUNDEF(VT);
4529   // fold (shl x, 0) -> x
4530   if (N1C && N1C->isNullValue())
4531     return N0;
4532   // fold (shl undef, x) -> 0
4533   if (N0.isUndef())
4534     return DAG.getConstant(0, SDLoc(N), VT);
4535   // if (shl x, c) is known to be zero, return 0
4536   if (DAG.MaskedValueIsZero(SDValue(N, 0),
4537                             APInt::getAllOnesValue(OpSizeInBits)))
4538     return DAG.getConstant(0, SDLoc(N), VT);
4539   // fold (shl x, (trunc (and y, c))) -> (shl x, (and (trunc y), (trunc c))).
4540   if (N1.getOpcode() == ISD::TRUNCATE &&
4541       N1.getOperand(0).getOpcode() == ISD::AND) {
4542     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
4543       return DAG.getNode(ISD::SHL, SDLoc(N), VT, N0, NewOp1);
4544   }
4545 
4546   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
4547     return SDValue(N, 0);
4548 
4549   // fold (shl (shl x, c1), c2) -> 0 or (shl x, (add c1, c2))
4550   if (N1C && N0.getOpcode() == ISD::SHL) {
4551     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4552       SDLoc DL(N);
4553       APInt c1 = N0C1->getAPIntValue();
4554       APInt c2 = N1C->getAPIntValue();
4555       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4556 
4557       APInt Sum = c1 + c2;
4558       if (Sum.uge(OpSizeInBits))
4559         return DAG.getConstant(0, DL, VT);
4560 
4561       return DAG.getNode(
4562           ISD::SHL, DL, VT, N0.getOperand(0),
4563           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4564     }
4565   }
4566 
4567   // fold (shl (ext (shl x, c1)), c2) -> (ext (shl x, (add c1, c2)))
4568   // For this to be valid, the second form must not preserve any of the bits
4569   // that are shifted out by the inner shift in the first form.  This means
4570   // the outer shift size must be >= the number of bits added by the ext.
4571   // As a corollary, we don't care what kind of ext it is.
4572   if (N1C && (N0.getOpcode() == ISD::ZERO_EXTEND ||
4573               N0.getOpcode() == ISD::ANY_EXTEND ||
4574               N0.getOpcode() == ISD::SIGN_EXTEND) &&
4575       N0.getOperand(0).getOpcode() == ISD::SHL) {
4576     SDValue N0Op0 = N0.getOperand(0);
4577     if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) {
4578       APInt c1 = N0Op0C1->getAPIntValue();
4579       APInt c2 = N1C->getAPIntValue();
4580       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4581 
4582       EVT InnerShiftVT = N0Op0.getValueType();
4583       uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits();
4584       if (c2.uge(OpSizeInBits - InnerShiftSize)) {
4585         SDLoc DL(N0);
4586         APInt Sum = c1 + c2;
4587         if (Sum.uge(OpSizeInBits))
4588           return DAG.getConstant(0, DL, VT);
4589 
4590         return DAG.getNode(
4591             ISD::SHL, DL, VT,
4592             DAG.getNode(N0.getOpcode(), DL, VT, N0Op0->getOperand(0)),
4593             DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4594       }
4595     }
4596   }
4597 
4598   // fold (shl (zext (srl x, C)), C) -> (zext (shl (srl x, C), C))
4599   // Only fold this if the inner zext has no other uses to avoid increasing
4600   // the total number of instructions.
4601   if (N1C && N0.getOpcode() == ISD::ZERO_EXTEND && N0.hasOneUse() &&
4602       N0.getOperand(0).getOpcode() == ISD::SRL) {
4603     SDValue N0Op0 = N0.getOperand(0);
4604     if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) {
4605       if (N0Op0C1->getAPIntValue().ult(VT.getScalarSizeInBits())) {
4606         uint64_t c1 = N0Op0C1->getZExtValue();
4607         uint64_t c2 = N1C->getZExtValue();
4608         if (c1 == c2) {
4609           SDValue NewOp0 = N0.getOperand(0);
4610           EVT CountVT = NewOp0.getOperand(1).getValueType();
4611           SDLoc DL(N);
4612           SDValue NewSHL = DAG.getNode(ISD::SHL, DL, NewOp0.getValueType(),
4613                                        NewOp0,
4614                                        DAG.getConstant(c2, DL, CountVT));
4615           AddToWorklist(NewSHL.getNode());
4616           return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N0), VT, NewSHL);
4617         }
4618       }
4619     }
4620   }
4621 
4622   // fold (shl (sr[la] exact X,  C1), C2) -> (shl    X, (C2-C1)) if C1 <= C2
4623   // fold (shl (sr[la] exact X,  C1), C2) -> (sr[la] X, (C2-C1)) if C1  > C2
4624   if (N1C && (N0.getOpcode() == ISD::SRL || N0.getOpcode() == ISD::SRA) &&
4625       cast<BinaryWithFlagsSDNode>(N0)->Flags.hasExact()) {
4626     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4627       uint64_t C1 = N0C1->getZExtValue();
4628       uint64_t C2 = N1C->getZExtValue();
4629       SDLoc DL(N);
4630       if (C1 <= C2)
4631         return DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
4632                            DAG.getConstant(C2 - C1, DL, N1.getValueType()));
4633       return DAG.getNode(N0.getOpcode(), DL, VT, N0.getOperand(0),
4634                          DAG.getConstant(C1 - C2, DL, N1.getValueType()));
4635     }
4636   }
4637 
4638   // fold (shl (srl x, c1), c2) -> (and (shl x, (sub c2, c1), MASK) or
4639   //                               (and (srl x, (sub c1, c2), MASK)
4640   // Only fold this if the inner shift has no other uses -- if it does, folding
4641   // this will increase the total number of instructions.
4642   if (N1C && N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
4643     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4644       uint64_t c1 = N0C1->getZExtValue();
4645       if (c1 < OpSizeInBits) {
4646         uint64_t c2 = N1C->getZExtValue();
4647         APInt Mask = APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - c1);
4648         SDValue Shift;
4649         if (c2 > c1) {
4650           Mask = Mask.shl(c2 - c1);
4651           SDLoc DL(N);
4652           Shift = DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
4653                               DAG.getConstant(c2 - c1, DL, N1.getValueType()));
4654         } else {
4655           Mask = Mask.lshr(c1 - c2);
4656           SDLoc DL(N);
4657           Shift = DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0),
4658                               DAG.getConstant(c1 - c2, DL, N1.getValueType()));
4659         }
4660         SDLoc DL(N0);
4661         return DAG.getNode(ISD::AND, DL, VT, Shift,
4662                            DAG.getConstant(Mask, DL, VT));
4663       }
4664     }
4665   }
4666   // fold (shl (sra x, c1), c1) -> (and x, (shl -1, c1))
4667   if (N1C && N0.getOpcode() == ISD::SRA && N1 == N0.getOperand(1)) {
4668     unsigned BitSize = VT.getScalarSizeInBits();
4669     SDLoc DL(N);
4670     SDValue HiBitsMask =
4671       DAG.getConstant(APInt::getHighBitsSet(BitSize,
4672                                             BitSize - N1C->getZExtValue()),
4673                       DL, VT);
4674     return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0),
4675                        HiBitsMask);
4676   }
4677 
4678   // fold (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
4679   // Variant of version done on multiply, except mul by a power of 2 is turned
4680   // into a shift.
4681   APInt Val;
4682   if (N1C && N0.getOpcode() == ISD::ADD && N0.getNode()->hasOneUse() &&
4683       (isa<ConstantSDNode>(N0.getOperand(1)) ||
4684        ISD::isConstantSplatVector(N0.getOperand(1).getNode(), Val))) {
4685     SDValue Shl0 = DAG.getNode(ISD::SHL, SDLoc(N0), VT, N0.getOperand(0), N1);
4686     SDValue Shl1 = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1);
4687     return DAG.getNode(ISD::ADD, SDLoc(N), VT, Shl0, Shl1);
4688   }
4689 
4690   // fold (shl (mul x, c1), c2) -> (mul x, c1 << c2)
4691   if (N1C && N0.getOpcode() == ISD::MUL && N0.getNode()->hasOneUse()) {
4692     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4693       if (SDValue Folded =
4694               DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N1), VT, N0C1, N1C))
4695         return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), Folded);
4696     }
4697   }
4698 
4699   if (N1C && !N1C->isOpaque())
4700     if (SDValue NewSHL = visitShiftByConstant(N, N1C))
4701       return NewSHL;
4702 
4703   return SDValue();
4704 }
4705 
4706 SDValue DAGCombiner::visitSRA(SDNode *N) {
4707   SDValue N0 = N->getOperand(0);
4708   SDValue N1 = N->getOperand(1);
4709   EVT VT = N0.getValueType();
4710   unsigned OpSizeInBits = VT.getScalarSizeInBits();
4711 
4712   // fold vector ops
4713   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
4714   if (VT.isVector()) {
4715     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4716       return FoldedVOp;
4717 
4718     N1C = isConstOrConstSplat(N1);
4719   }
4720 
4721   // fold (sra c1, c2) -> (sra c1, c2)
4722   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4723   if (N0C && N1C && !N1C->isOpaque())
4724     return DAG.FoldConstantArithmetic(ISD::SRA, SDLoc(N), VT, N0C, N1C);
4725   // fold (sra 0, x) -> 0
4726   if (isNullConstant(N0))
4727     return N0;
4728   // fold (sra -1, x) -> -1
4729   if (isAllOnesConstant(N0))
4730     return N0;
4731   // fold (sra x, c >= size(x)) -> undef
4732   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
4733     return DAG.getUNDEF(VT);
4734   // fold (sra x, 0) -> x
4735   if (N1C && N1C->isNullValue())
4736     return N0;
4737   // fold (sra (shl x, c1), c1) -> sext_inreg for some c1 and target supports
4738   // sext_inreg.
4739   if (N1C && N0.getOpcode() == ISD::SHL && N1 == N0.getOperand(1)) {
4740     unsigned LowBits = OpSizeInBits - (unsigned)N1C->getZExtValue();
4741     EVT ExtVT = EVT::getIntegerVT(*DAG.getContext(), LowBits);
4742     if (VT.isVector())
4743       ExtVT = EVT::getVectorVT(*DAG.getContext(),
4744                                ExtVT, VT.getVectorNumElements());
4745     if ((!LegalOperations ||
4746          TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, ExtVT)))
4747       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
4748                          N0.getOperand(0), DAG.getValueType(ExtVT));
4749   }
4750 
4751   // fold (sra (sra x, c1), c2) -> (sra x, (add c1, c2))
4752   if (N1C && N0.getOpcode() == ISD::SRA) {
4753     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4754       SDLoc DL(N);
4755       APInt c1 = N0C1->getAPIntValue();
4756       APInt c2 = N1C->getAPIntValue();
4757       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4758 
4759       APInt Sum = c1 + c2;
4760       if (Sum.uge(OpSizeInBits))
4761         Sum = APInt(OpSizeInBits, OpSizeInBits - 1);
4762 
4763       return DAG.getNode(
4764           ISD::SRA, DL, VT, N0.getOperand(0),
4765           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4766     }
4767   }
4768 
4769   // fold (sra (shl X, m), (sub result_size, n))
4770   // -> (sign_extend (trunc (shl X, (sub (sub result_size, n), m)))) for
4771   // result_size - n != m.
4772   // If truncate is free for the target sext(shl) is likely to result in better
4773   // code.
4774   if (N0.getOpcode() == ISD::SHL && N1C) {
4775     // Get the two constanst of the shifts, CN0 = m, CN = n.
4776     const ConstantSDNode *N01C = isConstOrConstSplat(N0.getOperand(1));
4777     if (N01C) {
4778       LLVMContext &Ctx = *DAG.getContext();
4779       // Determine what the truncate's result bitsize and type would be.
4780       EVT TruncVT = EVT::getIntegerVT(Ctx, OpSizeInBits - N1C->getZExtValue());
4781 
4782       if (VT.isVector())
4783         TruncVT = EVT::getVectorVT(Ctx, TruncVT, VT.getVectorNumElements());
4784 
4785       // Determine the residual right-shift amount.
4786       int ShiftAmt = N1C->getZExtValue() - N01C->getZExtValue();
4787 
4788       // If the shift is not a no-op (in which case this should be just a sign
4789       // extend already), the truncated to type is legal, sign_extend is legal
4790       // on that type, and the truncate to that type is both legal and free,
4791       // perform the transform.
4792       if ((ShiftAmt > 0) &&
4793           TLI.isOperationLegalOrCustom(ISD::SIGN_EXTEND, TruncVT) &&
4794           TLI.isOperationLegalOrCustom(ISD::TRUNCATE, VT) &&
4795           TLI.isTruncateFree(VT, TruncVT)) {
4796 
4797         SDLoc DL(N);
4798         SDValue Amt = DAG.getConstant(ShiftAmt, DL,
4799             getShiftAmountTy(N0.getOperand(0).getValueType()));
4800         SDValue Shift = DAG.getNode(ISD::SRL, DL, VT,
4801                                     N0.getOperand(0), Amt);
4802         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, TruncVT,
4803                                     Shift);
4804         return DAG.getNode(ISD::SIGN_EXTEND, DL,
4805                            N->getValueType(0), Trunc);
4806       }
4807     }
4808   }
4809 
4810   // fold (sra x, (trunc (and y, c))) -> (sra x, (and (trunc y), (trunc c))).
4811   if (N1.getOpcode() == ISD::TRUNCATE &&
4812       N1.getOperand(0).getOpcode() == ISD::AND) {
4813     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
4814       return DAG.getNode(ISD::SRA, SDLoc(N), VT, N0, NewOp1);
4815   }
4816 
4817   // fold (sra (trunc (srl x, c1)), c2) -> (trunc (sra x, c1 + c2))
4818   //      if c1 is equal to the number of bits the trunc removes
4819   if (N0.getOpcode() == ISD::TRUNCATE &&
4820       (N0.getOperand(0).getOpcode() == ISD::SRL ||
4821        N0.getOperand(0).getOpcode() == ISD::SRA) &&
4822       N0.getOperand(0).hasOneUse() &&
4823       N0.getOperand(0).getOperand(1).hasOneUse() &&
4824       N1C) {
4825     SDValue N0Op0 = N0.getOperand(0);
4826     if (ConstantSDNode *LargeShift = isConstOrConstSplat(N0Op0.getOperand(1))) {
4827       unsigned LargeShiftVal = LargeShift->getZExtValue();
4828       EVT LargeVT = N0Op0.getValueType();
4829 
4830       if (LargeVT.getScalarSizeInBits() - OpSizeInBits == LargeShiftVal) {
4831         SDLoc DL(N);
4832         SDValue Amt =
4833           DAG.getConstant(LargeShiftVal + N1C->getZExtValue(), DL,
4834                           getShiftAmountTy(N0Op0.getOperand(0).getValueType()));
4835         SDValue SRA = DAG.getNode(ISD::SRA, DL, LargeVT,
4836                                   N0Op0.getOperand(0), Amt);
4837         return DAG.getNode(ISD::TRUNCATE, DL, VT, SRA);
4838       }
4839     }
4840   }
4841 
4842   // Simplify, based on bits shifted out of the LHS.
4843   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
4844     return SDValue(N, 0);
4845 
4846 
4847   // If the sign bit is known to be zero, switch this to a SRL.
4848   if (DAG.SignBitIsZero(N0))
4849     return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, N1);
4850 
4851   if (N1C && !N1C->isOpaque())
4852     if (SDValue NewSRA = visitShiftByConstant(N, N1C))
4853       return NewSRA;
4854 
4855   return SDValue();
4856 }
4857 
4858 SDValue DAGCombiner::visitSRL(SDNode *N) {
4859   SDValue N0 = N->getOperand(0);
4860   SDValue N1 = N->getOperand(1);
4861   EVT VT = N0.getValueType();
4862   unsigned OpSizeInBits = VT.getScalarSizeInBits();
4863 
4864   // fold vector ops
4865   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
4866   if (VT.isVector()) {
4867     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4868       return FoldedVOp;
4869 
4870     N1C = isConstOrConstSplat(N1);
4871   }
4872 
4873   // fold (srl c1, c2) -> c1 >>u c2
4874   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4875   if (N0C && N1C && !N1C->isOpaque())
4876     return DAG.FoldConstantArithmetic(ISD::SRL, SDLoc(N), VT, N0C, N1C);
4877   // fold (srl 0, x) -> 0
4878   if (isNullConstant(N0))
4879     return N0;
4880   // fold (srl x, c >= size(x)) -> undef
4881   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
4882     return DAG.getUNDEF(VT);
4883   // fold (srl x, 0) -> x
4884   if (N1C && N1C->isNullValue())
4885     return N0;
4886   // if (srl x, c) is known to be zero, return 0
4887   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
4888                                    APInt::getAllOnesValue(OpSizeInBits)))
4889     return DAG.getConstant(0, SDLoc(N), VT);
4890 
4891   // fold (srl (srl x, c1), c2) -> 0 or (srl x, (add c1, c2))
4892   if (N1C && N0.getOpcode() == ISD::SRL) {
4893     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4894       SDLoc DL(N);
4895       APInt c1 = N0C1->getAPIntValue();
4896       APInt c2 = N1C->getAPIntValue();
4897       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4898 
4899       APInt Sum = c1 + c2;
4900       if (Sum.uge(OpSizeInBits))
4901         return DAG.getConstant(0, DL, VT);
4902 
4903       return DAG.getNode(
4904           ISD::SRL, DL, VT, N0.getOperand(0),
4905           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4906     }
4907   }
4908 
4909   // fold (srl (trunc (srl x, c1)), c2) -> 0 or (trunc (srl x, (add c1, c2)))
4910   if (N1C && N0.getOpcode() == ISD::TRUNCATE &&
4911       N0.getOperand(0).getOpcode() == ISD::SRL &&
4912       isa<ConstantSDNode>(N0.getOperand(0)->getOperand(1))) {
4913     uint64_t c1 =
4914       cast<ConstantSDNode>(N0.getOperand(0)->getOperand(1))->getZExtValue();
4915     uint64_t c2 = N1C->getZExtValue();
4916     EVT InnerShiftVT = N0.getOperand(0).getValueType();
4917     EVT ShiftCountVT = N0.getOperand(0)->getOperand(1).getValueType();
4918     uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits();
4919     // This is only valid if the OpSizeInBits + c1 = size of inner shift.
4920     if (c1 + OpSizeInBits == InnerShiftSize) {
4921       SDLoc DL(N0);
4922       if (c1 + c2 >= InnerShiftSize)
4923         return DAG.getConstant(0, DL, VT);
4924       return DAG.getNode(ISD::TRUNCATE, DL, VT,
4925                          DAG.getNode(ISD::SRL, DL, InnerShiftVT,
4926                                      N0.getOperand(0)->getOperand(0),
4927                                      DAG.getConstant(c1 + c2, DL,
4928                                                      ShiftCountVT)));
4929     }
4930   }
4931 
4932   // fold (srl (shl x, c), c) -> (and x, cst2)
4933   if (N1C && N0.getOpcode() == ISD::SHL && N0.getOperand(1) == N1) {
4934     unsigned BitSize = N0.getScalarValueSizeInBits();
4935     if (BitSize <= 64) {
4936       uint64_t ShAmt = N1C->getZExtValue() + 64 - BitSize;
4937       SDLoc DL(N);
4938       return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0),
4939                          DAG.getConstant(~0ULL >> ShAmt, DL, VT));
4940     }
4941   }
4942 
4943   // fold (srl (anyextend x), c) -> (and (anyextend (srl x, c)), mask)
4944   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
4945     // Shifting in all undef bits?
4946     EVT SmallVT = N0.getOperand(0).getValueType();
4947     unsigned BitSize = SmallVT.getScalarSizeInBits();
4948     if (N1C->getZExtValue() >= BitSize)
4949       return DAG.getUNDEF(VT);
4950 
4951     if (!LegalTypes || TLI.isTypeDesirableForOp(ISD::SRL, SmallVT)) {
4952       uint64_t ShiftAmt = N1C->getZExtValue();
4953       SDLoc DL0(N0);
4954       SDValue SmallShift = DAG.getNode(ISD::SRL, DL0, SmallVT,
4955                                        N0.getOperand(0),
4956                           DAG.getConstant(ShiftAmt, DL0,
4957                                           getShiftAmountTy(SmallVT)));
4958       AddToWorklist(SmallShift.getNode());
4959       APInt Mask = APInt::getAllOnesValue(OpSizeInBits).lshr(ShiftAmt);
4960       SDLoc DL(N);
4961       return DAG.getNode(ISD::AND, DL, VT,
4962                          DAG.getNode(ISD::ANY_EXTEND, DL, VT, SmallShift),
4963                          DAG.getConstant(Mask, DL, VT));
4964     }
4965   }
4966 
4967   // fold (srl (sra X, Y), 31) -> (srl X, 31).  This srl only looks at the sign
4968   // bit, which is unmodified by sra.
4969   if (N1C && N1C->getZExtValue() + 1 == OpSizeInBits) {
4970     if (N0.getOpcode() == ISD::SRA)
4971       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0.getOperand(0), N1);
4972   }
4973 
4974   // fold (srl (ctlz x), "5") -> x  iff x has one bit set (the low bit).
4975   if (N1C && N0.getOpcode() == ISD::CTLZ &&
4976       N1C->getAPIntValue() == Log2_32(OpSizeInBits)) {
4977     APInt KnownZero, KnownOne;
4978     DAG.computeKnownBits(N0.getOperand(0), KnownZero, KnownOne);
4979 
4980     // If any of the input bits are KnownOne, then the input couldn't be all
4981     // zeros, thus the result of the srl will always be zero.
4982     if (KnownOne.getBoolValue()) return DAG.getConstant(0, SDLoc(N0), VT);
4983 
4984     // If all of the bits input the to ctlz node are known to be zero, then
4985     // the result of the ctlz is "32" and the result of the shift is one.
4986     APInt UnknownBits = ~KnownZero;
4987     if (UnknownBits == 0) return DAG.getConstant(1, SDLoc(N0), VT);
4988 
4989     // Otherwise, check to see if there is exactly one bit input to the ctlz.
4990     if ((UnknownBits & (UnknownBits - 1)) == 0) {
4991       // Okay, we know that only that the single bit specified by UnknownBits
4992       // could be set on input to the CTLZ node. If this bit is set, the SRL
4993       // will return 0, if it is clear, it returns 1. Change the CTLZ/SRL pair
4994       // to an SRL/XOR pair, which is likely to simplify more.
4995       unsigned ShAmt = UnknownBits.countTrailingZeros();
4996       SDValue Op = N0.getOperand(0);
4997 
4998       if (ShAmt) {
4999         SDLoc DL(N0);
5000         Op = DAG.getNode(ISD::SRL, DL, VT, Op,
5001                   DAG.getConstant(ShAmt, DL,
5002                                   getShiftAmountTy(Op.getValueType())));
5003         AddToWorklist(Op.getNode());
5004       }
5005 
5006       SDLoc DL(N);
5007       return DAG.getNode(ISD::XOR, DL, VT,
5008                          Op, DAG.getConstant(1, DL, VT));
5009     }
5010   }
5011 
5012   // fold (srl x, (trunc (and y, c))) -> (srl x, (and (trunc y), (trunc c))).
5013   if (N1.getOpcode() == ISD::TRUNCATE &&
5014       N1.getOperand(0).getOpcode() == ISD::AND) {
5015     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
5016       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, NewOp1);
5017   }
5018 
5019   // fold operands of srl based on knowledge that the low bits are not
5020   // demanded.
5021   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
5022     return SDValue(N, 0);
5023 
5024   if (N1C && !N1C->isOpaque())
5025     if (SDValue NewSRL = visitShiftByConstant(N, N1C))
5026       return NewSRL;
5027 
5028   // Attempt to convert a srl of a load into a narrower zero-extending load.
5029   if (SDValue NarrowLoad = ReduceLoadWidth(N))
5030     return NarrowLoad;
5031 
5032   // Here is a common situation. We want to optimize:
5033   //
5034   //   %a = ...
5035   //   %b = and i32 %a, 2
5036   //   %c = srl i32 %b, 1
5037   //   brcond i32 %c ...
5038   //
5039   // into
5040   //
5041   //   %a = ...
5042   //   %b = and %a, 2
5043   //   %c = setcc eq %b, 0
5044   //   brcond %c ...
5045   //
5046   // However when after the source operand of SRL is optimized into AND, the SRL
5047   // itself may not be optimized further. Look for it and add the BRCOND into
5048   // the worklist.
5049   if (N->hasOneUse()) {
5050     SDNode *Use = *N->use_begin();
5051     if (Use->getOpcode() == ISD::BRCOND)
5052       AddToWorklist(Use);
5053     else if (Use->getOpcode() == ISD::TRUNCATE && Use->hasOneUse()) {
5054       // Also look pass the truncate.
5055       Use = *Use->use_begin();
5056       if (Use->getOpcode() == ISD::BRCOND)
5057         AddToWorklist(Use);
5058     }
5059   }
5060 
5061   return SDValue();
5062 }
5063 
5064 SDValue DAGCombiner::visitBSWAP(SDNode *N) {
5065   SDValue N0 = N->getOperand(0);
5066   EVT VT = N->getValueType(0);
5067 
5068   // fold (bswap c1) -> c2
5069   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5070     return DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N0);
5071   // fold (bswap (bswap x)) -> x
5072   if (N0.getOpcode() == ISD::BSWAP)
5073     return N0->getOperand(0);
5074   return SDValue();
5075 }
5076 
5077 SDValue DAGCombiner::visitBITREVERSE(SDNode *N) {
5078   SDValue N0 = N->getOperand(0);
5079 
5080   // fold (bitreverse (bitreverse x)) -> x
5081   if (N0.getOpcode() == ISD::BITREVERSE)
5082     return N0.getOperand(0);
5083   return SDValue();
5084 }
5085 
5086 SDValue DAGCombiner::visitCTLZ(SDNode *N) {
5087   SDValue N0 = N->getOperand(0);
5088   EVT VT = N->getValueType(0);
5089 
5090   // fold (ctlz c1) -> c2
5091   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5092     return DAG.getNode(ISD::CTLZ, SDLoc(N), VT, N0);
5093   return SDValue();
5094 }
5095 
5096 SDValue DAGCombiner::visitCTLZ_ZERO_UNDEF(SDNode *N) {
5097   SDValue N0 = N->getOperand(0);
5098   EVT VT = N->getValueType(0);
5099 
5100   // fold (ctlz_zero_undef c1) -> c2
5101   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5102     return DAG.getNode(ISD::CTLZ_ZERO_UNDEF, SDLoc(N), VT, N0);
5103   return SDValue();
5104 }
5105 
5106 SDValue DAGCombiner::visitCTTZ(SDNode *N) {
5107   SDValue N0 = N->getOperand(0);
5108   EVT VT = N->getValueType(0);
5109 
5110   // fold (cttz c1) -> c2
5111   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5112     return DAG.getNode(ISD::CTTZ, SDLoc(N), VT, N0);
5113   return SDValue();
5114 }
5115 
5116 SDValue DAGCombiner::visitCTTZ_ZERO_UNDEF(SDNode *N) {
5117   SDValue N0 = N->getOperand(0);
5118   EVT VT = N->getValueType(0);
5119 
5120   // fold (cttz_zero_undef c1) -> c2
5121   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5122     return DAG.getNode(ISD::CTTZ_ZERO_UNDEF, SDLoc(N), VT, N0);
5123   return SDValue();
5124 }
5125 
5126 SDValue DAGCombiner::visitCTPOP(SDNode *N) {
5127   SDValue N0 = N->getOperand(0);
5128   EVT VT = N->getValueType(0);
5129 
5130   // fold (ctpop c1) -> c2
5131   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5132     return DAG.getNode(ISD::CTPOP, SDLoc(N), VT, N0);
5133   return SDValue();
5134 }
5135 
5136 
5137 /// \brief Generate Min/Max node
5138 static SDValue combineMinNumMaxNum(const SDLoc &DL, EVT VT, SDValue LHS,
5139                                    SDValue RHS, SDValue True, SDValue False,
5140                                    ISD::CondCode CC, const TargetLowering &TLI,
5141                                    SelectionDAG &DAG) {
5142   if (!(LHS == True && RHS == False) && !(LHS == False && RHS == True))
5143     return SDValue();
5144 
5145   switch (CC) {
5146   case ISD::SETOLT:
5147   case ISD::SETOLE:
5148   case ISD::SETLT:
5149   case ISD::SETLE:
5150   case ISD::SETULT:
5151   case ISD::SETULE: {
5152     unsigned Opcode = (LHS == True) ? ISD::FMINNUM : ISD::FMAXNUM;
5153     if (TLI.isOperationLegal(Opcode, VT))
5154       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
5155     return SDValue();
5156   }
5157   case ISD::SETOGT:
5158   case ISD::SETOGE:
5159   case ISD::SETGT:
5160   case ISD::SETGE:
5161   case ISD::SETUGT:
5162   case ISD::SETUGE: {
5163     unsigned Opcode = (LHS == True) ? ISD::FMAXNUM : ISD::FMINNUM;
5164     if (TLI.isOperationLegal(Opcode, VT))
5165       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
5166     return SDValue();
5167   }
5168   default:
5169     return SDValue();
5170   }
5171 }
5172 
5173 // TODO: We should handle other cases of selecting between {-1,0,1} here.
5174 SDValue DAGCombiner::foldSelectOfConstants(SDNode *N) {
5175   SDValue Cond = N->getOperand(0);
5176   SDValue N1 = N->getOperand(1);
5177   SDValue N2 = N->getOperand(2);
5178   EVT VT = N->getValueType(0);
5179   EVT CondVT = Cond.getValueType();
5180   SDLoc DL(N);
5181 
5182   // fold (select Cond, 0, 1) -> (xor Cond, 1)
5183   // We can't do this reliably if integer based booleans have different contents
5184   // to floating point based booleans. This is because we can't tell whether we
5185   // have an integer-based boolean or a floating-point-based boolean unless we
5186   // can find the SETCC that produced it and inspect its operands. This is
5187   // fairly easy if C is the SETCC node, but it can potentially be
5188   // undiscoverable (or not reasonably discoverable). For example, it could be
5189   // in another basic block or it could require searching a complicated
5190   // expression.
5191   if (VT.isInteger() &&
5192       (CondVT == MVT::i1 || (CondVT.isInteger() &&
5193                              TLI.getBooleanContents(false, true) ==
5194                                  TargetLowering::ZeroOrOneBooleanContent &&
5195                              TLI.getBooleanContents(false, false) ==
5196                                  TargetLowering::ZeroOrOneBooleanContent)) &&
5197       isNullConstant(N1) && isOneConstant(N2)) {
5198     SDValue NotCond = DAG.getNode(ISD::XOR, DL, CondVT, Cond,
5199                                   DAG.getConstant(1, DL, CondVT));
5200     if (VT.bitsEq(CondVT))
5201       return NotCond;
5202     return DAG.getZExtOrTrunc(NotCond, DL, VT);
5203   }
5204 
5205   return SDValue();
5206 }
5207 
5208 SDValue DAGCombiner::visitSELECT(SDNode *N) {
5209   SDValue N0 = N->getOperand(0);
5210   SDValue N1 = N->getOperand(1);
5211   SDValue N2 = N->getOperand(2);
5212   EVT VT = N->getValueType(0);
5213   EVT VT0 = N0.getValueType();
5214 
5215   // fold (select C, X, X) -> X
5216   if (N1 == N2)
5217     return N1;
5218   if (const ConstantSDNode *N0C = dyn_cast<const ConstantSDNode>(N0)) {
5219     // fold (select true, X, Y) -> X
5220     // fold (select false, X, Y) -> Y
5221     return !N0C->isNullValue() ? N1 : N2;
5222   }
5223   // fold (select C, 1, X) -> (or C, X)
5224   if (VT == MVT::i1 && isOneConstant(N1))
5225     return DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N2);
5226 
5227   if (SDValue V = foldSelectOfConstants(N))
5228     return V;
5229 
5230   // fold (select C, 0, X) -> (and (not C), X)
5231   if (VT == VT0 && VT == MVT::i1 && isNullConstant(N1)) {
5232     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
5233     AddToWorklist(NOTNode.getNode());
5234     return DAG.getNode(ISD::AND, SDLoc(N), VT, NOTNode, N2);
5235   }
5236   // fold (select C, X, 1) -> (or (not C), X)
5237   if (VT == VT0 && VT == MVT::i1 && isOneConstant(N2)) {
5238     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
5239     AddToWorklist(NOTNode.getNode());
5240     return DAG.getNode(ISD::OR, SDLoc(N), VT, NOTNode, N1);
5241   }
5242   // fold (select C, X, 0) -> (and C, X)
5243   if (VT == MVT::i1 && isNullConstant(N2))
5244     return DAG.getNode(ISD::AND, SDLoc(N), VT, N0, N1);
5245   // fold (select X, X, Y) -> (or X, Y)
5246   // fold (select X, 1, Y) -> (or X, Y)
5247   if (VT == MVT::i1 && (N0 == N1 || isOneConstant(N1)))
5248     return DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N2);
5249   // fold (select X, Y, X) -> (and X, Y)
5250   // fold (select X, Y, 0) -> (and X, Y)
5251   if (VT == MVT::i1 && (N0 == N2 || isNullConstant(N2)))
5252     return DAG.getNode(ISD::AND, SDLoc(N), VT, N0, N1);
5253 
5254   // If we can fold this based on the true/false value, do so.
5255   if (SimplifySelectOps(N, N1, N2))
5256     return SDValue(N, 0);  // Don't revisit N.
5257 
5258   if (VT0 == MVT::i1) {
5259     // The code in this block deals with the following 2 equivalences:
5260     //    select(C0|C1, x, y) <=> select(C0, x, select(C1, x, y))
5261     //    select(C0&C1, x, y) <=> select(C0, select(C1, x, y), y)
5262     // The target can specify its prefered form with the
5263     // shouldNormalizeToSelectSequence() callback. However we always transform
5264     // to the right anyway if we find the inner select exists in the DAG anyway
5265     // and we always transform to the left side if we know that we can further
5266     // optimize the combination of the conditions.
5267     bool normalizeToSequence
5268       = TLI.shouldNormalizeToSelectSequence(*DAG.getContext(), VT);
5269     // select (and Cond0, Cond1), X, Y
5270     //   -> select Cond0, (select Cond1, X, Y), Y
5271     if (N0->getOpcode() == ISD::AND && N0->hasOneUse()) {
5272       SDValue Cond0 = N0->getOperand(0);
5273       SDValue Cond1 = N0->getOperand(1);
5274       SDValue InnerSelect = DAG.getNode(ISD::SELECT, SDLoc(N),
5275                                         N1.getValueType(), Cond1, N1, N2);
5276       if (normalizeToSequence || !InnerSelect.use_empty())
5277         return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Cond0,
5278                            InnerSelect, N2);
5279     }
5280     // select (or Cond0, Cond1), X, Y -> select Cond0, X, (select Cond1, X, Y)
5281     if (N0->getOpcode() == ISD::OR && N0->hasOneUse()) {
5282       SDValue Cond0 = N0->getOperand(0);
5283       SDValue Cond1 = N0->getOperand(1);
5284       SDValue InnerSelect = DAG.getNode(ISD::SELECT, SDLoc(N),
5285                                         N1.getValueType(), Cond1, N1, N2);
5286       if (normalizeToSequence || !InnerSelect.use_empty())
5287         return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Cond0, N1,
5288                            InnerSelect);
5289     }
5290 
5291     // select Cond0, (select Cond1, X, Y), Y -> select (and Cond0, Cond1), X, Y
5292     if (N1->getOpcode() == ISD::SELECT && N1->hasOneUse()) {
5293       SDValue N1_0 = N1->getOperand(0);
5294       SDValue N1_1 = N1->getOperand(1);
5295       SDValue N1_2 = N1->getOperand(2);
5296       if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
5297         // Create the actual and node if we can generate good code for it.
5298         if (!normalizeToSequence) {
5299           SDValue And = DAG.getNode(ISD::AND, SDLoc(N), N0.getValueType(),
5300                                     N0, N1_0);
5301           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), And,
5302                              N1_1, N2);
5303         }
5304         // Otherwise see if we can optimize the "and" to a better pattern.
5305         if (SDValue Combined = visitANDLike(N0, N1_0, N))
5306           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Combined,
5307                              N1_1, N2);
5308       }
5309     }
5310     // select Cond0, X, (select Cond1, X, Y) -> select (or Cond0, Cond1), X, Y
5311     if (N2->getOpcode() == ISD::SELECT && N2->hasOneUse()) {
5312       SDValue N2_0 = N2->getOperand(0);
5313       SDValue N2_1 = N2->getOperand(1);
5314       SDValue N2_2 = N2->getOperand(2);
5315       if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
5316         // Create the actual or node if we can generate good code for it.
5317         if (!normalizeToSequence) {
5318           SDValue Or = DAG.getNode(ISD::OR, SDLoc(N), N0.getValueType(),
5319                                    N0, N2_0);
5320           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Or,
5321                              N1, N2_2);
5322         }
5323         // Otherwise see if we can optimize to a better pattern.
5324         if (SDValue Combined = visitORLike(N0, N2_0, N))
5325           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Combined,
5326                              N1, N2_2);
5327       }
5328     }
5329   }
5330 
5331   // select (xor Cond, 1), X, Y -> select Cond, Y, X
5332   // select (xor Cond, 0), X, Y -> selext Cond, X, Y
5333   if (VT0 == MVT::i1) {
5334     if (N0->getOpcode() == ISD::XOR) {
5335       if (auto *C = dyn_cast<ConstantSDNode>(N0->getOperand(1))) {
5336         SDValue Cond0 = N0->getOperand(0);
5337         if (C->isOne())
5338           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(),
5339                              Cond0, N2, N1);
5340         else
5341           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(),
5342                              Cond0, N1, N2);
5343       }
5344     }
5345   }
5346 
5347   // fold selects based on a setcc into other things, such as min/max/abs
5348   if (N0.getOpcode() == ISD::SETCC) {
5349     // select x, y (fcmp lt x, y) -> fminnum x, y
5350     // select x, y (fcmp gt x, y) -> fmaxnum x, y
5351     //
5352     // This is OK if we don't care about what happens if either operand is a
5353     // NaN.
5354     //
5355 
5356     // FIXME: Instead of testing for UnsafeFPMath, this should be checking for
5357     // no signed zeros as well as no nans.
5358     const TargetOptions &Options = DAG.getTarget().Options;
5359     if (Options.UnsafeFPMath &&
5360         VT.isFloatingPoint() && N0.hasOneUse() &&
5361         DAG.isKnownNeverNaN(N1) && DAG.isKnownNeverNaN(N2)) {
5362       ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
5363 
5364       if (SDValue FMinMax = combineMinNumMaxNum(SDLoc(N), VT, N0.getOperand(0),
5365                                                 N0.getOperand(1), N1, N2, CC,
5366                                                 TLI, DAG))
5367         return FMinMax;
5368     }
5369 
5370     if ((!LegalOperations &&
5371          TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT)) ||
5372         TLI.isOperationLegal(ISD::SELECT_CC, VT))
5373       return DAG.getNode(ISD::SELECT_CC, SDLoc(N), VT,
5374                          N0.getOperand(0), N0.getOperand(1),
5375                          N1, N2, N0.getOperand(2));
5376     return SimplifySelect(SDLoc(N), N0, N1, N2);
5377   }
5378 
5379   return SDValue();
5380 }
5381 
5382 static
5383 std::pair<SDValue, SDValue> SplitVSETCC(const SDNode *N, SelectionDAG &DAG) {
5384   SDLoc DL(N);
5385   EVT LoVT, HiVT;
5386   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(N->getValueType(0));
5387 
5388   // Split the inputs.
5389   SDValue Lo, Hi, LL, LH, RL, RH;
5390   std::tie(LL, LH) = DAG.SplitVectorOperand(N, 0);
5391   std::tie(RL, RH) = DAG.SplitVectorOperand(N, 1);
5392 
5393   Lo = DAG.getNode(N->getOpcode(), DL, LoVT, LL, RL, N->getOperand(2));
5394   Hi = DAG.getNode(N->getOpcode(), DL, HiVT, LH, RH, N->getOperand(2));
5395 
5396   return std::make_pair(Lo, Hi);
5397 }
5398 
5399 // This function assumes all the vselect's arguments are CONCAT_VECTOR
5400 // nodes and that the condition is a BV of ConstantSDNodes (or undefs).
5401 static SDValue ConvertSelectToConcatVector(SDNode *N, SelectionDAG &DAG) {
5402   SDLoc DL(N);
5403   SDValue Cond = N->getOperand(0);
5404   SDValue LHS = N->getOperand(1);
5405   SDValue RHS = N->getOperand(2);
5406   EVT VT = N->getValueType(0);
5407   int NumElems = VT.getVectorNumElements();
5408   assert(LHS.getOpcode() == ISD::CONCAT_VECTORS &&
5409          RHS.getOpcode() == ISD::CONCAT_VECTORS &&
5410          Cond.getOpcode() == ISD::BUILD_VECTOR);
5411 
5412   // CONCAT_VECTOR can take an arbitrary number of arguments. We only care about
5413   // binary ones here.
5414   if (LHS->getNumOperands() != 2 || RHS->getNumOperands() != 2)
5415     return SDValue();
5416 
5417   // We're sure we have an even number of elements due to the
5418   // concat_vectors we have as arguments to vselect.
5419   // Skip BV elements until we find one that's not an UNDEF
5420   // After we find an UNDEF element, keep looping until we get to half the
5421   // length of the BV and see if all the non-undef nodes are the same.
5422   ConstantSDNode *BottomHalf = nullptr;
5423   for (int i = 0; i < NumElems / 2; ++i) {
5424     if (Cond->getOperand(i)->isUndef())
5425       continue;
5426 
5427     if (BottomHalf == nullptr)
5428       BottomHalf = cast<ConstantSDNode>(Cond.getOperand(i));
5429     else if (Cond->getOperand(i).getNode() != BottomHalf)
5430       return SDValue();
5431   }
5432 
5433   // Do the same for the second half of the BuildVector
5434   ConstantSDNode *TopHalf = nullptr;
5435   for (int i = NumElems / 2; i < NumElems; ++i) {
5436     if (Cond->getOperand(i)->isUndef())
5437       continue;
5438 
5439     if (TopHalf == nullptr)
5440       TopHalf = cast<ConstantSDNode>(Cond.getOperand(i));
5441     else if (Cond->getOperand(i).getNode() != TopHalf)
5442       return SDValue();
5443   }
5444 
5445   assert(TopHalf && BottomHalf &&
5446          "One half of the selector was all UNDEFs and the other was all the "
5447          "same value. This should have been addressed before this function.");
5448   return DAG.getNode(
5449       ISD::CONCAT_VECTORS, DL, VT,
5450       BottomHalf->isNullValue() ? RHS->getOperand(0) : LHS->getOperand(0),
5451       TopHalf->isNullValue() ? RHS->getOperand(1) : LHS->getOperand(1));
5452 }
5453 
5454 SDValue DAGCombiner::visitMSCATTER(SDNode *N) {
5455 
5456   if (Level >= AfterLegalizeTypes)
5457     return SDValue();
5458 
5459   MaskedScatterSDNode *MSC = cast<MaskedScatterSDNode>(N);
5460   SDValue Mask = MSC->getMask();
5461   SDValue Data  = MSC->getValue();
5462   SDLoc DL(N);
5463 
5464   // If the MSCATTER data type requires splitting and the mask is provided by a
5465   // SETCC, then split both nodes and its operands before legalization. This
5466   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5467   // and enables future optimizations (e.g. min/max pattern matching on X86).
5468   if (Mask.getOpcode() != ISD::SETCC)
5469     return SDValue();
5470 
5471   // Check if any splitting is required.
5472   if (TLI.getTypeAction(*DAG.getContext(), Data.getValueType()) !=
5473       TargetLowering::TypeSplitVector)
5474     return SDValue();
5475   SDValue MaskLo, MaskHi, Lo, Hi;
5476   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5477 
5478   EVT LoVT, HiVT;
5479   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MSC->getValueType(0));
5480 
5481   SDValue Chain = MSC->getChain();
5482 
5483   EVT MemoryVT = MSC->getMemoryVT();
5484   unsigned Alignment = MSC->getOriginalAlignment();
5485 
5486   EVT LoMemVT, HiMemVT;
5487   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5488 
5489   SDValue DataLo, DataHi;
5490   std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
5491 
5492   SDValue BasePtr = MSC->getBasePtr();
5493   SDValue IndexLo, IndexHi;
5494   std::tie(IndexLo, IndexHi) = DAG.SplitVector(MSC->getIndex(), DL);
5495 
5496   MachineMemOperand *MMO = DAG.getMachineFunction().
5497     getMachineMemOperand(MSC->getPointerInfo(),
5498                           MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
5499                           Alignment, MSC->getAAInfo(), MSC->getRanges());
5500 
5501   SDValue OpsLo[] = { Chain, DataLo, MaskLo, BasePtr, IndexLo };
5502   Lo = DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataLo.getValueType(),
5503                             DL, OpsLo, MMO);
5504 
5505   SDValue OpsHi[] = {Chain, DataHi, MaskHi, BasePtr, IndexHi};
5506   Hi = DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataHi.getValueType(),
5507                             DL, OpsHi, MMO);
5508 
5509   AddToWorklist(Lo.getNode());
5510   AddToWorklist(Hi.getNode());
5511 
5512   return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi);
5513 }
5514 
5515 SDValue DAGCombiner::visitMSTORE(SDNode *N) {
5516 
5517   if (Level >= AfterLegalizeTypes)
5518     return SDValue();
5519 
5520   MaskedStoreSDNode *MST = dyn_cast<MaskedStoreSDNode>(N);
5521   SDValue Mask = MST->getMask();
5522   SDValue Data  = MST->getValue();
5523   SDLoc DL(N);
5524 
5525   // If the MSTORE data type requires splitting and the mask is provided by a
5526   // SETCC, then split both nodes and its operands before legalization. This
5527   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5528   // and enables future optimizations (e.g. min/max pattern matching on X86).
5529   if (Mask.getOpcode() == ISD::SETCC) {
5530 
5531     // Check if any splitting is required.
5532     if (TLI.getTypeAction(*DAG.getContext(), Data.getValueType()) !=
5533         TargetLowering::TypeSplitVector)
5534       return SDValue();
5535 
5536     SDValue MaskLo, MaskHi, Lo, Hi;
5537     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5538 
5539     EVT LoVT, HiVT;
5540     std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MST->getValueType(0));
5541 
5542     SDValue Chain = MST->getChain();
5543     SDValue Ptr   = MST->getBasePtr();
5544 
5545     EVT MemoryVT = MST->getMemoryVT();
5546     unsigned Alignment = MST->getOriginalAlignment();
5547 
5548     // if Alignment is equal to the vector size,
5549     // take the half of it for the second part
5550     unsigned SecondHalfAlignment =
5551       (Alignment == Data->getValueType(0).getSizeInBits()/8) ?
5552          Alignment/2 : Alignment;
5553 
5554     EVT LoMemVT, HiMemVT;
5555     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5556 
5557     SDValue DataLo, DataHi;
5558     std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
5559 
5560     MachineMemOperand *MMO = DAG.getMachineFunction().
5561       getMachineMemOperand(MST->getPointerInfo(),
5562                            MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
5563                            Alignment, MST->getAAInfo(), MST->getRanges());
5564 
5565     Lo = DAG.getMaskedStore(Chain, DL, DataLo, Ptr, MaskLo, LoMemVT, MMO,
5566                             MST->isTruncatingStore());
5567 
5568     unsigned IncrementSize = LoMemVT.getSizeInBits()/8;
5569     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
5570                       DAG.getConstant(IncrementSize, DL, Ptr.getValueType()));
5571 
5572     MMO = DAG.getMachineFunction().
5573       getMachineMemOperand(MST->getPointerInfo(),
5574                            MachineMemOperand::MOStore,  HiMemVT.getStoreSize(),
5575                            SecondHalfAlignment, MST->getAAInfo(),
5576                            MST->getRanges());
5577 
5578     Hi = DAG.getMaskedStore(Chain, DL, DataHi, Ptr, MaskHi, HiMemVT, MMO,
5579                             MST->isTruncatingStore());
5580 
5581     AddToWorklist(Lo.getNode());
5582     AddToWorklist(Hi.getNode());
5583 
5584     return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi);
5585   }
5586   return SDValue();
5587 }
5588 
5589 SDValue DAGCombiner::visitMGATHER(SDNode *N) {
5590 
5591   if (Level >= AfterLegalizeTypes)
5592     return SDValue();
5593 
5594   MaskedGatherSDNode *MGT = dyn_cast<MaskedGatherSDNode>(N);
5595   SDValue Mask = MGT->getMask();
5596   SDLoc DL(N);
5597 
5598   // If the MGATHER result requires splitting and the mask is provided by a
5599   // SETCC, then split both nodes and its operands before legalization. This
5600   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5601   // and enables future optimizations (e.g. min/max pattern matching on X86).
5602 
5603   if (Mask.getOpcode() != ISD::SETCC)
5604     return SDValue();
5605 
5606   EVT VT = N->getValueType(0);
5607 
5608   // Check if any splitting is required.
5609   if (TLI.getTypeAction(*DAG.getContext(), VT) !=
5610       TargetLowering::TypeSplitVector)
5611     return SDValue();
5612 
5613   SDValue MaskLo, MaskHi, Lo, Hi;
5614   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5615 
5616   SDValue Src0 = MGT->getValue();
5617   SDValue Src0Lo, Src0Hi;
5618   std::tie(Src0Lo, Src0Hi) = DAG.SplitVector(Src0, DL);
5619 
5620   EVT LoVT, HiVT;
5621   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
5622 
5623   SDValue Chain = MGT->getChain();
5624   EVT MemoryVT = MGT->getMemoryVT();
5625   unsigned Alignment = MGT->getOriginalAlignment();
5626 
5627   EVT LoMemVT, HiMemVT;
5628   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5629 
5630   SDValue BasePtr = MGT->getBasePtr();
5631   SDValue Index = MGT->getIndex();
5632   SDValue IndexLo, IndexHi;
5633   std::tie(IndexLo, IndexHi) = DAG.SplitVector(Index, DL);
5634 
5635   MachineMemOperand *MMO = DAG.getMachineFunction().
5636     getMachineMemOperand(MGT->getPointerInfo(),
5637                           MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
5638                           Alignment, MGT->getAAInfo(), MGT->getRanges());
5639 
5640   SDValue OpsLo[] = { Chain, Src0Lo, MaskLo, BasePtr, IndexLo };
5641   Lo = DAG.getMaskedGather(DAG.getVTList(LoVT, MVT::Other), LoVT, DL, OpsLo,
5642                             MMO);
5643 
5644   SDValue OpsHi[] = {Chain, Src0Hi, MaskHi, BasePtr, IndexHi};
5645   Hi = DAG.getMaskedGather(DAG.getVTList(HiVT, MVT::Other), HiVT, DL, OpsHi,
5646                             MMO);
5647 
5648   AddToWorklist(Lo.getNode());
5649   AddToWorklist(Hi.getNode());
5650 
5651   // Build a factor node to remember that this load is independent of the
5652   // other one.
5653   Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
5654                       Hi.getValue(1));
5655 
5656   // Legalized the chain result - switch anything that used the old chain to
5657   // use the new one.
5658   DAG.ReplaceAllUsesOfValueWith(SDValue(MGT, 1), Chain);
5659 
5660   SDValue GatherRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
5661 
5662   SDValue RetOps[] = { GatherRes, Chain };
5663   return DAG.getMergeValues(RetOps, DL);
5664 }
5665 
5666 SDValue DAGCombiner::visitMLOAD(SDNode *N) {
5667 
5668   if (Level >= AfterLegalizeTypes)
5669     return SDValue();
5670 
5671   MaskedLoadSDNode *MLD = dyn_cast<MaskedLoadSDNode>(N);
5672   SDValue Mask = MLD->getMask();
5673   SDLoc DL(N);
5674 
5675   // If the MLOAD result requires splitting and the mask is provided by a
5676   // SETCC, then split both nodes and its operands before legalization. This
5677   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5678   // and enables future optimizations (e.g. min/max pattern matching on X86).
5679 
5680   if (Mask.getOpcode() == ISD::SETCC) {
5681     EVT VT = N->getValueType(0);
5682 
5683     // Check if any splitting is required.
5684     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
5685         TargetLowering::TypeSplitVector)
5686       return SDValue();
5687 
5688     SDValue MaskLo, MaskHi, Lo, Hi;
5689     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5690 
5691     SDValue Src0 = MLD->getSrc0();
5692     SDValue Src0Lo, Src0Hi;
5693     std::tie(Src0Lo, Src0Hi) = DAG.SplitVector(Src0, DL);
5694 
5695     EVT LoVT, HiVT;
5696     std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MLD->getValueType(0));
5697 
5698     SDValue Chain = MLD->getChain();
5699     SDValue Ptr   = MLD->getBasePtr();
5700     EVT MemoryVT = MLD->getMemoryVT();
5701     unsigned Alignment = MLD->getOriginalAlignment();
5702 
5703     // if Alignment is equal to the vector size,
5704     // take the half of it for the second part
5705     unsigned SecondHalfAlignment =
5706       (Alignment == MLD->getValueType(0).getSizeInBits()/8) ?
5707          Alignment/2 : Alignment;
5708 
5709     EVT LoMemVT, HiMemVT;
5710     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5711 
5712     MachineMemOperand *MMO = DAG.getMachineFunction().
5713     getMachineMemOperand(MLD->getPointerInfo(),
5714                          MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
5715                          Alignment, MLD->getAAInfo(), MLD->getRanges());
5716 
5717     Lo = DAG.getMaskedLoad(LoVT, DL, Chain, Ptr, MaskLo, Src0Lo, LoMemVT, MMO,
5718                            ISD::NON_EXTLOAD);
5719 
5720     unsigned IncrementSize = LoMemVT.getSizeInBits()/8;
5721     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
5722                       DAG.getConstant(IncrementSize, DL, Ptr.getValueType()));
5723 
5724     MMO = DAG.getMachineFunction().
5725     getMachineMemOperand(MLD->getPointerInfo(),
5726                          MachineMemOperand::MOLoad,  HiMemVT.getStoreSize(),
5727                          SecondHalfAlignment, MLD->getAAInfo(), MLD->getRanges());
5728 
5729     Hi = DAG.getMaskedLoad(HiVT, DL, Chain, Ptr, MaskHi, Src0Hi, HiMemVT, MMO,
5730                            ISD::NON_EXTLOAD);
5731 
5732     AddToWorklist(Lo.getNode());
5733     AddToWorklist(Hi.getNode());
5734 
5735     // Build a factor node to remember that this load is independent of the
5736     // other one.
5737     Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
5738                         Hi.getValue(1));
5739 
5740     // Legalized the chain result - switch anything that used the old chain to
5741     // use the new one.
5742     DAG.ReplaceAllUsesOfValueWith(SDValue(MLD, 1), Chain);
5743 
5744     SDValue LoadRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
5745 
5746     SDValue RetOps[] = { LoadRes, Chain };
5747     return DAG.getMergeValues(RetOps, DL);
5748   }
5749   return SDValue();
5750 }
5751 
5752 SDValue DAGCombiner::visitVSELECT(SDNode *N) {
5753   SDValue N0 = N->getOperand(0);
5754   SDValue N1 = N->getOperand(1);
5755   SDValue N2 = N->getOperand(2);
5756   SDLoc DL(N);
5757 
5758   // Canonicalize integer abs.
5759   // vselect (setg[te] X,  0),  X, -X ->
5760   // vselect (setgt    X, -1),  X, -X ->
5761   // vselect (setl[te] X,  0), -X,  X ->
5762   // Y = sra (X, size(X)-1); xor (add (X, Y), Y)
5763   if (N0.getOpcode() == ISD::SETCC) {
5764     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
5765     ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
5766     bool isAbs = false;
5767     bool RHSIsAllZeros = ISD::isBuildVectorAllZeros(RHS.getNode());
5768 
5769     if (((RHSIsAllZeros && (CC == ISD::SETGT || CC == ISD::SETGE)) ||
5770          (ISD::isBuildVectorAllOnes(RHS.getNode()) && CC == ISD::SETGT)) &&
5771         N1 == LHS && N2.getOpcode() == ISD::SUB && N1 == N2.getOperand(1))
5772       isAbs = ISD::isBuildVectorAllZeros(N2.getOperand(0).getNode());
5773     else if ((RHSIsAllZeros && (CC == ISD::SETLT || CC == ISD::SETLE)) &&
5774              N2 == LHS && N1.getOpcode() == ISD::SUB && N2 == N1.getOperand(1))
5775       isAbs = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
5776 
5777     if (isAbs) {
5778       EVT VT = LHS.getValueType();
5779       SDValue Shift = DAG.getNode(
5780           ISD::SRA, DL, VT, LHS,
5781           DAG.getConstant(VT.getScalarSizeInBits() - 1, DL, VT));
5782       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, LHS, Shift);
5783       AddToWorklist(Shift.getNode());
5784       AddToWorklist(Add.getNode());
5785       return DAG.getNode(ISD::XOR, DL, VT, Add, Shift);
5786     }
5787   }
5788 
5789   if (SimplifySelectOps(N, N1, N2))
5790     return SDValue(N, 0);  // Don't revisit N.
5791 
5792   // If the VSELECT result requires splitting and the mask is provided by a
5793   // SETCC, then split both nodes and its operands before legalization. This
5794   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5795   // and enables future optimizations (e.g. min/max pattern matching on X86).
5796   if (N0.getOpcode() == ISD::SETCC) {
5797     EVT VT = N->getValueType(0);
5798 
5799     // Check if any splitting is required.
5800     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
5801         TargetLowering::TypeSplitVector)
5802       return SDValue();
5803 
5804     SDValue Lo, Hi, CCLo, CCHi, LL, LH, RL, RH;
5805     std::tie(CCLo, CCHi) = SplitVSETCC(N0.getNode(), DAG);
5806     std::tie(LL, LH) = DAG.SplitVectorOperand(N, 1);
5807     std::tie(RL, RH) = DAG.SplitVectorOperand(N, 2);
5808 
5809     Lo = DAG.getNode(N->getOpcode(), DL, LL.getValueType(), CCLo, LL, RL);
5810     Hi = DAG.getNode(N->getOpcode(), DL, LH.getValueType(), CCHi, LH, RH);
5811 
5812     // Add the new VSELECT nodes to the work list in case they need to be split
5813     // again.
5814     AddToWorklist(Lo.getNode());
5815     AddToWorklist(Hi.getNode());
5816 
5817     return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
5818   }
5819 
5820   // Fold (vselect (build_vector all_ones), N1, N2) -> N1
5821   if (ISD::isBuildVectorAllOnes(N0.getNode()))
5822     return N1;
5823   // Fold (vselect (build_vector all_zeros), N1, N2) -> N2
5824   if (ISD::isBuildVectorAllZeros(N0.getNode()))
5825     return N2;
5826 
5827   // The ConvertSelectToConcatVector function is assuming both the above
5828   // checks for (vselect (build_vector all{ones,zeros) ...) have been made
5829   // and addressed.
5830   if (N1.getOpcode() == ISD::CONCAT_VECTORS &&
5831       N2.getOpcode() == ISD::CONCAT_VECTORS &&
5832       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())) {
5833     if (SDValue CV = ConvertSelectToConcatVector(N, DAG))
5834       return CV;
5835   }
5836 
5837   return SDValue();
5838 }
5839 
5840 SDValue DAGCombiner::visitSELECT_CC(SDNode *N) {
5841   SDValue N0 = N->getOperand(0);
5842   SDValue N1 = N->getOperand(1);
5843   SDValue N2 = N->getOperand(2);
5844   SDValue N3 = N->getOperand(3);
5845   SDValue N4 = N->getOperand(4);
5846   ISD::CondCode CC = cast<CondCodeSDNode>(N4)->get();
5847 
5848   // fold select_cc lhs, rhs, x, x, cc -> x
5849   if (N2 == N3)
5850     return N2;
5851 
5852   // Determine if the condition we're dealing with is constant
5853   if (SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()), N0, N1,
5854                                   CC, SDLoc(N), false)) {
5855     AddToWorklist(SCC.getNode());
5856 
5857     if (ConstantSDNode *SCCC = dyn_cast<ConstantSDNode>(SCC.getNode())) {
5858       if (!SCCC->isNullValue())
5859         return N2;    // cond always true -> true val
5860       else
5861         return N3;    // cond always false -> false val
5862     } else if (SCC->isUndef()) {
5863       // When the condition is UNDEF, just return the first operand. This is
5864       // coherent the DAG creation, no setcc node is created in this case
5865       return N2;
5866     } else if (SCC.getOpcode() == ISD::SETCC) {
5867       // Fold to a simpler select_cc
5868       return DAG.getNode(ISD::SELECT_CC, SDLoc(N), N2.getValueType(),
5869                          SCC.getOperand(0), SCC.getOperand(1), N2, N3,
5870                          SCC.getOperand(2));
5871     }
5872   }
5873 
5874   // If we can fold this based on the true/false value, do so.
5875   if (SimplifySelectOps(N, N2, N3))
5876     return SDValue(N, 0);  // Don't revisit N.
5877 
5878   // fold select_cc into other things, such as min/max/abs
5879   return SimplifySelectCC(SDLoc(N), N0, N1, N2, N3, CC);
5880 }
5881 
5882 SDValue DAGCombiner::visitSETCC(SDNode *N) {
5883   return SimplifySetCC(N->getValueType(0), N->getOperand(0), N->getOperand(1),
5884                        cast<CondCodeSDNode>(N->getOperand(2))->get(),
5885                        SDLoc(N));
5886 }
5887 
5888 SDValue DAGCombiner::visitSETCCE(SDNode *N) {
5889   SDValue LHS = N->getOperand(0);
5890   SDValue RHS = N->getOperand(1);
5891   SDValue Carry = N->getOperand(2);
5892   SDValue Cond = N->getOperand(3);
5893 
5894   // If Carry is false, fold to a regular SETCC.
5895   if (Carry.getOpcode() == ISD::CARRY_FALSE)
5896     return DAG.getNode(ISD::SETCC, SDLoc(N), N->getVTList(), LHS, RHS, Cond);
5897 
5898   return SDValue();
5899 }
5900 
5901 /// Try to fold a sext/zext/aext dag node into a ConstantSDNode or
5902 /// a build_vector of constants.
5903 /// This function is called by the DAGCombiner when visiting sext/zext/aext
5904 /// dag nodes (see for example method DAGCombiner::visitSIGN_EXTEND).
5905 /// Vector extends are not folded if operations are legal; this is to
5906 /// avoid introducing illegal build_vector dag nodes.
5907 static SDNode *tryToFoldExtendOfConstant(SDNode *N, const TargetLowering &TLI,
5908                                          SelectionDAG &DAG, bool LegalTypes,
5909                                          bool LegalOperations) {
5910   unsigned Opcode = N->getOpcode();
5911   SDValue N0 = N->getOperand(0);
5912   EVT VT = N->getValueType(0);
5913 
5914   assert((Opcode == ISD::SIGN_EXTEND || Opcode == ISD::ZERO_EXTEND ||
5915          Opcode == ISD::ANY_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG ||
5916          Opcode == ISD::ZERO_EXTEND_VECTOR_INREG)
5917          && "Expected EXTEND dag node in input!");
5918 
5919   // fold (sext c1) -> c1
5920   // fold (zext c1) -> c1
5921   // fold (aext c1) -> c1
5922   if (isa<ConstantSDNode>(N0))
5923     return DAG.getNode(Opcode, SDLoc(N), VT, N0).getNode();
5924 
5925   // fold (sext (build_vector AllConstants) -> (build_vector AllConstants)
5926   // fold (zext (build_vector AllConstants) -> (build_vector AllConstants)
5927   // fold (aext (build_vector AllConstants) -> (build_vector AllConstants)
5928   EVT SVT = VT.getScalarType();
5929   if (!(VT.isVector() &&
5930       (!LegalTypes || (!LegalOperations && TLI.isTypeLegal(SVT))) &&
5931       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())))
5932     return nullptr;
5933 
5934   // We can fold this node into a build_vector.
5935   unsigned VTBits = SVT.getSizeInBits();
5936   unsigned EVTBits = N0->getValueType(0).getScalarSizeInBits();
5937   SmallVector<SDValue, 8> Elts;
5938   unsigned NumElts = VT.getVectorNumElements();
5939   SDLoc DL(N);
5940 
5941   for (unsigned i=0; i != NumElts; ++i) {
5942     SDValue Op = N0->getOperand(i);
5943     if (Op->isUndef()) {
5944       Elts.push_back(DAG.getUNDEF(SVT));
5945       continue;
5946     }
5947 
5948     SDLoc DL(Op);
5949     // Get the constant value and if needed trunc it to the size of the type.
5950     // Nodes like build_vector might have constants wider than the scalar type.
5951     APInt C = cast<ConstantSDNode>(Op)->getAPIntValue().zextOrTrunc(EVTBits);
5952     if (Opcode == ISD::SIGN_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG)
5953       Elts.push_back(DAG.getConstant(C.sext(VTBits), DL, SVT));
5954     else
5955       Elts.push_back(DAG.getConstant(C.zext(VTBits), DL, SVT));
5956   }
5957 
5958   return DAG.getBuildVector(VT, DL, Elts).getNode();
5959 }
5960 
5961 // ExtendUsesToFormExtLoad - Trying to extend uses of a load to enable this:
5962 // "fold ({s|z|a}ext (load x)) -> ({s|z|a}ext (truncate ({s|z|a}extload x)))"
5963 // transformation. Returns true if extension are possible and the above
5964 // mentioned transformation is profitable.
5965 static bool ExtendUsesToFormExtLoad(SDNode *N, SDValue N0,
5966                                     unsigned ExtOpc,
5967                                     SmallVectorImpl<SDNode *> &ExtendNodes,
5968                                     const TargetLowering &TLI) {
5969   bool HasCopyToRegUses = false;
5970   bool isTruncFree = TLI.isTruncateFree(N->getValueType(0), N0.getValueType());
5971   for (SDNode::use_iterator UI = N0.getNode()->use_begin(),
5972                             UE = N0.getNode()->use_end();
5973        UI != UE; ++UI) {
5974     SDNode *User = *UI;
5975     if (User == N)
5976       continue;
5977     if (UI.getUse().getResNo() != N0.getResNo())
5978       continue;
5979     // FIXME: Only extend SETCC N, N and SETCC N, c for now.
5980     if (ExtOpc != ISD::ANY_EXTEND && User->getOpcode() == ISD::SETCC) {
5981       ISD::CondCode CC = cast<CondCodeSDNode>(User->getOperand(2))->get();
5982       if (ExtOpc == ISD::ZERO_EXTEND && ISD::isSignedIntSetCC(CC))
5983         // Sign bits will be lost after a zext.
5984         return false;
5985       bool Add = false;
5986       for (unsigned i = 0; i != 2; ++i) {
5987         SDValue UseOp = User->getOperand(i);
5988         if (UseOp == N0)
5989           continue;
5990         if (!isa<ConstantSDNode>(UseOp))
5991           return false;
5992         Add = true;
5993       }
5994       if (Add)
5995         ExtendNodes.push_back(User);
5996       continue;
5997     }
5998     // If truncates aren't free and there are users we can't
5999     // extend, it isn't worthwhile.
6000     if (!isTruncFree)
6001       return false;
6002     // Remember if this value is live-out.
6003     if (User->getOpcode() == ISD::CopyToReg)
6004       HasCopyToRegUses = true;
6005   }
6006 
6007   if (HasCopyToRegUses) {
6008     bool BothLiveOut = false;
6009     for (SDNode::use_iterator UI = N->use_begin(), UE = N->use_end();
6010          UI != UE; ++UI) {
6011       SDUse &Use = UI.getUse();
6012       if (Use.getResNo() == 0 && Use.getUser()->getOpcode() == ISD::CopyToReg) {
6013         BothLiveOut = true;
6014         break;
6015       }
6016     }
6017     if (BothLiveOut)
6018       // Both unextended and extended values are live out. There had better be
6019       // a good reason for the transformation.
6020       return ExtendNodes.size();
6021   }
6022   return true;
6023 }
6024 
6025 void DAGCombiner::ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs,
6026                                   SDValue Trunc, SDValue ExtLoad,
6027                                   const SDLoc &DL, ISD::NodeType ExtType) {
6028   // Extend SetCC uses if necessary.
6029   for (unsigned i = 0, e = SetCCs.size(); i != e; ++i) {
6030     SDNode *SetCC = SetCCs[i];
6031     SmallVector<SDValue, 4> Ops;
6032 
6033     for (unsigned j = 0; j != 2; ++j) {
6034       SDValue SOp = SetCC->getOperand(j);
6035       if (SOp == Trunc)
6036         Ops.push_back(ExtLoad);
6037       else
6038         Ops.push_back(DAG.getNode(ExtType, DL, ExtLoad->getValueType(0), SOp));
6039     }
6040 
6041     Ops.push_back(SetCC->getOperand(2));
6042     CombineTo(SetCC, DAG.getNode(ISD::SETCC, DL, SetCC->getValueType(0), Ops));
6043   }
6044 }
6045 
6046 // FIXME: Bring more similar combines here, common to sext/zext (maybe aext?).
6047 SDValue DAGCombiner::CombineExtLoad(SDNode *N) {
6048   SDValue N0 = N->getOperand(0);
6049   EVT DstVT = N->getValueType(0);
6050   EVT SrcVT = N0.getValueType();
6051 
6052   assert((N->getOpcode() == ISD::SIGN_EXTEND ||
6053           N->getOpcode() == ISD::ZERO_EXTEND) &&
6054          "Unexpected node type (not an extend)!");
6055 
6056   // fold (sext (load x)) to multiple smaller sextloads; same for zext.
6057   // For example, on a target with legal v4i32, but illegal v8i32, turn:
6058   //   (v8i32 (sext (v8i16 (load x))))
6059   // into:
6060   //   (v8i32 (concat_vectors (v4i32 (sextload x)),
6061   //                          (v4i32 (sextload (x + 16)))))
6062   // Where uses of the original load, i.e.:
6063   //   (v8i16 (load x))
6064   // are replaced with:
6065   //   (v8i16 (truncate
6066   //     (v8i32 (concat_vectors (v4i32 (sextload x)),
6067   //                            (v4i32 (sextload (x + 16)))))))
6068   //
6069   // This combine is only applicable to illegal, but splittable, vectors.
6070   // All legal types, and illegal non-vector types, are handled elsewhere.
6071   // This combine is controlled by TargetLowering::isVectorLoadExtDesirable.
6072   //
6073   if (N0->getOpcode() != ISD::LOAD)
6074     return SDValue();
6075 
6076   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6077 
6078   if (!ISD::isNON_EXTLoad(LN0) || !ISD::isUNINDEXEDLoad(LN0) ||
6079       !N0.hasOneUse() || LN0->isVolatile() || !DstVT.isVector() ||
6080       !DstVT.isPow2VectorType() || !TLI.isVectorLoadExtDesirable(SDValue(N, 0)))
6081     return SDValue();
6082 
6083   SmallVector<SDNode *, 4> SetCCs;
6084   if (!ExtendUsesToFormExtLoad(N, N0, N->getOpcode(), SetCCs, TLI))
6085     return SDValue();
6086 
6087   ISD::LoadExtType ExtType =
6088       N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
6089 
6090   // Try to split the vector types to get down to legal types.
6091   EVT SplitSrcVT = SrcVT;
6092   EVT SplitDstVT = DstVT;
6093   while (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT) &&
6094          SplitSrcVT.getVectorNumElements() > 1) {
6095     SplitDstVT = DAG.GetSplitDestVTs(SplitDstVT).first;
6096     SplitSrcVT = DAG.GetSplitDestVTs(SplitSrcVT).first;
6097   }
6098 
6099   if (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT))
6100     return SDValue();
6101 
6102   SDLoc DL(N);
6103   const unsigned NumSplits =
6104       DstVT.getVectorNumElements() / SplitDstVT.getVectorNumElements();
6105   const unsigned Stride = SplitSrcVT.getStoreSize();
6106   SmallVector<SDValue, 4> Loads;
6107   SmallVector<SDValue, 4> Chains;
6108 
6109   SDValue BasePtr = LN0->getBasePtr();
6110   for (unsigned Idx = 0; Idx < NumSplits; Idx++) {
6111     const unsigned Offset = Idx * Stride;
6112     const unsigned Align = MinAlign(LN0->getAlignment(), Offset);
6113 
6114     SDValue SplitLoad = DAG.getExtLoad(
6115         ExtType, DL, SplitDstVT, LN0->getChain(), BasePtr,
6116         LN0->getPointerInfo().getWithOffset(Offset), SplitSrcVT, Align,
6117         LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
6118 
6119     BasePtr = DAG.getNode(ISD::ADD, DL, BasePtr.getValueType(), BasePtr,
6120                           DAG.getConstant(Stride, DL, BasePtr.getValueType()));
6121 
6122     Loads.push_back(SplitLoad.getValue(0));
6123     Chains.push_back(SplitLoad.getValue(1));
6124   }
6125 
6126   SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
6127   SDValue NewValue = DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Loads);
6128 
6129   CombineTo(N, NewValue);
6130 
6131   // Replace uses of the original load (before extension)
6132   // with a truncate of the concatenated sextloaded vectors.
6133   SDValue Trunc =
6134       DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), NewValue);
6135   CombineTo(N0.getNode(), Trunc, NewChain);
6136   ExtendSetCCUses(SetCCs, Trunc, NewValue, DL,
6137                   (ISD::NodeType)N->getOpcode());
6138   return SDValue(N, 0); // Return N so it doesn't get rechecked!
6139 }
6140 
6141 SDValue DAGCombiner::visitSIGN_EXTEND(SDNode *N) {
6142   SDValue N0 = N->getOperand(0);
6143   EVT VT = N->getValueType(0);
6144 
6145   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6146                                               LegalOperations))
6147     return SDValue(Res, 0);
6148 
6149   // fold (sext (sext x)) -> (sext x)
6150   // fold (sext (aext x)) -> (sext x)
6151   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
6152     return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT,
6153                        N0.getOperand(0));
6154 
6155   if (N0.getOpcode() == ISD::TRUNCATE) {
6156     // fold (sext (truncate (load x))) -> (sext (smaller load x))
6157     // fold (sext (truncate (srl (load x), c))) -> (sext (smaller load (x+c/n)))
6158     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6159       SDNode* oye = N0.getNode()->getOperand(0).getNode();
6160       if (NarrowLoad.getNode() != N0.getNode()) {
6161         CombineTo(N0.getNode(), NarrowLoad);
6162         // CombineTo deleted the truncate, if needed, but not what's under it.
6163         AddToWorklist(oye);
6164       }
6165       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6166     }
6167 
6168     // See if the value being truncated is already sign extended.  If so, just
6169     // eliminate the trunc/sext pair.
6170     SDValue Op = N0.getOperand(0);
6171     unsigned OpBits   = Op.getScalarValueSizeInBits();
6172     unsigned MidBits  = N0.getScalarValueSizeInBits();
6173     unsigned DestBits = VT.getScalarSizeInBits();
6174     unsigned NumSignBits = DAG.ComputeNumSignBits(Op);
6175 
6176     if (OpBits == DestBits) {
6177       // Op is i32, Mid is i8, and Dest is i32.  If Op has more than 24 sign
6178       // bits, it is already ready.
6179       if (NumSignBits > DestBits-MidBits)
6180         return Op;
6181     } else if (OpBits < DestBits) {
6182       // Op is i32, Mid is i8, and Dest is i64.  If Op has more than 24 sign
6183       // bits, just sext from i32.
6184       if (NumSignBits > OpBits-MidBits)
6185         return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, Op);
6186     } else {
6187       // Op is i64, Mid is i8, and Dest is i32.  If Op has more than 56 sign
6188       // bits, just truncate to i32.
6189       if (NumSignBits > OpBits-MidBits)
6190         return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6191     }
6192 
6193     // fold (sext (truncate x)) -> (sextinreg x).
6194     if (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG,
6195                                                  N0.getValueType())) {
6196       if (OpBits < DestBits)
6197         Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N0), VT, Op);
6198       else if (OpBits > DestBits)
6199         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), VT, Op);
6200       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, Op,
6201                          DAG.getValueType(N0.getValueType()));
6202     }
6203   }
6204 
6205   // fold (sext (load x)) -> (sext (truncate (sextload x)))
6206   // Only generate vector extloads when 1) they're legal, and 2) they are
6207   // deemed desirable by the target.
6208   if (ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6209       ((!LegalOperations && !VT.isVector() &&
6210         !cast<LoadSDNode>(N0)->isVolatile()) ||
6211        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, N0.getValueType()))) {
6212     bool DoXform = true;
6213     SmallVector<SDNode*, 4> SetCCs;
6214     if (!N0.hasOneUse())
6215       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::SIGN_EXTEND, SetCCs, TLI);
6216     if (VT.isVector())
6217       DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0));
6218     if (DoXform) {
6219       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6220       SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
6221                                        LN0->getChain(),
6222                                        LN0->getBasePtr(), N0.getValueType(),
6223                                        LN0->getMemOperand());
6224       CombineTo(N, ExtLoad);
6225       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6226                                   N0.getValueType(), ExtLoad);
6227       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6228       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
6229                       ISD::SIGN_EXTEND);
6230       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6231     }
6232   }
6233 
6234   // fold (sext (load x)) to multiple smaller sextloads.
6235   // Only on illegal but splittable vectors.
6236   if (SDValue ExtLoad = CombineExtLoad(N))
6237     return ExtLoad;
6238 
6239   // fold (sext (sextload x)) -> (sext (truncate (sextload x)))
6240   // fold (sext ( extload x)) -> (sext (truncate (sextload x)))
6241   if ((ISD::isSEXTLoad(N0.getNode()) || ISD::isEXTLoad(N0.getNode())) &&
6242       ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) {
6243     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6244     EVT MemVT = LN0->getMemoryVT();
6245     if ((!LegalOperations && !LN0->isVolatile()) ||
6246         TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, MemVT)) {
6247       SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
6248                                        LN0->getChain(),
6249                                        LN0->getBasePtr(), MemVT,
6250                                        LN0->getMemOperand());
6251       CombineTo(N, ExtLoad);
6252       CombineTo(N0.getNode(),
6253                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6254                             N0.getValueType(), ExtLoad),
6255                 ExtLoad.getValue(1));
6256       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6257     }
6258   }
6259 
6260   // fold (sext (and/or/xor (load x), cst)) ->
6261   //      (and/or/xor (sextload x), (sext cst))
6262   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
6263        N0.getOpcode() == ISD::XOR) &&
6264       isa<LoadSDNode>(N0.getOperand(0)) &&
6265       N0.getOperand(1).getOpcode() == ISD::Constant &&
6266       TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, N0.getValueType()) &&
6267       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
6268     LoadSDNode *LN0 = cast<LoadSDNode>(N0.getOperand(0));
6269     if (LN0->getExtensionType() != ISD::ZEXTLOAD && LN0->isUnindexed()) {
6270       bool DoXform = true;
6271       SmallVector<SDNode*, 4> SetCCs;
6272       if (!N0.hasOneUse())
6273         DoXform = ExtendUsesToFormExtLoad(N, N0.getOperand(0), ISD::SIGN_EXTEND,
6274                                           SetCCs, TLI);
6275       if (DoXform) {
6276         SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(LN0), VT,
6277                                          LN0->getChain(), LN0->getBasePtr(),
6278                                          LN0->getMemoryVT(),
6279                                          LN0->getMemOperand());
6280         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6281         Mask = Mask.sext(VT.getSizeInBits());
6282         SDLoc DL(N);
6283         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
6284                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
6285         SDValue Trunc = DAG.getNode(ISD::TRUNCATE,
6286                                     SDLoc(N0.getOperand(0)),
6287                                     N0.getOperand(0).getValueType(), ExtLoad);
6288         CombineTo(N, And);
6289         CombineTo(N0.getOperand(0).getNode(), Trunc, ExtLoad.getValue(1));
6290         ExtendSetCCUses(SetCCs, Trunc, ExtLoad, DL,
6291                         ISD::SIGN_EXTEND);
6292         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6293       }
6294     }
6295   }
6296 
6297   if (N0.getOpcode() == ISD::SETCC) {
6298     EVT N0VT = N0.getOperand(0).getValueType();
6299     // sext(setcc) -> sext_in_reg(vsetcc) for vectors.
6300     // Only do this before legalize for now.
6301     if (VT.isVector() && !LegalOperations &&
6302         TLI.getBooleanContents(N0VT) ==
6303             TargetLowering::ZeroOrNegativeOneBooleanContent) {
6304       // On some architectures (such as SSE/NEON/etc) the SETCC result type is
6305       // of the same size as the compared operands. Only optimize sext(setcc())
6306       // if this is the case.
6307       EVT SVT = getSetCCResultType(N0VT);
6308 
6309       // We know that the # elements of the results is the same as the
6310       // # elements of the compare (and the # elements of the compare result
6311       // for that matter).  Check to see that they are the same size.  If so,
6312       // we know that the element size of the sext'd result matches the
6313       // element size of the compare operands.
6314       if (VT.getSizeInBits() == SVT.getSizeInBits())
6315         return DAG.getSetCC(SDLoc(N), VT, N0.getOperand(0),
6316                              N0.getOperand(1),
6317                              cast<CondCodeSDNode>(N0.getOperand(2))->get());
6318 
6319       // If the desired elements are smaller or larger than the source
6320       // elements we can use a matching integer vector type and then
6321       // truncate/sign extend
6322       EVT MatchingVectorType = N0VT.changeVectorElementTypeToInteger();
6323       if (SVT == MatchingVectorType) {
6324         SDValue VsetCC = DAG.getSetCC(SDLoc(N), MatchingVectorType,
6325                                N0.getOperand(0), N0.getOperand(1),
6326                                cast<CondCodeSDNode>(N0.getOperand(2))->get());
6327         return DAG.getSExtOrTrunc(VsetCC, SDLoc(N), VT);
6328       }
6329     }
6330 
6331     // sext(setcc x, y, cc) -> (select (setcc x, y, cc), T, 0)
6332     // Here, T can be 1 or -1, depending on the type of the setcc and
6333     // getBooleanContents().
6334     unsigned SetCCWidth = N0.getScalarValueSizeInBits();
6335 
6336     SDLoc DL(N);
6337     // To determine the "true" side of the select, we need to know the high bit
6338     // of the value returned by the setcc if it evaluates to true.
6339     // If the type of the setcc is i1, then the true case of the select is just
6340     // sext(i1 1), that is, -1.
6341     // If the type of the setcc is larger (say, i8) then the value of the high
6342     // bit depends on getBooleanContents(). So, ask TLI for a real "true" value
6343     // of the appropriate width.
6344     SDValue ExtTrueVal =
6345         (SetCCWidth == 1)
6346             ? DAG.getConstant(APInt::getAllOnesValue(VT.getScalarSizeInBits()),
6347                               DL, VT)
6348             : TLI.getConstTrueVal(DAG, VT, DL);
6349 
6350     if (SDValue SCC = SimplifySelectCC(
6351             DL, N0.getOperand(0), N0.getOperand(1), ExtTrueVal,
6352             DAG.getConstant(0, DL, VT),
6353             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
6354       return SCC;
6355 
6356     if (!VT.isVector()) {
6357       EVT SetCCVT = getSetCCResultType(N0.getOperand(0).getValueType());
6358       if (!LegalOperations ||
6359           TLI.isOperationLegal(ISD::SETCC, N0.getOperand(0).getValueType())) {
6360         SDLoc DL(N);
6361         ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
6362         SDValue SetCC =
6363             DAG.getSetCC(DL, SetCCVT, N0.getOperand(0), N0.getOperand(1), CC);
6364         return DAG.getSelect(DL, VT, SetCC, ExtTrueVal,
6365                              DAG.getConstant(0, DL, VT));
6366       }
6367     }
6368   }
6369 
6370   // fold (sext x) -> (zext x) if the sign bit is known zero.
6371   if ((!LegalOperations || TLI.isOperationLegal(ISD::ZERO_EXTEND, VT)) &&
6372       DAG.SignBitIsZero(N0))
6373     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, N0);
6374 
6375   return SDValue();
6376 }
6377 
6378 // isTruncateOf - If N is a truncate of some other value, return true, record
6379 // the value being truncated in Op and which of Op's bits are zero in KnownZero.
6380 // This function computes KnownZero to avoid a duplicated call to
6381 // computeKnownBits in the caller.
6382 static bool isTruncateOf(SelectionDAG &DAG, SDValue N, SDValue &Op,
6383                          APInt &KnownZero) {
6384   APInt KnownOne;
6385   if (N->getOpcode() == ISD::TRUNCATE) {
6386     Op = N->getOperand(0);
6387     DAG.computeKnownBits(Op, KnownZero, KnownOne);
6388     return true;
6389   }
6390 
6391   if (N->getOpcode() != ISD::SETCC || N->getValueType(0) != MVT::i1 ||
6392       cast<CondCodeSDNode>(N->getOperand(2))->get() != ISD::SETNE)
6393     return false;
6394 
6395   SDValue Op0 = N->getOperand(0);
6396   SDValue Op1 = N->getOperand(1);
6397   assert(Op0.getValueType() == Op1.getValueType());
6398 
6399   if (isNullConstant(Op0))
6400     Op = Op1;
6401   else if (isNullConstant(Op1))
6402     Op = Op0;
6403   else
6404     return false;
6405 
6406   DAG.computeKnownBits(Op, KnownZero, KnownOne);
6407 
6408   if (!(KnownZero | APInt(Op.getValueSizeInBits(), 1)).isAllOnesValue())
6409     return false;
6410 
6411   return true;
6412 }
6413 
6414 SDValue DAGCombiner::visitZERO_EXTEND(SDNode *N) {
6415   SDValue N0 = N->getOperand(0);
6416   EVT VT = N->getValueType(0);
6417 
6418   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6419                                               LegalOperations))
6420     return SDValue(Res, 0);
6421 
6422   // fold (zext (zext x)) -> (zext x)
6423   // fold (zext (aext x)) -> (zext x)
6424   if (N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
6425     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT,
6426                        N0.getOperand(0));
6427 
6428   // fold (zext (truncate x)) -> (zext x) or
6429   //      (zext (truncate x)) -> (truncate x)
6430   // This is valid when the truncated bits of x are already zero.
6431   // FIXME: We should extend this to work for vectors too.
6432   SDValue Op;
6433   APInt KnownZero;
6434   if (!VT.isVector() && isTruncateOf(DAG, N0, Op, KnownZero)) {
6435     APInt TruncatedBits =
6436       (Op.getValueSizeInBits() == N0.getValueSizeInBits()) ?
6437       APInt(Op.getValueSizeInBits(), 0) :
6438       APInt::getBitsSet(Op.getValueSizeInBits(),
6439                         N0.getValueSizeInBits(),
6440                         std::min(Op.getValueSizeInBits(),
6441                                  VT.getSizeInBits()));
6442     if (TruncatedBits == (KnownZero & TruncatedBits)) {
6443       if (VT.bitsGT(Op.getValueType()))
6444         return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, Op);
6445       if (VT.bitsLT(Op.getValueType()))
6446         return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6447 
6448       return Op;
6449     }
6450   }
6451 
6452   // fold (zext (truncate (load x))) -> (zext (smaller load x))
6453   // fold (zext (truncate (srl (load x), c))) -> (zext (small load (x+c/n)))
6454   if (N0.getOpcode() == ISD::TRUNCATE) {
6455     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6456       SDNode* oye = N0.getNode()->getOperand(0).getNode();
6457       if (NarrowLoad.getNode() != N0.getNode()) {
6458         CombineTo(N0.getNode(), NarrowLoad);
6459         // CombineTo deleted the truncate, if needed, but not what's under it.
6460         AddToWorklist(oye);
6461       }
6462       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6463     }
6464   }
6465 
6466   // fold (zext (truncate x)) -> (and x, mask)
6467   if (N0.getOpcode() == ISD::TRUNCATE) {
6468     // fold (zext (truncate (load x))) -> (zext (smaller load x))
6469     // fold (zext (truncate (srl (load x), c))) -> (zext (smaller load (x+c/n)))
6470     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6471       SDNode *oye = N0.getNode()->getOperand(0).getNode();
6472       if (NarrowLoad.getNode() != N0.getNode()) {
6473         CombineTo(N0.getNode(), NarrowLoad);
6474         // CombineTo deleted the truncate, if needed, but not what's under it.
6475         AddToWorklist(oye);
6476       }
6477       return SDValue(N, 0); // Return N so it doesn't get rechecked!
6478     }
6479 
6480     EVT SrcVT = N0.getOperand(0).getValueType();
6481     EVT MinVT = N0.getValueType();
6482 
6483     // Try to mask before the extension to avoid having to generate a larger mask,
6484     // possibly over several sub-vectors.
6485     if (SrcVT.bitsLT(VT)) {
6486       if (!LegalOperations || (TLI.isOperationLegal(ISD::AND, SrcVT) &&
6487                                TLI.isOperationLegal(ISD::ZERO_EXTEND, VT))) {
6488         SDValue Op = N0.getOperand(0);
6489         Op = DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
6490         AddToWorklist(Op.getNode());
6491         return DAG.getZExtOrTrunc(Op, SDLoc(N), VT);
6492       }
6493     }
6494 
6495     if (!LegalOperations || TLI.isOperationLegal(ISD::AND, VT)) {
6496       SDValue Op = N0.getOperand(0);
6497       if (SrcVT.bitsLT(VT)) {
6498         Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, Op);
6499         AddToWorklist(Op.getNode());
6500       } else if (SrcVT.bitsGT(VT)) {
6501         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6502         AddToWorklist(Op.getNode());
6503       }
6504       return DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
6505     }
6506   }
6507 
6508   // Fold (zext (and (trunc x), cst)) -> (and x, cst),
6509   // if either of the casts is not free.
6510   if (N0.getOpcode() == ISD::AND &&
6511       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
6512       N0.getOperand(1).getOpcode() == ISD::Constant &&
6513       (!TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
6514                            N0.getValueType()) ||
6515        !TLI.isZExtFree(N0.getValueType(), VT))) {
6516     SDValue X = N0.getOperand(0).getOperand(0);
6517     if (X.getValueType().bitsLT(VT)) {
6518       X = DAG.getNode(ISD::ANY_EXTEND, SDLoc(X), VT, X);
6519     } else if (X.getValueType().bitsGT(VT)) {
6520       X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
6521     }
6522     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6523     Mask = Mask.zext(VT.getSizeInBits());
6524     SDLoc DL(N);
6525     return DAG.getNode(ISD::AND, DL, VT,
6526                        X, DAG.getConstant(Mask, DL, VT));
6527   }
6528 
6529   // fold (zext (load x)) -> (zext (truncate (zextload x)))
6530   // Only generate vector extloads when 1) they're legal, and 2) they are
6531   // deemed desirable by the target.
6532   if (ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6533       ((!LegalOperations && !VT.isVector() &&
6534         !cast<LoadSDNode>(N0)->isVolatile()) ||
6535        TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, N0.getValueType()))) {
6536     bool DoXform = true;
6537     SmallVector<SDNode*, 4> SetCCs;
6538     if (!N0.hasOneUse())
6539       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::ZERO_EXTEND, SetCCs, TLI);
6540     if (VT.isVector())
6541       DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0));
6542     if (DoXform) {
6543       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6544       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N), VT,
6545                                        LN0->getChain(),
6546                                        LN0->getBasePtr(), N0.getValueType(),
6547                                        LN0->getMemOperand());
6548       CombineTo(N, ExtLoad);
6549       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6550                                   N0.getValueType(), ExtLoad);
6551       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6552 
6553       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
6554                       ISD::ZERO_EXTEND);
6555       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6556     }
6557   }
6558 
6559   // fold (zext (load x)) to multiple smaller zextloads.
6560   // Only on illegal but splittable vectors.
6561   if (SDValue ExtLoad = CombineExtLoad(N))
6562     return ExtLoad;
6563 
6564   // fold (zext (and/or/xor (load x), cst)) ->
6565   //      (and/or/xor (zextload x), (zext cst))
6566   // Unless (and (load x) cst) will match as a zextload already and has
6567   // additional users.
6568   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
6569        N0.getOpcode() == ISD::XOR) &&
6570       isa<LoadSDNode>(N0.getOperand(0)) &&
6571       N0.getOperand(1).getOpcode() == ISD::Constant &&
6572       TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, N0.getValueType()) &&
6573       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
6574     LoadSDNode *LN0 = cast<LoadSDNode>(N0.getOperand(0));
6575     if (LN0->getExtensionType() != ISD::SEXTLOAD && LN0->isUnindexed()) {
6576       bool DoXform = true;
6577       SmallVector<SDNode*, 4> SetCCs;
6578       if (!N0.hasOneUse()) {
6579         if (N0.getOpcode() == ISD::AND) {
6580           auto *AndC = cast<ConstantSDNode>(N0.getOperand(1));
6581           auto NarrowLoad = false;
6582           EVT LoadResultTy = AndC->getValueType(0);
6583           EVT ExtVT, LoadedVT;
6584           if (isAndLoadExtLoad(AndC, LN0, LoadResultTy, ExtVT, LoadedVT,
6585                                NarrowLoad))
6586             DoXform = false;
6587         }
6588         if (DoXform)
6589           DoXform = ExtendUsesToFormExtLoad(N, N0.getOperand(0),
6590                                             ISD::ZERO_EXTEND, SetCCs, TLI);
6591       }
6592       if (DoXform) {
6593         SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN0), VT,
6594                                          LN0->getChain(), LN0->getBasePtr(),
6595                                          LN0->getMemoryVT(),
6596                                          LN0->getMemOperand());
6597         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6598         Mask = Mask.zext(VT.getSizeInBits());
6599         SDLoc DL(N);
6600         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
6601                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
6602         SDValue Trunc = DAG.getNode(ISD::TRUNCATE,
6603                                     SDLoc(N0.getOperand(0)),
6604                                     N0.getOperand(0).getValueType(), ExtLoad);
6605         CombineTo(N, And);
6606         CombineTo(N0.getOperand(0).getNode(), Trunc, ExtLoad.getValue(1));
6607         ExtendSetCCUses(SetCCs, Trunc, ExtLoad, DL,
6608                         ISD::ZERO_EXTEND);
6609         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6610       }
6611     }
6612   }
6613 
6614   // fold (zext (zextload x)) -> (zext (truncate (zextload x)))
6615   // fold (zext ( extload x)) -> (zext (truncate (zextload x)))
6616   if ((ISD::isZEXTLoad(N0.getNode()) || ISD::isEXTLoad(N0.getNode())) &&
6617       ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) {
6618     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6619     EVT MemVT = LN0->getMemoryVT();
6620     if ((!LegalOperations && !LN0->isVolatile()) ||
6621         TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT)) {
6622       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N), VT,
6623                                        LN0->getChain(),
6624                                        LN0->getBasePtr(), MemVT,
6625                                        LN0->getMemOperand());
6626       CombineTo(N, ExtLoad);
6627       CombineTo(N0.getNode(),
6628                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(),
6629                             ExtLoad),
6630                 ExtLoad.getValue(1));
6631       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6632     }
6633   }
6634 
6635   if (N0.getOpcode() == ISD::SETCC) {
6636     // Only do this before legalize for now.
6637     if (!LegalOperations && VT.isVector() &&
6638         N0.getValueType().getVectorElementType() == MVT::i1) {
6639       EVT N00VT = N0.getOperand(0).getValueType();
6640       if (getSetCCResultType(N00VT) == N0.getValueType())
6641         return SDValue();
6642 
6643       // We know that the # elements of the results is the same as the #
6644       // elements of the compare (and the # elements of the compare result for
6645       // that matter). Check to see that they are the same size. If so, we know
6646       // that the element size of the sext'd result matches the element size of
6647       // the compare operands.
6648       SDLoc DL(N);
6649       SDValue VecOnes = DAG.getConstant(1, DL, VT);
6650       if (VT.getSizeInBits() == N00VT.getSizeInBits()) {
6651         // zext(setcc) -> (and (vsetcc), (1, 1, ...) for vectors.
6652         SDValue VSetCC = DAG.getNode(ISD::SETCC, DL, VT, N0.getOperand(0),
6653                                      N0.getOperand(1), N0.getOperand(2));
6654         return DAG.getNode(ISD::AND, DL, VT, VSetCC, VecOnes);
6655       }
6656 
6657       // If the desired elements are smaller or larger than the source
6658       // elements we can use a matching integer vector type and then
6659       // truncate/sign extend.
6660       EVT MatchingElementType = EVT::getIntegerVT(
6661           *DAG.getContext(), N00VT.getScalarSizeInBits());
6662       EVT MatchingVectorType = EVT::getVectorVT(
6663           *DAG.getContext(), MatchingElementType, N00VT.getVectorNumElements());
6664       SDValue VsetCC =
6665           DAG.getNode(ISD::SETCC, DL, MatchingVectorType, N0.getOperand(0),
6666                       N0.getOperand(1), N0.getOperand(2));
6667       return DAG.getNode(ISD::AND, DL, VT, DAG.getSExtOrTrunc(VsetCC, DL, VT),
6668                          VecOnes);
6669     }
6670 
6671     // zext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
6672     SDLoc DL(N);
6673     if (SDValue SCC = SimplifySelectCC(
6674             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
6675             DAG.getConstant(0, DL, VT),
6676             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
6677       return SCC;
6678   }
6679 
6680   // (zext (shl (zext x), cst)) -> (shl (zext x), cst)
6681   if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL) &&
6682       isa<ConstantSDNode>(N0.getOperand(1)) &&
6683       N0.getOperand(0).getOpcode() == ISD::ZERO_EXTEND &&
6684       N0.hasOneUse()) {
6685     SDValue ShAmt = N0.getOperand(1);
6686     unsigned ShAmtVal = cast<ConstantSDNode>(ShAmt)->getZExtValue();
6687     if (N0.getOpcode() == ISD::SHL) {
6688       SDValue InnerZExt = N0.getOperand(0);
6689       // If the original shl may be shifting out bits, do not perform this
6690       // transformation.
6691       unsigned KnownZeroBits = InnerZExt.getValueSizeInBits() -
6692         InnerZExt.getOperand(0).getValueSizeInBits();
6693       if (ShAmtVal > KnownZeroBits)
6694         return SDValue();
6695     }
6696 
6697     SDLoc DL(N);
6698 
6699     // Ensure that the shift amount is wide enough for the shifted value.
6700     if (VT.getSizeInBits() >= 256)
6701       ShAmt = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, ShAmt);
6702 
6703     return DAG.getNode(N0.getOpcode(), DL, VT,
6704                        DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0)),
6705                        ShAmt);
6706   }
6707 
6708   return SDValue();
6709 }
6710 
6711 SDValue DAGCombiner::visitANY_EXTEND(SDNode *N) {
6712   SDValue N0 = N->getOperand(0);
6713   EVT VT = N->getValueType(0);
6714 
6715   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6716                                               LegalOperations))
6717     return SDValue(Res, 0);
6718 
6719   // fold (aext (aext x)) -> (aext x)
6720   // fold (aext (zext x)) -> (zext x)
6721   // fold (aext (sext x)) -> (sext x)
6722   if (N0.getOpcode() == ISD::ANY_EXTEND  ||
6723       N0.getOpcode() == ISD::ZERO_EXTEND ||
6724       N0.getOpcode() == ISD::SIGN_EXTEND)
6725     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
6726 
6727   // fold (aext (truncate (load x))) -> (aext (smaller load x))
6728   // fold (aext (truncate (srl (load x), c))) -> (aext (small load (x+c/n)))
6729   if (N0.getOpcode() == ISD::TRUNCATE) {
6730     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6731       SDNode* oye = N0.getNode()->getOperand(0).getNode();
6732       if (NarrowLoad.getNode() != N0.getNode()) {
6733         CombineTo(N0.getNode(), NarrowLoad);
6734         // CombineTo deleted the truncate, if needed, but not what's under it.
6735         AddToWorklist(oye);
6736       }
6737       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6738     }
6739   }
6740 
6741   // fold (aext (truncate x))
6742   if (N0.getOpcode() == ISD::TRUNCATE) {
6743     SDValue TruncOp = N0.getOperand(0);
6744     if (TruncOp.getValueType() == VT)
6745       return TruncOp; // x iff x size == zext size.
6746     if (TruncOp.getValueType().bitsGT(VT))
6747       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, TruncOp);
6748     return DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, TruncOp);
6749   }
6750 
6751   // Fold (aext (and (trunc x), cst)) -> (and x, cst)
6752   // if the trunc is not free.
6753   if (N0.getOpcode() == ISD::AND &&
6754       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
6755       N0.getOperand(1).getOpcode() == ISD::Constant &&
6756       !TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
6757                           N0.getValueType())) {
6758     SDValue X = N0.getOperand(0).getOperand(0);
6759     if (X.getValueType().bitsLT(VT)) {
6760       X = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, X);
6761     } else if (X.getValueType().bitsGT(VT)) {
6762       X = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, X);
6763     }
6764     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6765     Mask = Mask.zext(VT.getSizeInBits());
6766     SDLoc DL(N);
6767     return DAG.getNode(ISD::AND, DL, VT,
6768                        X, DAG.getConstant(Mask, DL, VT));
6769   }
6770 
6771   // fold (aext (load x)) -> (aext (truncate (extload x)))
6772   // None of the supported targets knows how to perform load and any_ext
6773   // on vectors in one instruction.  We only perform this transformation on
6774   // scalars.
6775   if (ISD::isNON_EXTLoad(N0.getNode()) && !VT.isVector() &&
6776       ISD::isUNINDEXEDLoad(N0.getNode()) &&
6777       TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
6778     bool DoXform = true;
6779     SmallVector<SDNode*, 4> SetCCs;
6780     if (!N0.hasOneUse())
6781       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::ANY_EXTEND, SetCCs, TLI);
6782     if (DoXform) {
6783       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6784       SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
6785                                        LN0->getChain(),
6786                                        LN0->getBasePtr(), N0.getValueType(),
6787                                        LN0->getMemOperand());
6788       CombineTo(N, ExtLoad);
6789       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6790                                   N0.getValueType(), ExtLoad);
6791       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6792       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
6793                       ISD::ANY_EXTEND);
6794       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6795     }
6796   }
6797 
6798   // fold (aext (zextload x)) -> (aext (truncate (zextload x)))
6799   // fold (aext (sextload x)) -> (aext (truncate (sextload x)))
6800   // fold (aext ( extload x)) -> (aext (truncate (extload  x)))
6801   if (N0.getOpcode() == ISD::LOAD &&
6802       !ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6803       N0.hasOneUse()) {
6804     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6805     ISD::LoadExtType ExtType = LN0->getExtensionType();
6806     EVT MemVT = LN0->getMemoryVT();
6807     if (!LegalOperations || TLI.isLoadExtLegal(ExtType, VT, MemVT)) {
6808       SDValue ExtLoad = DAG.getExtLoad(ExtType, SDLoc(N),
6809                                        VT, LN0->getChain(), LN0->getBasePtr(),
6810                                        MemVT, LN0->getMemOperand());
6811       CombineTo(N, ExtLoad);
6812       CombineTo(N0.getNode(),
6813                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6814                             N0.getValueType(), ExtLoad),
6815                 ExtLoad.getValue(1));
6816       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6817     }
6818   }
6819 
6820   if (N0.getOpcode() == ISD::SETCC) {
6821     // For vectors:
6822     // aext(setcc) -> vsetcc
6823     // aext(setcc) -> truncate(vsetcc)
6824     // aext(setcc) -> aext(vsetcc)
6825     // Only do this before legalize for now.
6826     if (VT.isVector() && !LegalOperations) {
6827       EVT N0VT = N0.getOperand(0).getValueType();
6828         // We know that the # elements of the results is the same as the
6829         // # elements of the compare (and the # elements of the compare result
6830         // for that matter).  Check to see that they are the same size.  If so,
6831         // we know that the element size of the sext'd result matches the
6832         // element size of the compare operands.
6833       if (VT.getSizeInBits() == N0VT.getSizeInBits())
6834         return DAG.getSetCC(SDLoc(N), VT, N0.getOperand(0),
6835                              N0.getOperand(1),
6836                              cast<CondCodeSDNode>(N0.getOperand(2))->get());
6837       // If the desired elements are smaller or larger than the source
6838       // elements we can use a matching integer vector type and then
6839       // truncate/any extend
6840       else {
6841         EVT MatchingVectorType = N0VT.changeVectorElementTypeToInteger();
6842         SDValue VsetCC =
6843           DAG.getSetCC(SDLoc(N), MatchingVectorType, N0.getOperand(0),
6844                         N0.getOperand(1),
6845                         cast<CondCodeSDNode>(N0.getOperand(2))->get());
6846         return DAG.getAnyExtOrTrunc(VsetCC, SDLoc(N), VT);
6847       }
6848     }
6849 
6850     // aext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
6851     SDLoc DL(N);
6852     if (SDValue SCC = SimplifySelectCC(
6853             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
6854             DAG.getConstant(0, DL, VT),
6855             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
6856       return SCC;
6857   }
6858 
6859   return SDValue();
6860 }
6861 
6862 /// See if the specified operand can be simplified with the knowledge that only
6863 /// the bits specified by Mask are used.  If so, return the simpler operand,
6864 /// otherwise return a null SDValue.
6865 SDValue DAGCombiner::GetDemandedBits(SDValue V, const APInt &Mask) {
6866   switch (V.getOpcode()) {
6867   default: break;
6868   case ISD::Constant: {
6869     const ConstantSDNode *CV = cast<ConstantSDNode>(V.getNode());
6870     assert(CV && "Const value should be ConstSDNode.");
6871     const APInt &CVal = CV->getAPIntValue();
6872     APInt NewVal = CVal & Mask;
6873     if (NewVal != CVal)
6874       return DAG.getConstant(NewVal, SDLoc(V), V.getValueType());
6875     break;
6876   }
6877   case ISD::OR:
6878   case ISD::XOR:
6879     // If the LHS or RHS don't contribute bits to the or, drop them.
6880     if (DAG.MaskedValueIsZero(V.getOperand(0), Mask))
6881       return V.getOperand(1);
6882     if (DAG.MaskedValueIsZero(V.getOperand(1), Mask))
6883       return V.getOperand(0);
6884     break;
6885   case ISD::SRL:
6886     // Only look at single-use SRLs.
6887     if (!V.getNode()->hasOneUse())
6888       break;
6889     if (ConstantSDNode *RHSC = getAsNonOpaqueConstant(V.getOperand(1))) {
6890       // See if we can recursively simplify the LHS.
6891       unsigned Amt = RHSC->getZExtValue();
6892 
6893       // Watch out for shift count overflow though.
6894       if (Amt >= Mask.getBitWidth()) break;
6895       APInt NewMask = Mask << Amt;
6896       if (SDValue SimplifyLHS = GetDemandedBits(V.getOperand(0), NewMask))
6897         return DAG.getNode(ISD::SRL, SDLoc(V), V.getValueType(),
6898                            SimplifyLHS, V.getOperand(1));
6899     }
6900   }
6901   return SDValue();
6902 }
6903 
6904 /// If the result of a wider load is shifted to right of N  bits and then
6905 /// truncated to a narrower type and where N is a multiple of number of bits of
6906 /// the narrower type, transform it to a narrower load from address + N / num of
6907 /// bits of new type. If the result is to be extended, also fold the extension
6908 /// to form a extending load.
6909 SDValue DAGCombiner::ReduceLoadWidth(SDNode *N) {
6910   unsigned Opc = N->getOpcode();
6911 
6912   ISD::LoadExtType ExtType = ISD::NON_EXTLOAD;
6913   SDValue N0 = N->getOperand(0);
6914   EVT VT = N->getValueType(0);
6915   EVT ExtVT = VT;
6916 
6917   // This transformation isn't valid for vector loads.
6918   if (VT.isVector())
6919     return SDValue();
6920 
6921   // Special case: SIGN_EXTEND_INREG is basically truncating to ExtVT then
6922   // extended to VT.
6923   if (Opc == ISD::SIGN_EXTEND_INREG) {
6924     ExtType = ISD::SEXTLOAD;
6925     ExtVT = cast<VTSDNode>(N->getOperand(1))->getVT();
6926   } else if (Opc == ISD::SRL) {
6927     // Another special-case: SRL is basically zero-extending a narrower value.
6928     ExtType = ISD::ZEXTLOAD;
6929     N0 = SDValue(N, 0);
6930     ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
6931     if (!N01) return SDValue();
6932     ExtVT = EVT::getIntegerVT(*DAG.getContext(),
6933                               VT.getSizeInBits() - N01->getZExtValue());
6934   }
6935   if (LegalOperations && !TLI.isLoadExtLegal(ExtType, VT, ExtVT))
6936     return SDValue();
6937 
6938   unsigned EVTBits = ExtVT.getSizeInBits();
6939 
6940   // Do not generate loads of non-round integer types since these can
6941   // be expensive (and would be wrong if the type is not byte sized).
6942   if (!ExtVT.isRound())
6943     return SDValue();
6944 
6945   unsigned ShAmt = 0;
6946   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
6947     if (ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
6948       ShAmt = N01->getZExtValue();
6949       // Is the shift amount a multiple of size of VT?
6950       if ((ShAmt & (EVTBits-1)) == 0) {
6951         N0 = N0.getOperand(0);
6952         // Is the load width a multiple of size of VT?
6953         if ((N0.getValueSizeInBits() & (EVTBits-1)) != 0)
6954           return SDValue();
6955       }
6956 
6957       // At this point, we must have a load or else we can't do the transform.
6958       if (!isa<LoadSDNode>(N0)) return SDValue();
6959 
6960       // Because a SRL must be assumed to *need* to zero-extend the high bits
6961       // (as opposed to anyext the high bits), we can't combine the zextload
6962       // lowering of SRL and an sextload.
6963       if (cast<LoadSDNode>(N0)->getExtensionType() == ISD::SEXTLOAD)
6964         return SDValue();
6965 
6966       // If the shift amount is larger than the input type then we're not
6967       // accessing any of the loaded bytes.  If the load was a zextload/extload
6968       // then the result of the shift+trunc is zero/undef (handled elsewhere).
6969       if (ShAmt >= cast<LoadSDNode>(N0)->getMemoryVT().getSizeInBits())
6970         return SDValue();
6971     }
6972   }
6973 
6974   // If the load is shifted left (and the result isn't shifted back right),
6975   // we can fold the truncate through the shift.
6976   unsigned ShLeftAmt = 0;
6977   if (ShAmt == 0 && N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
6978       ExtVT == VT && TLI.isNarrowingProfitable(N0.getValueType(), VT)) {
6979     if (ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
6980       ShLeftAmt = N01->getZExtValue();
6981       N0 = N0.getOperand(0);
6982     }
6983   }
6984 
6985   // If we haven't found a load, we can't narrow it.  Don't transform one with
6986   // multiple uses, this would require adding a new load.
6987   if (!isa<LoadSDNode>(N0) || !N0.hasOneUse())
6988     return SDValue();
6989 
6990   // Don't change the width of a volatile load.
6991   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6992   if (LN0->isVolatile())
6993     return SDValue();
6994 
6995   // Verify that we are actually reducing a load width here.
6996   if (LN0->getMemoryVT().getSizeInBits() < EVTBits)
6997     return SDValue();
6998 
6999   // For the transform to be legal, the load must produce only two values
7000   // (the value loaded and the chain).  Don't transform a pre-increment
7001   // load, for example, which produces an extra value.  Otherwise the
7002   // transformation is not equivalent, and the downstream logic to replace
7003   // uses gets things wrong.
7004   if (LN0->getNumValues() > 2)
7005     return SDValue();
7006 
7007   // If the load that we're shrinking is an extload and we're not just
7008   // discarding the extension we can't simply shrink the load. Bail.
7009   // TODO: It would be possible to merge the extensions in some cases.
7010   if (LN0->getExtensionType() != ISD::NON_EXTLOAD &&
7011       LN0->getMemoryVT().getSizeInBits() < ExtVT.getSizeInBits() + ShAmt)
7012     return SDValue();
7013 
7014   if (!TLI.shouldReduceLoadWidth(LN0, ExtType, ExtVT))
7015     return SDValue();
7016 
7017   EVT PtrType = N0.getOperand(1).getValueType();
7018 
7019   if (PtrType == MVT::Untyped || PtrType.isExtended())
7020     // It's not possible to generate a constant of extended or untyped type.
7021     return SDValue();
7022 
7023   // For big endian targets, we need to adjust the offset to the pointer to
7024   // load the correct bytes.
7025   if (DAG.getDataLayout().isBigEndian()) {
7026     unsigned LVTStoreBits = LN0->getMemoryVT().getStoreSizeInBits();
7027     unsigned EVTStoreBits = ExtVT.getStoreSizeInBits();
7028     ShAmt = LVTStoreBits - EVTStoreBits - ShAmt;
7029   }
7030 
7031   uint64_t PtrOff = ShAmt / 8;
7032   unsigned NewAlign = MinAlign(LN0->getAlignment(), PtrOff);
7033   SDLoc DL(LN0);
7034   // The original load itself didn't wrap, so an offset within it doesn't.
7035   SDNodeFlags Flags;
7036   Flags.setNoUnsignedWrap(true);
7037   SDValue NewPtr = DAG.getNode(ISD::ADD, DL,
7038                                PtrType, LN0->getBasePtr(),
7039                                DAG.getConstant(PtrOff, DL, PtrType),
7040                                &Flags);
7041   AddToWorklist(NewPtr.getNode());
7042 
7043   SDValue Load;
7044   if (ExtType == ISD::NON_EXTLOAD)
7045     Load = DAG.getLoad(VT, SDLoc(N0), LN0->getChain(), NewPtr,
7046                        LN0->getPointerInfo().getWithOffset(PtrOff), NewAlign,
7047                        LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
7048   else
7049     Load = DAG.getExtLoad(ExtType, SDLoc(N0), VT, LN0->getChain(), NewPtr,
7050                           LN0->getPointerInfo().getWithOffset(PtrOff), ExtVT,
7051                           NewAlign, LN0->getMemOperand()->getFlags(),
7052                           LN0->getAAInfo());
7053 
7054   // Replace the old load's chain with the new load's chain.
7055   WorklistRemover DeadNodes(*this);
7056   DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
7057 
7058   // Shift the result left, if we've swallowed a left shift.
7059   SDValue Result = Load;
7060   if (ShLeftAmt != 0) {
7061     EVT ShImmTy = getShiftAmountTy(Result.getValueType());
7062     if (!isUIntN(ShImmTy.getSizeInBits(), ShLeftAmt))
7063       ShImmTy = VT;
7064     // If the shift amount is as large as the result size (but, presumably,
7065     // no larger than the source) then the useful bits of the result are
7066     // zero; we can't simply return the shortened shift, because the result
7067     // of that operation is undefined.
7068     SDLoc DL(N0);
7069     if (ShLeftAmt >= VT.getSizeInBits())
7070       Result = DAG.getConstant(0, DL, VT);
7071     else
7072       Result = DAG.getNode(ISD::SHL, DL, VT,
7073                           Result, DAG.getConstant(ShLeftAmt, DL, ShImmTy));
7074   }
7075 
7076   // Return the new loaded value.
7077   return Result;
7078 }
7079 
7080 SDValue DAGCombiner::visitSIGN_EXTEND_INREG(SDNode *N) {
7081   SDValue N0 = N->getOperand(0);
7082   SDValue N1 = N->getOperand(1);
7083   EVT VT = N->getValueType(0);
7084   EVT EVT = cast<VTSDNode>(N1)->getVT();
7085   unsigned VTBits = VT.getScalarSizeInBits();
7086   unsigned EVTBits = EVT.getScalarSizeInBits();
7087 
7088   if (N0.isUndef())
7089     return DAG.getUNDEF(VT);
7090 
7091   // fold (sext_in_reg c1) -> c1
7092   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7093     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, N0, N1);
7094 
7095   // If the input is already sign extended, just drop the extension.
7096   if (DAG.ComputeNumSignBits(N0) >= VTBits-EVTBits+1)
7097     return N0;
7098 
7099   // fold (sext_in_reg (sext_in_reg x, VT2), VT1) -> (sext_in_reg x, minVT) pt2
7100   if (N0.getOpcode() == ISD::SIGN_EXTEND_INREG &&
7101       EVT.bitsLT(cast<VTSDNode>(N0.getOperand(1))->getVT()))
7102     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
7103                        N0.getOperand(0), N1);
7104 
7105   // fold (sext_in_reg (sext x)) -> (sext x)
7106   // fold (sext_in_reg (aext x)) -> (sext x)
7107   // if x is small enough.
7108   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) {
7109     SDValue N00 = N0.getOperand(0);
7110     if (N00.getScalarValueSizeInBits() <= EVTBits &&
7111         (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT)))
7112       return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00, N1);
7113   }
7114 
7115   // fold (sext_in_reg x) -> (zext_in_reg x) if the sign bit is known zero.
7116   if (DAG.MaskedValueIsZero(N0, APInt::getBitsSet(VTBits, EVTBits-1, EVTBits)))
7117     return DAG.getZeroExtendInReg(N0, SDLoc(N), EVT.getScalarType());
7118 
7119   // fold operands of sext_in_reg based on knowledge that the top bits are not
7120   // demanded.
7121   if (SimplifyDemandedBits(SDValue(N, 0)))
7122     return SDValue(N, 0);
7123 
7124   // fold (sext_in_reg (load x)) -> (smaller sextload x)
7125   // fold (sext_in_reg (srl (load x), c)) -> (smaller sextload (x+c/evtbits))
7126   if (SDValue NarrowLoad = ReduceLoadWidth(N))
7127     return NarrowLoad;
7128 
7129   // fold (sext_in_reg (srl X, 24), i8) -> (sra X, 24)
7130   // fold (sext_in_reg (srl X, 23), i8) -> (sra X, 23) iff possible.
7131   // We already fold "(sext_in_reg (srl X, 25), i8) -> srl X, 25" above.
7132   if (N0.getOpcode() == ISD::SRL) {
7133     if (ConstantSDNode *ShAmt = dyn_cast<ConstantSDNode>(N0.getOperand(1)))
7134       if (ShAmt->getZExtValue()+EVTBits <= VTBits) {
7135         // We can turn this into an SRA iff the input to the SRL is already sign
7136         // extended enough.
7137         unsigned InSignBits = DAG.ComputeNumSignBits(N0.getOperand(0));
7138         if (VTBits-(ShAmt->getZExtValue()+EVTBits) < InSignBits)
7139           return DAG.getNode(ISD::SRA, SDLoc(N), VT,
7140                              N0.getOperand(0), N0.getOperand(1));
7141       }
7142   }
7143 
7144   // fold (sext_inreg (extload x)) -> (sextload x)
7145   if (ISD::isEXTLoad(N0.getNode()) &&
7146       ISD::isUNINDEXEDLoad(N0.getNode()) &&
7147       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
7148       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
7149        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
7150     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7151     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
7152                                      LN0->getChain(),
7153                                      LN0->getBasePtr(), EVT,
7154                                      LN0->getMemOperand());
7155     CombineTo(N, ExtLoad);
7156     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
7157     AddToWorklist(ExtLoad.getNode());
7158     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7159   }
7160   // fold (sext_inreg (zextload x)) -> (sextload x) iff load has one use
7161   if (ISD::isZEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
7162       N0.hasOneUse() &&
7163       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
7164       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
7165        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
7166     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7167     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
7168                                      LN0->getChain(),
7169                                      LN0->getBasePtr(), EVT,
7170                                      LN0->getMemOperand());
7171     CombineTo(N, ExtLoad);
7172     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
7173     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7174   }
7175 
7176   // Form (sext_inreg (bswap >> 16)) or (sext_inreg (rotl (bswap) 16))
7177   if (EVTBits <= 16 && N0.getOpcode() == ISD::OR) {
7178     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
7179                                            N0.getOperand(1), false))
7180       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
7181                          BSwap, N1);
7182   }
7183 
7184   return SDValue();
7185 }
7186 
7187 SDValue DAGCombiner::visitSIGN_EXTEND_VECTOR_INREG(SDNode *N) {
7188   SDValue N0 = N->getOperand(0);
7189   EVT VT = N->getValueType(0);
7190 
7191   if (N0.isUndef())
7192     return DAG.getUNDEF(VT);
7193 
7194   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
7195                                               LegalOperations))
7196     return SDValue(Res, 0);
7197 
7198   return SDValue();
7199 }
7200 
7201 SDValue DAGCombiner::visitZERO_EXTEND_VECTOR_INREG(SDNode *N) {
7202   SDValue N0 = N->getOperand(0);
7203   EVT VT = N->getValueType(0);
7204 
7205   if (N0.isUndef())
7206     return DAG.getUNDEF(VT);
7207 
7208   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
7209                                               LegalOperations))
7210     return SDValue(Res, 0);
7211 
7212   return SDValue();
7213 }
7214 
7215 SDValue DAGCombiner::visitTRUNCATE(SDNode *N) {
7216   SDValue N0 = N->getOperand(0);
7217   EVT VT = N->getValueType(0);
7218   bool isLE = DAG.getDataLayout().isLittleEndian();
7219 
7220   // noop truncate
7221   if (N0.getValueType() == N->getValueType(0))
7222     return N0;
7223   // fold (truncate c1) -> c1
7224   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7225     return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0);
7226   // fold (truncate (truncate x)) -> (truncate x)
7227   if (N0.getOpcode() == ISD::TRUNCATE)
7228     return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
7229   // fold (truncate (ext x)) -> (ext x) or (truncate x) or x
7230   if (N0.getOpcode() == ISD::ZERO_EXTEND ||
7231       N0.getOpcode() == ISD::SIGN_EXTEND ||
7232       N0.getOpcode() == ISD::ANY_EXTEND) {
7233     // if the source is smaller than the dest, we still need an extend.
7234     if (N0.getOperand(0).getValueType().bitsLT(VT))
7235       return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
7236     // if the source is larger than the dest, than we just need the truncate.
7237     if (N0.getOperand(0).getValueType().bitsGT(VT))
7238       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
7239     // if the source and dest are the same type, we can drop both the extend
7240     // and the truncate.
7241     return N0.getOperand(0);
7242   }
7243 
7244   // If this is anyext(trunc), don't fold it, allow ourselves to be folded.
7245   if (N->hasOneUse() && (N->use_begin()->getOpcode() == ISD::ANY_EXTEND))
7246     return SDValue();
7247 
7248   // Fold extract-and-trunc into a narrow extract. For example:
7249   //   i64 x = EXTRACT_VECTOR_ELT(v2i64 val, i32 1)
7250   //   i32 y = TRUNCATE(i64 x)
7251   //        -- becomes --
7252   //   v16i8 b = BITCAST (v2i64 val)
7253   //   i8 x = EXTRACT_VECTOR_ELT(v16i8 b, i32 8)
7254   //
7255   // Note: We only run this optimization after type legalization (which often
7256   // creates this pattern) and before operation legalization after which
7257   // we need to be more careful about the vector instructions that we generate.
7258   if (N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
7259       LegalTypes && !LegalOperations && N0->hasOneUse() && VT != MVT::i1) {
7260 
7261     EVT VecTy = N0.getOperand(0).getValueType();
7262     EVT ExTy = N0.getValueType();
7263     EVT TrTy = N->getValueType(0);
7264 
7265     unsigned NumElem = VecTy.getVectorNumElements();
7266     unsigned SizeRatio = ExTy.getSizeInBits()/TrTy.getSizeInBits();
7267 
7268     EVT NVT = EVT::getVectorVT(*DAG.getContext(), TrTy, SizeRatio * NumElem);
7269     assert(NVT.getSizeInBits() == VecTy.getSizeInBits() && "Invalid Size");
7270 
7271     SDValue EltNo = N0->getOperand(1);
7272     if (isa<ConstantSDNode>(EltNo) && isTypeLegal(NVT)) {
7273       int Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
7274       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
7275       int Index = isLE ? (Elt*SizeRatio) : (Elt*SizeRatio + (SizeRatio-1));
7276 
7277       SDLoc DL(N);
7278       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, TrTy,
7279                          DAG.getBitcast(NVT, N0.getOperand(0)),
7280                          DAG.getConstant(Index, DL, IndexTy));
7281     }
7282   }
7283 
7284   // trunc (select c, a, b) -> select c, (trunc a), (trunc b)
7285   if (N0.getOpcode() == ISD::SELECT) {
7286     EVT SrcVT = N0.getValueType();
7287     if ((!LegalOperations || TLI.isOperationLegal(ISD::SELECT, SrcVT)) &&
7288         TLI.isTruncateFree(SrcVT, VT)) {
7289       SDLoc SL(N0);
7290       SDValue Cond = N0.getOperand(0);
7291       SDValue TruncOp0 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1));
7292       SDValue TruncOp1 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(2));
7293       return DAG.getNode(ISD::SELECT, SDLoc(N), VT, Cond, TruncOp0, TruncOp1);
7294     }
7295   }
7296 
7297   // trunc (shl x, K) -> shl (trunc x), K => K < VT.getScalarSizeInBits()
7298   if (N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
7299       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::SHL, VT)) &&
7300       TLI.isTypeDesirableForOp(ISD::SHL, VT)) {
7301     if (const ConstantSDNode *CAmt = isConstOrConstSplat(N0.getOperand(1))) {
7302       uint64_t Amt = CAmt->getZExtValue();
7303       unsigned Size = VT.getScalarSizeInBits();
7304 
7305       if (Amt < Size) {
7306         SDLoc SL(N);
7307         EVT AmtVT = TLI.getShiftAmountTy(VT, DAG.getDataLayout());
7308 
7309         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0));
7310         return DAG.getNode(ISD::SHL, SL, VT, Trunc,
7311                            DAG.getConstant(Amt, SL, AmtVT));
7312       }
7313     }
7314   }
7315 
7316   // Fold a series of buildvector, bitcast, and truncate if possible.
7317   // For example fold
7318   //   (2xi32 trunc (bitcast ((4xi32)buildvector x, x, y, y) 2xi64)) to
7319   //   (2xi32 (buildvector x, y)).
7320   if (Level == AfterLegalizeVectorOps && VT.isVector() &&
7321       N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
7322       N0.getOperand(0).getOpcode() == ISD::BUILD_VECTOR &&
7323       N0.getOperand(0).hasOneUse()) {
7324 
7325     SDValue BuildVect = N0.getOperand(0);
7326     EVT BuildVectEltTy = BuildVect.getValueType().getVectorElementType();
7327     EVT TruncVecEltTy = VT.getVectorElementType();
7328 
7329     // Check that the element types match.
7330     if (BuildVectEltTy == TruncVecEltTy) {
7331       // Now we only need to compute the offset of the truncated elements.
7332       unsigned BuildVecNumElts =  BuildVect.getNumOperands();
7333       unsigned TruncVecNumElts = VT.getVectorNumElements();
7334       unsigned TruncEltOffset = BuildVecNumElts / TruncVecNumElts;
7335 
7336       assert((BuildVecNumElts % TruncVecNumElts) == 0 &&
7337              "Invalid number of elements");
7338 
7339       SmallVector<SDValue, 8> Opnds;
7340       for (unsigned i = 0, e = BuildVecNumElts; i != e; i += TruncEltOffset)
7341         Opnds.push_back(BuildVect.getOperand(i));
7342 
7343       return DAG.getBuildVector(VT, SDLoc(N), Opnds);
7344     }
7345   }
7346 
7347   // See if we can simplify the input to this truncate through knowledge that
7348   // only the low bits are being used.
7349   // For example "trunc (or (shl x, 8), y)" // -> trunc y
7350   // Currently we only perform this optimization on scalars because vectors
7351   // may have different active low bits.
7352   if (!VT.isVector()) {
7353     if (SDValue Shorter =
7354             GetDemandedBits(N0, APInt::getLowBitsSet(N0.getValueSizeInBits(),
7355                                                      VT.getSizeInBits())))
7356       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Shorter);
7357   }
7358   // fold (truncate (load x)) -> (smaller load x)
7359   // fold (truncate (srl (load x), c)) -> (smaller load (x+c/evtbits))
7360   if (!LegalTypes || TLI.isTypeDesirableForOp(N0.getOpcode(), VT)) {
7361     if (SDValue Reduced = ReduceLoadWidth(N))
7362       return Reduced;
7363 
7364     // Handle the case where the load remains an extending load even
7365     // after truncation.
7366     if (N0.hasOneUse() && ISD::isUNINDEXEDLoad(N0.getNode())) {
7367       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7368       if (!LN0->isVolatile() &&
7369           LN0->getMemoryVT().getStoreSizeInBits() < VT.getSizeInBits()) {
7370         SDValue NewLoad = DAG.getExtLoad(LN0->getExtensionType(), SDLoc(LN0),
7371                                          VT, LN0->getChain(), LN0->getBasePtr(),
7372                                          LN0->getMemoryVT(),
7373                                          LN0->getMemOperand());
7374         DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLoad.getValue(1));
7375         return NewLoad;
7376       }
7377     }
7378   }
7379   // fold (trunc (concat ... x ...)) -> (concat ..., (trunc x), ...)),
7380   // where ... are all 'undef'.
7381   if (N0.getOpcode() == ISD::CONCAT_VECTORS && !LegalTypes) {
7382     SmallVector<EVT, 8> VTs;
7383     SDValue V;
7384     unsigned Idx = 0;
7385     unsigned NumDefs = 0;
7386 
7387     for (unsigned i = 0, e = N0.getNumOperands(); i != e; ++i) {
7388       SDValue X = N0.getOperand(i);
7389       if (!X.isUndef()) {
7390         V = X;
7391         Idx = i;
7392         NumDefs++;
7393       }
7394       // Stop if more than one members are non-undef.
7395       if (NumDefs > 1)
7396         break;
7397       VTs.push_back(EVT::getVectorVT(*DAG.getContext(),
7398                                      VT.getVectorElementType(),
7399                                      X.getValueType().getVectorNumElements()));
7400     }
7401 
7402     if (NumDefs == 0)
7403       return DAG.getUNDEF(VT);
7404 
7405     if (NumDefs == 1) {
7406       assert(V.getNode() && "The single defined operand is empty!");
7407       SmallVector<SDValue, 8> Opnds;
7408       for (unsigned i = 0, e = VTs.size(); i != e; ++i) {
7409         if (i != Idx) {
7410           Opnds.push_back(DAG.getUNDEF(VTs[i]));
7411           continue;
7412         }
7413         SDValue NV = DAG.getNode(ISD::TRUNCATE, SDLoc(V), VTs[i], V);
7414         AddToWorklist(NV.getNode());
7415         Opnds.push_back(NV);
7416       }
7417       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Opnds);
7418     }
7419   }
7420 
7421   // Fold truncate of a bitcast of a vector to an extract of the low vector
7422   // element.
7423   //
7424   // e.g. trunc (i64 (bitcast v2i32:x)) -> extract_vector_elt v2i32:x, 0
7425   if (N0.getOpcode() == ISD::BITCAST && !VT.isVector()) {
7426     SDValue VecSrc = N0.getOperand(0);
7427     EVT SrcVT = VecSrc.getValueType();
7428     if (SrcVT.isVector() && SrcVT.getScalarType() == VT &&
7429         (!LegalOperations ||
7430          TLI.isOperationLegal(ISD::EXTRACT_VECTOR_ELT, SrcVT))) {
7431       SDLoc SL(N);
7432 
7433       EVT IdxVT = TLI.getVectorIdxTy(DAG.getDataLayout());
7434       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, VT,
7435                          VecSrc, DAG.getConstant(0, SL, IdxVT));
7436     }
7437   }
7438 
7439   // Simplify the operands using demanded-bits information.
7440   if (!VT.isVector() &&
7441       SimplifyDemandedBits(SDValue(N, 0)))
7442     return SDValue(N, 0);
7443 
7444   return SDValue();
7445 }
7446 
7447 static SDNode *getBuildPairElt(SDNode *N, unsigned i) {
7448   SDValue Elt = N->getOperand(i);
7449   if (Elt.getOpcode() != ISD::MERGE_VALUES)
7450     return Elt.getNode();
7451   return Elt.getOperand(Elt.getResNo()).getNode();
7452 }
7453 
7454 /// build_pair (load, load) -> load
7455 /// if load locations are consecutive.
7456 SDValue DAGCombiner::CombineConsecutiveLoads(SDNode *N, EVT VT) {
7457   assert(N->getOpcode() == ISD::BUILD_PAIR);
7458 
7459   LoadSDNode *LD1 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 0));
7460   LoadSDNode *LD2 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 1));
7461   if (!LD1 || !LD2 || !ISD::isNON_EXTLoad(LD1) || !LD1->hasOneUse() ||
7462       LD1->getAddressSpace() != LD2->getAddressSpace())
7463     return SDValue();
7464   EVT LD1VT = LD1->getValueType(0);
7465   unsigned LD1Bytes = LD1VT.getSizeInBits() / 8;
7466   if (ISD::isNON_EXTLoad(LD2) && LD2->hasOneUse() &&
7467       DAG.areNonVolatileConsecutiveLoads(LD2, LD1, LD1Bytes, 1)) {
7468     unsigned Align = LD1->getAlignment();
7469     unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
7470         VT.getTypeForEVT(*DAG.getContext()));
7471 
7472     if (NewAlign <= Align &&
7473         (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT)))
7474       return DAG.getLoad(VT, SDLoc(N), LD1->getChain(), LD1->getBasePtr(),
7475                          LD1->getPointerInfo(), Align);
7476   }
7477 
7478   return SDValue();
7479 }
7480 
7481 static unsigned getPPCf128HiElementSelector(const SelectionDAG &DAG) {
7482   // On little-endian machines, bitcasting from ppcf128 to i128 does swap the Hi
7483   // and Lo parts; on big-endian machines it doesn't.
7484   return DAG.getDataLayout().isBigEndian() ? 1 : 0;
7485 }
7486 
7487 static SDValue foldBitcastedFPLogic(SDNode *N, SelectionDAG &DAG,
7488                                     const TargetLowering &TLI) {
7489   // If this is not a bitcast to an FP type or if the target doesn't have
7490   // IEEE754-compliant FP logic, we're done.
7491   EVT VT = N->getValueType(0);
7492   if (!VT.isFloatingPoint() || !TLI.hasBitPreservingFPLogic(VT))
7493     return SDValue();
7494 
7495   // TODO: Use splat values for the constant-checking below and remove this
7496   // restriction.
7497   SDValue N0 = N->getOperand(0);
7498   EVT SourceVT = N0.getValueType();
7499   if (SourceVT.isVector())
7500     return SDValue();
7501 
7502   unsigned FPOpcode;
7503   APInt SignMask;
7504   switch (N0.getOpcode()) {
7505   case ISD::AND:
7506     FPOpcode = ISD::FABS;
7507     SignMask = ~APInt::getSignBit(SourceVT.getSizeInBits());
7508     break;
7509   case ISD::XOR:
7510     FPOpcode = ISD::FNEG;
7511     SignMask = APInt::getSignBit(SourceVT.getSizeInBits());
7512     break;
7513   // TODO: ISD::OR --> ISD::FNABS?
7514   default:
7515     return SDValue();
7516   }
7517 
7518   // Fold (bitcast int (and (bitcast fp X to int), 0x7fff...) to fp) -> fabs X
7519   // Fold (bitcast int (xor (bitcast fp X to int), 0x8000...) to fp) -> fneg X
7520   SDValue LogicOp0 = N0.getOperand(0);
7521   ConstantSDNode *LogicOp1 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7522   if (LogicOp1 && LogicOp1->getAPIntValue() == SignMask &&
7523       LogicOp0.getOpcode() == ISD::BITCAST &&
7524       LogicOp0->getOperand(0).getValueType() == VT)
7525     return DAG.getNode(FPOpcode, SDLoc(N), VT, LogicOp0->getOperand(0));
7526 
7527   return SDValue();
7528 }
7529 
7530 SDValue DAGCombiner::visitBITCAST(SDNode *N) {
7531   SDValue N0 = N->getOperand(0);
7532   EVT VT = N->getValueType(0);
7533 
7534   // If the input is a BUILD_VECTOR with all constant elements, fold this now.
7535   // Only do this before legalize, since afterward the target may be depending
7536   // on the bitconvert.
7537   // First check to see if this is all constant.
7538   if (!LegalTypes &&
7539       N0.getOpcode() == ISD::BUILD_VECTOR && N0.getNode()->hasOneUse() &&
7540       VT.isVector()) {
7541     bool isSimple = cast<BuildVectorSDNode>(N0)->isConstant();
7542 
7543     EVT DestEltVT = N->getValueType(0).getVectorElementType();
7544     assert(!DestEltVT.isVector() &&
7545            "Element type of vector ValueType must not be vector!");
7546     if (isSimple)
7547       return ConstantFoldBITCASTofBUILD_VECTOR(N0.getNode(), DestEltVT);
7548   }
7549 
7550   // If the input is a constant, let getNode fold it.
7551   if (isa<ConstantSDNode>(N0) || isa<ConstantFPSDNode>(N0)) {
7552     // If we can't allow illegal operations, we need to check that this is just
7553     // a fp -> int or int -> conversion and that the resulting operation will
7554     // be legal.
7555     if (!LegalOperations ||
7556         (isa<ConstantSDNode>(N0) && VT.isFloatingPoint() && !VT.isVector() &&
7557          TLI.isOperationLegal(ISD::ConstantFP, VT)) ||
7558         (isa<ConstantFPSDNode>(N0) && VT.isInteger() && !VT.isVector() &&
7559          TLI.isOperationLegal(ISD::Constant, VT)))
7560       return DAG.getBitcast(VT, N0);
7561   }
7562 
7563   // (conv (conv x, t1), t2) -> (conv x, t2)
7564   if (N0.getOpcode() == ISD::BITCAST)
7565     return DAG.getBitcast(VT, N0.getOperand(0));
7566 
7567   // fold (conv (load x)) -> (load (conv*)x)
7568   // If the resultant load doesn't need a higher alignment than the original!
7569   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
7570       // Do not change the width of a volatile load.
7571       !cast<LoadSDNode>(N0)->isVolatile() &&
7572       // Do not remove the cast if the types differ in endian layout.
7573       TLI.hasBigEndianPartOrdering(N0.getValueType(), DAG.getDataLayout()) ==
7574           TLI.hasBigEndianPartOrdering(VT, DAG.getDataLayout()) &&
7575       (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT)) &&
7576       TLI.isLoadBitCastBeneficial(N0.getValueType(), VT)) {
7577     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7578     unsigned OrigAlign = LN0->getAlignment();
7579 
7580     bool Fast = false;
7581     if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT,
7582                                LN0->getAddressSpace(), OrigAlign, &Fast) &&
7583         Fast) {
7584       SDValue Load =
7585           DAG.getLoad(VT, SDLoc(N), LN0->getChain(), LN0->getBasePtr(),
7586                       LN0->getPointerInfo(), OrigAlign,
7587                       LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
7588       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
7589       return Load;
7590     }
7591   }
7592 
7593   if (SDValue V = foldBitcastedFPLogic(N, DAG, TLI))
7594     return V;
7595 
7596   // fold (bitconvert (fneg x)) -> (xor (bitconvert x), signbit)
7597   // fold (bitconvert (fabs x)) -> (and (bitconvert x), (not signbit))
7598   //
7599   // For ppc_fp128:
7600   // fold (bitcast (fneg x)) ->
7601   //     flipbit = signbit
7602   //     (xor (bitcast x) (build_pair flipbit, flipbit))
7603   //
7604   // fold (bitcast (fabs x)) ->
7605   //     flipbit = (and (extract_element (bitcast x), 0), signbit)
7606   //     (xor (bitcast x) (build_pair flipbit, flipbit))
7607   // This often reduces constant pool loads.
7608   if (((N0.getOpcode() == ISD::FNEG && !TLI.isFNegFree(N0.getValueType())) ||
7609        (N0.getOpcode() == ISD::FABS && !TLI.isFAbsFree(N0.getValueType()))) &&
7610       N0.getNode()->hasOneUse() && VT.isInteger() &&
7611       !VT.isVector() && !N0.getValueType().isVector()) {
7612     SDValue NewConv = DAG.getBitcast(VT, N0.getOperand(0));
7613     AddToWorklist(NewConv.getNode());
7614 
7615     SDLoc DL(N);
7616     if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
7617       assert(VT.getSizeInBits() == 128);
7618       SDValue SignBit = DAG.getConstant(
7619           APInt::getSignBit(VT.getSizeInBits() / 2), SDLoc(N0), MVT::i64);
7620       SDValue FlipBit;
7621       if (N0.getOpcode() == ISD::FNEG) {
7622         FlipBit = SignBit;
7623         AddToWorklist(FlipBit.getNode());
7624       } else {
7625         assert(N0.getOpcode() == ISD::FABS);
7626         SDValue Hi =
7627             DAG.getNode(ISD::EXTRACT_ELEMENT, SDLoc(NewConv), MVT::i64, NewConv,
7628                         DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
7629                                               SDLoc(NewConv)));
7630         AddToWorklist(Hi.getNode());
7631         FlipBit = DAG.getNode(ISD::AND, SDLoc(N0), MVT::i64, Hi, SignBit);
7632         AddToWorklist(FlipBit.getNode());
7633       }
7634       SDValue FlipBits =
7635           DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
7636       AddToWorklist(FlipBits.getNode());
7637       return DAG.getNode(ISD::XOR, DL, VT, NewConv, FlipBits);
7638     }
7639     APInt SignBit = APInt::getSignBit(VT.getSizeInBits());
7640     if (N0.getOpcode() == ISD::FNEG)
7641       return DAG.getNode(ISD::XOR, DL, VT,
7642                          NewConv, DAG.getConstant(SignBit, DL, VT));
7643     assert(N0.getOpcode() == ISD::FABS);
7644     return DAG.getNode(ISD::AND, DL, VT,
7645                        NewConv, DAG.getConstant(~SignBit, DL, VT));
7646   }
7647 
7648   // fold (bitconvert (fcopysign cst, x)) ->
7649   //         (or (and (bitconvert x), sign), (and cst, (not sign)))
7650   // Note that we don't handle (copysign x, cst) because this can always be
7651   // folded to an fneg or fabs.
7652   //
7653   // For ppc_fp128:
7654   // fold (bitcast (fcopysign cst, x)) ->
7655   //     flipbit = (and (extract_element
7656   //                     (xor (bitcast cst), (bitcast x)), 0),
7657   //                    signbit)
7658   //     (xor (bitcast cst) (build_pair flipbit, flipbit))
7659   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse() &&
7660       isa<ConstantFPSDNode>(N0.getOperand(0)) &&
7661       VT.isInteger() && !VT.isVector()) {
7662     unsigned OrigXWidth = N0.getOperand(1).getValueSizeInBits();
7663     EVT IntXVT = EVT::getIntegerVT(*DAG.getContext(), OrigXWidth);
7664     if (isTypeLegal(IntXVT)) {
7665       SDValue X = DAG.getBitcast(IntXVT, N0.getOperand(1));
7666       AddToWorklist(X.getNode());
7667 
7668       // If X has a different width than the result/lhs, sext it or truncate it.
7669       unsigned VTWidth = VT.getSizeInBits();
7670       if (OrigXWidth < VTWidth) {
7671         X = DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, X);
7672         AddToWorklist(X.getNode());
7673       } else if (OrigXWidth > VTWidth) {
7674         // To get the sign bit in the right place, we have to shift it right
7675         // before truncating.
7676         SDLoc DL(X);
7677         X = DAG.getNode(ISD::SRL, DL,
7678                         X.getValueType(), X,
7679                         DAG.getConstant(OrigXWidth-VTWidth, DL,
7680                                         X.getValueType()));
7681         AddToWorklist(X.getNode());
7682         X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
7683         AddToWorklist(X.getNode());
7684       }
7685 
7686       if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
7687         APInt SignBit = APInt::getSignBit(VT.getSizeInBits() / 2);
7688         SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
7689         AddToWorklist(Cst.getNode());
7690         SDValue X = DAG.getBitcast(VT, N0.getOperand(1));
7691         AddToWorklist(X.getNode());
7692         SDValue XorResult = DAG.getNode(ISD::XOR, SDLoc(N0), VT, Cst, X);
7693         AddToWorklist(XorResult.getNode());
7694         SDValue XorResult64 = DAG.getNode(
7695             ISD::EXTRACT_ELEMENT, SDLoc(XorResult), MVT::i64, XorResult,
7696             DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
7697                                   SDLoc(XorResult)));
7698         AddToWorklist(XorResult64.getNode());
7699         SDValue FlipBit =
7700             DAG.getNode(ISD::AND, SDLoc(XorResult64), MVT::i64, XorResult64,
7701                         DAG.getConstant(SignBit, SDLoc(XorResult64), MVT::i64));
7702         AddToWorklist(FlipBit.getNode());
7703         SDValue FlipBits =
7704             DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
7705         AddToWorklist(FlipBits.getNode());
7706         return DAG.getNode(ISD::XOR, SDLoc(N), VT, Cst, FlipBits);
7707       }
7708       APInt SignBit = APInt::getSignBit(VT.getSizeInBits());
7709       X = DAG.getNode(ISD::AND, SDLoc(X), VT,
7710                       X, DAG.getConstant(SignBit, SDLoc(X), VT));
7711       AddToWorklist(X.getNode());
7712 
7713       SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
7714       Cst = DAG.getNode(ISD::AND, SDLoc(Cst), VT,
7715                         Cst, DAG.getConstant(~SignBit, SDLoc(Cst), VT));
7716       AddToWorklist(Cst.getNode());
7717 
7718       return DAG.getNode(ISD::OR, SDLoc(N), VT, X, Cst);
7719     }
7720   }
7721 
7722   // bitconvert(build_pair(ld, ld)) -> ld iff load locations are consecutive.
7723   if (N0.getOpcode() == ISD::BUILD_PAIR)
7724     if (SDValue CombineLD = CombineConsecutiveLoads(N0.getNode(), VT))
7725       return CombineLD;
7726 
7727   // Remove double bitcasts from shuffles - this is often a legacy of
7728   // XformToShuffleWithZero being used to combine bitmaskings (of
7729   // float vectors bitcast to integer vectors) into shuffles.
7730   // bitcast(shuffle(bitcast(s0),bitcast(s1))) -> shuffle(s0,s1)
7731   if (Level < AfterLegalizeDAG && TLI.isTypeLegal(VT) && VT.isVector() &&
7732       N0->getOpcode() == ISD::VECTOR_SHUFFLE &&
7733       VT.getVectorNumElements() >= N0.getValueType().getVectorNumElements() &&
7734       !(VT.getVectorNumElements() % N0.getValueType().getVectorNumElements())) {
7735     ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N0);
7736 
7737     // If operands are a bitcast, peek through if it casts the original VT.
7738     // If operands are a constant, just bitcast back to original VT.
7739     auto PeekThroughBitcast = [&](SDValue Op) {
7740       if (Op.getOpcode() == ISD::BITCAST &&
7741           Op.getOperand(0).getValueType() == VT)
7742         return SDValue(Op.getOperand(0));
7743       if (ISD::isBuildVectorOfConstantSDNodes(Op.getNode()) ||
7744           ISD::isBuildVectorOfConstantFPSDNodes(Op.getNode()))
7745         return DAG.getBitcast(VT, Op);
7746       return SDValue();
7747     };
7748 
7749     SDValue SV0 = PeekThroughBitcast(N0->getOperand(0));
7750     SDValue SV1 = PeekThroughBitcast(N0->getOperand(1));
7751     if (!(SV0 && SV1))
7752       return SDValue();
7753 
7754     int MaskScale =
7755         VT.getVectorNumElements() / N0.getValueType().getVectorNumElements();
7756     SmallVector<int, 8> NewMask;
7757     for (int M : SVN->getMask())
7758       for (int i = 0; i != MaskScale; ++i)
7759         NewMask.push_back(M < 0 ? -1 : M * MaskScale + i);
7760 
7761     bool LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
7762     if (!LegalMask) {
7763       std::swap(SV0, SV1);
7764       ShuffleVectorSDNode::commuteMask(NewMask);
7765       LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
7766     }
7767 
7768     if (LegalMask)
7769       return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, NewMask);
7770   }
7771 
7772   return SDValue();
7773 }
7774 
7775 SDValue DAGCombiner::visitBUILD_PAIR(SDNode *N) {
7776   EVT VT = N->getValueType(0);
7777   return CombineConsecutiveLoads(N, VT);
7778 }
7779 
7780 /// We know that BV is a build_vector node with Constant, ConstantFP or Undef
7781 /// operands. DstEltVT indicates the destination element value type.
7782 SDValue DAGCombiner::
7783 ConstantFoldBITCASTofBUILD_VECTOR(SDNode *BV, EVT DstEltVT) {
7784   EVT SrcEltVT = BV->getValueType(0).getVectorElementType();
7785 
7786   // If this is already the right type, we're done.
7787   if (SrcEltVT == DstEltVT) return SDValue(BV, 0);
7788 
7789   unsigned SrcBitSize = SrcEltVT.getSizeInBits();
7790   unsigned DstBitSize = DstEltVT.getSizeInBits();
7791 
7792   // If this is a conversion of N elements of one type to N elements of another
7793   // type, convert each element.  This handles FP<->INT cases.
7794   if (SrcBitSize == DstBitSize) {
7795     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
7796                               BV->getValueType(0).getVectorNumElements());
7797 
7798     // Due to the FP element handling below calling this routine recursively,
7799     // we can end up with a scalar-to-vector node here.
7800     if (BV->getOpcode() == ISD::SCALAR_TO_VECTOR)
7801       return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(BV), VT,
7802                          DAG.getBitcast(DstEltVT, BV->getOperand(0)));
7803 
7804     SmallVector<SDValue, 8> Ops;
7805     for (SDValue Op : BV->op_values()) {
7806       // If the vector element type is not legal, the BUILD_VECTOR operands
7807       // are promoted and implicitly truncated.  Make that explicit here.
7808       if (Op.getValueType() != SrcEltVT)
7809         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(BV), SrcEltVT, Op);
7810       Ops.push_back(DAG.getBitcast(DstEltVT, Op));
7811       AddToWorklist(Ops.back().getNode());
7812     }
7813     return DAG.getBuildVector(VT, SDLoc(BV), Ops);
7814   }
7815 
7816   // Otherwise, we're growing or shrinking the elements.  To avoid having to
7817   // handle annoying details of growing/shrinking FP values, we convert them to
7818   // int first.
7819   if (SrcEltVT.isFloatingPoint()) {
7820     // Convert the input float vector to a int vector where the elements are the
7821     // same sizes.
7822     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), SrcEltVT.getSizeInBits());
7823     BV = ConstantFoldBITCASTofBUILD_VECTOR(BV, IntVT).getNode();
7824     SrcEltVT = IntVT;
7825   }
7826 
7827   // Now we know the input is an integer vector.  If the output is a FP type,
7828   // convert to integer first, then to FP of the right size.
7829   if (DstEltVT.isFloatingPoint()) {
7830     EVT TmpVT = EVT::getIntegerVT(*DAG.getContext(), DstEltVT.getSizeInBits());
7831     SDNode *Tmp = ConstantFoldBITCASTofBUILD_VECTOR(BV, TmpVT).getNode();
7832 
7833     // Next, convert to FP elements of the same size.
7834     return ConstantFoldBITCASTofBUILD_VECTOR(Tmp, DstEltVT);
7835   }
7836 
7837   SDLoc DL(BV);
7838 
7839   // Okay, we know the src/dst types are both integers of differing types.
7840   // Handling growing first.
7841   assert(SrcEltVT.isInteger() && DstEltVT.isInteger());
7842   if (SrcBitSize < DstBitSize) {
7843     unsigned NumInputsPerOutput = DstBitSize/SrcBitSize;
7844 
7845     SmallVector<SDValue, 8> Ops;
7846     for (unsigned i = 0, e = BV->getNumOperands(); i != e;
7847          i += NumInputsPerOutput) {
7848       bool isLE = DAG.getDataLayout().isLittleEndian();
7849       APInt NewBits = APInt(DstBitSize, 0);
7850       bool EltIsUndef = true;
7851       for (unsigned j = 0; j != NumInputsPerOutput; ++j) {
7852         // Shift the previously computed bits over.
7853         NewBits <<= SrcBitSize;
7854         SDValue Op = BV->getOperand(i+ (isLE ? (NumInputsPerOutput-j-1) : j));
7855         if (Op.isUndef()) continue;
7856         EltIsUndef = false;
7857 
7858         NewBits |= cast<ConstantSDNode>(Op)->getAPIntValue().
7859                    zextOrTrunc(SrcBitSize).zext(DstBitSize);
7860       }
7861 
7862       if (EltIsUndef)
7863         Ops.push_back(DAG.getUNDEF(DstEltVT));
7864       else
7865         Ops.push_back(DAG.getConstant(NewBits, DL, DstEltVT));
7866     }
7867 
7868     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT, Ops.size());
7869     return DAG.getBuildVector(VT, DL, Ops);
7870   }
7871 
7872   // Finally, this must be the case where we are shrinking elements: each input
7873   // turns into multiple outputs.
7874   unsigned NumOutputsPerInput = SrcBitSize/DstBitSize;
7875   EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
7876                             NumOutputsPerInput*BV->getNumOperands());
7877   SmallVector<SDValue, 8> Ops;
7878 
7879   for (const SDValue &Op : BV->op_values()) {
7880     if (Op.isUndef()) {
7881       Ops.append(NumOutputsPerInput, DAG.getUNDEF(DstEltVT));
7882       continue;
7883     }
7884 
7885     APInt OpVal = cast<ConstantSDNode>(Op)->
7886                   getAPIntValue().zextOrTrunc(SrcBitSize);
7887 
7888     for (unsigned j = 0; j != NumOutputsPerInput; ++j) {
7889       APInt ThisVal = OpVal.trunc(DstBitSize);
7890       Ops.push_back(DAG.getConstant(ThisVal, DL, DstEltVT));
7891       OpVal = OpVal.lshr(DstBitSize);
7892     }
7893 
7894     // For big endian targets, swap the order of the pieces of each element.
7895     if (DAG.getDataLayout().isBigEndian())
7896       std::reverse(Ops.end()-NumOutputsPerInput, Ops.end());
7897   }
7898 
7899   return DAG.getBuildVector(VT, DL, Ops);
7900 }
7901 
7902 /// Try to perform FMA combining on a given FADD node.
7903 SDValue DAGCombiner::visitFADDForFMACombine(SDNode *N) {
7904   SDValue N0 = N->getOperand(0);
7905   SDValue N1 = N->getOperand(1);
7906   EVT VT = N->getValueType(0);
7907   SDLoc SL(N);
7908 
7909   const TargetOptions &Options = DAG.getTarget().Options;
7910   bool AllowFusion =
7911       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
7912 
7913   // Floating-point multiply-add with intermediate rounding.
7914   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
7915 
7916   // Floating-point multiply-add without intermediate rounding.
7917   bool HasFMA =
7918       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
7919       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
7920 
7921   // No valid opcode, do not combine.
7922   if (!HasFMAD && !HasFMA)
7923     return SDValue();
7924 
7925   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
7926   ;
7927   if (AllowFusion && STI && STI->generateFMAsInMachineCombiner(OptLevel))
7928     return SDValue();
7929 
7930   // Always prefer FMAD to FMA for precision.
7931   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
7932   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
7933   bool LookThroughFPExt = TLI.isFPExtFree(VT);
7934 
7935   // If we have two choices trying to fold (fadd (fmul u, v), (fmul x, y)),
7936   // prefer to fold the multiply with fewer uses.
7937   if (Aggressive && N0.getOpcode() == ISD::FMUL &&
7938       N1.getOpcode() == ISD::FMUL) {
7939     if (N0.getNode()->use_size() > N1.getNode()->use_size())
7940       std::swap(N0, N1);
7941   }
7942 
7943   // fold (fadd (fmul x, y), z) -> (fma x, y, z)
7944   if (N0.getOpcode() == ISD::FMUL &&
7945       (Aggressive || N0->hasOneUse())) {
7946     return DAG.getNode(PreferredFusedOpcode, SL, VT,
7947                        N0.getOperand(0), N0.getOperand(1), N1);
7948   }
7949 
7950   // fold (fadd x, (fmul y, z)) -> (fma y, z, x)
7951   // Note: Commutes FADD operands.
7952   if (N1.getOpcode() == ISD::FMUL &&
7953       (Aggressive || N1->hasOneUse())) {
7954     return DAG.getNode(PreferredFusedOpcode, SL, VT,
7955                        N1.getOperand(0), N1.getOperand(1), N0);
7956   }
7957 
7958   // Look through FP_EXTEND nodes to do more combining.
7959   if (AllowFusion && LookThroughFPExt) {
7960     // fold (fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext y), z)
7961     if (N0.getOpcode() == ISD::FP_EXTEND) {
7962       SDValue N00 = N0.getOperand(0);
7963       if (N00.getOpcode() == ISD::FMUL)
7964         return DAG.getNode(PreferredFusedOpcode, SL, VT,
7965                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7966                                        N00.getOperand(0)),
7967                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7968                                        N00.getOperand(1)), N1);
7969     }
7970 
7971     // fold (fadd x, (fpext (fmul y, z))) -> (fma (fpext y), (fpext z), x)
7972     // Note: Commutes FADD operands.
7973     if (N1.getOpcode() == ISD::FP_EXTEND) {
7974       SDValue N10 = N1.getOperand(0);
7975       if (N10.getOpcode() == ISD::FMUL)
7976         return DAG.getNode(PreferredFusedOpcode, SL, VT,
7977                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7978                                        N10.getOperand(0)),
7979                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7980                                        N10.getOperand(1)), N0);
7981     }
7982   }
7983 
7984   // More folding opportunities when target permits.
7985   if ((AllowFusion || HasFMAD)  && Aggressive) {
7986     // fold (fadd (fma x, y, (fmul u, v)), z) -> (fma x, y (fma u, v, z))
7987     if (N0.getOpcode() == PreferredFusedOpcode &&
7988         N0.getOperand(2).getOpcode() == ISD::FMUL) {
7989       return DAG.getNode(PreferredFusedOpcode, SL, VT,
7990                          N0.getOperand(0), N0.getOperand(1),
7991                          DAG.getNode(PreferredFusedOpcode, SL, VT,
7992                                      N0.getOperand(2).getOperand(0),
7993                                      N0.getOperand(2).getOperand(1),
7994                                      N1));
7995     }
7996 
7997     // fold (fadd x, (fma y, z, (fmul u, v)) -> (fma y, z (fma u, v, x))
7998     if (N1->getOpcode() == PreferredFusedOpcode &&
7999         N1.getOperand(2).getOpcode() == ISD::FMUL) {
8000       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8001                          N1.getOperand(0), N1.getOperand(1),
8002                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8003                                      N1.getOperand(2).getOperand(0),
8004                                      N1.getOperand(2).getOperand(1),
8005                                      N0));
8006     }
8007 
8008     if (AllowFusion && LookThroughFPExt) {
8009       // fold (fadd (fma x, y, (fpext (fmul u, v))), z)
8010       //   -> (fma x, y, (fma (fpext u), (fpext v), z))
8011       auto FoldFAddFMAFPExtFMul = [&] (
8012           SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z) {
8013         return DAG.getNode(PreferredFusedOpcode, SL, VT, X, Y,
8014                            DAG.getNode(PreferredFusedOpcode, SL, VT,
8015                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
8016                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
8017                                        Z));
8018       };
8019       if (N0.getOpcode() == PreferredFusedOpcode) {
8020         SDValue N02 = N0.getOperand(2);
8021         if (N02.getOpcode() == ISD::FP_EXTEND) {
8022           SDValue N020 = N02.getOperand(0);
8023           if (N020.getOpcode() == ISD::FMUL)
8024             return FoldFAddFMAFPExtFMul(N0.getOperand(0), N0.getOperand(1),
8025                                         N020.getOperand(0), N020.getOperand(1),
8026                                         N1);
8027         }
8028       }
8029 
8030       // fold (fadd (fpext (fma x, y, (fmul u, v))), z)
8031       //   -> (fma (fpext x), (fpext y), (fma (fpext u), (fpext v), z))
8032       // FIXME: This turns two single-precision and one double-precision
8033       // operation into two double-precision operations, which might not be
8034       // interesting for all targets, especially GPUs.
8035       auto FoldFAddFPExtFMAFMul = [&] (
8036           SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z) {
8037         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8038                            DAG.getNode(ISD::FP_EXTEND, SL, VT, X),
8039                            DAG.getNode(ISD::FP_EXTEND, SL, VT, Y),
8040                            DAG.getNode(PreferredFusedOpcode, SL, VT,
8041                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
8042                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
8043                                        Z));
8044       };
8045       if (N0.getOpcode() == ISD::FP_EXTEND) {
8046         SDValue N00 = N0.getOperand(0);
8047         if (N00.getOpcode() == PreferredFusedOpcode) {
8048           SDValue N002 = N00.getOperand(2);
8049           if (N002.getOpcode() == ISD::FMUL)
8050             return FoldFAddFPExtFMAFMul(N00.getOperand(0), N00.getOperand(1),
8051                                         N002.getOperand(0), N002.getOperand(1),
8052                                         N1);
8053         }
8054       }
8055 
8056       // fold (fadd x, (fma y, z, (fpext (fmul u, v)))
8057       //   -> (fma y, z, (fma (fpext u), (fpext v), x))
8058       if (N1.getOpcode() == PreferredFusedOpcode) {
8059         SDValue N12 = N1.getOperand(2);
8060         if (N12.getOpcode() == ISD::FP_EXTEND) {
8061           SDValue N120 = N12.getOperand(0);
8062           if (N120.getOpcode() == ISD::FMUL)
8063             return FoldFAddFMAFPExtFMul(N1.getOperand(0), N1.getOperand(1),
8064                                         N120.getOperand(0), N120.getOperand(1),
8065                                         N0);
8066         }
8067       }
8068 
8069       // fold (fadd x, (fpext (fma y, z, (fmul u, v)))
8070       //   -> (fma (fpext y), (fpext z), (fma (fpext u), (fpext v), x))
8071       // FIXME: This turns two single-precision and one double-precision
8072       // operation into two double-precision operations, which might not be
8073       // interesting for all targets, especially GPUs.
8074       if (N1.getOpcode() == ISD::FP_EXTEND) {
8075         SDValue N10 = N1.getOperand(0);
8076         if (N10.getOpcode() == PreferredFusedOpcode) {
8077           SDValue N102 = N10.getOperand(2);
8078           if (N102.getOpcode() == ISD::FMUL)
8079             return FoldFAddFPExtFMAFMul(N10.getOperand(0), N10.getOperand(1),
8080                                         N102.getOperand(0), N102.getOperand(1),
8081                                         N0);
8082         }
8083       }
8084     }
8085   }
8086 
8087   return SDValue();
8088 }
8089 
8090 /// Try to perform FMA combining on a given FSUB node.
8091 SDValue DAGCombiner::visitFSUBForFMACombine(SDNode *N) {
8092   SDValue N0 = N->getOperand(0);
8093   SDValue N1 = N->getOperand(1);
8094   EVT VT = N->getValueType(0);
8095   SDLoc SL(N);
8096 
8097   const TargetOptions &Options = DAG.getTarget().Options;
8098   bool AllowFusion =
8099       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
8100 
8101   // Floating-point multiply-add with intermediate rounding.
8102   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
8103 
8104   // Floating-point multiply-add without intermediate rounding.
8105   bool HasFMA =
8106       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
8107       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
8108 
8109   // No valid opcode, do not combine.
8110   if (!HasFMAD && !HasFMA)
8111     return SDValue();
8112 
8113   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
8114   if (AllowFusion && STI && STI->generateFMAsInMachineCombiner(OptLevel))
8115     return SDValue();
8116 
8117   // Always prefer FMAD to FMA for precision.
8118   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
8119   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
8120   bool LookThroughFPExt = TLI.isFPExtFree(VT);
8121 
8122   // fold (fsub (fmul x, y), z) -> (fma x, y, (fneg z))
8123   if (N0.getOpcode() == ISD::FMUL &&
8124       (Aggressive || N0->hasOneUse())) {
8125     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8126                        N0.getOperand(0), N0.getOperand(1),
8127                        DAG.getNode(ISD::FNEG, SL, VT, N1));
8128   }
8129 
8130   // fold (fsub x, (fmul y, z)) -> (fma (fneg y), z, x)
8131   // Note: Commutes FSUB operands.
8132   if (N1.getOpcode() == ISD::FMUL &&
8133       (Aggressive || N1->hasOneUse()))
8134     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8135                        DAG.getNode(ISD::FNEG, SL, VT,
8136                                    N1.getOperand(0)),
8137                        N1.getOperand(1), N0);
8138 
8139   // fold (fsub (fneg (fmul, x, y)), z) -> (fma (fneg x), y, (fneg z))
8140   if (N0.getOpcode() == ISD::FNEG &&
8141       N0.getOperand(0).getOpcode() == ISD::FMUL &&
8142       (Aggressive || (N0->hasOneUse() && N0.getOperand(0).hasOneUse()))) {
8143     SDValue N00 = N0.getOperand(0).getOperand(0);
8144     SDValue N01 = N0.getOperand(0).getOperand(1);
8145     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8146                        DAG.getNode(ISD::FNEG, SL, VT, N00), N01,
8147                        DAG.getNode(ISD::FNEG, SL, VT, N1));
8148   }
8149 
8150   // Look through FP_EXTEND nodes to do more combining.
8151   if (AllowFusion && LookThroughFPExt) {
8152     // fold (fsub (fpext (fmul x, y)), z)
8153     //   -> (fma (fpext x), (fpext y), (fneg z))
8154     if (N0.getOpcode() == ISD::FP_EXTEND) {
8155       SDValue N00 = N0.getOperand(0);
8156       if (N00.getOpcode() == ISD::FMUL)
8157         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8158                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8159                                        N00.getOperand(0)),
8160                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8161                                        N00.getOperand(1)),
8162                            DAG.getNode(ISD::FNEG, SL, VT, N1));
8163     }
8164 
8165     // fold (fsub x, (fpext (fmul y, z)))
8166     //   -> (fma (fneg (fpext y)), (fpext z), x)
8167     // Note: Commutes FSUB operands.
8168     if (N1.getOpcode() == ISD::FP_EXTEND) {
8169       SDValue N10 = N1.getOperand(0);
8170       if (N10.getOpcode() == ISD::FMUL)
8171         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8172                            DAG.getNode(ISD::FNEG, SL, VT,
8173                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
8174                                                    N10.getOperand(0))),
8175                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8176                                        N10.getOperand(1)),
8177                            N0);
8178     }
8179 
8180     // fold (fsub (fpext (fneg (fmul, x, y))), z)
8181     //   -> (fneg (fma (fpext x), (fpext y), z))
8182     // Note: This could be removed with appropriate canonicalization of the
8183     // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
8184     // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
8185     // from implementing the canonicalization in visitFSUB.
8186     if (N0.getOpcode() == ISD::FP_EXTEND) {
8187       SDValue N00 = N0.getOperand(0);
8188       if (N00.getOpcode() == ISD::FNEG) {
8189         SDValue N000 = N00.getOperand(0);
8190         if (N000.getOpcode() == ISD::FMUL) {
8191           return DAG.getNode(ISD::FNEG, SL, VT,
8192                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8193                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8194                                                      N000.getOperand(0)),
8195                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8196                                                      N000.getOperand(1)),
8197                                          N1));
8198         }
8199       }
8200     }
8201 
8202     // fold (fsub (fneg (fpext (fmul, x, y))), z)
8203     //   -> (fneg (fma (fpext x)), (fpext y), z)
8204     // Note: This could be removed with appropriate canonicalization of the
8205     // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
8206     // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
8207     // from implementing the canonicalization in visitFSUB.
8208     if (N0.getOpcode() == ISD::FNEG) {
8209       SDValue N00 = N0.getOperand(0);
8210       if (N00.getOpcode() == ISD::FP_EXTEND) {
8211         SDValue N000 = N00.getOperand(0);
8212         if (N000.getOpcode() == ISD::FMUL) {
8213           return DAG.getNode(ISD::FNEG, SL, VT,
8214                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8215                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8216                                                      N000.getOperand(0)),
8217                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8218                                                      N000.getOperand(1)),
8219                                          N1));
8220         }
8221       }
8222     }
8223 
8224   }
8225 
8226   // More folding opportunities when target permits.
8227   if ((AllowFusion || HasFMAD) && Aggressive) {
8228     // fold (fsub (fma x, y, (fmul u, v)), z)
8229     //   -> (fma x, y (fma u, v, (fneg z)))
8230     if (N0.getOpcode() == PreferredFusedOpcode &&
8231         N0.getOperand(2).getOpcode() == ISD::FMUL) {
8232       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8233                          N0.getOperand(0), N0.getOperand(1),
8234                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8235                                      N0.getOperand(2).getOperand(0),
8236                                      N0.getOperand(2).getOperand(1),
8237                                      DAG.getNode(ISD::FNEG, SL, VT,
8238                                                  N1)));
8239     }
8240 
8241     // fold (fsub x, (fma y, z, (fmul u, v)))
8242     //   -> (fma (fneg y), z, (fma (fneg u), v, x))
8243     if (N1.getOpcode() == PreferredFusedOpcode &&
8244         N1.getOperand(2).getOpcode() == ISD::FMUL) {
8245       SDValue N20 = N1.getOperand(2).getOperand(0);
8246       SDValue N21 = N1.getOperand(2).getOperand(1);
8247       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8248                          DAG.getNode(ISD::FNEG, SL, VT,
8249                                      N1.getOperand(0)),
8250                          N1.getOperand(1),
8251                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8252                                      DAG.getNode(ISD::FNEG, SL, VT, N20),
8253 
8254                                      N21, N0));
8255     }
8256 
8257     if (AllowFusion && LookThroughFPExt) {
8258       // fold (fsub (fma x, y, (fpext (fmul u, v))), z)
8259       //   -> (fma x, y (fma (fpext u), (fpext v), (fneg z)))
8260       if (N0.getOpcode() == PreferredFusedOpcode) {
8261         SDValue N02 = N0.getOperand(2);
8262         if (N02.getOpcode() == ISD::FP_EXTEND) {
8263           SDValue N020 = N02.getOperand(0);
8264           if (N020.getOpcode() == ISD::FMUL)
8265             return DAG.getNode(PreferredFusedOpcode, SL, VT,
8266                                N0.getOperand(0), N0.getOperand(1),
8267                                DAG.getNode(PreferredFusedOpcode, SL, VT,
8268                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8269                                                        N020.getOperand(0)),
8270                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8271                                                        N020.getOperand(1)),
8272                                            DAG.getNode(ISD::FNEG, SL, VT,
8273                                                        N1)));
8274         }
8275       }
8276 
8277       // fold (fsub (fpext (fma x, y, (fmul u, v))), z)
8278       //   -> (fma (fpext x), (fpext y),
8279       //           (fma (fpext u), (fpext v), (fneg z)))
8280       // FIXME: This turns two single-precision and one double-precision
8281       // operation into two double-precision operations, which might not be
8282       // interesting for all targets, especially GPUs.
8283       if (N0.getOpcode() == ISD::FP_EXTEND) {
8284         SDValue N00 = N0.getOperand(0);
8285         if (N00.getOpcode() == PreferredFusedOpcode) {
8286           SDValue N002 = N00.getOperand(2);
8287           if (N002.getOpcode() == ISD::FMUL)
8288             return DAG.getNode(PreferredFusedOpcode, SL, VT,
8289                                DAG.getNode(ISD::FP_EXTEND, SL, VT,
8290                                            N00.getOperand(0)),
8291                                DAG.getNode(ISD::FP_EXTEND, SL, VT,
8292                                            N00.getOperand(1)),
8293                                DAG.getNode(PreferredFusedOpcode, SL, VT,
8294                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8295                                                        N002.getOperand(0)),
8296                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8297                                                        N002.getOperand(1)),
8298                                            DAG.getNode(ISD::FNEG, SL, VT,
8299                                                        N1)));
8300         }
8301       }
8302 
8303       // fold (fsub x, (fma y, z, (fpext (fmul u, v))))
8304       //   -> (fma (fneg y), z, (fma (fneg (fpext u)), (fpext v), x))
8305       if (N1.getOpcode() == PreferredFusedOpcode &&
8306         N1.getOperand(2).getOpcode() == ISD::FP_EXTEND) {
8307         SDValue N120 = N1.getOperand(2).getOperand(0);
8308         if (N120.getOpcode() == ISD::FMUL) {
8309           SDValue N1200 = N120.getOperand(0);
8310           SDValue N1201 = N120.getOperand(1);
8311           return DAG.getNode(PreferredFusedOpcode, SL, VT,
8312                              DAG.getNode(ISD::FNEG, SL, VT, N1.getOperand(0)),
8313                              N1.getOperand(1),
8314                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8315                                          DAG.getNode(ISD::FNEG, SL, VT,
8316                                              DAG.getNode(ISD::FP_EXTEND, SL,
8317                                                          VT, N1200)),
8318                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8319                                                      N1201),
8320                                          N0));
8321         }
8322       }
8323 
8324       // fold (fsub x, (fpext (fma y, z, (fmul u, v))))
8325       //   -> (fma (fneg (fpext y)), (fpext z),
8326       //           (fma (fneg (fpext u)), (fpext v), x))
8327       // FIXME: This turns two single-precision and one double-precision
8328       // operation into two double-precision operations, which might not be
8329       // interesting for all targets, especially GPUs.
8330       if (N1.getOpcode() == ISD::FP_EXTEND &&
8331         N1.getOperand(0).getOpcode() == PreferredFusedOpcode) {
8332         SDValue N100 = N1.getOperand(0).getOperand(0);
8333         SDValue N101 = N1.getOperand(0).getOperand(1);
8334         SDValue N102 = N1.getOperand(0).getOperand(2);
8335         if (N102.getOpcode() == ISD::FMUL) {
8336           SDValue N1020 = N102.getOperand(0);
8337           SDValue N1021 = N102.getOperand(1);
8338           return DAG.getNode(PreferredFusedOpcode, SL, VT,
8339                              DAG.getNode(ISD::FNEG, SL, VT,
8340                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8341                                                      N100)),
8342                              DAG.getNode(ISD::FP_EXTEND, SL, VT, N101),
8343                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8344                                          DAG.getNode(ISD::FNEG, SL, VT,
8345                                              DAG.getNode(ISD::FP_EXTEND, SL,
8346                                                          VT, N1020)),
8347                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8348                                                      N1021),
8349                                          N0));
8350         }
8351       }
8352     }
8353   }
8354 
8355   return SDValue();
8356 }
8357 
8358 /// Try to perform FMA combining on a given FMUL node.
8359 SDValue DAGCombiner::visitFMULForFMACombine(SDNode *N) {
8360   SDValue N0 = N->getOperand(0);
8361   SDValue N1 = N->getOperand(1);
8362   EVT VT = N->getValueType(0);
8363   SDLoc SL(N);
8364 
8365   assert(N->getOpcode() == ISD::FMUL && "Expected FMUL Operation");
8366 
8367   const TargetOptions &Options = DAG.getTarget().Options;
8368   bool AllowFusion =
8369       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
8370 
8371   // Floating-point multiply-add with intermediate rounding.
8372   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
8373 
8374   // Floating-point multiply-add without intermediate rounding.
8375   bool HasFMA =
8376       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
8377       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
8378 
8379   // No valid opcode, do not combine.
8380   if (!HasFMAD && !HasFMA)
8381     return SDValue();
8382 
8383   // Always prefer FMAD to FMA for precision.
8384   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
8385   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
8386 
8387   // fold (fmul (fadd x, +1.0), y) -> (fma x, y, y)
8388   // fold (fmul (fadd x, -1.0), y) -> (fma x, y, (fneg y))
8389   auto FuseFADD = [&](SDValue X, SDValue Y) {
8390     if (X.getOpcode() == ISD::FADD && (Aggressive || X->hasOneUse())) {
8391       auto XC1 = isConstOrConstSplatFP(X.getOperand(1));
8392       if (XC1 && XC1->isExactlyValue(+1.0))
8393         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y);
8394       if (XC1 && XC1->isExactlyValue(-1.0))
8395         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
8396                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8397     }
8398     return SDValue();
8399   };
8400 
8401   if (SDValue FMA = FuseFADD(N0, N1))
8402     return FMA;
8403   if (SDValue FMA = FuseFADD(N1, N0))
8404     return FMA;
8405 
8406   // fold (fmul (fsub +1.0, x), y) -> (fma (fneg x), y, y)
8407   // fold (fmul (fsub -1.0, x), y) -> (fma (fneg x), y, (fneg y))
8408   // fold (fmul (fsub x, +1.0), y) -> (fma x, y, (fneg y))
8409   // fold (fmul (fsub x, -1.0), y) -> (fma x, y, y)
8410   auto FuseFSUB = [&](SDValue X, SDValue Y) {
8411     if (X.getOpcode() == ISD::FSUB && (Aggressive || X->hasOneUse())) {
8412       auto XC0 = isConstOrConstSplatFP(X.getOperand(0));
8413       if (XC0 && XC0->isExactlyValue(+1.0))
8414         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8415                            DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
8416                            Y);
8417       if (XC0 && XC0->isExactlyValue(-1.0))
8418         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8419                            DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
8420                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8421 
8422       auto XC1 = isConstOrConstSplatFP(X.getOperand(1));
8423       if (XC1 && XC1->isExactlyValue(+1.0))
8424         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
8425                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8426       if (XC1 && XC1->isExactlyValue(-1.0))
8427         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y);
8428     }
8429     return SDValue();
8430   };
8431 
8432   if (SDValue FMA = FuseFSUB(N0, N1))
8433     return FMA;
8434   if (SDValue FMA = FuseFSUB(N1, N0))
8435     return FMA;
8436 
8437   return SDValue();
8438 }
8439 
8440 SDValue DAGCombiner::visitFADD(SDNode *N) {
8441   SDValue N0 = N->getOperand(0);
8442   SDValue N1 = N->getOperand(1);
8443   bool N0CFP = isConstantFPBuildVectorOrConstantFP(N0);
8444   bool N1CFP = isConstantFPBuildVectorOrConstantFP(N1);
8445   EVT VT = N->getValueType(0);
8446   SDLoc DL(N);
8447   const TargetOptions &Options = DAG.getTarget().Options;
8448   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8449 
8450   // fold vector ops
8451   if (VT.isVector())
8452     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8453       return FoldedVOp;
8454 
8455   // fold (fadd c1, c2) -> c1 + c2
8456   if (N0CFP && N1CFP)
8457     return DAG.getNode(ISD::FADD, DL, VT, N0, N1, Flags);
8458 
8459   // canonicalize constant to RHS
8460   if (N0CFP && !N1CFP)
8461     return DAG.getNode(ISD::FADD, DL, VT, N1, N0, Flags);
8462 
8463   // fold (fadd A, (fneg B)) -> (fsub A, B)
8464   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
8465       isNegatibleForFree(N1, LegalOperations, TLI, &Options) == 2)
8466     return DAG.getNode(ISD::FSUB, DL, VT, N0,
8467                        GetNegatedExpression(N1, DAG, LegalOperations), Flags);
8468 
8469   // fold (fadd (fneg A), B) -> (fsub B, A)
8470   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
8471       isNegatibleForFree(N0, LegalOperations, TLI, &Options) == 2)
8472     return DAG.getNode(ISD::FSUB, DL, VT, N1,
8473                        GetNegatedExpression(N0, DAG, LegalOperations), Flags);
8474 
8475   // If 'unsafe math' is enabled, fold lots of things.
8476   if (Options.UnsafeFPMath) {
8477     // No FP constant should be created after legalization as Instruction
8478     // Selection pass has a hard time dealing with FP constants.
8479     bool AllowNewConst = (Level < AfterLegalizeDAG);
8480 
8481     // fold (fadd A, 0) -> A
8482     if (ConstantFPSDNode *N1C = isConstOrConstSplatFP(N1))
8483       if (N1C->isZero())
8484         return N0;
8485 
8486     // fold (fadd (fadd x, c1), c2) -> (fadd x, (fadd c1, c2))
8487     if (N1CFP && N0.getOpcode() == ISD::FADD && N0.getNode()->hasOneUse() &&
8488         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1)))
8489       return DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(0),
8490                          DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1), N1,
8491                                      Flags),
8492                          Flags);
8493 
8494     // If allowed, fold (fadd (fneg x), x) -> 0.0
8495     if (AllowNewConst && N0.getOpcode() == ISD::FNEG && N0.getOperand(0) == N1)
8496       return DAG.getConstantFP(0.0, DL, VT);
8497 
8498     // If allowed, fold (fadd x, (fneg x)) -> 0.0
8499     if (AllowNewConst && N1.getOpcode() == ISD::FNEG && N1.getOperand(0) == N0)
8500       return DAG.getConstantFP(0.0, DL, VT);
8501 
8502     // We can fold chains of FADD's of the same value into multiplications.
8503     // This transform is not safe in general because we are reducing the number
8504     // of rounding steps.
8505     if (TLI.isOperationLegalOrCustom(ISD::FMUL, VT) && !N0CFP && !N1CFP) {
8506       if (N0.getOpcode() == ISD::FMUL) {
8507         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
8508         bool CFP01 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(1));
8509 
8510         // (fadd (fmul x, c), x) -> (fmul x, c+1)
8511         if (CFP01 && !CFP00 && N0.getOperand(0) == N1) {
8512           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
8513                                        DAG.getConstantFP(1.0, DL, VT), Flags);
8514           return DAG.getNode(ISD::FMUL, DL, VT, N1, NewCFP, Flags);
8515         }
8516 
8517         // (fadd (fmul x, c), (fadd x, x)) -> (fmul x, c+2)
8518         if (CFP01 && !CFP00 && N1.getOpcode() == ISD::FADD &&
8519             N1.getOperand(0) == N1.getOperand(1) &&
8520             N0.getOperand(0) == N1.getOperand(0)) {
8521           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
8522                                        DAG.getConstantFP(2.0, DL, VT), Flags);
8523           return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), NewCFP, Flags);
8524         }
8525       }
8526 
8527       if (N1.getOpcode() == ISD::FMUL) {
8528         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
8529         bool CFP11 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(1));
8530 
8531         // (fadd x, (fmul x, c)) -> (fmul x, c+1)
8532         if (CFP11 && !CFP10 && N1.getOperand(0) == N0) {
8533           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
8534                                        DAG.getConstantFP(1.0, DL, VT), Flags);
8535           return DAG.getNode(ISD::FMUL, DL, VT, N0, NewCFP, Flags);
8536         }
8537 
8538         // (fadd (fadd x, x), (fmul x, c)) -> (fmul x, c+2)
8539         if (CFP11 && !CFP10 && N0.getOpcode() == ISD::FADD &&
8540             N0.getOperand(0) == N0.getOperand(1) &&
8541             N1.getOperand(0) == N0.getOperand(0)) {
8542           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
8543                                        DAG.getConstantFP(2.0, DL, VT), Flags);
8544           return DAG.getNode(ISD::FMUL, DL, VT, N1.getOperand(0), NewCFP, Flags);
8545         }
8546       }
8547 
8548       if (N0.getOpcode() == ISD::FADD && AllowNewConst) {
8549         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
8550         // (fadd (fadd x, x), x) -> (fmul x, 3.0)
8551         if (!CFP00 && N0.getOperand(0) == N0.getOperand(1) &&
8552             (N0.getOperand(0) == N1)) {
8553           return DAG.getNode(ISD::FMUL, DL, VT,
8554                              N1, DAG.getConstantFP(3.0, DL, VT), Flags);
8555         }
8556       }
8557 
8558       if (N1.getOpcode() == ISD::FADD && AllowNewConst) {
8559         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
8560         // (fadd x, (fadd x, x)) -> (fmul x, 3.0)
8561         if (!CFP10 && N1.getOperand(0) == N1.getOperand(1) &&
8562             N1.getOperand(0) == N0) {
8563           return DAG.getNode(ISD::FMUL, DL, VT,
8564                              N0, DAG.getConstantFP(3.0, DL, VT), Flags);
8565         }
8566       }
8567 
8568       // (fadd (fadd x, x), (fadd x, x)) -> (fmul x, 4.0)
8569       if (AllowNewConst &&
8570           N0.getOpcode() == ISD::FADD && N1.getOpcode() == ISD::FADD &&
8571           N0.getOperand(0) == N0.getOperand(1) &&
8572           N1.getOperand(0) == N1.getOperand(1) &&
8573           N0.getOperand(0) == N1.getOperand(0)) {
8574         return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0),
8575                            DAG.getConstantFP(4.0, DL, VT), Flags);
8576       }
8577     }
8578   } // enable-unsafe-fp-math
8579 
8580   // FADD -> FMA combines:
8581   if (SDValue Fused = visitFADDForFMACombine(N)) {
8582     AddToWorklist(Fused.getNode());
8583     return Fused;
8584   }
8585   return SDValue();
8586 }
8587 
8588 SDValue DAGCombiner::visitFSUB(SDNode *N) {
8589   SDValue N0 = N->getOperand(0);
8590   SDValue N1 = N->getOperand(1);
8591   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
8592   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
8593   EVT VT = N->getValueType(0);
8594   SDLoc DL(N);
8595   const TargetOptions &Options = DAG.getTarget().Options;
8596   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8597 
8598   // fold vector ops
8599   if (VT.isVector())
8600     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8601       return FoldedVOp;
8602 
8603   // fold (fsub c1, c2) -> c1-c2
8604   if (N0CFP && N1CFP)
8605     return DAG.getNode(ISD::FSUB, DL, VT, N0, N1, Flags);
8606 
8607   // fold (fsub A, (fneg B)) -> (fadd A, B)
8608   if (isNegatibleForFree(N1, LegalOperations, TLI, &Options))
8609     return DAG.getNode(ISD::FADD, DL, VT, N0,
8610                        GetNegatedExpression(N1, DAG, LegalOperations), Flags);
8611 
8612   // If 'unsafe math' is enabled, fold lots of things.
8613   if (Options.UnsafeFPMath) {
8614     // (fsub A, 0) -> A
8615     if (N1CFP && N1CFP->isZero())
8616       return N0;
8617 
8618     // (fsub 0, B) -> -B
8619     if (N0CFP && N0CFP->isZero()) {
8620       if (isNegatibleForFree(N1, LegalOperations, TLI, &Options))
8621         return GetNegatedExpression(N1, DAG, LegalOperations);
8622       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
8623         return DAG.getNode(ISD::FNEG, DL, VT, N1);
8624     }
8625 
8626     // (fsub x, x) -> 0.0
8627     if (N0 == N1)
8628       return DAG.getConstantFP(0.0f, DL, VT);
8629 
8630     // (fsub x, (fadd x, y)) -> (fneg y)
8631     // (fsub x, (fadd y, x)) -> (fneg y)
8632     if (N1.getOpcode() == ISD::FADD) {
8633       SDValue N10 = N1->getOperand(0);
8634       SDValue N11 = N1->getOperand(1);
8635 
8636       if (N10 == N0 && isNegatibleForFree(N11, LegalOperations, TLI, &Options))
8637         return GetNegatedExpression(N11, DAG, LegalOperations);
8638 
8639       if (N11 == N0 && isNegatibleForFree(N10, LegalOperations, TLI, &Options))
8640         return GetNegatedExpression(N10, DAG, LegalOperations);
8641     }
8642   }
8643 
8644   // FSUB -> FMA combines:
8645   if (SDValue Fused = visitFSUBForFMACombine(N)) {
8646     AddToWorklist(Fused.getNode());
8647     return Fused;
8648   }
8649 
8650   return SDValue();
8651 }
8652 
8653 SDValue DAGCombiner::visitFMUL(SDNode *N) {
8654   SDValue N0 = N->getOperand(0);
8655   SDValue N1 = N->getOperand(1);
8656   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
8657   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
8658   EVT VT = N->getValueType(0);
8659   SDLoc DL(N);
8660   const TargetOptions &Options = DAG.getTarget().Options;
8661   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8662 
8663   // fold vector ops
8664   if (VT.isVector()) {
8665     // This just handles C1 * C2 for vectors. Other vector folds are below.
8666     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8667       return FoldedVOp;
8668   }
8669 
8670   // fold (fmul c1, c2) -> c1*c2
8671   if (N0CFP && N1CFP)
8672     return DAG.getNode(ISD::FMUL, DL, VT, N0, N1, Flags);
8673 
8674   // canonicalize constant to RHS
8675   if (isConstantFPBuildVectorOrConstantFP(N0) &&
8676      !isConstantFPBuildVectorOrConstantFP(N1))
8677     return DAG.getNode(ISD::FMUL, DL, VT, N1, N0, Flags);
8678 
8679   // fold (fmul A, 1.0) -> A
8680   if (N1CFP && N1CFP->isExactlyValue(1.0))
8681     return N0;
8682 
8683   if (Options.UnsafeFPMath) {
8684     // fold (fmul A, 0) -> 0
8685     if (N1CFP && N1CFP->isZero())
8686       return N1;
8687 
8688     // fold (fmul (fmul x, c1), c2) -> (fmul x, (fmul c1, c2))
8689     if (N0.getOpcode() == ISD::FMUL) {
8690       // Fold scalars or any vector constants (not just splats).
8691       // This fold is done in general by InstCombine, but extra fmul insts
8692       // may have been generated during lowering.
8693       SDValue N00 = N0.getOperand(0);
8694       SDValue N01 = N0.getOperand(1);
8695       auto *BV1 = dyn_cast<BuildVectorSDNode>(N1);
8696       auto *BV00 = dyn_cast<BuildVectorSDNode>(N00);
8697       auto *BV01 = dyn_cast<BuildVectorSDNode>(N01);
8698 
8699       // Check 1: Make sure that the first operand of the inner multiply is NOT
8700       // a constant. Otherwise, we may induce infinite looping.
8701       if (!(isConstOrConstSplatFP(N00) || (BV00 && BV00->isConstant()))) {
8702         // Check 2: Make sure that the second operand of the inner multiply and
8703         // the second operand of the outer multiply are constants.
8704         if ((N1CFP && isConstOrConstSplatFP(N01)) ||
8705             (BV1 && BV01 && BV1->isConstant() && BV01->isConstant())) {
8706           SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, N01, N1, Flags);
8707           return DAG.getNode(ISD::FMUL, DL, VT, N00, MulConsts, Flags);
8708         }
8709       }
8710     }
8711 
8712     // fold (fmul (fadd x, x), c) -> (fmul x, (fmul 2.0, c))
8713     // Undo the fmul 2.0, x -> fadd x, x transformation, since if it occurs
8714     // during an early run of DAGCombiner can prevent folding with fmuls
8715     // inserted during lowering.
8716     if (N0.getOpcode() == ISD::FADD &&
8717         (N0.getOperand(0) == N0.getOperand(1)) &&
8718         N0.hasOneUse()) {
8719       const SDValue Two = DAG.getConstantFP(2.0, DL, VT);
8720       SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, Two, N1, Flags);
8721       return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), MulConsts, Flags);
8722     }
8723   }
8724 
8725   // fold (fmul X, 2.0) -> (fadd X, X)
8726   if (N1CFP && N1CFP->isExactlyValue(+2.0))
8727     return DAG.getNode(ISD::FADD, DL, VT, N0, N0, Flags);
8728 
8729   // fold (fmul X, -1.0) -> (fneg X)
8730   if (N1CFP && N1CFP->isExactlyValue(-1.0))
8731     if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
8732       return DAG.getNode(ISD::FNEG, DL, VT, N0);
8733 
8734   // fold (fmul (fneg X), (fneg Y)) -> (fmul X, Y)
8735   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) {
8736     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) {
8737       // Both can be negated for free, check to see if at least one is cheaper
8738       // negated.
8739       if (LHSNeg == 2 || RHSNeg == 2)
8740         return DAG.getNode(ISD::FMUL, DL, VT,
8741                            GetNegatedExpression(N0, DAG, LegalOperations),
8742                            GetNegatedExpression(N1, DAG, LegalOperations),
8743                            Flags);
8744     }
8745   }
8746 
8747   // FMUL -> FMA combines:
8748   if (SDValue Fused = visitFMULForFMACombine(N)) {
8749     AddToWorklist(Fused.getNode());
8750     return Fused;
8751   }
8752 
8753   return SDValue();
8754 }
8755 
8756 SDValue DAGCombiner::visitFMA(SDNode *N) {
8757   SDValue N0 = N->getOperand(0);
8758   SDValue N1 = N->getOperand(1);
8759   SDValue N2 = N->getOperand(2);
8760   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
8761   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
8762   EVT VT = N->getValueType(0);
8763   SDLoc DL(N);
8764   const TargetOptions &Options = DAG.getTarget().Options;
8765 
8766   // Constant fold FMA.
8767   if (isa<ConstantFPSDNode>(N0) &&
8768       isa<ConstantFPSDNode>(N1) &&
8769       isa<ConstantFPSDNode>(N2)) {
8770     return DAG.getNode(ISD::FMA, DL, VT, N0, N1, N2);
8771   }
8772 
8773   if (Options.UnsafeFPMath) {
8774     if (N0CFP && N0CFP->isZero())
8775       return N2;
8776     if (N1CFP && N1CFP->isZero())
8777       return N2;
8778   }
8779   // TODO: The FMA node should have flags that propagate to these nodes.
8780   if (N0CFP && N0CFP->isExactlyValue(1.0))
8781     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N1, N2);
8782   if (N1CFP && N1CFP->isExactlyValue(1.0))
8783     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N0, N2);
8784 
8785   // Canonicalize (fma c, x, y) -> (fma x, c, y)
8786   if (isConstantFPBuildVectorOrConstantFP(N0) &&
8787      !isConstantFPBuildVectorOrConstantFP(N1))
8788     return DAG.getNode(ISD::FMA, SDLoc(N), VT, N1, N0, N2);
8789 
8790   // TODO: FMA nodes should have flags that propagate to the created nodes.
8791   // For now, create a Flags object for use with all unsafe math transforms.
8792   SDNodeFlags Flags;
8793   Flags.setUnsafeAlgebra(true);
8794 
8795   if (Options.UnsafeFPMath) {
8796     // (fma x, c1, (fmul x, c2)) -> (fmul x, c1+c2)
8797     if (N2.getOpcode() == ISD::FMUL && N0 == N2.getOperand(0) &&
8798         isConstantFPBuildVectorOrConstantFP(N1) &&
8799         isConstantFPBuildVectorOrConstantFP(N2.getOperand(1))) {
8800       return DAG.getNode(ISD::FMUL, DL, VT, N0,
8801                          DAG.getNode(ISD::FADD, DL, VT, N1, N2.getOperand(1),
8802                                      &Flags), &Flags);
8803     }
8804 
8805     // (fma (fmul x, c1), c2, y) -> (fma x, c1*c2, y)
8806     if (N0.getOpcode() == ISD::FMUL &&
8807         isConstantFPBuildVectorOrConstantFP(N1) &&
8808         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1))) {
8809       return DAG.getNode(ISD::FMA, DL, VT,
8810                          N0.getOperand(0),
8811                          DAG.getNode(ISD::FMUL, DL, VT, N1, N0.getOperand(1),
8812                                      &Flags),
8813                          N2);
8814     }
8815   }
8816 
8817   // (fma x, 1, y) -> (fadd x, y)
8818   // (fma x, -1, y) -> (fadd (fneg x), y)
8819   if (N1CFP) {
8820     if (N1CFP->isExactlyValue(1.0))
8821       // TODO: The FMA node should have flags that propagate to this node.
8822       return DAG.getNode(ISD::FADD, DL, VT, N0, N2);
8823 
8824     if (N1CFP->isExactlyValue(-1.0) &&
8825         (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))) {
8826       SDValue RHSNeg = DAG.getNode(ISD::FNEG, DL, VT, N0);
8827       AddToWorklist(RHSNeg.getNode());
8828       // TODO: The FMA node should have flags that propagate to this node.
8829       return DAG.getNode(ISD::FADD, DL, VT, N2, RHSNeg);
8830     }
8831   }
8832 
8833   if (Options.UnsafeFPMath) {
8834     // (fma x, c, x) -> (fmul x, (c+1))
8835     if (N1CFP && N0 == N2) {
8836       return DAG.getNode(ISD::FMUL, DL, VT, N0,
8837                          DAG.getNode(ISD::FADD, DL, VT, N1,
8838                                      DAG.getConstantFP(1.0, DL, VT), &Flags),
8839                          &Flags);
8840     }
8841 
8842     // (fma x, c, (fneg x)) -> (fmul x, (c-1))
8843     if (N1CFP && N2.getOpcode() == ISD::FNEG && N2.getOperand(0) == N0) {
8844       return DAG.getNode(ISD::FMUL, DL, VT, N0,
8845                          DAG.getNode(ISD::FADD, DL, VT, N1,
8846                                      DAG.getConstantFP(-1.0, DL, VT), &Flags),
8847                          &Flags);
8848     }
8849   }
8850 
8851   return SDValue();
8852 }
8853 
8854 // Combine multiple FDIVs with the same divisor into multiple FMULs by the
8855 // reciprocal.
8856 // E.g., (a / D; b / D;) -> (recip = 1.0 / D; a * recip; b * recip)
8857 // Notice that this is not always beneficial. One reason is different target
8858 // may have different costs for FDIV and FMUL, so sometimes the cost of two
8859 // FDIVs may be lower than the cost of one FDIV and two FMULs. Another reason
8860 // is the critical path is increased from "one FDIV" to "one FDIV + one FMUL".
8861 SDValue DAGCombiner::combineRepeatedFPDivisors(SDNode *N) {
8862   bool UnsafeMath = DAG.getTarget().Options.UnsafeFPMath;
8863   const SDNodeFlags *Flags = N->getFlags();
8864   if (!UnsafeMath && !Flags->hasAllowReciprocal())
8865     return SDValue();
8866 
8867   // Skip if current node is a reciprocal.
8868   SDValue N0 = N->getOperand(0);
8869   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
8870   if (N0CFP && N0CFP->isExactlyValue(1.0))
8871     return SDValue();
8872 
8873   // Exit early if the target does not want this transform or if there can't
8874   // possibly be enough uses of the divisor to make the transform worthwhile.
8875   SDValue N1 = N->getOperand(1);
8876   unsigned MinUses = TLI.combineRepeatedFPDivisors();
8877   if (!MinUses || N1->use_size() < MinUses)
8878     return SDValue();
8879 
8880   // Find all FDIV users of the same divisor.
8881   // Use a set because duplicates may be present in the user list.
8882   SetVector<SDNode *> Users;
8883   for (auto *U : N1->uses()) {
8884     if (U->getOpcode() == ISD::FDIV && U->getOperand(1) == N1) {
8885       // This division is eligible for optimization only if global unsafe math
8886       // is enabled or if this division allows reciprocal formation.
8887       if (UnsafeMath || U->getFlags()->hasAllowReciprocal())
8888         Users.insert(U);
8889     }
8890   }
8891 
8892   // Now that we have the actual number of divisor uses, make sure it meets
8893   // the minimum threshold specified by the target.
8894   if (Users.size() < MinUses)
8895     return SDValue();
8896 
8897   EVT VT = N->getValueType(0);
8898   SDLoc DL(N);
8899   SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
8900   SDValue Reciprocal = DAG.getNode(ISD::FDIV, DL, VT, FPOne, N1, Flags);
8901 
8902   // Dividend / Divisor -> Dividend * Reciprocal
8903   for (auto *U : Users) {
8904     SDValue Dividend = U->getOperand(0);
8905     if (Dividend != FPOne) {
8906       SDValue NewNode = DAG.getNode(ISD::FMUL, SDLoc(U), VT, Dividend,
8907                                     Reciprocal, Flags);
8908       CombineTo(U, NewNode);
8909     } else if (U != Reciprocal.getNode()) {
8910       // In the absence of fast-math-flags, this user node is always the
8911       // same node as Reciprocal, but with FMF they may be different nodes.
8912       CombineTo(U, Reciprocal);
8913     }
8914   }
8915   return SDValue(N, 0);  // N was replaced.
8916 }
8917 
8918 SDValue DAGCombiner::visitFDIV(SDNode *N) {
8919   SDValue N0 = N->getOperand(0);
8920   SDValue N1 = N->getOperand(1);
8921   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
8922   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
8923   EVT VT = N->getValueType(0);
8924   SDLoc DL(N);
8925   const TargetOptions &Options = DAG.getTarget().Options;
8926   SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8927 
8928   // fold vector ops
8929   if (VT.isVector())
8930     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8931       return FoldedVOp;
8932 
8933   // fold (fdiv c1, c2) -> c1/c2
8934   if (N0CFP && N1CFP)
8935     return DAG.getNode(ISD::FDIV, SDLoc(N), VT, N0, N1, Flags);
8936 
8937   if (Options.UnsafeFPMath) {
8938     // fold (fdiv X, c2) -> fmul X, 1/c2 if losing precision is acceptable.
8939     if (N1CFP) {
8940       // Compute the reciprocal 1.0 / c2.
8941       const APFloat &N1APF = N1CFP->getValueAPF();
8942       APFloat Recip(N1APF.getSemantics(), 1); // 1.0
8943       APFloat::opStatus st = Recip.divide(N1APF, APFloat::rmNearestTiesToEven);
8944       // Only do the transform if the reciprocal is a legal fp immediate that
8945       // isn't too nasty (eg NaN, denormal, ...).
8946       if ((st == APFloat::opOK || st == APFloat::opInexact) && // Not too nasty
8947           (!LegalOperations ||
8948            // FIXME: custom lowering of ConstantFP might fail (see e.g. ARM
8949            // backend)... we should handle this gracefully after Legalize.
8950            // TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT) ||
8951            TLI.isOperationLegal(llvm::ISD::ConstantFP, VT) ||
8952            TLI.isFPImmLegal(Recip, VT)))
8953         return DAG.getNode(ISD::FMUL, DL, VT, N0,
8954                            DAG.getConstantFP(Recip, DL, VT), Flags);
8955     }
8956 
8957     // If this FDIV is part of a reciprocal square root, it may be folded
8958     // into a target-specific square root estimate instruction.
8959     if (N1.getOpcode() == ISD::FSQRT) {
8960       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0), Flags)) {
8961         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8962       }
8963     } else if (N1.getOpcode() == ISD::FP_EXTEND &&
8964                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
8965       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
8966                                           Flags)) {
8967         RV = DAG.getNode(ISD::FP_EXTEND, SDLoc(N1), VT, RV);
8968         AddToWorklist(RV.getNode());
8969         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8970       }
8971     } else if (N1.getOpcode() == ISD::FP_ROUND &&
8972                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
8973       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
8974                                           Flags)) {
8975         RV = DAG.getNode(ISD::FP_ROUND, SDLoc(N1), VT, RV, N1.getOperand(1));
8976         AddToWorklist(RV.getNode());
8977         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8978       }
8979     } else if (N1.getOpcode() == ISD::FMUL) {
8980       // Look through an FMUL. Even though this won't remove the FDIV directly,
8981       // it's still worthwhile to get rid of the FSQRT if possible.
8982       SDValue SqrtOp;
8983       SDValue OtherOp;
8984       if (N1.getOperand(0).getOpcode() == ISD::FSQRT) {
8985         SqrtOp = N1.getOperand(0);
8986         OtherOp = N1.getOperand(1);
8987       } else if (N1.getOperand(1).getOpcode() == ISD::FSQRT) {
8988         SqrtOp = N1.getOperand(1);
8989         OtherOp = N1.getOperand(0);
8990       }
8991       if (SqrtOp.getNode()) {
8992         // We found a FSQRT, so try to make this fold:
8993         // x / (y * sqrt(z)) -> x * (rsqrt(z) / y)
8994         if (SDValue RV = buildRsqrtEstimate(SqrtOp.getOperand(0), Flags)) {
8995           RV = DAG.getNode(ISD::FDIV, SDLoc(N1), VT, RV, OtherOp, Flags);
8996           AddToWorklist(RV.getNode());
8997           return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8998         }
8999       }
9000     }
9001 
9002     // Fold into a reciprocal estimate and multiply instead of a real divide.
9003     if (SDValue RV = BuildReciprocalEstimate(N1, Flags)) {
9004       AddToWorklist(RV.getNode());
9005       return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
9006     }
9007   }
9008 
9009   // (fdiv (fneg X), (fneg Y)) -> (fdiv X, Y)
9010   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) {
9011     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) {
9012       // Both can be negated for free, check to see if at least one is cheaper
9013       // negated.
9014       if (LHSNeg == 2 || RHSNeg == 2)
9015         return DAG.getNode(ISD::FDIV, SDLoc(N), VT,
9016                            GetNegatedExpression(N0, DAG, LegalOperations),
9017                            GetNegatedExpression(N1, DAG, LegalOperations),
9018                            Flags);
9019     }
9020   }
9021 
9022   if (SDValue CombineRepeatedDivisors = combineRepeatedFPDivisors(N))
9023     return CombineRepeatedDivisors;
9024 
9025   return SDValue();
9026 }
9027 
9028 SDValue DAGCombiner::visitFREM(SDNode *N) {
9029   SDValue N0 = N->getOperand(0);
9030   SDValue N1 = N->getOperand(1);
9031   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9032   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
9033   EVT VT = N->getValueType(0);
9034 
9035   // fold (frem c1, c2) -> fmod(c1,c2)
9036   if (N0CFP && N1CFP)
9037     return DAG.getNode(ISD::FREM, SDLoc(N), VT, N0, N1,
9038                        &cast<BinaryWithFlagsSDNode>(N)->Flags);
9039 
9040   return SDValue();
9041 }
9042 
9043 SDValue DAGCombiner::visitFSQRT(SDNode *N) {
9044   if (!DAG.getTarget().Options.UnsafeFPMath)
9045     return SDValue();
9046 
9047   SDValue N0 = N->getOperand(0);
9048   if (TLI.isFsqrtCheap(N0, DAG))
9049     return SDValue();
9050 
9051   // TODO: FSQRT nodes should have flags that propagate to the created nodes.
9052   // For now, create a Flags object for use with all unsafe math transforms.
9053   SDNodeFlags Flags;
9054   Flags.setUnsafeAlgebra(true);
9055   return buildSqrtEstimate(N0, &Flags);
9056 }
9057 
9058 /// copysign(x, fp_extend(y)) -> copysign(x, y)
9059 /// copysign(x, fp_round(y)) -> copysign(x, y)
9060 static inline bool CanCombineFCOPYSIGN_EXTEND_ROUND(SDNode *N) {
9061   SDValue N1 = N->getOperand(1);
9062   if ((N1.getOpcode() == ISD::FP_EXTEND ||
9063        N1.getOpcode() == ISD::FP_ROUND)) {
9064     // Do not optimize out type conversion of f128 type yet.
9065     // For some targets like x86_64, configuration is changed to keep one f128
9066     // value in one SSE register, but instruction selection cannot handle
9067     // FCOPYSIGN on SSE registers yet.
9068     EVT N1VT = N1->getValueType(0);
9069     EVT N1Op0VT = N1->getOperand(0)->getValueType(0);
9070     return (N1VT == N1Op0VT || N1Op0VT != MVT::f128);
9071   }
9072   return false;
9073 }
9074 
9075 SDValue DAGCombiner::visitFCOPYSIGN(SDNode *N) {
9076   SDValue N0 = N->getOperand(0);
9077   SDValue N1 = N->getOperand(1);
9078   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9079   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
9080   EVT VT = N->getValueType(0);
9081 
9082   if (N0CFP && N1CFP) // Constant fold
9083     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1);
9084 
9085   if (N1CFP) {
9086     const APFloat &V = N1CFP->getValueAPF();
9087     // copysign(x, c1) -> fabs(x)       iff ispos(c1)
9088     // copysign(x, c1) -> fneg(fabs(x)) iff isneg(c1)
9089     if (!V.isNegative()) {
9090       if (!LegalOperations || TLI.isOperationLegal(ISD::FABS, VT))
9091         return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9092     } else {
9093       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
9094         return DAG.getNode(ISD::FNEG, SDLoc(N), VT,
9095                            DAG.getNode(ISD::FABS, SDLoc(N0), VT, N0));
9096     }
9097   }
9098 
9099   // copysign(fabs(x), y) -> copysign(x, y)
9100   // copysign(fneg(x), y) -> copysign(x, y)
9101   // copysign(copysign(x,z), y) -> copysign(x, y)
9102   if (N0.getOpcode() == ISD::FABS || N0.getOpcode() == ISD::FNEG ||
9103       N0.getOpcode() == ISD::FCOPYSIGN)
9104     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0.getOperand(0), N1);
9105 
9106   // copysign(x, abs(y)) -> abs(x)
9107   if (N1.getOpcode() == ISD::FABS)
9108     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9109 
9110   // copysign(x, copysign(y,z)) -> copysign(x, z)
9111   if (N1.getOpcode() == ISD::FCOPYSIGN)
9112     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(1));
9113 
9114   // copysign(x, fp_extend(y)) -> copysign(x, y)
9115   // copysign(x, fp_round(y)) -> copysign(x, y)
9116   if (CanCombineFCOPYSIGN_EXTEND_ROUND(N))
9117     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(0));
9118 
9119   return SDValue();
9120 }
9121 
9122 SDValue DAGCombiner::visitSINT_TO_FP(SDNode *N) {
9123   SDValue N0 = N->getOperand(0);
9124   EVT VT = N->getValueType(0);
9125   EVT OpVT = N0.getValueType();
9126 
9127   // fold (sint_to_fp c1) -> c1fp
9128   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
9129       // ...but only if the target supports immediate floating-point values
9130       (!LegalOperations ||
9131        TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT)))
9132     return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
9133 
9134   // If the input is a legal type, and SINT_TO_FP is not legal on this target,
9135   // but UINT_TO_FP is legal on this target, try to convert.
9136   if (!TLI.isOperationLegalOrCustom(ISD::SINT_TO_FP, OpVT) &&
9137       TLI.isOperationLegalOrCustom(ISD::UINT_TO_FP, OpVT)) {
9138     // If the sign bit is known to be zero, we can change this to UINT_TO_FP.
9139     if (DAG.SignBitIsZero(N0))
9140       return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
9141   }
9142 
9143   // The next optimizations are desirable only if SELECT_CC can be lowered.
9144   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
9145     // fold (sint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
9146     if (N0.getOpcode() == ISD::SETCC && N0.getValueType() == MVT::i1 &&
9147         !VT.isVector() &&
9148         (!LegalOperations ||
9149          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9150       SDLoc DL(N);
9151       SDValue Ops[] =
9152         { N0.getOperand(0), N0.getOperand(1),
9153           DAG.getConstantFP(-1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9154           N0.getOperand(2) };
9155       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9156     }
9157 
9158     // fold (sint_to_fp (zext (setcc x, y, cc))) ->
9159     //      (select_cc x, y, 1.0, 0.0,, cc)
9160     if (N0.getOpcode() == ISD::ZERO_EXTEND &&
9161         N0.getOperand(0).getOpcode() == ISD::SETCC &&!VT.isVector() &&
9162         (!LegalOperations ||
9163          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9164       SDLoc DL(N);
9165       SDValue Ops[] =
9166         { N0.getOperand(0).getOperand(0), N0.getOperand(0).getOperand(1),
9167           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9168           N0.getOperand(0).getOperand(2) };
9169       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9170     }
9171   }
9172 
9173   return SDValue();
9174 }
9175 
9176 SDValue DAGCombiner::visitUINT_TO_FP(SDNode *N) {
9177   SDValue N0 = N->getOperand(0);
9178   EVT VT = N->getValueType(0);
9179   EVT OpVT = N0.getValueType();
9180 
9181   // fold (uint_to_fp c1) -> c1fp
9182   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
9183       // ...but only if the target supports immediate floating-point values
9184       (!LegalOperations ||
9185        TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT)))
9186     return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
9187 
9188   // If the input is a legal type, and UINT_TO_FP is not legal on this target,
9189   // but SINT_TO_FP is legal on this target, try to convert.
9190   if (!TLI.isOperationLegalOrCustom(ISD::UINT_TO_FP, OpVT) &&
9191       TLI.isOperationLegalOrCustom(ISD::SINT_TO_FP, OpVT)) {
9192     // If the sign bit is known to be zero, we can change this to SINT_TO_FP.
9193     if (DAG.SignBitIsZero(N0))
9194       return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
9195   }
9196 
9197   // The next optimizations are desirable only if SELECT_CC can be lowered.
9198   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
9199     // fold (uint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
9200 
9201     if (N0.getOpcode() == ISD::SETCC && !VT.isVector() &&
9202         (!LegalOperations ||
9203          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9204       SDLoc DL(N);
9205       SDValue Ops[] =
9206         { N0.getOperand(0), N0.getOperand(1),
9207           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9208           N0.getOperand(2) };
9209       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9210     }
9211   }
9212 
9213   return SDValue();
9214 }
9215 
9216 // Fold (fp_to_{s/u}int ({s/u}int_to_fpx)) -> zext x, sext x, trunc x, or x
9217 static SDValue FoldIntToFPToInt(SDNode *N, SelectionDAG &DAG) {
9218   SDValue N0 = N->getOperand(0);
9219   EVT VT = N->getValueType(0);
9220 
9221   if (N0.getOpcode() != ISD::UINT_TO_FP && N0.getOpcode() != ISD::SINT_TO_FP)
9222     return SDValue();
9223 
9224   SDValue Src = N0.getOperand(0);
9225   EVT SrcVT = Src.getValueType();
9226   bool IsInputSigned = N0.getOpcode() == ISD::SINT_TO_FP;
9227   bool IsOutputSigned = N->getOpcode() == ISD::FP_TO_SINT;
9228 
9229   // We can safely assume the conversion won't overflow the output range,
9230   // because (for example) (uint8_t)18293.f is undefined behavior.
9231 
9232   // Since we can assume the conversion won't overflow, our decision as to
9233   // whether the input will fit in the float should depend on the minimum
9234   // of the input range and output range.
9235 
9236   // This means this is also safe for a signed input and unsigned output, since
9237   // a negative input would lead to undefined behavior.
9238   unsigned InputSize = (int)SrcVT.getScalarSizeInBits() - IsInputSigned;
9239   unsigned OutputSize = (int)VT.getScalarSizeInBits() - IsOutputSigned;
9240   unsigned ActualSize = std::min(InputSize, OutputSize);
9241   const fltSemantics &sem = DAG.EVTToAPFloatSemantics(N0.getValueType());
9242 
9243   // We can only fold away the float conversion if the input range can be
9244   // represented exactly in the float range.
9245   if (APFloat::semanticsPrecision(sem) >= ActualSize) {
9246     if (VT.getScalarSizeInBits() > SrcVT.getScalarSizeInBits()) {
9247       unsigned ExtOp = IsInputSigned && IsOutputSigned ? ISD::SIGN_EXTEND
9248                                                        : ISD::ZERO_EXTEND;
9249       return DAG.getNode(ExtOp, SDLoc(N), VT, Src);
9250     }
9251     if (VT.getScalarSizeInBits() < SrcVT.getScalarSizeInBits())
9252       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Src);
9253     return DAG.getBitcast(VT, Src);
9254   }
9255   return SDValue();
9256 }
9257 
9258 SDValue DAGCombiner::visitFP_TO_SINT(SDNode *N) {
9259   SDValue N0 = N->getOperand(0);
9260   EVT VT = N->getValueType(0);
9261 
9262   // fold (fp_to_sint c1fp) -> c1
9263   if (isConstantFPBuildVectorOrConstantFP(N0))
9264     return DAG.getNode(ISD::FP_TO_SINT, SDLoc(N), VT, N0);
9265 
9266   return FoldIntToFPToInt(N, DAG);
9267 }
9268 
9269 SDValue DAGCombiner::visitFP_TO_UINT(SDNode *N) {
9270   SDValue N0 = N->getOperand(0);
9271   EVT VT = N->getValueType(0);
9272 
9273   // fold (fp_to_uint c1fp) -> c1
9274   if (isConstantFPBuildVectorOrConstantFP(N0))
9275     return DAG.getNode(ISD::FP_TO_UINT, SDLoc(N), VT, N0);
9276 
9277   return FoldIntToFPToInt(N, DAG);
9278 }
9279 
9280 SDValue DAGCombiner::visitFP_ROUND(SDNode *N) {
9281   SDValue N0 = N->getOperand(0);
9282   SDValue N1 = N->getOperand(1);
9283   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9284   EVT VT = N->getValueType(0);
9285 
9286   // fold (fp_round c1fp) -> c1fp
9287   if (N0CFP)
9288     return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT, N0, N1);
9289 
9290   // fold (fp_round (fp_extend x)) -> x
9291   if (N0.getOpcode() == ISD::FP_EXTEND && VT == N0.getOperand(0).getValueType())
9292     return N0.getOperand(0);
9293 
9294   // fold (fp_round (fp_round x)) -> (fp_round x)
9295   if (N0.getOpcode() == ISD::FP_ROUND) {
9296     const bool NIsTrunc = N->getConstantOperandVal(1) == 1;
9297     const bool N0IsTrunc = N0.getNode()->getConstantOperandVal(1) == 1;
9298 
9299     // Skip this folding if it results in an fp_round from f80 to f16.
9300     //
9301     // f80 to f16 always generates an expensive (and as yet, unimplemented)
9302     // libcall to __truncxfhf2 instead of selecting native f16 conversion
9303     // instructions from f32 or f64.  Moreover, the first (value-preserving)
9304     // fp_round from f80 to either f32 or f64 may become a NOP in platforms like
9305     // x86.
9306     if (N0.getOperand(0).getValueType() == MVT::f80 && VT == MVT::f16)
9307       return SDValue();
9308 
9309     // If the first fp_round isn't a value preserving truncation, it might
9310     // introduce a tie in the second fp_round, that wouldn't occur in the
9311     // single-step fp_round we want to fold to.
9312     // In other words, double rounding isn't the same as rounding.
9313     // Also, this is a value preserving truncation iff both fp_round's are.
9314     if (DAG.getTarget().Options.UnsafeFPMath || N0IsTrunc) {
9315       SDLoc DL(N);
9316       return DAG.getNode(ISD::FP_ROUND, DL, VT, N0.getOperand(0),
9317                          DAG.getIntPtrConstant(NIsTrunc && N0IsTrunc, DL));
9318     }
9319   }
9320 
9321   // fold (fp_round (copysign X, Y)) -> (copysign (fp_round X), Y)
9322   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse()) {
9323     SDValue Tmp = DAG.getNode(ISD::FP_ROUND, SDLoc(N0), VT,
9324                               N0.getOperand(0), N1);
9325     AddToWorklist(Tmp.getNode());
9326     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT,
9327                        Tmp, N0.getOperand(1));
9328   }
9329 
9330   return SDValue();
9331 }
9332 
9333 SDValue DAGCombiner::visitFP_ROUND_INREG(SDNode *N) {
9334   SDValue N0 = N->getOperand(0);
9335   EVT VT = N->getValueType(0);
9336   EVT EVT = cast<VTSDNode>(N->getOperand(1))->getVT();
9337   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9338 
9339   // fold (fp_round_inreg c1fp) -> c1fp
9340   if (N0CFP && isTypeLegal(EVT)) {
9341     SDLoc DL(N);
9342     SDValue Round = DAG.getConstantFP(*N0CFP->getConstantFPValue(), DL, EVT);
9343     return DAG.getNode(ISD::FP_EXTEND, DL, VT, Round);
9344   }
9345 
9346   return SDValue();
9347 }
9348 
9349 SDValue DAGCombiner::visitFP_EXTEND(SDNode *N) {
9350   SDValue N0 = N->getOperand(0);
9351   EVT VT = N->getValueType(0);
9352 
9353   // If this is fp_round(fpextend), don't fold it, allow ourselves to be folded.
9354   if (N->hasOneUse() &&
9355       N->use_begin()->getOpcode() == ISD::FP_ROUND)
9356     return SDValue();
9357 
9358   // fold (fp_extend c1fp) -> c1fp
9359   if (isConstantFPBuildVectorOrConstantFP(N0))
9360     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, N0);
9361 
9362   // fold (fp_extend (fp16_to_fp op)) -> (fp16_to_fp op)
9363   if (N0.getOpcode() == ISD::FP16_TO_FP &&
9364       TLI.getOperationAction(ISD::FP16_TO_FP, VT) == TargetLowering::Legal)
9365     return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), VT, N0.getOperand(0));
9366 
9367   // Turn fp_extend(fp_round(X, 1)) -> x since the fp_round doesn't affect the
9368   // value of X.
9369   if (N0.getOpcode() == ISD::FP_ROUND
9370       && N0.getNode()->getConstantOperandVal(1) == 1) {
9371     SDValue In = N0.getOperand(0);
9372     if (In.getValueType() == VT) return In;
9373     if (VT.bitsLT(In.getValueType()))
9374       return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT,
9375                          In, N0.getOperand(1));
9376     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, In);
9377   }
9378 
9379   // fold (fpext (load x)) -> (fpext (fptrunc (extload x)))
9380   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
9381        TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
9382     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9383     SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
9384                                      LN0->getChain(),
9385                                      LN0->getBasePtr(), N0.getValueType(),
9386                                      LN0->getMemOperand());
9387     CombineTo(N, ExtLoad);
9388     CombineTo(N0.getNode(),
9389               DAG.getNode(ISD::FP_ROUND, SDLoc(N0),
9390                           N0.getValueType(), ExtLoad,
9391                           DAG.getIntPtrConstant(1, SDLoc(N0))),
9392               ExtLoad.getValue(1));
9393     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9394   }
9395 
9396   return SDValue();
9397 }
9398 
9399 SDValue DAGCombiner::visitFCEIL(SDNode *N) {
9400   SDValue N0 = N->getOperand(0);
9401   EVT VT = N->getValueType(0);
9402 
9403   // fold (fceil c1) -> fceil(c1)
9404   if (isConstantFPBuildVectorOrConstantFP(N0))
9405     return DAG.getNode(ISD::FCEIL, SDLoc(N), VT, N0);
9406 
9407   return SDValue();
9408 }
9409 
9410 SDValue DAGCombiner::visitFTRUNC(SDNode *N) {
9411   SDValue N0 = N->getOperand(0);
9412   EVT VT = N->getValueType(0);
9413 
9414   // fold (ftrunc c1) -> ftrunc(c1)
9415   if (isConstantFPBuildVectorOrConstantFP(N0))
9416     return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0);
9417 
9418   return SDValue();
9419 }
9420 
9421 SDValue DAGCombiner::visitFFLOOR(SDNode *N) {
9422   SDValue N0 = N->getOperand(0);
9423   EVT VT = N->getValueType(0);
9424 
9425   // fold (ffloor c1) -> ffloor(c1)
9426   if (isConstantFPBuildVectorOrConstantFP(N0))
9427     return DAG.getNode(ISD::FFLOOR, SDLoc(N), VT, N0);
9428 
9429   return SDValue();
9430 }
9431 
9432 // FIXME: FNEG and FABS have a lot in common; refactor.
9433 SDValue DAGCombiner::visitFNEG(SDNode *N) {
9434   SDValue N0 = N->getOperand(0);
9435   EVT VT = N->getValueType(0);
9436 
9437   // Constant fold FNEG.
9438   if (isConstantFPBuildVectorOrConstantFP(N0))
9439     return DAG.getNode(ISD::FNEG, SDLoc(N), VT, N0);
9440 
9441   if (isNegatibleForFree(N0, LegalOperations, DAG.getTargetLoweringInfo(),
9442                          &DAG.getTarget().Options))
9443     return GetNegatedExpression(N0, DAG, LegalOperations);
9444 
9445   // Transform fneg(bitconvert(x)) -> bitconvert(x ^ sign) to avoid loading
9446   // constant pool values.
9447   if (!TLI.isFNegFree(VT) &&
9448       N0.getOpcode() == ISD::BITCAST &&
9449       N0.getNode()->hasOneUse()) {
9450     SDValue Int = N0.getOperand(0);
9451     EVT IntVT = Int.getValueType();
9452     if (IntVT.isInteger() && !IntVT.isVector()) {
9453       APInt SignMask;
9454       if (N0.getValueType().isVector()) {
9455         // For a vector, get a mask such as 0x80... per scalar element
9456         // and splat it.
9457         SignMask = APInt::getSignBit(N0.getScalarValueSizeInBits());
9458         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
9459       } else {
9460         // For a scalar, just generate 0x80...
9461         SignMask = APInt::getSignBit(IntVT.getSizeInBits());
9462       }
9463       SDLoc DL0(N0);
9464       Int = DAG.getNode(ISD::XOR, DL0, IntVT, Int,
9465                         DAG.getConstant(SignMask, DL0, IntVT));
9466       AddToWorklist(Int.getNode());
9467       return DAG.getBitcast(VT, Int);
9468     }
9469   }
9470 
9471   // (fneg (fmul c, x)) -> (fmul -c, x)
9472   if (N0.getOpcode() == ISD::FMUL &&
9473       (N0.getNode()->hasOneUse() || !TLI.isFNegFree(VT))) {
9474     ConstantFPSDNode *CFP1 = dyn_cast<ConstantFPSDNode>(N0.getOperand(1));
9475     if (CFP1) {
9476       APFloat CVal = CFP1->getValueAPF();
9477       CVal.changeSign();
9478       if (Level >= AfterLegalizeDAG &&
9479           (TLI.isFPImmLegal(CVal, VT) ||
9480            TLI.isOperationLegal(ISD::ConstantFP, VT)))
9481         return DAG.getNode(ISD::FMUL, SDLoc(N), VT, N0.getOperand(0),
9482                            DAG.getNode(ISD::FNEG, SDLoc(N), VT,
9483                                        N0.getOperand(1)),
9484                            &cast<BinaryWithFlagsSDNode>(N0)->Flags);
9485     }
9486   }
9487 
9488   return SDValue();
9489 }
9490 
9491 SDValue DAGCombiner::visitFMINNUM(SDNode *N) {
9492   SDValue N0 = N->getOperand(0);
9493   SDValue N1 = N->getOperand(1);
9494   EVT VT = N->getValueType(0);
9495   const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9496   const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9497 
9498   if (N0CFP && N1CFP) {
9499     const APFloat &C0 = N0CFP->getValueAPF();
9500     const APFloat &C1 = N1CFP->getValueAPF();
9501     return DAG.getConstantFP(minnum(C0, C1), SDLoc(N), VT);
9502   }
9503 
9504   // Canonicalize to constant on RHS.
9505   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9506      !isConstantFPBuildVectorOrConstantFP(N1))
9507     return DAG.getNode(ISD::FMINNUM, SDLoc(N), VT, N1, N0);
9508 
9509   return SDValue();
9510 }
9511 
9512 SDValue DAGCombiner::visitFMAXNUM(SDNode *N) {
9513   SDValue N0 = N->getOperand(0);
9514   SDValue N1 = N->getOperand(1);
9515   EVT VT = N->getValueType(0);
9516   const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9517   const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9518 
9519   if (N0CFP && N1CFP) {
9520     const APFloat &C0 = N0CFP->getValueAPF();
9521     const APFloat &C1 = N1CFP->getValueAPF();
9522     return DAG.getConstantFP(maxnum(C0, C1), SDLoc(N), VT);
9523   }
9524 
9525   // Canonicalize to constant on RHS.
9526   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9527      !isConstantFPBuildVectorOrConstantFP(N1))
9528     return DAG.getNode(ISD::FMAXNUM, SDLoc(N), VT, N1, N0);
9529 
9530   return SDValue();
9531 }
9532 
9533 SDValue DAGCombiner::visitFABS(SDNode *N) {
9534   SDValue N0 = N->getOperand(0);
9535   EVT VT = N->getValueType(0);
9536 
9537   // fold (fabs c1) -> fabs(c1)
9538   if (isConstantFPBuildVectorOrConstantFP(N0))
9539     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9540 
9541   // fold (fabs (fabs x)) -> (fabs x)
9542   if (N0.getOpcode() == ISD::FABS)
9543     return N->getOperand(0);
9544 
9545   // fold (fabs (fneg x)) -> (fabs x)
9546   // fold (fabs (fcopysign x, y)) -> (fabs x)
9547   if (N0.getOpcode() == ISD::FNEG || N0.getOpcode() == ISD::FCOPYSIGN)
9548     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0.getOperand(0));
9549 
9550   // Transform fabs(bitconvert(x)) -> bitconvert(x & ~sign) to avoid loading
9551   // constant pool values.
9552   if (!TLI.isFAbsFree(VT) &&
9553       N0.getOpcode() == ISD::BITCAST &&
9554       N0.getNode()->hasOneUse()) {
9555     SDValue Int = N0.getOperand(0);
9556     EVT IntVT = Int.getValueType();
9557     if (IntVT.isInteger() && !IntVT.isVector()) {
9558       APInt SignMask;
9559       if (N0.getValueType().isVector()) {
9560         // For a vector, get a mask such as 0x7f... per scalar element
9561         // and splat it.
9562         SignMask = ~APInt::getSignBit(N0.getScalarValueSizeInBits());
9563         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
9564       } else {
9565         // For a scalar, just generate 0x7f...
9566         SignMask = ~APInt::getSignBit(IntVT.getSizeInBits());
9567       }
9568       SDLoc DL(N0);
9569       Int = DAG.getNode(ISD::AND, DL, IntVT, Int,
9570                         DAG.getConstant(SignMask, DL, IntVT));
9571       AddToWorklist(Int.getNode());
9572       return DAG.getBitcast(N->getValueType(0), Int);
9573     }
9574   }
9575 
9576   return SDValue();
9577 }
9578 
9579 SDValue DAGCombiner::visitBRCOND(SDNode *N) {
9580   SDValue Chain = N->getOperand(0);
9581   SDValue N1 = N->getOperand(1);
9582   SDValue N2 = N->getOperand(2);
9583 
9584   // If N is a constant we could fold this into a fallthrough or unconditional
9585   // branch. However that doesn't happen very often in normal code, because
9586   // Instcombine/SimplifyCFG should have handled the available opportunities.
9587   // If we did this folding here, it would be necessary to update the
9588   // MachineBasicBlock CFG, which is awkward.
9589 
9590   // fold a brcond with a setcc condition into a BR_CC node if BR_CC is legal
9591   // on the target.
9592   if (N1.getOpcode() == ISD::SETCC &&
9593       TLI.isOperationLegalOrCustom(ISD::BR_CC,
9594                                    N1.getOperand(0).getValueType())) {
9595     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
9596                        Chain, N1.getOperand(2),
9597                        N1.getOperand(0), N1.getOperand(1), N2);
9598   }
9599 
9600   if ((N1.hasOneUse() && N1.getOpcode() == ISD::SRL) ||
9601       ((N1.getOpcode() == ISD::TRUNCATE && N1.hasOneUse()) &&
9602        (N1.getOperand(0).hasOneUse() &&
9603         N1.getOperand(0).getOpcode() == ISD::SRL))) {
9604     SDNode *Trunc = nullptr;
9605     if (N1.getOpcode() == ISD::TRUNCATE) {
9606       // Look pass the truncate.
9607       Trunc = N1.getNode();
9608       N1 = N1.getOperand(0);
9609     }
9610 
9611     // Match this pattern so that we can generate simpler code:
9612     //
9613     //   %a = ...
9614     //   %b = and i32 %a, 2
9615     //   %c = srl i32 %b, 1
9616     //   brcond i32 %c ...
9617     //
9618     // into
9619     //
9620     //   %a = ...
9621     //   %b = and i32 %a, 2
9622     //   %c = setcc eq %b, 0
9623     //   brcond %c ...
9624     //
9625     // This applies only when the AND constant value has one bit set and the
9626     // SRL constant is equal to the log2 of the AND constant. The back-end is
9627     // smart enough to convert the result into a TEST/JMP sequence.
9628     SDValue Op0 = N1.getOperand(0);
9629     SDValue Op1 = N1.getOperand(1);
9630 
9631     if (Op0.getOpcode() == ISD::AND &&
9632         Op1.getOpcode() == ISD::Constant) {
9633       SDValue AndOp1 = Op0.getOperand(1);
9634 
9635       if (AndOp1.getOpcode() == ISD::Constant) {
9636         const APInt &AndConst = cast<ConstantSDNode>(AndOp1)->getAPIntValue();
9637 
9638         if (AndConst.isPowerOf2() &&
9639             cast<ConstantSDNode>(Op1)->getAPIntValue()==AndConst.logBase2()) {
9640           SDLoc DL(N);
9641           SDValue SetCC =
9642             DAG.getSetCC(DL,
9643                          getSetCCResultType(Op0.getValueType()),
9644                          Op0, DAG.getConstant(0, DL, Op0.getValueType()),
9645                          ISD::SETNE);
9646 
9647           SDValue NewBRCond = DAG.getNode(ISD::BRCOND, DL,
9648                                           MVT::Other, Chain, SetCC, N2);
9649           // Don't add the new BRCond into the worklist or else SimplifySelectCC
9650           // will convert it back to (X & C1) >> C2.
9651           CombineTo(N, NewBRCond, false);
9652           // Truncate is dead.
9653           if (Trunc)
9654             deleteAndRecombine(Trunc);
9655           // Replace the uses of SRL with SETCC
9656           WorklistRemover DeadNodes(*this);
9657           DAG.ReplaceAllUsesOfValueWith(N1, SetCC);
9658           deleteAndRecombine(N1.getNode());
9659           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9660         }
9661       }
9662     }
9663 
9664     if (Trunc)
9665       // Restore N1 if the above transformation doesn't match.
9666       N1 = N->getOperand(1);
9667   }
9668 
9669   // Transform br(xor(x, y)) -> br(x != y)
9670   // Transform br(xor(xor(x,y), 1)) -> br (x == y)
9671   if (N1.hasOneUse() && N1.getOpcode() == ISD::XOR) {
9672     SDNode *TheXor = N1.getNode();
9673     SDValue Op0 = TheXor->getOperand(0);
9674     SDValue Op1 = TheXor->getOperand(1);
9675     if (Op0.getOpcode() == Op1.getOpcode()) {
9676       // Avoid missing important xor optimizations.
9677       if (SDValue Tmp = visitXOR(TheXor)) {
9678         if (Tmp.getNode() != TheXor) {
9679           DEBUG(dbgs() << "\nReplacing.8 ";
9680                 TheXor->dump(&DAG);
9681                 dbgs() << "\nWith: ";
9682                 Tmp.getNode()->dump(&DAG);
9683                 dbgs() << '\n');
9684           WorklistRemover DeadNodes(*this);
9685           DAG.ReplaceAllUsesOfValueWith(N1, Tmp);
9686           deleteAndRecombine(TheXor);
9687           return DAG.getNode(ISD::BRCOND, SDLoc(N),
9688                              MVT::Other, Chain, Tmp, N2);
9689         }
9690 
9691         // visitXOR has changed XOR's operands or replaced the XOR completely,
9692         // bail out.
9693         return SDValue(N, 0);
9694       }
9695     }
9696 
9697     if (Op0.getOpcode() != ISD::SETCC && Op1.getOpcode() != ISD::SETCC) {
9698       bool Equal = false;
9699       if (isOneConstant(Op0) && Op0.hasOneUse() &&
9700           Op0.getOpcode() == ISD::XOR) {
9701         TheXor = Op0.getNode();
9702         Equal = true;
9703       }
9704 
9705       EVT SetCCVT = N1.getValueType();
9706       if (LegalTypes)
9707         SetCCVT = getSetCCResultType(SetCCVT);
9708       SDValue SetCC = DAG.getSetCC(SDLoc(TheXor),
9709                                    SetCCVT,
9710                                    Op0, Op1,
9711                                    Equal ? ISD::SETEQ : ISD::SETNE);
9712       // Replace the uses of XOR with SETCC
9713       WorklistRemover DeadNodes(*this);
9714       DAG.ReplaceAllUsesOfValueWith(N1, SetCC);
9715       deleteAndRecombine(N1.getNode());
9716       return DAG.getNode(ISD::BRCOND, SDLoc(N),
9717                          MVT::Other, Chain, SetCC, N2);
9718     }
9719   }
9720 
9721   return SDValue();
9722 }
9723 
9724 // Operand List for BR_CC: Chain, CondCC, CondLHS, CondRHS, DestBB.
9725 //
9726 SDValue DAGCombiner::visitBR_CC(SDNode *N) {
9727   CondCodeSDNode *CC = cast<CondCodeSDNode>(N->getOperand(1));
9728   SDValue CondLHS = N->getOperand(2), CondRHS = N->getOperand(3);
9729 
9730   // If N is a constant we could fold this into a fallthrough or unconditional
9731   // branch. However that doesn't happen very often in normal code, because
9732   // Instcombine/SimplifyCFG should have handled the available opportunities.
9733   // If we did this folding here, it would be necessary to update the
9734   // MachineBasicBlock CFG, which is awkward.
9735 
9736   // Use SimplifySetCC to simplify SETCC's.
9737   SDValue Simp = SimplifySetCC(getSetCCResultType(CondLHS.getValueType()),
9738                                CondLHS, CondRHS, CC->get(), SDLoc(N),
9739                                false);
9740   if (Simp.getNode()) AddToWorklist(Simp.getNode());
9741 
9742   // fold to a simpler setcc
9743   if (Simp.getNode() && Simp.getOpcode() == ISD::SETCC)
9744     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
9745                        N->getOperand(0), Simp.getOperand(2),
9746                        Simp.getOperand(0), Simp.getOperand(1),
9747                        N->getOperand(4));
9748 
9749   return SDValue();
9750 }
9751 
9752 /// Return true if 'Use' is a load or a store that uses N as its base pointer
9753 /// and that N may be folded in the load / store addressing mode.
9754 static bool canFoldInAddressingMode(SDNode *N, SDNode *Use,
9755                                     SelectionDAG &DAG,
9756                                     const TargetLowering &TLI) {
9757   EVT VT;
9758   unsigned AS;
9759 
9760   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(Use)) {
9761     if (LD->isIndexed() || LD->getBasePtr().getNode() != N)
9762       return false;
9763     VT = LD->getMemoryVT();
9764     AS = LD->getAddressSpace();
9765   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(Use)) {
9766     if (ST->isIndexed() || ST->getBasePtr().getNode() != N)
9767       return false;
9768     VT = ST->getMemoryVT();
9769     AS = ST->getAddressSpace();
9770   } else
9771     return false;
9772 
9773   TargetLowering::AddrMode AM;
9774   if (N->getOpcode() == ISD::ADD) {
9775     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
9776     if (Offset)
9777       // [reg +/- imm]
9778       AM.BaseOffs = Offset->getSExtValue();
9779     else
9780       // [reg +/- reg]
9781       AM.Scale = 1;
9782   } else if (N->getOpcode() == ISD::SUB) {
9783     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
9784     if (Offset)
9785       // [reg +/- imm]
9786       AM.BaseOffs = -Offset->getSExtValue();
9787     else
9788       // [reg +/- reg]
9789       AM.Scale = 1;
9790   } else
9791     return false;
9792 
9793   return TLI.isLegalAddressingMode(DAG.getDataLayout(), AM,
9794                                    VT.getTypeForEVT(*DAG.getContext()), AS);
9795 }
9796 
9797 /// Try turning a load/store into a pre-indexed load/store when the base
9798 /// pointer is an add or subtract and it has other uses besides the load/store.
9799 /// After the transformation, the new indexed load/store has effectively folded
9800 /// the add/subtract in and all of its other uses are redirected to the
9801 /// new load/store.
9802 bool DAGCombiner::CombineToPreIndexedLoadStore(SDNode *N) {
9803   if (Level < AfterLegalizeDAG)
9804     return false;
9805 
9806   bool isLoad = true;
9807   SDValue Ptr;
9808   EVT VT;
9809   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
9810     if (LD->isIndexed())
9811       return false;
9812     VT = LD->getMemoryVT();
9813     if (!TLI.isIndexedLoadLegal(ISD::PRE_INC, VT) &&
9814         !TLI.isIndexedLoadLegal(ISD::PRE_DEC, VT))
9815       return false;
9816     Ptr = LD->getBasePtr();
9817   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
9818     if (ST->isIndexed())
9819       return false;
9820     VT = ST->getMemoryVT();
9821     if (!TLI.isIndexedStoreLegal(ISD::PRE_INC, VT) &&
9822         !TLI.isIndexedStoreLegal(ISD::PRE_DEC, VT))
9823       return false;
9824     Ptr = ST->getBasePtr();
9825     isLoad = false;
9826   } else {
9827     return false;
9828   }
9829 
9830   // If the pointer is not an add/sub, or if it doesn't have multiple uses, bail
9831   // out.  There is no reason to make this a preinc/predec.
9832   if ((Ptr.getOpcode() != ISD::ADD && Ptr.getOpcode() != ISD::SUB) ||
9833       Ptr.getNode()->hasOneUse())
9834     return false;
9835 
9836   // Ask the target to do addressing mode selection.
9837   SDValue BasePtr;
9838   SDValue Offset;
9839   ISD::MemIndexedMode AM = ISD::UNINDEXED;
9840   if (!TLI.getPreIndexedAddressParts(N, BasePtr, Offset, AM, DAG))
9841     return false;
9842 
9843   // Backends without true r+i pre-indexed forms may need to pass a
9844   // constant base with a variable offset so that constant coercion
9845   // will work with the patterns in canonical form.
9846   bool Swapped = false;
9847   if (isa<ConstantSDNode>(BasePtr)) {
9848     std::swap(BasePtr, Offset);
9849     Swapped = true;
9850   }
9851 
9852   // Don't create a indexed load / store with zero offset.
9853   if (isNullConstant(Offset))
9854     return false;
9855 
9856   // Try turning it into a pre-indexed load / store except when:
9857   // 1) The new base ptr is a frame index.
9858   // 2) If N is a store and the new base ptr is either the same as or is a
9859   //    predecessor of the value being stored.
9860   // 3) Another use of old base ptr is a predecessor of N. If ptr is folded
9861   //    that would create a cycle.
9862   // 4) All uses are load / store ops that use it as old base ptr.
9863 
9864   // Check #1.  Preinc'ing a frame index would require copying the stack pointer
9865   // (plus the implicit offset) to a register to preinc anyway.
9866   if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
9867     return false;
9868 
9869   // Check #2.
9870   if (!isLoad) {
9871     SDValue Val = cast<StoreSDNode>(N)->getValue();
9872     if (Val == BasePtr || BasePtr.getNode()->isPredecessorOf(Val.getNode()))
9873       return false;
9874   }
9875 
9876   // Caches for hasPredecessorHelper.
9877   SmallPtrSet<const SDNode *, 32> Visited;
9878   SmallVector<const SDNode *, 16> Worklist;
9879   Worklist.push_back(N);
9880 
9881   // If the offset is a constant, there may be other adds of constants that
9882   // can be folded with this one. We should do this to avoid having to keep
9883   // a copy of the original base pointer.
9884   SmallVector<SDNode *, 16> OtherUses;
9885   if (isa<ConstantSDNode>(Offset))
9886     for (SDNode::use_iterator UI = BasePtr.getNode()->use_begin(),
9887                               UE = BasePtr.getNode()->use_end();
9888          UI != UE; ++UI) {
9889       SDUse &Use = UI.getUse();
9890       // Skip the use that is Ptr and uses of other results from BasePtr's
9891       // node (important for nodes that return multiple results).
9892       if (Use.getUser() == Ptr.getNode() || Use != BasePtr)
9893         continue;
9894 
9895       if (SDNode::hasPredecessorHelper(Use.getUser(), Visited, Worklist))
9896         continue;
9897 
9898       if (Use.getUser()->getOpcode() != ISD::ADD &&
9899           Use.getUser()->getOpcode() != ISD::SUB) {
9900         OtherUses.clear();
9901         break;
9902       }
9903 
9904       SDValue Op1 = Use.getUser()->getOperand((UI.getOperandNo() + 1) & 1);
9905       if (!isa<ConstantSDNode>(Op1)) {
9906         OtherUses.clear();
9907         break;
9908       }
9909 
9910       // FIXME: In some cases, we can be smarter about this.
9911       if (Op1.getValueType() != Offset.getValueType()) {
9912         OtherUses.clear();
9913         break;
9914       }
9915 
9916       OtherUses.push_back(Use.getUser());
9917     }
9918 
9919   if (Swapped)
9920     std::swap(BasePtr, Offset);
9921 
9922   // Now check for #3 and #4.
9923   bool RealUse = false;
9924 
9925   for (SDNode *Use : Ptr.getNode()->uses()) {
9926     if (Use == N)
9927       continue;
9928     if (SDNode::hasPredecessorHelper(Use, Visited, Worklist))
9929       return false;
9930 
9931     // If Ptr may be folded in addressing mode of other use, then it's
9932     // not profitable to do this transformation.
9933     if (!canFoldInAddressingMode(Ptr.getNode(), Use, DAG, TLI))
9934       RealUse = true;
9935   }
9936 
9937   if (!RealUse)
9938     return false;
9939 
9940   SDValue Result;
9941   if (isLoad)
9942     Result = DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
9943                                 BasePtr, Offset, AM);
9944   else
9945     Result = DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
9946                                  BasePtr, Offset, AM);
9947   ++PreIndexedNodes;
9948   ++NodesCombined;
9949   DEBUG(dbgs() << "\nReplacing.4 ";
9950         N->dump(&DAG);
9951         dbgs() << "\nWith: ";
9952         Result.getNode()->dump(&DAG);
9953         dbgs() << '\n');
9954   WorklistRemover DeadNodes(*this);
9955   if (isLoad) {
9956     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
9957     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
9958   } else {
9959     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
9960   }
9961 
9962   // Finally, since the node is now dead, remove it from the graph.
9963   deleteAndRecombine(N);
9964 
9965   if (Swapped)
9966     std::swap(BasePtr, Offset);
9967 
9968   // Replace other uses of BasePtr that can be updated to use Ptr
9969   for (unsigned i = 0, e = OtherUses.size(); i != e; ++i) {
9970     unsigned OffsetIdx = 1;
9971     if (OtherUses[i]->getOperand(OffsetIdx).getNode() == BasePtr.getNode())
9972       OffsetIdx = 0;
9973     assert(OtherUses[i]->getOperand(!OffsetIdx).getNode() ==
9974            BasePtr.getNode() && "Expected BasePtr operand");
9975 
9976     // We need to replace ptr0 in the following expression:
9977     //   x0 * offset0 + y0 * ptr0 = t0
9978     // knowing that
9979     //   x1 * offset1 + y1 * ptr0 = t1 (the indexed load/store)
9980     //
9981     // where x0, x1, y0 and y1 in {-1, 1} are given by the types of the
9982     // indexed load/store and the expresion that needs to be re-written.
9983     //
9984     // Therefore, we have:
9985     //   t0 = (x0 * offset0 - x1 * y0 * y1 *offset1) + (y0 * y1) * t1
9986 
9987     ConstantSDNode *CN =
9988       cast<ConstantSDNode>(OtherUses[i]->getOperand(OffsetIdx));
9989     int X0, X1, Y0, Y1;
9990     const APInt &Offset0 = CN->getAPIntValue();
9991     APInt Offset1 = cast<ConstantSDNode>(Offset)->getAPIntValue();
9992 
9993     X0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 1) ? -1 : 1;
9994     Y0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 0) ? -1 : 1;
9995     X1 = (AM == ISD::PRE_DEC && !Swapped) ? -1 : 1;
9996     Y1 = (AM == ISD::PRE_DEC && Swapped) ? -1 : 1;
9997 
9998     unsigned Opcode = (Y0 * Y1 < 0) ? ISD::SUB : ISD::ADD;
9999 
10000     APInt CNV = Offset0;
10001     if (X0 < 0) CNV = -CNV;
10002     if (X1 * Y0 * Y1 < 0) CNV = CNV + Offset1;
10003     else CNV = CNV - Offset1;
10004 
10005     SDLoc DL(OtherUses[i]);
10006 
10007     // We can now generate the new expression.
10008     SDValue NewOp1 = DAG.getConstant(CNV, DL, CN->getValueType(0));
10009     SDValue NewOp2 = Result.getValue(isLoad ? 1 : 0);
10010 
10011     SDValue NewUse = DAG.getNode(Opcode,
10012                                  DL,
10013                                  OtherUses[i]->getValueType(0), NewOp1, NewOp2);
10014     DAG.ReplaceAllUsesOfValueWith(SDValue(OtherUses[i], 0), NewUse);
10015     deleteAndRecombine(OtherUses[i]);
10016   }
10017 
10018   // Replace the uses of Ptr with uses of the updated base value.
10019   DAG.ReplaceAllUsesOfValueWith(Ptr, Result.getValue(isLoad ? 1 : 0));
10020   deleteAndRecombine(Ptr.getNode());
10021 
10022   return true;
10023 }
10024 
10025 /// Try to combine a load/store with a add/sub of the base pointer node into a
10026 /// post-indexed load/store. The transformation folded the add/subtract into the
10027 /// new indexed load/store effectively and all of its uses are redirected to the
10028 /// new load/store.
10029 bool DAGCombiner::CombineToPostIndexedLoadStore(SDNode *N) {
10030   if (Level < AfterLegalizeDAG)
10031     return false;
10032 
10033   bool isLoad = true;
10034   SDValue Ptr;
10035   EVT VT;
10036   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
10037     if (LD->isIndexed())
10038       return false;
10039     VT = LD->getMemoryVT();
10040     if (!TLI.isIndexedLoadLegal(ISD::POST_INC, VT) &&
10041         !TLI.isIndexedLoadLegal(ISD::POST_DEC, VT))
10042       return false;
10043     Ptr = LD->getBasePtr();
10044   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
10045     if (ST->isIndexed())
10046       return false;
10047     VT = ST->getMemoryVT();
10048     if (!TLI.isIndexedStoreLegal(ISD::POST_INC, VT) &&
10049         !TLI.isIndexedStoreLegal(ISD::POST_DEC, VT))
10050       return false;
10051     Ptr = ST->getBasePtr();
10052     isLoad = false;
10053   } else {
10054     return false;
10055   }
10056 
10057   if (Ptr.getNode()->hasOneUse())
10058     return false;
10059 
10060   for (SDNode *Op : Ptr.getNode()->uses()) {
10061     if (Op == N ||
10062         (Op->getOpcode() != ISD::ADD && Op->getOpcode() != ISD::SUB))
10063       continue;
10064 
10065     SDValue BasePtr;
10066     SDValue Offset;
10067     ISD::MemIndexedMode AM = ISD::UNINDEXED;
10068     if (TLI.getPostIndexedAddressParts(N, Op, BasePtr, Offset, AM, DAG)) {
10069       // Don't create a indexed load / store with zero offset.
10070       if (isNullConstant(Offset))
10071         continue;
10072 
10073       // Try turning it into a post-indexed load / store except when
10074       // 1) All uses are load / store ops that use it as base ptr (and
10075       //    it may be folded as addressing mmode).
10076       // 2) Op must be independent of N, i.e. Op is neither a predecessor
10077       //    nor a successor of N. Otherwise, if Op is folded that would
10078       //    create a cycle.
10079 
10080       if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
10081         continue;
10082 
10083       // Check for #1.
10084       bool TryNext = false;
10085       for (SDNode *Use : BasePtr.getNode()->uses()) {
10086         if (Use == Ptr.getNode())
10087           continue;
10088 
10089         // If all the uses are load / store addresses, then don't do the
10090         // transformation.
10091         if (Use->getOpcode() == ISD::ADD || Use->getOpcode() == ISD::SUB){
10092           bool RealUse = false;
10093           for (SDNode *UseUse : Use->uses()) {
10094             if (!canFoldInAddressingMode(Use, UseUse, DAG, TLI))
10095               RealUse = true;
10096           }
10097 
10098           if (!RealUse) {
10099             TryNext = true;
10100             break;
10101           }
10102         }
10103       }
10104 
10105       if (TryNext)
10106         continue;
10107 
10108       // Check for #2
10109       if (!Op->isPredecessorOf(N) && !N->isPredecessorOf(Op)) {
10110         SDValue Result = isLoad
10111           ? DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
10112                                BasePtr, Offset, AM)
10113           : DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
10114                                 BasePtr, Offset, AM);
10115         ++PostIndexedNodes;
10116         ++NodesCombined;
10117         DEBUG(dbgs() << "\nReplacing.5 ";
10118               N->dump(&DAG);
10119               dbgs() << "\nWith: ";
10120               Result.getNode()->dump(&DAG);
10121               dbgs() << '\n');
10122         WorklistRemover DeadNodes(*this);
10123         if (isLoad) {
10124           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
10125           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
10126         } else {
10127           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
10128         }
10129 
10130         // Finally, since the node is now dead, remove it from the graph.
10131         deleteAndRecombine(N);
10132 
10133         // Replace the uses of Use with uses of the updated base value.
10134         DAG.ReplaceAllUsesOfValueWith(SDValue(Op, 0),
10135                                       Result.getValue(isLoad ? 1 : 0));
10136         deleteAndRecombine(Op);
10137         return true;
10138       }
10139     }
10140   }
10141 
10142   return false;
10143 }
10144 
10145 /// \brief Return the base-pointer arithmetic from an indexed \p LD.
10146 SDValue DAGCombiner::SplitIndexingFromLoad(LoadSDNode *LD) {
10147   ISD::MemIndexedMode AM = LD->getAddressingMode();
10148   assert(AM != ISD::UNINDEXED);
10149   SDValue BP = LD->getOperand(1);
10150   SDValue Inc = LD->getOperand(2);
10151 
10152   // Some backends use TargetConstants for load offsets, but don't expect
10153   // TargetConstants in general ADD nodes. We can convert these constants into
10154   // regular Constants (if the constant is not opaque).
10155   assert((Inc.getOpcode() != ISD::TargetConstant ||
10156           !cast<ConstantSDNode>(Inc)->isOpaque()) &&
10157          "Cannot split out indexing using opaque target constants");
10158   if (Inc.getOpcode() == ISD::TargetConstant) {
10159     ConstantSDNode *ConstInc = cast<ConstantSDNode>(Inc);
10160     Inc = DAG.getConstant(*ConstInc->getConstantIntValue(), SDLoc(Inc),
10161                           ConstInc->getValueType(0));
10162   }
10163 
10164   unsigned Opc =
10165       (AM == ISD::PRE_INC || AM == ISD::POST_INC ? ISD::ADD : ISD::SUB);
10166   return DAG.getNode(Opc, SDLoc(LD), BP.getSimpleValueType(), BP, Inc);
10167 }
10168 
10169 SDValue DAGCombiner::visitLOAD(SDNode *N) {
10170   LoadSDNode *LD  = cast<LoadSDNode>(N);
10171   SDValue Chain = LD->getChain();
10172   SDValue Ptr   = LD->getBasePtr();
10173 
10174   // If load is not volatile and there are no uses of the loaded value (and
10175   // the updated indexed value in case of indexed loads), change uses of the
10176   // chain value into uses of the chain input (i.e. delete the dead load).
10177   if (!LD->isVolatile()) {
10178     if (N->getValueType(1) == MVT::Other) {
10179       // Unindexed loads.
10180       if (!N->hasAnyUseOfValue(0)) {
10181         // It's not safe to use the two value CombineTo variant here. e.g.
10182         // v1, chain2 = load chain1, loc
10183         // v2, chain3 = load chain2, loc
10184         // v3         = add v2, c
10185         // Now we replace use of chain2 with chain1.  This makes the second load
10186         // isomorphic to the one we are deleting, and thus makes this load live.
10187         DEBUG(dbgs() << "\nReplacing.6 ";
10188               N->dump(&DAG);
10189               dbgs() << "\nWith chain: ";
10190               Chain.getNode()->dump(&DAG);
10191               dbgs() << "\n");
10192         WorklistRemover DeadNodes(*this);
10193         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
10194 
10195         if (N->use_empty())
10196           deleteAndRecombine(N);
10197 
10198         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10199       }
10200     } else {
10201       // Indexed loads.
10202       assert(N->getValueType(2) == MVT::Other && "Malformed indexed loads?");
10203 
10204       // If this load has an opaque TargetConstant offset, then we cannot split
10205       // the indexing into an add/sub directly (that TargetConstant may not be
10206       // valid for a different type of node, and we cannot convert an opaque
10207       // target constant into a regular constant).
10208       bool HasOTCInc = LD->getOperand(2).getOpcode() == ISD::TargetConstant &&
10209                        cast<ConstantSDNode>(LD->getOperand(2))->isOpaque();
10210 
10211       if (!N->hasAnyUseOfValue(0) &&
10212           ((MaySplitLoadIndex && !HasOTCInc) || !N->hasAnyUseOfValue(1))) {
10213         SDValue Undef = DAG.getUNDEF(N->getValueType(0));
10214         SDValue Index;
10215         if (N->hasAnyUseOfValue(1) && MaySplitLoadIndex && !HasOTCInc) {
10216           Index = SplitIndexingFromLoad(LD);
10217           // Try to fold the base pointer arithmetic into subsequent loads and
10218           // stores.
10219           AddUsersToWorklist(N);
10220         } else
10221           Index = DAG.getUNDEF(N->getValueType(1));
10222         DEBUG(dbgs() << "\nReplacing.7 ";
10223               N->dump(&DAG);
10224               dbgs() << "\nWith: ";
10225               Undef.getNode()->dump(&DAG);
10226               dbgs() << " and 2 other values\n");
10227         WorklistRemover DeadNodes(*this);
10228         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Undef);
10229         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Index);
10230         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 2), Chain);
10231         deleteAndRecombine(N);
10232         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10233       }
10234     }
10235   }
10236 
10237   // If this load is directly stored, replace the load value with the stored
10238   // value.
10239   // TODO: Handle store large -> read small portion.
10240   // TODO: Handle TRUNCSTORE/LOADEXT
10241   if (OptLevel != CodeGenOpt::None &&
10242       ISD::isNormalLoad(N) && !LD->isVolatile()) {
10243     if (ISD::isNON_TRUNCStore(Chain.getNode())) {
10244       StoreSDNode *PrevST = cast<StoreSDNode>(Chain);
10245       if (PrevST->getBasePtr() == Ptr &&
10246           PrevST->getValue().getValueType() == N->getValueType(0))
10247       return CombineTo(N, Chain.getOperand(1), Chain);
10248     }
10249   }
10250 
10251   // Try to infer better alignment information than the load already has.
10252   if (OptLevel != CodeGenOpt::None && LD->isUnindexed()) {
10253     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
10254       if (Align > LD->getMemOperand()->getBaseAlignment()) {
10255         SDValue NewLoad = DAG.getExtLoad(
10256             LD->getExtensionType(), SDLoc(N), LD->getValueType(0), Chain, Ptr,
10257             LD->getPointerInfo(), LD->getMemoryVT(), Align,
10258             LD->getMemOperand()->getFlags(), LD->getAAInfo());
10259         if (NewLoad.getNode() != N)
10260           return CombineTo(N, NewLoad, SDValue(NewLoad.getNode(), 1), true);
10261       }
10262     }
10263   }
10264 
10265   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
10266                                                   : DAG.getSubtarget().useAA();
10267 #ifndef NDEBUG
10268   if (CombinerAAOnlyFunc.getNumOccurrences() &&
10269       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
10270     UseAA = false;
10271 #endif
10272   if (UseAA && LD->isUnindexed()) {
10273     // Walk up chain skipping non-aliasing memory nodes.
10274     SDValue BetterChain = FindBetterChain(N, Chain);
10275 
10276     // If there is a better chain.
10277     if (Chain != BetterChain) {
10278       SDValue ReplLoad;
10279 
10280       // Replace the chain to void dependency.
10281       if (LD->getExtensionType() == ISD::NON_EXTLOAD) {
10282         ReplLoad = DAG.getLoad(N->getValueType(0), SDLoc(LD),
10283                                BetterChain, Ptr, LD->getMemOperand());
10284       } else {
10285         ReplLoad = DAG.getExtLoad(LD->getExtensionType(), SDLoc(LD),
10286                                   LD->getValueType(0),
10287                                   BetterChain, Ptr, LD->getMemoryVT(),
10288                                   LD->getMemOperand());
10289       }
10290 
10291       // Create token factor to keep old chain connected.
10292       SDValue Token = DAG.getNode(ISD::TokenFactor, SDLoc(N),
10293                                   MVT::Other, Chain, ReplLoad.getValue(1));
10294 
10295       // Make sure the new and old chains are cleaned up.
10296       AddToWorklist(Token.getNode());
10297 
10298       // Replace uses with load result and token factor. Don't add users
10299       // to work list.
10300       return CombineTo(N, ReplLoad.getValue(0), Token, false);
10301     }
10302   }
10303 
10304   // Try transforming N to an indexed load.
10305   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
10306     return SDValue(N, 0);
10307 
10308   // Try to slice up N to more direct loads if the slices are mapped to
10309   // different register banks or pairing can take place.
10310   if (SliceUpLoad(N))
10311     return SDValue(N, 0);
10312 
10313   return SDValue();
10314 }
10315 
10316 namespace {
10317 /// \brief Helper structure used to slice a load in smaller loads.
10318 /// Basically a slice is obtained from the following sequence:
10319 /// Origin = load Ty1, Base
10320 /// Shift = srl Ty1 Origin, CstTy Amount
10321 /// Inst = trunc Shift to Ty2
10322 ///
10323 /// Then, it will be rewriten into:
10324 /// Slice = load SliceTy, Base + SliceOffset
10325 /// [Inst = zext Slice to Ty2], only if SliceTy <> Ty2
10326 ///
10327 /// SliceTy is deduced from the number of bits that are actually used to
10328 /// build Inst.
10329 struct LoadedSlice {
10330   /// \brief Helper structure used to compute the cost of a slice.
10331   struct Cost {
10332     /// Are we optimizing for code size.
10333     bool ForCodeSize;
10334     /// Various cost.
10335     unsigned Loads;
10336     unsigned Truncates;
10337     unsigned CrossRegisterBanksCopies;
10338     unsigned ZExts;
10339     unsigned Shift;
10340 
10341     Cost(bool ForCodeSize = false)
10342         : ForCodeSize(ForCodeSize), Loads(0), Truncates(0),
10343           CrossRegisterBanksCopies(0), ZExts(0), Shift(0) {}
10344 
10345     /// \brief Get the cost of one isolated slice.
10346     Cost(const LoadedSlice &LS, bool ForCodeSize = false)
10347         : ForCodeSize(ForCodeSize), Loads(1), Truncates(0),
10348           CrossRegisterBanksCopies(0), ZExts(0), Shift(0) {
10349       EVT TruncType = LS.Inst->getValueType(0);
10350       EVT LoadedType = LS.getLoadedType();
10351       if (TruncType != LoadedType &&
10352           !LS.DAG->getTargetLoweringInfo().isZExtFree(LoadedType, TruncType))
10353         ZExts = 1;
10354     }
10355 
10356     /// \brief Account for slicing gain in the current cost.
10357     /// Slicing provide a few gains like removing a shift or a
10358     /// truncate. This method allows to grow the cost of the original
10359     /// load with the gain from this slice.
10360     void addSliceGain(const LoadedSlice &LS) {
10361       // Each slice saves a truncate.
10362       const TargetLowering &TLI = LS.DAG->getTargetLoweringInfo();
10363       if (!TLI.isTruncateFree(LS.Inst->getOperand(0).getValueType(),
10364                               LS.Inst->getValueType(0)))
10365         ++Truncates;
10366       // If there is a shift amount, this slice gets rid of it.
10367       if (LS.Shift)
10368         ++Shift;
10369       // If this slice can merge a cross register bank copy, account for it.
10370       if (LS.canMergeExpensiveCrossRegisterBankCopy())
10371         ++CrossRegisterBanksCopies;
10372     }
10373 
10374     Cost &operator+=(const Cost &RHS) {
10375       Loads += RHS.Loads;
10376       Truncates += RHS.Truncates;
10377       CrossRegisterBanksCopies += RHS.CrossRegisterBanksCopies;
10378       ZExts += RHS.ZExts;
10379       Shift += RHS.Shift;
10380       return *this;
10381     }
10382 
10383     bool operator==(const Cost &RHS) const {
10384       return Loads == RHS.Loads && Truncates == RHS.Truncates &&
10385              CrossRegisterBanksCopies == RHS.CrossRegisterBanksCopies &&
10386              ZExts == RHS.ZExts && Shift == RHS.Shift;
10387     }
10388 
10389     bool operator!=(const Cost &RHS) const { return !(*this == RHS); }
10390 
10391     bool operator<(const Cost &RHS) const {
10392       // Assume cross register banks copies are as expensive as loads.
10393       // FIXME: Do we want some more target hooks?
10394       unsigned ExpensiveOpsLHS = Loads + CrossRegisterBanksCopies;
10395       unsigned ExpensiveOpsRHS = RHS.Loads + RHS.CrossRegisterBanksCopies;
10396       // Unless we are optimizing for code size, consider the
10397       // expensive operation first.
10398       if (!ForCodeSize && ExpensiveOpsLHS != ExpensiveOpsRHS)
10399         return ExpensiveOpsLHS < ExpensiveOpsRHS;
10400       return (Truncates + ZExts + Shift + ExpensiveOpsLHS) <
10401              (RHS.Truncates + RHS.ZExts + RHS.Shift + ExpensiveOpsRHS);
10402     }
10403 
10404     bool operator>(const Cost &RHS) const { return RHS < *this; }
10405 
10406     bool operator<=(const Cost &RHS) const { return !(RHS < *this); }
10407 
10408     bool operator>=(const Cost &RHS) const { return !(*this < RHS); }
10409   };
10410   // The last instruction that represent the slice. This should be a
10411   // truncate instruction.
10412   SDNode *Inst;
10413   // The original load instruction.
10414   LoadSDNode *Origin;
10415   // The right shift amount in bits from the original load.
10416   unsigned Shift;
10417   // The DAG from which Origin came from.
10418   // This is used to get some contextual information about legal types, etc.
10419   SelectionDAG *DAG;
10420 
10421   LoadedSlice(SDNode *Inst = nullptr, LoadSDNode *Origin = nullptr,
10422               unsigned Shift = 0, SelectionDAG *DAG = nullptr)
10423       : Inst(Inst), Origin(Origin), Shift(Shift), DAG(DAG) {}
10424 
10425   /// \brief Get the bits used in a chunk of bits \p BitWidth large.
10426   /// \return Result is \p BitWidth and has used bits set to 1 and
10427   ///         not used bits set to 0.
10428   APInt getUsedBits() const {
10429     // Reproduce the trunc(lshr) sequence:
10430     // - Start from the truncated value.
10431     // - Zero extend to the desired bit width.
10432     // - Shift left.
10433     assert(Origin && "No original load to compare against.");
10434     unsigned BitWidth = Origin->getValueSizeInBits(0);
10435     assert(Inst && "This slice is not bound to an instruction");
10436     assert(Inst->getValueSizeInBits(0) <= BitWidth &&
10437            "Extracted slice is bigger than the whole type!");
10438     APInt UsedBits(Inst->getValueSizeInBits(0), 0);
10439     UsedBits.setAllBits();
10440     UsedBits = UsedBits.zext(BitWidth);
10441     UsedBits <<= Shift;
10442     return UsedBits;
10443   }
10444 
10445   /// \brief Get the size of the slice to be loaded in bytes.
10446   unsigned getLoadedSize() const {
10447     unsigned SliceSize = getUsedBits().countPopulation();
10448     assert(!(SliceSize & 0x7) && "Size is not a multiple of a byte.");
10449     return SliceSize / 8;
10450   }
10451 
10452   /// \brief Get the type that will be loaded for this slice.
10453   /// Note: This may not be the final type for the slice.
10454   EVT getLoadedType() const {
10455     assert(DAG && "Missing context");
10456     LLVMContext &Ctxt = *DAG->getContext();
10457     return EVT::getIntegerVT(Ctxt, getLoadedSize() * 8);
10458   }
10459 
10460   /// \brief Get the alignment of the load used for this slice.
10461   unsigned getAlignment() const {
10462     unsigned Alignment = Origin->getAlignment();
10463     unsigned Offset = getOffsetFromBase();
10464     if (Offset != 0)
10465       Alignment = MinAlign(Alignment, Alignment + Offset);
10466     return Alignment;
10467   }
10468 
10469   /// \brief Check if this slice can be rewritten with legal operations.
10470   bool isLegal() const {
10471     // An invalid slice is not legal.
10472     if (!Origin || !Inst || !DAG)
10473       return false;
10474 
10475     // Offsets are for indexed load only, we do not handle that.
10476     if (!Origin->getOffset().isUndef())
10477       return false;
10478 
10479     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
10480 
10481     // Check that the type is legal.
10482     EVT SliceType = getLoadedType();
10483     if (!TLI.isTypeLegal(SliceType))
10484       return false;
10485 
10486     // Check that the load is legal for this type.
10487     if (!TLI.isOperationLegal(ISD::LOAD, SliceType))
10488       return false;
10489 
10490     // Check that the offset can be computed.
10491     // 1. Check its type.
10492     EVT PtrType = Origin->getBasePtr().getValueType();
10493     if (PtrType == MVT::Untyped || PtrType.isExtended())
10494       return false;
10495 
10496     // 2. Check that it fits in the immediate.
10497     if (!TLI.isLegalAddImmediate(getOffsetFromBase()))
10498       return false;
10499 
10500     // 3. Check that the computation is legal.
10501     if (!TLI.isOperationLegal(ISD::ADD, PtrType))
10502       return false;
10503 
10504     // Check that the zext is legal if it needs one.
10505     EVT TruncateType = Inst->getValueType(0);
10506     if (TruncateType != SliceType &&
10507         !TLI.isOperationLegal(ISD::ZERO_EXTEND, TruncateType))
10508       return false;
10509 
10510     return true;
10511   }
10512 
10513   /// \brief Get the offset in bytes of this slice in the original chunk of
10514   /// bits.
10515   /// \pre DAG != nullptr.
10516   uint64_t getOffsetFromBase() const {
10517     assert(DAG && "Missing context.");
10518     bool IsBigEndian = DAG->getDataLayout().isBigEndian();
10519     assert(!(Shift & 0x7) && "Shifts not aligned on Bytes are not supported.");
10520     uint64_t Offset = Shift / 8;
10521     unsigned TySizeInBytes = Origin->getValueSizeInBits(0) / 8;
10522     assert(!(Origin->getValueSizeInBits(0) & 0x7) &&
10523            "The size of the original loaded type is not a multiple of a"
10524            " byte.");
10525     // If Offset is bigger than TySizeInBytes, it means we are loading all
10526     // zeros. This should have been optimized before in the process.
10527     assert(TySizeInBytes > Offset &&
10528            "Invalid shift amount for given loaded size");
10529     if (IsBigEndian)
10530       Offset = TySizeInBytes - Offset - getLoadedSize();
10531     return Offset;
10532   }
10533 
10534   /// \brief Generate the sequence of instructions to load the slice
10535   /// represented by this object and redirect the uses of this slice to
10536   /// this new sequence of instructions.
10537   /// \pre this->Inst && this->Origin are valid Instructions and this
10538   /// object passed the legal check: LoadedSlice::isLegal returned true.
10539   /// \return The last instruction of the sequence used to load the slice.
10540   SDValue loadSlice() const {
10541     assert(Inst && Origin && "Unable to replace a non-existing slice.");
10542     const SDValue &OldBaseAddr = Origin->getBasePtr();
10543     SDValue BaseAddr = OldBaseAddr;
10544     // Get the offset in that chunk of bytes w.r.t. the endianess.
10545     int64_t Offset = static_cast<int64_t>(getOffsetFromBase());
10546     assert(Offset >= 0 && "Offset too big to fit in int64_t!");
10547     if (Offset) {
10548       // BaseAddr = BaseAddr + Offset.
10549       EVT ArithType = BaseAddr.getValueType();
10550       SDLoc DL(Origin);
10551       BaseAddr = DAG->getNode(ISD::ADD, DL, ArithType, BaseAddr,
10552                               DAG->getConstant(Offset, DL, ArithType));
10553     }
10554 
10555     // Create the type of the loaded slice according to its size.
10556     EVT SliceType = getLoadedType();
10557 
10558     // Create the load for the slice.
10559     SDValue LastInst =
10560         DAG->getLoad(SliceType, SDLoc(Origin), Origin->getChain(), BaseAddr,
10561                      Origin->getPointerInfo().getWithOffset(Offset),
10562                      getAlignment(), Origin->getMemOperand()->getFlags());
10563     // If the final type is not the same as the loaded type, this means that
10564     // we have to pad with zero. Create a zero extend for that.
10565     EVT FinalType = Inst->getValueType(0);
10566     if (SliceType != FinalType)
10567       LastInst =
10568           DAG->getNode(ISD::ZERO_EXTEND, SDLoc(LastInst), FinalType, LastInst);
10569     return LastInst;
10570   }
10571 
10572   /// \brief Check if this slice can be merged with an expensive cross register
10573   /// bank copy. E.g.,
10574   /// i = load i32
10575   /// f = bitcast i32 i to float
10576   bool canMergeExpensiveCrossRegisterBankCopy() const {
10577     if (!Inst || !Inst->hasOneUse())
10578       return false;
10579     SDNode *Use = *Inst->use_begin();
10580     if (Use->getOpcode() != ISD::BITCAST)
10581       return false;
10582     assert(DAG && "Missing context");
10583     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
10584     EVT ResVT = Use->getValueType(0);
10585     const TargetRegisterClass *ResRC = TLI.getRegClassFor(ResVT.getSimpleVT());
10586     const TargetRegisterClass *ArgRC =
10587         TLI.getRegClassFor(Use->getOperand(0).getValueType().getSimpleVT());
10588     if (ArgRC == ResRC || !TLI.isOperationLegal(ISD::LOAD, ResVT))
10589       return false;
10590 
10591     // At this point, we know that we perform a cross-register-bank copy.
10592     // Check if it is expensive.
10593     const TargetRegisterInfo *TRI = DAG->getSubtarget().getRegisterInfo();
10594     // Assume bitcasts are cheap, unless both register classes do not
10595     // explicitly share a common sub class.
10596     if (!TRI || TRI->getCommonSubClass(ArgRC, ResRC))
10597       return false;
10598 
10599     // Check if it will be merged with the load.
10600     // 1. Check the alignment constraint.
10601     unsigned RequiredAlignment = DAG->getDataLayout().getABITypeAlignment(
10602         ResVT.getTypeForEVT(*DAG->getContext()));
10603 
10604     if (RequiredAlignment > getAlignment())
10605       return false;
10606 
10607     // 2. Check that the load is a legal operation for that type.
10608     if (!TLI.isOperationLegal(ISD::LOAD, ResVT))
10609       return false;
10610 
10611     // 3. Check that we do not have a zext in the way.
10612     if (Inst->getValueType(0) != getLoadedType())
10613       return false;
10614 
10615     return true;
10616   }
10617 };
10618 }
10619 
10620 /// \brief Check that all bits set in \p UsedBits form a dense region, i.e.,
10621 /// \p UsedBits looks like 0..0 1..1 0..0.
10622 static bool areUsedBitsDense(const APInt &UsedBits) {
10623   // If all the bits are one, this is dense!
10624   if (UsedBits.isAllOnesValue())
10625     return true;
10626 
10627   // Get rid of the unused bits on the right.
10628   APInt NarrowedUsedBits = UsedBits.lshr(UsedBits.countTrailingZeros());
10629   // Get rid of the unused bits on the left.
10630   if (NarrowedUsedBits.countLeadingZeros())
10631     NarrowedUsedBits = NarrowedUsedBits.trunc(NarrowedUsedBits.getActiveBits());
10632   // Check that the chunk of bits is completely used.
10633   return NarrowedUsedBits.isAllOnesValue();
10634 }
10635 
10636 /// \brief Check whether or not \p First and \p Second are next to each other
10637 /// in memory. This means that there is no hole between the bits loaded
10638 /// by \p First and the bits loaded by \p Second.
10639 static bool areSlicesNextToEachOther(const LoadedSlice &First,
10640                                      const LoadedSlice &Second) {
10641   assert(First.Origin == Second.Origin && First.Origin &&
10642          "Unable to match different memory origins.");
10643   APInt UsedBits = First.getUsedBits();
10644   assert((UsedBits & Second.getUsedBits()) == 0 &&
10645          "Slices are not supposed to overlap.");
10646   UsedBits |= Second.getUsedBits();
10647   return areUsedBitsDense(UsedBits);
10648 }
10649 
10650 /// \brief Adjust the \p GlobalLSCost according to the target
10651 /// paring capabilities and the layout of the slices.
10652 /// \pre \p GlobalLSCost should account for at least as many loads as
10653 /// there is in the slices in \p LoadedSlices.
10654 static void adjustCostForPairing(SmallVectorImpl<LoadedSlice> &LoadedSlices,
10655                                  LoadedSlice::Cost &GlobalLSCost) {
10656   unsigned NumberOfSlices = LoadedSlices.size();
10657   // If there is less than 2 elements, no pairing is possible.
10658   if (NumberOfSlices < 2)
10659     return;
10660 
10661   // Sort the slices so that elements that are likely to be next to each
10662   // other in memory are next to each other in the list.
10663   std::sort(LoadedSlices.begin(), LoadedSlices.end(),
10664             [](const LoadedSlice &LHS, const LoadedSlice &RHS) {
10665     assert(LHS.Origin == RHS.Origin && "Different bases not implemented.");
10666     return LHS.getOffsetFromBase() < RHS.getOffsetFromBase();
10667   });
10668   const TargetLowering &TLI = LoadedSlices[0].DAG->getTargetLoweringInfo();
10669   // First (resp. Second) is the first (resp. Second) potentially candidate
10670   // to be placed in a paired load.
10671   const LoadedSlice *First = nullptr;
10672   const LoadedSlice *Second = nullptr;
10673   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice,
10674                 // Set the beginning of the pair.
10675                                                            First = Second) {
10676 
10677     Second = &LoadedSlices[CurrSlice];
10678 
10679     // If First is NULL, it means we start a new pair.
10680     // Get to the next slice.
10681     if (!First)
10682       continue;
10683 
10684     EVT LoadedType = First->getLoadedType();
10685 
10686     // If the types of the slices are different, we cannot pair them.
10687     if (LoadedType != Second->getLoadedType())
10688       continue;
10689 
10690     // Check if the target supplies paired loads for this type.
10691     unsigned RequiredAlignment = 0;
10692     if (!TLI.hasPairedLoad(LoadedType, RequiredAlignment)) {
10693       // move to the next pair, this type is hopeless.
10694       Second = nullptr;
10695       continue;
10696     }
10697     // Check if we meet the alignment requirement.
10698     if (RequiredAlignment > First->getAlignment())
10699       continue;
10700 
10701     // Check that both loads are next to each other in memory.
10702     if (!areSlicesNextToEachOther(*First, *Second))
10703       continue;
10704 
10705     assert(GlobalLSCost.Loads > 0 && "We save more loads than we created!");
10706     --GlobalLSCost.Loads;
10707     // Move to the next pair.
10708     Second = nullptr;
10709   }
10710 }
10711 
10712 /// \brief Check the profitability of all involved LoadedSlice.
10713 /// Currently, it is considered profitable if there is exactly two
10714 /// involved slices (1) which are (2) next to each other in memory, and
10715 /// whose cost (\see LoadedSlice::Cost) is smaller than the original load (3).
10716 ///
10717 /// Note: The order of the elements in \p LoadedSlices may be modified, but not
10718 /// the elements themselves.
10719 ///
10720 /// FIXME: When the cost model will be mature enough, we can relax
10721 /// constraints (1) and (2).
10722 static bool isSlicingProfitable(SmallVectorImpl<LoadedSlice> &LoadedSlices,
10723                                 const APInt &UsedBits, bool ForCodeSize) {
10724   unsigned NumberOfSlices = LoadedSlices.size();
10725   if (StressLoadSlicing)
10726     return NumberOfSlices > 1;
10727 
10728   // Check (1).
10729   if (NumberOfSlices != 2)
10730     return false;
10731 
10732   // Check (2).
10733   if (!areUsedBitsDense(UsedBits))
10734     return false;
10735 
10736   // Check (3).
10737   LoadedSlice::Cost OrigCost(ForCodeSize), GlobalSlicingCost(ForCodeSize);
10738   // The original code has one big load.
10739   OrigCost.Loads = 1;
10740   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice) {
10741     const LoadedSlice &LS = LoadedSlices[CurrSlice];
10742     // Accumulate the cost of all the slices.
10743     LoadedSlice::Cost SliceCost(LS, ForCodeSize);
10744     GlobalSlicingCost += SliceCost;
10745 
10746     // Account as cost in the original configuration the gain obtained
10747     // with the current slices.
10748     OrigCost.addSliceGain(LS);
10749   }
10750 
10751   // If the target supports paired load, adjust the cost accordingly.
10752   adjustCostForPairing(LoadedSlices, GlobalSlicingCost);
10753   return OrigCost > GlobalSlicingCost;
10754 }
10755 
10756 /// \brief If the given load, \p LI, is used only by trunc or trunc(lshr)
10757 /// operations, split it in the various pieces being extracted.
10758 ///
10759 /// This sort of thing is introduced by SROA.
10760 /// This slicing takes care not to insert overlapping loads.
10761 /// \pre LI is a simple load (i.e., not an atomic or volatile load).
10762 bool DAGCombiner::SliceUpLoad(SDNode *N) {
10763   if (Level < AfterLegalizeDAG)
10764     return false;
10765 
10766   LoadSDNode *LD = cast<LoadSDNode>(N);
10767   if (LD->isVolatile() || !ISD::isNormalLoad(LD) ||
10768       !LD->getValueType(0).isInteger())
10769     return false;
10770 
10771   // Keep track of already used bits to detect overlapping values.
10772   // In that case, we will just abort the transformation.
10773   APInt UsedBits(LD->getValueSizeInBits(0), 0);
10774 
10775   SmallVector<LoadedSlice, 4> LoadedSlices;
10776 
10777   // Check if this load is used as several smaller chunks of bits.
10778   // Basically, look for uses in trunc or trunc(lshr) and record a new chain
10779   // of computation for each trunc.
10780   for (SDNode::use_iterator UI = LD->use_begin(), UIEnd = LD->use_end();
10781        UI != UIEnd; ++UI) {
10782     // Skip the uses of the chain.
10783     if (UI.getUse().getResNo() != 0)
10784       continue;
10785 
10786     SDNode *User = *UI;
10787     unsigned Shift = 0;
10788 
10789     // Check if this is a trunc(lshr).
10790     if (User->getOpcode() == ISD::SRL && User->hasOneUse() &&
10791         isa<ConstantSDNode>(User->getOperand(1))) {
10792       Shift = cast<ConstantSDNode>(User->getOperand(1))->getZExtValue();
10793       User = *User->use_begin();
10794     }
10795 
10796     // At this point, User is a Truncate, iff we encountered, trunc or
10797     // trunc(lshr).
10798     if (User->getOpcode() != ISD::TRUNCATE)
10799       return false;
10800 
10801     // The width of the type must be a power of 2 and greater than 8-bits.
10802     // Otherwise the load cannot be represented in LLVM IR.
10803     // Moreover, if we shifted with a non-8-bits multiple, the slice
10804     // will be across several bytes. We do not support that.
10805     unsigned Width = User->getValueSizeInBits(0);
10806     if (Width < 8 || !isPowerOf2_32(Width) || (Shift & 0x7))
10807       return 0;
10808 
10809     // Build the slice for this chain of computations.
10810     LoadedSlice LS(User, LD, Shift, &DAG);
10811     APInt CurrentUsedBits = LS.getUsedBits();
10812 
10813     // Check if this slice overlaps with another.
10814     if ((CurrentUsedBits & UsedBits) != 0)
10815       return false;
10816     // Update the bits used globally.
10817     UsedBits |= CurrentUsedBits;
10818 
10819     // Check if the new slice would be legal.
10820     if (!LS.isLegal())
10821       return false;
10822 
10823     // Record the slice.
10824     LoadedSlices.push_back(LS);
10825   }
10826 
10827   // Abort slicing if it does not seem to be profitable.
10828   if (!isSlicingProfitable(LoadedSlices, UsedBits, ForCodeSize))
10829     return false;
10830 
10831   ++SlicedLoads;
10832 
10833   // Rewrite each chain to use an independent load.
10834   // By construction, each chain can be represented by a unique load.
10835 
10836   // Prepare the argument for the new token factor for all the slices.
10837   SmallVector<SDValue, 8> ArgChains;
10838   for (SmallVectorImpl<LoadedSlice>::const_iterator
10839            LSIt = LoadedSlices.begin(),
10840            LSItEnd = LoadedSlices.end();
10841        LSIt != LSItEnd; ++LSIt) {
10842     SDValue SliceInst = LSIt->loadSlice();
10843     CombineTo(LSIt->Inst, SliceInst, true);
10844     if (SliceInst.getOpcode() != ISD::LOAD)
10845       SliceInst = SliceInst.getOperand(0);
10846     assert(SliceInst->getOpcode() == ISD::LOAD &&
10847            "It takes more than a zext to get to the loaded slice!!");
10848     ArgChains.push_back(SliceInst.getValue(1));
10849   }
10850 
10851   SDValue Chain = DAG.getNode(ISD::TokenFactor, SDLoc(LD), MVT::Other,
10852                               ArgChains);
10853   DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
10854   return true;
10855 }
10856 
10857 /// Check to see if V is (and load (ptr), imm), where the load is having
10858 /// specific bytes cleared out.  If so, return the byte size being masked out
10859 /// and the shift amount.
10860 static std::pair<unsigned, unsigned>
10861 CheckForMaskedLoad(SDValue V, SDValue Ptr, SDValue Chain) {
10862   std::pair<unsigned, unsigned> Result(0, 0);
10863 
10864   // Check for the structure we're looking for.
10865   if (V->getOpcode() != ISD::AND ||
10866       !isa<ConstantSDNode>(V->getOperand(1)) ||
10867       !ISD::isNormalLoad(V->getOperand(0).getNode()))
10868     return Result;
10869 
10870   // Check the chain and pointer.
10871   LoadSDNode *LD = cast<LoadSDNode>(V->getOperand(0));
10872   if (LD->getBasePtr() != Ptr) return Result;  // Not from same pointer.
10873 
10874   // The store should be chained directly to the load or be an operand of a
10875   // tokenfactor.
10876   if (LD == Chain.getNode())
10877     ; // ok.
10878   else if (Chain->getOpcode() != ISD::TokenFactor)
10879     return Result; // Fail.
10880   else {
10881     bool isOk = false;
10882     for (const SDValue &ChainOp : Chain->op_values())
10883       if (ChainOp.getNode() == LD) {
10884         isOk = true;
10885         break;
10886       }
10887     if (!isOk) return Result;
10888   }
10889 
10890   // This only handles simple types.
10891   if (V.getValueType() != MVT::i16 &&
10892       V.getValueType() != MVT::i32 &&
10893       V.getValueType() != MVT::i64)
10894     return Result;
10895 
10896   // Check the constant mask.  Invert it so that the bits being masked out are
10897   // 0 and the bits being kept are 1.  Use getSExtValue so that leading bits
10898   // follow the sign bit for uniformity.
10899   uint64_t NotMask = ~cast<ConstantSDNode>(V->getOperand(1))->getSExtValue();
10900   unsigned NotMaskLZ = countLeadingZeros(NotMask);
10901   if (NotMaskLZ & 7) return Result;  // Must be multiple of a byte.
10902   unsigned NotMaskTZ = countTrailingZeros(NotMask);
10903   if (NotMaskTZ & 7) return Result;  // Must be multiple of a byte.
10904   if (NotMaskLZ == 64) return Result;  // All zero mask.
10905 
10906   // See if we have a continuous run of bits.  If so, we have 0*1+0*
10907   if (countTrailingOnes(NotMask >> NotMaskTZ) + NotMaskTZ + NotMaskLZ != 64)
10908     return Result;
10909 
10910   // Adjust NotMaskLZ down to be from the actual size of the int instead of i64.
10911   if (V.getValueType() != MVT::i64 && NotMaskLZ)
10912     NotMaskLZ -= 64-V.getValueSizeInBits();
10913 
10914   unsigned MaskedBytes = (V.getValueSizeInBits()-NotMaskLZ-NotMaskTZ)/8;
10915   switch (MaskedBytes) {
10916   case 1:
10917   case 2:
10918   case 4: break;
10919   default: return Result; // All one mask, or 5-byte mask.
10920   }
10921 
10922   // Verify that the first bit starts at a multiple of mask so that the access
10923   // is aligned the same as the access width.
10924   if (NotMaskTZ && NotMaskTZ/8 % MaskedBytes) return Result;
10925 
10926   Result.first = MaskedBytes;
10927   Result.second = NotMaskTZ/8;
10928   return Result;
10929 }
10930 
10931 
10932 /// Check to see if IVal is something that provides a value as specified by
10933 /// MaskInfo. If so, replace the specified store with a narrower store of
10934 /// truncated IVal.
10935 static SDNode *
10936 ShrinkLoadReplaceStoreWithStore(const std::pair<unsigned, unsigned> &MaskInfo,
10937                                 SDValue IVal, StoreSDNode *St,
10938                                 DAGCombiner *DC) {
10939   unsigned NumBytes = MaskInfo.first;
10940   unsigned ByteShift = MaskInfo.second;
10941   SelectionDAG &DAG = DC->getDAG();
10942 
10943   // Check to see if IVal is all zeros in the part being masked in by the 'or'
10944   // that uses this.  If not, this is not a replacement.
10945   APInt Mask = ~APInt::getBitsSet(IVal.getValueSizeInBits(),
10946                                   ByteShift*8, (ByteShift+NumBytes)*8);
10947   if (!DAG.MaskedValueIsZero(IVal, Mask)) return nullptr;
10948 
10949   // Check that it is legal on the target to do this.  It is legal if the new
10950   // VT we're shrinking to (i8/i16/i32) is legal or we're still before type
10951   // legalization.
10952   MVT VT = MVT::getIntegerVT(NumBytes*8);
10953   if (!DC->isTypeLegal(VT))
10954     return nullptr;
10955 
10956   // Okay, we can do this!  Replace the 'St' store with a store of IVal that is
10957   // shifted by ByteShift and truncated down to NumBytes.
10958   if (ByteShift) {
10959     SDLoc DL(IVal);
10960     IVal = DAG.getNode(ISD::SRL, DL, IVal.getValueType(), IVal,
10961                        DAG.getConstant(ByteShift*8, DL,
10962                                     DC->getShiftAmountTy(IVal.getValueType())));
10963   }
10964 
10965   // Figure out the offset for the store and the alignment of the access.
10966   unsigned StOffset;
10967   unsigned NewAlign = St->getAlignment();
10968 
10969   if (DAG.getDataLayout().isLittleEndian())
10970     StOffset = ByteShift;
10971   else
10972     StOffset = IVal.getValueType().getStoreSize() - ByteShift - NumBytes;
10973 
10974   SDValue Ptr = St->getBasePtr();
10975   if (StOffset) {
10976     SDLoc DL(IVal);
10977     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(),
10978                       Ptr, DAG.getConstant(StOffset, DL, Ptr.getValueType()));
10979     NewAlign = MinAlign(NewAlign, StOffset);
10980   }
10981 
10982   // Truncate down to the new size.
10983   IVal = DAG.getNode(ISD::TRUNCATE, SDLoc(IVal), VT, IVal);
10984 
10985   ++OpsNarrowed;
10986   return DAG
10987       .getStore(St->getChain(), SDLoc(St), IVal, Ptr,
10988                 St->getPointerInfo().getWithOffset(StOffset), NewAlign)
10989       .getNode();
10990 }
10991 
10992 
10993 /// Look for sequence of load / op / store where op is one of 'or', 'xor', and
10994 /// 'and' of immediates. If 'op' is only touching some of the loaded bits, try
10995 /// narrowing the load and store if it would end up being a win for performance
10996 /// or code size.
10997 SDValue DAGCombiner::ReduceLoadOpStoreWidth(SDNode *N) {
10998   StoreSDNode *ST  = cast<StoreSDNode>(N);
10999   if (ST->isVolatile())
11000     return SDValue();
11001 
11002   SDValue Chain = ST->getChain();
11003   SDValue Value = ST->getValue();
11004   SDValue Ptr   = ST->getBasePtr();
11005   EVT VT = Value.getValueType();
11006 
11007   if (ST->isTruncatingStore() || VT.isVector() || !Value.hasOneUse())
11008     return SDValue();
11009 
11010   unsigned Opc = Value.getOpcode();
11011 
11012   // If this is "store (or X, Y), P" and X is "(and (load P), cst)", where cst
11013   // is a byte mask indicating a consecutive number of bytes, check to see if
11014   // Y is known to provide just those bytes.  If so, we try to replace the
11015   // load + replace + store sequence with a single (narrower) store, which makes
11016   // the load dead.
11017   if (Opc == ISD::OR) {
11018     std::pair<unsigned, unsigned> MaskedLoad;
11019     MaskedLoad = CheckForMaskedLoad(Value.getOperand(0), Ptr, Chain);
11020     if (MaskedLoad.first)
11021       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
11022                                                   Value.getOperand(1), ST,this))
11023         return SDValue(NewST, 0);
11024 
11025     // Or is commutative, so try swapping X and Y.
11026     MaskedLoad = CheckForMaskedLoad(Value.getOperand(1), Ptr, Chain);
11027     if (MaskedLoad.first)
11028       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
11029                                                   Value.getOperand(0), ST,this))
11030         return SDValue(NewST, 0);
11031   }
11032 
11033   if ((Opc != ISD::OR && Opc != ISD::XOR && Opc != ISD::AND) ||
11034       Value.getOperand(1).getOpcode() != ISD::Constant)
11035     return SDValue();
11036 
11037   SDValue N0 = Value.getOperand(0);
11038   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
11039       Chain == SDValue(N0.getNode(), 1)) {
11040     LoadSDNode *LD = cast<LoadSDNode>(N0);
11041     if (LD->getBasePtr() != Ptr ||
11042         LD->getPointerInfo().getAddrSpace() !=
11043         ST->getPointerInfo().getAddrSpace())
11044       return SDValue();
11045 
11046     // Find the type to narrow it the load / op / store to.
11047     SDValue N1 = Value.getOperand(1);
11048     unsigned BitWidth = N1.getValueSizeInBits();
11049     APInt Imm = cast<ConstantSDNode>(N1)->getAPIntValue();
11050     if (Opc == ISD::AND)
11051       Imm ^= APInt::getAllOnesValue(BitWidth);
11052     if (Imm == 0 || Imm.isAllOnesValue())
11053       return SDValue();
11054     unsigned ShAmt = Imm.countTrailingZeros();
11055     unsigned MSB = BitWidth - Imm.countLeadingZeros() - 1;
11056     unsigned NewBW = NextPowerOf2(MSB - ShAmt);
11057     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
11058     // The narrowing should be profitable, the load/store operation should be
11059     // legal (or custom) and the store size should be equal to the NewVT width.
11060     while (NewBW < BitWidth &&
11061            (NewVT.getStoreSizeInBits() != NewBW ||
11062             !TLI.isOperationLegalOrCustom(Opc, NewVT) ||
11063             !TLI.isNarrowingProfitable(VT, NewVT))) {
11064       NewBW = NextPowerOf2(NewBW);
11065       NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
11066     }
11067     if (NewBW >= BitWidth)
11068       return SDValue();
11069 
11070     // If the lsb changed does not start at the type bitwidth boundary,
11071     // start at the previous one.
11072     if (ShAmt % NewBW)
11073       ShAmt = (((ShAmt + NewBW - 1) / NewBW) * NewBW) - NewBW;
11074     APInt Mask = APInt::getBitsSet(BitWidth, ShAmt,
11075                                    std::min(BitWidth, ShAmt + NewBW));
11076     if ((Imm & Mask) == Imm) {
11077       APInt NewImm = (Imm & Mask).lshr(ShAmt).trunc(NewBW);
11078       if (Opc == ISD::AND)
11079         NewImm ^= APInt::getAllOnesValue(NewBW);
11080       uint64_t PtrOff = ShAmt / 8;
11081       // For big endian targets, we need to adjust the offset to the pointer to
11082       // load the correct bytes.
11083       if (DAG.getDataLayout().isBigEndian())
11084         PtrOff = (BitWidth + 7 - NewBW) / 8 - PtrOff;
11085 
11086       unsigned NewAlign = MinAlign(LD->getAlignment(), PtrOff);
11087       Type *NewVTTy = NewVT.getTypeForEVT(*DAG.getContext());
11088       if (NewAlign < DAG.getDataLayout().getABITypeAlignment(NewVTTy))
11089         return SDValue();
11090 
11091       SDValue NewPtr = DAG.getNode(ISD::ADD, SDLoc(LD),
11092                                    Ptr.getValueType(), Ptr,
11093                                    DAG.getConstant(PtrOff, SDLoc(LD),
11094                                                    Ptr.getValueType()));
11095       SDValue NewLD =
11096           DAG.getLoad(NewVT, SDLoc(N0), LD->getChain(), NewPtr,
11097                       LD->getPointerInfo().getWithOffset(PtrOff), NewAlign,
11098                       LD->getMemOperand()->getFlags(), LD->getAAInfo());
11099       SDValue NewVal = DAG.getNode(Opc, SDLoc(Value), NewVT, NewLD,
11100                                    DAG.getConstant(NewImm, SDLoc(Value),
11101                                                    NewVT));
11102       SDValue NewST =
11103           DAG.getStore(Chain, SDLoc(N), NewVal, NewPtr,
11104                        ST->getPointerInfo().getWithOffset(PtrOff), NewAlign);
11105 
11106       AddToWorklist(NewPtr.getNode());
11107       AddToWorklist(NewLD.getNode());
11108       AddToWorklist(NewVal.getNode());
11109       WorklistRemover DeadNodes(*this);
11110       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLD.getValue(1));
11111       ++OpsNarrowed;
11112       return NewST;
11113     }
11114   }
11115 
11116   return SDValue();
11117 }
11118 
11119 /// For a given floating point load / store pair, if the load value isn't used
11120 /// by any other operations, then consider transforming the pair to integer
11121 /// load / store operations if the target deems the transformation profitable.
11122 SDValue DAGCombiner::TransformFPLoadStorePair(SDNode *N) {
11123   StoreSDNode *ST  = cast<StoreSDNode>(N);
11124   SDValue Chain = ST->getChain();
11125   SDValue Value = ST->getValue();
11126   if (ISD::isNormalStore(ST) && ISD::isNormalLoad(Value.getNode()) &&
11127       Value.hasOneUse() &&
11128       Chain == SDValue(Value.getNode(), 1)) {
11129     LoadSDNode *LD = cast<LoadSDNode>(Value);
11130     EVT VT = LD->getMemoryVT();
11131     if (!VT.isFloatingPoint() ||
11132         VT != ST->getMemoryVT() ||
11133         LD->isNonTemporal() ||
11134         ST->isNonTemporal() ||
11135         LD->getPointerInfo().getAddrSpace() != 0 ||
11136         ST->getPointerInfo().getAddrSpace() != 0)
11137       return SDValue();
11138 
11139     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits());
11140     if (!TLI.isOperationLegal(ISD::LOAD, IntVT) ||
11141         !TLI.isOperationLegal(ISD::STORE, IntVT) ||
11142         !TLI.isDesirableToTransformToIntegerOp(ISD::LOAD, VT) ||
11143         !TLI.isDesirableToTransformToIntegerOp(ISD::STORE, VT))
11144       return SDValue();
11145 
11146     unsigned LDAlign = LD->getAlignment();
11147     unsigned STAlign = ST->getAlignment();
11148     Type *IntVTTy = IntVT.getTypeForEVT(*DAG.getContext());
11149     unsigned ABIAlign = DAG.getDataLayout().getABITypeAlignment(IntVTTy);
11150     if (LDAlign < ABIAlign || STAlign < ABIAlign)
11151       return SDValue();
11152 
11153     SDValue NewLD =
11154         DAG.getLoad(IntVT, SDLoc(Value), LD->getChain(), LD->getBasePtr(),
11155                     LD->getPointerInfo(), LDAlign);
11156 
11157     SDValue NewST =
11158         DAG.getStore(NewLD.getValue(1), SDLoc(N), NewLD, ST->getBasePtr(),
11159                      ST->getPointerInfo(), STAlign);
11160 
11161     AddToWorklist(NewLD.getNode());
11162     AddToWorklist(NewST.getNode());
11163     WorklistRemover DeadNodes(*this);
11164     DAG.ReplaceAllUsesOfValueWith(Value.getValue(1), NewLD.getValue(1));
11165     ++LdStFP2Int;
11166     return NewST;
11167   }
11168 
11169   return SDValue();
11170 }
11171 
11172 namespace {
11173 /// Helper struct to parse and store a memory address as base + index + offset.
11174 /// We ignore sign extensions when it is safe to do so.
11175 /// The following two expressions are not equivalent. To differentiate we need
11176 /// to store whether there was a sign extension involved in the index
11177 /// computation.
11178 ///  (load (i64 add (i64 copyfromreg %c)
11179 ///                 (i64 signextend (add (i8 load %index)
11180 ///                                      (i8 1))))
11181 /// vs
11182 ///
11183 /// (load (i64 add (i64 copyfromreg %c)
11184 ///                (i64 signextend (i32 add (i32 signextend (i8 load %index))
11185 ///                                         (i32 1)))))
11186 struct BaseIndexOffset {
11187   SDValue Base;
11188   SDValue Index;
11189   int64_t Offset;
11190   bool IsIndexSignExt;
11191 
11192   BaseIndexOffset() : Offset(0), IsIndexSignExt(false) {}
11193 
11194   BaseIndexOffset(SDValue Base, SDValue Index, int64_t Offset,
11195                   bool IsIndexSignExt) :
11196     Base(Base), Index(Index), Offset(Offset), IsIndexSignExt(IsIndexSignExt) {}
11197 
11198   bool equalBaseIndex(const BaseIndexOffset &Other) {
11199     return Other.Base == Base && Other.Index == Index &&
11200       Other.IsIndexSignExt == IsIndexSignExt;
11201   }
11202 
11203   /// Parses tree in Ptr for base, index, offset addresses.
11204   static BaseIndexOffset match(SDValue Ptr, SelectionDAG &DAG) {
11205     bool IsIndexSignExt = false;
11206 
11207     // Split up a folded GlobalAddress+Offset into its component parts.
11208     if (GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(Ptr))
11209       if (GA->getOpcode() == ISD::GlobalAddress && GA->getOffset() != 0) {
11210         return BaseIndexOffset(DAG.getGlobalAddress(GA->getGlobal(),
11211                                                     SDLoc(GA),
11212                                                     GA->getValueType(0),
11213                                                     /*Offset=*/0,
11214                                                     /*isTargetGA=*/false,
11215                                                     GA->getTargetFlags()),
11216                                SDValue(),
11217                                GA->getOffset(),
11218                                IsIndexSignExt);
11219       }
11220 
11221     // We only can pattern match BASE + INDEX + OFFSET. If Ptr is not an ADD
11222     // instruction, then it could be just the BASE or everything else we don't
11223     // know how to handle. Just use Ptr as BASE and give up.
11224     if (Ptr->getOpcode() != ISD::ADD)
11225       return BaseIndexOffset(Ptr, SDValue(), 0, IsIndexSignExt);
11226 
11227     // We know that we have at least an ADD instruction. Try to pattern match
11228     // the simple case of BASE + OFFSET.
11229     if (isa<ConstantSDNode>(Ptr->getOperand(1))) {
11230       int64_t Offset = cast<ConstantSDNode>(Ptr->getOperand(1))->getSExtValue();
11231       return  BaseIndexOffset(Ptr->getOperand(0), SDValue(), Offset,
11232                               IsIndexSignExt);
11233     }
11234 
11235     // Inside a loop the current BASE pointer is calculated using an ADD and a
11236     // MUL instruction. In this case Ptr is the actual BASE pointer.
11237     // (i64 add (i64 %array_ptr)
11238     //          (i64 mul (i64 %induction_var)
11239     //                   (i64 %element_size)))
11240     if (Ptr->getOperand(1)->getOpcode() == ISD::MUL)
11241       return BaseIndexOffset(Ptr, SDValue(), 0, IsIndexSignExt);
11242 
11243     // Look at Base + Index + Offset cases.
11244     SDValue Base = Ptr->getOperand(0);
11245     SDValue IndexOffset = Ptr->getOperand(1);
11246 
11247     // Skip signextends.
11248     if (IndexOffset->getOpcode() == ISD::SIGN_EXTEND) {
11249       IndexOffset = IndexOffset->getOperand(0);
11250       IsIndexSignExt = true;
11251     }
11252 
11253     // Either the case of Base + Index (no offset) or something else.
11254     if (IndexOffset->getOpcode() != ISD::ADD)
11255       return BaseIndexOffset(Base, IndexOffset, 0, IsIndexSignExt);
11256 
11257     // Now we have the case of Base + Index + offset.
11258     SDValue Index = IndexOffset->getOperand(0);
11259     SDValue Offset = IndexOffset->getOperand(1);
11260 
11261     if (!isa<ConstantSDNode>(Offset))
11262       return BaseIndexOffset(Ptr, SDValue(), 0, IsIndexSignExt);
11263 
11264     // Ignore signextends.
11265     if (Index->getOpcode() == ISD::SIGN_EXTEND) {
11266       Index = Index->getOperand(0);
11267       IsIndexSignExt = true;
11268     } else IsIndexSignExt = false;
11269 
11270     int64_t Off = cast<ConstantSDNode>(Offset)->getSExtValue();
11271     return BaseIndexOffset(Base, Index, Off, IsIndexSignExt);
11272   }
11273 };
11274 } // namespace
11275 
11276 // This is a helper function for visitMUL to check the profitability
11277 // of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
11278 // MulNode is the original multiply, AddNode is (add x, c1),
11279 // and ConstNode is c2.
11280 //
11281 // If the (add x, c1) has multiple uses, we could increase
11282 // the number of adds if we make this transformation.
11283 // It would only be worth doing this if we can remove a
11284 // multiply in the process. Check for that here.
11285 // To illustrate:
11286 //     (A + c1) * c3
11287 //     (A + c2) * c3
11288 // We're checking for cases where we have common "c3 * A" expressions.
11289 bool DAGCombiner::isMulAddWithConstProfitable(SDNode *MulNode,
11290                                               SDValue &AddNode,
11291                                               SDValue &ConstNode) {
11292   APInt Val;
11293 
11294   // If the add only has one use, this would be OK to do.
11295   if (AddNode.getNode()->hasOneUse())
11296     return true;
11297 
11298   // Walk all the users of the constant with which we're multiplying.
11299   for (SDNode *Use : ConstNode->uses()) {
11300 
11301     if (Use == MulNode) // This use is the one we're on right now. Skip it.
11302       continue;
11303 
11304     if (Use->getOpcode() == ISD::MUL) { // We have another multiply use.
11305       SDNode *OtherOp;
11306       SDNode *MulVar = AddNode.getOperand(0).getNode();
11307 
11308       // OtherOp is what we're multiplying against the constant.
11309       if (Use->getOperand(0) == ConstNode)
11310         OtherOp = Use->getOperand(1).getNode();
11311       else
11312         OtherOp = Use->getOperand(0).getNode();
11313 
11314       // Check to see if multiply is with the same operand of our "add".
11315       //
11316       //     ConstNode  = CONST
11317       //     Use = ConstNode * A  <-- visiting Use. OtherOp is A.
11318       //     ...
11319       //     AddNode  = (A + c1)  <-- MulVar is A.
11320       //         = AddNode * ConstNode   <-- current visiting instruction.
11321       //
11322       // If we make this transformation, we will have a common
11323       // multiply (ConstNode * A) that we can save.
11324       if (OtherOp == MulVar)
11325         return true;
11326 
11327       // Now check to see if a future expansion will give us a common
11328       // multiply.
11329       //
11330       //     ConstNode  = CONST
11331       //     AddNode    = (A + c1)
11332       //     ...   = AddNode * ConstNode <-- current visiting instruction.
11333       //     ...
11334       //     OtherOp = (A + c2)
11335       //     Use     = OtherOp * ConstNode <-- visiting Use.
11336       //
11337       // If we make this transformation, we will have a common
11338       // multiply (CONST * A) after we also do the same transformation
11339       // to the "t2" instruction.
11340       if (OtherOp->getOpcode() == ISD::ADD &&
11341           DAG.isConstantIntBuildVectorOrConstantInt(OtherOp->getOperand(1)) &&
11342           OtherOp->getOperand(0).getNode() == MulVar)
11343         return true;
11344     }
11345   }
11346 
11347   // Didn't find a case where this would be profitable.
11348   return false;
11349 }
11350 
11351 SDValue DAGCombiner::getMergedConstantVectorStore(
11352     SelectionDAG &DAG, const SDLoc &SL, ArrayRef<MemOpLink> Stores,
11353     SmallVectorImpl<SDValue> &Chains, EVT Ty) const {
11354   SmallVector<SDValue, 8> BuildVector;
11355 
11356   for (unsigned I = 0, E = Ty.getVectorNumElements(); I != E; ++I) {
11357     StoreSDNode *St = cast<StoreSDNode>(Stores[I].MemNode);
11358     Chains.push_back(St->getChain());
11359     BuildVector.push_back(St->getValue());
11360   }
11361 
11362   return DAG.getBuildVector(Ty, SL, BuildVector);
11363 }
11364 
11365 bool DAGCombiner::MergeStoresOfConstantsOrVecElts(
11366                   SmallVectorImpl<MemOpLink> &StoreNodes, EVT MemVT,
11367                   unsigned NumStores, bool IsConstantSrc, bool UseVector) {
11368   // Make sure we have something to merge.
11369   if (NumStores < 2)
11370     return false;
11371 
11372   int64_t ElementSizeBytes = MemVT.getSizeInBits() / 8;
11373   LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
11374   unsigned LatestNodeUsed = 0;
11375 
11376   for (unsigned i=0; i < NumStores; ++i) {
11377     // Find a chain for the new wide-store operand. Notice that some
11378     // of the store nodes that we found may not be selected for inclusion
11379     // in the wide store. The chain we use needs to be the chain of the
11380     // latest store node which is *used* and replaced by the wide store.
11381     if (StoreNodes[i].SequenceNum < StoreNodes[LatestNodeUsed].SequenceNum)
11382       LatestNodeUsed = i;
11383   }
11384 
11385   SmallVector<SDValue, 8> Chains;
11386 
11387   // The latest Node in the DAG.
11388   LSBaseSDNode *LatestOp = StoreNodes[LatestNodeUsed].MemNode;
11389   SDLoc DL(StoreNodes[0].MemNode);
11390 
11391   SDValue StoredVal;
11392   if (UseVector) {
11393     bool IsVec = MemVT.isVector();
11394     unsigned Elts = NumStores;
11395     if (IsVec) {
11396       // When merging vector stores, get the total number of elements.
11397       Elts *= MemVT.getVectorNumElements();
11398     }
11399     // Get the type for the merged vector store.
11400     EVT Ty = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
11401     assert(TLI.isTypeLegal(Ty) && "Illegal vector store");
11402 
11403     if (IsConstantSrc) {
11404       StoredVal = getMergedConstantVectorStore(DAG, DL, StoreNodes, Chains, Ty);
11405     } else {
11406       SmallVector<SDValue, 8> Ops;
11407       for (unsigned i = 0; i < NumStores; ++i) {
11408         StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
11409         SDValue Val = St->getValue();
11410         // All operands of BUILD_VECTOR / CONCAT_VECTOR must have the same type.
11411         if (Val.getValueType() != MemVT)
11412           return false;
11413         Ops.push_back(Val);
11414         Chains.push_back(St->getChain());
11415       }
11416 
11417       // Build the extracted vector elements back into a vector.
11418       StoredVal = DAG.getNode(IsVec ? ISD::CONCAT_VECTORS : ISD::BUILD_VECTOR,
11419                               DL, Ty, Ops);    }
11420   } else {
11421     // We should always use a vector store when merging extracted vector
11422     // elements, so this path implies a store of constants.
11423     assert(IsConstantSrc && "Merged vector elements should use vector store");
11424 
11425     unsigned SizeInBits = NumStores * ElementSizeBytes * 8;
11426     APInt StoreInt(SizeInBits, 0);
11427 
11428     // Construct a single integer constant which is made of the smaller
11429     // constant inputs.
11430     bool IsLE = DAG.getDataLayout().isLittleEndian();
11431     for (unsigned i = 0; i < NumStores; ++i) {
11432       unsigned Idx = IsLE ? (NumStores - 1 - i) : i;
11433       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[Idx].MemNode);
11434       Chains.push_back(St->getChain());
11435 
11436       SDValue Val = St->getValue();
11437       StoreInt <<= ElementSizeBytes * 8;
11438       if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Val)) {
11439         StoreInt |= C->getAPIntValue().zext(SizeInBits);
11440       } else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(Val)) {
11441         StoreInt |= C->getValueAPF().bitcastToAPInt().zext(SizeInBits);
11442       } else {
11443         llvm_unreachable("Invalid constant element type");
11444       }
11445     }
11446 
11447     // Create the new Load and Store operations.
11448     EVT StoreTy = EVT::getIntegerVT(*DAG.getContext(), SizeInBits);
11449     StoredVal = DAG.getConstant(StoreInt, DL, StoreTy);
11450   }
11451 
11452   assert(!Chains.empty());
11453 
11454   SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
11455   SDValue NewStore = DAG.getStore(NewChain, DL, StoredVal,
11456                                   FirstInChain->getBasePtr(),
11457                                   FirstInChain->getPointerInfo(),
11458                                   FirstInChain->getAlignment());
11459 
11460   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11461                                                   : DAG.getSubtarget().useAA();
11462   if (UseAA) {
11463     // Replace all merged stores with the new store.
11464     for (unsigned i = 0; i < NumStores; ++i)
11465       CombineTo(StoreNodes[i].MemNode, NewStore);
11466   } else {
11467     // Replace the last store with the new store.
11468     CombineTo(LatestOp, NewStore);
11469     // Erase all other stores.
11470     for (unsigned i = 0; i < NumStores; ++i) {
11471       if (StoreNodes[i].MemNode == LatestOp)
11472         continue;
11473       StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
11474       // ReplaceAllUsesWith will replace all uses that existed when it was
11475       // called, but graph optimizations may cause new ones to appear. For
11476       // example, the case in pr14333 looks like
11477       //
11478       //  St's chain -> St -> another store -> X
11479       //
11480       // And the only difference from St to the other store is the chain.
11481       // When we change it's chain to be St's chain they become identical,
11482       // get CSEed and the net result is that X is now a use of St.
11483       // Since we know that St is redundant, just iterate.
11484       while (!St->use_empty())
11485         DAG.ReplaceAllUsesWith(SDValue(St, 0), St->getChain());
11486       deleteAndRecombine(St);
11487     }
11488   }
11489 
11490   return true;
11491 }
11492 
11493 void DAGCombiner::getStoreMergeAndAliasCandidates(
11494     StoreSDNode* St, SmallVectorImpl<MemOpLink> &StoreNodes,
11495     SmallVectorImpl<LSBaseSDNode*> &AliasLoadNodes) {
11496   // This holds the base pointer, index, and the offset in bytes from the base
11497   // pointer.
11498   BaseIndexOffset BasePtr = BaseIndexOffset::match(St->getBasePtr(), DAG);
11499 
11500   // We must have a base and an offset.
11501   if (!BasePtr.Base.getNode())
11502     return;
11503 
11504   // Do not handle stores to undef base pointers.
11505   if (BasePtr.Base.isUndef())
11506     return;
11507 
11508   // Walk up the chain and look for nodes with offsets from the same
11509   // base pointer. Stop when reaching an instruction with a different kind
11510   // or instruction which has a different base pointer.
11511   EVT MemVT = St->getMemoryVT();
11512   unsigned Seq = 0;
11513   StoreSDNode *Index = St;
11514 
11515 
11516   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11517                                                   : DAG.getSubtarget().useAA();
11518 
11519   if (UseAA) {
11520     // Look at other users of the same chain. Stores on the same chain do not
11521     // alias. If combiner-aa is enabled, non-aliasing stores are canonicalized
11522     // to be on the same chain, so don't bother looking at adjacent chains.
11523 
11524     SDValue Chain = St->getChain();
11525     for (auto I = Chain->use_begin(), E = Chain->use_end(); I != E; ++I) {
11526       if (StoreSDNode *OtherST = dyn_cast<StoreSDNode>(*I)) {
11527         if (I.getOperandNo() != 0)
11528           continue;
11529 
11530         if (OtherST->isVolatile() || OtherST->isIndexed())
11531           continue;
11532 
11533         if (OtherST->getMemoryVT() != MemVT)
11534           continue;
11535 
11536         BaseIndexOffset Ptr = BaseIndexOffset::match(OtherST->getBasePtr(), DAG);
11537 
11538         if (Ptr.equalBaseIndex(BasePtr))
11539           StoreNodes.push_back(MemOpLink(OtherST, Ptr.Offset, Seq++));
11540       }
11541     }
11542 
11543     return;
11544   }
11545 
11546   while (Index) {
11547     // If the chain has more than one use, then we can't reorder the mem ops.
11548     if (Index != St && !SDValue(Index, 0)->hasOneUse())
11549       break;
11550 
11551     // Find the base pointer and offset for this memory node.
11552     BaseIndexOffset Ptr = BaseIndexOffset::match(Index->getBasePtr(), DAG);
11553 
11554     // Check that the base pointer is the same as the original one.
11555     if (!Ptr.equalBaseIndex(BasePtr))
11556       break;
11557 
11558     // The memory operands must not be volatile.
11559     if (Index->isVolatile() || Index->isIndexed())
11560       break;
11561 
11562     // No truncation.
11563     if (Index->isTruncatingStore())
11564       break;
11565 
11566     // The stored memory type must be the same.
11567     if (Index->getMemoryVT() != MemVT)
11568       break;
11569 
11570     // We do not allow under-aligned stores in order to prevent
11571     // overriding stores. NOTE: this is a bad hack. Alignment SHOULD
11572     // be irrelevant here; what MATTERS is that we not move memory
11573     // operations that potentially overlap past each-other.
11574     if (Index->getAlignment() < MemVT.getStoreSize())
11575       break;
11576 
11577     // We found a potential memory operand to merge.
11578     StoreNodes.push_back(MemOpLink(Index, Ptr.Offset, Seq++));
11579 
11580     // Find the next memory operand in the chain. If the next operand in the
11581     // chain is a store then move up and continue the scan with the next
11582     // memory operand. If the next operand is a load save it and use alias
11583     // information to check if it interferes with anything.
11584     SDNode *NextInChain = Index->getChain().getNode();
11585     while (1) {
11586       if (StoreSDNode *STn = dyn_cast<StoreSDNode>(NextInChain)) {
11587         // We found a store node. Use it for the next iteration.
11588         Index = STn;
11589         break;
11590       } else if (LoadSDNode *Ldn = dyn_cast<LoadSDNode>(NextInChain)) {
11591         if (Ldn->isVolatile()) {
11592           Index = nullptr;
11593           break;
11594         }
11595 
11596         // Save the load node for later. Continue the scan.
11597         AliasLoadNodes.push_back(Ldn);
11598         NextInChain = Ldn->getChain().getNode();
11599         continue;
11600       } else {
11601         Index = nullptr;
11602         break;
11603       }
11604     }
11605   }
11606 }
11607 
11608 // We need to check that merging these stores does not cause a loop
11609 // in the DAG. Any store candidate may depend on another candidate
11610 // indirectly through its operand (we already consider dependencies
11611 // through the chain). Check in parallel by searching up from
11612 // non-chain operands of candidates.
11613 bool DAGCombiner::checkMergeStoreCandidatesForDependencies(
11614     SmallVectorImpl<MemOpLink> &StoreNodes) {
11615   SmallPtrSet<const SDNode *, 16> Visited;
11616   SmallVector<const SDNode *, 8> Worklist;
11617   // search ops of store candidates
11618   for (unsigned i = 0; i < StoreNodes.size(); ++i) {
11619     SDNode *n = StoreNodes[i].MemNode;
11620     // Potential loops may happen only through non-chain operands
11621     for (unsigned j = 1; j < n->getNumOperands(); ++j)
11622       Worklist.push_back(n->getOperand(j).getNode());
11623   }
11624   // search through DAG. We can stop early if we find a storenode
11625   for (unsigned i = 0; i < StoreNodes.size(); ++i) {
11626     if (SDNode::hasPredecessorHelper(StoreNodes[i].MemNode, Visited, Worklist))
11627       return false;
11628   }
11629   return true;
11630 }
11631 
11632 bool DAGCombiner::MergeConsecutiveStores(StoreSDNode* St) {
11633   if (OptLevel == CodeGenOpt::None)
11634     return false;
11635 
11636   EVT MemVT = St->getMemoryVT();
11637   int64_t ElementSizeBytes = MemVT.getSizeInBits() / 8;
11638   bool NoVectors = DAG.getMachineFunction().getFunction()->hasFnAttribute(
11639       Attribute::NoImplicitFloat);
11640 
11641   // This function cannot currently deal with non-byte-sized memory sizes.
11642   if (ElementSizeBytes * 8 != MemVT.getSizeInBits())
11643     return false;
11644 
11645   if (!MemVT.isSimple())
11646     return false;
11647 
11648   // Perform an early exit check. Do not bother looking at stored values that
11649   // are not constants, loads, or extracted vector elements.
11650   SDValue StoredVal = St->getValue();
11651   bool IsLoadSrc = isa<LoadSDNode>(StoredVal);
11652   bool IsConstantSrc = isa<ConstantSDNode>(StoredVal) ||
11653                        isa<ConstantFPSDNode>(StoredVal);
11654   bool IsExtractVecSrc = (StoredVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT ||
11655                           StoredVal.getOpcode() == ISD::EXTRACT_SUBVECTOR);
11656 
11657   if (!IsConstantSrc && !IsLoadSrc && !IsExtractVecSrc)
11658     return false;
11659 
11660   // Don't merge vectors into wider vectors if the source data comes from loads.
11661   // TODO: This restriction can be lifted by using logic similar to the
11662   // ExtractVecSrc case.
11663   if (MemVT.isVector() && IsLoadSrc)
11664     return false;
11665 
11666   // Only look at ends of store sequences.
11667   SDValue Chain = SDValue(St, 0);
11668   if (Chain->hasOneUse() && Chain->use_begin()->getOpcode() == ISD::STORE)
11669     return false;
11670 
11671   // Save the LoadSDNodes that we find in the chain.
11672   // We need to make sure that these nodes do not interfere with
11673   // any of the store nodes.
11674   SmallVector<LSBaseSDNode*, 8> AliasLoadNodes;
11675 
11676   // Save the StoreSDNodes that we find in the chain.
11677   SmallVector<MemOpLink, 8> StoreNodes;
11678 
11679   getStoreMergeAndAliasCandidates(St, StoreNodes, AliasLoadNodes);
11680 
11681   // Check if there is anything to merge.
11682   if (StoreNodes.size() < 2)
11683     return false;
11684 
11685   // only do dependence check in AA case
11686   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11687                                                   : DAG.getSubtarget().useAA();
11688   if (UseAA && !checkMergeStoreCandidatesForDependencies(StoreNodes))
11689     return false;
11690 
11691   // Sort the memory operands according to their distance from the
11692   // base pointer.  As a secondary criteria: make sure stores coming
11693   // later in the code come first in the list. This is important for
11694   // the non-UseAA case, because we're merging stores into the FINAL
11695   // store along a chain which potentially contains aliasing stores.
11696   // Thus, if there are multiple stores to the same address, the last
11697   // one can be considered for merging but not the others.
11698   std::sort(StoreNodes.begin(), StoreNodes.end(),
11699             [](MemOpLink LHS, MemOpLink RHS) {
11700     return LHS.OffsetFromBase < RHS.OffsetFromBase ||
11701            (LHS.OffsetFromBase == RHS.OffsetFromBase &&
11702             LHS.SequenceNum < RHS.SequenceNum);
11703   });
11704 
11705   // Scan the memory operations on the chain and find the first non-consecutive
11706   // store memory address.
11707   unsigned LastConsecutiveStore = 0;
11708   int64_t StartAddress = StoreNodes[0].OffsetFromBase;
11709   for (unsigned i = 0, e = StoreNodes.size(); i < e; ++i) {
11710 
11711     // Check that the addresses are consecutive starting from the second
11712     // element in the list of stores.
11713     if (i > 0) {
11714       int64_t CurrAddress = StoreNodes[i].OffsetFromBase;
11715       if (CurrAddress - StartAddress != (ElementSizeBytes * i))
11716         break;
11717     }
11718 
11719     // Check if this store interferes with any of the loads that we found.
11720     // If we find a load that alias with this store. Stop the sequence.
11721     if (any_of(AliasLoadNodes, [&](LSBaseSDNode *Ldn) {
11722           return isAlias(Ldn, StoreNodes[i].MemNode);
11723         }))
11724       break;
11725 
11726     // Mark this node as useful.
11727     LastConsecutiveStore = i;
11728   }
11729 
11730   // The node with the lowest store address.
11731   LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
11732   unsigned FirstStoreAS = FirstInChain->getAddressSpace();
11733   unsigned FirstStoreAlign = FirstInChain->getAlignment();
11734   LLVMContext &Context = *DAG.getContext();
11735   const DataLayout &DL = DAG.getDataLayout();
11736 
11737   // Store the constants into memory as one consecutive store.
11738   if (IsConstantSrc) {
11739     unsigned LastLegalType = 0;
11740     unsigned LastLegalVectorType = 0;
11741     bool NonZero = false;
11742     for (unsigned i=0; i<LastConsecutiveStore+1; ++i) {
11743       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
11744       SDValue StoredVal = St->getValue();
11745 
11746       if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(StoredVal)) {
11747         NonZero |= !C->isNullValue();
11748       } else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(StoredVal)) {
11749         NonZero |= !C->getConstantFPValue()->isNullValue();
11750       } else {
11751         // Non-constant.
11752         break;
11753       }
11754 
11755       // Find a legal type for the constant store.
11756       unsigned SizeInBits = (i+1) * ElementSizeBytes * 8;
11757       EVT StoreTy = EVT::getIntegerVT(Context, SizeInBits);
11758       bool IsFast;
11759       if (TLI.isTypeLegal(StoreTy) &&
11760           TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
11761                                  FirstStoreAlign, &IsFast) && IsFast) {
11762         LastLegalType = i+1;
11763       // Or check whether a truncstore is legal.
11764       } else if (TLI.getTypeAction(Context, StoreTy) ==
11765                  TargetLowering::TypePromoteInteger) {
11766         EVT LegalizedStoredValueTy =
11767           TLI.getTypeToTransformTo(Context, StoredVal.getValueType());
11768         if (TLI.isTruncStoreLegal(LegalizedStoredValueTy, StoreTy) &&
11769             TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
11770                                    FirstStoreAS, FirstStoreAlign, &IsFast) &&
11771             IsFast) {
11772           LastLegalType = i + 1;
11773         }
11774       }
11775 
11776       // We only use vectors if the constant is known to be zero or the target
11777       // allows it and the function is not marked with the noimplicitfloat
11778       // attribute.
11779       if ((!NonZero || TLI.storeOfVectorConstantIsCheap(MemVT, i+1,
11780                                                         FirstStoreAS)) &&
11781           !NoVectors) {
11782         // Find a legal type for the vector store.
11783         EVT Ty = EVT::getVectorVT(Context, MemVT, i+1);
11784         if (TLI.isTypeLegal(Ty) &&
11785             TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS,
11786                                    FirstStoreAlign, &IsFast) && IsFast)
11787           LastLegalVectorType = i + 1;
11788       }
11789     }
11790 
11791     // Check if we found a legal integer type to store.
11792     if (LastLegalType == 0 && LastLegalVectorType == 0)
11793       return false;
11794 
11795     bool UseVector = (LastLegalVectorType > LastLegalType) && !NoVectors;
11796     unsigned NumElem = UseVector ? LastLegalVectorType : LastLegalType;
11797 
11798     return MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumElem,
11799                                            true, UseVector);
11800   }
11801 
11802   // When extracting multiple vector elements, try to store them
11803   // in one vector store rather than a sequence of scalar stores.
11804   if (IsExtractVecSrc) {
11805     unsigned NumStoresToMerge = 0;
11806     bool IsVec = MemVT.isVector();
11807     for (unsigned i = 0; i < LastConsecutiveStore + 1; ++i) {
11808       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
11809       unsigned StoreValOpcode = St->getValue().getOpcode();
11810       // This restriction could be loosened.
11811       // Bail out if any stored values are not elements extracted from a vector.
11812       // It should be possible to handle mixed sources, but load sources need
11813       // more careful handling (see the block of code below that handles
11814       // consecutive loads).
11815       if (StoreValOpcode != ISD::EXTRACT_VECTOR_ELT &&
11816           StoreValOpcode != ISD::EXTRACT_SUBVECTOR)
11817         return false;
11818 
11819       // Find a legal type for the vector store.
11820       unsigned Elts = i + 1;
11821       if (IsVec) {
11822         // When merging vector stores, get the total number of elements.
11823         Elts *= MemVT.getVectorNumElements();
11824       }
11825       EVT Ty = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
11826       bool IsFast;
11827       if (TLI.isTypeLegal(Ty) &&
11828           TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS,
11829                                  FirstStoreAlign, &IsFast) && IsFast)
11830         NumStoresToMerge = i + 1;
11831     }
11832 
11833     return MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumStoresToMerge,
11834                                            false, true);
11835   }
11836 
11837   // Below we handle the case of multiple consecutive stores that
11838   // come from multiple consecutive loads. We merge them into a single
11839   // wide load and a single wide store.
11840 
11841   // Look for load nodes which are used by the stored values.
11842   SmallVector<MemOpLink, 8> LoadNodes;
11843 
11844   // Find acceptable loads. Loads need to have the same chain (token factor),
11845   // must not be zext, volatile, indexed, and they must be consecutive.
11846   BaseIndexOffset LdBasePtr;
11847   for (unsigned i=0; i<LastConsecutiveStore+1; ++i) {
11848     StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
11849     LoadSDNode *Ld = dyn_cast<LoadSDNode>(St->getValue());
11850     if (!Ld) break;
11851 
11852     // Loads must only have one use.
11853     if (!Ld->hasNUsesOfValue(1, 0))
11854       break;
11855 
11856     // The memory operands must not be volatile.
11857     if (Ld->isVolatile() || Ld->isIndexed())
11858       break;
11859 
11860     // We do not accept ext loads.
11861     if (Ld->getExtensionType() != ISD::NON_EXTLOAD)
11862       break;
11863 
11864     // The stored memory type must be the same.
11865     if (Ld->getMemoryVT() != MemVT)
11866       break;
11867 
11868     BaseIndexOffset LdPtr = BaseIndexOffset::match(Ld->getBasePtr(), DAG);
11869     // If this is not the first ptr that we check.
11870     if (LdBasePtr.Base.getNode()) {
11871       // The base ptr must be the same.
11872       if (!LdPtr.equalBaseIndex(LdBasePtr))
11873         break;
11874     } else {
11875       // Check that all other base pointers are the same as this one.
11876       LdBasePtr = LdPtr;
11877     }
11878 
11879     // We found a potential memory operand to merge.
11880     LoadNodes.push_back(MemOpLink(Ld, LdPtr.Offset, 0));
11881   }
11882 
11883   if (LoadNodes.size() < 2)
11884     return false;
11885 
11886   // If we have load/store pair instructions and we only have two values,
11887   // don't bother.
11888   unsigned RequiredAlignment;
11889   if (LoadNodes.size() == 2 && TLI.hasPairedLoad(MemVT, RequiredAlignment) &&
11890       St->getAlignment() >= RequiredAlignment)
11891     return false;
11892 
11893   LoadSDNode *FirstLoad = cast<LoadSDNode>(LoadNodes[0].MemNode);
11894   unsigned FirstLoadAS = FirstLoad->getAddressSpace();
11895   unsigned FirstLoadAlign = FirstLoad->getAlignment();
11896 
11897   // Scan the memory operations on the chain and find the first non-consecutive
11898   // load memory address. These variables hold the index in the store node
11899   // array.
11900   unsigned LastConsecutiveLoad = 0;
11901   // This variable refers to the size and not index in the array.
11902   unsigned LastLegalVectorType = 0;
11903   unsigned LastLegalIntegerType = 0;
11904   StartAddress = LoadNodes[0].OffsetFromBase;
11905   SDValue FirstChain = FirstLoad->getChain();
11906   for (unsigned i = 1; i < LoadNodes.size(); ++i) {
11907     // All loads must share the same chain.
11908     if (LoadNodes[i].MemNode->getChain() != FirstChain)
11909       break;
11910 
11911     int64_t CurrAddress = LoadNodes[i].OffsetFromBase;
11912     if (CurrAddress - StartAddress != (ElementSizeBytes * i))
11913       break;
11914     LastConsecutiveLoad = i;
11915     // Find a legal type for the vector store.
11916     EVT StoreTy = EVT::getVectorVT(Context, MemVT, i+1);
11917     bool IsFastSt, IsFastLd;
11918     if (TLI.isTypeLegal(StoreTy) &&
11919         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
11920                                FirstStoreAlign, &IsFastSt) && IsFastSt &&
11921         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS,
11922                                FirstLoadAlign, &IsFastLd) && IsFastLd) {
11923       LastLegalVectorType = i + 1;
11924     }
11925 
11926     // Find a legal type for the integer store.
11927     unsigned SizeInBits = (i+1) * ElementSizeBytes * 8;
11928     StoreTy = EVT::getIntegerVT(Context, SizeInBits);
11929     if (TLI.isTypeLegal(StoreTy) &&
11930         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
11931                                FirstStoreAlign, &IsFastSt) && IsFastSt &&
11932         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS,
11933                                FirstLoadAlign, &IsFastLd) && IsFastLd)
11934       LastLegalIntegerType = i + 1;
11935     // Or check whether a truncstore and extload is legal.
11936     else if (TLI.getTypeAction(Context, StoreTy) ==
11937              TargetLowering::TypePromoteInteger) {
11938       EVT LegalizedStoredValueTy =
11939         TLI.getTypeToTransformTo(Context, StoreTy);
11940       if (TLI.isTruncStoreLegal(LegalizedStoredValueTy, StoreTy) &&
11941           TLI.isLoadExtLegal(ISD::ZEXTLOAD, LegalizedStoredValueTy, StoreTy) &&
11942           TLI.isLoadExtLegal(ISD::SEXTLOAD, LegalizedStoredValueTy, StoreTy) &&
11943           TLI.isLoadExtLegal(ISD::EXTLOAD, LegalizedStoredValueTy, StoreTy) &&
11944           TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
11945                                  FirstStoreAS, FirstStoreAlign, &IsFastSt) &&
11946           IsFastSt &&
11947           TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
11948                                  FirstLoadAS, FirstLoadAlign, &IsFastLd) &&
11949           IsFastLd)
11950         LastLegalIntegerType = i+1;
11951     }
11952   }
11953 
11954   // Only use vector types if the vector type is larger than the integer type.
11955   // If they are the same, use integers.
11956   bool UseVectorTy = LastLegalVectorType > LastLegalIntegerType && !NoVectors;
11957   unsigned LastLegalType = std::max(LastLegalVectorType, LastLegalIntegerType);
11958 
11959   // We add +1 here because the LastXXX variables refer to location while
11960   // the NumElem refers to array/index size.
11961   unsigned NumElem = std::min(LastConsecutiveStore, LastConsecutiveLoad) + 1;
11962   NumElem = std::min(LastLegalType, NumElem);
11963 
11964   if (NumElem < 2)
11965     return false;
11966 
11967   // Collect the chains from all merged stores.
11968   SmallVector<SDValue, 8> MergeStoreChains;
11969   MergeStoreChains.push_back(StoreNodes[0].MemNode->getChain());
11970 
11971   // The latest Node in the DAG.
11972   unsigned LatestNodeUsed = 0;
11973   for (unsigned i=1; i<NumElem; ++i) {
11974     // Find a chain for the new wide-store operand. Notice that some
11975     // of the store nodes that we found may not be selected for inclusion
11976     // in the wide store. The chain we use needs to be the chain of the
11977     // latest store node which is *used* and replaced by the wide store.
11978     if (StoreNodes[i].SequenceNum < StoreNodes[LatestNodeUsed].SequenceNum)
11979       LatestNodeUsed = i;
11980 
11981     MergeStoreChains.push_back(StoreNodes[i].MemNode->getChain());
11982   }
11983 
11984   LSBaseSDNode *LatestOp = StoreNodes[LatestNodeUsed].MemNode;
11985 
11986   // Find if it is better to use vectors or integers to load and store
11987   // to memory.
11988   EVT JointMemOpVT;
11989   if (UseVectorTy) {
11990     JointMemOpVT = EVT::getVectorVT(Context, MemVT, NumElem);
11991   } else {
11992     unsigned SizeInBits = NumElem * ElementSizeBytes * 8;
11993     JointMemOpVT = EVT::getIntegerVT(Context, SizeInBits);
11994   }
11995 
11996   SDLoc LoadDL(LoadNodes[0].MemNode);
11997   SDLoc StoreDL(StoreNodes[0].MemNode);
11998 
11999   // The merged loads are required to have the same incoming chain, so
12000   // using the first's chain is acceptable.
12001   SDValue NewLoad = DAG.getLoad(JointMemOpVT, LoadDL, FirstLoad->getChain(),
12002                                 FirstLoad->getBasePtr(),
12003                                 FirstLoad->getPointerInfo(), FirstLoadAlign);
12004 
12005   SDValue NewStoreChain =
12006     DAG.getNode(ISD::TokenFactor, StoreDL, MVT::Other, MergeStoreChains);
12007 
12008   SDValue NewStore =
12009       DAG.getStore(NewStoreChain, StoreDL, NewLoad, FirstInChain->getBasePtr(),
12010                    FirstInChain->getPointerInfo(), FirstStoreAlign);
12011 
12012   // Transfer chain users from old loads to the new load.
12013   for (unsigned i = 0; i < NumElem; ++i) {
12014     LoadSDNode *Ld = cast<LoadSDNode>(LoadNodes[i].MemNode);
12015     DAG.ReplaceAllUsesOfValueWith(SDValue(Ld, 1),
12016                                   SDValue(NewLoad.getNode(), 1));
12017   }
12018 
12019   if (UseAA) {
12020     // Replace the all stores with the new store.
12021     for (unsigned i = 0; i < NumElem; ++i)
12022       CombineTo(StoreNodes[i].MemNode, NewStore);
12023   } else {
12024     // Replace the last store with the new store.
12025     CombineTo(LatestOp, NewStore);
12026     // Erase all other stores.
12027     for (unsigned i = 0; i < NumElem; ++i) {
12028       // Remove all Store nodes.
12029       if (StoreNodes[i].MemNode == LatestOp)
12030         continue;
12031       StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
12032       DAG.ReplaceAllUsesOfValueWith(SDValue(St, 0), St->getChain());
12033       deleteAndRecombine(St);
12034     }
12035   }
12036 
12037   return true;
12038 }
12039 
12040 SDValue DAGCombiner::replaceStoreChain(StoreSDNode *ST, SDValue BetterChain) {
12041   SDLoc SL(ST);
12042   SDValue ReplStore;
12043 
12044   // Replace the chain to avoid dependency.
12045   if (ST->isTruncatingStore()) {
12046     ReplStore = DAG.getTruncStore(BetterChain, SL, ST->getValue(),
12047                                   ST->getBasePtr(), ST->getMemoryVT(),
12048                                   ST->getMemOperand());
12049   } else {
12050     ReplStore = DAG.getStore(BetterChain, SL, ST->getValue(), ST->getBasePtr(),
12051                              ST->getMemOperand());
12052   }
12053 
12054   // Create token to keep both nodes around.
12055   SDValue Token = DAG.getNode(ISD::TokenFactor, SL,
12056                               MVT::Other, ST->getChain(), ReplStore);
12057 
12058   // Make sure the new and old chains are cleaned up.
12059   AddToWorklist(Token.getNode());
12060 
12061   // Don't add users to work list.
12062   return CombineTo(ST, Token, false);
12063 }
12064 
12065 SDValue DAGCombiner::replaceStoreOfFPConstant(StoreSDNode *ST) {
12066   SDValue Value = ST->getValue();
12067   if (Value.getOpcode() == ISD::TargetConstantFP)
12068     return SDValue();
12069 
12070   SDLoc DL(ST);
12071 
12072   SDValue Chain = ST->getChain();
12073   SDValue Ptr = ST->getBasePtr();
12074 
12075   const ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Value);
12076 
12077   // NOTE: If the original store is volatile, this transform must not increase
12078   // the number of stores.  For example, on x86-32 an f64 can be stored in one
12079   // processor operation but an i64 (which is not legal) requires two.  So the
12080   // transform should not be done in this case.
12081 
12082   SDValue Tmp;
12083   switch (CFP->getSimpleValueType(0).SimpleTy) {
12084   default:
12085     llvm_unreachable("Unknown FP type");
12086   case MVT::f16:    // We don't do this for these yet.
12087   case MVT::f80:
12088   case MVT::f128:
12089   case MVT::ppcf128:
12090     return SDValue();
12091   case MVT::f32:
12092     if ((isTypeLegal(MVT::i32) && !LegalOperations && !ST->isVolatile()) ||
12093         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
12094       ;
12095       Tmp = DAG.getConstant((uint32_t)CFP->getValueAPF().
12096                             bitcastToAPInt().getZExtValue(), SDLoc(CFP),
12097                             MVT::i32);
12098       return DAG.getStore(Chain, DL, Tmp, Ptr, ST->getMemOperand());
12099     }
12100 
12101     return SDValue();
12102   case MVT::f64:
12103     if ((TLI.isTypeLegal(MVT::i64) && !LegalOperations &&
12104          !ST->isVolatile()) ||
12105         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i64)) {
12106       ;
12107       Tmp = DAG.getConstant(CFP->getValueAPF().bitcastToAPInt().
12108                             getZExtValue(), SDLoc(CFP), MVT::i64);
12109       return DAG.getStore(Chain, DL, Tmp,
12110                           Ptr, ST->getMemOperand());
12111     }
12112 
12113     if (!ST->isVolatile() &&
12114         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
12115       // Many FP stores are not made apparent until after legalize, e.g. for
12116       // argument passing.  Since this is so common, custom legalize the
12117       // 64-bit integer store into two 32-bit stores.
12118       uint64_t Val = CFP->getValueAPF().bitcastToAPInt().getZExtValue();
12119       SDValue Lo = DAG.getConstant(Val & 0xFFFFFFFF, SDLoc(CFP), MVT::i32);
12120       SDValue Hi = DAG.getConstant(Val >> 32, SDLoc(CFP), MVT::i32);
12121       if (DAG.getDataLayout().isBigEndian())
12122         std::swap(Lo, Hi);
12123 
12124       unsigned Alignment = ST->getAlignment();
12125       MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
12126       AAMDNodes AAInfo = ST->getAAInfo();
12127 
12128       SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
12129                                  ST->getAlignment(), MMOFlags, AAInfo);
12130       Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
12131                         DAG.getConstant(4, DL, Ptr.getValueType()));
12132       Alignment = MinAlign(Alignment, 4U);
12133       SDValue St1 = DAG.getStore(Chain, DL, Hi, Ptr,
12134                                  ST->getPointerInfo().getWithOffset(4),
12135                                  Alignment, MMOFlags, AAInfo);
12136       return DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
12137                          St0, St1);
12138     }
12139 
12140     return SDValue();
12141   }
12142 }
12143 
12144 SDValue DAGCombiner::visitSTORE(SDNode *N) {
12145   StoreSDNode *ST  = cast<StoreSDNode>(N);
12146   SDValue Chain = ST->getChain();
12147   SDValue Value = ST->getValue();
12148   SDValue Ptr   = ST->getBasePtr();
12149 
12150   // If this is a store of a bit convert, store the input value if the
12151   // resultant store does not need a higher alignment than the original.
12152   if (Value.getOpcode() == ISD::BITCAST && !ST->isTruncatingStore() &&
12153       ST->isUnindexed()) {
12154     EVT SVT = Value.getOperand(0).getValueType();
12155     if (((!LegalOperations && !ST->isVolatile()) ||
12156          TLI.isOperationLegalOrCustom(ISD::STORE, SVT)) &&
12157         TLI.isStoreBitCastBeneficial(Value.getValueType(), SVT)) {
12158       unsigned OrigAlign = ST->getAlignment();
12159       bool Fast = false;
12160       if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), SVT,
12161                                  ST->getAddressSpace(), OrigAlign, &Fast) &&
12162           Fast) {
12163         return DAG.getStore(Chain, SDLoc(N), Value.getOperand(0), Ptr,
12164                             ST->getPointerInfo(), OrigAlign,
12165                             ST->getMemOperand()->getFlags(), ST->getAAInfo());
12166       }
12167     }
12168   }
12169 
12170   // Turn 'store undef, Ptr' -> nothing.
12171   if (Value.isUndef() && ST->isUnindexed())
12172     return Chain;
12173 
12174   // Try to infer better alignment information than the store already has.
12175   if (OptLevel != CodeGenOpt::None && ST->isUnindexed()) {
12176     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
12177       if (Align > ST->getAlignment()) {
12178         SDValue NewStore =
12179             DAG.getTruncStore(Chain, SDLoc(N), Value, Ptr, ST->getPointerInfo(),
12180                               ST->getMemoryVT(), Align,
12181                               ST->getMemOperand()->getFlags(), ST->getAAInfo());
12182         if (NewStore.getNode() != N)
12183           return CombineTo(ST, NewStore, true);
12184       }
12185     }
12186   }
12187 
12188   // Try transforming a pair floating point load / store ops to integer
12189   // load / store ops.
12190   if (SDValue NewST = TransformFPLoadStorePair(N))
12191     return NewST;
12192 
12193   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
12194                                                   : DAG.getSubtarget().useAA();
12195 #ifndef NDEBUG
12196   if (CombinerAAOnlyFunc.getNumOccurrences() &&
12197       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
12198     UseAA = false;
12199 #endif
12200   if (UseAA && ST->isUnindexed()) {
12201     // FIXME: We should do this even without AA enabled. AA will just allow
12202     // FindBetterChain to work in more situations. The problem with this is that
12203     // any combine that expects memory operations to be on consecutive chains
12204     // first needs to be updated to look for users of the same chain.
12205 
12206     // Walk up chain skipping non-aliasing memory nodes, on this store and any
12207     // adjacent stores.
12208     if (findBetterNeighborChains(ST)) {
12209       // replaceStoreChain uses CombineTo, which handled all of the worklist
12210       // manipulation. Return the original node to not do anything else.
12211       return SDValue(ST, 0);
12212     }
12213     Chain = ST->getChain();
12214   }
12215 
12216   // Try transforming N to an indexed store.
12217   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
12218     return SDValue(N, 0);
12219 
12220   // FIXME: is there such a thing as a truncating indexed store?
12221   if (ST->isTruncatingStore() && ST->isUnindexed() &&
12222       Value.getValueType().isInteger()) {
12223     // See if we can simplify the input to this truncstore with knowledge that
12224     // only the low bits are being used.  For example:
12225     // "truncstore (or (shl x, 8), y), i8"  -> "truncstore y, i8"
12226     SDValue Shorter = GetDemandedBits(
12227         Value, APInt::getLowBitsSet(Value.getScalarValueSizeInBits(),
12228                                     ST->getMemoryVT().getScalarSizeInBits()));
12229     AddToWorklist(Value.getNode());
12230     if (Shorter.getNode())
12231       return DAG.getTruncStore(Chain, SDLoc(N), Shorter,
12232                                Ptr, ST->getMemoryVT(), ST->getMemOperand());
12233 
12234     // Otherwise, see if we can simplify the operation with
12235     // SimplifyDemandedBits, which only works if the value has a single use.
12236     if (SimplifyDemandedBits(
12237             Value,
12238             APInt::getLowBitsSet(Value.getScalarValueSizeInBits(),
12239                                  ST->getMemoryVT().getScalarSizeInBits())))
12240       return SDValue(N, 0);
12241   }
12242 
12243   // If this is a load followed by a store to the same location, then the store
12244   // is dead/noop.
12245   if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Value)) {
12246     if (Ld->getBasePtr() == Ptr && ST->getMemoryVT() == Ld->getMemoryVT() &&
12247         ST->isUnindexed() && !ST->isVolatile() &&
12248         // There can't be any side effects between the load and store, such as
12249         // a call or store.
12250         Chain.reachesChainWithoutSideEffects(SDValue(Ld, 1))) {
12251       // The store is dead, remove it.
12252       return Chain;
12253     }
12254   }
12255 
12256   // If this is a store followed by a store with the same value to the same
12257   // location, then the store is dead/noop.
12258   if (StoreSDNode *ST1 = dyn_cast<StoreSDNode>(Chain)) {
12259     if (ST1->getBasePtr() == Ptr && ST->getMemoryVT() == ST1->getMemoryVT() &&
12260         ST1->getValue() == Value && ST->isUnindexed() && !ST->isVolatile() &&
12261         ST1->isUnindexed() && !ST1->isVolatile()) {
12262       // The store is dead, remove it.
12263       return Chain;
12264     }
12265   }
12266 
12267   // If this is an FP_ROUND or TRUNC followed by a store, fold this into a
12268   // truncating store.  We can do this even if this is already a truncstore.
12269   if ((Value.getOpcode() == ISD::FP_ROUND || Value.getOpcode() == ISD::TRUNCATE)
12270       && Value.getNode()->hasOneUse() && ST->isUnindexed() &&
12271       TLI.isTruncStoreLegal(Value.getOperand(0).getValueType(),
12272                             ST->getMemoryVT())) {
12273     return DAG.getTruncStore(Chain, SDLoc(N), Value.getOperand(0),
12274                              Ptr, ST->getMemoryVT(), ST->getMemOperand());
12275   }
12276 
12277   // Only perform this optimization before the types are legal, because we
12278   // don't want to perform this optimization on every DAGCombine invocation.
12279   if (!LegalTypes) {
12280     bool EverChanged = false;
12281 
12282     do {
12283       // There can be multiple store sequences on the same chain.
12284       // Keep trying to merge store sequences until we are unable to do so
12285       // or until we merge the last store on the chain.
12286       bool Changed = MergeConsecutiveStores(ST);
12287       EverChanged |= Changed;
12288       if (!Changed) break;
12289     } while (ST->getOpcode() != ISD::DELETED_NODE);
12290 
12291     if (EverChanged)
12292       return SDValue(N, 0);
12293   }
12294 
12295   // Turn 'store float 1.0, Ptr' -> 'store int 0x12345678, Ptr'
12296   //
12297   // Make sure to do this only after attempting to merge stores in order to
12298   //  avoid changing the types of some subset of stores due to visit order,
12299   //  preventing their merging.
12300   if (isa<ConstantFPSDNode>(Value)) {
12301     if (SDValue NewSt = replaceStoreOfFPConstant(ST))
12302       return NewSt;
12303   }
12304 
12305   if (SDValue NewSt = splitMergedValStore(ST))
12306     return NewSt;
12307 
12308   return ReduceLoadOpStoreWidth(N);
12309 }
12310 
12311 /// For the instruction sequence of store below, F and I values
12312 /// are bundled together as an i64 value before being stored into memory.
12313 /// Sometimes it is more efficent to generate separate stores for F and I,
12314 /// which can remove the bitwise instructions or sink them to colder places.
12315 ///
12316 ///   (store (or (zext (bitcast F to i32) to i64),
12317 ///              (shl (zext I to i64), 32)), addr)  -->
12318 ///   (store F, addr) and (store I, addr+4)
12319 ///
12320 /// Similarly, splitting for other merged store can also be beneficial, like:
12321 /// For pair of {i32, i32}, i64 store --> two i32 stores.
12322 /// For pair of {i32, i16}, i64 store --> two i32 stores.
12323 /// For pair of {i16, i16}, i32 store --> two i16 stores.
12324 /// For pair of {i16, i8},  i32 store --> two i16 stores.
12325 /// For pair of {i8, i8},   i16 store --> two i8 stores.
12326 ///
12327 /// We allow each target to determine specifically which kind of splitting is
12328 /// supported.
12329 ///
12330 /// The store patterns are commonly seen from the simple code snippet below
12331 /// if only std::make_pair(...) is sroa transformed before inlined into hoo.
12332 ///   void goo(const std::pair<int, float> &);
12333 ///   hoo() {
12334 ///     ...
12335 ///     goo(std::make_pair(tmp, ftmp));
12336 ///     ...
12337 ///   }
12338 ///
12339 SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
12340   if (OptLevel == CodeGenOpt::None)
12341     return SDValue();
12342 
12343   SDValue Val = ST->getValue();
12344   SDLoc DL(ST);
12345 
12346   // Match OR operand.
12347   if (!Val.getValueType().isScalarInteger() || Val.getOpcode() != ISD::OR)
12348     return SDValue();
12349 
12350   // Match SHL operand and get Lower and Higher parts of Val.
12351   SDValue Op1 = Val.getOperand(0);
12352   SDValue Op2 = Val.getOperand(1);
12353   SDValue Lo, Hi;
12354   if (Op1.getOpcode() != ISD::SHL) {
12355     std::swap(Op1, Op2);
12356     if (Op1.getOpcode() != ISD::SHL)
12357       return SDValue();
12358   }
12359   Lo = Op2;
12360   Hi = Op1.getOperand(0);
12361   if (!Op1.hasOneUse())
12362     return SDValue();
12363 
12364   // Match shift amount to HalfValBitSize.
12365   unsigned HalfValBitSize = Val.getValueSizeInBits() / 2;
12366   ConstantSDNode *ShAmt = dyn_cast<ConstantSDNode>(Op1.getOperand(1));
12367   if (!ShAmt || ShAmt->getAPIntValue() != HalfValBitSize)
12368     return SDValue();
12369 
12370   // Lo and Hi are zero-extended from int with size less equal than 32
12371   // to i64.
12372   if (Lo.getOpcode() != ISD::ZERO_EXTEND || !Lo.hasOneUse() ||
12373       !Lo.getOperand(0).getValueType().isScalarInteger() ||
12374       Lo.getOperand(0).getValueSizeInBits() > HalfValBitSize ||
12375       Hi.getOpcode() != ISD::ZERO_EXTEND || !Hi.hasOneUse() ||
12376       !Hi.getOperand(0).getValueType().isScalarInteger() ||
12377       Hi.getOperand(0).getValueSizeInBits() > HalfValBitSize)
12378     return SDValue();
12379 
12380   if (!TLI.isMultiStoresCheaperThanBitsMerge(Lo.getOperand(0),
12381                                              Hi.getOperand(0)))
12382     return SDValue();
12383 
12384   // Start to split store.
12385   unsigned Alignment = ST->getAlignment();
12386   MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
12387   AAMDNodes AAInfo = ST->getAAInfo();
12388 
12389   // Change the sizes of Lo and Hi's value types to HalfValBitSize.
12390   EVT VT = EVT::getIntegerVT(*DAG.getContext(), HalfValBitSize);
12391   Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Lo.getOperand(0));
12392   Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Hi.getOperand(0));
12393 
12394   SDValue Chain = ST->getChain();
12395   SDValue Ptr = ST->getBasePtr();
12396   // Lower value store.
12397   SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
12398                              ST->getAlignment(), MMOFlags, AAInfo);
12399   Ptr =
12400       DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
12401                   DAG.getConstant(HalfValBitSize / 8, DL, Ptr.getValueType()));
12402   // Higher value store.
12403   SDValue St1 =
12404       DAG.getStore(St0, DL, Hi, Ptr,
12405                    ST->getPointerInfo().getWithOffset(HalfValBitSize / 8),
12406                    Alignment / 2, MMOFlags, AAInfo);
12407   return St1;
12408 }
12409 
12410 SDValue DAGCombiner::visitINSERT_VECTOR_ELT(SDNode *N) {
12411   SDValue InVec = N->getOperand(0);
12412   SDValue InVal = N->getOperand(1);
12413   SDValue EltNo = N->getOperand(2);
12414   SDLoc DL(N);
12415 
12416   // If the inserted element is an UNDEF, just use the input vector.
12417   if (InVal.isUndef())
12418     return InVec;
12419 
12420   EVT VT = InVec.getValueType();
12421 
12422   // If we can't generate a legal BUILD_VECTOR, exit
12423   if (LegalOperations && !TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
12424     return SDValue();
12425 
12426   // Check that we know which element is being inserted
12427   if (!isa<ConstantSDNode>(EltNo))
12428     return SDValue();
12429   unsigned Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
12430 
12431   // Canonicalize insert_vector_elt dag nodes.
12432   // Example:
12433   // (insert_vector_elt (insert_vector_elt A, Idx0), Idx1)
12434   // -> (insert_vector_elt (insert_vector_elt A, Idx1), Idx0)
12435   //
12436   // Do this only if the child insert_vector node has one use; also
12437   // do this only if indices are both constants and Idx1 < Idx0.
12438   if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT && InVec.hasOneUse()
12439       && isa<ConstantSDNode>(InVec.getOperand(2))) {
12440     unsigned OtherElt =
12441       cast<ConstantSDNode>(InVec.getOperand(2))->getZExtValue();
12442     if (Elt < OtherElt) {
12443       // Swap nodes.
12444       SDValue NewOp = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT,
12445                                   InVec.getOperand(0), InVal, EltNo);
12446       AddToWorklist(NewOp.getNode());
12447       return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(InVec.getNode()),
12448                          VT, NewOp, InVec.getOperand(1), InVec.getOperand(2));
12449     }
12450   }
12451 
12452   // Check that the operand is a BUILD_VECTOR (or UNDEF, which can essentially
12453   // be converted to a BUILD_VECTOR).  Fill in the Ops vector with the
12454   // vector elements.
12455   SmallVector<SDValue, 8> Ops;
12456   // Do not combine these two vectors if the output vector will not replace
12457   // the input vector.
12458   if (InVec.getOpcode() == ISD::BUILD_VECTOR && InVec.hasOneUse()) {
12459     Ops.append(InVec.getNode()->op_begin(),
12460                InVec.getNode()->op_end());
12461   } else if (InVec.isUndef()) {
12462     unsigned NElts = VT.getVectorNumElements();
12463     Ops.append(NElts, DAG.getUNDEF(InVal.getValueType()));
12464   } else {
12465     return SDValue();
12466   }
12467 
12468   // Insert the element
12469   if (Elt < Ops.size()) {
12470     // All the operands of BUILD_VECTOR must have the same type;
12471     // we enforce that here.
12472     EVT OpVT = Ops[0].getValueType();
12473     if (InVal.getValueType() != OpVT)
12474       InVal = OpVT.bitsGT(InVal.getValueType()) ?
12475                 DAG.getNode(ISD::ANY_EXTEND, DL, OpVT, InVal) :
12476                 DAG.getNode(ISD::TRUNCATE, DL, OpVT, InVal);
12477     Ops[Elt] = InVal;
12478   }
12479 
12480   // Return the new vector
12481   return DAG.getBuildVector(VT, DL, Ops);
12482 }
12483 
12484 SDValue DAGCombiner::ReplaceExtractVectorEltOfLoadWithNarrowedLoad(
12485     SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad) {
12486   assert(!OriginalLoad->isVolatile());
12487 
12488   EVT ResultVT = EVE->getValueType(0);
12489   EVT VecEltVT = InVecVT.getVectorElementType();
12490   unsigned Align = OriginalLoad->getAlignment();
12491   unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
12492       VecEltVT.getTypeForEVT(*DAG.getContext()));
12493 
12494   if (NewAlign > Align || !TLI.isOperationLegalOrCustom(ISD::LOAD, VecEltVT))
12495     return SDValue();
12496 
12497   Align = NewAlign;
12498 
12499   SDValue NewPtr = OriginalLoad->getBasePtr();
12500   SDValue Offset;
12501   EVT PtrType = NewPtr.getValueType();
12502   MachinePointerInfo MPI;
12503   SDLoc DL(EVE);
12504   if (auto *ConstEltNo = dyn_cast<ConstantSDNode>(EltNo)) {
12505     int Elt = ConstEltNo->getZExtValue();
12506     unsigned PtrOff = VecEltVT.getSizeInBits() * Elt / 8;
12507     Offset = DAG.getConstant(PtrOff, DL, PtrType);
12508     MPI = OriginalLoad->getPointerInfo().getWithOffset(PtrOff);
12509   } else {
12510     Offset = DAG.getZExtOrTrunc(EltNo, DL, PtrType);
12511     Offset = DAG.getNode(
12512         ISD::MUL, DL, PtrType, Offset,
12513         DAG.getConstant(VecEltVT.getStoreSize(), DL, PtrType));
12514     MPI = OriginalLoad->getPointerInfo();
12515   }
12516   NewPtr = DAG.getNode(ISD::ADD, DL, PtrType, NewPtr, Offset);
12517 
12518   // The replacement we need to do here is a little tricky: we need to
12519   // replace an extractelement of a load with a load.
12520   // Use ReplaceAllUsesOfValuesWith to do the replacement.
12521   // Note that this replacement assumes that the extractvalue is the only
12522   // use of the load; that's okay because we don't want to perform this
12523   // transformation in other cases anyway.
12524   SDValue Load;
12525   SDValue Chain;
12526   if (ResultVT.bitsGT(VecEltVT)) {
12527     // If the result type of vextract is wider than the load, then issue an
12528     // extending load instead.
12529     ISD::LoadExtType ExtType = TLI.isLoadExtLegal(ISD::ZEXTLOAD, ResultVT,
12530                                                   VecEltVT)
12531                                    ? ISD::ZEXTLOAD
12532                                    : ISD::EXTLOAD;
12533     Load = DAG.getExtLoad(ExtType, SDLoc(EVE), ResultVT,
12534                           OriginalLoad->getChain(), NewPtr, MPI, VecEltVT,
12535                           Align, OriginalLoad->getMemOperand()->getFlags(),
12536                           OriginalLoad->getAAInfo());
12537     Chain = Load.getValue(1);
12538   } else {
12539     Load = DAG.getLoad(VecEltVT, SDLoc(EVE), OriginalLoad->getChain(), NewPtr,
12540                        MPI, Align, OriginalLoad->getMemOperand()->getFlags(),
12541                        OriginalLoad->getAAInfo());
12542     Chain = Load.getValue(1);
12543     if (ResultVT.bitsLT(VecEltVT))
12544       Load = DAG.getNode(ISD::TRUNCATE, SDLoc(EVE), ResultVT, Load);
12545     else
12546       Load = DAG.getBitcast(ResultVT, Load);
12547   }
12548   WorklistRemover DeadNodes(*this);
12549   SDValue From[] = { SDValue(EVE, 0), SDValue(OriginalLoad, 1) };
12550   SDValue To[] = { Load, Chain };
12551   DAG.ReplaceAllUsesOfValuesWith(From, To, 2);
12552   // Since we're explicitly calling ReplaceAllUses, add the new node to the
12553   // worklist explicitly as well.
12554   AddToWorklist(Load.getNode());
12555   AddUsersToWorklist(Load.getNode()); // Add users too
12556   // Make sure to revisit this node to clean it up; it will usually be dead.
12557   AddToWorklist(EVE);
12558   ++OpsNarrowed;
12559   return SDValue(EVE, 0);
12560 }
12561 
12562 SDValue DAGCombiner::visitEXTRACT_VECTOR_ELT(SDNode *N) {
12563   // (vextract (scalar_to_vector val, 0) -> val
12564   SDValue InVec = N->getOperand(0);
12565   EVT VT = InVec.getValueType();
12566   EVT NVT = N->getValueType(0);
12567 
12568   if (InVec.getOpcode() == ISD::SCALAR_TO_VECTOR) {
12569     // Check if the result type doesn't match the inserted element type. A
12570     // SCALAR_TO_VECTOR may truncate the inserted element and the
12571     // EXTRACT_VECTOR_ELT may widen the extracted vector.
12572     SDValue InOp = InVec.getOperand(0);
12573     if (InOp.getValueType() != NVT) {
12574       assert(InOp.getValueType().isInteger() && NVT.isInteger());
12575       return DAG.getSExtOrTrunc(InOp, SDLoc(InVec), NVT);
12576     }
12577     return InOp;
12578   }
12579 
12580   SDValue EltNo = N->getOperand(1);
12581   ConstantSDNode *ConstEltNo = dyn_cast<ConstantSDNode>(EltNo);
12582 
12583   // extract_vector_elt (build_vector x, y), 1 -> y
12584   if (ConstEltNo &&
12585       InVec.getOpcode() == ISD::BUILD_VECTOR &&
12586       TLI.isTypeLegal(VT) &&
12587       (InVec.hasOneUse() ||
12588        TLI.aggressivelyPreferBuildVectorSources(VT))) {
12589     SDValue Elt = InVec.getOperand(ConstEltNo->getZExtValue());
12590     EVT InEltVT = Elt.getValueType();
12591 
12592     // Sometimes build_vector's scalar input types do not match result type.
12593     if (NVT == InEltVT)
12594       return Elt;
12595 
12596     // TODO: It may be useful to truncate if free if the build_vector implicitly
12597     // converts.
12598   }
12599 
12600   // extract_vector_elt (v2i32 (bitcast i64:x)), 0 -> i32 (trunc i64:x)
12601   if (ConstEltNo && InVec.getOpcode() == ISD::BITCAST && InVec.hasOneUse() &&
12602       ConstEltNo->isNullValue() && VT.isInteger()) {
12603     SDValue BCSrc = InVec.getOperand(0);
12604     if (BCSrc.getValueType().isScalarInteger())
12605       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), NVT, BCSrc);
12606   }
12607 
12608   // extract_vector_elt (insert_vector_elt vec, val, idx), idx) -> val
12609   //
12610   // This only really matters if the index is non-constant since other combines
12611   // on the constant elements already work.
12612   if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT &&
12613       EltNo == InVec.getOperand(2)) {
12614     SDValue Elt = InVec.getOperand(1);
12615     return VT.isInteger() ? DAG.getAnyExtOrTrunc(Elt, SDLoc(N), NVT) : Elt;
12616   }
12617 
12618   // Transform: (EXTRACT_VECTOR_ELT( VECTOR_SHUFFLE )) -> EXTRACT_VECTOR_ELT.
12619   // We only perform this optimization before the op legalization phase because
12620   // we may introduce new vector instructions which are not backed by TD
12621   // patterns. For example on AVX, extracting elements from a wide vector
12622   // without using extract_subvector. However, if we can find an underlying
12623   // scalar value, then we can always use that.
12624   if (ConstEltNo && InVec.getOpcode() == ISD::VECTOR_SHUFFLE) {
12625     int NumElem = VT.getVectorNumElements();
12626     ShuffleVectorSDNode *SVOp = cast<ShuffleVectorSDNode>(InVec);
12627     // Find the new index to extract from.
12628     int OrigElt = SVOp->getMaskElt(ConstEltNo->getZExtValue());
12629 
12630     // Extracting an undef index is undef.
12631     if (OrigElt == -1)
12632       return DAG.getUNDEF(NVT);
12633 
12634     // Select the right vector half to extract from.
12635     SDValue SVInVec;
12636     if (OrigElt < NumElem) {
12637       SVInVec = InVec->getOperand(0);
12638     } else {
12639       SVInVec = InVec->getOperand(1);
12640       OrigElt -= NumElem;
12641     }
12642 
12643     if (SVInVec.getOpcode() == ISD::BUILD_VECTOR) {
12644       SDValue InOp = SVInVec.getOperand(OrigElt);
12645       if (InOp.getValueType() != NVT) {
12646         assert(InOp.getValueType().isInteger() && NVT.isInteger());
12647         InOp = DAG.getSExtOrTrunc(InOp, SDLoc(SVInVec), NVT);
12648       }
12649 
12650       return InOp;
12651     }
12652 
12653     // FIXME: We should handle recursing on other vector shuffles and
12654     // scalar_to_vector here as well.
12655 
12656     if (!LegalOperations) {
12657       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
12658       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(N), NVT, SVInVec,
12659                          DAG.getConstant(OrigElt, SDLoc(SVOp), IndexTy));
12660     }
12661   }
12662 
12663   bool BCNumEltsChanged = false;
12664   EVT ExtVT = VT.getVectorElementType();
12665   EVT LVT = ExtVT;
12666 
12667   // If the result of load has to be truncated, then it's not necessarily
12668   // profitable.
12669   if (NVT.bitsLT(LVT) && !TLI.isTruncateFree(LVT, NVT))
12670     return SDValue();
12671 
12672   if (InVec.getOpcode() == ISD::BITCAST) {
12673     // Don't duplicate a load with other uses.
12674     if (!InVec.hasOneUse())
12675       return SDValue();
12676 
12677     EVT BCVT = InVec.getOperand(0).getValueType();
12678     if (!BCVT.isVector() || ExtVT.bitsGT(BCVT.getVectorElementType()))
12679       return SDValue();
12680     if (VT.getVectorNumElements() != BCVT.getVectorNumElements())
12681       BCNumEltsChanged = true;
12682     InVec = InVec.getOperand(0);
12683     ExtVT = BCVT.getVectorElementType();
12684   }
12685 
12686   // (vextract (vN[if]M load $addr), i) -> ([if]M load $addr + i * size)
12687   if (!LegalOperations && !ConstEltNo && InVec.hasOneUse() &&
12688       ISD::isNormalLoad(InVec.getNode()) &&
12689       !N->getOperand(1)->hasPredecessor(InVec.getNode())) {
12690     SDValue Index = N->getOperand(1);
12691     if (LoadSDNode *OrigLoad = dyn_cast<LoadSDNode>(InVec)) {
12692       if (!OrigLoad->isVolatile()) {
12693         return ReplaceExtractVectorEltOfLoadWithNarrowedLoad(N, VT, Index,
12694                                                              OrigLoad);
12695       }
12696     }
12697   }
12698 
12699   // Perform only after legalization to ensure build_vector / vector_shuffle
12700   // optimizations have already been done.
12701   if (!LegalOperations) return SDValue();
12702 
12703   // (vextract (v4f32 load $addr), c) -> (f32 load $addr+c*size)
12704   // (vextract (v4f32 s2v (f32 load $addr)), c) -> (f32 load $addr+c*size)
12705   // (vextract (v4f32 shuffle (load $addr), <1,u,u,u>), 0) -> (f32 load $addr)
12706 
12707   if (ConstEltNo) {
12708     int Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
12709 
12710     LoadSDNode *LN0 = nullptr;
12711     const ShuffleVectorSDNode *SVN = nullptr;
12712     if (ISD::isNormalLoad(InVec.getNode())) {
12713       LN0 = cast<LoadSDNode>(InVec);
12714     } else if (InVec.getOpcode() == ISD::SCALAR_TO_VECTOR &&
12715                InVec.getOperand(0).getValueType() == ExtVT &&
12716                ISD::isNormalLoad(InVec.getOperand(0).getNode())) {
12717       // Don't duplicate a load with other uses.
12718       if (!InVec.hasOneUse())
12719         return SDValue();
12720 
12721       LN0 = cast<LoadSDNode>(InVec.getOperand(0));
12722     } else if ((SVN = dyn_cast<ShuffleVectorSDNode>(InVec))) {
12723       // (vextract (vector_shuffle (load $addr), v2, <1, u, u, u>), 1)
12724       // =>
12725       // (load $addr+1*size)
12726 
12727       // Don't duplicate a load with other uses.
12728       if (!InVec.hasOneUse())
12729         return SDValue();
12730 
12731       // If the bit convert changed the number of elements, it is unsafe
12732       // to examine the mask.
12733       if (BCNumEltsChanged)
12734         return SDValue();
12735 
12736       // Select the input vector, guarding against out of range extract vector.
12737       unsigned NumElems = VT.getVectorNumElements();
12738       int Idx = (Elt > (int)NumElems) ? -1 : SVN->getMaskElt(Elt);
12739       InVec = (Idx < (int)NumElems) ? InVec.getOperand(0) : InVec.getOperand(1);
12740 
12741       if (InVec.getOpcode() == ISD::BITCAST) {
12742         // Don't duplicate a load with other uses.
12743         if (!InVec.hasOneUse())
12744           return SDValue();
12745 
12746         InVec = InVec.getOperand(0);
12747       }
12748       if (ISD::isNormalLoad(InVec.getNode())) {
12749         LN0 = cast<LoadSDNode>(InVec);
12750         Elt = (Idx < (int)NumElems) ? Idx : Idx - (int)NumElems;
12751         EltNo = DAG.getConstant(Elt, SDLoc(EltNo), EltNo.getValueType());
12752       }
12753     }
12754 
12755     // Make sure we found a non-volatile load and the extractelement is
12756     // the only use.
12757     if (!LN0 || !LN0->hasNUsesOfValue(1,0) || LN0->isVolatile())
12758       return SDValue();
12759 
12760     // If Idx was -1 above, Elt is going to be -1, so just return undef.
12761     if (Elt == -1)
12762       return DAG.getUNDEF(LVT);
12763 
12764     return ReplaceExtractVectorEltOfLoadWithNarrowedLoad(N, VT, EltNo, LN0);
12765   }
12766 
12767   return SDValue();
12768 }
12769 
12770 // Simplify (build_vec (ext )) to (bitcast (build_vec ))
12771 SDValue DAGCombiner::reduceBuildVecExtToExtBuildVec(SDNode *N) {
12772   // We perform this optimization post type-legalization because
12773   // the type-legalizer often scalarizes integer-promoted vectors.
12774   // Performing this optimization before may create bit-casts which
12775   // will be type-legalized to complex code sequences.
12776   // We perform this optimization only before the operation legalizer because we
12777   // may introduce illegal operations.
12778   if (Level != AfterLegalizeVectorOps && Level != AfterLegalizeTypes)
12779     return SDValue();
12780 
12781   unsigned NumInScalars = N->getNumOperands();
12782   SDLoc DL(N);
12783   EVT VT = N->getValueType(0);
12784 
12785   // Check to see if this is a BUILD_VECTOR of a bunch of values
12786   // which come from any_extend or zero_extend nodes. If so, we can create
12787   // a new BUILD_VECTOR using bit-casts which may enable other BUILD_VECTOR
12788   // optimizations. We do not handle sign-extend because we can't fill the sign
12789   // using shuffles.
12790   EVT SourceType = MVT::Other;
12791   bool AllAnyExt = true;
12792 
12793   for (unsigned i = 0; i != NumInScalars; ++i) {
12794     SDValue In = N->getOperand(i);
12795     // Ignore undef inputs.
12796     if (In.isUndef()) continue;
12797 
12798     bool AnyExt  = In.getOpcode() == ISD::ANY_EXTEND;
12799     bool ZeroExt = In.getOpcode() == ISD::ZERO_EXTEND;
12800 
12801     // Abort if the element is not an extension.
12802     if (!ZeroExt && !AnyExt) {
12803       SourceType = MVT::Other;
12804       break;
12805     }
12806 
12807     // The input is a ZeroExt or AnyExt. Check the original type.
12808     EVT InTy = In.getOperand(0).getValueType();
12809 
12810     // Check that all of the widened source types are the same.
12811     if (SourceType == MVT::Other)
12812       // First time.
12813       SourceType = InTy;
12814     else if (InTy != SourceType) {
12815       // Multiple income types. Abort.
12816       SourceType = MVT::Other;
12817       break;
12818     }
12819 
12820     // Check if all of the extends are ANY_EXTENDs.
12821     AllAnyExt &= AnyExt;
12822   }
12823 
12824   // In order to have valid types, all of the inputs must be extended from the
12825   // same source type and all of the inputs must be any or zero extend.
12826   // Scalar sizes must be a power of two.
12827   EVT OutScalarTy = VT.getScalarType();
12828   bool ValidTypes = SourceType != MVT::Other &&
12829                  isPowerOf2_32(OutScalarTy.getSizeInBits()) &&
12830                  isPowerOf2_32(SourceType.getSizeInBits());
12831 
12832   // Create a new simpler BUILD_VECTOR sequence which other optimizations can
12833   // turn into a single shuffle instruction.
12834   if (!ValidTypes)
12835     return SDValue();
12836 
12837   bool isLE = DAG.getDataLayout().isLittleEndian();
12838   unsigned ElemRatio = OutScalarTy.getSizeInBits()/SourceType.getSizeInBits();
12839   assert(ElemRatio > 1 && "Invalid element size ratio");
12840   SDValue Filler = AllAnyExt ? DAG.getUNDEF(SourceType):
12841                                DAG.getConstant(0, DL, SourceType);
12842 
12843   unsigned NewBVElems = ElemRatio * VT.getVectorNumElements();
12844   SmallVector<SDValue, 8> Ops(NewBVElems, Filler);
12845 
12846   // Populate the new build_vector
12847   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
12848     SDValue Cast = N->getOperand(i);
12849     assert((Cast.getOpcode() == ISD::ANY_EXTEND ||
12850             Cast.getOpcode() == ISD::ZERO_EXTEND ||
12851             Cast.isUndef()) && "Invalid cast opcode");
12852     SDValue In;
12853     if (Cast.isUndef())
12854       In = DAG.getUNDEF(SourceType);
12855     else
12856       In = Cast->getOperand(0);
12857     unsigned Index = isLE ? (i * ElemRatio) :
12858                             (i * ElemRatio + (ElemRatio - 1));
12859 
12860     assert(Index < Ops.size() && "Invalid index");
12861     Ops[Index] = In;
12862   }
12863 
12864   // The type of the new BUILD_VECTOR node.
12865   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SourceType, NewBVElems);
12866   assert(VecVT.getSizeInBits() == VT.getSizeInBits() &&
12867          "Invalid vector size");
12868   // Check if the new vector type is legal.
12869   if (!isTypeLegal(VecVT)) return SDValue();
12870 
12871   // Make the new BUILD_VECTOR.
12872   SDValue BV = DAG.getBuildVector(VecVT, DL, Ops);
12873 
12874   // The new BUILD_VECTOR node has the potential to be further optimized.
12875   AddToWorklist(BV.getNode());
12876   // Bitcast to the desired type.
12877   return DAG.getBitcast(VT, BV);
12878 }
12879 
12880 SDValue DAGCombiner::reduceBuildVecConvertToConvertBuildVec(SDNode *N) {
12881   EVT VT = N->getValueType(0);
12882 
12883   unsigned NumInScalars = N->getNumOperands();
12884   SDLoc DL(N);
12885 
12886   EVT SrcVT = MVT::Other;
12887   unsigned Opcode = ISD::DELETED_NODE;
12888   unsigned NumDefs = 0;
12889 
12890   for (unsigned i = 0; i != NumInScalars; ++i) {
12891     SDValue In = N->getOperand(i);
12892     unsigned Opc = In.getOpcode();
12893 
12894     if (Opc == ISD::UNDEF)
12895       continue;
12896 
12897     // If all scalar values are floats and converted from integers.
12898     if (Opcode == ISD::DELETED_NODE &&
12899         (Opc == ISD::UINT_TO_FP || Opc == ISD::SINT_TO_FP)) {
12900       Opcode = Opc;
12901     }
12902 
12903     if (Opc != Opcode)
12904       return SDValue();
12905 
12906     EVT InVT = In.getOperand(0).getValueType();
12907 
12908     // If all scalar values are typed differently, bail out. It's chosen to
12909     // simplify BUILD_VECTOR of integer types.
12910     if (SrcVT == MVT::Other)
12911       SrcVT = InVT;
12912     if (SrcVT != InVT)
12913       return SDValue();
12914     NumDefs++;
12915   }
12916 
12917   // If the vector has just one element defined, it's not worth to fold it into
12918   // a vectorized one.
12919   if (NumDefs < 2)
12920     return SDValue();
12921 
12922   assert((Opcode == ISD::UINT_TO_FP || Opcode == ISD::SINT_TO_FP)
12923          && "Should only handle conversion from integer to float.");
12924   assert(SrcVT != MVT::Other && "Cannot determine source type!");
12925 
12926   EVT NVT = EVT::getVectorVT(*DAG.getContext(), SrcVT, NumInScalars);
12927 
12928   if (!TLI.isOperationLegalOrCustom(Opcode, NVT))
12929     return SDValue();
12930 
12931   // Just because the floating-point vector type is legal does not necessarily
12932   // mean that the corresponding integer vector type is.
12933   if (!isTypeLegal(NVT))
12934     return SDValue();
12935 
12936   SmallVector<SDValue, 8> Opnds;
12937   for (unsigned i = 0; i != NumInScalars; ++i) {
12938     SDValue In = N->getOperand(i);
12939 
12940     if (In.isUndef())
12941       Opnds.push_back(DAG.getUNDEF(SrcVT));
12942     else
12943       Opnds.push_back(In.getOperand(0));
12944   }
12945   SDValue BV = DAG.getBuildVector(NVT, DL, Opnds);
12946   AddToWorklist(BV.getNode());
12947 
12948   return DAG.getNode(Opcode, DL, VT, BV);
12949 }
12950 
12951 SDValue DAGCombiner::createBuildVecShuffle(SDLoc DL, SDNode *N,
12952                                            ArrayRef<int> VectorMask,
12953                                            SDValue VecIn1, SDValue VecIn2,
12954                                            unsigned LeftIdx) {
12955   MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout());
12956   SDValue ZeroIdx = DAG.getConstant(0, DL, IdxTy);
12957 
12958   EVT VT = N->getValueType(0);
12959   EVT InVT1 = VecIn1.getValueType();
12960   EVT InVT2 = VecIn2.getNode() ? VecIn2.getValueType() : InVT1;
12961 
12962   unsigned Vec2Offset = InVT1.getVectorNumElements();
12963   unsigned NumElems = VT.getVectorNumElements();
12964   unsigned ShuffleNumElems = NumElems;
12965 
12966   // We can't generate a shuffle node with mismatched input and output types.
12967   // Try to make the types match the type of the output.
12968   if (InVT1 != VT || InVT2 != VT) {
12969     if (InVT1.getSizeInBits() * 2 == VT.getSizeInBits() && InVT1 == InVT2) {
12970       // If both input vectors are exactly half the size of the output, concat
12971       // them. If we have only one (non-zero) input, concat it with undef.
12972       VecIn1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, VecIn1,
12973                            VecIn2.getNode() ? VecIn2 : DAG.getUNDEF(InVT1));
12974       VecIn2 = SDValue();
12975     } else if (InVT1.getSizeInBits() == VT.getSizeInBits() * 2) {
12976       if (!TLI.isExtractSubvectorCheap(VT, NumElems))
12977         return SDValue();
12978 
12979       if (!VecIn2.getNode()) {
12980         // If we only have one input vector, and it's twice the size of the
12981         // output, split it in two.
12982         VecIn2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1,
12983                              DAG.getConstant(NumElems, DL, IdxTy));
12984         VecIn1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1, ZeroIdx);
12985         // Since we now have shorter input vectors, adjust the offset of the
12986         // second vector's start.
12987         Vec2Offset = NumElems;
12988       } else if (InVT2.getSizeInBits() <= InVT1.getSizeInBits()) {
12989         // VecIn1 is wider than the output, and we have another, possibly
12990         // smaller input. Pad the smaller input with undefs, shuffle at the
12991         // input vector width, and extract the output.
12992         // The shuffle type is different than VT, so check legality again.
12993         if (LegalOperations &&
12994             !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, InVT1))
12995           return SDValue();
12996 
12997         if (InVT1 != InVT2)
12998           VecIn2 = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, InVT1,
12999                                DAG.getUNDEF(InVT1), VecIn2, ZeroIdx);
13000         ShuffleNumElems = NumElems * 2;
13001       } else {
13002         // Both VecIn1 and VecIn2 are wider than the output, and VecIn2 is wider
13003         // than VecIn1. We can't handle this for now - this case will disappear
13004         // when we start sorting the vectors by type.
13005         return SDValue();
13006       }
13007     } else {
13008       // TODO: Support cases where the length mismatch isn't exactly by a
13009       // factor of 2.
13010       // TODO: Move this check upwards, so that if we have bad type
13011       // mismatches, we don't create any DAG nodes.
13012       return SDValue();
13013     }
13014   }
13015 
13016   // Initialize mask to undef.
13017   SmallVector<int, 8> Mask(ShuffleNumElems, -1);
13018 
13019   // Only need to run up to the number of elements actually used, not the
13020   // total number of elements in the shuffle - if we are shuffling a wider
13021   // vector, the high lanes should be set to undef.
13022   for (unsigned i = 0; i != NumElems; ++i) {
13023     if (VectorMask[i] <= 0)
13024       continue;
13025 
13026     SDValue Extract = N->getOperand(i);
13027     unsigned ExtIndex =
13028         cast<ConstantSDNode>(Extract.getOperand(1))->getZExtValue();
13029 
13030     if (VectorMask[i] == (int)LeftIdx) {
13031       Mask[i] = ExtIndex;
13032     } else if (VectorMask[i] == (int)LeftIdx + 1) {
13033       Mask[i] = Vec2Offset + ExtIndex;
13034     }
13035   }
13036 
13037   // The type the input vectors may have changed above.
13038   InVT1 = VecIn1.getValueType();
13039 
13040   // If we already have a VecIn2, it should have the same type as VecIn1.
13041   // If we don't, get an undef/zero vector of the appropriate type.
13042   VecIn2 = VecIn2.getNode() ? VecIn2 : DAG.getUNDEF(InVT1);
13043   assert(InVT1 == VecIn2.getValueType() && "Unexpected second input type.");
13044 
13045   SDValue Shuffle = DAG.getVectorShuffle(InVT1, DL, VecIn1, VecIn2, Mask);
13046   if (ShuffleNumElems > NumElems)
13047     Shuffle = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Shuffle, ZeroIdx);
13048 
13049   return Shuffle;
13050 }
13051 
13052 // Check to see if this is a BUILD_VECTOR of a bunch of EXTRACT_VECTOR_ELT
13053 // operations. If the types of the vectors we're extracting from allow it,
13054 // turn this into a vector_shuffle node.
13055 SDValue DAGCombiner::reduceBuildVecToShuffle(SDNode *N) {
13056   SDLoc DL(N);
13057   EVT VT = N->getValueType(0);
13058 
13059   // Only type-legal BUILD_VECTOR nodes are converted to shuffle nodes.
13060   if (!isTypeLegal(VT))
13061     return SDValue();
13062 
13063   // May only combine to shuffle after legalize if shuffle is legal.
13064   if (LegalOperations && !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, VT))
13065     return SDValue();
13066 
13067   bool UsesZeroVector = false;
13068   unsigned NumElems = N->getNumOperands();
13069 
13070   // Record, for each element of the newly built vector, which input vector
13071   // that element comes from. -1 stands for undef, 0 for the zero vector,
13072   // and positive values for the input vectors.
13073   // VectorMask maps each element to its vector number, and VecIn maps vector
13074   // numbers to their initial SDValues.
13075 
13076   SmallVector<int, 8> VectorMask(NumElems, -1);
13077   SmallVector<SDValue, 8> VecIn;
13078   VecIn.push_back(SDValue());
13079 
13080   for (unsigned i = 0; i != NumElems; ++i) {
13081     SDValue Op = N->getOperand(i);
13082 
13083     if (Op.isUndef())
13084       continue;
13085 
13086     // See if we can use a blend with a zero vector.
13087     // TODO: Should we generalize this to a blend with an arbitrary constant
13088     // vector?
13089     if (isNullConstant(Op) || isNullFPConstant(Op)) {
13090       UsesZeroVector = true;
13091       VectorMask[i] = 0;
13092       continue;
13093     }
13094 
13095     // Not an undef or zero. If the input is something other than an
13096     // EXTRACT_VECTOR_ELT with a constant index, bail out.
13097     if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
13098         !isa<ConstantSDNode>(Op.getOperand(1)))
13099       return SDValue();
13100 
13101     SDValue ExtractedFromVec = Op.getOperand(0);
13102 
13103     // All inputs must have the same element type as the output.
13104     if (VT.getVectorElementType() !=
13105         ExtractedFromVec.getValueType().getVectorElementType())
13106       return SDValue();
13107 
13108     // Have we seen this input vector before?
13109     // The vectors are expected to be tiny (usually 1 or 2 elements), so using
13110     // a map back from SDValues to numbers isn't worth it.
13111     unsigned Idx = std::distance(
13112         VecIn.begin(), std::find(VecIn.begin(), VecIn.end(), ExtractedFromVec));
13113     if (Idx == VecIn.size())
13114       VecIn.push_back(ExtractedFromVec);
13115 
13116     VectorMask[i] = Idx;
13117   }
13118 
13119   // If we didn't find at least one input vector, bail out.
13120   if (VecIn.size() < 2)
13121     return SDValue();
13122 
13123   // TODO: We want to sort the vectors by descending length, so that adjacent
13124   // pairs have similar length, and the longer vector is always first in the
13125   // pair.
13126 
13127   // TODO: Should this fire if some of the input vectors has illegal type (like
13128   // it does now), or should we let legalization run its course first?
13129 
13130   // Shuffle phase:
13131   // Take pairs of vectors, and shuffle them so that the result has elements
13132   // from these vectors in the correct places.
13133   // For example, given:
13134   // t10: i32 = extract_vector_elt t1, Constant:i64<0>
13135   // t11: i32 = extract_vector_elt t2, Constant:i64<0>
13136   // t12: i32 = extract_vector_elt t3, Constant:i64<0>
13137   // t13: i32 = extract_vector_elt t1, Constant:i64<1>
13138   // t14: v4i32 = BUILD_VECTOR t10, t11, t12, t13
13139   // We will generate:
13140   // t20: v4i32 = vector_shuffle<0,4,u,1> t1, t2
13141   // t21: v4i32 = vector_shuffle<u,u,0,u> t3, undef
13142   SmallVector<SDValue, 4> Shuffles;
13143   for (unsigned In = 0, Len = (VecIn.size() / 2); In < Len; ++In) {
13144     unsigned LeftIdx = 2 * In + 1;
13145     SDValue VecLeft = VecIn[LeftIdx];
13146     SDValue VecRight =
13147         (LeftIdx + 1) < VecIn.size() ? VecIn[LeftIdx + 1] : SDValue();
13148 
13149     if (SDValue Shuffle = createBuildVecShuffle(DL, N, VectorMask, VecLeft,
13150                                                 VecRight, LeftIdx))
13151       Shuffles.push_back(Shuffle);
13152     else
13153       return SDValue();
13154   }
13155 
13156   // If we need the zero vector as an "ingredient" in the blend tree, add it
13157   // to the list of shuffles.
13158   if (UsesZeroVector)
13159     Shuffles.push_back(VT.isInteger() ? DAG.getConstant(0, DL, VT)
13160                                       : DAG.getConstantFP(0.0, DL, VT));
13161 
13162   // If we only have one shuffle, we're done.
13163   if (Shuffles.size() == 1)
13164     return Shuffles[0];
13165 
13166   // Update the vector mask to point to the post-shuffle vectors.
13167   for (int &Vec : VectorMask)
13168     if (Vec == 0)
13169       Vec = Shuffles.size() - 1;
13170     else
13171       Vec = (Vec - 1) / 2;
13172 
13173   // More than one shuffle. Generate a binary tree of blends, e.g. if from
13174   // the previous step we got the set of shuffles t10, t11, t12, t13, we will
13175   // generate:
13176   // t10: v8i32 = vector_shuffle<0,8,u,u,u,u,u,u> t1, t2
13177   // t11: v8i32 = vector_shuffle<u,u,0,8,u,u,u,u> t3, t4
13178   // t12: v8i32 = vector_shuffle<u,u,u,u,0,8,u,u> t5, t6
13179   // t13: v8i32 = vector_shuffle<u,u,u,u,u,u,0,8> t7, t8
13180   // t20: v8i32 = vector_shuffle<0,1,10,11,u,u,u,u> t10, t11
13181   // t21: v8i32 = vector_shuffle<u,u,u,u,4,5,14,15> t12, t13
13182   // t30: v8i32 = vector_shuffle<0,1,2,3,12,13,14,15> t20, t21
13183 
13184   // Make sure the initial size of the shuffle list is even.
13185   if (Shuffles.size() % 2)
13186     Shuffles.push_back(DAG.getUNDEF(VT));
13187 
13188   for (unsigned CurSize = Shuffles.size(); CurSize > 1; CurSize /= 2) {
13189     if (CurSize % 2) {
13190       Shuffles[CurSize] = DAG.getUNDEF(VT);
13191       CurSize++;
13192     }
13193     for (unsigned In = 0, Len = CurSize / 2; In < Len; ++In) {
13194       int Left = 2 * In;
13195       int Right = 2 * In + 1;
13196       SmallVector<int, 8> Mask(NumElems, -1);
13197       for (unsigned i = 0; i != NumElems; ++i) {
13198         if (VectorMask[i] == Left) {
13199           Mask[i] = i;
13200           VectorMask[i] = In;
13201         } else if (VectorMask[i] == Right) {
13202           Mask[i] = i + NumElems;
13203           VectorMask[i] = In;
13204         }
13205       }
13206 
13207       Shuffles[In] =
13208           DAG.getVectorShuffle(VT, DL, Shuffles[Left], Shuffles[Right], Mask);
13209     }
13210   }
13211 
13212   return Shuffles[0];
13213 }
13214 
13215 SDValue DAGCombiner::visitBUILD_VECTOR(SDNode *N) {
13216   EVT VT = N->getValueType(0);
13217 
13218   // A vector built entirely of undefs is undef.
13219   if (ISD::allOperandsUndef(N))
13220     return DAG.getUNDEF(VT);
13221 
13222   if (SDValue V = reduceBuildVecExtToExtBuildVec(N))
13223     return V;
13224 
13225   if (SDValue V = reduceBuildVecConvertToConvertBuildVec(N))
13226     return V;
13227 
13228   if (SDValue V = reduceBuildVecToShuffle(N))
13229     return V;
13230 
13231   return SDValue();
13232 }
13233 
13234 static SDValue combineConcatVectorOfScalars(SDNode *N, SelectionDAG &DAG) {
13235   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
13236   EVT OpVT = N->getOperand(0).getValueType();
13237 
13238   // If the operands are legal vectors, leave them alone.
13239   if (TLI.isTypeLegal(OpVT))
13240     return SDValue();
13241 
13242   SDLoc DL(N);
13243   EVT VT = N->getValueType(0);
13244   SmallVector<SDValue, 8> Ops;
13245 
13246   EVT SVT = EVT::getIntegerVT(*DAG.getContext(), OpVT.getSizeInBits());
13247   SDValue ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
13248 
13249   // Keep track of what we encounter.
13250   bool AnyInteger = false;
13251   bool AnyFP = false;
13252   for (const SDValue &Op : N->ops()) {
13253     if (ISD::BITCAST == Op.getOpcode() &&
13254         !Op.getOperand(0).getValueType().isVector())
13255       Ops.push_back(Op.getOperand(0));
13256     else if (ISD::UNDEF == Op.getOpcode())
13257       Ops.push_back(ScalarUndef);
13258     else
13259       return SDValue();
13260 
13261     // Note whether we encounter an integer or floating point scalar.
13262     // If it's neither, bail out, it could be something weird like x86mmx.
13263     EVT LastOpVT = Ops.back().getValueType();
13264     if (LastOpVT.isFloatingPoint())
13265       AnyFP = true;
13266     else if (LastOpVT.isInteger())
13267       AnyInteger = true;
13268     else
13269       return SDValue();
13270   }
13271 
13272   // If any of the operands is a floating point scalar bitcast to a vector,
13273   // use floating point types throughout, and bitcast everything.
13274   // Replace UNDEFs by another scalar UNDEF node, of the final desired type.
13275   if (AnyFP) {
13276     SVT = EVT::getFloatingPointVT(OpVT.getSizeInBits());
13277     ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
13278     if (AnyInteger) {
13279       for (SDValue &Op : Ops) {
13280         if (Op.getValueType() == SVT)
13281           continue;
13282         if (Op.isUndef())
13283           Op = ScalarUndef;
13284         else
13285           Op = DAG.getBitcast(SVT, Op);
13286       }
13287     }
13288   }
13289 
13290   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SVT,
13291                                VT.getSizeInBits() / SVT.getSizeInBits());
13292   return DAG.getBitcast(VT, DAG.getBuildVector(VecVT, DL, Ops));
13293 }
13294 
13295 // Check to see if this is a CONCAT_VECTORS of a bunch of EXTRACT_SUBVECTOR
13296 // operations. If so, and if the EXTRACT_SUBVECTOR vector inputs come from at
13297 // most two distinct vectors the same size as the result, attempt to turn this
13298 // into a legal shuffle.
13299 static SDValue combineConcatVectorOfExtracts(SDNode *N, SelectionDAG &DAG) {
13300   EVT VT = N->getValueType(0);
13301   EVT OpVT = N->getOperand(0).getValueType();
13302   int NumElts = VT.getVectorNumElements();
13303   int NumOpElts = OpVT.getVectorNumElements();
13304 
13305   SDValue SV0 = DAG.getUNDEF(VT), SV1 = DAG.getUNDEF(VT);
13306   SmallVector<int, 8> Mask;
13307 
13308   for (SDValue Op : N->ops()) {
13309     // Peek through any bitcast.
13310     while (Op.getOpcode() == ISD::BITCAST)
13311       Op = Op.getOperand(0);
13312 
13313     // UNDEF nodes convert to UNDEF shuffle mask values.
13314     if (Op.isUndef()) {
13315       Mask.append((unsigned)NumOpElts, -1);
13316       continue;
13317     }
13318 
13319     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
13320       return SDValue();
13321 
13322     // What vector are we extracting the subvector from and at what index?
13323     SDValue ExtVec = Op.getOperand(0);
13324 
13325     // We want the EVT of the original extraction to correctly scale the
13326     // extraction index.
13327     EVT ExtVT = ExtVec.getValueType();
13328 
13329     // Peek through any bitcast.
13330     while (ExtVec.getOpcode() == ISD::BITCAST)
13331       ExtVec = ExtVec.getOperand(0);
13332 
13333     // UNDEF nodes convert to UNDEF shuffle mask values.
13334     if (ExtVec.isUndef()) {
13335       Mask.append((unsigned)NumOpElts, -1);
13336       continue;
13337     }
13338 
13339     if (!isa<ConstantSDNode>(Op.getOperand(1)))
13340       return SDValue();
13341     int ExtIdx = cast<ConstantSDNode>(Op.getOperand(1))->getZExtValue();
13342 
13343     // Ensure that we are extracting a subvector from a vector the same
13344     // size as the result.
13345     if (ExtVT.getSizeInBits() != VT.getSizeInBits())
13346       return SDValue();
13347 
13348     // Scale the subvector index to account for any bitcast.
13349     int NumExtElts = ExtVT.getVectorNumElements();
13350     if (0 == (NumExtElts % NumElts))
13351       ExtIdx /= (NumExtElts / NumElts);
13352     else if (0 == (NumElts % NumExtElts))
13353       ExtIdx *= (NumElts / NumExtElts);
13354     else
13355       return SDValue();
13356 
13357     // At most we can reference 2 inputs in the final shuffle.
13358     if (SV0.isUndef() || SV0 == ExtVec) {
13359       SV0 = ExtVec;
13360       for (int i = 0; i != NumOpElts; ++i)
13361         Mask.push_back(i + ExtIdx);
13362     } else if (SV1.isUndef() || SV1 == ExtVec) {
13363       SV1 = ExtVec;
13364       for (int i = 0; i != NumOpElts; ++i)
13365         Mask.push_back(i + ExtIdx + NumElts);
13366     } else {
13367       return SDValue();
13368     }
13369   }
13370 
13371   if (!DAG.getTargetLoweringInfo().isShuffleMaskLegal(Mask, VT))
13372     return SDValue();
13373 
13374   return DAG.getVectorShuffle(VT, SDLoc(N), DAG.getBitcast(VT, SV0),
13375                               DAG.getBitcast(VT, SV1), Mask);
13376 }
13377 
13378 SDValue DAGCombiner::visitCONCAT_VECTORS(SDNode *N) {
13379   // If we only have one input vector, we don't need to do any concatenation.
13380   if (N->getNumOperands() == 1)
13381     return N->getOperand(0);
13382 
13383   // Check if all of the operands are undefs.
13384   EVT VT = N->getValueType(0);
13385   if (ISD::allOperandsUndef(N))
13386     return DAG.getUNDEF(VT);
13387 
13388   // Optimize concat_vectors where all but the first of the vectors are undef.
13389   if (std::all_of(std::next(N->op_begin()), N->op_end(), [](const SDValue &Op) {
13390         return Op.isUndef();
13391       })) {
13392     SDValue In = N->getOperand(0);
13393     assert(In.getValueType().isVector() && "Must concat vectors");
13394 
13395     // Transform: concat_vectors(scalar, undef) -> scalar_to_vector(sclr).
13396     if (In->getOpcode() == ISD::BITCAST &&
13397         !In->getOperand(0)->getValueType(0).isVector()) {
13398       SDValue Scalar = In->getOperand(0);
13399 
13400       // If the bitcast type isn't legal, it might be a trunc of a legal type;
13401       // look through the trunc so we can still do the transform:
13402       //   concat_vectors(trunc(scalar), undef) -> scalar_to_vector(scalar)
13403       if (Scalar->getOpcode() == ISD::TRUNCATE &&
13404           !TLI.isTypeLegal(Scalar.getValueType()) &&
13405           TLI.isTypeLegal(Scalar->getOperand(0).getValueType()))
13406         Scalar = Scalar->getOperand(0);
13407 
13408       EVT SclTy = Scalar->getValueType(0);
13409 
13410       if (!SclTy.isFloatingPoint() && !SclTy.isInteger())
13411         return SDValue();
13412 
13413       EVT NVT = EVT::getVectorVT(*DAG.getContext(), SclTy,
13414                                  VT.getSizeInBits() / SclTy.getSizeInBits());
13415       if (!TLI.isTypeLegal(NVT) || !TLI.isTypeLegal(Scalar.getValueType()))
13416         return SDValue();
13417 
13418       SDValue Res = DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), NVT, Scalar);
13419       return DAG.getBitcast(VT, Res);
13420     }
13421   }
13422 
13423   // Fold any combination of BUILD_VECTOR or UNDEF nodes into one BUILD_VECTOR.
13424   // We have already tested above for an UNDEF only concatenation.
13425   // fold (concat_vectors (BUILD_VECTOR A, B, ...), (BUILD_VECTOR C, D, ...))
13426   // -> (BUILD_VECTOR A, B, ..., C, D, ...)
13427   auto IsBuildVectorOrUndef = [](const SDValue &Op) {
13428     return ISD::UNDEF == Op.getOpcode() || ISD::BUILD_VECTOR == Op.getOpcode();
13429   };
13430   if (llvm::all_of(N->ops(), IsBuildVectorOrUndef)) {
13431     SmallVector<SDValue, 8> Opnds;
13432     EVT SVT = VT.getScalarType();
13433 
13434     EVT MinVT = SVT;
13435     if (!SVT.isFloatingPoint()) {
13436       // If BUILD_VECTOR are from built from integer, they may have different
13437       // operand types. Get the smallest type and truncate all operands to it.
13438       bool FoundMinVT = false;
13439       for (const SDValue &Op : N->ops())
13440         if (ISD::BUILD_VECTOR == Op.getOpcode()) {
13441           EVT OpSVT = Op.getOperand(0)->getValueType(0);
13442           MinVT = (!FoundMinVT || OpSVT.bitsLE(MinVT)) ? OpSVT : MinVT;
13443           FoundMinVT = true;
13444         }
13445       assert(FoundMinVT && "Concat vector type mismatch");
13446     }
13447 
13448     for (const SDValue &Op : N->ops()) {
13449       EVT OpVT = Op.getValueType();
13450       unsigned NumElts = OpVT.getVectorNumElements();
13451 
13452       if (ISD::UNDEF == Op.getOpcode())
13453         Opnds.append(NumElts, DAG.getUNDEF(MinVT));
13454 
13455       if (ISD::BUILD_VECTOR == Op.getOpcode()) {
13456         if (SVT.isFloatingPoint()) {
13457           assert(SVT == OpVT.getScalarType() && "Concat vector type mismatch");
13458           Opnds.append(Op->op_begin(), Op->op_begin() + NumElts);
13459         } else {
13460           for (unsigned i = 0; i != NumElts; ++i)
13461             Opnds.push_back(
13462                 DAG.getNode(ISD::TRUNCATE, SDLoc(N), MinVT, Op.getOperand(i)));
13463         }
13464       }
13465     }
13466 
13467     assert(VT.getVectorNumElements() == Opnds.size() &&
13468            "Concat vector type mismatch");
13469     return DAG.getBuildVector(VT, SDLoc(N), Opnds);
13470   }
13471 
13472   // Fold CONCAT_VECTORS of only bitcast scalars (or undef) to BUILD_VECTOR.
13473   if (SDValue V = combineConcatVectorOfScalars(N, DAG))
13474     return V;
13475 
13476   // Fold CONCAT_VECTORS of EXTRACT_SUBVECTOR (or undef) to VECTOR_SHUFFLE.
13477   if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT))
13478     if (SDValue V = combineConcatVectorOfExtracts(N, DAG))
13479       return V;
13480 
13481   // Type legalization of vectors and DAG canonicalization of SHUFFLE_VECTOR
13482   // nodes often generate nop CONCAT_VECTOR nodes.
13483   // Scan the CONCAT_VECTOR operands and look for a CONCAT operations that
13484   // place the incoming vectors at the exact same location.
13485   SDValue SingleSource = SDValue();
13486   unsigned PartNumElem = N->getOperand(0).getValueType().getVectorNumElements();
13487 
13488   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
13489     SDValue Op = N->getOperand(i);
13490 
13491     if (Op.isUndef())
13492       continue;
13493 
13494     // Check if this is the identity extract:
13495     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
13496       return SDValue();
13497 
13498     // Find the single incoming vector for the extract_subvector.
13499     if (SingleSource.getNode()) {
13500       if (Op.getOperand(0) != SingleSource)
13501         return SDValue();
13502     } else {
13503       SingleSource = Op.getOperand(0);
13504 
13505       // Check the source type is the same as the type of the result.
13506       // If not, this concat may extend the vector, so we can not
13507       // optimize it away.
13508       if (SingleSource.getValueType() != N->getValueType(0))
13509         return SDValue();
13510     }
13511 
13512     unsigned IdentityIndex = i * PartNumElem;
13513     ConstantSDNode *CS = dyn_cast<ConstantSDNode>(Op.getOperand(1));
13514     // The extract index must be constant.
13515     if (!CS)
13516       return SDValue();
13517 
13518     // Check that we are reading from the identity index.
13519     if (CS->getZExtValue() != IdentityIndex)
13520       return SDValue();
13521   }
13522 
13523   if (SingleSource.getNode())
13524     return SingleSource;
13525 
13526   return SDValue();
13527 }
13528 
13529 SDValue DAGCombiner::visitEXTRACT_SUBVECTOR(SDNode* N) {
13530   EVT NVT = N->getValueType(0);
13531   SDValue V = N->getOperand(0);
13532 
13533   if (V->getOpcode() == ISD::CONCAT_VECTORS) {
13534     // Combine:
13535     //    (extract_subvec (concat V1, V2, ...), i)
13536     // Into:
13537     //    Vi if possible
13538     // Only operand 0 is checked as 'concat' assumes all inputs of the same
13539     // type.
13540     if (V->getOperand(0).getValueType() != NVT)
13541       return SDValue();
13542     unsigned Idx = N->getConstantOperandVal(1);
13543     unsigned NumElems = NVT.getVectorNumElements();
13544     assert((Idx % NumElems) == 0 &&
13545            "IDX in concat is not a multiple of the result vector length.");
13546     return V->getOperand(Idx / NumElems);
13547   }
13548 
13549   // Skip bitcasting
13550   if (V->getOpcode() == ISD::BITCAST)
13551     V = V.getOperand(0);
13552 
13553   if (V->getOpcode() == ISD::INSERT_SUBVECTOR) {
13554     // Handle only simple case where vector being inserted and vector
13555     // being extracted are of same type, and are half size of larger vectors.
13556     EVT BigVT = V->getOperand(0).getValueType();
13557     EVT SmallVT = V->getOperand(1).getValueType();
13558     if (!NVT.bitsEq(SmallVT) || NVT.getSizeInBits()*2 != BigVT.getSizeInBits())
13559       return SDValue();
13560 
13561     // Only handle cases where both indexes are constants with the same type.
13562     ConstantSDNode *ExtIdx = dyn_cast<ConstantSDNode>(N->getOperand(1));
13563     ConstantSDNode *InsIdx = dyn_cast<ConstantSDNode>(V->getOperand(2));
13564 
13565     if (InsIdx && ExtIdx &&
13566         InsIdx->getValueType(0).getSizeInBits() <= 64 &&
13567         ExtIdx->getValueType(0).getSizeInBits() <= 64) {
13568       // Combine:
13569       //    (extract_subvec (insert_subvec V1, V2, InsIdx), ExtIdx)
13570       // Into:
13571       //    indices are equal or bit offsets are equal => V1
13572       //    otherwise => (extract_subvec V1, ExtIdx)
13573       if (InsIdx->getZExtValue() * SmallVT.getScalarSizeInBits() ==
13574           ExtIdx->getZExtValue() * NVT.getScalarSizeInBits())
13575         return DAG.getBitcast(NVT, V->getOperand(1));
13576       return DAG.getNode(
13577           ISD::EXTRACT_SUBVECTOR, SDLoc(N), NVT,
13578           DAG.getBitcast(N->getOperand(0).getValueType(), V->getOperand(0)),
13579           N->getOperand(1));
13580     }
13581   }
13582 
13583   return SDValue();
13584 }
13585 
13586 static SDValue simplifyShuffleOperandRecursively(SmallBitVector &UsedElements,
13587                                                  SDValue V, SelectionDAG &DAG) {
13588   SDLoc DL(V);
13589   EVT VT = V.getValueType();
13590 
13591   switch (V.getOpcode()) {
13592   default:
13593     return V;
13594 
13595   case ISD::CONCAT_VECTORS: {
13596     EVT OpVT = V->getOperand(0).getValueType();
13597     int OpSize = OpVT.getVectorNumElements();
13598     SmallBitVector OpUsedElements(OpSize, false);
13599     bool FoundSimplification = false;
13600     SmallVector<SDValue, 4> NewOps;
13601     NewOps.reserve(V->getNumOperands());
13602     for (int i = 0, NumOps = V->getNumOperands(); i < NumOps; ++i) {
13603       SDValue Op = V->getOperand(i);
13604       bool OpUsed = false;
13605       for (int j = 0; j < OpSize; ++j)
13606         if (UsedElements[i * OpSize + j]) {
13607           OpUsedElements[j] = true;
13608           OpUsed = true;
13609         }
13610       NewOps.push_back(
13611           OpUsed ? simplifyShuffleOperandRecursively(OpUsedElements, Op, DAG)
13612                  : DAG.getUNDEF(OpVT));
13613       FoundSimplification |= Op == NewOps.back();
13614       OpUsedElements.reset();
13615     }
13616     if (FoundSimplification)
13617       V = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, NewOps);
13618     return V;
13619   }
13620 
13621   case ISD::INSERT_SUBVECTOR: {
13622     SDValue BaseV = V->getOperand(0);
13623     SDValue SubV = V->getOperand(1);
13624     auto *IdxN = dyn_cast<ConstantSDNode>(V->getOperand(2));
13625     if (!IdxN)
13626       return V;
13627 
13628     int SubSize = SubV.getValueType().getVectorNumElements();
13629     int Idx = IdxN->getZExtValue();
13630     bool SubVectorUsed = false;
13631     SmallBitVector SubUsedElements(SubSize, false);
13632     for (int i = 0; i < SubSize; ++i)
13633       if (UsedElements[i + Idx]) {
13634         SubVectorUsed = true;
13635         SubUsedElements[i] = true;
13636         UsedElements[i + Idx] = false;
13637       }
13638 
13639     // Now recurse on both the base and sub vectors.
13640     SDValue SimplifiedSubV =
13641         SubVectorUsed
13642             ? simplifyShuffleOperandRecursively(SubUsedElements, SubV, DAG)
13643             : DAG.getUNDEF(SubV.getValueType());
13644     SDValue SimplifiedBaseV = simplifyShuffleOperandRecursively(UsedElements, BaseV, DAG);
13645     if (SimplifiedSubV != SubV || SimplifiedBaseV != BaseV)
13646       V = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VT,
13647                       SimplifiedBaseV, SimplifiedSubV, V->getOperand(2));
13648     return V;
13649   }
13650   }
13651 }
13652 
13653 static SDValue simplifyShuffleOperands(ShuffleVectorSDNode *SVN, SDValue N0,
13654                                        SDValue N1, SelectionDAG &DAG) {
13655   EVT VT = SVN->getValueType(0);
13656   int NumElts = VT.getVectorNumElements();
13657   SmallBitVector N0UsedElements(NumElts, false), N1UsedElements(NumElts, false);
13658   for (int M : SVN->getMask())
13659     if (M >= 0 && M < NumElts)
13660       N0UsedElements[M] = true;
13661     else if (M >= NumElts)
13662       N1UsedElements[M - NumElts] = true;
13663 
13664   SDValue S0 = simplifyShuffleOperandRecursively(N0UsedElements, N0, DAG);
13665   SDValue S1 = simplifyShuffleOperandRecursively(N1UsedElements, N1, DAG);
13666   if (S0 == N0 && S1 == N1)
13667     return SDValue();
13668 
13669   return DAG.getVectorShuffle(VT, SDLoc(SVN), S0, S1, SVN->getMask());
13670 }
13671 
13672 // Tries to turn a shuffle of two CONCAT_VECTORS into a single concat,
13673 // or turn a shuffle of a single concat into simpler shuffle then concat.
13674 static SDValue partitionShuffleOfConcats(SDNode *N, SelectionDAG &DAG) {
13675   EVT VT = N->getValueType(0);
13676   unsigned NumElts = VT.getVectorNumElements();
13677 
13678   SDValue N0 = N->getOperand(0);
13679   SDValue N1 = N->getOperand(1);
13680   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
13681 
13682   SmallVector<SDValue, 4> Ops;
13683   EVT ConcatVT = N0.getOperand(0).getValueType();
13684   unsigned NumElemsPerConcat = ConcatVT.getVectorNumElements();
13685   unsigned NumConcats = NumElts / NumElemsPerConcat;
13686 
13687   // Special case: shuffle(concat(A,B)) can be more efficiently represented
13688   // as concat(shuffle(A,B),UNDEF) if the shuffle doesn't set any of the high
13689   // half vector elements.
13690   if (NumElemsPerConcat * 2 == NumElts && N1.isUndef() &&
13691       std::all_of(SVN->getMask().begin() + NumElemsPerConcat,
13692                   SVN->getMask().end(), [](int i) { return i == -1; })) {
13693     N0 = DAG.getVectorShuffle(ConcatVT, SDLoc(N), N0.getOperand(0), N0.getOperand(1),
13694                               makeArrayRef(SVN->getMask().begin(), NumElemsPerConcat));
13695     N1 = DAG.getUNDEF(ConcatVT);
13696     return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N0, N1);
13697   }
13698 
13699   // Look at every vector that's inserted. We're looking for exact
13700   // subvector-sized copies from a concatenated vector
13701   for (unsigned I = 0; I != NumConcats; ++I) {
13702     // Make sure we're dealing with a copy.
13703     unsigned Begin = I * NumElemsPerConcat;
13704     bool AllUndef = true, NoUndef = true;
13705     for (unsigned J = Begin; J != Begin + NumElemsPerConcat; ++J) {
13706       if (SVN->getMaskElt(J) >= 0)
13707         AllUndef = false;
13708       else
13709         NoUndef = false;
13710     }
13711 
13712     if (NoUndef) {
13713       if (SVN->getMaskElt(Begin) % NumElemsPerConcat != 0)
13714         return SDValue();
13715 
13716       for (unsigned J = 1; J != NumElemsPerConcat; ++J)
13717         if (SVN->getMaskElt(Begin + J - 1) + 1 != SVN->getMaskElt(Begin + J))
13718           return SDValue();
13719 
13720       unsigned FirstElt = SVN->getMaskElt(Begin) / NumElemsPerConcat;
13721       if (FirstElt < N0.getNumOperands())
13722         Ops.push_back(N0.getOperand(FirstElt));
13723       else
13724         Ops.push_back(N1.getOperand(FirstElt - N0.getNumOperands()));
13725 
13726     } else if (AllUndef) {
13727       Ops.push_back(DAG.getUNDEF(N0.getOperand(0).getValueType()));
13728     } else { // Mixed with general masks and undefs, can't do optimization.
13729       return SDValue();
13730     }
13731   }
13732 
13733   return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops);
13734 }
13735 
13736 SDValue DAGCombiner::visitVECTOR_SHUFFLE(SDNode *N) {
13737   EVT VT = N->getValueType(0);
13738   unsigned NumElts = VT.getVectorNumElements();
13739 
13740   SDValue N0 = N->getOperand(0);
13741   SDValue N1 = N->getOperand(1);
13742 
13743   assert(N0.getValueType() == VT && "Vector shuffle must be normalized in DAG");
13744 
13745   // Canonicalize shuffle undef, undef -> undef
13746   if (N0.isUndef() && N1.isUndef())
13747     return DAG.getUNDEF(VT);
13748 
13749   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
13750 
13751   // Canonicalize shuffle v, v -> v, undef
13752   if (N0 == N1) {
13753     SmallVector<int, 8> NewMask;
13754     for (unsigned i = 0; i != NumElts; ++i) {
13755       int Idx = SVN->getMaskElt(i);
13756       if (Idx >= (int)NumElts) Idx -= NumElts;
13757       NewMask.push_back(Idx);
13758     }
13759     return DAG.getVectorShuffle(VT, SDLoc(N), N0, DAG.getUNDEF(VT), NewMask);
13760   }
13761 
13762   // Canonicalize shuffle undef, v -> v, undef.  Commute the shuffle mask.
13763   if (N0.isUndef())
13764     return DAG.getCommutedVectorShuffle(*SVN);
13765 
13766   // Remove references to rhs if it is undef
13767   if (N1.isUndef()) {
13768     bool Changed = false;
13769     SmallVector<int, 8> NewMask;
13770     for (unsigned i = 0; i != NumElts; ++i) {
13771       int Idx = SVN->getMaskElt(i);
13772       if (Idx >= (int)NumElts) {
13773         Idx = -1;
13774         Changed = true;
13775       }
13776       NewMask.push_back(Idx);
13777     }
13778     if (Changed)
13779       return DAG.getVectorShuffle(VT, SDLoc(N), N0, N1, NewMask);
13780   }
13781 
13782   // If it is a splat, check if the argument vector is another splat or a
13783   // build_vector.
13784   if (SVN->isSplat() && SVN->getSplatIndex() < (int)NumElts) {
13785     SDNode *V = N0.getNode();
13786 
13787     // If this is a bit convert that changes the element type of the vector but
13788     // not the number of vector elements, look through it.  Be careful not to
13789     // look though conversions that change things like v4f32 to v2f64.
13790     if (V->getOpcode() == ISD::BITCAST) {
13791       SDValue ConvInput = V->getOperand(0);
13792       if (ConvInput.getValueType().isVector() &&
13793           ConvInput.getValueType().getVectorNumElements() == NumElts)
13794         V = ConvInput.getNode();
13795     }
13796 
13797     if (V->getOpcode() == ISD::BUILD_VECTOR) {
13798       assert(V->getNumOperands() == NumElts &&
13799              "BUILD_VECTOR has wrong number of operands");
13800       SDValue Base;
13801       bool AllSame = true;
13802       for (unsigned i = 0; i != NumElts; ++i) {
13803         if (!V->getOperand(i).isUndef()) {
13804           Base = V->getOperand(i);
13805           break;
13806         }
13807       }
13808       // Splat of <u, u, u, u>, return <u, u, u, u>
13809       if (!Base.getNode())
13810         return N0;
13811       for (unsigned i = 0; i != NumElts; ++i) {
13812         if (V->getOperand(i) != Base) {
13813           AllSame = false;
13814           break;
13815         }
13816       }
13817       // Splat of <x, x, x, x>, return <x, x, x, x>
13818       if (AllSame)
13819         return N0;
13820 
13821       // Canonicalize any other splat as a build_vector.
13822       const SDValue &Splatted = V->getOperand(SVN->getSplatIndex());
13823       SmallVector<SDValue, 8> Ops(NumElts, Splatted);
13824       SDValue NewBV = DAG.getBuildVector(V->getValueType(0), SDLoc(N), Ops);
13825 
13826       // We may have jumped through bitcasts, so the type of the
13827       // BUILD_VECTOR may not match the type of the shuffle.
13828       if (V->getValueType(0) != VT)
13829         NewBV = DAG.getBitcast(VT, NewBV);
13830       return NewBV;
13831     }
13832   }
13833 
13834   // There are various patterns used to build up a vector from smaller vectors,
13835   // subvectors, or elements. Scan chains of these and replace unused insertions
13836   // or components with undef.
13837   if (SDValue S = simplifyShuffleOperands(SVN, N0, N1, DAG))
13838     return S;
13839 
13840   if (N0.getOpcode() == ISD::CONCAT_VECTORS &&
13841       Level < AfterLegalizeVectorOps &&
13842       (N1.isUndef() ||
13843       (N1.getOpcode() == ISD::CONCAT_VECTORS &&
13844        N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType()))) {
13845     if (SDValue V = partitionShuffleOfConcats(N, DAG))
13846       return V;
13847   }
13848 
13849   // Attempt to combine a shuffle of 2 inputs of 'scalar sources' -
13850   // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR.
13851   if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT)) {
13852     SmallVector<SDValue, 8> Ops;
13853     for (int M : SVN->getMask()) {
13854       SDValue Op = DAG.getUNDEF(VT.getScalarType());
13855       if (M >= 0) {
13856         int Idx = M % NumElts;
13857         SDValue &S = (M < (int)NumElts ? N0 : N1);
13858         if (S.getOpcode() == ISD::BUILD_VECTOR && S.hasOneUse()) {
13859           Op = S.getOperand(Idx);
13860         } else if (S.getOpcode() == ISD::SCALAR_TO_VECTOR && S.hasOneUse()) {
13861           if (Idx == 0)
13862             Op = S.getOperand(0);
13863         } else {
13864           // Operand can't be combined - bail out.
13865           break;
13866         }
13867       }
13868       Ops.push_back(Op);
13869     }
13870     if (Ops.size() == VT.getVectorNumElements()) {
13871       // BUILD_VECTOR requires all inputs to be of the same type, find the
13872       // maximum type and extend them all.
13873       EVT SVT = VT.getScalarType();
13874       if (SVT.isInteger())
13875         for (SDValue &Op : Ops)
13876           SVT = (SVT.bitsLT(Op.getValueType()) ? Op.getValueType() : SVT);
13877       if (SVT != VT.getScalarType())
13878         for (SDValue &Op : Ops)
13879           Op = TLI.isZExtFree(Op.getValueType(), SVT)
13880                    ? DAG.getZExtOrTrunc(Op, SDLoc(N), SVT)
13881                    : DAG.getSExtOrTrunc(Op, SDLoc(N), SVT);
13882       return DAG.getBuildVector(VT, SDLoc(N), Ops);
13883     }
13884   }
13885 
13886   // If this shuffle only has a single input that is a bitcasted shuffle,
13887   // attempt to merge the 2 shuffles and suitably bitcast the inputs/output
13888   // back to their original types.
13889   if (N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
13890       N1.isUndef() && Level < AfterLegalizeVectorOps &&
13891       TLI.isTypeLegal(VT)) {
13892 
13893     // Peek through the bitcast only if there is one user.
13894     SDValue BC0 = N0;
13895     while (BC0.getOpcode() == ISD::BITCAST) {
13896       if (!BC0.hasOneUse())
13897         break;
13898       BC0 = BC0.getOperand(0);
13899     }
13900 
13901     auto ScaleShuffleMask = [](ArrayRef<int> Mask, int Scale) {
13902       if (Scale == 1)
13903         return SmallVector<int, 8>(Mask.begin(), Mask.end());
13904 
13905       SmallVector<int, 8> NewMask;
13906       for (int M : Mask)
13907         for (int s = 0; s != Scale; ++s)
13908           NewMask.push_back(M < 0 ? -1 : Scale * M + s);
13909       return NewMask;
13910     };
13911 
13912     if (BC0.getOpcode() == ISD::VECTOR_SHUFFLE && BC0.hasOneUse()) {
13913       EVT SVT = VT.getScalarType();
13914       EVT InnerVT = BC0->getValueType(0);
13915       EVT InnerSVT = InnerVT.getScalarType();
13916 
13917       // Determine which shuffle works with the smaller scalar type.
13918       EVT ScaleVT = SVT.bitsLT(InnerSVT) ? VT : InnerVT;
13919       EVT ScaleSVT = ScaleVT.getScalarType();
13920 
13921       if (TLI.isTypeLegal(ScaleVT) &&
13922           0 == (InnerSVT.getSizeInBits() % ScaleSVT.getSizeInBits()) &&
13923           0 == (SVT.getSizeInBits() % ScaleSVT.getSizeInBits())) {
13924 
13925         int InnerScale = InnerSVT.getSizeInBits() / ScaleSVT.getSizeInBits();
13926         int OuterScale = SVT.getSizeInBits() / ScaleSVT.getSizeInBits();
13927 
13928         // Scale the shuffle masks to the smaller scalar type.
13929         ShuffleVectorSDNode *InnerSVN = cast<ShuffleVectorSDNode>(BC0);
13930         SmallVector<int, 8> InnerMask =
13931             ScaleShuffleMask(InnerSVN->getMask(), InnerScale);
13932         SmallVector<int, 8> OuterMask =
13933             ScaleShuffleMask(SVN->getMask(), OuterScale);
13934 
13935         // Merge the shuffle masks.
13936         SmallVector<int, 8> NewMask;
13937         for (int M : OuterMask)
13938           NewMask.push_back(M < 0 ? -1 : InnerMask[M]);
13939 
13940         // Test for shuffle mask legality over both commutations.
13941         SDValue SV0 = BC0->getOperand(0);
13942         SDValue SV1 = BC0->getOperand(1);
13943         bool LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
13944         if (!LegalMask) {
13945           std::swap(SV0, SV1);
13946           ShuffleVectorSDNode::commuteMask(NewMask);
13947           LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
13948         }
13949 
13950         if (LegalMask) {
13951           SV0 = DAG.getBitcast(ScaleVT, SV0);
13952           SV1 = DAG.getBitcast(ScaleVT, SV1);
13953           return DAG.getBitcast(
13954               VT, DAG.getVectorShuffle(ScaleVT, SDLoc(N), SV0, SV1, NewMask));
13955         }
13956       }
13957     }
13958   }
13959 
13960   // Canonicalize shuffles according to rules:
13961   //  shuffle(A, shuffle(A, B)) -> shuffle(shuffle(A,B), A)
13962   //  shuffle(B, shuffle(A, B)) -> shuffle(shuffle(A,B), B)
13963   //  shuffle(B, shuffle(A, Undef)) -> shuffle(shuffle(A, Undef), B)
13964   if (N1.getOpcode() == ISD::VECTOR_SHUFFLE &&
13965       N0.getOpcode() != ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG &&
13966       TLI.isTypeLegal(VT)) {
13967     // The incoming shuffle must be of the same type as the result of the
13968     // current shuffle.
13969     assert(N1->getOperand(0).getValueType() == VT &&
13970            "Shuffle types don't match");
13971 
13972     SDValue SV0 = N1->getOperand(0);
13973     SDValue SV1 = N1->getOperand(1);
13974     bool HasSameOp0 = N0 == SV0;
13975     bool IsSV1Undef = SV1.isUndef();
13976     if (HasSameOp0 || IsSV1Undef || N0 == SV1)
13977       // Commute the operands of this shuffle so that next rule
13978       // will trigger.
13979       return DAG.getCommutedVectorShuffle(*SVN);
13980   }
13981 
13982   // Try to fold according to rules:
13983   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
13984   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
13985   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
13986   // Don't try to fold shuffles with illegal type.
13987   // Only fold if this shuffle is the only user of the other shuffle.
13988   if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && N->isOnlyUserOf(N0.getNode()) &&
13989       Level < AfterLegalizeDAG && TLI.isTypeLegal(VT)) {
13990     ShuffleVectorSDNode *OtherSV = cast<ShuffleVectorSDNode>(N0);
13991 
13992     // The incoming shuffle must be of the same type as the result of the
13993     // current shuffle.
13994     assert(OtherSV->getOperand(0).getValueType() == VT &&
13995            "Shuffle types don't match");
13996 
13997     SDValue SV0, SV1;
13998     SmallVector<int, 4> Mask;
13999     // Compute the combined shuffle mask for a shuffle with SV0 as the first
14000     // operand, and SV1 as the second operand.
14001     for (unsigned i = 0; i != NumElts; ++i) {
14002       int Idx = SVN->getMaskElt(i);
14003       if (Idx < 0) {
14004         // Propagate Undef.
14005         Mask.push_back(Idx);
14006         continue;
14007       }
14008 
14009       SDValue CurrentVec;
14010       if (Idx < (int)NumElts) {
14011         // This shuffle index refers to the inner shuffle N0. Lookup the inner
14012         // shuffle mask to identify which vector is actually referenced.
14013         Idx = OtherSV->getMaskElt(Idx);
14014         if (Idx < 0) {
14015           // Propagate Undef.
14016           Mask.push_back(Idx);
14017           continue;
14018         }
14019 
14020         CurrentVec = (Idx < (int) NumElts) ? OtherSV->getOperand(0)
14021                                            : OtherSV->getOperand(1);
14022       } else {
14023         // This shuffle index references an element within N1.
14024         CurrentVec = N1;
14025       }
14026 
14027       // Simple case where 'CurrentVec' is UNDEF.
14028       if (CurrentVec.isUndef()) {
14029         Mask.push_back(-1);
14030         continue;
14031       }
14032 
14033       // Canonicalize the shuffle index. We don't know yet if CurrentVec
14034       // will be the first or second operand of the combined shuffle.
14035       Idx = Idx % NumElts;
14036       if (!SV0.getNode() || SV0 == CurrentVec) {
14037         // Ok. CurrentVec is the left hand side.
14038         // Update the mask accordingly.
14039         SV0 = CurrentVec;
14040         Mask.push_back(Idx);
14041         continue;
14042       }
14043 
14044       // Bail out if we cannot convert the shuffle pair into a single shuffle.
14045       if (SV1.getNode() && SV1 != CurrentVec)
14046         return SDValue();
14047 
14048       // Ok. CurrentVec is the right hand side.
14049       // Update the mask accordingly.
14050       SV1 = CurrentVec;
14051       Mask.push_back(Idx + NumElts);
14052     }
14053 
14054     // Check if all indices in Mask are Undef. In case, propagate Undef.
14055     bool isUndefMask = true;
14056     for (unsigned i = 0; i != NumElts && isUndefMask; ++i)
14057       isUndefMask &= Mask[i] < 0;
14058 
14059     if (isUndefMask)
14060       return DAG.getUNDEF(VT);
14061 
14062     if (!SV0.getNode())
14063       SV0 = DAG.getUNDEF(VT);
14064     if (!SV1.getNode())
14065       SV1 = DAG.getUNDEF(VT);
14066 
14067     // Avoid introducing shuffles with illegal mask.
14068     if (!TLI.isShuffleMaskLegal(Mask, VT)) {
14069       ShuffleVectorSDNode::commuteMask(Mask);
14070 
14071       if (!TLI.isShuffleMaskLegal(Mask, VT))
14072         return SDValue();
14073 
14074       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, A, M2)
14075       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, A, M2)
14076       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, B, M2)
14077       std::swap(SV0, SV1);
14078     }
14079 
14080     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
14081     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
14082     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
14083     return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, Mask);
14084   }
14085 
14086   return SDValue();
14087 }
14088 
14089 SDValue DAGCombiner::visitSCALAR_TO_VECTOR(SDNode *N) {
14090   SDValue InVal = N->getOperand(0);
14091   EVT VT = N->getValueType(0);
14092 
14093   // Replace a SCALAR_TO_VECTOR(EXTRACT_VECTOR_ELT(V,C0)) pattern
14094   // with a VECTOR_SHUFFLE.
14095   if (InVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
14096     SDValue InVec = InVal->getOperand(0);
14097     SDValue EltNo = InVal->getOperand(1);
14098 
14099     // FIXME: We could support implicit truncation if the shuffle can be
14100     // scaled to a smaller vector scalar type.
14101     ConstantSDNode *C0 = dyn_cast<ConstantSDNode>(EltNo);
14102     if (C0 && VT == InVec.getValueType() &&
14103         VT.getScalarType() == InVal.getValueType()) {
14104       SmallVector<int, 8> NewMask(VT.getVectorNumElements(), -1);
14105       int Elt = C0->getZExtValue();
14106       NewMask[0] = Elt;
14107 
14108       if (TLI.isShuffleMaskLegal(NewMask, VT))
14109         return DAG.getVectorShuffle(VT, SDLoc(N), InVec, DAG.getUNDEF(VT),
14110                                     NewMask);
14111     }
14112   }
14113 
14114   return SDValue();
14115 }
14116 
14117 SDValue DAGCombiner::visitINSERT_SUBVECTOR(SDNode *N) {
14118   EVT VT = N->getValueType(0);
14119   SDValue N0 = N->getOperand(0);
14120   SDValue N1 = N->getOperand(1);
14121   SDValue N2 = N->getOperand(2);
14122 
14123   // Combine INSERT_SUBVECTORs where we are inserting to the same index.
14124   // INSERT_SUBVECTOR( INSERT_SUBVECTOR( Vec, SubOld, Idx ), SubNew, Idx )
14125   // --> INSERT_SUBVECTOR( Vec, SubNew, Idx )
14126   if (N0.getOpcode() == ISD::INSERT_SUBVECTOR &&
14127       N0.getOperand(1).getValueType() == N1.getValueType() &&
14128       N0.getOperand(2) == N2)
14129     return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0.getOperand(0),
14130                        N1, N2);
14131 
14132   if (N0.getValueType() != N1.getValueType())
14133     return SDValue();
14134 
14135   // If the input vector is a concatenation, and the insert replaces
14136   // one of the halves, we can optimize into a single concat_vectors.
14137   if (N0.getOpcode() == ISD::CONCAT_VECTORS && N0->getNumOperands() == 2 &&
14138       N2.getOpcode() == ISD::Constant) {
14139     APInt InsIdx = cast<ConstantSDNode>(N2)->getAPIntValue();
14140 
14141     // Lower half: fold (insert_subvector (concat_vectors X, Y), Z) ->
14142     // (concat_vectors Z, Y)
14143     if (InsIdx == 0)
14144       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N1,
14145                          N0.getOperand(1));
14146 
14147     // Upper half: fold (insert_subvector (concat_vectors X, Y), Z) ->
14148     // (concat_vectors X, Z)
14149     if (InsIdx == VT.getVectorNumElements() / 2)
14150       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N0.getOperand(0),
14151                          N1);
14152   }
14153 
14154   return SDValue();
14155 }
14156 
14157 SDValue DAGCombiner::visitFP_TO_FP16(SDNode *N) {
14158   SDValue N0 = N->getOperand(0);
14159 
14160   // fold (fp_to_fp16 (fp16_to_fp op)) -> op
14161   if (N0->getOpcode() == ISD::FP16_TO_FP)
14162     return N0->getOperand(0);
14163 
14164   return SDValue();
14165 }
14166 
14167 SDValue DAGCombiner::visitFP16_TO_FP(SDNode *N) {
14168   SDValue N0 = N->getOperand(0);
14169 
14170   // fold fp16_to_fp(op & 0xffff) -> fp16_to_fp(op)
14171   if (N0->getOpcode() == ISD::AND) {
14172     ConstantSDNode *AndConst = getAsNonOpaqueConstant(N0.getOperand(1));
14173     if (AndConst && AndConst->getAPIntValue() == 0xffff) {
14174       return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), N->getValueType(0),
14175                          N0.getOperand(0));
14176     }
14177   }
14178 
14179   return SDValue();
14180 }
14181 
14182 /// Returns a vector_shuffle if it able to transform an AND to a vector_shuffle
14183 /// with the destination vector and a zero vector.
14184 /// e.g. AND V, <0xffffffff, 0, 0xffffffff, 0>. ==>
14185 ///      vector_shuffle V, Zero, <0, 4, 2, 4>
14186 SDValue DAGCombiner::XformToShuffleWithZero(SDNode *N) {
14187   EVT VT = N->getValueType(0);
14188   SDValue LHS = N->getOperand(0);
14189   SDValue RHS = N->getOperand(1);
14190   SDLoc DL(N);
14191 
14192   // Make sure we're not running after operation legalization where it
14193   // may have custom lowered the vector shuffles.
14194   if (LegalOperations)
14195     return SDValue();
14196 
14197   if (N->getOpcode() != ISD::AND)
14198     return SDValue();
14199 
14200   if (RHS.getOpcode() == ISD::BITCAST)
14201     RHS = RHS.getOperand(0);
14202 
14203   if (RHS.getOpcode() != ISD::BUILD_VECTOR)
14204     return SDValue();
14205 
14206   EVT RVT = RHS.getValueType();
14207   unsigned NumElts = RHS.getNumOperands();
14208 
14209   // Attempt to create a valid clear mask, splitting the mask into
14210   // sub elements and checking to see if each is
14211   // all zeros or all ones - suitable for shuffle masking.
14212   auto BuildClearMask = [&](int Split) {
14213     int NumSubElts = NumElts * Split;
14214     int NumSubBits = RVT.getScalarSizeInBits() / Split;
14215 
14216     SmallVector<int, 8> Indices;
14217     for (int i = 0; i != NumSubElts; ++i) {
14218       int EltIdx = i / Split;
14219       int SubIdx = i % Split;
14220       SDValue Elt = RHS.getOperand(EltIdx);
14221       if (Elt.isUndef()) {
14222         Indices.push_back(-1);
14223         continue;
14224       }
14225 
14226       APInt Bits;
14227       if (isa<ConstantSDNode>(Elt))
14228         Bits = cast<ConstantSDNode>(Elt)->getAPIntValue();
14229       else if (isa<ConstantFPSDNode>(Elt))
14230         Bits = cast<ConstantFPSDNode>(Elt)->getValueAPF().bitcastToAPInt();
14231       else
14232         return SDValue();
14233 
14234       // Extract the sub element from the constant bit mask.
14235       if (DAG.getDataLayout().isBigEndian()) {
14236         Bits = Bits.lshr((Split - SubIdx - 1) * NumSubBits);
14237       } else {
14238         Bits = Bits.lshr(SubIdx * NumSubBits);
14239       }
14240 
14241       if (Split > 1)
14242         Bits = Bits.trunc(NumSubBits);
14243 
14244       if (Bits.isAllOnesValue())
14245         Indices.push_back(i);
14246       else if (Bits == 0)
14247         Indices.push_back(i + NumSubElts);
14248       else
14249         return SDValue();
14250     }
14251 
14252     // Let's see if the target supports this vector_shuffle.
14253     EVT ClearSVT = EVT::getIntegerVT(*DAG.getContext(), NumSubBits);
14254     EVT ClearVT = EVT::getVectorVT(*DAG.getContext(), ClearSVT, NumSubElts);
14255     if (!TLI.isVectorClearMaskLegal(Indices, ClearVT))
14256       return SDValue();
14257 
14258     SDValue Zero = DAG.getConstant(0, DL, ClearVT);
14259     return DAG.getBitcast(VT, DAG.getVectorShuffle(ClearVT, DL,
14260                                                    DAG.getBitcast(ClearVT, LHS),
14261                                                    Zero, Indices));
14262   };
14263 
14264   // Determine maximum split level (byte level masking).
14265   int MaxSplit = 1;
14266   if (RVT.getScalarSizeInBits() % 8 == 0)
14267     MaxSplit = RVT.getScalarSizeInBits() / 8;
14268 
14269   for (int Split = 1; Split <= MaxSplit; ++Split)
14270     if (RVT.getScalarSizeInBits() % Split == 0)
14271       if (SDValue S = BuildClearMask(Split))
14272         return S;
14273 
14274   return SDValue();
14275 }
14276 
14277 /// Visit a binary vector operation, like ADD.
14278 SDValue DAGCombiner::SimplifyVBinOp(SDNode *N) {
14279   assert(N->getValueType(0).isVector() &&
14280          "SimplifyVBinOp only works on vectors!");
14281 
14282   SDValue LHS = N->getOperand(0);
14283   SDValue RHS = N->getOperand(1);
14284   SDValue Ops[] = {LHS, RHS};
14285 
14286   // See if we can constant fold the vector operation.
14287   if (SDValue Fold = DAG.FoldConstantVectorArithmetic(
14288           N->getOpcode(), SDLoc(LHS), LHS.getValueType(), Ops, N->getFlags()))
14289     return Fold;
14290 
14291   // Try to convert a constant mask AND into a shuffle clear mask.
14292   if (SDValue Shuffle = XformToShuffleWithZero(N))
14293     return Shuffle;
14294 
14295   // Type legalization might introduce new shuffles in the DAG.
14296   // Fold (VBinOp (shuffle (A, Undef, Mask)), (shuffle (B, Undef, Mask)))
14297   //   -> (shuffle (VBinOp (A, B)), Undef, Mask).
14298   if (LegalTypes && isa<ShuffleVectorSDNode>(LHS) &&
14299       isa<ShuffleVectorSDNode>(RHS) && LHS.hasOneUse() && RHS.hasOneUse() &&
14300       LHS.getOperand(1).isUndef() &&
14301       RHS.getOperand(1).isUndef()) {
14302     ShuffleVectorSDNode *SVN0 = cast<ShuffleVectorSDNode>(LHS);
14303     ShuffleVectorSDNode *SVN1 = cast<ShuffleVectorSDNode>(RHS);
14304 
14305     if (SVN0->getMask().equals(SVN1->getMask())) {
14306       EVT VT = N->getValueType(0);
14307       SDValue UndefVector = LHS.getOperand(1);
14308       SDValue NewBinOp = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
14309                                      LHS.getOperand(0), RHS.getOperand(0),
14310                                      N->getFlags());
14311       AddUsersToWorklist(N);
14312       return DAG.getVectorShuffle(VT, SDLoc(N), NewBinOp, UndefVector,
14313                                   SVN0->getMask());
14314     }
14315   }
14316 
14317   return SDValue();
14318 }
14319 
14320 SDValue DAGCombiner::SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1,
14321                                     SDValue N2) {
14322   assert(N0.getOpcode() ==ISD::SETCC && "First argument must be a SetCC node!");
14323 
14324   SDValue SCC = SimplifySelectCC(DL, N0.getOperand(0), N0.getOperand(1), N1, N2,
14325                                  cast<CondCodeSDNode>(N0.getOperand(2))->get());
14326 
14327   // If we got a simplified select_cc node back from SimplifySelectCC, then
14328   // break it down into a new SETCC node, and a new SELECT node, and then return
14329   // the SELECT node, since we were called with a SELECT node.
14330   if (SCC.getNode()) {
14331     // Check to see if we got a select_cc back (to turn into setcc/select).
14332     // Otherwise, just return whatever node we got back, like fabs.
14333     if (SCC.getOpcode() == ISD::SELECT_CC) {
14334       SDValue SETCC = DAG.getNode(ISD::SETCC, SDLoc(N0),
14335                                   N0.getValueType(),
14336                                   SCC.getOperand(0), SCC.getOperand(1),
14337                                   SCC.getOperand(4));
14338       AddToWorklist(SETCC.getNode());
14339       return DAG.getSelect(SDLoc(SCC), SCC.getValueType(), SETCC,
14340                            SCC.getOperand(2), SCC.getOperand(3));
14341     }
14342 
14343     return SCC;
14344   }
14345   return SDValue();
14346 }
14347 
14348 /// Given a SELECT or a SELECT_CC node, where LHS and RHS are the two values
14349 /// being selected between, see if we can simplify the select.  Callers of this
14350 /// should assume that TheSelect is deleted if this returns true.  As such, they
14351 /// should return the appropriate thing (e.g. the node) back to the top-level of
14352 /// the DAG combiner loop to avoid it being looked at.
14353 bool DAGCombiner::SimplifySelectOps(SDNode *TheSelect, SDValue LHS,
14354                                     SDValue RHS) {
14355 
14356   // fold (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
14357   // The select + setcc is redundant, because fsqrt returns NaN for X < 0.
14358   if (const ConstantFPSDNode *NaN = isConstOrConstSplatFP(LHS)) {
14359     if (NaN->isNaN() && RHS.getOpcode() == ISD::FSQRT) {
14360       // We have: (select (setcc ?, ?, ?), NaN, (fsqrt ?))
14361       SDValue Sqrt = RHS;
14362       ISD::CondCode CC;
14363       SDValue CmpLHS;
14364       const ConstantFPSDNode *Zero = nullptr;
14365 
14366       if (TheSelect->getOpcode() == ISD::SELECT_CC) {
14367         CC = dyn_cast<CondCodeSDNode>(TheSelect->getOperand(4))->get();
14368         CmpLHS = TheSelect->getOperand(0);
14369         Zero = isConstOrConstSplatFP(TheSelect->getOperand(1));
14370       } else {
14371         // SELECT or VSELECT
14372         SDValue Cmp = TheSelect->getOperand(0);
14373         if (Cmp.getOpcode() == ISD::SETCC) {
14374           CC = dyn_cast<CondCodeSDNode>(Cmp.getOperand(2))->get();
14375           CmpLHS = Cmp.getOperand(0);
14376           Zero = isConstOrConstSplatFP(Cmp.getOperand(1));
14377         }
14378       }
14379       if (Zero && Zero->isZero() &&
14380           Sqrt.getOperand(0) == CmpLHS && (CC == ISD::SETOLT ||
14381           CC == ISD::SETULT || CC == ISD::SETLT)) {
14382         // We have: (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
14383         CombineTo(TheSelect, Sqrt);
14384         return true;
14385       }
14386     }
14387   }
14388   // Cannot simplify select with vector condition
14389   if (TheSelect->getOperand(0).getValueType().isVector()) return false;
14390 
14391   // If this is a select from two identical things, try to pull the operation
14392   // through the select.
14393   if (LHS.getOpcode() != RHS.getOpcode() ||
14394       !LHS.hasOneUse() || !RHS.hasOneUse())
14395     return false;
14396 
14397   // If this is a load and the token chain is identical, replace the select
14398   // of two loads with a load through a select of the address to load from.
14399   // This triggers in things like "select bool X, 10.0, 123.0" after the FP
14400   // constants have been dropped into the constant pool.
14401   if (LHS.getOpcode() == ISD::LOAD) {
14402     LoadSDNode *LLD = cast<LoadSDNode>(LHS);
14403     LoadSDNode *RLD = cast<LoadSDNode>(RHS);
14404 
14405     // Token chains must be identical.
14406     if (LHS.getOperand(0) != RHS.getOperand(0) ||
14407         // Do not let this transformation reduce the number of volatile loads.
14408         LLD->isVolatile() || RLD->isVolatile() ||
14409         // FIXME: If either is a pre/post inc/dec load,
14410         // we'd need to split out the address adjustment.
14411         LLD->isIndexed() || RLD->isIndexed() ||
14412         // If this is an EXTLOAD, the VT's must match.
14413         LLD->getMemoryVT() != RLD->getMemoryVT() ||
14414         // If this is an EXTLOAD, the kind of extension must match.
14415         (LLD->getExtensionType() != RLD->getExtensionType() &&
14416          // The only exception is if one of the extensions is anyext.
14417          LLD->getExtensionType() != ISD::EXTLOAD &&
14418          RLD->getExtensionType() != ISD::EXTLOAD) ||
14419         // FIXME: this discards src value information.  This is
14420         // over-conservative. It would be beneficial to be able to remember
14421         // both potential memory locations.  Since we are discarding
14422         // src value info, don't do the transformation if the memory
14423         // locations are not in the default address space.
14424         LLD->getPointerInfo().getAddrSpace() != 0 ||
14425         RLD->getPointerInfo().getAddrSpace() != 0 ||
14426         !TLI.isOperationLegalOrCustom(TheSelect->getOpcode(),
14427                                       LLD->getBasePtr().getValueType()))
14428       return false;
14429 
14430     // Check that the select condition doesn't reach either load.  If so,
14431     // folding this will induce a cycle into the DAG.  If not, this is safe to
14432     // xform, so create a select of the addresses.
14433     SDValue Addr;
14434     if (TheSelect->getOpcode() == ISD::SELECT) {
14435       SDNode *CondNode = TheSelect->getOperand(0).getNode();
14436       if ((LLD->hasAnyUseOfValue(1) && LLD->isPredecessorOf(CondNode)) ||
14437           (RLD->hasAnyUseOfValue(1) && RLD->isPredecessorOf(CondNode)))
14438         return false;
14439       // The loads must not depend on one another.
14440       if (LLD->isPredecessorOf(RLD) ||
14441           RLD->isPredecessorOf(LLD))
14442         return false;
14443       Addr = DAG.getSelect(SDLoc(TheSelect),
14444                            LLD->getBasePtr().getValueType(),
14445                            TheSelect->getOperand(0), LLD->getBasePtr(),
14446                            RLD->getBasePtr());
14447     } else {  // Otherwise SELECT_CC
14448       SDNode *CondLHS = TheSelect->getOperand(0).getNode();
14449       SDNode *CondRHS = TheSelect->getOperand(1).getNode();
14450 
14451       if ((LLD->hasAnyUseOfValue(1) &&
14452            (LLD->isPredecessorOf(CondLHS) || LLD->isPredecessorOf(CondRHS))) ||
14453           (RLD->hasAnyUseOfValue(1) &&
14454            (RLD->isPredecessorOf(CondLHS) || RLD->isPredecessorOf(CondRHS))))
14455         return false;
14456 
14457       Addr = DAG.getNode(ISD::SELECT_CC, SDLoc(TheSelect),
14458                          LLD->getBasePtr().getValueType(),
14459                          TheSelect->getOperand(0),
14460                          TheSelect->getOperand(1),
14461                          LLD->getBasePtr(), RLD->getBasePtr(),
14462                          TheSelect->getOperand(4));
14463     }
14464 
14465     SDValue Load;
14466     // It is safe to replace the two loads if they have different alignments,
14467     // but the new load must be the minimum (most restrictive) alignment of the
14468     // inputs.
14469     unsigned Alignment = std::min(LLD->getAlignment(), RLD->getAlignment());
14470     MachineMemOperand::Flags MMOFlags = LLD->getMemOperand()->getFlags();
14471     if (!RLD->isInvariant())
14472       MMOFlags &= ~MachineMemOperand::MOInvariant;
14473     if (!RLD->isDereferenceable())
14474       MMOFlags &= ~MachineMemOperand::MODereferenceable;
14475     if (LLD->getExtensionType() == ISD::NON_EXTLOAD) {
14476       // FIXME: Discards pointer and AA info.
14477       Load = DAG.getLoad(TheSelect->getValueType(0), SDLoc(TheSelect),
14478                          LLD->getChain(), Addr, MachinePointerInfo(), Alignment,
14479                          MMOFlags);
14480     } else {
14481       // FIXME: Discards pointer and AA info.
14482       Load = DAG.getExtLoad(
14483           LLD->getExtensionType() == ISD::EXTLOAD ? RLD->getExtensionType()
14484                                                   : LLD->getExtensionType(),
14485           SDLoc(TheSelect), TheSelect->getValueType(0), LLD->getChain(), Addr,
14486           MachinePointerInfo(), LLD->getMemoryVT(), Alignment, MMOFlags);
14487     }
14488 
14489     // Users of the select now use the result of the load.
14490     CombineTo(TheSelect, Load);
14491 
14492     // Users of the old loads now use the new load's chain.  We know the
14493     // old-load value is dead now.
14494     CombineTo(LHS.getNode(), Load.getValue(0), Load.getValue(1));
14495     CombineTo(RHS.getNode(), Load.getValue(0), Load.getValue(1));
14496     return true;
14497   }
14498 
14499   return false;
14500 }
14501 
14502 /// Simplify an expression of the form (N0 cond N1) ? N2 : N3
14503 /// where 'cond' is the comparison specified by CC.
14504 SDValue DAGCombiner::SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
14505                                       SDValue N2, SDValue N3, ISD::CondCode CC,
14506                                       bool NotExtCompare) {
14507   // (x ? y : y) -> y.
14508   if (N2 == N3) return N2;
14509 
14510   EVT VT = N2.getValueType();
14511   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1.getNode());
14512   ConstantSDNode *N2C = dyn_cast<ConstantSDNode>(N2.getNode());
14513 
14514   // Determine if the condition we're dealing with is constant
14515   SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()),
14516                               N0, N1, CC, DL, false);
14517   if (SCC.getNode()) AddToWorklist(SCC.getNode());
14518 
14519   if (ConstantSDNode *SCCC = dyn_cast_or_null<ConstantSDNode>(SCC.getNode())) {
14520     // fold select_cc true, x, y -> x
14521     // fold select_cc false, x, y -> y
14522     return !SCCC->isNullValue() ? N2 : N3;
14523   }
14524 
14525   // Check to see if we can simplify the select into an fabs node
14526   if (ConstantFPSDNode *CFP = dyn_cast<ConstantFPSDNode>(N1)) {
14527     // Allow either -0.0 or 0.0
14528     if (CFP->isZero()) {
14529       // select (setg[te] X, +/-0.0), X, fneg(X) -> fabs
14530       if ((CC == ISD::SETGE || CC == ISD::SETGT) &&
14531           N0 == N2 && N3.getOpcode() == ISD::FNEG &&
14532           N2 == N3.getOperand(0))
14533         return DAG.getNode(ISD::FABS, DL, VT, N0);
14534 
14535       // select (setl[te] X, +/-0.0), fneg(X), X -> fabs
14536       if ((CC == ISD::SETLT || CC == ISD::SETLE) &&
14537           N0 == N3 && N2.getOpcode() == ISD::FNEG &&
14538           N2.getOperand(0) == N3)
14539         return DAG.getNode(ISD::FABS, DL, VT, N3);
14540     }
14541   }
14542 
14543   // Turn "(a cond b) ? 1.0f : 2.0f" into "load (tmp + ((a cond b) ? 0 : 4)"
14544   // where "tmp" is a constant pool entry containing an array with 1.0 and 2.0
14545   // in it.  This is a win when the constant is not otherwise available because
14546   // it replaces two constant pool loads with one.  We only do this if the FP
14547   // type is known to be legal, because if it isn't, then we are before legalize
14548   // types an we want the other legalization to happen first (e.g. to avoid
14549   // messing with soft float) and if the ConstantFP is not legal, because if
14550   // it is legal, we may not need to store the FP constant in a constant pool.
14551   if (ConstantFPSDNode *TV = dyn_cast<ConstantFPSDNode>(N2))
14552     if (ConstantFPSDNode *FV = dyn_cast<ConstantFPSDNode>(N3)) {
14553       if (TLI.isTypeLegal(N2.getValueType()) &&
14554           (TLI.getOperationAction(ISD::ConstantFP, N2.getValueType()) !=
14555                TargetLowering::Legal &&
14556            !TLI.isFPImmLegal(TV->getValueAPF(), TV->getValueType(0)) &&
14557            !TLI.isFPImmLegal(FV->getValueAPF(), FV->getValueType(0))) &&
14558           // If both constants have multiple uses, then we won't need to do an
14559           // extra load, they are likely around in registers for other users.
14560           (TV->hasOneUse() || FV->hasOneUse())) {
14561         Constant *Elts[] = {
14562           const_cast<ConstantFP*>(FV->getConstantFPValue()),
14563           const_cast<ConstantFP*>(TV->getConstantFPValue())
14564         };
14565         Type *FPTy = Elts[0]->getType();
14566         const DataLayout &TD = DAG.getDataLayout();
14567 
14568         // Create a ConstantArray of the two constants.
14569         Constant *CA = ConstantArray::get(ArrayType::get(FPTy, 2), Elts);
14570         SDValue CPIdx =
14571             DAG.getConstantPool(CA, TLI.getPointerTy(DAG.getDataLayout()),
14572                                 TD.getPrefTypeAlignment(FPTy));
14573         unsigned Alignment = cast<ConstantPoolSDNode>(CPIdx)->getAlignment();
14574 
14575         // Get the offsets to the 0 and 1 element of the array so that we can
14576         // select between them.
14577         SDValue Zero = DAG.getIntPtrConstant(0, DL);
14578         unsigned EltSize = (unsigned)TD.getTypeAllocSize(Elts[0]->getType());
14579         SDValue One = DAG.getIntPtrConstant(EltSize, SDLoc(FV));
14580 
14581         SDValue Cond = DAG.getSetCC(DL,
14582                                     getSetCCResultType(N0.getValueType()),
14583                                     N0, N1, CC);
14584         AddToWorklist(Cond.getNode());
14585         SDValue CstOffset = DAG.getSelect(DL, Zero.getValueType(),
14586                                           Cond, One, Zero);
14587         AddToWorklist(CstOffset.getNode());
14588         CPIdx = DAG.getNode(ISD::ADD, DL, CPIdx.getValueType(), CPIdx,
14589                             CstOffset);
14590         AddToWorklist(CPIdx.getNode());
14591         return DAG.getLoad(
14592             TV->getValueType(0), DL, DAG.getEntryNode(), CPIdx,
14593             MachinePointerInfo::getConstantPool(DAG.getMachineFunction()),
14594             Alignment);
14595       }
14596     }
14597 
14598   // Check to see if we can perform the "gzip trick", transforming
14599   // (select_cc setlt X, 0, A, 0) -> (and (sra X, (sub size(X), 1), A)
14600   if (isNullConstant(N3) && CC == ISD::SETLT &&
14601       (isNullConstant(N1) ||                 // (a < 0) ? b : 0
14602        (isOneConstant(N1) && N0 == N2))) {   // (a < 1) ? a : 0
14603     EVT XType = N0.getValueType();
14604     EVT AType = N2.getValueType();
14605     if (XType.bitsGE(AType)) {
14606       // and (sra X, size(X)-1, A) -> "and (srl X, C2), A" iff A is a
14607       // single-bit constant.
14608       if (N2C && ((N2C->getAPIntValue() & (N2C->getAPIntValue() - 1)) == 0)) {
14609         unsigned ShCtV = N2C->getAPIntValue().logBase2();
14610         ShCtV = XType.getSizeInBits() - ShCtV - 1;
14611         SDValue ShCt = DAG.getConstant(ShCtV, SDLoc(N0),
14612                                        getShiftAmountTy(N0.getValueType()));
14613         SDValue Shift = DAG.getNode(ISD::SRL, SDLoc(N0),
14614                                     XType, N0, ShCt);
14615         AddToWorklist(Shift.getNode());
14616 
14617         if (XType.bitsGT(AType)) {
14618           Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
14619           AddToWorklist(Shift.getNode());
14620         }
14621 
14622         return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
14623       }
14624 
14625       SDValue Shift = DAG.getNode(ISD::SRA, SDLoc(N0),
14626                                   XType, N0,
14627                                   DAG.getConstant(XType.getSizeInBits() - 1,
14628                                                   SDLoc(N0),
14629                                          getShiftAmountTy(N0.getValueType())));
14630       AddToWorklist(Shift.getNode());
14631 
14632       if (XType.bitsGT(AType)) {
14633         Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
14634         AddToWorklist(Shift.getNode());
14635       }
14636 
14637       return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
14638     }
14639   }
14640 
14641   // fold (select_cc seteq (and x, y), 0, 0, A) -> (and (shr (shl x)) A)
14642   // where y is has a single bit set.
14643   // A plaintext description would be, we can turn the SELECT_CC into an AND
14644   // when the condition can be materialized as an all-ones register.  Any
14645   // single bit-test can be materialized as an all-ones register with
14646   // shift-left and shift-right-arith.
14647   if (CC == ISD::SETEQ && N0->getOpcode() == ISD::AND &&
14648       N0->getValueType(0) == VT && isNullConstant(N1) && isNullConstant(N2)) {
14649     SDValue AndLHS = N0->getOperand(0);
14650     ConstantSDNode *ConstAndRHS = dyn_cast<ConstantSDNode>(N0->getOperand(1));
14651     if (ConstAndRHS && ConstAndRHS->getAPIntValue().countPopulation() == 1) {
14652       // Shift the tested bit over the sign bit.
14653       const APInt &AndMask = ConstAndRHS->getAPIntValue();
14654       SDValue ShlAmt =
14655         DAG.getConstant(AndMask.countLeadingZeros(), SDLoc(AndLHS),
14656                         getShiftAmountTy(AndLHS.getValueType()));
14657       SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N0), VT, AndLHS, ShlAmt);
14658 
14659       // Now arithmetic right shift it all the way over, so the result is either
14660       // all-ones, or zero.
14661       SDValue ShrAmt =
14662         DAG.getConstant(AndMask.getBitWidth() - 1, SDLoc(Shl),
14663                         getShiftAmountTy(Shl.getValueType()));
14664       SDValue Shr = DAG.getNode(ISD::SRA, SDLoc(N0), VT, Shl, ShrAmt);
14665 
14666       return DAG.getNode(ISD::AND, DL, VT, Shr, N3);
14667     }
14668   }
14669 
14670   // fold select C, 16, 0 -> shl C, 4
14671   if (N2C && isNullConstant(N3) && N2C->getAPIntValue().isPowerOf2() &&
14672       TLI.getBooleanContents(N0.getValueType()) ==
14673           TargetLowering::ZeroOrOneBooleanContent) {
14674 
14675     // If the caller doesn't want us to simplify this into a zext of a compare,
14676     // don't do it.
14677     if (NotExtCompare && N2C->isOne())
14678       return SDValue();
14679 
14680     // Get a SetCC of the condition
14681     // NOTE: Don't create a SETCC if it's not legal on this target.
14682     if (!LegalOperations ||
14683         TLI.isOperationLegal(ISD::SETCC, N0.getValueType())) {
14684       SDValue Temp, SCC;
14685       // cast from setcc result type to select result type
14686       if (LegalTypes) {
14687         SCC  = DAG.getSetCC(DL, getSetCCResultType(N0.getValueType()),
14688                             N0, N1, CC);
14689         if (N2.getValueType().bitsLT(SCC.getValueType()))
14690           Temp = DAG.getZeroExtendInReg(SCC, SDLoc(N2),
14691                                         N2.getValueType());
14692         else
14693           Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2),
14694                              N2.getValueType(), SCC);
14695       } else {
14696         SCC  = DAG.getSetCC(SDLoc(N0), MVT::i1, N0, N1, CC);
14697         Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2),
14698                            N2.getValueType(), SCC);
14699       }
14700 
14701       AddToWorklist(SCC.getNode());
14702       AddToWorklist(Temp.getNode());
14703 
14704       if (N2C->isOne())
14705         return Temp;
14706 
14707       // shl setcc result by log2 n2c
14708       return DAG.getNode(
14709           ISD::SHL, DL, N2.getValueType(), Temp,
14710           DAG.getConstant(N2C->getAPIntValue().logBase2(), SDLoc(Temp),
14711                           getShiftAmountTy(Temp.getValueType())));
14712     }
14713   }
14714 
14715   // Check to see if this is an integer abs.
14716   // select_cc setg[te] X,  0,  X, -X ->
14717   // select_cc setgt    X, -1,  X, -X ->
14718   // select_cc setl[te] X,  0, -X,  X ->
14719   // select_cc setlt    X,  1, -X,  X ->
14720   // Y = sra (X, size(X)-1); xor (add (X, Y), Y)
14721   if (N1C) {
14722     ConstantSDNode *SubC = nullptr;
14723     if (((N1C->isNullValue() && (CC == ISD::SETGT || CC == ISD::SETGE)) ||
14724          (N1C->isAllOnesValue() && CC == ISD::SETGT)) &&
14725         N0 == N2 && N3.getOpcode() == ISD::SUB && N0 == N3.getOperand(1))
14726       SubC = dyn_cast<ConstantSDNode>(N3.getOperand(0));
14727     else if (((N1C->isNullValue() && (CC == ISD::SETLT || CC == ISD::SETLE)) ||
14728               (N1C->isOne() && CC == ISD::SETLT)) &&
14729              N0 == N3 && N2.getOpcode() == ISD::SUB && N0 == N2.getOperand(1))
14730       SubC = dyn_cast<ConstantSDNode>(N2.getOperand(0));
14731 
14732     EVT XType = N0.getValueType();
14733     if (SubC && SubC->isNullValue() && XType.isInteger()) {
14734       SDLoc DL(N0);
14735       SDValue Shift = DAG.getNode(ISD::SRA, DL, XType,
14736                                   N0,
14737                                   DAG.getConstant(XType.getSizeInBits() - 1, DL,
14738                                          getShiftAmountTy(N0.getValueType())));
14739       SDValue Add = DAG.getNode(ISD::ADD, DL,
14740                                 XType, N0, Shift);
14741       AddToWorklist(Shift.getNode());
14742       AddToWorklist(Add.getNode());
14743       return DAG.getNode(ISD::XOR, DL, XType, Add, Shift);
14744     }
14745   }
14746 
14747   // select_cc seteq X, 0, sizeof(X), ctlz(X) -> ctlz(X)
14748   // select_cc seteq X, 0, sizeof(X), ctlz_zero_undef(X) -> ctlz(X)
14749   // select_cc seteq X, 0, sizeof(X), cttz(X) -> cttz(X)
14750   // select_cc seteq X, 0, sizeof(X), cttz_zero_undef(X) -> cttz(X)
14751   // select_cc setne X, 0, ctlz(X), sizeof(X) -> ctlz(X)
14752   // select_cc setne X, 0, ctlz_zero_undef(X), sizeof(X) -> ctlz(X)
14753   // select_cc setne X, 0, cttz(X), sizeof(X) -> cttz(X)
14754   // select_cc setne X, 0, cttz_zero_undef(X), sizeof(X) -> cttz(X)
14755   if (N1C && N1C->isNullValue() && (CC == ISD::SETEQ || CC == ISD::SETNE)) {
14756     SDValue ValueOnZero = N2;
14757     SDValue Count = N3;
14758     // If the condition is NE instead of E, swap the operands.
14759     if (CC == ISD::SETNE)
14760       std::swap(ValueOnZero, Count);
14761     // Check if the value on zero is a constant equal to the bits in the type.
14762     if (auto *ValueOnZeroC = dyn_cast<ConstantSDNode>(ValueOnZero)) {
14763       if (ValueOnZeroC->getAPIntValue() == VT.getSizeInBits()) {
14764         // If the other operand is cttz/cttz_zero_undef of N0, and cttz is
14765         // legal, combine to just cttz.
14766         if ((Count.getOpcode() == ISD::CTTZ ||
14767              Count.getOpcode() == ISD::CTTZ_ZERO_UNDEF) &&
14768             N0 == Count.getOperand(0) &&
14769             (!LegalOperations || TLI.isOperationLegal(ISD::CTTZ, VT)))
14770           return DAG.getNode(ISD::CTTZ, DL, VT, N0);
14771         // If the other operand is ctlz/ctlz_zero_undef of N0, and ctlz is
14772         // legal, combine to just ctlz.
14773         if ((Count.getOpcode() == ISD::CTLZ ||
14774              Count.getOpcode() == ISD::CTLZ_ZERO_UNDEF) &&
14775             N0 == Count.getOperand(0) &&
14776             (!LegalOperations || TLI.isOperationLegal(ISD::CTLZ, VT)))
14777           return DAG.getNode(ISD::CTLZ, DL, VT, N0);
14778       }
14779     }
14780   }
14781 
14782   return SDValue();
14783 }
14784 
14785 /// This is a stub for TargetLowering::SimplifySetCC.
14786 SDValue DAGCombiner::SimplifySetCC(EVT VT, SDValue N0, SDValue N1,
14787                                    ISD::CondCode Cond, const SDLoc &DL,
14788                                    bool foldBooleans) {
14789   TargetLowering::DAGCombinerInfo
14790     DagCombineInfo(DAG, Level, false, this);
14791   return TLI.SimplifySetCC(VT, N0, N1, Cond, foldBooleans, DagCombineInfo, DL);
14792 }
14793 
14794 /// Given an ISD::SDIV node expressing a divide by constant, return
14795 /// a DAG expression to select that will generate the same value by multiplying
14796 /// by a magic number.
14797 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
14798 SDValue DAGCombiner::BuildSDIV(SDNode *N) {
14799   // when optimising for minimum size, we don't want to expand a div to a mul
14800   // and a shift.
14801   if (DAG.getMachineFunction().getFunction()->optForMinSize())
14802     return SDValue();
14803 
14804   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
14805   if (!C)
14806     return SDValue();
14807 
14808   // Avoid division by zero.
14809   if (C->isNullValue())
14810     return SDValue();
14811 
14812   std::vector<SDNode*> Built;
14813   SDValue S =
14814       TLI.BuildSDIV(N, C->getAPIntValue(), DAG, LegalOperations, &Built);
14815 
14816   for (SDNode *N : Built)
14817     AddToWorklist(N);
14818   return S;
14819 }
14820 
14821 /// Given an ISD::SDIV node expressing a divide by constant power of 2, return a
14822 /// DAG expression that will generate the same value by right shifting.
14823 SDValue DAGCombiner::BuildSDIVPow2(SDNode *N) {
14824   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
14825   if (!C)
14826     return SDValue();
14827 
14828   // Avoid division by zero.
14829   if (C->isNullValue())
14830     return SDValue();
14831 
14832   std::vector<SDNode *> Built;
14833   SDValue S = TLI.BuildSDIVPow2(N, C->getAPIntValue(), DAG, &Built);
14834 
14835   for (SDNode *N : Built)
14836     AddToWorklist(N);
14837   return S;
14838 }
14839 
14840 /// Given an ISD::UDIV node expressing a divide by constant, return a DAG
14841 /// expression that will generate the same value by multiplying by a magic
14842 /// number.
14843 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
14844 SDValue DAGCombiner::BuildUDIV(SDNode *N) {
14845   // when optimising for minimum size, we don't want to expand a div to a mul
14846   // and a shift.
14847   if (DAG.getMachineFunction().getFunction()->optForMinSize())
14848     return SDValue();
14849 
14850   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
14851   if (!C)
14852     return SDValue();
14853 
14854   // Avoid division by zero.
14855   if (C->isNullValue())
14856     return SDValue();
14857 
14858   std::vector<SDNode*> Built;
14859   SDValue S =
14860       TLI.BuildUDIV(N, C->getAPIntValue(), DAG, LegalOperations, &Built);
14861 
14862   for (SDNode *N : Built)
14863     AddToWorklist(N);
14864   return S;
14865 }
14866 
14867 SDValue DAGCombiner::BuildReciprocalEstimate(SDValue Op, SDNodeFlags *Flags) {
14868   if (Level >= AfterLegalizeDAG)
14869     return SDValue();
14870 
14871   // Expose the DAG combiner to the target combiner implementations.
14872   TargetLowering::DAGCombinerInfo DCI(DAG, Level, false, this);
14873 
14874   unsigned Iterations = 0;
14875   if (SDValue Est = TLI.getRecipEstimate(Op, DCI, Iterations)) {
14876     if (Iterations) {
14877       // Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
14878       // For the reciprocal, we need to find the zero of the function:
14879       //   F(X) = A X - 1 [which has a zero at X = 1/A]
14880       //     =>
14881       //   X_{i+1} = X_i (2 - A X_i) = X_i + X_i (1 - A X_i) [this second form
14882       //     does not require additional intermediate precision]
14883       EVT VT = Op.getValueType();
14884       SDLoc DL(Op);
14885       SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
14886 
14887       AddToWorklist(Est.getNode());
14888 
14889       // Newton iterations: Est = Est + Est (1 - Arg * Est)
14890       for (unsigned i = 0; i < Iterations; ++i) {
14891         SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Op, Est, Flags);
14892         AddToWorklist(NewEst.getNode());
14893 
14894         NewEst = DAG.getNode(ISD::FSUB, DL, VT, FPOne, NewEst, Flags);
14895         AddToWorklist(NewEst.getNode());
14896 
14897         NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
14898         AddToWorklist(NewEst.getNode());
14899 
14900         Est = DAG.getNode(ISD::FADD, DL, VT, Est, NewEst, Flags);
14901         AddToWorklist(Est.getNode());
14902       }
14903     }
14904     return Est;
14905   }
14906 
14907   return SDValue();
14908 }
14909 
14910 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
14911 /// For the reciprocal sqrt, we need to find the zero of the function:
14912 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
14913 ///     =>
14914 ///   X_{i+1} = X_i (1.5 - A X_i^2 / 2)
14915 /// As a result, we precompute A/2 prior to the iteration loop.
14916 SDValue DAGCombiner::buildSqrtNROneConst(SDValue Arg, SDValue Est,
14917                                          unsigned Iterations,
14918                                          SDNodeFlags *Flags, bool Reciprocal) {
14919   EVT VT = Arg.getValueType();
14920   SDLoc DL(Arg);
14921   SDValue ThreeHalves = DAG.getConstantFP(1.5, DL, VT);
14922 
14923   // We now need 0.5 * Arg which we can write as (1.5 * Arg - Arg) so that
14924   // this entire sequence requires only one FP constant.
14925   SDValue HalfArg = DAG.getNode(ISD::FMUL, DL, VT, ThreeHalves, Arg, Flags);
14926   AddToWorklist(HalfArg.getNode());
14927 
14928   HalfArg = DAG.getNode(ISD::FSUB, DL, VT, HalfArg, Arg, Flags);
14929   AddToWorklist(HalfArg.getNode());
14930 
14931   // Newton iterations: Est = Est * (1.5 - HalfArg * Est * Est)
14932   for (unsigned i = 0; i < Iterations; ++i) {
14933     SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, Est, Flags);
14934     AddToWorklist(NewEst.getNode());
14935 
14936     NewEst = DAG.getNode(ISD::FMUL, DL, VT, HalfArg, NewEst, Flags);
14937     AddToWorklist(NewEst.getNode());
14938 
14939     NewEst = DAG.getNode(ISD::FSUB, DL, VT, ThreeHalves, NewEst, Flags);
14940     AddToWorklist(NewEst.getNode());
14941 
14942     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
14943     AddToWorklist(Est.getNode());
14944   }
14945 
14946   // If non-reciprocal square root is requested, multiply the result by Arg.
14947   if (!Reciprocal) {
14948     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, Arg, Flags);
14949     AddToWorklist(Est.getNode());
14950   }
14951 
14952   return Est;
14953 }
14954 
14955 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
14956 /// For the reciprocal sqrt, we need to find the zero of the function:
14957 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
14958 ///     =>
14959 ///   X_{i+1} = (-0.5 * X_i) * (A * X_i * X_i + (-3.0))
14960 SDValue DAGCombiner::buildSqrtNRTwoConst(SDValue Arg, SDValue Est,
14961                                          unsigned Iterations,
14962                                          SDNodeFlags *Flags, bool Reciprocal) {
14963   EVT VT = Arg.getValueType();
14964   SDLoc DL(Arg);
14965   SDValue MinusThree = DAG.getConstantFP(-3.0, DL, VT);
14966   SDValue MinusHalf = DAG.getConstantFP(-0.5, DL, VT);
14967 
14968   // This routine must enter the loop below to work correctly
14969   // when (Reciprocal == false).
14970   assert(Iterations > 0);
14971 
14972   // Newton iterations for reciprocal square root:
14973   // E = (E * -0.5) * ((A * E) * E + -3.0)
14974   for (unsigned i = 0; i < Iterations; ++i) {
14975     SDValue AE = DAG.getNode(ISD::FMUL, DL, VT, Arg, Est, Flags);
14976     AddToWorklist(AE.getNode());
14977 
14978     SDValue AEE = DAG.getNode(ISD::FMUL, DL, VT, AE, Est, Flags);
14979     AddToWorklist(AEE.getNode());
14980 
14981     SDValue RHS = DAG.getNode(ISD::FADD, DL, VT, AEE, MinusThree, Flags);
14982     AddToWorklist(RHS.getNode());
14983 
14984     // When calculating a square root at the last iteration build:
14985     // S = ((A * E) * -0.5) * ((A * E) * E + -3.0)
14986     // (notice a common subexpression)
14987     SDValue LHS;
14988     if (Reciprocal || (i + 1) < Iterations) {
14989       // RSQRT: LHS = (E * -0.5)
14990       LHS = DAG.getNode(ISD::FMUL, DL, VT, Est, MinusHalf, Flags);
14991     } else {
14992       // SQRT: LHS = (A * E) * -0.5
14993       LHS = DAG.getNode(ISD::FMUL, DL, VT, AE, MinusHalf, Flags);
14994     }
14995     AddToWorklist(LHS.getNode());
14996 
14997     Est = DAG.getNode(ISD::FMUL, DL, VT, LHS, RHS, Flags);
14998     AddToWorklist(Est.getNode());
14999   }
15000 
15001   return Est;
15002 }
15003 
15004 /// Build code to calculate either rsqrt(Op) or sqrt(Op). In the latter case
15005 /// Op*rsqrt(Op) is actually computed, so additional postprocessing is needed if
15006 /// Op can be zero.
15007 SDValue DAGCombiner::buildSqrtEstimateImpl(SDValue Op, SDNodeFlags *Flags,
15008                                            bool Reciprocal) {
15009   if (Level >= AfterLegalizeDAG)
15010     return SDValue();
15011 
15012   // Expose the DAG combiner to the target combiner implementations.
15013   TargetLowering::DAGCombinerInfo DCI(DAG, Level, false, this);
15014   unsigned Iterations = 0;
15015   bool UseOneConstNR = false;
15016   if (SDValue Est = TLI.getRsqrtEstimate(Op, DCI, Iterations, UseOneConstNR)) {
15017     AddToWorklist(Est.getNode());
15018     if (Iterations) {
15019       Est = UseOneConstNR
15020                 ? buildSqrtNROneConst(Op, Est, Iterations, Flags, Reciprocal)
15021                 : buildSqrtNRTwoConst(Op, Est, Iterations, Flags, Reciprocal);
15022     }
15023     return Est;
15024   }
15025 
15026   return SDValue();
15027 }
15028 
15029 SDValue DAGCombiner::buildRsqrtEstimate(SDValue Op, SDNodeFlags *Flags) {
15030   return buildSqrtEstimateImpl(Op, Flags, true);
15031 }
15032 
15033 SDValue DAGCombiner::buildSqrtEstimate(SDValue Op, SDNodeFlags *Flags) {
15034   SDValue Est = buildSqrtEstimateImpl(Op, Flags, false);
15035   if (!Est)
15036     return SDValue();
15037 
15038   // Unfortunately, Est is now NaN if the input was exactly 0.
15039   // Select out this case and force the answer to 0.
15040   EVT VT = Est.getValueType();
15041   SDLoc DL(Op);
15042   SDValue Zero = DAG.getConstantFP(0.0, DL, VT);
15043   EVT CCVT = getSetCCResultType(VT);
15044   SDValue ZeroCmp = DAG.getSetCC(DL, CCVT, Op, Zero, ISD::SETEQ);
15045   AddToWorklist(ZeroCmp.getNode());
15046 
15047   Est = DAG.getNode(VT.isVector() ? ISD::VSELECT : ISD::SELECT, DL, VT, ZeroCmp,
15048                     Zero, Est);
15049   AddToWorklist(Est.getNode());
15050   return Est;
15051 }
15052 
15053 /// Return true if base is a frame index, which is known not to alias with
15054 /// anything but itself.  Provides base object and offset as results.
15055 static bool FindBaseOffset(SDValue Ptr, SDValue &Base, int64_t &Offset,
15056                            const GlobalValue *&GV, const void *&CV) {
15057   // Assume it is a primitive operation.
15058   Base = Ptr; Offset = 0; GV = nullptr; CV = nullptr;
15059 
15060   // If it's an adding a simple constant then integrate the offset.
15061   if (Base.getOpcode() == ISD::ADD) {
15062     if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Base.getOperand(1))) {
15063       Base = Base.getOperand(0);
15064       Offset += C->getZExtValue();
15065     }
15066   }
15067 
15068   // Return the underlying GlobalValue, and update the Offset.  Return false
15069   // for GlobalAddressSDNode since the same GlobalAddress may be represented
15070   // by multiple nodes with different offsets.
15071   if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Base)) {
15072     GV = G->getGlobal();
15073     Offset += G->getOffset();
15074     return false;
15075   }
15076 
15077   // Return the underlying Constant value, and update the Offset.  Return false
15078   // for ConstantSDNodes since the same constant pool entry may be represented
15079   // by multiple nodes with different offsets.
15080   if (ConstantPoolSDNode *C = dyn_cast<ConstantPoolSDNode>(Base)) {
15081     CV = C->isMachineConstantPoolEntry() ? (const void *)C->getMachineCPVal()
15082                                          : (const void *)C->getConstVal();
15083     Offset += C->getOffset();
15084     return false;
15085   }
15086   // If it's any of the following then it can't alias with anything but itself.
15087   return isa<FrameIndexSDNode>(Base);
15088 }
15089 
15090 /// Return true if there is any possibility that the two addresses overlap.
15091 bool DAGCombiner::isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const {
15092   // If they are the same then they must be aliases.
15093   if (Op0->getBasePtr() == Op1->getBasePtr()) return true;
15094 
15095   // If they are both volatile then they cannot be reordered.
15096   if (Op0->isVolatile() && Op1->isVolatile()) return true;
15097 
15098   // If one operation reads from invariant memory, and the other may store, they
15099   // cannot alias. These should really be checking the equivalent of mayWrite,
15100   // but it only matters for memory nodes other than load /store.
15101   if (Op0->isInvariant() && Op1->writeMem())
15102     return false;
15103 
15104   if (Op1->isInvariant() && Op0->writeMem())
15105     return false;
15106 
15107   // Gather base node and offset information.
15108   SDValue Base1, Base2;
15109   int64_t Offset1, Offset2;
15110   const GlobalValue *GV1, *GV2;
15111   const void *CV1, *CV2;
15112   bool isFrameIndex1 = FindBaseOffset(Op0->getBasePtr(),
15113                                       Base1, Offset1, GV1, CV1);
15114   bool isFrameIndex2 = FindBaseOffset(Op1->getBasePtr(),
15115                                       Base2, Offset2, GV2, CV2);
15116 
15117   // If they have a same base address then check to see if they overlap.
15118   if (Base1 == Base2 || (GV1 && (GV1 == GV2)) || (CV1 && (CV1 == CV2)))
15119     return !((Offset1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= Offset2 ||
15120              (Offset2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= Offset1);
15121 
15122   // It is possible for different frame indices to alias each other, mostly
15123   // when tail call optimization reuses return address slots for arguments.
15124   // To catch this case, look up the actual index of frame indices to compute
15125   // the real alias relationship.
15126   if (isFrameIndex1 && isFrameIndex2) {
15127     MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
15128     Offset1 += MFI.getObjectOffset(cast<FrameIndexSDNode>(Base1)->getIndex());
15129     Offset2 += MFI.getObjectOffset(cast<FrameIndexSDNode>(Base2)->getIndex());
15130     return !((Offset1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= Offset2 ||
15131              (Offset2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= Offset1);
15132   }
15133 
15134   // Otherwise, if we know what the bases are, and they aren't identical, then
15135   // we know they cannot alias.
15136   if ((isFrameIndex1 || CV1 || GV1) && (isFrameIndex2 || CV2 || GV2))
15137     return false;
15138 
15139   // If we know required SrcValue1 and SrcValue2 have relatively large alignment
15140   // compared to the size and offset of the access, we may be able to prove they
15141   // do not alias.  This check is conservative for now to catch cases created by
15142   // splitting vector types.
15143   if ((Op0->getOriginalAlignment() == Op1->getOriginalAlignment()) &&
15144       (Op0->getSrcValueOffset() != Op1->getSrcValueOffset()) &&
15145       (Op0->getMemoryVT().getSizeInBits() >> 3 ==
15146        Op1->getMemoryVT().getSizeInBits() >> 3) &&
15147       (Op0->getOriginalAlignment() > (Op0->getMemoryVT().getSizeInBits() >> 3))) {
15148     int64_t OffAlign1 = Op0->getSrcValueOffset() % Op0->getOriginalAlignment();
15149     int64_t OffAlign2 = Op1->getSrcValueOffset() % Op1->getOriginalAlignment();
15150 
15151     // There is no overlap between these relatively aligned accesses of similar
15152     // size, return no alias.
15153     if ((OffAlign1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= OffAlign2 ||
15154         (OffAlign2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= OffAlign1)
15155       return false;
15156   }
15157 
15158   bool UseAA = CombinerGlobalAA.getNumOccurrences() > 0
15159                    ? CombinerGlobalAA
15160                    : DAG.getSubtarget().useAA();
15161 #ifndef NDEBUG
15162   if (CombinerAAOnlyFunc.getNumOccurrences() &&
15163       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
15164     UseAA = false;
15165 #endif
15166   if (UseAA &&
15167       Op0->getMemOperand()->getValue() && Op1->getMemOperand()->getValue()) {
15168     // Use alias analysis information.
15169     int64_t MinOffset = std::min(Op0->getSrcValueOffset(),
15170                                  Op1->getSrcValueOffset());
15171     int64_t Overlap1 = (Op0->getMemoryVT().getSizeInBits() >> 3) +
15172         Op0->getSrcValueOffset() - MinOffset;
15173     int64_t Overlap2 = (Op1->getMemoryVT().getSizeInBits() >> 3) +
15174         Op1->getSrcValueOffset() - MinOffset;
15175     AliasResult AAResult =
15176         AA.alias(MemoryLocation(Op0->getMemOperand()->getValue(), Overlap1,
15177                                 UseTBAA ? Op0->getAAInfo() : AAMDNodes()),
15178                  MemoryLocation(Op1->getMemOperand()->getValue(), Overlap2,
15179                                 UseTBAA ? Op1->getAAInfo() : AAMDNodes()));
15180     if (AAResult == NoAlias)
15181       return false;
15182   }
15183 
15184   // Otherwise we have to assume they alias.
15185   return true;
15186 }
15187 
15188 /// Walk up chain skipping non-aliasing memory nodes,
15189 /// looking for aliasing nodes and adding them to the Aliases vector.
15190 void DAGCombiner::GatherAllAliases(SDNode *N, SDValue OriginalChain,
15191                                    SmallVectorImpl<SDValue> &Aliases) {
15192   SmallVector<SDValue, 8> Chains;     // List of chains to visit.
15193   SmallPtrSet<SDNode *, 16> Visited;  // Visited node set.
15194 
15195   // Get alias information for node.
15196   bool IsLoad = isa<LoadSDNode>(N) && !cast<LSBaseSDNode>(N)->isVolatile();
15197 
15198   // Starting off.
15199   Chains.push_back(OriginalChain);
15200   unsigned Depth = 0;
15201 
15202   // Look at each chain and determine if it is an alias.  If so, add it to the
15203   // aliases list.  If not, then continue up the chain looking for the next
15204   // candidate.
15205   while (!Chains.empty()) {
15206     SDValue Chain = Chains.pop_back_val();
15207 
15208     // For TokenFactor nodes, look at each operand and only continue up the
15209     // chain until we reach the depth limit.
15210     //
15211     // FIXME: The depth check could be made to return the last non-aliasing
15212     // chain we found before we hit a tokenfactor rather than the original
15213     // chain.
15214     if (Depth > TLI.getGatherAllAliasesMaxDepth()) {
15215       Aliases.clear();
15216       Aliases.push_back(OriginalChain);
15217       return;
15218     }
15219 
15220     // Don't bother if we've been before.
15221     if (!Visited.insert(Chain.getNode()).second)
15222       continue;
15223 
15224     switch (Chain.getOpcode()) {
15225     case ISD::EntryToken:
15226       // Entry token is ideal chain operand, but handled in FindBetterChain.
15227       break;
15228 
15229     case ISD::LOAD:
15230     case ISD::STORE: {
15231       // Get alias information for Chain.
15232       bool IsOpLoad = isa<LoadSDNode>(Chain.getNode()) &&
15233           !cast<LSBaseSDNode>(Chain.getNode())->isVolatile();
15234 
15235       // If chain is alias then stop here.
15236       if (!(IsLoad && IsOpLoad) &&
15237           isAlias(cast<LSBaseSDNode>(N), cast<LSBaseSDNode>(Chain.getNode()))) {
15238         Aliases.push_back(Chain);
15239       } else {
15240         // Look further up the chain.
15241         Chains.push_back(Chain.getOperand(0));
15242         ++Depth;
15243       }
15244       break;
15245     }
15246 
15247     case ISD::TokenFactor:
15248       // We have to check each of the operands of the token factor for "small"
15249       // token factors, so we queue them up.  Adding the operands to the queue
15250       // (stack) in reverse order maintains the original order and increases the
15251       // likelihood that getNode will find a matching token factor (CSE.)
15252       if (Chain.getNumOperands() > 16) {
15253         Aliases.push_back(Chain);
15254         break;
15255       }
15256       for (unsigned n = Chain.getNumOperands(); n;)
15257         Chains.push_back(Chain.getOperand(--n));
15258       ++Depth;
15259       break;
15260 
15261     default:
15262       // For all other instructions we will just have to take what we can get.
15263       Aliases.push_back(Chain);
15264       break;
15265     }
15266   }
15267 }
15268 
15269 /// Walk up chain skipping non-aliasing memory nodes, looking for a better chain
15270 /// (aliasing node.)
15271 SDValue DAGCombiner::FindBetterChain(SDNode *N, SDValue OldChain) {
15272   SmallVector<SDValue, 8> Aliases;  // Ops for replacing token factor.
15273 
15274   // Accumulate all the aliases to this node.
15275   GatherAllAliases(N, OldChain, Aliases);
15276 
15277   // If no operands then chain to entry token.
15278   if (Aliases.size() == 0)
15279     return DAG.getEntryNode();
15280 
15281   // If a single operand then chain to it.  We don't need to revisit it.
15282   if (Aliases.size() == 1)
15283     return Aliases[0];
15284 
15285   // Construct a custom tailored token factor.
15286   return DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Aliases);
15287 }
15288 
15289 bool DAGCombiner::findBetterNeighborChains(StoreSDNode *St) {
15290   // This holds the base pointer, index, and the offset in bytes from the base
15291   // pointer.
15292   BaseIndexOffset BasePtr = BaseIndexOffset::match(St->getBasePtr(), DAG);
15293 
15294   // We must have a base and an offset.
15295   if (!BasePtr.Base.getNode())
15296     return false;
15297 
15298   // Do not handle stores to undef base pointers.
15299   if (BasePtr.Base.isUndef())
15300     return false;
15301 
15302   SmallVector<StoreSDNode *, 8> ChainedStores;
15303   ChainedStores.push_back(St);
15304 
15305   // Walk up the chain and look for nodes with offsets from the same
15306   // base pointer. Stop when reaching an instruction with a different kind
15307   // or instruction which has a different base pointer.
15308   StoreSDNode *Index = St;
15309   while (Index) {
15310     // If the chain has more than one use, then we can't reorder the mem ops.
15311     if (Index != St && !SDValue(Index, 0)->hasOneUse())
15312       break;
15313 
15314     if (Index->isVolatile() || Index->isIndexed())
15315       break;
15316 
15317     // Find the base pointer and offset for this memory node.
15318     BaseIndexOffset Ptr = BaseIndexOffset::match(Index->getBasePtr(), DAG);
15319 
15320     // Check that the base pointer is the same as the original one.
15321     if (!Ptr.equalBaseIndex(BasePtr))
15322       break;
15323 
15324     // Find the next memory operand in the chain. If the next operand in the
15325     // chain is a store then move up and continue the scan with the next
15326     // memory operand. If the next operand is a load save it and use alias
15327     // information to check if it interferes with anything.
15328     SDNode *NextInChain = Index->getChain().getNode();
15329     while (true) {
15330       if (StoreSDNode *STn = dyn_cast<StoreSDNode>(NextInChain)) {
15331         // We found a store node. Use it for the next iteration.
15332         if (STn->isVolatile() || STn->isIndexed()) {
15333           Index = nullptr;
15334           break;
15335         }
15336         ChainedStores.push_back(STn);
15337         Index = STn;
15338         break;
15339       } else if (LoadSDNode *Ldn = dyn_cast<LoadSDNode>(NextInChain)) {
15340         NextInChain = Ldn->getChain().getNode();
15341         continue;
15342       } else {
15343         Index = nullptr;
15344         break;
15345       }
15346     }
15347   }
15348 
15349   bool MadeChangeToSt = false;
15350   SmallVector<std::pair<StoreSDNode *, SDValue>, 8> BetterChains;
15351 
15352   for (StoreSDNode *ChainedStore : ChainedStores) {
15353     SDValue Chain = ChainedStore->getChain();
15354     SDValue BetterChain = FindBetterChain(ChainedStore, Chain);
15355 
15356     if (Chain != BetterChain) {
15357       if (ChainedStore == St)
15358         MadeChangeToSt = true;
15359       BetterChains.push_back(std::make_pair(ChainedStore, BetterChain));
15360     }
15361   }
15362 
15363   // Do all replacements after finding the replacements to make to avoid making
15364   // the chains more complicated by introducing new TokenFactors.
15365   for (auto Replacement : BetterChains)
15366     replaceStoreChain(Replacement.first, Replacement.second);
15367 
15368   return MadeChangeToSt;
15369 }
15370 
15371 /// This is the entry point for the file.
15372 void SelectionDAG::Combine(CombineLevel Level, AliasAnalysis &AA,
15373                            CodeGenOpt::Level OptLevel) {
15374   /// This is the main entry point to this class.
15375   DAGCombiner(*this, AA, OptLevel).Run(Level);
15376 }
15377