1 //===-- DAGCombiner.cpp - Implement a DAG node combiner -------------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 // This pass combines dag nodes to form fewer, simpler DAG nodes.  It can be run
11 // both before and after the DAG is legalized.
12 //
13 // This pass is not a substitute for the LLVM IR instcombine pass. This pass is
14 // primarily intended to handle simplification opportunities that are implicit
15 // in the LLVM IR and exposed by the various codegen lowering phases.
16 //
17 //===----------------------------------------------------------------------===//
18 
19 #include "llvm/CodeGen/SelectionDAG.h"
20 #include "llvm/ADT/SetVector.h"
21 #include "llvm/ADT/SmallBitVector.h"
22 #include "llvm/ADT/SmallPtrSet.h"
23 #include "llvm/ADT/Statistic.h"
24 #include "llvm/Analysis/AliasAnalysis.h"
25 #include "llvm/CodeGen/MachineFrameInfo.h"
26 #include "llvm/CodeGen/MachineFunction.h"
27 #include "llvm/CodeGen/SelectionDAGTargetInfo.h"
28 #include "llvm/IR/DataLayout.h"
29 #include "llvm/IR/DerivedTypes.h"
30 #include "llvm/IR/Function.h"
31 #include "llvm/IR/LLVMContext.h"
32 #include "llvm/Support/CommandLine.h"
33 #include "llvm/Support/Debug.h"
34 #include "llvm/Support/ErrorHandling.h"
35 #include "llvm/Support/MathExtras.h"
36 #include "llvm/Support/raw_ostream.h"
37 #include "llvm/Target/TargetLowering.h"
38 #include "llvm/Target/TargetOptions.h"
39 #include "llvm/Target/TargetRegisterInfo.h"
40 #include "llvm/Target/TargetSubtargetInfo.h"
41 #include <algorithm>
42 using namespace llvm;
43 
44 #define DEBUG_TYPE "dagcombine"
45 
46 STATISTIC(NodesCombined   , "Number of dag nodes combined");
47 STATISTIC(PreIndexedNodes , "Number of pre-indexed nodes created");
48 STATISTIC(PostIndexedNodes, "Number of post-indexed nodes created");
49 STATISTIC(OpsNarrowed     , "Number of load/op/store narrowed");
50 STATISTIC(LdStFP2Int      , "Number of fp load/store pairs transformed to int");
51 STATISTIC(SlicedLoads, "Number of load sliced");
52 
53 namespace {
54   static cl::opt<bool>
55     CombinerAA("combiner-alias-analysis", cl::Hidden,
56                cl::desc("Enable DAG combiner alias-analysis heuristics"));
57 
58   static cl::opt<bool>
59     CombinerGlobalAA("combiner-global-alias-analysis", cl::Hidden,
60                cl::desc("Enable DAG combiner's use of IR alias analysis"));
61 
62   static cl::opt<bool>
63     UseTBAA("combiner-use-tbaa", cl::Hidden, cl::init(true),
64                cl::desc("Enable DAG combiner's use of TBAA"));
65 
66 #ifndef NDEBUG
67   static cl::opt<std::string>
68     CombinerAAOnlyFunc("combiner-aa-only-func", cl::Hidden,
69                cl::desc("Only use DAG-combiner alias analysis in this"
70                         " function"));
71 #endif
72 
73   /// Hidden option to stress test load slicing, i.e., when this option
74   /// is enabled, load slicing bypasses most of its profitability guards.
75   static cl::opt<bool>
76   StressLoadSlicing("combiner-stress-load-slicing", cl::Hidden,
77                     cl::desc("Bypass the profitability model of load "
78                              "slicing"),
79                     cl::init(false));
80 
81   static cl::opt<bool>
82     MaySplitLoadIndex("combiner-split-load-index", cl::Hidden, cl::init(true),
83                       cl::desc("DAG combiner may split indexing from loads"));
84 
85 //------------------------------ DAGCombiner ---------------------------------//
86 
87   class DAGCombiner {
88     SelectionDAG &DAG;
89     const TargetLowering &TLI;
90     CombineLevel Level;
91     CodeGenOpt::Level OptLevel;
92     bool LegalOperations;
93     bool LegalTypes;
94     bool ForCodeSize;
95 
96     /// \brief Worklist of all of the nodes that need to be simplified.
97     ///
98     /// This must behave as a stack -- new nodes to process are pushed onto the
99     /// back and when processing we pop off of the back.
100     ///
101     /// The worklist will not contain duplicates but may contain null entries
102     /// due to nodes being deleted from the underlying DAG.
103     SmallVector<SDNode *, 64> Worklist;
104 
105     /// \brief Mapping from an SDNode to its position on the worklist.
106     ///
107     /// This is used to find and remove nodes from the worklist (by nulling
108     /// them) when they are deleted from the underlying DAG. It relies on
109     /// stable indices of nodes within the worklist.
110     DenseMap<SDNode *, unsigned> WorklistMap;
111 
112     /// \brief Set of nodes which have been combined (at least once).
113     ///
114     /// This is used to allow us to reliably add any operands of a DAG node
115     /// which have not yet been combined to the worklist.
116     SmallPtrSet<SDNode *, 32> CombinedNodes;
117 
118     // AA - Used for DAG load/store alias analysis.
119     AliasAnalysis &AA;
120 
121     /// When an instruction is simplified, add all users of the instruction to
122     /// the work lists because they might get more simplified now.
123     void AddUsersToWorklist(SDNode *N) {
124       for (SDNode *Node : N->uses())
125         AddToWorklist(Node);
126     }
127 
128     /// Call the node-specific routine that folds each particular type of node.
129     SDValue visit(SDNode *N);
130 
131   public:
132     /// Add to the worklist making sure its instance is at the back (next to be
133     /// processed.)
134     void AddToWorklist(SDNode *N) {
135       // Skip handle nodes as they can't usefully be combined and confuse the
136       // zero-use deletion strategy.
137       if (N->getOpcode() == ISD::HANDLENODE)
138         return;
139 
140       if (WorklistMap.insert(std::make_pair(N, Worklist.size())).second)
141         Worklist.push_back(N);
142     }
143 
144     /// Remove all instances of N from the worklist.
145     void removeFromWorklist(SDNode *N) {
146       CombinedNodes.erase(N);
147 
148       auto It = WorklistMap.find(N);
149       if (It == WorklistMap.end())
150         return; // Not in the worklist.
151 
152       // Null out the entry rather than erasing it to avoid a linear operation.
153       Worklist[It->second] = nullptr;
154       WorklistMap.erase(It);
155     }
156 
157     void deleteAndRecombine(SDNode *N);
158     bool recursivelyDeleteUnusedNodes(SDNode *N);
159 
160     /// Replaces all uses of the results of one DAG node with new values.
161     SDValue CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
162                       bool AddTo = true);
163 
164     /// Replaces all uses of the results of one DAG node with new values.
165     SDValue CombineTo(SDNode *N, SDValue Res, bool AddTo = true) {
166       return CombineTo(N, &Res, 1, AddTo);
167     }
168 
169     /// Replaces all uses of the results of one DAG node with new values.
170     SDValue CombineTo(SDNode *N, SDValue Res0, SDValue Res1,
171                       bool AddTo = true) {
172       SDValue To[] = { Res0, Res1 };
173       return CombineTo(N, To, 2, AddTo);
174     }
175 
176     void CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO);
177 
178   private:
179 
180     /// Check the specified integer node value to see if it can be simplified or
181     /// if things it uses can be simplified by bit propagation.
182     /// If so, return true.
183     bool SimplifyDemandedBits(SDValue Op) {
184       unsigned BitWidth = Op.getScalarValueSizeInBits();
185       APInt Demanded = APInt::getAllOnesValue(BitWidth);
186       return SimplifyDemandedBits(Op, Demanded);
187     }
188 
189     bool SimplifyDemandedBits(SDValue Op, const APInt &Demanded);
190 
191     bool CombineToPreIndexedLoadStore(SDNode *N);
192     bool CombineToPostIndexedLoadStore(SDNode *N);
193     SDValue SplitIndexingFromLoad(LoadSDNode *LD);
194     bool SliceUpLoad(SDNode *N);
195 
196     /// \brief Replace an ISD::EXTRACT_VECTOR_ELT of a load with a narrowed
197     ///   load.
198     ///
199     /// \param EVE ISD::EXTRACT_VECTOR_ELT to be replaced.
200     /// \param InVecVT type of the input vector to EVE with bitcasts resolved.
201     /// \param EltNo index of the vector element to load.
202     /// \param OriginalLoad load that EVE came from to be replaced.
203     /// \returns EVE on success SDValue() on failure.
204     SDValue ReplaceExtractVectorEltOfLoadWithNarrowedLoad(
205         SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad);
206     void ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad);
207     SDValue PromoteOperand(SDValue Op, EVT PVT, bool &Replace);
208     SDValue SExtPromoteOperand(SDValue Op, EVT PVT);
209     SDValue ZExtPromoteOperand(SDValue Op, EVT PVT);
210     SDValue PromoteIntBinOp(SDValue Op);
211     SDValue PromoteIntShiftOp(SDValue Op);
212     SDValue PromoteExtend(SDValue Op);
213     bool PromoteLoad(SDValue Op);
214 
215     void ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs, SDValue Trunc,
216                          SDValue ExtLoad, const SDLoc &DL,
217                          ISD::NodeType ExtType);
218 
219     /// Call the node-specific routine that knows how to fold each
220     /// particular type of node. If that doesn't do anything, try the
221     /// target-specific DAG combines.
222     SDValue combine(SDNode *N);
223 
224     // Visitation implementation - Implement dag node combining for different
225     // node types.  The semantics are as follows:
226     // Return Value:
227     //   SDValue.getNode() == 0 - No change was made
228     //   SDValue.getNode() == N - N was replaced, is dead and has been handled.
229     //   otherwise              - N should be replaced by the returned Operand.
230     //
231     SDValue visitTokenFactor(SDNode *N);
232     SDValue visitMERGE_VALUES(SDNode *N);
233     SDValue visitADD(SDNode *N);
234     SDValue visitSUB(SDNode *N);
235     SDValue visitADDC(SDNode *N);
236     SDValue visitSUBC(SDNode *N);
237     SDValue visitADDE(SDNode *N);
238     SDValue visitSUBE(SDNode *N);
239     SDValue visitMUL(SDNode *N);
240     SDValue useDivRem(SDNode *N);
241     SDValue visitSDIV(SDNode *N);
242     SDValue visitUDIV(SDNode *N);
243     SDValue visitREM(SDNode *N);
244     SDValue visitMULHU(SDNode *N);
245     SDValue visitMULHS(SDNode *N);
246     SDValue visitSMUL_LOHI(SDNode *N);
247     SDValue visitUMUL_LOHI(SDNode *N);
248     SDValue visitSMULO(SDNode *N);
249     SDValue visitUMULO(SDNode *N);
250     SDValue visitIMINMAX(SDNode *N);
251     SDValue visitAND(SDNode *N);
252     SDValue visitANDLike(SDValue N0, SDValue N1, SDNode *LocReference);
253     SDValue visitOR(SDNode *N);
254     SDValue visitORLike(SDValue N0, SDValue N1, SDNode *LocReference);
255     SDValue visitXOR(SDNode *N);
256     SDValue SimplifyVBinOp(SDNode *N);
257     SDValue visitSHL(SDNode *N);
258     SDValue visitSRA(SDNode *N);
259     SDValue visitSRL(SDNode *N);
260     SDValue visitRotate(SDNode *N);
261     SDValue visitBSWAP(SDNode *N);
262     SDValue visitBITREVERSE(SDNode *N);
263     SDValue visitCTLZ(SDNode *N);
264     SDValue visitCTLZ_ZERO_UNDEF(SDNode *N);
265     SDValue visitCTTZ(SDNode *N);
266     SDValue visitCTTZ_ZERO_UNDEF(SDNode *N);
267     SDValue visitCTPOP(SDNode *N);
268     SDValue visitSELECT(SDNode *N);
269     SDValue visitVSELECT(SDNode *N);
270     SDValue visitSELECT_CC(SDNode *N);
271     SDValue visitSETCC(SDNode *N);
272     SDValue visitSETCCE(SDNode *N);
273     SDValue visitSIGN_EXTEND(SDNode *N);
274     SDValue visitZERO_EXTEND(SDNode *N);
275     SDValue visitANY_EXTEND(SDNode *N);
276     SDValue visitSIGN_EXTEND_INREG(SDNode *N);
277     SDValue visitSIGN_EXTEND_VECTOR_INREG(SDNode *N);
278     SDValue visitZERO_EXTEND_VECTOR_INREG(SDNode *N);
279     SDValue visitTRUNCATE(SDNode *N);
280     SDValue visitBITCAST(SDNode *N);
281     SDValue visitBUILD_PAIR(SDNode *N);
282     SDValue visitFADD(SDNode *N);
283     SDValue visitFSUB(SDNode *N);
284     SDValue visitFMUL(SDNode *N);
285     SDValue visitFMA(SDNode *N);
286     SDValue visitFDIV(SDNode *N);
287     SDValue visitFREM(SDNode *N);
288     SDValue visitFSQRT(SDNode *N);
289     SDValue visitFCOPYSIGN(SDNode *N);
290     SDValue visitSINT_TO_FP(SDNode *N);
291     SDValue visitUINT_TO_FP(SDNode *N);
292     SDValue visitFP_TO_SINT(SDNode *N);
293     SDValue visitFP_TO_UINT(SDNode *N);
294     SDValue visitFP_ROUND(SDNode *N);
295     SDValue visitFP_ROUND_INREG(SDNode *N);
296     SDValue visitFP_EXTEND(SDNode *N);
297     SDValue visitFNEG(SDNode *N);
298     SDValue visitFABS(SDNode *N);
299     SDValue visitFCEIL(SDNode *N);
300     SDValue visitFTRUNC(SDNode *N);
301     SDValue visitFFLOOR(SDNode *N);
302     SDValue visitFMINNUM(SDNode *N);
303     SDValue visitFMAXNUM(SDNode *N);
304     SDValue visitBRCOND(SDNode *N);
305     SDValue visitBR_CC(SDNode *N);
306     SDValue visitLOAD(SDNode *N);
307 
308     SDValue replaceStoreChain(StoreSDNode *ST, SDValue BetterChain);
309     SDValue replaceStoreOfFPConstant(StoreSDNode *ST);
310 
311     SDValue visitSTORE(SDNode *N);
312     SDValue visitINSERT_VECTOR_ELT(SDNode *N);
313     SDValue visitEXTRACT_VECTOR_ELT(SDNode *N);
314     SDValue visitBUILD_VECTOR(SDNode *N);
315     SDValue visitCONCAT_VECTORS(SDNode *N);
316     SDValue visitEXTRACT_SUBVECTOR(SDNode *N);
317     SDValue visitVECTOR_SHUFFLE(SDNode *N);
318     SDValue visitSCALAR_TO_VECTOR(SDNode *N);
319     SDValue visitINSERT_SUBVECTOR(SDNode *N);
320     SDValue visitMLOAD(SDNode *N);
321     SDValue visitMSTORE(SDNode *N);
322     SDValue visitMGATHER(SDNode *N);
323     SDValue visitMSCATTER(SDNode *N);
324     SDValue visitFP_TO_FP16(SDNode *N);
325     SDValue visitFP16_TO_FP(SDNode *N);
326 
327     SDValue visitFADDForFMACombine(SDNode *N);
328     SDValue visitFSUBForFMACombine(SDNode *N);
329     SDValue visitFMULForFMACombine(SDNode *N);
330 
331     SDValue XformToShuffleWithZero(SDNode *N);
332     SDValue ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue LHS,
333                            SDValue RHS);
334 
335     SDValue visitShiftByConstant(SDNode *N, ConstantSDNode *Amt);
336 
337     SDValue foldSelectOfConstants(SDNode *N);
338     bool SimplifySelectOps(SDNode *SELECT, SDValue LHS, SDValue RHS);
339     SDValue SimplifyBinOpWithSameOpcodeHands(SDNode *N);
340     SDValue SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2);
341     SDValue SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
342                              SDValue N2, SDValue N3, ISD::CondCode CC,
343                              bool NotExtCompare = false);
344     SDValue SimplifySetCC(EVT VT, SDValue N0, SDValue N1, ISD::CondCode Cond,
345                           const SDLoc &DL, bool foldBooleans = true);
346 
347     bool isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
348                            SDValue &CC) const;
349     bool isOneUseSetCC(SDValue N) const;
350 
351     SDValue SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
352                                          unsigned HiOp);
353     SDValue CombineConsecutiveLoads(SDNode *N, EVT VT);
354     SDValue CombineExtLoad(SDNode *N);
355     SDValue combineRepeatedFPDivisors(SDNode *N);
356     SDValue ConstantFoldBITCASTofBUILD_VECTOR(SDNode *, EVT);
357     SDValue BuildSDIV(SDNode *N);
358     SDValue BuildSDIVPow2(SDNode *N);
359     SDValue BuildUDIV(SDNode *N);
360     SDValue BuildReciprocalEstimate(SDValue Op, SDNodeFlags *Flags);
361     SDValue buildRsqrtEstimate(SDValue Op, SDNodeFlags *Flags);
362     SDValue buildSqrtEstimate(SDValue Op, SDNodeFlags *Flags);
363     SDValue buildSqrtEstimateImpl(SDValue Op, SDNodeFlags *Flags, bool Recip);
364     SDValue buildSqrtNROneConst(SDValue Op, SDValue Est, unsigned Iterations,
365                                 SDNodeFlags *Flags, bool Reciprocal);
366     SDValue buildSqrtNRTwoConst(SDValue Op, SDValue Est, unsigned Iterations,
367                                 SDNodeFlags *Flags, bool Reciprocal);
368     SDValue MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
369                                bool DemandHighBits = true);
370     SDValue MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1);
371     SDNode *MatchRotatePosNeg(SDValue Shifted, SDValue Pos, SDValue Neg,
372                               SDValue InnerPos, SDValue InnerNeg,
373                               unsigned PosOpcode, unsigned NegOpcode,
374                               const SDLoc &DL);
375     SDNode *MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL);
376     SDValue ReduceLoadWidth(SDNode *N);
377     SDValue ReduceLoadOpStoreWidth(SDNode *N);
378     SDValue splitMergedValStore(StoreSDNode *ST);
379     SDValue TransformFPLoadStorePair(SDNode *N);
380     SDValue reduceBuildVecExtToExtBuildVec(SDNode *N);
381     SDValue reduceBuildVecConvertToConvertBuildVec(SDNode *N);
382     SDValue reduceBuildVecToShuffle(SDNode *N);
383     SDValue createBuildVecShuffle(SDLoc DL, SDNode *N, ArrayRef<int> VectorMask,
384                                   SDValue VecIn1, SDValue VecIn2,
385                                   unsigned LeftIdx);
386 
387     SDValue GetDemandedBits(SDValue V, const APInt &Mask);
388 
389     /// Walk up chain skipping non-aliasing memory nodes,
390     /// looking for aliasing nodes and adding them to the Aliases vector.
391     void GatherAllAliases(SDNode *N, SDValue OriginalChain,
392                           SmallVectorImpl<SDValue> &Aliases);
393 
394     /// Return true if there is any possibility that the two addresses overlap.
395     bool isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const;
396 
397     /// Walk up chain skipping non-aliasing memory nodes, looking for a better
398     /// chain (aliasing node.)
399     SDValue FindBetterChain(SDNode *N, SDValue Chain);
400 
401     /// Try to replace a store and any possibly adjacent stores on
402     /// consecutive chains with better chains. Return true only if St is
403     /// replaced.
404     ///
405     /// Notice that other chains may still be replaced even if the function
406     /// returns false.
407     bool findBetterNeighborChains(StoreSDNode *St);
408 
409     /// Match "(X shl/srl V1) & V2" where V2 may not be present.
410     bool MatchRotateHalf(SDValue Op, SDValue &Shift, SDValue &Mask);
411 
412     /// Holds a pointer to an LSBaseSDNode as well as information on where it
413     /// is located in a sequence of memory operations connected by a chain.
414     struct MemOpLink {
415       MemOpLink (LSBaseSDNode *N, int64_t Offset, unsigned Seq):
416       MemNode(N), OffsetFromBase(Offset), SequenceNum(Seq) { }
417       // Ptr to the mem node.
418       LSBaseSDNode *MemNode;
419       // Offset from the base ptr.
420       int64_t OffsetFromBase;
421       // What is the sequence number of this mem node.
422       // Lowest mem operand in the DAG starts at zero.
423       unsigned SequenceNum;
424     };
425 
426     /// This is a helper function for visitMUL to check the profitability
427     /// of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
428     /// MulNode is the original multiply, AddNode is (add x, c1),
429     /// and ConstNode is c2.
430     bool isMulAddWithConstProfitable(SDNode *MulNode,
431                                      SDValue &AddNode,
432                                      SDValue &ConstNode);
433 
434     /// This is a helper function for MergeStoresOfConstantsOrVecElts. Returns a
435     /// constant build_vector of the stored constant values in Stores.
436     SDValue getMergedConstantVectorStore(SelectionDAG &DAG, const SDLoc &SL,
437                                          ArrayRef<MemOpLink> Stores,
438                                          SmallVectorImpl<SDValue> &Chains,
439                                          EVT Ty) const;
440 
441     /// This is a helper function for visitAND and visitZERO_EXTEND.  Returns
442     /// true if the (and (load x) c) pattern matches an extload.  ExtVT returns
443     /// the type of the loaded value to be extended.  LoadedVT returns the type
444     /// of the original loaded value.  NarrowLoad returns whether the load would
445     /// need to be narrowed in order to match.
446     bool isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
447                           EVT LoadResultTy, EVT &ExtVT, EVT &LoadedVT,
448                           bool &NarrowLoad);
449 
450     /// This is a helper function for MergeConsecutiveStores. When the source
451     /// elements of the consecutive stores are all constants or all extracted
452     /// vector elements, try to merge them into one larger store.
453     /// \return True if a merged store was created.
454     bool MergeStoresOfConstantsOrVecElts(SmallVectorImpl<MemOpLink> &StoreNodes,
455                                          EVT MemVT, unsigned NumStores,
456                                          bool IsConstantSrc, bool UseVector);
457 
458     /// This is a helper function for MergeConsecutiveStores.
459     /// Stores that may be merged are placed in StoreNodes.
460     /// Loads that may alias with those stores are placed in AliasLoadNodes.
461     void getStoreMergeAndAliasCandidates(
462         StoreSDNode* St, SmallVectorImpl<MemOpLink> &StoreNodes,
463         SmallVectorImpl<LSBaseSDNode*> &AliasLoadNodes);
464 
465     /// Helper function for MergeConsecutiveStores. Checks if
466     /// Candidate stores have indirect dependency through their
467     /// operands. \return True if safe to merge
468     bool checkMergeStoreCandidatesForDependencies(
469         SmallVectorImpl<MemOpLink> &StoreNodes);
470 
471     /// Merge consecutive store operations into a wide store.
472     /// This optimization uses wide integers or vectors when possible.
473     /// \return True if some memory operations were changed.
474     bool MergeConsecutiveStores(StoreSDNode *N);
475 
476     /// \brief Try to transform a truncation where C is a constant:
477     ///     (trunc (and X, C)) -> (and (trunc X), (trunc C))
478     ///
479     /// \p N needs to be a truncation and its first operand an AND. Other
480     /// requirements are checked by the function (e.g. that trunc is
481     /// single-use) and if missed an empty SDValue is returned.
482     SDValue distributeTruncateThroughAnd(SDNode *N);
483 
484   public:
485     DAGCombiner(SelectionDAG &D, AliasAnalysis &A, CodeGenOpt::Level OL)
486         : DAG(D), TLI(D.getTargetLoweringInfo()), Level(BeforeLegalizeTypes),
487           OptLevel(OL), LegalOperations(false), LegalTypes(false), AA(A) {
488       ForCodeSize = DAG.getMachineFunction().getFunction()->optForSize();
489     }
490 
491     /// Runs the dag combiner on all nodes in the work list
492     void Run(CombineLevel AtLevel);
493 
494     SelectionDAG &getDAG() const { return DAG; }
495 
496     /// Returns a type large enough to hold any valid shift amount - before type
497     /// legalization these can be huge.
498     EVT getShiftAmountTy(EVT LHSTy) {
499       assert(LHSTy.isInteger() && "Shift amount is not an integer type!");
500       if (LHSTy.isVector())
501         return LHSTy;
502       auto &DL = DAG.getDataLayout();
503       return LegalTypes ? TLI.getScalarShiftAmountTy(DL, LHSTy)
504                         : TLI.getPointerTy(DL);
505     }
506 
507     /// This method returns true if we are running before type legalization or
508     /// if the specified VT is legal.
509     bool isTypeLegal(const EVT &VT) {
510       if (!LegalTypes) return true;
511       return TLI.isTypeLegal(VT);
512     }
513 
514     /// Convenience wrapper around TargetLowering::getSetCCResultType
515     EVT getSetCCResultType(EVT VT) const {
516       return TLI.getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
517     }
518   };
519 }
520 
521 
522 namespace {
523 /// This class is a DAGUpdateListener that removes any deleted
524 /// nodes from the worklist.
525 class WorklistRemover : public SelectionDAG::DAGUpdateListener {
526   DAGCombiner &DC;
527 public:
528   explicit WorklistRemover(DAGCombiner &dc)
529     : SelectionDAG::DAGUpdateListener(dc.getDAG()), DC(dc) {}
530 
531   void NodeDeleted(SDNode *N, SDNode *E) override {
532     DC.removeFromWorklist(N);
533   }
534 };
535 }
536 
537 //===----------------------------------------------------------------------===//
538 //  TargetLowering::DAGCombinerInfo implementation
539 //===----------------------------------------------------------------------===//
540 
541 void TargetLowering::DAGCombinerInfo::AddToWorklist(SDNode *N) {
542   ((DAGCombiner*)DC)->AddToWorklist(N);
543 }
544 
545 SDValue TargetLowering::DAGCombinerInfo::
546 CombineTo(SDNode *N, ArrayRef<SDValue> To, bool AddTo) {
547   return ((DAGCombiner*)DC)->CombineTo(N, &To[0], To.size(), AddTo);
548 }
549 
550 SDValue TargetLowering::DAGCombinerInfo::
551 CombineTo(SDNode *N, SDValue Res, bool AddTo) {
552   return ((DAGCombiner*)DC)->CombineTo(N, Res, AddTo);
553 }
554 
555 
556 SDValue TargetLowering::DAGCombinerInfo::
557 CombineTo(SDNode *N, SDValue Res0, SDValue Res1, bool AddTo) {
558   return ((DAGCombiner*)DC)->CombineTo(N, Res0, Res1, AddTo);
559 }
560 
561 void TargetLowering::DAGCombinerInfo::
562 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
563   return ((DAGCombiner*)DC)->CommitTargetLoweringOpt(TLO);
564 }
565 
566 //===----------------------------------------------------------------------===//
567 // Helper Functions
568 //===----------------------------------------------------------------------===//
569 
570 void DAGCombiner::deleteAndRecombine(SDNode *N) {
571   removeFromWorklist(N);
572 
573   // If the operands of this node are only used by the node, they will now be
574   // dead. Make sure to re-visit them and recursively delete dead nodes.
575   for (const SDValue &Op : N->ops())
576     // For an operand generating multiple values, one of the values may
577     // become dead allowing further simplification (e.g. split index
578     // arithmetic from an indexed load).
579     if (Op->hasOneUse() || Op->getNumValues() > 1)
580       AddToWorklist(Op.getNode());
581 
582   DAG.DeleteNode(N);
583 }
584 
585 /// Return 1 if we can compute the negated form of the specified expression for
586 /// the same cost as the expression itself, or 2 if we can compute the negated
587 /// form more cheaply than the expression itself.
588 static char isNegatibleForFree(SDValue Op, bool LegalOperations,
589                                const TargetLowering &TLI,
590                                const TargetOptions *Options,
591                                unsigned Depth = 0) {
592   // fneg is removable even if it has multiple uses.
593   if (Op.getOpcode() == ISD::FNEG) return 2;
594 
595   // Don't allow anything with multiple uses.
596   if (!Op.hasOneUse()) return 0;
597 
598   // Don't recurse exponentially.
599   if (Depth > 6) return 0;
600 
601   switch (Op.getOpcode()) {
602   default: return false;
603   case ISD::ConstantFP:
604     // Don't invert constant FP values after legalize.  The negated constant
605     // isn't necessarily legal.
606     return LegalOperations ? 0 : 1;
607   case ISD::FADD:
608     // FIXME: determine better conditions for this xform.
609     if (!Options->UnsafeFPMath) return 0;
610 
611     // After operation legalization, it might not be legal to create new FSUBs.
612     if (LegalOperations &&
613         !TLI.isOperationLegalOrCustom(ISD::FSUB,  Op.getValueType()))
614       return 0;
615 
616     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
617     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
618                                     Options, Depth + 1))
619       return V;
620     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
621     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
622                               Depth + 1);
623   case ISD::FSUB:
624     // We can't turn -(A-B) into B-A when we honor signed zeros.
625     if (!Options->UnsafeFPMath) return 0;
626 
627     // fold (fneg (fsub A, B)) -> (fsub B, A)
628     return 1;
629 
630   case ISD::FMUL:
631   case ISD::FDIV:
632     if (Options->HonorSignDependentRoundingFPMath()) return 0;
633 
634     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y) or (fmul X, (fneg Y))
635     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
636                                     Options, Depth + 1))
637       return V;
638 
639     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
640                               Depth + 1);
641 
642   case ISD::FP_EXTEND:
643   case ISD::FP_ROUND:
644   case ISD::FSIN:
645     return isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI, Options,
646                               Depth + 1);
647   }
648 }
649 
650 /// If isNegatibleForFree returns true, return the newly negated expression.
651 static SDValue GetNegatedExpression(SDValue Op, SelectionDAG &DAG,
652                                     bool LegalOperations, unsigned Depth = 0) {
653   const TargetOptions &Options = DAG.getTarget().Options;
654   // fneg is removable even if it has multiple uses.
655   if (Op.getOpcode() == ISD::FNEG) return Op.getOperand(0);
656 
657   // Don't allow anything with multiple uses.
658   assert(Op.hasOneUse() && "Unknown reuse!");
659 
660   assert(Depth <= 6 && "GetNegatedExpression doesn't match isNegatibleForFree");
661 
662   const SDNodeFlags *Flags = Op.getNode()->getFlags();
663 
664   switch (Op.getOpcode()) {
665   default: llvm_unreachable("Unknown code");
666   case ISD::ConstantFP: {
667     APFloat V = cast<ConstantFPSDNode>(Op)->getValueAPF();
668     V.changeSign();
669     return DAG.getConstantFP(V, SDLoc(Op), Op.getValueType());
670   }
671   case ISD::FADD:
672     // FIXME: determine better conditions for this xform.
673     assert(Options.UnsafeFPMath);
674 
675     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
676     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
677                            DAG.getTargetLoweringInfo(), &Options, Depth+1))
678       return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
679                          GetNegatedExpression(Op.getOperand(0), DAG,
680                                               LegalOperations, Depth+1),
681                          Op.getOperand(1), Flags);
682     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
683     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
684                        GetNegatedExpression(Op.getOperand(1), DAG,
685                                             LegalOperations, Depth+1),
686                        Op.getOperand(0), Flags);
687   case ISD::FSUB:
688     // We can't turn -(A-B) into B-A when we honor signed zeros.
689     assert(Options.UnsafeFPMath);
690 
691     // fold (fneg (fsub 0, B)) -> B
692     if (ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(Op.getOperand(0)))
693       if (N0CFP->isZero())
694         return Op.getOperand(1);
695 
696     // fold (fneg (fsub A, B)) -> (fsub B, A)
697     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
698                        Op.getOperand(1), Op.getOperand(0), Flags);
699 
700   case ISD::FMUL:
701   case ISD::FDIV:
702     assert(!Options.HonorSignDependentRoundingFPMath());
703 
704     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y)
705     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
706                            DAG.getTargetLoweringInfo(), &Options, Depth+1))
707       return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
708                          GetNegatedExpression(Op.getOperand(0), DAG,
709                                               LegalOperations, Depth+1),
710                          Op.getOperand(1), Flags);
711 
712     // fold (fneg (fmul X, Y)) -> (fmul X, (fneg Y))
713     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
714                        Op.getOperand(0),
715                        GetNegatedExpression(Op.getOperand(1), DAG,
716                                             LegalOperations, Depth+1), Flags);
717 
718   case ISD::FP_EXTEND:
719   case ISD::FSIN:
720     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
721                        GetNegatedExpression(Op.getOperand(0), DAG,
722                                             LegalOperations, Depth+1));
723   case ISD::FP_ROUND:
724       return DAG.getNode(ISD::FP_ROUND, SDLoc(Op), Op.getValueType(),
725                          GetNegatedExpression(Op.getOperand(0), DAG,
726                                               LegalOperations, Depth+1),
727                          Op.getOperand(1));
728   }
729 }
730 
731 // APInts must be the same size for most operations, this helper
732 // function zero extends the shorter of the pair so that they match.
733 // We provide an Offset so that we can create bitwidths that won't overflow.
734 static void zeroExtendToMatch(APInt &LHS, APInt &RHS, unsigned Offset = 0) {
735   unsigned Bits = Offset + std::max(LHS.getBitWidth(), RHS.getBitWidth());
736   LHS = LHS.zextOrSelf(Bits);
737   RHS = RHS.zextOrSelf(Bits);
738 }
739 
740 // Return true if this node is a setcc, or is a select_cc
741 // that selects between the target values used for true and false, making it
742 // equivalent to a setcc. Also, set the incoming LHS, RHS, and CC references to
743 // the appropriate nodes based on the type of node we are checking. This
744 // simplifies life a bit for the callers.
745 bool DAGCombiner::isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
746                                     SDValue &CC) const {
747   if (N.getOpcode() == ISD::SETCC) {
748     LHS = N.getOperand(0);
749     RHS = N.getOperand(1);
750     CC  = N.getOperand(2);
751     return true;
752   }
753 
754   if (N.getOpcode() != ISD::SELECT_CC ||
755       !TLI.isConstTrueVal(N.getOperand(2).getNode()) ||
756       !TLI.isConstFalseVal(N.getOperand(3).getNode()))
757     return false;
758 
759   if (TLI.getBooleanContents(N.getValueType()) ==
760       TargetLowering::UndefinedBooleanContent)
761     return false;
762 
763   LHS = N.getOperand(0);
764   RHS = N.getOperand(1);
765   CC  = N.getOperand(4);
766   return true;
767 }
768 
769 /// Return true if this is a SetCC-equivalent operation with only one use.
770 /// If this is true, it allows the users to invert the operation for free when
771 /// it is profitable to do so.
772 bool DAGCombiner::isOneUseSetCC(SDValue N) const {
773   SDValue N0, N1, N2;
774   if (isSetCCEquivalent(N, N0, N1, N2) && N.getNode()->hasOneUse())
775     return true;
776   return false;
777 }
778 
779 // \brief Returns the SDNode if it is a constant float BuildVector
780 // or constant float.
781 static SDNode *isConstantFPBuildVectorOrConstantFP(SDValue N) {
782   if (isa<ConstantFPSDNode>(N))
783     return N.getNode();
784   if (ISD::isBuildVectorOfConstantFPSDNodes(N.getNode()))
785     return N.getNode();
786   return nullptr;
787 }
788 
789 // \brief Returns the SDNode if it is a constant splat BuildVector or constant
790 // int.
791 static ConstantSDNode *isConstOrConstSplat(SDValue N) {
792   if (ConstantSDNode *CN = dyn_cast<ConstantSDNode>(N))
793     return CN;
794 
795   if (BuildVectorSDNode *BV = dyn_cast<BuildVectorSDNode>(N)) {
796     BitVector UndefElements;
797     ConstantSDNode *CN = BV->getConstantSplatNode(&UndefElements);
798 
799     // BuildVectors can truncate their operands. Ignore that case here.
800     // FIXME: We blindly ignore splats which include undef which is overly
801     // pessimistic.
802     if (CN && UndefElements.none() &&
803         CN->getValueType(0) == N.getValueType().getScalarType())
804       return CN;
805   }
806 
807   return nullptr;
808 }
809 
810 // \brief Returns the SDNode if it is a constant splat BuildVector or constant
811 // float.
812 static ConstantFPSDNode *isConstOrConstSplatFP(SDValue N) {
813   if (ConstantFPSDNode *CN = dyn_cast<ConstantFPSDNode>(N))
814     return CN;
815 
816   if (BuildVectorSDNode *BV = dyn_cast<BuildVectorSDNode>(N)) {
817     BitVector UndefElements;
818     ConstantFPSDNode *CN = BV->getConstantFPSplatNode(&UndefElements);
819 
820     if (CN && UndefElements.none())
821       return CN;
822   }
823 
824   return nullptr;
825 }
826 
827 SDValue DAGCombiner::ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue N0,
828                                     SDValue N1) {
829   EVT VT = N0.getValueType();
830   if (N0.getOpcode() == Opc) {
831     if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1))) {
832       if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
833         // reassoc. (op (op x, c1), c2) -> (op x, (op c1, c2))
834         if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, L, R))
835           return DAG.getNode(Opc, DL, VT, N0.getOperand(0), OpNode);
836         return SDValue();
837       }
838       if (N0.hasOneUse()) {
839         // reassoc. (op (op x, c1), y) -> (op (op x, y), c1) iff x+c1 has one
840         // use
841         SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0.getOperand(0), N1);
842         if (!OpNode.getNode())
843           return SDValue();
844         AddToWorklist(OpNode.getNode());
845         return DAG.getNode(Opc, DL, VT, OpNode, N0.getOperand(1));
846       }
847     }
848   }
849 
850   if (N1.getOpcode() == Opc) {
851     if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1.getOperand(1))) {
852       if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
853         // reassoc. (op c2, (op x, c1)) -> (op x, (op c1, c2))
854         if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, R, L))
855           return DAG.getNode(Opc, DL, VT, N1.getOperand(0), OpNode);
856         return SDValue();
857       }
858       if (N1.hasOneUse()) {
859         // reassoc. (op x, (op y, c1)) -> (op (op x, y), c1) iff x+c1 has one
860         // use
861         SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0, N1.getOperand(0));
862         if (!OpNode.getNode())
863           return SDValue();
864         AddToWorklist(OpNode.getNode());
865         return DAG.getNode(Opc, DL, VT, OpNode, N1.getOperand(1));
866       }
867     }
868   }
869 
870   return SDValue();
871 }
872 
873 SDValue DAGCombiner::CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
874                                bool AddTo) {
875   assert(N->getNumValues() == NumTo && "Broken CombineTo call!");
876   ++NodesCombined;
877   DEBUG(dbgs() << "\nReplacing.1 ";
878         N->dump(&DAG);
879         dbgs() << "\nWith: ";
880         To[0].getNode()->dump(&DAG);
881         dbgs() << " and " << NumTo-1 << " other values\n");
882   for (unsigned i = 0, e = NumTo; i != e; ++i)
883     assert((!To[i].getNode() ||
884             N->getValueType(i) == To[i].getValueType()) &&
885            "Cannot combine value to value of different type!");
886 
887   WorklistRemover DeadNodes(*this);
888   DAG.ReplaceAllUsesWith(N, To);
889   if (AddTo) {
890     // Push the new nodes and any users onto the worklist
891     for (unsigned i = 0, e = NumTo; i != e; ++i) {
892       if (To[i].getNode()) {
893         AddToWorklist(To[i].getNode());
894         AddUsersToWorklist(To[i].getNode());
895       }
896     }
897   }
898 
899   // Finally, if the node is now dead, remove it from the graph.  The node
900   // may not be dead if the replacement process recursively simplified to
901   // something else needing this node.
902   if (N->use_empty())
903     deleteAndRecombine(N);
904   return SDValue(N, 0);
905 }
906 
907 void DAGCombiner::
908 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
909   // Replace all uses.  If any nodes become isomorphic to other nodes and
910   // are deleted, make sure to remove them from our worklist.
911   WorklistRemover DeadNodes(*this);
912   DAG.ReplaceAllUsesOfValueWith(TLO.Old, TLO.New);
913 
914   // Push the new node and any (possibly new) users onto the worklist.
915   AddToWorklist(TLO.New.getNode());
916   AddUsersToWorklist(TLO.New.getNode());
917 
918   // Finally, if the node is now dead, remove it from the graph.  The node
919   // may not be dead if the replacement process recursively simplified to
920   // something else needing this node.
921   if (TLO.Old.getNode()->use_empty())
922     deleteAndRecombine(TLO.Old.getNode());
923 }
924 
925 /// Check the specified integer node value to see if it can be simplified or if
926 /// things it uses can be simplified by bit propagation. If so, return true.
927 bool DAGCombiner::SimplifyDemandedBits(SDValue Op, const APInt &Demanded) {
928   TargetLowering::TargetLoweringOpt TLO(DAG, LegalTypes, LegalOperations);
929   APInt KnownZero, KnownOne;
930   if (!TLI.SimplifyDemandedBits(Op, Demanded, KnownZero, KnownOne, TLO))
931     return false;
932 
933   // Revisit the node.
934   AddToWorklist(Op.getNode());
935 
936   // Replace the old value with the new one.
937   ++NodesCombined;
938   DEBUG(dbgs() << "\nReplacing.2 ";
939         TLO.Old.getNode()->dump(&DAG);
940         dbgs() << "\nWith: ";
941         TLO.New.getNode()->dump(&DAG);
942         dbgs() << '\n');
943 
944   CommitTargetLoweringOpt(TLO);
945   return true;
946 }
947 
948 void DAGCombiner::ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad) {
949   SDLoc DL(Load);
950   EVT VT = Load->getValueType(0);
951   SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, VT, SDValue(ExtLoad, 0));
952 
953   DEBUG(dbgs() << "\nReplacing.9 ";
954         Load->dump(&DAG);
955         dbgs() << "\nWith: ";
956         Trunc.getNode()->dump(&DAG);
957         dbgs() << '\n');
958   WorklistRemover DeadNodes(*this);
959   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 0), Trunc);
960   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), SDValue(ExtLoad, 1));
961   deleteAndRecombine(Load);
962   AddToWorklist(Trunc.getNode());
963 }
964 
965 SDValue DAGCombiner::PromoteOperand(SDValue Op, EVT PVT, bool &Replace) {
966   Replace = false;
967   SDLoc DL(Op);
968   if (ISD::isUNINDEXEDLoad(Op.getNode())) {
969     LoadSDNode *LD = cast<LoadSDNode>(Op);
970     EVT MemVT = LD->getMemoryVT();
971     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD)
972       ? (TLI.isLoadExtLegal(ISD::ZEXTLOAD, PVT, MemVT) ? ISD::ZEXTLOAD
973                                                        : ISD::EXTLOAD)
974       : LD->getExtensionType();
975     Replace = true;
976     return DAG.getExtLoad(ExtType, DL, PVT,
977                           LD->getChain(), LD->getBasePtr(),
978                           MemVT, LD->getMemOperand());
979   }
980 
981   unsigned Opc = Op.getOpcode();
982   switch (Opc) {
983   default: break;
984   case ISD::AssertSext:
985     return DAG.getNode(ISD::AssertSext, DL, PVT,
986                        SExtPromoteOperand(Op.getOperand(0), PVT),
987                        Op.getOperand(1));
988   case ISD::AssertZext:
989     return DAG.getNode(ISD::AssertZext, DL, PVT,
990                        ZExtPromoteOperand(Op.getOperand(0), PVT),
991                        Op.getOperand(1));
992   case ISD::Constant: {
993     unsigned ExtOpc =
994       Op.getValueType().isByteSized() ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
995     return DAG.getNode(ExtOpc, DL, PVT, Op);
996   }
997   }
998 
999   if (!TLI.isOperationLegal(ISD::ANY_EXTEND, PVT))
1000     return SDValue();
1001   return DAG.getNode(ISD::ANY_EXTEND, DL, PVT, Op);
1002 }
1003 
1004 SDValue DAGCombiner::SExtPromoteOperand(SDValue Op, EVT PVT) {
1005   if (!TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, PVT))
1006     return SDValue();
1007   EVT OldVT = Op.getValueType();
1008   SDLoc DL(Op);
1009   bool Replace = false;
1010   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1011   if (!NewOp.getNode())
1012     return SDValue();
1013   AddToWorklist(NewOp.getNode());
1014 
1015   if (Replace)
1016     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1017   return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, NewOp.getValueType(), NewOp,
1018                      DAG.getValueType(OldVT));
1019 }
1020 
1021 SDValue DAGCombiner::ZExtPromoteOperand(SDValue Op, EVT PVT) {
1022   EVT OldVT = Op.getValueType();
1023   SDLoc DL(Op);
1024   bool Replace = false;
1025   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1026   if (!NewOp.getNode())
1027     return SDValue();
1028   AddToWorklist(NewOp.getNode());
1029 
1030   if (Replace)
1031     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1032   return DAG.getZeroExtendInReg(NewOp, DL, OldVT);
1033 }
1034 
1035 /// Promote the specified integer binary operation if the target indicates it is
1036 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1037 /// i32 since i16 instructions are longer.
1038 SDValue DAGCombiner::PromoteIntBinOp(SDValue Op) {
1039   if (!LegalOperations)
1040     return SDValue();
1041 
1042   EVT VT = Op.getValueType();
1043   if (VT.isVector() || !VT.isInteger())
1044     return SDValue();
1045 
1046   // If operation type is 'undesirable', e.g. i16 on x86, consider
1047   // promoting it.
1048   unsigned Opc = Op.getOpcode();
1049   if (TLI.isTypeDesirableForOp(Opc, VT))
1050     return SDValue();
1051 
1052   EVT PVT = VT;
1053   // Consult target whether it is a good idea to promote this operation and
1054   // what's the right type to promote it to.
1055   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1056     assert(PVT != VT && "Don't know what type to promote to!");
1057 
1058     bool Replace0 = false;
1059     SDValue N0 = Op.getOperand(0);
1060     SDValue NN0 = PromoteOperand(N0, PVT, Replace0);
1061     if (!NN0.getNode())
1062       return SDValue();
1063 
1064     bool Replace1 = false;
1065     SDValue N1 = Op.getOperand(1);
1066     SDValue NN1;
1067     if (N0 == N1)
1068       NN1 = NN0;
1069     else {
1070       NN1 = PromoteOperand(N1, PVT, Replace1);
1071       if (!NN1.getNode())
1072         return SDValue();
1073     }
1074 
1075     AddToWorklist(NN0.getNode());
1076     if (NN1.getNode())
1077       AddToWorklist(NN1.getNode());
1078 
1079     if (Replace0)
1080       ReplaceLoadWithPromotedLoad(N0.getNode(), NN0.getNode());
1081     if (Replace1)
1082       ReplaceLoadWithPromotedLoad(N1.getNode(), NN1.getNode());
1083 
1084     DEBUG(dbgs() << "\nPromoting ";
1085           Op.getNode()->dump(&DAG));
1086     SDLoc DL(Op);
1087     return DAG.getNode(ISD::TRUNCATE, DL, VT,
1088                        DAG.getNode(Opc, DL, PVT, NN0, NN1));
1089   }
1090   return SDValue();
1091 }
1092 
1093 /// Promote the specified integer shift operation if the target indicates it is
1094 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1095 /// i32 since i16 instructions are longer.
1096 SDValue DAGCombiner::PromoteIntShiftOp(SDValue Op) {
1097   if (!LegalOperations)
1098     return SDValue();
1099 
1100   EVT VT = Op.getValueType();
1101   if (VT.isVector() || !VT.isInteger())
1102     return SDValue();
1103 
1104   // If operation type is 'undesirable', e.g. i16 on x86, consider
1105   // promoting it.
1106   unsigned Opc = Op.getOpcode();
1107   if (TLI.isTypeDesirableForOp(Opc, VT))
1108     return SDValue();
1109 
1110   EVT PVT = VT;
1111   // Consult target whether it is a good idea to promote this operation and
1112   // what's the right type to promote it to.
1113   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1114     assert(PVT != VT && "Don't know what type to promote to!");
1115 
1116     bool Replace = false;
1117     SDValue N0 = Op.getOperand(0);
1118     if (Opc == ISD::SRA)
1119       N0 = SExtPromoteOperand(Op.getOperand(0), PVT);
1120     else if (Opc == ISD::SRL)
1121       N0 = ZExtPromoteOperand(Op.getOperand(0), PVT);
1122     else
1123       N0 = PromoteOperand(N0, PVT, Replace);
1124     if (!N0.getNode())
1125       return SDValue();
1126 
1127     AddToWorklist(N0.getNode());
1128     if (Replace)
1129       ReplaceLoadWithPromotedLoad(Op.getOperand(0).getNode(), N0.getNode());
1130 
1131     DEBUG(dbgs() << "\nPromoting ";
1132           Op.getNode()->dump(&DAG));
1133     SDLoc DL(Op);
1134     return DAG.getNode(ISD::TRUNCATE, DL, VT,
1135                        DAG.getNode(Opc, DL, PVT, N0, Op.getOperand(1)));
1136   }
1137   return SDValue();
1138 }
1139 
1140 SDValue DAGCombiner::PromoteExtend(SDValue Op) {
1141   if (!LegalOperations)
1142     return SDValue();
1143 
1144   EVT VT = Op.getValueType();
1145   if (VT.isVector() || !VT.isInteger())
1146     return SDValue();
1147 
1148   // If operation type is 'undesirable', e.g. i16 on x86, consider
1149   // promoting it.
1150   unsigned Opc = Op.getOpcode();
1151   if (TLI.isTypeDesirableForOp(Opc, VT))
1152     return SDValue();
1153 
1154   EVT PVT = VT;
1155   // Consult target whether it is a good idea to promote this operation and
1156   // what's the right type to promote it to.
1157   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1158     assert(PVT != VT && "Don't know what type to promote to!");
1159     // fold (aext (aext x)) -> (aext x)
1160     // fold (aext (zext x)) -> (zext x)
1161     // fold (aext (sext x)) -> (sext x)
1162     DEBUG(dbgs() << "\nPromoting ";
1163           Op.getNode()->dump(&DAG));
1164     return DAG.getNode(Op.getOpcode(), SDLoc(Op), VT, Op.getOperand(0));
1165   }
1166   return SDValue();
1167 }
1168 
1169 bool DAGCombiner::PromoteLoad(SDValue Op) {
1170   if (!LegalOperations)
1171     return false;
1172 
1173   if (!ISD::isUNINDEXEDLoad(Op.getNode()))
1174     return false;
1175 
1176   EVT VT = Op.getValueType();
1177   if (VT.isVector() || !VT.isInteger())
1178     return false;
1179 
1180   // If operation type is 'undesirable', e.g. i16 on x86, consider
1181   // promoting it.
1182   unsigned Opc = Op.getOpcode();
1183   if (TLI.isTypeDesirableForOp(Opc, VT))
1184     return false;
1185 
1186   EVT PVT = VT;
1187   // Consult target whether it is a good idea to promote this operation and
1188   // what's the right type to promote it to.
1189   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1190     assert(PVT != VT && "Don't know what type to promote to!");
1191 
1192     SDLoc DL(Op);
1193     SDNode *N = Op.getNode();
1194     LoadSDNode *LD = cast<LoadSDNode>(N);
1195     EVT MemVT = LD->getMemoryVT();
1196     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD)
1197       ? (TLI.isLoadExtLegal(ISD::ZEXTLOAD, PVT, MemVT) ? ISD::ZEXTLOAD
1198                                                        : ISD::EXTLOAD)
1199       : LD->getExtensionType();
1200     SDValue NewLD = DAG.getExtLoad(ExtType, DL, PVT,
1201                                    LD->getChain(), LD->getBasePtr(),
1202                                    MemVT, LD->getMemOperand());
1203     SDValue Result = DAG.getNode(ISD::TRUNCATE, DL, VT, NewLD);
1204 
1205     DEBUG(dbgs() << "\nPromoting ";
1206           N->dump(&DAG);
1207           dbgs() << "\nTo: ";
1208           Result.getNode()->dump(&DAG);
1209           dbgs() << '\n');
1210     WorklistRemover DeadNodes(*this);
1211     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result);
1212     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), NewLD.getValue(1));
1213     deleteAndRecombine(N);
1214     AddToWorklist(Result.getNode());
1215     return true;
1216   }
1217   return false;
1218 }
1219 
1220 /// \brief Recursively delete a node which has no uses and any operands for
1221 /// which it is the only use.
1222 ///
1223 /// Note that this both deletes the nodes and removes them from the worklist.
1224 /// It also adds any nodes who have had a user deleted to the worklist as they
1225 /// may now have only one use and subject to other combines.
1226 bool DAGCombiner::recursivelyDeleteUnusedNodes(SDNode *N) {
1227   if (!N->use_empty())
1228     return false;
1229 
1230   SmallSetVector<SDNode *, 16> Nodes;
1231   Nodes.insert(N);
1232   do {
1233     N = Nodes.pop_back_val();
1234     if (!N)
1235       continue;
1236 
1237     if (N->use_empty()) {
1238       for (const SDValue &ChildN : N->op_values())
1239         Nodes.insert(ChildN.getNode());
1240 
1241       removeFromWorklist(N);
1242       DAG.DeleteNode(N);
1243     } else {
1244       AddToWorklist(N);
1245     }
1246   } while (!Nodes.empty());
1247   return true;
1248 }
1249 
1250 //===----------------------------------------------------------------------===//
1251 //  Main DAG Combiner implementation
1252 //===----------------------------------------------------------------------===//
1253 
1254 void DAGCombiner::Run(CombineLevel AtLevel) {
1255   // set the instance variables, so that the various visit routines may use it.
1256   Level = AtLevel;
1257   LegalOperations = Level >= AfterLegalizeVectorOps;
1258   LegalTypes = Level >= AfterLegalizeTypes;
1259 
1260   // Add all the dag nodes to the worklist.
1261   for (SDNode &Node : DAG.allnodes())
1262     AddToWorklist(&Node);
1263 
1264   // Create a dummy node (which is not added to allnodes), that adds a reference
1265   // to the root node, preventing it from being deleted, and tracking any
1266   // changes of the root.
1267   HandleSDNode Dummy(DAG.getRoot());
1268 
1269   // While the worklist isn't empty, find a node and try to combine it.
1270   while (!WorklistMap.empty()) {
1271     SDNode *N;
1272     // The Worklist holds the SDNodes in order, but it may contain null entries.
1273     do {
1274       N = Worklist.pop_back_val();
1275     } while (!N);
1276 
1277     bool GoodWorklistEntry = WorklistMap.erase(N);
1278     (void)GoodWorklistEntry;
1279     assert(GoodWorklistEntry &&
1280            "Found a worklist entry without a corresponding map entry!");
1281 
1282     // If N has no uses, it is dead.  Make sure to revisit all N's operands once
1283     // N is deleted from the DAG, since they too may now be dead or may have a
1284     // reduced number of uses, allowing other xforms.
1285     if (recursivelyDeleteUnusedNodes(N))
1286       continue;
1287 
1288     WorklistRemover DeadNodes(*this);
1289 
1290     // If this combine is running after legalizing the DAG, re-legalize any
1291     // nodes pulled off the worklist.
1292     if (Level == AfterLegalizeDAG) {
1293       SmallSetVector<SDNode *, 16> UpdatedNodes;
1294       bool NIsValid = DAG.LegalizeOp(N, UpdatedNodes);
1295 
1296       for (SDNode *LN : UpdatedNodes) {
1297         AddToWorklist(LN);
1298         AddUsersToWorklist(LN);
1299       }
1300       if (!NIsValid)
1301         continue;
1302     }
1303 
1304     DEBUG(dbgs() << "\nCombining: "; N->dump(&DAG));
1305 
1306     // Add any operands of the new node which have not yet been combined to the
1307     // worklist as well. Because the worklist uniques things already, this
1308     // won't repeatedly process the same operand.
1309     CombinedNodes.insert(N);
1310     for (const SDValue &ChildN : N->op_values())
1311       if (!CombinedNodes.count(ChildN.getNode()))
1312         AddToWorklist(ChildN.getNode());
1313 
1314     SDValue RV = combine(N);
1315 
1316     if (!RV.getNode())
1317       continue;
1318 
1319     ++NodesCombined;
1320 
1321     // If we get back the same node we passed in, rather than a new node or
1322     // zero, we know that the node must have defined multiple values and
1323     // CombineTo was used.  Since CombineTo takes care of the worklist
1324     // mechanics for us, we have no work to do in this case.
1325     if (RV.getNode() == N)
1326       continue;
1327 
1328     assert(N->getOpcode() != ISD::DELETED_NODE &&
1329            RV.getOpcode() != ISD::DELETED_NODE &&
1330            "Node was deleted but visit returned new node!");
1331 
1332     DEBUG(dbgs() << " ... into: ";
1333           RV.getNode()->dump(&DAG));
1334 
1335     if (N->getNumValues() == RV.getNode()->getNumValues())
1336       DAG.ReplaceAllUsesWith(N, RV.getNode());
1337     else {
1338       assert(N->getValueType(0) == RV.getValueType() &&
1339              N->getNumValues() == 1 && "Type mismatch");
1340       SDValue OpV = RV;
1341       DAG.ReplaceAllUsesWith(N, &OpV);
1342     }
1343 
1344     // Push the new node and any users onto the worklist
1345     AddToWorklist(RV.getNode());
1346     AddUsersToWorklist(RV.getNode());
1347 
1348     // Finally, if the node is now dead, remove it from the graph.  The node
1349     // may not be dead if the replacement process recursively simplified to
1350     // something else needing this node. This will also take care of adding any
1351     // operands which have lost a user to the worklist.
1352     recursivelyDeleteUnusedNodes(N);
1353   }
1354 
1355   // If the root changed (e.g. it was a dead load, update the root).
1356   DAG.setRoot(Dummy.getValue());
1357   DAG.RemoveDeadNodes();
1358 }
1359 
1360 SDValue DAGCombiner::visit(SDNode *N) {
1361   switch (N->getOpcode()) {
1362   default: break;
1363   case ISD::TokenFactor:        return visitTokenFactor(N);
1364   case ISD::MERGE_VALUES:       return visitMERGE_VALUES(N);
1365   case ISD::ADD:                return visitADD(N);
1366   case ISD::SUB:                return visitSUB(N);
1367   case ISD::ADDC:               return visitADDC(N);
1368   case ISD::SUBC:               return visitSUBC(N);
1369   case ISD::ADDE:               return visitADDE(N);
1370   case ISD::SUBE:               return visitSUBE(N);
1371   case ISD::MUL:                return visitMUL(N);
1372   case ISD::SDIV:               return visitSDIV(N);
1373   case ISD::UDIV:               return visitUDIV(N);
1374   case ISD::SREM:
1375   case ISD::UREM:               return visitREM(N);
1376   case ISD::MULHU:              return visitMULHU(N);
1377   case ISD::MULHS:              return visitMULHS(N);
1378   case ISD::SMUL_LOHI:          return visitSMUL_LOHI(N);
1379   case ISD::UMUL_LOHI:          return visitUMUL_LOHI(N);
1380   case ISD::SMULO:              return visitSMULO(N);
1381   case ISD::UMULO:              return visitUMULO(N);
1382   case ISD::SMIN:
1383   case ISD::SMAX:
1384   case ISD::UMIN:
1385   case ISD::UMAX:               return visitIMINMAX(N);
1386   case ISD::AND:                return visitAND(N);
1387   case ISD::OR:                 return visitOR(N);
1388   case ISD::XOR:                return visitXOR(N);
1389   case ISD::SHL:                return visitSHL(N);
1390   case ISD::SRA:                return visitSRA(N);
1391   case ISD::SRL:                return visitSRL(N);
1392   case ISD::ROTR:
1393   case ISD::ROTL:               return visitRotate(N);
1394   case ISD::BSWAP:              return visitBSWAP(N);
1395   case ISD::BITREVERSE:         return visitBITREVERSE(N);
1396   case ISD::CTLZ:               return visitCTLZ(N);
1397   case ISD::CTLZ_ZERO_UNDEF:    return visitCTLZ_ZERO_UNDEF(N);
1398   case ISD::CTTZ:               return visitCTTZ(N);
1399   case ISD::CTTZ_ZERO_UNDEF:    return visitCTTZ_ZERO_UNDEF(N);
1400   case ISD::CTPOP:              return visitCTPOP(N);
1401   case ISD::SELECT:             return visitSELECT(N);
1402   case ISD::VSELECT:            return visitVSELECT(N);
1403   case ISD::SELECT_CC:          return visitSELECT_CC(N);
1404   case ISD::SETCC:              return visitSETCC(N);
1405   case ISD::SETCCE:             return visitSETCCE(N);
1406   case ISD::SIGN_EXTEND:        return visitSIGN_EXTEND(N);
1407   case ISD::ZERO_EXTEND:        return visitZERO_EXTEND(N);
1408   case ISD::ANY_EXTEND:         return visitANY_EXTEND(N);
1409   case ISD::SIGN_EXTEND_INREG:  return visitSIGN_EXTEND_INREG(N);
1410   case ISD::SIGN_EXTEND_VECTOR_INREG: return visitSIGN_EXTEND_VECTOR_INREG(N);
1411   case ISD::ZERO_EXTEND_VECTOR_INREG: return visitZERO_EXTEND_VECTOR_INREG(N);
1412   case ISD::TRUNCATE:           return visitTRUNCATE(N);
1413   case ISD::BITCAST:            return visitBITCAST(N);
1414   case ISD::BUILD_PAIR:         return visitBUILD_PAIR(N);
1415   case ISD::FADD:               return visitFADD(N);
1416   case ISD::FSUB:               return visitFSUB(N);
1417   case ISD::FMUL:               return visitFMUL(N);
1418   case ISD::FMA:                return visitFMA(N);
1419   case ISD::FDIV:               return visitFDIV(N);
1420   case ISD::FREM:               return visitFREM(N);
1421   case ISD::FSQRT:              return visitFSQRT(N);
1422   case ISD::FCOPYSIGN:          return visitFCOPYSIGN(N);
1423   case ISD::SINT_TO_FP:         return visitSINT_TO_FP(N);
1424   case ISD::UINT_TO_FP:         return visitUINT_TO_FP(N);
1425   case ISD::FP_TO_SINT:         return visitFP_TO_SINT(N);
1426   case ISD::FP_TO_UINT:         return visitFP_TO_UINT(N);
1427   case ISD::FP_ROUND:           return visitFP_ROUND(N);
1428   case ISD::FP_ROUND_INREG:     return visitFP_ROUND_INREG(N);
1429   case ISD::FP_EXTEND:          return visitFP_EXTEND(N);
1430   case ISD::FNEG:               return visitFNEG(N);
1431   case ISD::FABS:               return visitFABS(N);
1432   case ISD::FFLOOR:             return visitFFLOOR(N);
1433   case ISD::FMINNUM:            return visitFMINNUM(N);
1434   case ISD::FMAXNUM:            return visitFMAXNUM(N);
1435   case ISD::FCEIL:              return visitFCEIL(N);
1436   case ISD::FTRUNC:             return visitFTRUNC(N);
1437   case ISD::BRCOND:             return visitBRCOND(N);
1438   case ISD::BR_CC:              return visitBR_CC(N);
1439   case ISD::LOAD:               return visitLOAD(N);
1440   case ISD::STORE:              return visitSTORE(N);
1441   case ISD::INSERT_VECTOR_ELT:  return visitINSERT_VECTOR_ELT(N);
1442   case ISD::EXTRACT_VECTOR_ELT: return visitEXTRACT_VECTOR_ELT(N);
1443   case ISD::BUILD_VECTOR:       return visitBUILD_VECTOR(N);
1444   case ISD::CONCAT_VECTORS:     return visitCONCAT_VECTORS(N);
1445   case ISD::EXTRACT_SUBVECTOR:  return visitEXTRACT_SUBVECTOR(N);
1446   case ISD::VECTOR_SHUFFLE:     return visitVECTOR_SHUFFLE(N);
1447   case ISD::SCALAR_TO_VECTOR:   return visitSCALAR_TO_VECTOR(N);
1448   case ISD::INSERT_SUBVECTOR:   return visitINSERT_SUBVECTOR(N);
1449   case ISD::MGATHER:            return visitMGATHER(N);
1450   case ISD::MLOAD:              return visitMLOAD(N);
1451   case ISD::MSCATTER:           return visitMSCATTER(N);
1452   case ISD::MSTORE:             return visitMSTORE(N);
1453   case ISD::FP_TO_FP16:         return visitFP_TO_FP16(N);
1454   case ISD::FP16_TO_FP:         return visitFP16_TO_FP(N);
1455   }
1456   return SDValue();
1457 }
1458 
1459 SDValue DAGCombiner::combine(SDNode *N) {
1460   SDValue RV = visit(N);
1461 
1462   // If nothing happened, try a target-specific DAG combine.
1463   if (!RV.getNode()) {
1464     assert(N->getOpcode() != ISD::DELETED_NODE &&
1465            "Node was deleted but visit returned NULL!");
1466 
1467     if (N->getOpcode() >= ISD::BUILTIN_OP_END ||
1468         TLI.hasTargetDAGCombine((ISD::NodeType)N->getOpcode())) {
1469 
1470       // Expose the DAG combiner to the target combiner impls.
1471       TargetLowering::DAGCombinerInfo
1472         DagCombineInfo(DAG, Level, false, this);
1473 
1474       RV = TLI.PerformDAGCombine(N, DagCombineInfo);
1475     }
1476   }
1477 
1478   // If nothing happened still, try promoting the operation.
1479   if (!RV.getNode()) {
1480     switch (N->getOpcode()) {
1481     default: break;
1482     case ISD::ADD:
1483     case ISD::SUB:
1484     case ISD::MUL:
1485     case ISD::AND:
1486     case ISD::OR:
1487     case ISD::XOR:
1488       RV = PromoteIntBinOp(SDValue(N, 0));
1489       break;
1490     case ISD::SHL:
1491     case ISD::SRA:
1492     case ISD::SRL:
1493       RV = PromoteIntShiftOp(SDValue(N, 0));
1494       break;
1495     case ISD::SIGN_EXTEND:
1496     case ISD::ZERO_EXTEND:
1497     case ISD::ANY_EXTEND:
1498       RV = PromoteExtend(SDValue(N, 0));
1499       break;
1500     case ISD::LOAD:
1501       if (PromoteLoad(SDValue(N, 0)))
1502         RV = SDValue(N, 0);
1503       break;
1504     }
1505   }
1506 
1507   // If N is a commutative binary node, try commuting it to enable more
1508   // sdisel CSE.
1509   if (!RV.getNode() && SelectionDAG::isCommutativeBinOp(N->getOpcode()) &&
1510       N->getNumValues() == 1) {
1511     SDValue N0 = N->getOperand(0);
1512     SDValue N1 = N->getOperand(1);
1513 
1514     // Constant operands are canonicalized to RHS.
1515     if (isa<ConstantSDNode>(N0) || !isa<ConstantSDNode>(N1)) {
1516       SDValue Ops[] = {N1, N0};
1517       SDNode *CSENode = DAG.getNodeIfExists(N->getOpcode(), N->getVTList(), Ops,
1518                                             N->getFlags());
1519       if (CSENode)
1520         return SDValue(CSENode, 0);
1521     }
1522   }
1523 
1524   return RV;
1525 }
1526 
1527 /// Given a node, return its input chain if it has one, otherwise return a null
1528 /// sd operand.
1529 static SDValue getInputChainForNode(SDNode *N) {
1530   if (unsigned NumOps = N->getNumOperands()) {
1531     if (N->getOperand(0).getValueType() == MVT::Other)
1532       return N->getOperand(0);
1533     if (N->getOperand(NumOps-1).getValueType() == MVT::Other)
1534       return N->getOperand(NumOps-1);
1535     for (unsigned i = 1; i < NumOps-1; ++i)
1536       if (N->getOperand(i).getValueType() == MVT::Other)
1537         return N->getOperand(i);
1538   }
1539   return SDValue();
1540 }
1541 
1542 SDValue DAGCombiner::visitTokenFactor(SDNode *N) {
1543   // If N has two operands, where one has an input chain equal to the other,
1544   // the 'other' chain is redundant.
1545   if (N->getNumOperands() == 2) {
1546     if (getInputChainForNode(N->getOperand(0).getNode()) == N->getOperand(1))
1547       return N->getOperand(0);
1548     if (getInputChainForNode(N->getOperand(1).getNode()) == N->getOperand(0))
1549       return N->getOperand(1);
1550   }
1551 
1552   SmallVector<SDNode *, 8> TFs;     // List of token factors to visit.
1553   SmallVector<SDValue, 8> Ops;    // Ops for replacing token factor.
1554   SmallPtrSet<SDNode*, 16> SeenOps;
1555   bool Changed = false;             // If we should replace this token factor.
1556 
1557   // Start out with this token factor.
1558   TFs.push_back(N);
1559 
1560   // Iterate through token factors.  The TFs grows when new token factors are
1561   // encountered.
1562   for (unsigned i = 0; i < TFs.size(); ++i) {
1563     SDNode *TF = TFs[i];
1564 
1565     // Check each of the operands.
1566     for (const SDValue &Op : TF->op_values()) {
1567 
1568       switch (Op.getOpcode()) {
1569       case ISD::EntryToken:
1570         // Entry tokens don't need to be added to the list. They are
1571         // redundant.
1572         Changed = true;
1573         break;
1574 
1575       case ISD::TokenFactor:
1576         if (Op.hasOneUse() && !is_contained(TFs, Op.getNode())) {
1577           // Queue up for processing.
1578           TFs.push_back(Op.getNode());
1579           // Clean up in case the token factor is removed.
1580           AddToWorklist(Op.getNode());
1581           Changed = true;
1582           break;
1583         }
1584         LLVM_FALLTHROUGH;
1585 
1586       default:
1587         // Only add if it isn't already in the list.
1588         if (SeenOps.insert(Op.getNode()).second)
1589           Ops.push_back(Op);
1590         else
1591           Changed = true;
1592         break;
1593       }
1594     }
1595   }
1596 
1597   SDValue Result;
1598 
1599   // If we've changed things around then replace token factor.
1600   if (Changed) {
1601     if (Ops.empty()) {
1602       // The entry token is the only possible outcome.
1603       Result = DAG.getEntryNode();
1604     } else {
1605       // New and improved token factor.
1606       Result = DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Ops);
1607     }
1608 
1609     // Add users to worklist if AA is enabled, since it may introduce
1610     // a lot of new chained token factors while removing memory deps.
1611     bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
1612       : DAG.getSubtarget().useAA();
1613     return CombineTo(N, Result, UseAA /*add to worklist*/);
1614   }
1615 
1616   return Result;
1617 }
1618 
1619 /// MERGE_VALUES can always be eliminated.
1620 SDValue DAGCombiner::visitMERGE_VALUES(SDNode *N) {
1621   WorklistRemover DeadNodes(*this);
1622   // Replacing results may cause a different MERGE_VALUES to suddenly
1623   // be CSE'd with N, and carry its uses with it. Iterate until no
1624   // uses remain, to ensure that the node can be safely deleted.
1625   // First add the users of this node to the work list so that they
1626   // can be tried again once they have new operands.
1627   AddUsersToWorklist(N);
1628   do {
1629     for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i)
1630       DAG.ReplaceAllUsesOfValueWith(SDValue(N, i), N->getOperand(i));
1631   } while (!N->use_empty());
1632   deleteAndRecombine(N);
1633   return SDValue(N, 0);   // Return N so it doesn't get rechecked!
1634 }
1635 
1636 /// If \p N is a ConstantSDNode with isOpaque() == false return it casted to a
1637 /// ConstantSDNode pointer else nullptr.
1638 static ConstantSDNode *getAsNonOpaqueConstant(SDValue N) {
1639   ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N);
1640   return Const != nullptr && !Const->isOpaque() ? Const : nullptr;
1641 }
1642 
1643 SDValue DAGCombiner::visitADD(SDNode *N) {
1644   SDValue N0 = N->getOperand(0);
1645   SDValue N1 = N->getOperand(1);
1646   EVT VT = N0.getValueType();
1647 
1648   // fold vector ops
1649   if (VT.isVector()) {
1650     if (SDValue FoldedVOp = SimplifyVBinOp(N))
1651       return FoldedVOp;
1652 
1653     // fold (add x, 0) -> x, vector edition
1654     if (ISD::isBuildVectorAllZeros(N1.getNode()))
1655       return N0;
1656     if (ISD::isBuildVectorAllZeros(N0.getNode()))
1657       return N1;
1658   }
1659 
1660   // fold (add x, undef) -> undef
1661   if (N0.isUndef())
1662     return N0;
1663   if (N1.isUndef())
1664     return N1;
1665   if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
1666     // canonicalize constant to RHS
1667     if (!DAG.isConstantIntBuildVectorOrConstantInt(N1))
1668       return DAG.getNode(ISD::ADD, SDLoc(N), VT, N1, N0);
1669     // fold (add c1, c2) -> c1+c2
1670     return DAG.FoldConstantArithmetic(ISD::ADD, SDLoc(N), VT,
1671                                       N0.getNode(), N1.getNode());
1672   }
1673   // fold (add x, 0) -> x
1674   if (isNullConstant(N1))
1675     return N0;
1676   // fold ((c1-A)+c2) -> (c1+c2)-A
1677   if (ConstantSDNode *N1C = getAsNonOpaqueConstant(N1)) {
1678     if (N0.getOpcode() == ISD::SUB)
1679       if (ConstantSDNode *N0C = getAsNonOpaqueConstant(N0.getOperand(0))) {
1680         SDLoc DL(N);
1681         return DAG.getNode(ISD::SUB, DL, VT,
1682                            DAG.getConstant(N1C->getAPIntValue()+
1683                                            N0C->getAPIntValue(), DL, VT),
1684                            N0.getOperand(1));
1685       }
1686   }
1687   // reassociate add
1688   if (SDValue RADD = ReassociateOps(ISD::ADD, SDLoc(N), N0, N1))
1689     return RADD;
1690   // fold ((0-A) + B) -> B-A
1691   if (N0.getOpcode() == ISD::SUB && isNullConstant(N0.getOperand(0)))
1692     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1, N0.getOperand(1));
1693   // fold (A + (0-B)) -> A-B
1694   if (N1.getOpcode() == ISD::SUB && isNullConstant(N1.getOperand(0)))
1695     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N0, N1.getOperand(1));
1696   // fold (A+(B-A)) -> B
1697   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(1))
1698     return N1.getOperand(0);
1699   // fold ((B-A)+A) -> B
1700   if (N0.getOpcode() == ISD::SUB && N1 == N0.getOperand(1))
1701     return N0.getOperand(0);
1702   // fold (A+(B-(A+C))) to (B-C)
1703   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
1704       N0 == N1.getOperand(1).getOperand(0))
1705     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1.getOperand(0),
1706                        N1.getOperand(1).getOperand(1));
1707   // fold (A+(B-(C+A))) to (B-C)
1708   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
1709       N0 == N1.getOperand(1).getOperand(1))
1710     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1.getOperand(0),
1711                        N1.getOperand(1).getOperand(0));
1712   // fold (A+((B-A)+or-C)) to (B+or-C)
1713   if ((N1.getOpcode() == ISD::SUB || N1.getOpcode() == ISD::ADD) &&
1714       N1.getOperand(0).getOpcode() == ISD::SUB &&
1715       N0 == N1.getOperand(0).getOperand(1))
1716     return DAG.getNode(N1.getOpcode(), SDLoc(N), VT,
1717                        N1.getOperand(0).getOperand(0), N1.getOperand(1));
1718 
1719   // fold (A-B)+(C-D) to (A+C)-(B+D) when A or C is constant
1720   if (N0.getOpcode() == ISD::SUB && N1.getOpcode() == ISD::SUB) {
1721     SDValue N00 = N0.getOperand(0);
1722     SDValue N01 = N0.getOperand(1);
1723     SDValue N10 = N1.getOperand(0);
1724     SDValue N11 = N1.getOperand(1);
1725 
1726     if (isa<ConstantSDNode>(N00) || isa<ConstantSDNode>(N10))
1727       return DAG.getNode(ISD::SUB, SDLoc(N), VT,
1728                          DAG.getNode(ISD::ADD, SDLoc(N0), VT, N00, N10),
1729                          DAG.getNode(ISD::ADD, SDLoc(N1), VT, N01, N11));
1730   }
1731 
1732   if (!VT.isVector() && SimplifyDemandedBits(SDValue(N, 0)))
1733     return SDValue(N, 0);
1734 
1735   // fold (a+b) -> (a|b) iff a and b share no bits.
1736   if ((!LegalOperations || TLI.isOperationLegal(ISD::OR, VT)) &&
1737       VT.isInteger() && !VT.isVector() && DAG.haveNoCommonBitsSet(N0, N1))
1738     return DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N1);
1739 
1740   // fold (add x, shl(0 - y, n)) -> sub(x, shl(y, n))
1741   if (N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::SUB &&
1742       isNullConstant(N1.getOperand(0).getOperand(0)))
1743     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N0,
1744                        DAG.getNode(ISD::SHL, SDLoc(N), VT,
1745                                    N1.getOperand(0).getOperand(1),
1746                                    N1.getOperand(1)));
1747   if (N0.getOpcode() == ISD::SHL && N0.getOperand(0).getOpcode() == ISD::SUB &&
1748       isNullConstant(N0.getOperand(0).getOperand(0)))
1749     return DAG.getNode(ISD::SUB, SDLoc(N), VT, N1,
1750                        DAG.getNode(ISD::SHL, SDLoc(N), VT,
1751                                    N0.getOperand(0).getOperand(1),
1752                                    N0.getOperand(1)));
1753 
1754   if (N1.getOpcode() == ISD::AND) {
1755     SDValue AndOp0 = N1.getOperand(0);
1756     unsigned NumSignBits = DAG.ComputeNumSignBits(AndOp0);
1757     unsigned DestBits = VT.getScalarSizeInBits();
1758 
1759     // (add z, (and (sbbl x, x), 1)) -> (sub z, (sbbl x, x))
1760     // and similar xforms where the inner op is either ~0 or 0.
1761     if (NumSignBits == DestBits && isOneConstant(N1->getOperand(1))) {
1762       SDLoc DL(N);
1763       return DAG.getNode(ISD::SUB, DL, VT, N->getOperand(0), AndOp0);
1764     }
1765   }
1766 
1767   // add (sext i1), X -> sub X, (zext i1)
1768   if (N0.getOpcode() == ISD::SIGN_EXTEND &&
1769       N0.getOperand(0).getValueType() == MVT::i1 &&
1770       !TLI.isOperationLegal(ISD::SIGN_EXTEND, MVT::i1)) {
1771     SDLoc DL(N);
1772     SDValue ZExt = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0));
1773     return DAG.getNode(ISD::SUB, DL, VT, N1, ZExt);
1774   }
1775 
1776   // add X, (sextinreg Y i1) -> sub X, (and Y 1)
1777   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
1778     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
1779     if (TN->getVT() == MVT::i1) {
1780       SDLoc DL(N);
1781       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
1782                                  DAG.getConstant(1, DL, VT));
1783       return DAG.getNode(ISD::SUB, DL, VT, N0, ZExt);
1784     }
1785   }
1786 
1787   return SDValue();
1788 }
1789 
1790 SDValue DAGCombiner::visitADDC(SDNode *N) {
1791   SDValue N0 = N->getOperand(0);
1792   SDValue N1 = N->getOperand(1);
1793   EVT VT = N0.getValueType();
1794 
1795   // If the flag result is dead, turn this into an ADD.
1796   if (!N->hasAnyUseOfValue(1))
1797     return CombineTo(N, DAG.getNode(ISD::ADD, SDLoc(N), VT, N0, N1),
1798                      DAG.getNode(ISD::CARRY_FALSE,
1799                                  SDLoc(N), MVT::Glue));
1800 
1801   // canonicalize constant to RHS.
1802   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
1803   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
1804   if (N0C && !N1C)
1805     return DAG.getNode(ISD::ADDC, SDLoc(N), N->getVTList(), N1, N0);
1806 
1807   // fold (addc x, 0) -> x + no carry out
1808   if (isNullConstant(N1))
1809     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE,
1810                                         SDLoc(N), MVT::Glue));
1811 
1812   // fold (addc a, b) -> (or a, b), CARRY_FALSE iff a and b share no bits.
1813   APInt LHSZero, LHSOne;
1814   APInt RHSZero, RHSOne;
1815   DAG.computeKnownBits(N0, LHSZero, LHSOne);
1816 
1817   if (LHSZero.getBoolValue()) {
1818     DAG.computeKnownBits(N1, RHSZero, RHSOne);
1819 
1820     // If all possibly-set bits on the LHS are clear on the RHS, return an OR.
1821     // If all possibly-set bits on the RHS are clear on the LHS, return an OR.
1822     if ((RHSZero & ~LHSZero) == ~LHSZero || (LHSZero & ~RHSZero) == ~RHSZero)
1823       return CombineTo(N, DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N1),
1824                        DAG.getNode(ISD::CARRY_FALSE,
1825                                    SDLoc(N), MVT::Glue));
1826   }
1827 
1828   return SDValue();
1829 }
1830 
1831 SDValue DAGCombiner::visitADDE(SDNode *N) {
1832   SDValue N0 = N->getOperand(0);
1833   SDValue N1 = N->getOperand(1);
1834   SDValue CarryIn = N->getOperand(2);
1835 
1836   // canonicalize constant to RHS
1837   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
1838   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
1839   if (N0C && !N1C)
1840     return DAG.getNode(ISD::ADDE, SDLoc(N), N->getVTList(),
1841                        N1, N0, CarryIn);
1842 
1843   // fold (adde x, y, false) -> (addc x, y)
1844   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
1845     return DAG.getNode(ISD::ADDC, SDLoc(N), N->getVTList(), N0, N1);
1846 
1847   return SDValue();
1848 }
1849 
1850 // Since it may not be valid to emit a fold to zero for vector initializers
1851 // check if we can before folding.
1852 static SDValue tryFoldToZero(const SDLoc &DL, const TargetLowering &TLI, EVT VT,
1853                              SelectionDAG &DAG, bool LegalOperations,
1854                              bool LegalTypes) {
1855   if (!VT.isVector())
1856     return DAG.getConstant(0, DL, VT);
1857   if (!LegalOperations || TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
1858     return DAG.getConstant(0, DL, VT);
1859   return SDValue();
1860 }
1861 
1862 SDValue DAGCombiner::visitSUB(SDNode *N) {
1863   SDValue N0 = N->getOperand(0);
1864   SDValue N1 = N->getOperand(1);
1865   EVT VT = N0.getValueType();
1866   SDLoc DL(N);
1867 
1868   // fold vector ops
1869   if (VT.isVector()) {
1870     if (SDValue FoldedVOp = SimplifyVBinOp(N))
1871       return FoldedVOp;
1872 
1873     // fold (sub x, 0) -> x, vector edition
1874     if (ISD::isBuildVectorAllZeros(N1.getNode()))
1875       return N0;
1876   }
1877 
1878   // fold (sub x, x) -> 0
1879   // FIXME: Refactor this and xor and other similar operations together.
1880   if (N0 == N1)
1881     return tryFoldToZero(DL, TLI, VT, DAG, LegalOperations, LegalTypes);
1882   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
1883       DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
1884     // fold (sub c1, c2) -> c1-c2
1885     return DAG.FoldConstantArithmetic(ISD::SUB, DL, VT, N0.getNode(),
1886                                       N1.getNode());
1887   }
1888 
1889   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
1890   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
1891 
1892   // fold (sub x, c) -> (add x, -c)
1893   if (N1C) {
1894     return DAG.getNode(ISD::ADD, DL, VT, N0,
1895                        DAG.getConstant(-N1C->getAPIntValue(), DL, VT));
1896   }
1897 
1898   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1)
1899   if (isAllOnesConstant(N0))
1900     return DAG.getNode(ISD::XOR, DL, VT, N1, N0);
1901 
1902   // fold A-(A-B) -> B
1903   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(0))
1904     return N1.getOperand(1);
1905 
1906   // fold (A+B)-A -> B
1907   if (N0.getOpcode() == ISD::ADD && N0.getOperand(0) == N1)
1908     return N0.getOperand(1);
1909 
1910   // fold (A+B)-B -> A
1911   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1) == N1)
1912     return N0.getOperand(0);
1913 
1914   // fold C2-(A+C1) -> (C2-C1)-A
1915   if (N1.getOpcode() == ISD::ADD && N0C) {
1916     if (auto *N1C1 = dyn_cast<ConstantSDNode>(N1.getOperand(1).getNode())) {
1917       SDValue NewC =
1918           DAG.getConstant(N0C->getAPIntValue() - N1C1->getAPIntValue(), DL, VT);
1919       return DAG.getNode(ISD::SUB, DL, VT, NewC, N1.getOperand(0));
1920     }
1921   }
1922 
1923   // fold ((A+(B+or-C))-B) -> A+or-C
1924   if (N0.getOpcode() == ISD::ADD &&
1925       (N0.getOperand(1).getOpcode() == ISD::SUB ||
1926        N0.getOperand(1).getOpcode() == ISD::ADD) &&
1927       N0.getOperand(1).getOperand(0) == N1)
1928     return DAG.getNode(N0.getOperand(1).getOpcode(), DL, VT, N0.getOperand(0),
1929                        N0.getOperand(1).getOperand(1));
1930 
1931   // fold ((A+(C+B))-B) -> A+C
1932   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1).getOpcode() == ISD::ADD &&
1933       N0.getOperand(1).getOperand(1) == N1)
1934     return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0),
1935                        N0.getOperand(1).getOperand(0));
1936 
1937   // fold ((A-(B-C))-C) -> A-B
1938   if (N0.getOpcode() == ISD::SUB && N0.getOperand(1).getOpcode() == ISD::SUB &&
1939       N0.getOperand(1).getOperand(1) == N1)
1940     return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0),
1941                        N0.getOperand(1).getOperand(0));
1942 
1943   // If either operand of a sub is undef, the result is undef
1944   if (N0.isUndef())
1945     return N0;
1946   if (N1.isUndef())
1947     return N1;
1948 
1949   // If the relocation model supports it, consider symbol offsets.
1950   if (GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(N0))
1951     if (!LegalOperations && TLI.isOffsetFoldingLegal(GA)) {
1952       // fold (sub Sym, c) -> Sym-c
1953       if (N1C && GA->getOpcode() == ISD::GlobalAddress)
1954         return DAG.getGlobalAddress(GA->getGlobal(), SDLoc(N1C), VT,
1955                                     GA->getOffset() -
1956                                         (uint64_t)N1C->getSExtValue());
1957       // fold (sub Sym+c1, Sym+c2) -> c1-c2
1958       if (GlobalAddressSDNode *GB = dyn_cast<GlobalAddressSDNode>(N1))
1959         if (GA->getGlobal() == GB->getGlobal())
1960           return DAG.getConstant((uint64_t)GA->getOffset() - GB->getOffset(),
1961                                  DL, VT);
1962     }
1963 
1964   // sub X, (sextinreg Y i1) -> add X, (and Y 1)
1965   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
1966     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
1967     if (TN->getVT() == MVT::i1) {
1968       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
1969                                  DAG.getConstant(1, DL, VT));
1970       return DAG.getNode(ISD::ADD, DL, VT, N0, ZExt);
1971     }
1972   }
1973 
1974   return SDValue();
1975 }
1976 
1977 SDValue DAGCombiner::visitSUBC(SDNode *N) {
1978   SDValue N0 = N->getOperand(0);
1979   SDValue N1 = N->getOperand(1);
1980   EVT VT = N0.getValueType();
1981   SDLoc DL(N);
1982 
1983   // If the flag result is dead, turn this into an SUB.
1984   if (!N->hasAnyUseOfValue(1))
1985     return CombineTo(N, DAG.getNode(ISD::SUB, DL, VT, N0, N1),
1986                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
1987 
1988   // fold (subc x, x) -> 0 + no borrow
1989   if (N0 == N1)
1990     return CombineTo(N, DAG.getConstant(0, DL, VT),
1991                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
1992 
1993   // fold (subc x, 0) -> x + no borrow
1994   if (isNullConstant(N1))
1995     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
1996 
1997   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1) + no borrow
1998   if (isAllOnesConstant(N0))
1999     return CombineTo(N, DAG.getNode(ISD::XOR, DL, VT, N1, N0),
2000                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2001 
2002   return SDValue();
2003 }
2004 
2005 SDValue DAGCombiner::visitSUBE(SDNode *N) {
2006   SDValue N0 = N->getOperand(0);
2007   SDValue N1 = N->getOperand(1);
2008   SDValue CarryIn = N->getOperand(2);
2009 
2010   // fold (sube x, y, false) -> (subc x, y)
2011   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
2012     return DAG.getNode(ISD::SUBC, SDLoc(N), N->getVTList(), N0, N1);
2013 
2014   return SDValue();
2015 }
2016 
2017 SDValue DAGCombiner::visitMUL(SDNode *N) {
2018   SDValue N0 = N->getOperand(0);
2019   SDValue N1 = N->getOperand(1);
2020   EVT VT = N0.getValueType();
2021 
2022   // fold (mul x, undef) -> 0
2023   if (N0.isUndef() || N1.isUndef())
2024     return DAG.getConstant(0, SDLoc(N), VT);
2025 
2026   bool N0IsConst = false;
2027   bool N1IsConst = false;
2028   bool N1IsOpaqueConst = false;
2029   bool N0IsOpaqueConst = false;
2030   APInt ConstValue0, ConstValue1;
2031   // fold vector ops
2032   if (VT.isVector()) {
2033     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2034       return FoldedVOp;
2035 
2036     N0IsConst = ISD::isConstantSplatVector(N0.getNode(), ConstValue0);
2037     N1IsConst = ISD::isConstantSplatVector(N1.getNode(), ConstValue1);
2038   } else {
2039     N0IsConst = isa<ConstantSDNode>(N0);
2040     if (N0IsConst) {
2041       ConstValue0 = cast<ConstantSDNode>(N0)->getAPIntValue();
2042       N0IsOpaqueConst = cast<ConstantSDNode>(N0)->isOpaque();
2043     }
2044     N1IsConst = isa<ConstantSDNode>(N1);
2045     if (N1IsConst) {
2046       ConstValue1 = cast<ConstantSDNode>(N1)->getAPIntValue();
2047       N1IsOpaqueConst = cast<ConstantSDNode>(N1)->isOpaque();
2048     }
2049   }
2050 
2051   // fold (mul c1, c2) -> c1*c2
2052   if (N0IsConst && N1IsConst && !N0IsOpaqueConst && !N1IsOpaqueConst)
2053     return DAG.FoldConstantArithmetic(ISD::MUL, SDLoc(N), VT,
2054                                       N0.getNode(), N1.getNode());
2055 
2056   // canonicalize constant to RHS (vector doesn't have to splat)
2057   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2058      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
2059     return DAG.getNode(ISD::MUL, SDLoc(N), VT, N1, N0);
2060   // fold (mul x, 0) -> 0
2061   if (N1IsConst && ConstValue1 == 0)
2062     return N1;
2063   // We require a splat of the entire scalar bit width for non-contiguous
2064   // bit patterns.
2065   bool IsFullSplat =
2066     ConstValue1.getBitWidth() == VT.getScalarSizeInBits();
2067   // fold (mul x, 1) -> x
2068   if (N1IsConst && ConstValue1 == 1 && IsFullSplat)
2069     return N0;
2070   // fold (mul x, -1) -> 0-x
2071   if (N1IsConst && ConstValue1.isAllOnesValue()) {
2072     SDLoc DL(N);
2073     return DAG.getNode(ISD::SUB, DL, VT,
2074                        DAG.getConstant(0, DL, VT), N0);
2075   }
2076   // fold (mul x, (1 << c)) -> x << c
2077   if (N1IsConst && !N1IsOpaqueConst && ConstValue1.isPowerOf2() &&
2078       IsFullSplat) {
2079     SDLoc DL(N);
2080     return DAG.getNode(ISD::SHL, DL, VT, N0,
2081                        DAG.getConstant(ConstValue1.logBase2(), DL,
2082                                        getShiftAmountTy(N0.getValueType())));
2083   }
2084   // fold (mul x, -(1 << c)) -> -(x << c) or (-x) << c
2085   if (N1IsConst && !N1IsOpaqueConst && (-ConstValue1).isPowerOf2() &&
2086       IsFullSplat) {
2087     unsigned Log2Val = (-ConstValue1).logBase2();
2088     SDLoc DL(N);
2089     // FIXME: If the input is something that is easily negated (e.g. a
2090     // single-use add), we should put the negate there.
2091     return DAG.getNode(ISD::SUB, DL, VT,
2092                        DAG.getConstant(0, DL, VT),
2093                        DAG.getNode(ISD::SHL, DL, VT, N0,
2094                             DAG.getConstant(Log2Val, DL,
2095                                       getShiftAmountTy(N0.getValueType()))));
2096   }
2097 
2098   APInt Val;
2099   // (mul (shl X, c1), c2) -> (mul X, c2 << c1)
2100   if (N1IsConst && N0.getOpcode() == ISD::SHL &&
2101       (ISD::isConstantSplatVector(N0.getOperand(1).getNode(), Val) ||
2102        isa<ConstantSDNode>(N0.getOperand(1)))) {
2103     SDValue C3 = DAG.getNode(ISD::SHL, SDLoc(N), VT, N1, N0.getOperand(1));
2104     AddToWorklist(C3.getNode());
2105     return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), C3);
2106   }
2107 
2108   // Change (mul (shl X, C), Y) -> (shl (mul X, Y), C) when the shift has one
2109   // use.
2110   {
2111     SDValue Sh(nullptr, 0), Y(nullptr, 0);
2112     // Check for both (mul (shl X, C), Y)  and  (mul Y, (shl X, C)).
2113     if (N0.getOpcode() == ISD::SHL &&
2114         (ISD::isConstantSplatVector(N0.getOperand(1).getNode(), Val) ||
2115          isa<ConstantSDNode>(N0.getOperand(1))) &&
2116         N0.getNode()->hasOneUse()) {
2117       Sh = N0; Y = N1;
2118     } else if (N1.getOpcode() == ISD::SHL &&
2119                isa<ConstantSDNode>(N1.getOperand(1)) &&
2120                N1.getNode()->hasOneUse()) {
2121       Sh = N1; Y = N0;
2122     }
2123 
2124     if (Sh.getNode()) {
2125       SDValue Mul = DAG.getNode(ISD::MUL, SDLoc(N), VT, Sh.getOperand(0), Y);
2126       return DAG.getNode(ISD::SHL, SDLoc(N), VT, Mul, Sh.getOperand(1));
2127     }
2128   }
2129 
2130   // fold (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2)
2131   if (DAG.isConstantIntBuildVectorOrConstantInt(N1) &&
2132       N0.getOpcode() == ISD::ADD &&
2133       DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1)) &&
2134       isMulAddWithConstProfitable(N, N0, N1))
2135       return DAG.getNode(ISD::ADD, SDLoc(N), VT,
2136                          DAG.getNode(ISD::MUL, SDLoc(N0), VT,
2137                                      N0.getOperand(0), N1),
2138                          DAG.getNode(ISD::MUL, SDLoc(N1), VT,
2139                                      N0.getOperand(1), N1));
2140 
2141   // reassociate mul
2142   if (SDValue RMUL = ReassociateOps(ISD::MUL, SDLoc(N), N0, N1))
2143     return RMUL;
2144 
2145   return SDValue();
2146 }
2147 
2148 /// Return true if divmod libcall is available.
2149 static bool isDivRemLibcallAvailable(SDNode *Node, bool isSigned,
2150                                      const TargetLowering &TLI) {
2151   RTLIB::Libcall LC;
2152   EVT NodeType = Node->getValueType(0);
2153   if (!NodeType.isSimple())
2154     return false;
2155   switch (NodeType.getSimpleVT().SimpleTy) {
2156   default: return false; // No libcall for vector types.
2157   case MVT::i8:   LC= isSigned ? RTLIB::SDIVREM_I8  : RTLIB::UDIVREM_I8;  break;
2158   case MVT::i16:  LC= isSigned ? RTLIB::SDIVREM_I16 : RTLIB::UDIVREM_I16; break;
2159   case MVT::i32:  LC= isSigned ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32; break;
2160   case MVT::i64:  LC= isSigned ? RTLIB::SDIVREM_I64 : RTLIB::UDIVREM_I64; break;
2161   case MVT::i128: LC= isSigned ? RTLIB::SDIVREM_I128:RTLIB::UDIVREM_I128; break;
2162   }
2163 
2164   return TLI.getLibcallName(LC) != nullptr;
2165 }
2166 
2167 /// Issue divrem if both quotient and remainder are needed.
2168 SDValue DAGCombiner::useDivRem(SDNode *Node) {
2169   if (Node->use_empty())
2170     return SDValue(); // This is a dead node, leave it alone.
2171 
2172   unsigned Opcode = Node->getOpcode();
2173   bool isSigned = (Opcode == ISD::SDIV) || (Opcode == ISD::SREM);
2174   unsigned DivRemOpc = isSigned ? ISD::SDIVREM : ISD::UDIVREM;
2175 
2176   // DivMod lib calls can still work on non-legal types if using lib-calls.
2177   EVT VT = Node->getValueType(0);
2178   if (VT.isVector() || !VT.isInteger())
2179     return SDValue();
2180 
2181   if (!TLI.isTypeLegal(VT) && !TLI.isOperationCustom(DivRemOpc, VT))
2182     return SDValue();
2183 
2184   // If DIVREM is going to get expanded into a libcall,
2185   // but there is no libcall available, then don't combine.
2186   if (!TLI.isOperationLegalOrCustom(DivRemOpc, VT) &&
2187       !isDivRemLibcallAvailable(Node, isSigned, TLI))
2188     return SDValue();
2189 
2190   // If div is legal, it's better to do the normal expansion
2191   unsigned OtherOpcode = 0;
2192   if ((Opcode == ISD::SDIV) || (Opcode == ISD::UDIV)) {
2193     OtherOpcode = isSigned ? ISD::SREM : ISD::UREM;
2194     if (TLI.isOperationLegalOrCustom(Opcode, VT))
2195       return SDValue();
2196   } else {
2197     OtherOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
2198     if (TLI.isOperationLegalOrCustom(OtherOpcode, VT))
2199       return SDValue();
2200   }
2201 
2202   SDValue Op0 = Node->getOperand(0);
2203   SDValue Op1 = Node->getOperand(1);
2204   SDValue combined;
2205   for (SDNode::use_iterator UI = Op0.getNode()->use_begin(),
2206          UE = Op0.getNode()->use_end(); UI != UE; ++UI) {
2207     SDNode *User = *UI;
2208     if (User == Node || User->use_empty())
2209       continue;
2210     // Convert the other matching node(s), too;
2211     // otherwise, the DIVREM may get target-legalized into something
2212     // target-specific that we won't be able to recognize.
2213     unsigned UserOpc = User->getOpcode();
2214     if ((UserOpc == Opcode || UserOpc == OtherOpcode || UserOpc == DivRemOpc) &&
2215         User->getOperand(0) == Op0 &&
2216         User->getOperand(1) == Op1) {
2217       if (!combined) {
2218         if (UserOpc == OtherOpcode) {
2219           SDVTList VTs = DAG.getVTList(VT, VT);
2220           combined = DAG.getNode(DivRemOpc, SDLoc(Node), VTs, Op0, Op1);
2221         } else if (UserOpc == DivRemOpc) {
2222           combined = SDValue(User, 0);
2223         } else {
2224           assert(UserOpc == Opcode);
2225           continue;
2226         }
2227       }
2228       if (UserOpc == ISD::SDIV || UserOpc == ISD::UDIV)
2229         CombineTo(User, combined);
2230       else if (UserOpc == ISD::SREM || UserOpc == ISD::UREM)
2231         CombineTo(User, combined.getValue(1));
2232     }
2233   }
2234   return combined;
2235 }
2236 
2237 SDValue DAGCombiner::visitSDIV(SDNode *N) {
2238   SDValue N0 = N->getOperand(0);
2239   SDValue N1 = N->getOperand(1);
2240   EVT VT = N->getValueType(0);
2241 
2242   // fold vector ops
2243   if (VT.isVector())
2244     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2245       return FoldedVOp;
2246 
2247   SDLoc DL(N);
2248 
2249   // fold (sdiv c1, c2) -> c1/c2
2250   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2251   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2252   if (N0C && N1C && !N0C->isOpaque() && !N1C->isOpaque())
2253     return DAG.FoldConstantArithmetic(ISD::SDIV, DL, VT, N0C, N1C);
2254   // fold (sdiv X, 1) -> X
2255   if (N1C && N1C->isOne())
2256     return N0;
2257   // fold (sdiv X, -1) -> 0-X
2258   if (N1C && N1C->isAllOnesValue())
2259     return DAG.getNode(ISD::SUB, DL, VT,
2260                        DAG.getConstant(0, DL, VT), N0);
2261 
2262   // If we know the sign bits of both operands are zero, strength reduce to a
2263   // udiv instead.  Handles (X&15) /s 4 -> X&15 >> 2
2264   if (!VT.isVector()) {
2265     if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
2266       return DAG.getNode(ISD::UDIV, DL, N1.getValueType(), N0, N1);
2267   }
2268 
2269   // fold (sdiv X, pow2) -> simple ops after legalize
2270   // FIXME: We check for the exact bit here because the generic lowering gives
2271   // better results in that case. The target-specific lowering should learn how
2272   // to handle exact sdivs efficiently.
2273   if (N1C && !N1C->isNullValue() && !N1C->isOpaque() &&
2274       !cast<BinaryWithFlagsSDNode>(N)->Flags.hasExact() &&
2275       (N1C->getAPIntValue().isPowerOf2() ||
2276        (-N1C->getAPIntValue()).isPowerOf2())) {
2277     // Target-specific implementation of sdiv x, pow2.
2278     if (SDValue Res = BuildSDIVPow2(N))
2279       return Res;
2280 
2281     unsigned lg2 = N1C->getAPIntValue().countTrailingZeros();
2282 
2283     // Splat the sign bit into the register
2284     SDValue SGN =
2285         DAG.getNode(ISD::SRA, DL, VT, N0,
2286                     DAG.getConstant(VT.getScalarSizeInBits() - 1, DL,
2287                                     getShiftAmountTy(N0.getValueType())));
2288     AddToWorklist(SGN.getNode());
2289 
2290     // Add (N0 < 0) ? abs2 - 1 : 0;
2291     SDValue SRL =
2292         DAG.getNode(ISD::SRL, DL, VT, SGN,
2293                     DAG.getConstant(VT.getScalarSizeInBits() - lg2, DL,
2294                                     getShiftAmountTy(SGN.getValueType())));
2295     SDValue ADD = DAG.getNode(ISD::ADD, DL, VT, N0, SRL);
2296     AddToWorklist(SRL.getNode());
2297     AddToWorklist(ADD.getNode());    // Divide by pow2
2298     SDValue SRA = DAG.getNode(ISD::SRA, DL, VT, ADD,
2299                   DAG.getConstant(lg2, DL,
2300                                   getShiftAmountTy(ADD.getValueType())));
2301 
2302     // If we're dividing by a positive value, we're done.  Otherwise, we must
2303     // negate the result.
2304     if (N1C->getAPIntValue().isNonNegative())
2305       return SRA;
2306 
2307     AddToWorklist(SRA.getNode());
2308     return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), SRA);
2309   }
2310 
2311   // If integer divide is expensive and we satisfy the requirements, emit an
2312   // alternate sequence.  Targets may check function attributes for size/speed
2313   // trade-offs.
2314   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2315   if (N1C && !TLI.isIntDivCheap(N->getValueType(0), Attr))
2316     if (SDValue Op = BuildSDIV(N))
2317       return Op;
2318 
2319   // sdiv, srem -> sdivrem
2320   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is true.
2321   // Otherwise, we break the simplification logic in visitREM().
2322   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
2323     if (SDValue DivRem = useDivRem(N))
2324         return DivRem;
2325 
2326   // undef / X -> 0
2327   if (N0.isUndef())
2328     return DAG.getConstant(0, DL, VT);
2329   // X / undef -> undef
2330   if (N1.isUndef())
2331     return N1;
2332 
2333   return SDValue();
2334 }
2335 
2336 SDValue DAGCombiner::visitUDIV(SDNode *N) {
2337   SDValue N0 = N->getOperand(0);
2338   SDValue N1 = N->getOperand(1);
2339   EVT VT = N->getValueType(0);
2340 
2341   // fold vector ops
2342   if (VT.isVector())
2343     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2344       return FoldedVOp;
2345 
2346   SDLoc DL(N);
2347 
2348   // fold (udiv c1, c2) -> c1/c2
2349   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2350   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2351   if (N0C && N1C)
2352     if (SDValue Folded = DAG.FoldConstantArithmetic(ISD::UDIV, DL, VT,
2353                                                     N0C, N1C))
2354       return Folded;
2355   // fold (udiv x, (1 << c)) -> x >>u c
2356   if (N1C && !N1C->isOpaque() && N1C->getAPIntValue().isPowerOf2())
2357     return DAG.getNode(ISD::SRL, DL, VT, N0,
2358                        DAG.getConstant(N1C->getAPIntValue().logBase2(), DL,
2359                                        getShiftAmountTy(N0.getValueType())));
2360 
2361   // fold (udiv x, (shl c, y)) -> x >>u (log2(c)+y) iff c is power of 2
2362   if (N1.getOpcode() == ISD::SHL) {
2363     if (ConstantSDNode *SHC = getAsNonOpaqueConstant(N1.getOperand(0))) {
2364       if (SHC->getAPIntValue().isPowerOf2()) {
2365         EVT ADDVT = N1.getOperand(1).getValueType();
2366         SDValue Add = DAG.getNode(ISD::ADD, DL, ADDVT,
2367                                   N1.getOperand(1),
2368                                   DAG.getConstant(SHC->getAPIntValue()
2369                                                                   .logBase2(),
2370                                                   DL, ADDVT));
2371         AddToWorklist(Add.getNode());
2372         return DAG.getNode(ISD::SRL, DL, VT, N0, Add);
2373       }
2374     }
2375   }
2376 
2377   // fold (udiv x, c) -> alternate
2378   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2379   if (N1C && !TLI.isIntDivCheap(N->getValueType(0), Attr))
2380     if (SDValue Op = BuildUDIV(N))
2381       return Op;
2382 
2383   // sdiv, srem -> sdivrem
2384   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is true.
2385   // Otherwise, we break the simplification logic in visitREM().
2386   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
2387     if (SDValue DivRem = useDivRem(N))
2388         return DivRem;
2389 
2390   // undef / X -> 0
2391   if (N0.isUndef())
2392     return DAG.getConstant(0, DL, VT);
2393   // X / undef -> undef
2394   if (N1.isUndef())
2395     return N1;
2396 
2397   return SDValue();
2398 }
2399 
2400 // handles ISD::SREM and ISD::UREM
2401 SDValue DAGCombiner::visitREM(SDNode *N) {
2402   unsigned Opcode = N->getOpcode();
2403   SDValue N0 = N->getOperand(0);
2404   SDValue N1 = N->getOperand(1);
2405   EVT VT = N->getValueType(0);
2406   bool isSigned = (Opcode == ISD::SREM);
2407   SDLoc DL(N);
2408 
2409   // fold (rem c1, c2) -> c1%c2
2410   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2411   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2412   if (N0C && N1C)
2413     if (SDValue Folded = DAG.FoldConstantArithmetic(Opcode, DL, VT, N0C, N1C))
2414       return Folded;
2415 
2416   if (isSigned) {
2417     // If we know the sign bits of both operands are zero, strength reduce to a
2418     // urem instead.  Handles (X & 0x0FFFFFFF) %s 16 -> X&15
2419     if (!VT.isVector()) {
2420       if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
2421         return DAG.getNode(ISD::UREM, DL, VT, N0, N1);
2422     }
2423   } else {
2424     // fold (urem x, pow2) -> (and x, pow2-1)
2425     if (N1C && !N1C->isNullValue() && !N1C->isOpaque() &&
2426         N1C->getAPIntValue().isPowerOf2()) {
2427       return DAG.getNode(ISD::AND, DL, VT, N0,
2428                          DAG.getConstant(N1C->getAPIntValue() - 1, DL, VT));
2429     }
2430     // fold (urem x, (shl pow2, y)) -> (and x, (add (shl pow2, y), -1))
2431     if (N1.getOpcode() == ISD::SHL) {
2432       ConstantSDNode *SHC = getAsNonOpaqueConstant(N1.getOperand(0));
2433       if (SHC && SHC->getAPIntValue().isPowerOf2()) {
2434         APInt NegOne = APInt::getAllOnesValue(VT.getSizeInBits());
2435         SDValue Add =
2436             DAG.getNode(ISD::ADD, DL, VT, N1, DAG.getConstant(NegOne, DL, VT));
2437         AddToWorklist(Add.getNode());
2438         return DAG.getNode(ISD::AND, DL, VT, N0, Add);
2439       }
2440     }
2441   }
2442 
2443   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2444 
2445   // If X/C can be simplified by the division-by-constant logic, lower
2446   // X%C to the equivalent of X-X/C*C.
2447   // To avoid mangling nodes, this simplification requires that the combine()
2448   // call for the speculative DIV must not cause a DIVREM conversion.  We guard
2449   // against this by skipping the simplification if isIntDivCheap().  When
2450   // div is not cheap, combine will not return a DIVREM.  Regardless,
2451   // checking cheapness here makes sense since the simplification results in
2452   // fatter code.
2453   if (N1C && !N1C->isNullValue() && !TLI.isIntDivCheap(VT, Attr)) {
2454     unsigned DivOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
2455     SDValue Div = DAG.getNode(DivOpcode, DL, VT, N0, N1);
2456     AddToWorklist(Div.getNode());
2457     SDValue OptimizedDiv = combine(Div.getNode());
2458     if (OptimizedDiv.getNode() && OptimizedDiv.getNode() != Div.getNode()) {
2459       assert((OptimizedDiv.getOpcode() != ISD::UDIVREM) &&
2460              (OptimizedDiv.getOpcode() != ISD::SDIVREM));
2461       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, OptimizedDiv, N1);
2462       SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul);
2463       AddToWorklist(Mul.getNode());
2464       return Sub;
2465     }
2466   }
2467 
2468   // sdiv, srem -> sdivrem
2469   if (SDValue DivRem = useDivRem(N))
2470     return DivRem.getValue(1);
2471 
2472   // undef % X -> 0
2473   if (N0.isUndef())
2474     return DAG.getConstant(0, DL, VT);
2475   // X % undef -> undef
2476   if (N1.isUndef())
2477     return N1;
2478 
2479   return SDValue();
2480 }
2481 
2482 SDValue DAGCombiner::visitMULHS(SDNode *N) {
2483   SDValue N0 = N->getOperand(0);
2484   SDValue N1 = N->getOperand(1);
2485   EVT VT = N->getValueType(0);
2486   SDLoc DL(N);
2487 
2488   // fold (mulhs x, 0) -> 0
2489   if (isNullConstant(N1))
2490     return N1;
2491   // fold (mulhs x, 1) -> (sra x, size(x)-1)
2492   if (isOneConstant(N1)) {
2493     SDLoc DL(N);
2494     return DAG.getNode(ISD::SRA, DL, N0.getValueType(), N0,
2495                        DAG.getConstant(N0.getValueSizeInBits() - 1, DL,
2496                                        getShiftAmountTy(N0.getValueType())));
2497   }
2498   // fold (mulhs x, undef) -> 0
2499   if (N0.isUndef() || N1.isUndef())
2500     return DAG.getConstant(0, SDLoc(N), VT);
2501 
2502   // If the type twice as wide is legal, transform the mulhs to a wider multiply
2503   // plus a shift.
2504   if (VT.isSimple() && !VT.isVector()) {
2505     MVT Simple = VT.getSimpleVT();
2506     unsigned SimpleSize = Simple.getSizeInBits();
2507     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2508     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2509       N0 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N0);
2510       N1 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N1);
2511       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
2512       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
2513             DAG.getConstant(SimpleSize, DL,
2514                             getShiftAmountTy(N1.getValueType())));
2515       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
2516     }
2517   }
2518 
2519   return SDValue();
2520 }
2521 
2522 SDValue DAGCombiner::visitMULHU(SDNode *N) {
2523   SDValue N0 = N->getOperand(0);
2524   SDValue N1 = N->getOperand(1);
2525   EVT VT = N->getValueType(0);
2526   SDLoc DL(N);
2527 
2528   // fold (mulhu x, 0) -> 0
2529   if (isNullConstant(N1))
2530     return N1;
2531   // fold (mulhu x, 1) -> 0
2532   if (isOneConstant(N1))
2533     return DAG.getConstant(0, DL, N0.getValueType());
2534   // fold (mulhu x, undef) -> 0
2535   if (N0.isUndef() || N1.isUndef())
2536     return DAG.getConstant(0, DL, VT);
2537 
2538   // If the type twice as wide is legal, transform the mulhu to a wider multiply
2539   // plus a shift.
2540   if (VT.isSimple() && !VT.isVector()) {
2541     MVT Simple = VT.getSimpleVT();
2542     unsigned SimpleSize = Simple.getSizeInBits();
2543     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2544     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2545       N0 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N0);
2546       N1 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N1);
2547       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
2548       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
2549             DAG.getConstant(SimpleSize, DL,
2550                             getShiftAmountTy(N1.getValueType())));
2551       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
2552     }
2553   }
2554 
2555   return SDValue();
2556 }
2557 
2558 /// Perform optimizations common to nodes that compute two values. LoOp and HiOp
2559 /// give the opcodes for the two computations that are being performed. Return
2560 /// true if a simplification was made.
2561 SDValue DAGCombiner::SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
2562                                                 unsigned HiOp) {
2563   // If the high half is not needed, just compute the low half.
2564   bool HiExists = N->hasAnyUseOfValue(1);
2565   if (!HiExists &&
2566       (!LegalOperations ||
2567        TLI.isOperationLegalOrCustom(LoOp, N->getValueType(0)))) {
2568     SDValue Res = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
2569     return CombineTo(N, Res, Res);
2570   }
2571 
2572   // If the low half is not needed, just compute the high half.
2573   bool LoExists = N->hasAnyUseOfValue(0);
2574   if (!LoExists &&
2575       (!LegalOperations ||
2576        TLI.isOperationLegal(HiOp, N->getValueType(1)))) {
2577     SDValue Res = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
2578     return CombineTo(N, Res, Res);
2579   }
2580 
2581   // If both halves are used, return as it is.
2582   if (LoExists && HiExists)
2583     return SDValue();
2584 
2585   // If the two computed results can be simplified separately, separate them.
2586   if (LoExists) {
2587     SDValue Lo = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
2588     AddToWorklist(Lo.getNode());
2589     SDValue LoOpt = combine(Lo.getNode());
2590     if (LoOpt.getNode() && LoOpt.getNode() != Lo.getNode() &&
2591         (!LegalOperations ||
2592          TLI.isOperationLegal(LoOpt.getOpcode(), LoOpt.getValueType())))
2593       return CombineTo(N, LoOpt, LoOpt);
2594   }
2595 
2596   if (HiExists) {
2597     SDValue Hi = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
2598     AddToWorklist(Hi.getNode());
2599     SDValue HiOpt = combine(Hi.getNode());
2600     if (HiOpt.getNode() && HiOpt != Hi &&
2601         (!LegalOperations ||
2602          TLI.isOperationLegal(HiOpt.getOpcode(), HiOpt.getValueType())))
2603       return CombineTo(N, HiOpt, HiOpt);
2604   }
2605 
2606   return SDValue();
2607 }
2608 
2609 SDValue DAGCombiner::visitSMUL_LOHI(SDNode *N) {
2610   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHS))
2611     return Res;
2612 
2613   EVT VT = N->getValueType(0);
2614   SDLoc DL(N);
2615 
2616   // If the type is twice as wide is legal, transform the mulhu to a wider
2617   // multiply plus a shift.
2618   if (VT.isSimple() && !VT.isVector()) {
2619     MVT Simple = VT.getSimpleVT();
2620     unsigned SimpleSize = Simple.getSizeInBits();
2621     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2622     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2623       SDValue Lo = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(0));
2624       SDValue Hi = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(1));
2625       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
2626       // Compute the high part as N1.
2627       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
2628             DAG.getConstant(SimpleSize, DL,
2629                             getShiftAmountTy(Lo.getValueType())));
2630       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
2631       // Compute the low part as N0.
2632       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
2633       return CombineTo(N, Lo, Hi);
2634     }
2635   }
2636 
2637   return SDValue();
2638 }
2639 
2640 SDValue DAGCombiner::visitUMUL_LOHI(SDNode *N) {
2641   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHU))
2642     return Res;
2643 
2644   EVT VT = N->getValueType(0);
2645   SDLoc DL(N);
2646 
2647   // If the type is twice as wide is legal, transform the mulhu to a wider
2648   // multiply plus a shift.
2649   if (VT.isSimple() && !VT.isVector()) {
2650     MVT Simple = VT.getSimpleVT();
2651     unsigned SimpleSize = Simple.getSizeInBits();
2652     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2653     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2654       SDValue Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(0));
2655       SDValue Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(1));
2656       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
2657       // Compute the high part as N1.
2658       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
2659             DAG.getConstant(SimpleSize, DL,
2660                             getShiftAmountTy(Lo.getValueType())));
2661       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
2662       // Compute the low part as N0.
2663       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
2664       return CombineTo(N, Lo, Hi);
2665     }
2666   }
2667 
2668   return SDValue();
2669 }
2670 
2671 SDValue DAGCombiner::visitSMULO(SDNode *N) {
2672   // (smulo x, 2) -> (saddo x, x)
2673   if (ConstantSDNode *C2 = dyn_cast<ConstantSDNode>(N->getOperand(1)))
2674     if (C2->getAPIntValue() == 2)
2675       return DAG.getNode(ISD::SADDO, SDLoc(N), N->getVTList(),
2676                          N->getOperand(0), N->getOperand(0));
2677 
2678   return SDValue();
2679 }
2680 
2681 SDValue DAGCombiner::visitUMULO(SDNode *N) {
2682   // (umulo x, 2) -> (uaddo x, x)
2683   if (ConstantSDNode *C2 = dyn_cast<ConstantSDNode>(N->getOperand(1)))
2684     if (C2->getAPIntValue() == 2)
2685       return DAG.getNode(ISD::UADDO, SDLoc(N), N->getVTList(),
2686                          N->getOperand(0), N->getOperand(0));
2687 
2688   return SDValue();
2689 }
2690 
2691 SDValue DAGCombiner::visitIMINMAX(SDNode *N) {
2692   SDValue N0 = N->getOperand(0);
2693   SDValue N1 = N->getOperand(1);
2694   EVT VT = N0.getValueType();
2695 
2696   // fold vector ops
2697   if (VT.isVector())
2698     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2699       return FoldedVOp;
2700 
2701   // fold (add c1, c2) -> c1+c2
2702   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
2703   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
2704   if (N0C && N1C)
2705     return DAG.FoldConstantArithmetic(N->getOpcode(), SDLoc(N), VT, N0C, N1C);
2706 
2707   // canonicalize constant to RHS
2708   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2709      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
2710     return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N1, N0);
2711 
2712   return SDValue();
2713 }
2714 
2715 /// If this is a binary operator with two operands of the same opcode, try to
2716 /// simplify it.
2717 SDValue DAGCombiner::SimplifyBinOpWithSameOpcodeHands(SDNode *N) {
2718   SDValue N0 = N->getOperand(0), N1 = N->getOperand(1);
2719   EVT VT = N0.getValueType();
2720   assert(N0.getOpcode() == N1.getOpcode() && "Bad input!");
2721 
2722   // Bail early if none of these transforms apply.
2723   if (N0.getNode()->getNumOperands() == 0) return SDValue();
2724 
2725   // For each of OP in AND/OR/XOR:
2726   // fold (OP (zext x), (zext y)) -> (zext (OP x, y))
2727   // fold (OP (sext x), (sext y)) -> (sext (OP x, y))
2728   // fold (OP (aext x), (aext y)) -> (aext (OP x, y))
2729   // fold (OP (bswap x), (bswap y)) -> (bswap (OP x, y))
2730   // fold (OP (trunc x), (trunc y)) -> (trunc (OP x, y)) (if trunc isn't free)
2731   //
2732   // do not sink logical op inside of a vector extend, since it may combine
2733   // into a vsetcc.
2734   EVT Op0VT = N0.getOperand(0).getValueType();
2735   if ((N0.getOpcode() == ISD::ZERO_EXTEND ||
2736        N0.getOpcode() == ISD::SIGN_EXTEND ||
2737        N0.getOpcode() == ISD::BSWAP ||
2738        // Avoid infinite looping with PromoteIntBinOp.
2739        (N0.getOpcode() == ISD::ANY_EXTEND &&
2740         (!LegalTypes || TLI.isTypeDesirableForOp(N->getOpcode(), Op0VT))) ||
2741        (N0.getOpcode() == ISD::TRUNCATE &&
2742         (!TLI.isZExtFree(VT, Op0VT) ||
2743          !TLI.isTruncateFree(Op0VT, VT)) &&
2744         TLI.isTypeLegal(Op0VT))) &&
2745       !VT.isVector() &&
2746       Op0VT == N1.getOperand(0).getValueType() &&
2747       (!LegalOperations || TLI.isOperationLegal(N->getOpcode(), Op0VT))) {
2748     SDValue ORNode = DAG.getNode(N->getOpcode(), SDLoc(N0),
2749                                  N0.getOperand(0).getValueType(),
2750                                  N0.getOperand(0), N1.getOperand(0));
2751     AddToWorklist(ORNode.getNode());
2752     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, ORNode);
2753   }
2754 
2755   // For each of OP in SHL/SRL/SRA/AND...
2756   //   fold (and (OP x, z), (OP y, z)) -> (OP (and x, y), z)
2757   //   fold (or  (OP x, z), (OP y, z)) -> (OP (or  x, y), z)
2758   //   fold (xor (OP x, z), (OP y, z)) -> (OP (xor x, y), z)
2759   if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL ||
2760        N0.getOpcode() == ISD::SRA || N0.getOpcode() == ISD::AND) &&
2761       N0.getOperand(1) == N1.getOperand(1)) {
2762     SDValue ORNode = DAG.getNode(N->getOpcode(), SDLoc(N0),
2763                                  N0.getOperand(0).getValueType(),
2764                                  N0.getOperand(0), N1.getOperand(0));
2765     AddToWorklist(ORNode.getNode());
2766     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT,
2767                        ORNode, N0.getOperand(1));
2768   }
2769 
2770   // Simplify xor/and/or (bitcast(A), bitcast(B)) -> bitcast(op (A,B))
2771   // Only perform this optimization up until type legalization, before
2772   // LegalizeVectorOprs. LegalizeVectorOprs promotes vector operations by
2773   // adding bitcasts. For example (xor v4i32) is promoted to (v2i64), and
2774   // we don't want to undo this promotion.
2775   // We also handle SCALAR_TO_VECTOR because xor/or/and operations are cheaper
2776   // on scalars.
2777   if ((N0.getOpcode() == ISD::BITCAST ||
2778        N0.getOpcode() == ISD::SCALAR_TO_VECTOR) &&
2779        Level <= AfterLegalizeTypes) {
2780     SDValue In0 = N0.getOperand(0);
2781     SDValue In1 = N1.getOperand(0);
2782     EVT In0Ty = In0.getValueType();
2783     EVT In1Ty = In1.getValueType();
2784     SDLoc DL(N);
2785     // If both incoming values are integers, and the original types are the
2786     // same.
2787     if (In0Ty.isInteger() && In1Ty.isInteger() && In0Ty == In1Ty) {
2788       SDValue Op = DAG.getNode(N->getOpcode(), DL, In0Ty, In0, In1);
2789       SDValue BC = DAG.getNode(N0.getOpcode(), DL, VT, Op);
2790       AddToWorklist(Op.getNode());
2791       return BC;
2792     }
2793   }
2794 
2795   // Xor/and/or are indifferent to the swizzle operation (shuffle of one value).
2796   // Simplify xor/and/or (shuff(A), shuff(B)) -> shuff(op (A,B))
2797   // If both shuffles use the same mask, and both shuffle within a single
2798   // vector, then it is worthwhile to move the swizzle after the operation.
2799   // The type-legalizer generates this pattern when loading illegal
2800   // vector types from memory. In many cases this allows additional shuffle
2801   // optimizations.
2802   // There are other cases where moving the shuffle after the xor/and/or
2803   // is profitable even if shuffles don't perform a swizzle.
2804   // If both shuffles use the same mask, and both shuffles have the same first
2805   // or second operand, then it might still be profitable to move the shuffle
2806   // after the xor/and/or operation.
2807   if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG) {
2808     ShuffleVectorSDNode *SVN0 = cast<ShuffleVectorSDNode>(N0);
2809     ShuffleVectorSDNode *SVN1 = cast<ShuffleVectorSDNode>(N1);
2810 
2811     assert(N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType() &&
2812            "Inputs to shuffles are not the same type");
2813 
2814     // Check that both shuffles use the same mask. The masks are known to be of
2815     // the same length because the result vector type is the same.
2816     // Check also that shuffles have only one use to avoid introducing extra
2817     // instructions.
2818     if (SVN0->hasOneUse() && SVN1->hasOneUse() &&
2819         SVN0->getMask().equals(SVN1->getMask())) {
2820       SDValue ShOp = N0->getOperand(1);
2821 
2822       // Don't try to fold this node if it requires introducing a
2823       // build vector of all zeros that might be illegal at this stage.
2824       if (N->getOpcode() == ISD::XOR && !ShOp.isUndef()) {
2825         if (!LegalTypes)
2826           ShOp = DAG.getConstant(0, SDLoc(N), VT);
2827         else
2828           ShOp = SDValue();
2829       }
2830 
2831       // (AND (shuf (A, C), shuf (B, C)) -> shuf (AND (A, B), C)
2832       // (OR  (shuf (A, C), shuf (B, C)) -> shuf (OR  (A, B), C)
2833       // (XOR (shuf (A, C), shuf (B, C)) -> shuf (XOR (A, B), V_0)
2834       if (N0.getOperand(1) == N1.getOperand(1) && ShOp.getNode()) {
2835         SDValue NewNode = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
2836                                       N0->getOperand(0), N1->getOperand(0));
2837         AddToWorklist(NewNode.getNode());
2838         return DAG.getVectorShuffle(VT, SDLoc(N), NewNode, ShOp,
2839                                     SVN0->getMask());
2840       }
2841 
2842       // Don't try to fold this node if it requires introducing a
2843       // build vector of all zeros that might be illegal at this stage.
2844       ShOp = N0->getOperand(0);
2845       if (N->getOpcode() == ISD::XOR && !ShOp.isUndef()) {
2846         if (!LegalTypes)
2847           ShOp = DAG.getConstant(0, SDLoc(N), VT);
2848         else
2849           ShOp = SDValue();
2850       }
2851 
2852       // (AND (shuf (C, A), shuf (C, B)) -> shuf (C, AND (A, B))
2853       // (OR  (shuf (C, A), shuf (C, B)) -> shuf (C, OR  (A, B))
2854       // (XOR (shuf (C, A), shuf (C, B)) -> shuf (V_0, XOR (A, B))
2855       if (N0->getOperand(0) == N1->getOperand(0) && ShOp.getNode()) {
2856         SDValue NewNode = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
2857                                       N0->getOperand(1), N1->getOperand(1));
2858         AddToWorklist(NewNode.getNode());
2859         return DAG.getVectorShuffle(VT, SDLoc(N), ShOp, NewNode,
2860                                     SVN0->getMask());
2861       }
2862     }
2863   }
2864 
2865   return SDValue();
2866 }
2867 
2868 /// This contains all DAGCombine rules which reduce two values combined by
2869 /// an And operation to a single value. This makes them reusable in the context
2870 /// of visitSELECT(). Rules involving constants are not included as
2871 /// visitSELECT() already handles those cases.
2872 SDValue DAGCombiner::visitANDLike(SDValue N0, SDValue N1,
2873                                   SDNode *LocReference) {
2874   EVT VT = N1.getValueType();
2875 
2876   // fold (and x, undef) -> 0
2877   if (N0.isUndef() || N1.isUndef())
2878     return DAG.getConstant(0, SDLoc(LocReference), VT);
2879   // fold (and (setcc x), (setcc y)) -> (setcc (and x, y))
2880   SDValue LL, LR, RL, RR, CC0, CC1;
2881   if (isSetCCEquivalent(N0, LL, LR, CC0) && isSetCCEquivalent(N1, RL, RR, CC1)){
2882     ISD::CondCode Op0 = cast<CondCodeSDNode>(CC0)->get();
2883     ISD::CondCode Op1 = cast<CondCodeSDNode>(CC1)->get();
2884 
2885     if (LR == RR && isa<ConstantSDNode>(LR) && Op0 == Op1 &&
2886         LL.getValueType().isInteger()) {
2887       // fold (and (seteq X, 0), (seteq Y, 0)) -> (seteq (or X, Y), 0)
2888       if (isNullConstant(LR) && Op1 == ISD::SETEQ) {
2889         EVT CCVT = getSetCCResultType(LR.getValueType());
2890         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2891           SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(N0),
2892                                        LR.getValueType(), LL, RL);
2893           AddToWorklist(ORNode.getNode());
2894           return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
2895         }
2896       }
2897       if (isAllOnesConstant(LR)) {
2898         // fold (and (seteq X, -1), (seteq Y, -1)) -> (seteq (and X, Y), -1)
2899         if (Op1 == ISD::SETEQ) {
2900           EVT CCVT = getSetCCResultType(LR.getValueType());
2901           if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2902             SDValue ANDNode = DAG.getNode(ISD::AND, SDLoc(N0),
2903                                           LR.getValueType(), LL, RL);
2904             AddToWorklist(ANDNode.getNode());
2905             return DAG.getSetCC(SDLoc(LocReference), VT, ANDNode, LR, Op1);
2906           }
2907         }
2908         // fold (and (setgt X, -1), (setgt Y, -1)) -> (setgt (or X, Y), -1)
2909         if (Op1 == ISD::SETGT) {
2910           EVT CCVT = getSetCCResultType(LR.getValueType());
2911           if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2912             SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(N0),
2913                                          LR.getValueType(), LL, RL);
2914             AddToWorklist(ORNode.getNode());
2915             return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
2916           }
2917         }
2918       }
2919     }
2920     // Simplify (and (setne X, 0), (setne X, -1)) -> (setuge (add X, 1), 2)
2921     if (LL == RL && isa<ConstantSDNode>(LR) && isa<ConstantSDNode>(RR) &&
2922         Op0 == Op1 && LL.getValueType().isInteger() &&
2923       Op0 == ISD::SETNE && ((isNullConstant(LR) && isAllOnesConstant(RR)) ||
2924                             (isAllOnesConstant(LR) && isNullConstant(RR)))) {
2925       EVT CCVT = getSetCCResultType(LL.getValueType());
2926       if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2927         SDLoc DL(N0);
2928         SDValue ADDNode = DAG.getNode(ISD::ADD, DL, LL.getValueType(),
2929                                       LL, DAG.getConstant(1, DL,
2930                                                           LL.getValueType()));
2931         AddToWorklist(ADDNode.getNode());
2932         return DAG.getSetCC(SDLoc(LocReference), VT, ADDNode,
2933                             DAG.getConstant(2, DL, LL.getValueType()),
2934                             ISD::SETUGE);
2935       }
2936     }
2937     // canonicalize equivalent to ll == rl
2938     if (LL == RR && LR == RL) {
2939       Op1 = ISD::getSetCCSwappedOperands(Op1);
2940       std::swap(RL, RR);
2941     }
2942     if (LL == RL && LR == RR) {
2943       bool isInteger = LL.getValueType().isInteger();
2944       ISD::CondCode Result = ISD::getSetCCAndOperation(Op0, Op1, isInteger);
2945       if (Result != ISD::SETCC_INVALID &&
2946           (!LegalOperations ||
2947            (TLI.isCondCodeLegal(Result, LL.getSimpleValueType()) &&
2948             TLI.isOperationLegal(ISD::SETCC, LL.getValueType())))) {
2949         EVT CCVT = getSetCCResultType(LL.getValueType());
2950         if (N0.getValueType() == CCVT ||
2951             (!LegalOperations && N0.getValueType() == MVT::i1))
2952           return DAG.getSetCC(SDLoc(LocReference), N0.getValueType(),
2953                               LL, LR, Result);
2954       }
2955     }
2956   }
2957 
2958   if (N0.getOpcode() == ISD::ADD && N1.getOpcode() == ISD::SRL &&
2959       VT.getSizeInBits() <= 64) {
2960     if (ConstantSDNode *ADDI = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
2961       APInt ADDC = ADDI->getAPIntValue();
2962       if (!TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
2963         // Look for (and (add x, c1), (lshr y, c2)). If C1 wasn't a legal
2964         // immediate for an add, but it is legal if its top c2 bits are set,
2965         // transform the ADD so the immediate doesn't need to be materialized
2966         // in a register.
2967         if (ConstantSDNode *SRLI = dyn_cast<ConstantSDNode>(N1.getOperand(1))) {
2968           APInt Mask = APInt::getHighBitsSet(VT.getSizeInBits(),
2969                                              SRLI->getZExtValue());
2970           if (DAG.MaskedValueIsZero(N0.getOperand(1), Mask)) {
2971             ADDC |= Mask;
2972             if (TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
2973               SDLoc DL(N0);
2974               SDValue NewAdd =
2975                 DAG.getNode(ISD::ADD, DL, VT,
2976                             N0.getOperand(0), DAG.getConstant(ADDC, DL, VT));
2977               CombineTo(N0.getNode(), NewAdd);
2978               // Return N so it doesn't get rechecked!
2979               return SDValue(LocReference, 0);
2980             }
2981           }
2982         }
2983       }
2984     }
2985   }
2986 
2987   // Reduce bit extract of low half of an integer to the narrower type.
2988   // (and (srl i64:x, K), KMask) ->
2989   //   (i64 zero_extend (and (srl (i32 (trunc i64:x)), K)), KMask)
2990   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
2991     if (ConstantSDNode *CAnd = dyn_cast<ConstantSDNode>(N1)) {
2992       if (ConstantSDNode *CShift = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
2993         unsigned Size = VT.getSizeInBits();
2994         const APInt &AndMask = CAnd->getAPIntValue();
2995         unsigned ShiftBits = CShift->getZExtValue();
2996         unsigned MaskBits = AndMask.countTrailingOnes();
2997         EVT HalfVT = EVT::getIntegerVT(*DAG.getContext(), Size / 2);
2998 
2999         if (APIntOps::isMask(AndMask) &&
3000             // Required bits must not span the two halves of the integer and
3001             // must fit in the half size type.
3002             (ShiftBits + MaskBits <= Size / 2) &&
3003             TLI.isNarrowingProfitable(VT, HalfVT) &&
3004             TLI.isTypeDesirableForOp(ISD::AND, HalfVT) &&
3005             TLI.isTypeDesirableForOp(ISD::SRL, HalfVT) &&
3006             TLI.isTruncateFree(VT, HalfVT) &&
3007             TLI.isZExtFree(HalfVT, VT)) {
3008           // The isNarrowingProfitable is to avoid regressions on PPC and
3009           // AArch64 which match a few 64-bit bit insert / bit extract patterns
3010           // on downstream users of this. Those patterns could probably be
3011           // extended to handle extensions mixed in.
3012 
3013           SDValue SL(N0);
3014           assert(ShiftBits != 0 && MaskBits <= Size);
3015 
3016           // Extracting the highest bit of the low half.
3017           EVT ShiftVT = TLI.getShiftAmountTy(HalfVT, DAG.getDataLayout());
3018           SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, HalfVT,
3019                                       N0.getOperand(0));
3020 
3021           SDValue NewMask = DAG.getConstant(AndMask.trunc(Size / 2), SL, HalfVT);
3022           SDValue ShiftK = DAG.getConstant(ShiftBits, SL, ShiftVT);
3023           SDValue Shift = DAG.getNode(ISD::SRL, SL, HalfVT, Trunc, ShiftK);
3024           SDValue And = DAG.getNode(ISD::AND, SL, HalfVT, Shift, NewMask);
3025           return DAG.getNode(ISD::ZERO_EXTEND, SL, VT, And);
3026         }
3027       }
3028     }
3029   }
3030 
3031   return SDValue();
3032 }
3033 
3034 bool DAGCombiner::isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
3035                                    EVT LoadResultTy, EVT &ExtVT, EVT &LoadedVT,
3036                                    bool &NarrowLoad) {
3037   uint32_t ActiveBits = AndC->getAPIntValue().getActiveBits();
3038 
3039   if (ActiveBits == 0 || !APIntOps::isMask(ActiveBits, AndC->getAPIntValue()))
3040     return false;
3041 
3042   ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits);
3043   LoadedVT = LoadN->getMemoryVT();
3044 
3045   if (ExtVT == LoadedVT &&
3046       (!LegalOperations ||
3047        TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))) {
3048     // ZEXTLOAD will match without needing to change the size of the value being
3049     // loaded.
3050     NarrowLoad = false;
3051     return true;
3052   }
3053 
3054   // Do not change the width of a volatile load.
3055   if (LoadN->isVolatile())
3056     return false;
3057 
3058   // Do not generate loads of non-round integer types since these can
3059   // be expensive (and would be wrong if the type is not byte sized).
3060   if (!LoadedVT.bitsGT(ExtVT) || !ExtVT.isRound())
3061     return false;
3062 
3063   if (LegalOperations &&
3064       !TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))
3065     return false;
3066 
3067   if (!TLI.shouldReduceLoadWidth(LoadN, ISD::ZEXTLOAD, ExtVT))
3068     return false;
3069 
3070   NarrowLoad = true;
3071   return true;
3072 }
3073 
3074 SDValue DAGCombiner::visitAND(SDNode *N) {
3075   SDValue N0 = N->getOperand(0);
3076   SDValue N1 = N->getOperand(1);
3077   EVT VT = N1.getValueType();
3078 
3079   // fold vector ops
3080   if (VT.isVector()) {
3081     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3082       return FoldedVOp;
3083 
3084     // fold (and x, 0) -> 0, vector edition
3085     if (ISD::isBuildVectorAllZeros(N0.getNode()))
3086       // do not return N0, because undef node may exist in N0
3087       return DAG.getConstant(APInt::getNullValue(N0.getScalarValueSizeInBits()),
3088                              SDLoc(N), N0.getValueType());
3089     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3090       // do not return N1, because undef node may exist in N1
3091       return DAG.getConstant(APInt::getNullValue(N1.getScalarValueSizeInBits()),
3092                              SDLoc(N), N1.getValueType());
3093 
3094     // fold (and x, -1) -> x, vector edition
3095     if (ISD::isBuildVectorAllOnes(N0.getNode()))
3096       return N1;
3097     if (ISD::isBuildVectorAllOnes(N1.getNode()))
3098       return N0;
3099   }
3100 
3101   // fold (and c1, c2) -> c1&c2
3102   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
3103   ConstantSDNode *N1C = isConstOrConstSplat(N1);
3104   if (N0C && N1C && !N1C->isOpaque())
3105     return DAG.FoldConstantArithmetic(ISD::AND, SDLoc(N), VT, N0C, N1C);
3106   // canonicalize constant to RHS
3107   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3108      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
3109     return DAG.getNode(ISD::AND, SDLoc(N), VT, N1, N0);
3110   // fold (and x, -1) -> x
3111   if (isAllOnesConstant(N1))
3112     return N0;
3113   // if (and x, c) is known to be zero, return 0
3114   unsigned BitWidth = VT.getScalarSizeInBits();
3115   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
3116                                    APInt::getAllOnesValue(BitWidth)))
3117     return DAG.getConstant(0, SDLoc(N), VT);
3118   // reassociate and
3119   if (SDValue RAND = ReassociateOps(ISD::AND, SDLoc(N), N0, N1))
3120     return RAND;
3121   // fold (and (or x, C), D) -> D if (C & D) == D
3122   if (N1C && N0.getOpcode() == ISD::OR)
3123     if (ConstantSDNode *ORI = isConstOrConstSplat(N0.getOperand(1)))
3124       if ((ORI->getAPIntValue() & N1C->getAPIntValue()) == N1C->getAPIntValue())
3125         return N1;
3126   // fold (and (any_ext V), c) -> (zero_ext V) if 'and' only clears top bits.
3127   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
3128     SDValue N0Op0 = N0.getOperand(0);
3129     APInt Mask = ~N1C->getAPIntValue();
3130     Mask = Mask.trunc(N0Op0.getScalarValueSizeInBits());
3131     if (DAG.MaskedValueIsZero(N0Op0, Mask)) {
3132       SDValue Zext = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N),
3133                                  N0.getValueType(), N0Op0);
3134 
3135       // Replace uses of the AND with uses of the Zero extend node.
3136       CombineTo(N, Zext);
3137 
3138       // We actually want to replace all uses of the any_extend with the
3139       // zero_extend, to avoid duplicating things.  This will later cause this
3140       // AND to be folded.
3141       CombineTo(N0.getNode(), Zext);
3142       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3143     }
3144   }
3145   // similarly fold (and (X (load ([non_ext|any_ext|zero_ext] V))), c) ->
3146   // (X (load ([non_ext|zero_ext] V))) if 'and' only clears top bits which must
3147   // already be zero by virtue of the width of the base type of the load.
3148   //
3149   // the 'X' node here can either be nothing or an extract_vector_elt to catch
3150   // more cases.
3151   if ((N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
3152        N0.getValueSizeInBits() == N0.getOperand(0).getScalarValueSizeInBits() &&
3153        N0.getOperand(0).getOpcode() == ISD::LOAD &&
3154        N0.getOperand(0).getResNo() == 0) ||
3155       (N0.getOpcode() == ISD::LOAD && N0.getResNo() == 0)) {
3156     LoadSDNode *Load = cast<LoadSDNode>( (N0.getOpcode() == ISD::LOAD) ?
3157                                          N0 : N0.getOperand(0) );
3158 
3159     // Get the constant (if applicable) the zero'th operand is being ANDed with.
3160     // This can be a pure constant or a vector splat, in which case we treat the
3161     // vector as a scalar and use the splat value.
3162     APInt Constant = APInt::getNullValue(1);
3163     if (const ConstantSDNode *C = dyn_cast<ConstantSDNode>(N1)) {
3164       Constant = C->getAPIntValue();
3165     } else if (BuildVectorSDNode *Vector = dyn_cast<BuildVectorSDNode>(N1)) {
3166       APInt SplatValue, SplatUndef;
3167       unsigned SplatBitSize;
3168       bool HasAnyUndefs;
3169       bool IsSplat = Vector->isConstantSplat(SplatValue, SplatUndef,
3170                                              SplatBitSize, HasAnyUndefs);
3171       if (IsSplat) {
3172         // Undef bits can contribute to a possible optimisation if set, so
3173         // set them.
3174         SplatValue |= SplatUndef;
3175 
3176         // The splat value may be something like "0x00FFFFFF", which means 0 for
3177         // the first vector value and FF for the rest, repeating. We need a mask
3178         // that will apply equally to all members of the vector, so AND all the
3179         // lanes of the constant together.
3180         EVT VT = Vector->getValueType(0);
3181         unsigned BitWidth = VT.getScalarSizeInBits();
3182 
3183         // If the splat value has been compressed to a bitlength lower
3184         // than the size of the vector lane, we need to re-expand it to
3185         // the lane size.
3186         if (BitWidth > SplatBitSize)
3187           for (SplatValue = SplatValue.zextOrTrunc(BitWidth);
3188                SplatBitSize < BitWidth;
3189                SplatBitSize = SplatBitSize * 2)
3190             SplatValue |= SplatValue.shl(SplatBitSize);
3191 
3192         // Make sure that variable 'Constant' is only set if 'SplatBitSize' is a
3193         // multiple of 'BitWidth'. Otherwise, we could propagate a wrong value.
3194         if (SplatBitSize % BitWidth == 0) {
3195           Constant = APInt::getAllOnesValue(BitWidth);
3196           for (unsigned i = 0, n = SplatBitSize/BitWidth; i < n; ++i)
3197             Constant &= SplatValue.lshr(i*BitWidth).zextOrTrunc(BitWidth);
3198         }
3199       }
3200     }
3201 
3202     // If we want to change an EXTLOAD to a ZEXTLOAD, ensure a ZEXTLOAD is
3203     // actually legal and isn't going to get expanded, else this is a false
3204     // optimisation.
3205     bool CanZextLoadProfitably = TLI.isLoadExtLegal(ISD::ZEXTLOAD,
3206                                                     Load->getValueType(0),
3207                                                     Load->getMemoryVT());
3208 
3209     // Resize the constant to the same size as the original memory access before
3210     // extension. If it is still the AllOnesValue then this AND is completely
3211     // unneeded.
3212     Constant = Constant.zextOrTrunc(Load->getMemoryVT().getScalarSizeInBits());
3213 
3214     bool B;
3215     switch (Load->getExtensionType()) {
3216     default: B = false; break;
3217     case ISD::EXTLOAD: B = CanZextLoadProfitably; break;
3218     case ISD::ZEXTLOAD:
3219     case ISD::NON_EXTLOAD: B = true; break;
3220     }
3221 
3222     if (B && Constant.isAllOnesValue()) {
3223       // If the load type was an EXTLOAD, convert to ZEXTLOAD in order to
3224       // preserve semantics once we get rid of the AND.
3225       SDValue NewLoad(Load, 0);
3226       if (Load->getExtensionType() == ISD::EXTLOAD) {
3227         NewLoad = DAG.getLoad(Load->getAddressingMode(), ISD::ZEXTLOAD,
3228                               Load->getValueType(0), SDLoc(Load),
3229                               Load->getChain(), Load->getBasePtr(),
3230                               Load->getOffset(), Load->getMemoryVT(),
3231                               Load->getMemOperand());
3232         // Replace uses of the EXTLOAD with the new ZEXTLOAD.
3233         if (Load->getNumValues() == 3) {
3234           // PRE/POST_INC loads have 3 values.
3235           SDValue To[] = { NewLoad.getValue(0), NewLoad.getValue(1),
3236                            NewLoad.getValue(2) };
3237           CombineTo(Load, To, 3, true);
3238         } else {
3239           CombineTo(Load, NewLoad.getValue(0), NewLoad.getValue(1));
3240         }
3241       }
3242 
3243       // Fold the AND away, taking care not to fold to the old load node if we
3244       // replaced it.
3245       CombineTo(N, (N0.getNode() == Load) ? NewLoad : N0);
3246 
3247       return SDValue(N, 0); // Return N so it doesn't get rechecked!
3248     }
3249   }
3250 
3251   // fold (and (load x), 255) -> (zextload x, i8)
3252   // fold (and (extload x, i16), 255) -> (zextload x, i8)
3253   // fold (and (any_ext (extload x, i16)), 255) -> (zextload x, i8)
3254   if (!VT.isVector() && N1C && (N0.getOpcode() == ISD::LOAD ||
3255                                 (N0.getOpcode() == ISD::ANY_EXTEND &&
3256                                  N0.getOperand(0).getOpcode() == ISD::LOAD))) {
3257     bool HasAnyExt = N0.getOpcode() == ISD::ANY_EXTEND;
3258     LoadSDNode *LN0 = HasAnyExt
3259       ? cast<LoadSDNode>(N0.getOperand(0))
3260       : cast<LoadSDNode>(N0);
3261     if (LN0->getExtensionType() != ISD::SEXTLOAD &&
3262         LN0->isUnindexed() && N0.hasOneUse() && SDValue(LN0, 0).hasOneUse()) {
3263       auto NarrowLoad = false;
3264       EVT LoadResultTy = HasAnyExt ? LN0->getValueType(0) : VT;
3265       EVT ExtVT, LoadedVT;
3266       if (isAndLoadExtLoad(N1C, LN0, LoadResultTy, ExtVT, LoadedVT,
3267                            NarrowLoad)) {
3268         if (!NarrowLoad) {
3269           SDValue NewLoad =
3270             DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN0), LoadResultTy,
3271                            LN0->getChain(), LN0->getBasePtr(), ExtVT,
3272                            LN0->getMemOperand());
3273           AddToWorklist(N);
3274           CombineTo(LN0, NewLoad, NewLoad.getValue(1));
3275           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3276         } else {
3277           EVT PtrType = LN0->getOperand(1).getValueType();
3278 
3279           unsigned Alignment = LN0->getAlignment();
3280           SDValue NewPtr = LN0->getBasePtr();
3281 
3282           // For big endian targets, we need to add an offset to the pointer
3283           // to load the correct bytes.  For little endian systems, we merely
3284           // need to read fewer bytes from the same pointer.
3285           if (DAG.getDataLayout().isBigEndian()) {
3286             unsigned LVTStoreBytes = LoadedVT.getStoreSize();
3287             unsigned EVTStoreBytes = ExtVT.getStoreSize();
3288             unsigned PtrOff = LVTStoreBytes - EVTStoreBytes;
3289             SDLoc DL(LN0);
3290             NewPtr = DAG.getNode(ISD::ADD, DL, PtrType,
3291                                  NewPtr, DAG.getConstant(PtrOff, DL, PtrType));
3292             Alignment = MinAlign(Alignment, PtrOff);
3293           }
3294 
3295           AddToWorklist(NewPtr.getNode());
3296 
3297           SDValue Load = DAG.getExtLoad(
3298               ISD::ZEXTLOAD, SDLoc(LN0), LoadResultTy, LN0->getChain(), NewPtr,
3299               LN0->getPointerInfo(), ExtVT, Alignment,
3300               LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
3301           AddToWorklist(N);
3302           CombineTo(LN0, Load, Load.getValue(1));
3303           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3304         }
3305       }
3306     }
3307   }
3308 
3309   if (SDValue Combined = visitANDLike(N0, N1, N))
3310     return Combined;
3311 
3312   // Simplify: (and (op x...), (op y...))  -> (op (and x, y))
3313   if (N0.getOpcode() == N1.getOpcode())
3314     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
3315       return Tmp;
3316 
3317   // Masking the negated extension of a boolean is just the zero-extended
3318   // boolean:
3319   // and (sub 0, zext(bool X)), 1 --> zext(bool X)
3320   // and (sub 0, sext(bool X)), 1 --> zext(bool X)
3321   //
3322   // Note: the SimplifyDemandedBits fold below can make an information-losing
3323   // transform, and then we have no way to find this better fold.
3324   if (N1C && N1C->isOne() && N0.getOpcode() == ISD::SUB) {
3325     ConstantSDNode *SubLHS = isConstOrConstSplat(N0.getOperand(0));
3326     SDValue SubRHS = N0.getOperand(1);
3327     if (SubLHS && SubLHS->isNullValue()) {
3328       if (SubRHS.getOpcode() == ISD::ZERO_EXTEND &&
3329           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
3330         return SubRHS;
3331       if (SubRHS.getOpcode() == ISD::SIGN_EXTEND &&
3332           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
3333         return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, SubRHS.getOperand(0));
3334     }
3335   }
3336 
3337   // fold (and (sign_extend_inreg x, i16 to i32), 1) -> (and x, 1)
3338   // fold (and (sra)) -> (and (srl)) when possible.
3339   if (!VT.isVector() && SimplifyDemandedBits(SDValue(N, 0)))
3340     return SDValue(N, 0);
3341 
3342   // fold (zext_inreg (extload x)) -> (zextload x)
3343   if (ISD::isEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode())) {
3344     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
3345     EVT MemVT = LN0->getMemoryVT();
3346     // If we zero all the possible extended bits, then we can turn this into
3347     // a zextload if we are running before legalize or the operation is legal.
3348     unsigned BitWidth = N1.getScalarValueSizeInBits();
3349     if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth,
3350                            BitWidth - MemVT.getScalarSizeInBits())) &&
3351         ((!LegalOperations && !LN0->isVolatile()) ||
3352          TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) {
3353       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT,
3354                                        LN0->getChain(), LN0->getBasePtr(),
3355                                        MemVT, LN0->getMemOperand());
3356       AddToWorklist(N);
3357       CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
3358       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3359     }
3360   }
3361   // fold (zext_inreg (sextload x)) -> (zextload x) iff load has one use
3362   if (ISD::isSEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
3363       N0.hasOneUse()) {
3364     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
3365     EVT MemVT = LN0->getMemoryVT();
3366     // If we zero all the possible extended bits, then we can turn this into
3367     // a zextload if we are running before legalize or the operation is legal.
3368     unsigned BitWidth = N1.getScalarValueSizeInBits();
3369     if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth,
3370                            BitWidth - MemVT.getScalarSizeInBits())) &&
3371         ((!LegalOperations && !LN0->isVolatile()) ||
3372          TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) {
3373       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT,
3374                                        LN0->getChain(), LN0->getBasePtr(),
3375                                        MemVT, LN0->getMemOperand());
3376       AddToWorklist(N);
3377       CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
3378       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3379     }
3380   }
3381   // fold (and (or (srl N, 8), (shl N, 8)), 0xffff) -> (srl (bswap N), const)
3382   if (N1C && N1C->getAPIntValue() == 0xffff && N0.getOpcode() == ISD::OR) {
3383     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
3384                                            N0.getOperand(1), false))
3385       return BSwap;
3386   }
3387 
3388   return SDValue();
3389 }
3390 
3391 /// Match (a >> 8) | (a << 8) as (bswap a) >> 16.
3392 SDValue DAGCombiner::MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
3393                                         bool DemandHighBits) {
3394   if (!LegalOperations)
3395     return SDValue();
3396 
3397   EVT VT = N->getValueType(0);
3398   if (VT != MVT::i64 && VT != MVT::i32 && VT != MVT::i16)
3399     return SDValue();
3400   if (!TLI.isOperationLegal(ISD::BSWAP, VT))
3401     return SDValue();
3402 
3403   // Recognize (and (shl a, 8), 0xff), (and (srl a, 8), 0xff00)
3404   bool LookPassAnd0 = false;
3405   bool LookPassAnd1 = false;
3406   if (N0.getOpcode() == ISD::AND && N0.getOperand(0).getOpcode() == ISD::SRL)
3407       std::swap(N0, N1);
3408   if (N1.getOpcode() == ISD::AND && N1.getOperand(0).getOpcode() == ISD::SHL)
3409       std::swap(N0, N1);
3410   if (N0.getOpcode() == ISD::AND) {
3411     if (!N0.getNode()->hasOneUse())
3412       return SDValue();
3413     ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3414     if (!N01C || N01C->getZExtValue() != 0xFF00)
3415       return SDValue();
3416     N0 = N0.getOperand(0);
3417     LookPassAnd0 = true;
3418   }
3419 
3420   if (N1.getOpcode() == ISD::AND) {
3421     if (!N1.getNode()->hasOneUse())
3422       return SDValue();
3423     ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
3424     if (!N11C || N11C->getZExtValue() != 0xFF)
3425       return SDValue();
3426     N1 = N1.getOperand(0);
3427     LookPassAnd1 = true;
3428   }
3429 
3430   if (N0.getOpcode() == ISD::SRL && N1.getOpcode() == ISD::SHL)
3431     std::swap(N0, N1);
3432   if (N0.getOpcode() != ISD::SHL || N1.getOpcode() != ISD::SRL)
3433     return SDValue();
3434   if (!N0.getNode()->hasOneUse() || !N1.getNode()->hasOneUse())
3435     return SDValue();
3436 
3437   ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3438   ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
3439   if (!N01C || !N11C)
3440     return SDValue();
3441   if (N01C->getZExtValue() != 8 || N11C->getZExtValue() != 8)
3442     return SDValue();
3443 
3444   // Look for (shl (and a, 0xff), 8), (srl (and a, 0xff00), 8)
3445   SDValue N00 = N0->getOperand(0);
3446   if (!LookPassAnd0 && N00.getOpcode() == ISD::AND) {
3447     if (!N00.getNode()->hasOneUse())
3448       return SDValue();
3449     ConstantSDNode *N001C = dyn_cast<ConstantSDNode>(N00.getOperand(1));
3450     if (!N001C || N001C->getZExtValue() != 0xFF)
3451       return SDValue();
3452     N00 = N00.getOperand(0);
3453     LookPassAnd0 = true;
3454   }
3455 
3456   SDValue N10 = N1->getOperand(0);
3457   if (!LookPassAnd1 && N10.getOpcode() == ISD::AND) {
3458     if (!N10.getNode()->hasOneUse())
3459       return SDValue();
3460     ConstantSDNode *N101C = dyn_cast<ConstantSDNode>(N10.getOperand(1));
3461     if (!N101C || N101C->getZExtValue() != 0xFF00)
3462       return SDValue();
3463     N10 = N10.getOperand(0);
3464     LookPassAnd1 = true;
3465   }
3466 
3467   if (N00 != N10)
3468     return SDValue();
3469 
3470   // Make sure everything beyond the low halfword gets set to zero since the SRL
3471   // 16 will clear the top bits.
3472   unsigned OpSizeInBits = VT.getSizeInBits();
3473   if (DemandHighBits && OpSizeInBits > 16) {
3474     // If the left-shift isn't masked out then the only way this is a bswap is
3475     // if all bits beyond the low 8 are 0. In that case the entire pattern
3476     // reduces to a left shift anyway: leave it for other parts of the combiner.
3477     if (!LookPassAnd0)
3478       return SDValue();
3479 
3480     // However, if the right shift isn't masked out then it might be because
3481     // it's not needed. See if we can spot that too.
3482     if (!LookPassAnd1 &&
3483         !DAG.MaskedValueIsZero(
3484             N10, APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - 16)))
3485       return SDValue();
3486   }
3487 
3488   SDValue Res = DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N00);
3489   if (OpSizeInBits > 16) {
3490     SDLoc DL(N);
3491     Res = DAG.getNode(ISD::SRL, DL, VT, Res,
3492                       DAG.getConstant(OpSizeInBits - 16, DL,
3493                                       getShiftAmountTy(VT)));
3494   }
3495   return Res;
3496 }
3497 
3498 /// Return true if the specified node is an element that makes up a 32-bit
3499 /// packed halfword byteswap.
3500 /// ((x & 0x000000ff) << 8) |
3501 /// ((x & 0x0000ff00) >> 8) |
3502 /// ((x & 0x00ff0000) << 8) |
3503 /// ((x & 0xff000000) >> 8)
3504 static bool isBSwapHWordElement(SDValue N, MutableArrayRef<SDNode *> Parts) {
3505   if (!N.getNode()->hasOneUse())
3506     return false;
3507 
3508   unsigned Opc = N.getOpcode();
3509   if (Opc != ISD::AND && Opc != ISD::SHL && Opc != ISD::SRL)
3510     return false;
3511 
3512   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3513   if (!N1C)
3514     return false;
3515 
3516   unsigned Num;
3517   switch (N1C->getZExtValue()) {
3518   default:
3519     return false;
3520   case 0xFF:       Num = 0; break;
3521   case 0xFF00:     Num = 1; break;
3522   case 0xFF0000:   Num = 2; break;
3523   case 0xFF000000: Num = 3; break;
3524   }
3525 
3526   // Look for (x & 0xff) << 8 as well as ((x << 8) & 0xff00).
3527   SDValue N0 = N.getOperand(0);
3528   if (Opc == ISD::AND) {
3529     if (Num == 0 || Num == 2) {
3530       // (x >> 8) & 0xff
3531       // (x >> 8) & 0xff0000
3532       if (N0.getOpcode() != ISD::SRL)
3533         return false;
3534       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3535       if (!C || C->getZExtValue() != 8)
3536         return false;
3537     } else {
3538       // (x << 8) & 0xff00
3539       // (x << 8) & 0xff000000
3540       if (N0.getOpcode() != ISD::SHL)
3541         return false;
3542       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3543       if (!C || C->getZExtValue() != 8)
3544         return false;
3545     }
3546   } else if (Opc == ISD::SHL) {
3547     // (x & 0xff) << 8
3548     // (x & 0xff0000) << 8
3549     if (Num != 0 && Num != 2)
3550       return false;
3551     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3552     if (!C || C->getZExtValue() != 8)
3553       return false;
3554   } else { // Opc == ISD::SRL
3555     // (x & 0xff00) >> 8
3556     // (x & 0xff000000) >> 8
3557     if (Num != 1 && Num != 3)
3558       return false;
3559     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3560     if (!C || C->getZExtValue() != 8)
3561       return false;
3562   }
3563 
3564   if (Parts[Num])
3565     return false;
3566 
3567   Parts[Num] = N0.getOperand(0).getNode();
3568   return true;
3569 }
3570 
3571 /// Match a 32-bit packed halfword bswap. That is
3572 /// ((x & 0x000000ff) << 8) |
3573 /// ((x & 0x0000ff00) >> 8) |
3574 /// ((x & 0x00ff0000) << 8) |
3575 /// ((x & 0xff000000) >> 8)
3576 /// => (rotl (bswap x), 16)
3577 SDValue DAGCombiner::MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1) {
3578   if (!LegalOperations)
3579     return SDValue();
3580 
3581   EVT VT = N->getValueType(0);
3582   if (VT != MVT::i32)
3583     return SDValue();
3584   if (!TLI.isOperationLegal(ISD::BSWAP, VT))
3585     return SDValue();
3586 
3587   // Look for either
3588   // (or (or (and), (and)), (or (and), (and)))
3589   // (or (or (or (and), (and)), (and)), (and))
3590   if (N0.getOpcode() != ISD::OR)
3591     return SDValue();
3592   SDValue N00 = N0.getOperand(0);
3593   SDValue N01 = N0.getOperand(1);
3594   SDNode *Parts[4] = {};
3595 
3596   if (N1.getOpcode() == ISD::OR &&
3597       N00.getNumOperands() == 2 && N01.getNumOperands() == 2) {
3598     // (or (or (and), (and)), (or (and), (and)))
3599     SDValue N000 = N00.getOperand(0);
3600     if (!isBSwapHWordElement(N000, Parts))
3601       return SDValue();
3602 
3603     SDValue N001 = N00.getOperand(1);
3604     if (!isBSwapHWordElement(N001, Parts))
3605       return SDValue();
3606     SDValue N010 = N01.getOperand(0);
3607     if (!isBSwapHWordElement(N010, Parts))
3608       return SDValue();
3609     SDValue N011 = N01.getOperand(1);
3610     if (!isBSwapHWordElement(N011, Parts))
3611       return SDValue();
3612   } else {
3613     // (or (or (or (and), (and)), (and)), (and))
3614     if (!isBSwapHWordElement(N1, Parts))
3615       return SDValue();
3616     if (!isBSwapHWordElement(N01, Parts))
3617       return SDValue();
3618     if (N00.getOpcode() != ISD::OR)
3619       return SDValue();
3620     SDValue N000 = N00.getOperand(0);
3621     if (!isBSwapHWordElement(N000, Parts))
3622       return SDValue();
3623     SDValue N001 = N00.getOperand(1);
3624     if (!isBSwapHWordElement(N001, Parts))
3625       return SDValue();
3626   }
3627 
3628   // Make sure the parts are all coming from the same node.
3629   if (Parts[0] != Parts[1] || Parts[0] != Parts[2] || Parts[0] != Parts[3])
3630     return SDValue();
3631 
3632   SDLoc DL(N);
3633   SDValue BSwap = DAG.getNode(ISD::BSWAP, DL, VT,
3634                               SDValue(Parts[0], 0));
3635 
3636   // Result of the bswap should be rotated by 16. If it's not legal, then
3637   // do  (x << 16) | (x >> 16).
3638   SDValue ShAmt = DAG.getConstant(16, DL, getShiftAmountTy(VT));
3639   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT))
3640     return DAG.getNode(ISD::ROTL, DL, VT, BSwap, ShAmt);
3641   if (TLI.isOperationLegalOrCustom(ISD::ROTR, VT))
3642     return DAG.getNode(ISD::ROTR, DL, VT, BSwap, ShAmt);
3643   return DAG.getNode(ISD::OR, DL, VT,
3644                      DAG.getNode(ISD::SHL, DL, VT, BSwap, ShAmt),
3645                      DAG.getNode(ISD::SRL, DL, VT, BSwap, ShAmt));
3646 }
3647 
3648 /// This contains all DAGCombine rules which reduce two values combined by
3649 /// an Or operation to a single value \see visitANDLike().
3650 SDValue DAGCombiner::visitORLike(SDValue N0, SDValue N1, SDNode *LocReference) {
3651   EVT VT = N1.getValueType();
3652   // fold (or x, undef) -> -1
3653   if (!LegalOperations &&
3654       (N0.isUndef() || N1.isUndef())) {
3655     EVT EltVT = VT.isVector() ? VT.getVectorElementType() : VT;
3656     return DAG.getConstant(APInt::getAllOnesValue(EltVT.getSizeInBits()),
3657                            SDLoc(LocReference), VT);
3658   }
3659   // fold (or (setcc x), (setcc y)) -> (setcc (or x, y))
3660   SDValue LL, LR, RL, RR, CC0, CC1;
3661   if (isSetCCEquivalent(N0, LL, LR, CC0) && isSetCCEquivalent(N1, RL, RR, CC1)){
3662     ISD::CondCode Op0 = cast<CondCodeSDNode>(CC0)->get();
3663     ISD::CondCode Op1 = cast<CondCodeSDNode>(CC1)->get();
3664 
3665     if (LR == RR && Op0 == Op1 && LL.getValueType().isInteger()) {
3666       // fold (or (setne X, 0), (setne Y, 0)) -> (setne (or X, Y), 0)
3667       // fold (or (setlt X, 0), (setlt Y, 0)) -> (setne (or X, Y), 0)
3668       if (isNullConstant(LR) && (Op1 == ISD::SETNE || Op1 == ISD::SETLT)) {
3669         EVT CCVT = getSetCCResultType(LR.getValueType());
3670         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
3671           SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(LR),
3672                                        LR.getValueType(), LL, RL);
3673           AddToWorklist(ORNode.getNode());
3674           return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
3675         }
3676       }
3677       // fold (or (setne X, -1), (setne Y, -1)) -> (setne (and X, Y), -1)
3678       // fold (or (setgt X, -1), (setgt Y  -1)) -> (setgt (and X, Y), -1)
3679       if (isAllOnesConstant(LR) && (Op1 == ISD::SETNE || Op1 == ISD::SETGT)) {
3680         EVT CCVT = getSetCCResultType(LR.getValueType());
3681         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
3682           SDValue ANDNode = DAG.getNode(ISD::AND, SDLoc(LR),
3683                                         LR.getValueType(), LL, RL);
3684           AddToWorklist(ANDNode.getNode());
3685           return DAG.getSetCC(SDLoc(LocReference), VT, ANDNode, LR, Op1);
3686         }
3687       }
3688     }
3689     // canonicalize equivalent to ll == rl
3690     if (LL == RR && LR == RL) {
3691       Op1 = ISD::getSetCCSwappedOperands(Op1);
3692       std::swap(RL, RR);
3693     }
3694     if (LL == RL && LR == RR) {
3695       bool isInteger = LL.getValueType().isInteger();
3696       ISD::CondCode Result = ISD::getSetCCOrOperation(Op0, Op1, isInteger);
3697       if (Result != ISD::SETCC_INVALID &&
3698           (!LegalOperations ||
3699            (TLI.isCondCodeLegal(Result, LL.getSimpleValueType()) &&
3700             TLI.isOperationLegal(ISD::SETCC, LL.getValueType())))) {
3701         EVT CCVT = getSetCCResultType(LL.getValueType());
3702         if (N0.getValueType() == CCVT ||
3703             (!LegalOperations && N0.getValueType() == MVT::i1))
3704           return DAG.getSetCC(SDLoc(LocReference), N0.getValueType(),
3705                               LL, LR, Result);
3706       }
3707     }
3708   }
3709 
3710   // (or (and X, C1), (and Y, C2))  -> (and (or X, Y), C3) if possible.
3711   if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND &&
3712       // Don't increase # computations.
3713       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
3714     // We can only do this xform if we know that bits from X that are set in C2
3715     // but not in C1 are already zero.  Likewise for Y.
3716     if (const ConstantSDNode *N0O1C =
3717         getAsNonOpaqueConstant(N0.getOperand(1))) {
3718       if (const ConstantSDNode *N1O1C =
3719           getAsNonOpaqueConstant(N1.getOperand(1))) {
3720         // We can only do this xform if we know that bits from X that are set in
3721         // C2 but not in C1 are already zero.  Likewise for Y.
3722         const APInt &LHSMask = N0O1C->getAPIntValue();
3723         const APInt &RHSMask = N1O1C->getAPIntValue();
3724 
3725         if (DAG.MaskedValueIsZero(N0.getOperand(0), RHSMask&~LHSMask) &&
3726             DAG.MaskedValueIsZero(N1.getOperand(0), LHSMask&~RHSMask)) {
3727           SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
3728                                   N0.getOperand(0), N1.getOperand(0));
3729           SDLoc DL(LocReference);
3730           return DAG.getNode(ISD::AND, DL, VT, X,
3731                              DAG.getConstant(LHSMask | RHSMask, DL, VT));
3732         }
3733       }
3734     }
3735   }
3736 
3737   // (or (and X, M), (and X, N)) -> (and X, (or M, N))
3738   if (N0.getOpcode() == ISD::AND &&
3739       N1.getOpcode() == ISD::AND &&
3740       N0.getOperand(0) == N1.getOperand(0) &&
3741       // Don't increase # computations.
3742       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
3743     SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
3744                             N0.getOperand(1), N1.getOperand(1));
3745     return DAG.getNode(ISD::AND, SDLoc(LocReference), VT, N0.getOperand(0), X);
3746   }
3747 
3748   return SDValue();
3749 }
3750 
3751 SDValue DAGCombiner::visitOR(SDNode *N) {
3752   SDValue N0 = N->getOperand(0);
3753   SDValue N1 = N->getOperand(1);
3754   EVT VT = N1.getValueType();
3755 
3756   // fold vector ops
3757   if (VT.isVector()) {
3758     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3759       return FoldedVOp;
3760 
3761     // fold (or x, 0) -> x, vector edition
3762     if (ISD::isBuildVectorAllZeros(N0.getNode()))
3763       return N1;
3764     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3765       return N0;
3766 
3767     // fold (or x, -1) -> -1, vector edition
3768     if (ISD::isBuildVectorAllOnes(N0.getNode()))
3769       // do not return N0, because undef node may exist in N0
3770       return DAG.getConstant(
3771           APInt::getAllOnesValue(N0.getScalarValueSizeInBits()), SDLoc(N),
3772           N0.getValueType());
3773     if (ISD::isBuildVectorAllOnes(N1.getNode()))
3774       // do not return N1, because undef node may exist in N1
3775       return DAG.getConstant(
3776           APInt::getAllOnesValue(N1.getScalarValueSizeInBits()), SDLoc(N),
3777           N1.getValueType());
3778 
3779     // fold (or (shuf A, V_0, MA), (shuf B, V_0, MB)) -> (shuf A, B, Mask)
3780     // Do this only if the resulting shuffle is legal.
3781     if (isa<ShuffleVectorSDNode>(N0) &&
3782         isa<ShuffleVectorSDNode>(N1) &&
3783         // Avoid folding a node with illegal type.
3784         TLI.isTypeLegal(VT)) {
3785       bool ZeroN00 = ISD::isBuildVectorAllZeros(N0.getOperand(0).getNode());
3786       bool ZeroN01 = ISD::isBuildVectorAllZeros(N0.getOperand(1).getNode());
3787       bool ZeroN10 = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
3788       bool ZeroN11 = ISD::isBuildVectorAllZeros(N1.getOperand(1).getNode());
3789       // Ensure both shuffles have a zero input.
3790       if ((ZeroN00 || ZeroN01) && (ZeroN10 || ZeroN11)) {
3791         assert((!ZeroN00 || !ZeroN01) && "Both inputs zero!");
3792         assert((!ZeroN10 || !ZeroN11) && "Both inputs zero!");
3793         const ShuffleVectorSDNode *SV0 = cast<ShuffleVectorSDNode>(N0);
3794         const ShuffleVectorSDNode *SV1 = cast<ShuffleVectorSDNode>(N1);
3795         bool CanFold = true;
3796         int NumElts = VT.getVectorNumElements();
3797         SmallVector<int, 4> Mask(NumElts);
3798 
3799         for (int i = 0; i != NumElts; ++i) {
3800           int M0 = SV0->getMaskElt(i);
3801           int M1 = SV1->getMaskElt(i);
3802 
3803           // Determine if either index is pointing to a zero vector.
3804           bool M0Zero = M0 < 0 || (ZeroN00 == (M0 < NumElts));
3805           bool M1Zero = M1 < 0 || (ZeroN10 == (M1 < NumElts));
3806 
3807           // If one element is zero and the otherside is undef, keep undef.
3808           // This also handles the case that both are undef.
3809           if ((M0Zero && M1 < 0) || (M1Zero && M0 < 0)) {
3810             Mask[i] = -1;
3811             continue;
3812           }
3813 
3814           // Make sure only one of the elements is zero.
3815           if (M0Zero == M1Zero) {
3816             CanFold = false;
3817             break;
3818           }
3819 
3820           assert((M0 >= 0 || M1 >= 0) && "Undef index!");
3821 
3822           // We have a zero and non-zero element. If the non-zero came from
3823           // SV0 make the index a LHS index. If it came from SV1, make it
3824           // a RHS index. We need to mod by NumElts because we don't care
3825           // which operand it came from in the original shuffles.
3826           Mask[i] = M1Zero ? M0 % NumElts : (M1 % NumElts) + NumElts;
3827         }
3828 
3829         if (CanFold) {
3830           SDValue NewLHS = ZeroN00 ? N0.getOperand(1) : N0.getOperand(0);
3831           SDValue NewRHS = ZeroN10 ? N1.getOperand(1) : N1.getOperand(0);
3832 
3833           bool LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
3834           if (!LegalMask) {
3835             std::swap(NewLHS, NewRHS);
3836             ShuffleVectorSDNode::commuteMask(Mask);
3837             LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
3838           }
3839 
3840           if (LegalMask)
3841             return DAG.getVectorShuffle(VT, SDLoc(N), NewLHS, NewRHS, Mask);
3842         }
3843       }
3844     }
3845   }
3846 
3847   // fold (or c1, c2) -> c1|c2
3848   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
3849   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
3850   if (N0C && N1C && !N1C->isOpaque())
3851     return DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N), VT, N0C, N1C);
3852   // canonicalize constant to RHS
3853   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3854      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
3855     return DAG.getNode(ISD::OR, SDLoc(N), VT, N1, N0);
3856   // fold (or x, 0) -> x
3857   if (isNullConstant(N1))
3858     return N0;
3859   // fold (or x, -1) -> -1
3860   if (isAllOnesConstant(N1))
3861     return N1;
3862   // fold (or x, c) -> c iff (x & ~c) == 0
3863   if (N1C && DAG.MaskedValueIsZero(N0, ~N1C->getAPIntValue()))
3864     return N1;
3865 
3866   if (SDValue Combined = visitORLike(N0, N1, N))
3867     return Combined;
3868 
3869   // Recognize halfword bswaps as (bswap + rotl 16) or (bswap + shl 16)
3870   if (SDValue BSwap = MatchBSwapHWord(N, N0, N1))
3871     return BSwap;
3872   if (SDValue BSwap = MatchBSwapHWordLow(N, N0, N1))
3873     return BSwap;
3874 
3875   // reassociate or
3876   if (SDValue ROR = ReassociateOps(ISD::OR, SDLoc(N), N0, N1))
3877     return ROR;
3878   // Canonicalize (or (and X, c1), c2) -> (and (or X, c2), c1|c2)
3879   // iff (c1 & c2) == 0.
3880   if (N1C && N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
3881              isa<ConstantSDNode>(N0.getOperand(1))) {
3882     ConstantSDNode *C1 = cast<ConstantSDNode>(N0.getOperand(1));
3883     if ((C1->getAPIntValue() & N1C->getAPIntValue()) != 0) {
3884       if (SDValue COR = DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N1), VT,
3885                                                    N1C, C1))
3886         return DAG.getNode(
3887             ISD::AND, SDLoc(N), VT,
3888             DAG.getNode(ISD::OR, SDLoc(N0), VT, N0.getOperand(0), N1), COR);
3889       return SDValue();
3890     }
3891   }
3892   // Simplify: (or (op x...), (op y...))  -> (op (or x, y))
3893   if (N0.getOpcode() == N1.getOpcode())
3894     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
3895       return Tmp;
3896 
3897   // See if this is some rotate idiom.
3898   if (SDNode *Rot = MatchRotate(N0, N1, SDLoc(N)))
3899     return SDValue(Rot, 0);
3900 
3901   // Simplify the operands using demanded-bits information.
3902   if (!VT.isVector() &&
3903       SimplifyDemandedBits(SDValue(N, 0)))
3904     return SDValue(N, 0);
3905 
3906   return SDValue();
3907 }
3908 
3909 /// Match "(X shl/srl V1) & V2" where V2 may not be present.
3910 bool DAGCombiner::MatchRotateHalf(SDValue Op, SDValue &Shift, SDValue &Mask) {
3911   if (Op.getOpcode() == ISD::AND) {
3912     if (DAG.isConstantIntBuildVectorOrConstantInt(Op.getOperand(1))) {
3913       Mask = Op.getOperand(1);
3914       Op = Op.getOperand(0);
3915     } else {
3916       return false;
3917     }
3918   }
3919 
3920   if (Op.getOpcode() == ISD::SRL || Op.getOpcode() == ISD::SHL) {
3921     Shift = Op;
3922     return true;
3923   }
3924 
3925   return false;
3926 }
3927 
3928 // Return true if we can prove that, whenever Neg and Pos are both in the
3929 // range [0, EltSize), Neg == (Pos == 0 ? 0 : EltSize - Pos).  This means that
3930 // for two opposing shifts shift1 and shift2 and a value X with OpBits bits:
3931 //
3932 //     (or (shift1 X, Neg), (shift2 X, Pos))
3933 //
3934 // reduces to a rotate in direction shift2 by Pos or (equivalently) a rotate
3935 // in direction shift1 by Neg.  The range [0, EltSize) means that we only need
3936 // to consider shift amounts with defined behavior.
3937 static bool matchRotateSub(SDValue Pos, SDValue Neg, unsigned EltSize) {
3938   // If EltSize is a power of 2 then:
3939   //
3940   //  (a) (Pos == 0 ? 0 : EltSize - Pos) == (EltSize - Pos) & (EltSize - 1)
3941   //  (b) Neg == Neg & (EltSize - 1) whenever Neg is in [0, EltSize).
3942   //
3943   // So if EltSize is a power of 2 and Neg is (and Neg', EltSize-1), we check
3944   // for the stronger condition:
3945   //
3946   //     Neg & (EltSize - 1) == (EltSize - Pos) & (EltSize - 1)    [A]
3947   //
3948   // for all Neg and Pos.  Since Neg & (EltSize - 1) == Neg' & (EltSize - 1)
3949   // we can just replace Neg with Neg' for the rest of the function.
3950   //
3951   // In other cases we check for the even stronger condition:
3952   //
3953   //     Neg == EltSize - Pos                                    [B]
3954   //
3955   // for all Neg and Pos.  Note that the (or ...) then invokes undefined
3956   // behavior if Pos == 0 (and consequently Neg == EltSize).
3957   //
3958   // We could actually use [A] whenever EltSize is a power of 2, but the
3959   // only extra cases that it would match are those uninteresting ones
3960   // where Neg and Pos are never in range at the same time.  E.g. for
3961   // EltSize == 32, using [A] would allow a Neg of the form (sub 64, Pos)
3962   // as well as (sub 32, Pos), but:
3963   //
3964   //     (or (shift1 X, (sub 64, Pos)), (shift2 X, Pos))
3965   //
3966   // always invokes undefined behavior for 32-bit X.
3967   //
3968   // Below, Mask == EltSize - 1 when using [A] and is all-ones otherwise.
3969   unsigned MaskLoBits = 0;
3970   if (Neg.getOpcode() == ISD::AND && isPowerOf2_64(EltSize)) {
3971     if (ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(1))) {
3972       if (NegC->getAPIntValue() == EltSize - 1) {
3973         Neg = Neg.getOperand(0);
3974         MaskLoBits = Log2_64(EltSize);
3975       }
3976     }
3977   }
3978 
3979   // Check whether Neg has the form (sub NegC, NegOp1) for some NegC and NegOp1.
3980   if (Neg.getOpcode() != ISD::SUB)
3981     return false;
3982   ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(0));
3983   if (!NegC)
3984     return false;
3985   SDValue NegOp1 = Neg.getOperand(1);
3986 
3987   // On the RHS of [A], if Pos is Pos' & (EltSize - 1), just replace Pos with
3988   // Pos'.  The truncation is redundant for the purpose of the equality.
3989   if (MaskLoBits && Pos.getOpcode() == ISD::AND)
3990     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1)))
3991       if (PosC->getAPIntValue() == EltSize - 1)
3992         Pos = Pos.getOperand(0);
3993 
3994   // The condition we need is now:
3995   //
3996   //     (NegC - NegOp1) & Mask == (EltSize - Pos) & Mask
3997   //
3998   // If NegOp1 == Pos then we need:
3999   //
4000   //              EltSize & Mask == NegC & Mask
4001   //
4002   // (because "x & Mask" is a truncation and distributes through subtraction).
4003   APInt Width;
4004   if (Pos == NegOp1)
4005     Width = NegC->getAPIntValue();
4006 
4007   // Check for cases where Pos has the form (add NegOp1, PosC) for some PosC.
4008   // Then the condition we want to prove becomes:
4009   //
4010   //     (NegC - NegOp1) & Mask == (EltSize - (NegOp1 + PosC)) & Mask
4011   //
4012   // which, again because "x & Mask" is a truncation, becomes:
4013   //
4014   //                NegC & Mask == (EltSize - PosC) & Mask
4015   //             EltSize & Mask == (NegC + PosC) & Mask
4016   else if (Pos.getOpcode() == ISD::ADD && Pos.getOperand(0) == NegOp1) {
4017     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1)))
4018       Width = PosC->getAPIntValue() + NegC->getAPIntValue();
4019     else
4020       return false;
4021   } else
4022     return false;
4023 
4024   // Now we just need to check that EltSize & Mask == Width & Mask.
4025   if (MaskLoBits)
4026     // EltSize & Mask is 0 since Mask is EltSize - 1.
4027     return Width.getLoBits(MaskLoBits) == 0;
4028   return Width == EltSize;
4029 }
4030 
4031 // A subroutine of MatchRotate used once we have found an OR of two opposite
4032 // shifts of Shifted.  If Neg == <operand size> - Pos then the OR reduces
4033 // to both (PosOpcode Shifted, Pos) and (NegOpcode Shifted, Neg), with the
4034 // former being preferred if supported.  InnerPos and InnerNeg are Pos and
4035 // Neg with outer conversions stripped away.
4036 SDNode *DAGCombiner::MatchRotatePosNeg(SDValue Shifted, SDValue Pos,
4037                                        SDValue Neg, SDValue InnerPos,
4038                                        SDValue InnerNeg, unsigned PosOpcode,
4039                                        unsigned NegOpcode, const SDLoc &DL) {
4040   // fold (or (shl x, (*ext y)),
4041   //          (srl x, (*ext (sub 32, y)))) ->
4042   //   (rotl x, y) or (rotr x, (sub 32, y))
4043   //
4044   // fold (or (shl x, (*ext (sub 32, y))),
4045   //          (srl x, (*ext y))) ->
4046   //   (rotr x, y) or (rotl x, (sub 32, y))
4047   EVT VT = Shifted.getValueType();
4048   if (matchRotateSub(InnerPos, InnerNeg, VT.getScalarSizeInBits())) {
4049     bool HasPos = TLI.isOperationLegalOrCustom(PosOpcode, VT);
4050     return DAG.getNode(HasPos ? PosOpcode : NegOpcode, DL, VT, Shifted,
4051                        HasPos ? Pos : Neg).getNode();
4052   }
4053 
4054   return nullptr;
4055 }
4056 
4057 // MatchRotate - Handle an 'or' of two operands.  If this is one of the many
4058 // idioms for rotate, and if the target supports rotation instructions, generate
4059 // a rot[lr].
4060 SDNode *DAGCombiner::MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL) {
4061   // Must be a legal type.  Expanded 'n promoted things won't work with rotates.
4062   EVT VT = LHS.getValueType();
4063   if (!TLI.isTypeLegal(VT)) return nullptr;
4064 
4065   // The target must have at least one rotate flavor.
4066   bool HasROTL = TLI.isOperationLegalOrCustom(ISD::ROTL, VT);
4067   bool HasROTR = TLI.isOperationLegalOrCustom(ISD::ROTR, VT);
4068   if (!HasROTL && !HasROTR) return nullptr;
4069 
4070   // Match "(X shl/srl V1) & V2" where V2 may not be present.
4071   SDValue LHSShift;   // The shift.
4072   SDValue LHSMask;    // AND value if any.
4073   if (!MatchRotateHalf(LHS, LHSShift, LHSMask))
4074     return nullptr; // Not part of a rotate.
4075 
4076   SDValue RHSShift;   // The shift.
4077   SDValue RHSMask;    // AND value if any.
4078   if (!MatchRotateHalf(RHS, RHSShift, RHSMask))
4079     return nullptr; // Not part of a rotate.
4080 
4081   if (LHSShift.getOperand(0) != RHSShift.getOperand(0))
4082     return nullptr;   // Not shifting the same value.
4083 
4084   if (LHSShift.getOpcode() == RHSShift.getOpcode())
4085     return nullptr;   // Shifts must disagree.
4086 
4087   // Canonicalize shl to left side in a shl/srl pair.
4088   if (RHSShift.getOpcode() == ISD::SHL) {
4089     std::swap(LHS, RHS);
4090     std::swap(LHSShift, RHSShift);
4091     std::swap(LHSMask, RHSMask);
4092   }
4093 
4094   unsigned EltSizeInBits = VT.getScalarSizeInBits();
4095   SDValue LHSShiftArg = LHSShift.getOperand(0);
4096   SDValue LHSShiftAmt = LHSShift.getOperand(1);
4097   SDValue RHSShiftArg = RHSShift.getOperand(0);
4098   SDValue RHSShiftAmt = RHSShift.getOperand(1);
4099 
4100   // fold (or (shl x, C1), (srl x, C2)) -> (rotl x, C1)
4101   // fold (or (shl x, C1), (srl x, C2)) -> (rotr x, C2)
4102   if (isConstOrConstSplat(LHSShiftAmt) && isConstOrConstSplat(RHSShiftAmt)) {
4103     uint64_t LShVal = isConstOrConstSplat(LHSShiftAmt)->getZExtValue();
4104     uint64_t RShVal = isConstOrConstSplat(RHSShiftAmt)->getZExtValue();
4105     if ((LShVal + RShVal) != EltSizeInBits)
4106       return nullptr;
4107 
4108     SDValue Rot = DAG.getNode(HasROTL ? ISD::ROTL : ISD::ROTR, DL, VT,
4109                               LHSShiftArg, HasROTL ? LHSShiftAmt : RHSShiftAmt);
4110 
4111     // If there is an AND of either shifted operand, apply it to the result.
4112     if (LHSMask.getNode() || RHSMask.getNode()) {
4113       APInt AllBits = APInt::getAllOnesValue(EltSizeInBits);
4114       SDValue Mask = DAG.getConstant(AllBits, DL, VT);
4115 
4116       if (LHSMask.getNode()) {
4117         APInt RHSBits = APInt::getLowBitsSet(EltSizeInBits, LShVal);
4118         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
4119                            DAG.getNode(ISD::OR, DL, VT, LHSMask,
4120                                        DAG.getConstant(RHSBits, DL, VT)));
4121       }
4122       if (RHSMask.getNode()) {
4123         APInt LHSBits = APInt::getHighBitsSet(EltSizeInBits, RShVal);
4124         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
4125                            DAG.getNode(ISD::OR, DL, VT, RHSMask,
4126                                        DAG.getConstant(LHSBits, DL, VT)));
4127       }
4128 
4129       Rot = DAG.getNode(ISD::AND, DL, VT, Rot, Mask);
4130     }
4131 
4132     return Rot.getNode();
4133   }
4134 
4135   // If there is a mask here, and we have a variable shift, we can't be sure
4136   // that we're masking out the right stuff.
4137   if (LHSMask.getNode() || RHSMask.getNode())
4138     return nullptr;
4139 
4140   // If the shift amount is sign/zext/any-extended just peel it off.
4141   SDValue LExtOp0 = LHSShiftAmt;
4142   SDValue RExtOp0 = RHSShiftAmt;
4143   if ((LHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
4144        LHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
4145        LHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
4146        LHSShiftAmt.getOpcode() == ISD::TRUNCATE) &&
4147       (RHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
4148        RHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
4149        RHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
4150        RHSShiftAmt.getOpcode() == ISD::TRUNCATE)) {
4151     LExtOp0 = LHSShiftAmt.getOperand(0);
4152     RExtOp0 = RHSShiftAmt.getOperand(0);
4153   }
4154 
4155   SDNode *TryL = MatchRotatePosNeg(LHSShiftArg, LHSShiftAmt, RHSShiftAmt,
4156                                    LExtOp0, RExtOp0, ISD::ROTL, ISD::ROTR, DL);
4157   if (TryL)
4158     return TryL;
4159 
4160   SDNode *TryR = MatchRotatePosNeg(RHSShiftArg, RHSShiftAmt, LHSShiftAmt,
4161                                    RExtOp0, LExtOp0, ISD::ROTR, ISD::ROTL, DL);
4162   if (TryR)
4163     return TryR;
4164 
4165   return nullptr;
4166 }
4167 
4168 SDValue DAGCombiner::visitXOR(SDNode *N) {
4169   SDValue N0 = N->getOperand(0);
4170   SDValue N1 = N->getOperand(1);
4171   EVT VT = N0.getValueType();
4172 
4173   // fold vector ops
4174   if (VT.isVector()) {
4175     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4176       return FoldedVOp;
4177 
4178     // fold (xor x, 0) -> x, vector edition
4179     if (ISD::isBuildVectorAllZeros(N0.getNode()))
4180       return N1;
4181     if (ISD::isBuildVectorAllZeros(N1.getNode()))
4182       return N0;
4183   }
4184 
4185   // fold (xor undef, undef) -> 0. This is a common idiom (misuse).
4186   if (N0.isUndef() && N1.isUndef())
4187     return DAG.getConstant(0, SDLoc(N), VT);
4188   // fold (xor x, undef) -> undef
4189   if (N0.isUndef())
4190     return N0;
4191   if (N1.isUndef())
4192     return N1;
4193   // fold (xor c1, c2) -> c1^c2
4194   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4195   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
4196   if (N0C && N1C)
4197     return DAG.FoldConstantArithmetic(ISD::XOR, SDLoc(N), VT, N0C, N1C);
4198   // canonicalize constant to RHS
4199   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
4200      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
4201     return DAG.getNode(ISD::XOR, SDLoc(N), VT, N1, N0);
4202   // fold (xor x, 0) -> x
4203   if (isNullConstant(N1))
4204     return N0;
4205   // reassociate xor
4206   if (SDValue RXOR = ReassociateOps(ISD::XOR, SDLoc(N), N0, N1))
4207     return RXOR;
4208 
4209   // fold !(x cc y) -> (x !cc y)
4210   SDValue LHS, RHS, CC;
4211   if (TLI.isConstTrueVal(N1.getNode()) && isSetCCEquivalent(N0, LHS, RHS, CC)) {
4212     bool isInt = LHS.getValueType().isInteger();
4213     ISD::CondCode NotCC = ISD::getSetCCInverse(cast<CondCodeSDNode>(CC)->get(),
4214                                                isInt);
4215 
4216     if (!LegalOperations ||
4217         TLI.isCondCodeLegal(NotCC, LHS.getSimpleValueType())) {
4218       switch (N0.getOpcode()) {
4219       default:
4220         llvm_unreachable("Unhandled SetCC Equivalent!");
4221       case ISD::SETCC:
4222         return DAG.getSetCC(SDLoc(N), VT, LHS, RHS, NotCC);
4223       case ISD::SELECT_CC:
4224         return DAG.getSelectCC(SDLoc(N), LHS, RHS, N0.getOperand(2),
4225                                N0.getOperand(3), NotCC);
4226       }
4227     }
4228   }
4229 
4230   // fold (not (zext (setcc x, y))) -> (zext (not (setcc x, y)))
4231   if (isOneConstant(N1) && N0.getOpcode() == ISD::ZERO_EXTEND &&
4232       N0.getNode()->hasOneUse() &&
4233       isSetCCEquivalent(N0.getOperand(0), LHS, RHS, CC)){
4234     SDValue V = N0.getOperand(0);
4235     SDLoc DL(N0);
4236     V = DAG.getNode(ISD::XOR, DL, V.getValueType(), V,
4237                     DAG.getConstant(1, DL, V.getValueType()));
4238     AddToWorklist(V.getNode());
4239     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, V);
4240   }
4241 
4242   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are setcc
4243   if (isOneConstant(N1) && VT == MVT::i1 &&
4244       (N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::AND)) {
4245     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
4246     if (isOneUseSetCC(RHS) || isOneUseSetCC(LHS)) {
4247       unsigned NewOpcode = N0.getOpcode() == ISD::AND ? ISD::OR : ISD::AND;
4248       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
4249       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
4250       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
4251       return DAG.getNode(NewOpcode, SDLoc(N), VT, LHS, RHS);
4252     }
4253   }
4254   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are constants
4255   if (isAllOnesConstant(N1) &&
4256       (N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::AND)) {
4257     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
4258     if (isa<ConstantSDNode>(RHS) || isa<ConstantSDNode>(LHS)) {
4259       unsigned NewOpcode = N0.getOpcode() == ISD::AND ? ISD::OR : ISD::AND;
4260       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
4261       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
4262       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
4263       return DAG.getNode(NewOpcode, SDLoc(N), VT, LHS, RHS);
4264     }
4265   }
4266   // fold (xor (and x, y), y) -> (and (not x), y)
4267   if (N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
4268       N0->getOperand(1) == N1) {
4269     SDValue X = N0->getOperand(0);
4270     SDValue NotX = DAG.getNOT(SDLoc(X), X, VT);
4271     AddToWorklist(NotX.getNode());
4272     return DAG.getNode(ISD::AND, SDLoc(N), VT, NotX, N1);
4273   }
4274   // fold (xor (xor x, c1), c2) -> (xor x, (xor c1, c2))
4275   if (N1C && N0.getOpcode() == ISD::XOR) {
4276     if (const ConstantSDNode *N00C = getAsNonOpaqueConstant(N0.getOperand(0))) {
4277       SDLoc DL(N);
4278       return DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(1),
4279                          DAG.getConstant(N1C->getAPIntValue() ^
4280                                          N00C->getAPIntValue(), DL, VT));
4281     }
4282     if (const ConstantSDNode *N01C = getAsNonOpaqueConstant(N0.getOperand(1))) {
4283       SDLoc DL(N);
4284       return DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(0),
4285                          DAG.getConstant(N1C->getAPIntValue() ^
4286                                          N01C->getAPIntValue(), DL, VT));
4287     }
4288   }
4289   // fold (xor x, x) -> 0
4290   if (N0 == N1)
4291     return tryFoldToZero(SDLoc(N), TLI, VT, DAG, LegalOperations, LegalTypes);
4292 
4293   // fold (xor (shl 1, x), -1) -> (rotl ~1, x)
4294   // Here is a concrete example of this equivalence:
4295   // i16   x ==  14
4296   // i16 shl ==   1 << 14  == 16384 == 0b0100000000000000
4297   // i16 xor == ~(1 << 14) == 49151 == 0b1011111111111111
4298   //
4299   // =>
4300   //
4301   // i16     ~1      == 0b1111111111111110
4302   // i16 rol(~1, 14) == 0b1011111111111111
4303   //
4304   // Some additional tips to help conceptualize this transform:
4305   // - Try to see the operation as placing a single zero in a value of all ones.
4306   // - There exists no value for x which would allow the result to contain zero.
4307   // - Values of x larger than the bitwidth are undefined and do not require a
4308   //   consistent result.
4309   // - Pushing the zero left requires shifting one bits in from the right.
4310   // A rotate left of ~1 is a nice way of achieving the desired result.
4311   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT) && N0.getOpcode() == ISD::SHL
4312       && isAllOnesConstant(N1) && isOneConstant(N0.getOperand(0))) {
4313     SDLoc DL(N);
4314     return DAG.getNode(ISD::ROTL, DL, VT, DAG.getConstant(~1, DL, VT),
4315                        N0.getOperand(1));
4316   }
4317 
4318   // Simplify: xor (op x...), (op y...)  -> (op (xor x, y))
4319   if (N0.getOpcode() == N1.getOpcode())
4320     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
4321       return Tmp;
4322 
4323   // Simplify the expression using non-local knowledge.
4324   if (!VT.isVector() &&
4325       SimplifyDemandedBits(SDValue(N, 0)))
4326     return SDValue(N, 0);
4327 
4328   return SDValue();
4329 }
4330 
4331 /// Handle transforms common to the three shifts, when the shift amount is a
4332 /// constant.
4333 SDValue DAGCombiner::visitShiftByConstant(SDNode *N, ConstantSDNode *Amt) {
4334   SDNode *LHS = N->getOperand(0).getNode();
4335   if (!LHS->hasOneUse()) return SDValue();
4336 
4337   // We want to pull some binops through shifts, so that we have (and (shift))
4338   // instead of (shift (and)), likewise for add, or, xor, etc.  This sort of
4339   // thing happens with address calculations, so it's important to canonicalize
4340   // it.
4341   bool HighBitSet = false;  // Can we transform this if the high bit is set?
4342 
4343   switch (LHS->getOpcode()) {
4344   default: return SDValue();
4345   case ISD::OR:
4346   case ISD::XOR:
4347     HighBitSet = false; // We can only transform sra if the high bit is clear.
4348     break;
4349   case ISD::AND:
4350     HighBitSet = true;  // We can only transform sra if the high bit is set.
4351     break;
4352   case ISD::ADD:
4353     if (N->getOpcode() != ISD::SHL)
4354       return SDValue(); // only shl(add) not sr[al](add).
4355     HighBitSet = false; // We can only transform sra if the high bit is clear.
4356     break;
4357   }
4358 
4359   // We require the RHS of the binop to be a constant and not opaque as well.
4360   ConstantSDNode *BinOpCst = getAsNonOpaqueConstant(LHS->getOperand(1));
4361   if (!BinOpCst) return SDValue();
4362 
4363   // FIXME: disable this unless the input to the binop is a shift by a constant.
4364   // If it is not a shift, it pessimizes some common cases like:
4365   //
4366   //    void foo(int *X, int i) { X[i & 1235] = 1; }
4367   //    int bar(int *X, int i) { return X[i & 255]; }
4368   SDNode *BinOpLHSVal = LHS->getOperand(0).getNode();
4369   if ((BinOpLHSVal->getOpcode() != ISD::SHL &&
4370        BinOpLHSVal->getOpcode() != ISD::SRA &&
4371        BinOpLHSVal->getOpcode() != ISD::SRL) ||
4372       !isa<ConstantSDNode>(BinOpLHSVal->getOperand(1)))
4373     return SDValue();
4374 
4375   EVT VT = N->getValueType(0);
4376 
4377   // If this is a signed shift right, and the high bit is modified by the
4378   // logical operation, do not perform the transformation. The highBitSet
4379   // boolean indicates the value of the high bit of the constant which would
4380   // cause it to be modified for this operation.
4381   if (N->getOpcode() == ISD::SRA) {
4382     bool BinOpRHSSignSet = BinOpCst->getAPIntValue().isNegative();
4383     if (BinOpRHSSignSet != HighBitSet)
4384       return SDValue();
4385   }
4386 
4387   if (!TLI.isDesirableToCommuteWithShift(LHS))
4388     return SDValue();
4389 
4390   // Fold the constants, shifting the binop RHS by the shift amount.
4391   SDValue NewRHS = DAG.getNode(N->getOpcode(), SDLoc(LHS->getOperand(1)),
4392                                N->getValueType(0),
4393                                LHS->getOperand(1), N->getOperand(1));
4394   assert(isa<ConstantSDNode>(NewRHS) && "Folding was not successful!");
4395 
4396   // Create the new shift.
4397   SDValue NewShift = DAG.getNode(N->getOpcode(),
4398                                  SDLoc(LHS->getOperand(0)),
4399                                  VT, LHS->getOperand(0), N->getOperand(1));
4400 
4401   // Create the new binop.
4402   return DAG.getNode(LHS->getOpcode(), SDLoc(N), VT, NewShift, NewRHS);
4403 }
4404 
4405 SDValue DAGCombiner::distributeTruncateThroughAnd(SDNode *N) {
4406   assert(N->getOpcode() == ISD::TRUNCATE);
4407   assert(N->getOperand(0).getOpcode() == ISD::AND);
4408 
4409   // (truncate:TruncVT (and N00, N01C)) -> (and (truncate:TruncVT N00), TruncC)
4410   if (N->hasOneUse() && N->getOperand(0).hasOneUse()) {
4411     SDValue N01 = N->getOperand(0).getOperand(1);
4412 
4413     if (ConstantSDNode *N01C = isConstOrConstSplat(N01)) {
4414       if (!N01C->isOpaque()) {
4415         EVT TruncVT = N->getValueType(0);
4416         SDValue N00 = N->getOperand(0).getOperand(0);
4417         APInt TruncC = N01C->getAPIntValue();
4418         TruncC = TruncC.trunc(TruncVT.getScalarSizeInBits());
4419         SDLoc DL(N);
4420 
4421         return DAG.getNode(ISD::AND, DL, TruncVT,
4422                            DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N00),
4423                            DAG.getConstant(TruncC, DL, TruncVT));
4424       }
4425     }
4426   }
4427 
4428   return SDValue();
4429 }
4430 
4431 SDValue DAGCombiner::visitRotate(SDNode *N) {
4432   // fold (rot* x, (trunc (and y, c))) -> (rot* x, (and (trunc y), (trunc c))).
4433   if (N->getOperand(1).getOpcode() == ISD::TRUNCATE &&
4434       N->getOperand(1).getOperand(0).getOpcode() == ISD::AND) {
4435     if (SDValue NewOp1 =
4436             distributeTruncateThroughAnd(N->getOperand(1).getNode()))
4437       return DAG.getNode(N->getOpcode(), SDLoc(N), N->getValueType(0),
4438                          N->getOperand(0), NewOp1);
4439   }
4440   return SDValue();
4441 }
4442 
4443 SDValue DAGCombiner::visitSHL(SDNode *N) {
4444   SDValue N0 = N->getOperand(0);
4445   SDValue N1 = N->getOperand(1);
4446   EVT VT = N0.getValueType();
4447   unsigned OpSizeInBits = VT.getScalarSizeInBits();
4448 
4449   // fold vector ops
4450   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
4451   if (VT.isVector()) {
4452     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4453       return FoldedVOp;
4454 
4455     BuildVectorSDNode *N1CV = dyn_cast<BuildVectorSDNode>(N1);
4456     // If setcc produces all-one true value then:
4457     // (shl (and (setcc) N01CV) N1CV) -> (and (setcc) N01CV<<N1CV)
4458     if (N1CV && N1CV->isConstant()) {
4459       if (N0.getOpcode() == ISD::AND) {
4460         SDValue N00 = N0->getOperand(0);
4461         SDValue N01 = N0->getOperand(1);
4462         BuildVectorSDNode *N01CV = dyn_cast<BuildVectorSDNode>(N01);
4463 
4464         if (N01CV && N01CV->isConstant() && N00.getOpcode() == ISD::SETCC &&
4465             TLI.getBooleanContents(N00.getOperand(0).getValueType()) ==
4466                 TargetLowering::ZeroOrNegativeOneBooleanContent) {
4467           if (SDValue C = DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT,
4468                                                      N01CV, N1CV))
4469             return DAG.getNode(ISD::AND, SDLoc(N), VT, N00, C);
4470         }
4471       } else {
4472         N1C = isConstOrConstSplat(N1);
4473       }
4474     }
4475   }
4476 
4477   // fold (shl c1, c2) -> c1<<c2
4478   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4479   if (N0C && N1C && !N1C->isOpaque())
4480     return DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT, N0C, N1C);
4481   // fold (shl 0, x) -> 0
4482   if (isNullConstant(N0))
4483     return N0;
4484   // fold (shl x, c >= size(x)) -> undef
4485   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
4486     return DAG.getUNDEF(VT);
4487   // fold (shl x, 0) -> x
4488   if (N1C && N1C->isNullValue())
4489     return N0;
4490   // fold (shl undef, x) -> 0
4491   if (N0.isUndef())
4492     return DAG.getConstant(0, SDLoc(N), VT);
4493   // if (shl x, c) is known to be zero, return 0
4494   if (DAG.MaskedValueIsZero(SDValue(N, 0),
4495                             APInt::getAllOnesValue(OpSizeInBits)))
4496     return DAG.getConstant(0, SDLoc(N), VT);
4497   // fold (shl x, (trunc (and y, c))) -> (shl x, (and (trunc y), (trunc c))).
4498   if (N1.getOpcode() == ISD::TRUNCATE &&
4499       N1.getOperand(0).getOpcode() == ISD::AND) {
4500     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
4501       return DAG.getNode(ISD::SHL, SDLoc(N), VT, N0, NewOp1);
4502   }
4503 
4504   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
4505     return SDValue(N, 0);
4506 
4507   // fold (shl (shl x, c1), c2) -> 0 or (shl x, (add c1, c2))
4508   if (N1C && N0.getOpcode() == ISD::SHL) {
4509     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4510       SDLoc DL(N);
4511       APInt c1 = N0C1->getAPIntValue();
4512       APInt c2 = N1C->getAPIntValue();
4513       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4514 
4515       APInt Sum = c1 + c2;
4516       if (Sum.uge(OpSizeInBits))
4517         return DAG.getConstant(0, DL, VT);
4518 
4519       return DAG.getNode(
4520           ISD::SHL, DL, VT, N0.getOperand(0),
4521           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4522     }
4523   }
4524 
4525   // fold (shl (ext (shl x, c1)), c2) -> (ext (shl x, (add c1, c2)))
4526   // For this to be valid, the second form must not preserve any of the bits
4527   // that are shifted out by the inner shift in the first form.  This means
4528   // the outer shift size must be >= the number of bits added by the ext.
4529   // As a corollary, we don't care what kind of ext it is.
4530   if (N1C && (N0.getOpcode() == ISD::ZERO_EXTEND ||
4531               N0.getOpcode() == ISD::ANY_EXTEND ||
4532               N0.getOpcode() == ISD::SIGN_EXTEND) &&
4533       N0.getOperand(0).getOpcode() == ISD::SHL) {
4534     SDValue N0Op0 = N0.getOperand(0);
4535     if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) {
4536       APInt c1 = N0Op0C1->getAPIntValue();
4537       APInt c2 = N1C->getAPIntValue();
4538       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4539 
4540       EVT InnerShiftVT = N0Op0.getValueType();
4541       uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits();
4542       if (c2.uge(OpSizeInBits - InnerShiftSize)) {
4543         SDLoc DL(N0);
4544         APInt Sum = c1 + c2;
4545         if (Sum.uge(OpSizeInBits))
4546           return DAG.getConstant(0, DL, VT);
4547 
4548         return DAG.getNode(
4549             ISD::SHL, DL, VT,
4550             DAG.getNode(N0.getOpcode(), DL, VT, N0Op0->getOperand(0)),
4551             DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4552       }
4553     }
4554   }
4555 
4556   // fold (shl (zext (srl x, C)), C) -> (zext (shl (srl x, C), C))
4557   // Only fold this if the inner zext has no other uses to avoid increasing
4558   // the total number of instructions.
4559   if (N1C && N0.getOpcode() == ISD::ZERO_EXTEND && N0.hasOneUse() &&
4560       N0.getOperand(0).getOpcode() == ISD::SRL) {
4561     SDValue N0Op0 = N0.getOperand(0);
4562     if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) {
4563       if (N0Op0C1->getAPIntValue().ult(VT.getScalarSizeInBits())) {
4564         uint64_t c1 = N0Op0C1->getZExtValue();
4565         uint64_t c2 = N1C->getZExtValue();
4566         if (c1 == c2) {
4567           SDValue NewOp0 = N0.getOperand(0);
4568           EVT CountVT = NewOp0.getOperand(1).getValueType();
4569           SDLoc DL(N);
4570           SDValue NewSHL = DAG.getNode(ISD::SHL, DL, NewOp0.getValueType(),
4571                                        NewOp0,
4572                                        DAG.getConstant(c2, DL, CountVT));
4573           AddToWorklist(NewSHL.getNode());
4574           return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N0), VT, NewSHL);
4575         }
4576       }
4577     }
4578   }
4579 
4580   // fold (shl (sr[la] exact X,  C1), C2) -> (shl    X, (C2-C1)) if C1 <= C2
4581   // fold (shl (sr[la] exact X,  C1), C2) -> (sr[la] X, (C2-C1)) if C1  > C2
4582   if (N1C && (N0.getOpcode() == ISD::SRL || N0.getOpcode() == ISD::SRA) &&
4583       cast<BinaryWithFlagsSDNode>(N0)->Flags.hasExact()) {
4584     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4585       uint64_t C1 = N0C1->getZExtValue();
4586       uint64_t C2 = N1C->getZExtValue();
4587       SDLoc DL(N);
4588       if (C1 <= C2)
4589         return DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
4590                            DAG.getConstant(C2 - C1, DL, N1.getValueType()));
4591       return DAG.getNode(N0.getOpcode(), DL, VT, N0.getOperand(0),
4592                          DAG.getConstant(C1 - C2, DL, N1.getValueType()));
4593     }
4594   }
4595 
4596   // fold (shl (srl x, c1), c2) -> (and (shl x, (sub c2, c1), MASK) or
4597   //                               (and (srl x, (sub c1, c2), MASK)
4598   // Only fold this if the inner shift has no other uses -- if it does, folding
4599   // this will increase the total number of instructions.
4600   if (N1C && N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
4601     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4602       uint64_t c1 = N0C1->getZExtValue();
4603       if (c1 < OpSizeInBits) {
4604         uint64_t c2 = N1C->getZExtValue();
4605         APInt Mask = APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - c1);
4606         SDValue Shift;
4607         if (c2 > c1) {
4608           Mask = Mask.shl(c2 - c1);
4609           SDLoc DL(N);
4610           Shift = DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
4611                               DAG.getConstant(c2 - c1, DL, N1.getValueType()));
4612         } else {
4613           Mask = Mask.lshr(c1 - c2);
4614           SDLoc DL(N);
4615           Shift = DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0),
4616                               DAG.getConstant(c1 - c2, DL, N1.getValueType()));
4617         }
4618         SDLoc DL(N0);
4619         return DAG.getNode(ISD::AND, DL, VT, Shift,
4620                            DAG.getConstant(Mask, DL, VT));
4621       }
4622     }
4623   }
4624   // fold (shl (sra x, c1), c1) -> (and x, (shl -1, c1))
4625   if (N1C && N0.getOpcode() == ISD::SRA && N1 == N0.getOperand(1)) {
4626     unsigned BitSize = VT.getScalarSizeInBits();
4627     SDLoc DL(N);
4628     SDValue HiBitsMask =
4629       DAG.getConstant(APInt::getHighBitsSet(BitSize,
4630                                             BitSize - N1C->getZExtValue()),
4631                       DL, VT);
4632     return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0),
4633                        HiBitsMask);
4634   }
4635 
4636   // fold (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
4637   // Variant of version done on multiply, except mul by a power of 2 is turned
4638   // into a shift.
4639   APInt Val;
4640   if (N1C && N0.getOpcode() == ISD::ADD && N0.getNode()->hasOneUse() &&
4641       (isa<ConstantSDNode>(N0.getOperand(1)) ||
4642        ISD::isConstantSplatVector(N0.getOperand(1).getNode(), Val))) {
4643     SDValue Shl0 = DAG.getNode(ISD::SHL, SDLoc(N0), VT, N0.getOperand(0), N1);
4644     SDValue Shl1 = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1);
4645     return DAG.getNode(ISD::ADD, SDLoc(N), VT, Shl0, Shl1);
4646   }
4647 
4648   // fold (shl (mul x, c1), c2) -> (mul x, c1 << c2)
4649   if (N1C && N0.getOpcode() == ISD::MUL && N0.getNode()->hasOneUse()) {
4650     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4651       if (SDValue Folded =
4652               DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N1), VT, N0C1, N1C))
4653         return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), Folded);
4654     }
4655   }
4656 
4657   if (N1C && !N1C->isOpaque())
4658     if (SDValue NewSHL = visitShiftByConstant(N, N1C))
4659       return NewSHL;
4660 
4661   return SDValue();
4662 }
4663 
4664 SDValue DAGCombiner::visitSRA(SDNode *N) {
4665   SDValue N0 = N->getOperand(0);
4666   SDValue N1 = N->getOperand(1);
4667   EVT VT = N0.getValueType();
4668   unsigned OpSizeInBits = VT.getScalarSizeInBits();
4669 
4670   // fold vector ops
4671   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
4672   if (VT.isVector()) {
4673     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4674       return FoldedVOp;
4675 
4676     N1C = isConstOrConstSplat(N1);
4677   }
4678 
4679   // fold (sra c1, c2) -> (sra c1, c2)
4680   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4681   if (N0C && N1C && !N1C->isOpaque())
4682     return DAG.FoldConstantArithmetic(ISD::SRA, SDLoc(N), VT, N0C, N1C);
4683   // fold (sra 0, x) -> 0
4684   if (isNullConstant(N0))
4685     return N0;
4686   // fold (sra -1, x) -> -1
4687   if (isAllOnesConstant(N0))
4688     return N0;
4689   // fold (sra x, c >= size(x)) -> undef
4690   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
4691     return DAG.getUNDEF(VT);
4692   // fold (sra x, 0) -> x
4693   if (N1C && N1C->isNullValue())
4694     return N0;
4695   // fold (sra (shl x, c1), c1) -> sext_inreg for some c1 and target supports
4696   // sext_inreg.
4697   if (N1C && N0.getOpcode() == ISD::SHL && N1 == N0.getOperand(1)) {
4698     unsigned LowBits = OpSizeInBits - (unsigned)N1C->getZExtValue();
4699     EVT ExtVT = EVT::getIntegerVT(*DAG.getContext(), LowBits);
4700     if (VT.isVector())
4701       ExtVT = EVT::getVectorVT(*DAG.getContext(),
4702                                ExtVT, VT.getVectorNumElements());
4703     if ((!LegalOperations ||
4704          TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, ExtVT)))
4705       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
4706                          N0.getOperand(0), DAG.getValueType(ExtVT));
4707   }
4708 
4709   // fold (sra (sra x, c1), c2) -> (sra x, (add c1, c2))
4710   if (N1C && N0.getOpcode() == ISD::SRA) {
4711     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4712       SDLoc DL(N);
4713       APInt c1 = N0C1->getAPIntValue();
4714       APInt c2 = N1C->getAPIntValue();
4715       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4716 
4717       APInt Sum = c1 + c2;
4718       if (Sum.uge(OpSizeInBits))
4719         Sum = APInt(OpSizeInBits, OpSizeInBits - 1);
4720 
4721       return DAG.getNode(
4722           ISD::SRA, DL, VT, N0.getOperand(0),
4723           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4724     }
4725   }
4726 
4727   // fold (sra (shl X, m), (sub result_size, n))
4728   // -> (sign_extend (trunc (shl X, (sub (sub result_size, n), m)))) for
4729   // result_size - n != m.
4730   // If truncate is free for the target sext(shl) is likely to result in better
4731   // code.
4732   if (N0.getOpcode() == ISD::SHL && N1C) {
4733     // Get the two constanst of the shifts, CN0 = m, CN = n.
4734     const ConstantSDNode *N01C = isConstOrConstSplat(N0.getOperand(1));
4735     if (N01C) {
4736       LLVMContext &Ctx = *DAG.getContext();
4737       // Determine what the truncate's result bitsize and type would be.
4738       EVT TruncVT = EVT::getIntegerVT(Ctx, OpSizeInBits - N1C->getZExtValue());
4739 
4740       if (VT.isVector())
4741         TruncVT = EVT::getVectorVT(Ctx, TruncVT, VT.getVectorNumElements());
4742 
4743       // Determine the residual right-shift amount.
4744       int ShiftAmt = N1C->getZExtValue() - N01C->getZExtValue();
4745 
4746       // If the shift is not a no-op (in which case this should be just a sign
4747       // extend already), the truncated to type is legal, sign_extend is legal
4748       // on that type, and the truncate to that type is both legal and free,
4749       // perform the transform.
4750       if ((ShiftAmt > 0) &&
4751           TLI.isOperationLegalOrCustom(ISD::SIGN_EXTEND, TruncVT) &&
4752           TLI.isOperationLegalOrCustom(ISD::TRUNCATE, VT) &&
4753           TLI.isTruncateFree(VT, TruncVT)) {
4754 
4755         SDLoc DL(N);
4756         SDValue Amt = DAG.getConstant(ShiftAmt, DL,
4757             getShiftAmountTy(N0.getOperand(0).getValueType()));
4758         SDValue Shift = DAG.getNode(ISD::SRL, DL, VT,
4759                                     N0.getOperand(0), Amt);
4760         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, TruncVT,
4761                                     Shift);
4762         return DAG.getNode(ISD::SIGN_EXTEND, DL,
4763                            N->getValueType(0), Trunc);
4764       }
4765     }
4766   }
4767 
4768   // fold (sra x, (trunc (and y, c))) -> (sra x, (and (trunc y), (trunc c))).
4769   if (N1.getOpcode() == ISD::TRUNCATE &&
4770       N1.getOperand(0).getOpcode() == ISD::AND) {
4771     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
4772       return DAG.getNode(ISD::SRA, SDLoc(N), VT, N0, NewOp1);
4773   }
4774 
4775   // fold (sra (trunc (srl x, c1)), c2) -> (trunc (sra x, c1 + c2))
4776   //      if c1 is equal to the number of bits the trunc removes
4777   if (N0.getOpcode() == ISD::TRUNCATE &&
4778       (N0.getOperand(0).getOpcode() == ISD::SRL ||
4779        N0.getOperand(0).getOpcode() == ISD::SRA) &&
4780       N0.getOperand(0).hasOneUse() &&
4781       N0.getOperand(0).getOperand(1).hasOneUse() &&
4782       N1C) {
4783     SDValue N0Op0 = N0.getOperand(0);
4784     if (ConstantSDNode *LargeShift = isConstOrConstSplat(N0Op0.getOperand(1))) {
4785       unsigned LargeShiftVal = LargeShift->getZExtValue();
4786       EVT LargeVT = N0Op0.getValueType();
4787 
4788       if (LargeVT.getScalarSizeInBits() - OpSizeInBits == LargeShiftVal) {
4789         SDLoc DL(N);
4790         SDValue Amt =
4791           DAG.getConstant(LargeShiftVal + N1C->getZExtValue(), DL,
4792                           getShiftAmountTy(N0Op0.getOperand(0).getValueType()));
4793         SDValue SRA = DAG.getNode(ISD::SRA, DL, LargeVT,
4794                                   N0Op0.getOperand(0), Amt);
4795         return DAG.getNode(ISD::TRUNCATE, DL, VT, SRA);
4796       }
4797     }
4798   }
4799 
4800   // Simplify, based on bits shifted out of the LHS.
4801   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
4802     return SDValue(N, 0);
4803 
4804 
4805   // If the sign bit is known to be zero, switch this to a SRL.
4806   if (DAG.SignBitIsZero(N0))
4807     return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, N1);
4808 
4809   if (N1C && !N1C->isOpaque())
4810     if (SDValue NewSRA = visitShiftByConstant(N, N1C))
4811       return NewSRA;
4812 
4813   return SDValue();
4814 }
4815 
4816 SDValue DAGCombiner::visitSRL(SDNode *N) {
4817   SDValue N0 = N->getOperand(0);
4818   SDValue N1 = N->getOperand(1);
4819   EVT VT = N0.getValueType();
4820   unsigned OpSizeInBits = VT.getScalarSizeInBits();
4821 
4822   // fold vector ops
4823   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
4824   if (VT.isVector()) {
4825     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4826       return FoldedVOp;
4827 
4828     N1C = isConstOrConstSplat(N1);
4829   }
4830 
4831   // fold (srl c1, c2) -> c1 >>u c2
4832   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4833   if (N0C && N1C && !N1C->isOpaque())
4834     return DAG.FoldConstantArithmetic(ISD::SRL, SDLoc(N), VT, N0C, N1C);
4835   // fold (srl 0, x) -> 0
4836   if (isNullConstant(N0))
4837     return N0;
4838   // fold (srl x, c >= size(x)) -> undef
4839   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
4840     return DAG.getUNDEF(VT);
4841   // fold (srl x, 0) -> x
4842   if (N1C && N1C->isNullValue())
4843     return N0;
4844   // if (srl x, c) is known to be zero, return 0
4845   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
4846                                    APInt::getAllOnesValue(OpSizeInBits)))
4847     return DAG.getConstant(0, SDLoc(N), VT);
4848 
4849   // fold (srl (srl x, c1), c2) -> 0 or (srl x, (add c1, c2))
4850   if (N1C && N0.getOpcode() == ISD::SRL) {
4851     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4852       SDLoc DL(N);
4853       APInt c1 = N0C1->getAPIntValue();
4854       APInt c2 = N1C->getAPIntValue();
4855       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4856 
4857       APInt Sum = c1 + c2;
4858       if (Sum.uge(OpSizeInBits))
4859         return DAG.getConstant(0, DL, VT);
4860 
4861       return DAG.getNode(
4862           ISD::SRL, DL, VT, N0.getOperand(0),
4863           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4864     }
4865   }
4866 
4867   // fold (srl (trunc (srl x, c1)), c2) -> 0 or (trunc (srl x, (add c1, c2)))
4868   if (N1C && N0.getOpcode() == ISD::TRUNCATE &&
4869       N0.getOperand(0).getOpcode() == ISD::SRL &&
4870       isa<ConstantSDNode>(N0.getOperand(0)->getOperand(1))) {
4871     uint64_t c1 =
4872       cast<ConstantSDNode>(N0.getOperand(0)->getOperand(1))->getZExtValue();
4873     uint64_t c2 = N1C->getZExtValue();
4874     EVT InnerShiftVT = N0.getOperand(0).getValueType();
4875     EVT ShiftCountVT = N0.getOperand(0)->getOperand(1).getValueType();
4876     uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits();
4877     // This is only valid if the OpSizeInBits + c1 = size of inner shift.
4878     if (c1 + OpSizeInBits == InnerShiftSize) {
4879       SDLoc DL(N0);
4880       if (c1 + c2 >= InnerShiftSize)
4881         return DAG.getConstant(0, DL, VT);
4882       return DAG.getNode(ISD::TRUNCATE, DL, VT,
4883                          DAG.getNode(ISD::SRL, DL, InnerShiftVT,
4884                                      N0.getOperand(0)->getOperand(0),
4885                                      DAG.getConstant(c1 + c2, DL,
4886                                                      ShiftCountVT)));
4887     }
4888   }
4889 
4890   // fold (srl (shl x, c), c) -> (and x, cst2)
4891   if (N1C && N0.getOpcode() == ISD::SHL && N0.getOperand(1) == N1) {
4892     unsigned BitSize = N0.getScalarValueSizeInBits();
4893     if (BitSize <= 64) {
4894       uint64_t ShAmt = N1C->getZExtValue() + 64 - BitSize;
4895       SDLoc DL(N);
4896       return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0),
4897                          DAG.getConstant(~0ULL >> ShAmt, DL, VT));
4898     }
4899   }
4900 
4901   // fold (srl (anyextend x), c) -> (and (anyextend (srl x, c)), mask)
4902   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
4903     // Shifting in all undef bits?
4904     EVT SmallVT = N0.getOperand(0).getValueType();
4905     unsigned BitSize = SmallVT.getScalarSizeInBits();
4906     if (N1C->getZExtValue() >= BitSize)
4907       return DAG.getUNDEF(VT);
4908 
4909     if (!LegalTypes || TLI.isTypeDesirableForOp(ISD::SRL, SmallVT)) {
4910       uint64_t ShiftAmt = N1C->getZExtValue();
4911       SDLoc DL0(N0);
4912       SDValue SmallShift = DAG.getNode(ISD::SRL, DL0, SmallVT,
4913                                        N0.getOperand(0),
4914                           DAG.getConstant(ShiftAmt, DL0,
4915                                           getShiftAmountTy(SmallVT)));
4916       AddToWorklist(SmallShift.getNode());
4917       APInt Mask = APInt::getAllOnesValue(OpSizeInBits).lshr(ShiftAmt);
4918       SDLoc DL(N);
4919       return DAG.getNode(ISD::AND, DL, VT,
4920                          DAG.getNode(ISD::ANY_EXTEND, DL, VT, SmallShift),
4921                          DAG.getConstant(Mask, DL, VT));
4922     }
4923   }
4924 
4925   // fold (srl (sra X, Y), 31) -> (srl X, 31).  This srl only looks at the sign
4926   // bit, which is unmodified by sra.
4927   if (N1C && N1C->getZExtValue() + 1 == OpSizeInBits) {
4928     if (N0.getOpcode() == ISD::SRA)
4929       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0.getOperand(0), N1);
4930   }
4931 
4932   // fold (srl (ctlz x), "5") -> x  iff x has one bit set (the low bit).
4933   if (N1C && N0.getOpcode() == ISD::CTLZ &&
4934       N1C->getAPIntValue() == Log2_32(OpSizeInBits)) {
4935     APInt KnownZero, KnownOne;
4936     DAG.computeKnownBits(N0.getOperand(0), KnownZero, KnownOne);
4937 
4938     // If any of the input bits are KnownOne, then the input couldn't be all
4939     // zeros, thus the result of the srl will always be zero.
4940     if (KnownOne.getBoolValue()) return DAG.getConstant(0, SDLoc(N0), VT);
4941 
4942     // If all of the bits input the to ctlz node are known to be zero, then
4943     // the result of the ctlz is "32" and the result of the shift is one.
4944     APInt UnknownBits = ~KnownZero;
4945     if (UnknownBits == 0) return DAG.getConstant(1, SDLoc(N0), VT);
4946 
4947     // Otherwise, check to see if there is exactly one bit input to the ctlz.
4948     if ((UnknownBits & (UnknownBits - 1)) == 0) {
4949       // Okay, we know that only that the single bit specified by UnknownBits
4950       // could be set on input to the CTLZ node. If this bit is set, the SRL
4951       // will return 0, if it is clear, it returns 1. Change the CTLZ/SRL pair
4952       // to an SRL/XOR pair, which is likely to simplify more.
4953       unsigned ShAmt = UnknownBits.countTrailingZeros();
4954       SDValue Op = N0.getOperand(0);
4955 
4956       if (ShAmt) {
4957         SDLoc DL(N0);
4958         Op = DAG.getNode(ISD::SRL, DL, VT, Op,
4959                   DAG.getConstant(ShAmt, DL,
4960                                   getShiftAmountTy(Op.getValueType())));
4961         AddToWorklist(Op.getNode());
4962       }
4963 
4964       SDLoc DL(N);
4965       return DAG.getNode(ISD::XOR, DL, VT,
4966                          Op, DAG.getConstant(1, DL, VT));
4967     }
4968   }
4969 
4970   // fold (srl x, (trunc (and y, c))) -> (srl x, (and (trunc y), (trunc c))).
4971   if (N1.getOpcode() == ISD::TRUNCATE &&
4972       N1.getOperand(0).getOpcode() == ISD::AND) {
4973     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
4974       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, NewOp1);
4975   }
4976 
4977   // fold operands of srl based on knowledge that the low bits are not
4978   // demanded.
4979   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
4980     return SDValue(N, 0);
4981 
4982   if (N1C && !N1C->isOpaque())
4983     if (SDValue NewSRL = visitShiftByConstant(N, N1C))
4984       return NewSRL;
4985 
4986   // Attempt to convert a srl of a load into a narrower zero-extending load.
4987   if (SDValue NarrowLoad = ReduceLoadWidth(N))
4988     return NarrowLoad;
4989 
4990   // Here is a common situation. We want to optimize:
4991   //
4992   //   %a = ...
4993   //   %b = and i32 %a, 2
4994   //   %c = srl i32 %b, 1
4995   //   brcond i32 %c ...
4996   //
4997   // into
4998   //
4999   //   %a = ...
5000   //   %b = and %a, 2
5001   //   %c = setcc eq %b, 0
5002   //   brcond %c ...
5003   //
5004   // However when after the source operand of SRL is optimized into AND, the SRL
5005   // itself may not be optimized further. Look for it and add the BRCOND into
5006   // the worklist.
5007   if (N->hasOneUse()) {
5008     SDNode *Use = *N->use_begin();
5009     if (Use->getOpcode() == ISD::BRCOND)
5010       AddToWorklist(Use);
5011     else if (Use->getOpcode() == ISD::TRUNCATE && Use->hasOneUse()) {
5012       // Also look pass the truncate.
5013       Use = *Use->use_begin();
5014       if (Use->getOpcode() == ISD::BRCOND)
5015         AddToWorklist(Use);
5016     }
5017   }
5018 
5019   return SDValue();
5020 }
5021 
5022 SDValue DAGCombiner::visitBSWAP(SDNode *N) {
5023   SDValue N0 = N->getOperand(0);
5024   EVT VT = N->getValueType(0);
5025 
5026   // fold (bswap c1) -> c2
5027   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5028     return DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N0);
5029   // fold (bswap (bswap x)) -> x
5030   if (N0.getOpcode() == ISD::BSWAP)
5031     return N0->getOperand(0);
5032   return SDValue();
5033 }
5034 
5035 SDValue DAGCombiner::visitBITREVERSE(SDNode *N) {
5036   SDValue N0 = N->getOperand(0);
5037 
5038   // fold (bitreverse (bitreverse x)) -> x
5039   if (N0.getOpcode() == ISD::BITREVERSE)
5040     return N0.getOperand(0);
5041   return SDValue();
5042 }
5043 
5044 SDValue DAGCombiner::visitCTLZ(SDNode *N) {
5045   SDValue N0 = N->getOperand(0);
5046   EVT VT = N->getValueType(0);
5047 
5048   // fold (ctlz c1) -> c2
5049   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5050     return DAG.getNode(ISD::CTLZ, SDLoc(N), VT, N0);
5051   return SDValue();
5052 }
5053 
5054 SDValue DAGCombiner::visitCTLZ_ZERO_UNDEF(SDNode *N) {
5055   SDValue N0 = N->getOperand(0);
5056   EVT VT = N->getValueType(0);
5057 
5058   // fold (ctlz_zero_undef c1) -> c2
5059   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5060     return DAG.getNode(ISD::CTLZ_ZERO_UNDEF, SDLoc(N), VT, N0);
5061   return SDValue();
5062 }
5063 
5064 SDValue DAGCombiner::visitCTTZ(SDNode *N) {
5065   SDValue N0 = N->getOperand(0);
5066   EVT VT = N->getValueType(0);
5067 
5068   // fold (cttz c1) -> c2
5069   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5070     return DAG.getNode(ISD::CTTZ, SDLoc(N), VT, N0);
5071   return SDValue();
5072 }
5073 
5074 SDValue DAGCombiner::visitCTTZ_ZERO_UNDEF(SDNode *N) {
5075   SDValue N0 = N->getOperand(0);
5076   EVT VT = N->getValueType(0);
5077 
5078   // fold (cttz_zero_undef c1) -> c2
5079   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5080     return DAG.getNode(ISD::CTTZ_ZERO_UNDEF, SDLoc(N), VT, N0);
5081   return SDValue();
5082 }
5083 
5084 SDValue DAGCombiner::visitCTPOP(SDNode *N) {
5085   SDValue N0 = N->getOperand(0);
5086   EVT VT = N->getValueType(0);
5087 
5088   // fold (ctpop c1) -> c2
5089   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5090     return DAG.getNode(ISD::CTPOP, SDLoc(N), VT, N0);
5091   return SDValue();
5092 }
5093 
5094 
5095 /// \brief Generate Min/Max node
5096 static SDValue combineMinNumMaxNum(const SDLoc &DL, EVT VT, SDValue LHS,
5097                                    SDValue RHS, SDValue True, SDValue False,
5098                                    ISD::CondCode CC, const TargetLowering &TLI,
5099                                    SelectionDAG &DAG) {
5100   if (!(LHS == True && RHS == False) && !(LHS == False && RHS == True))
5101     return SDValue();
5102 
5103   switch (CC) {
5104   case ISD::SETOLT:
5105   case ISD::SETOLE:
5106   case ISD::SETLT:
5107   case ISD::SETLE:
5108   case ISD::SETULT:
5109   case ISD::SETULE: {
5110     unsigned Opcode = (LHS == True) ? ISD::FMINNUM : ISD::FMAXNUM;
5111     if (TLI.isOperationLegal(Opcode, VT))
5112       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
5113     return SDValue();
5114   }
5115   case ISD::SETOGT:
5116   case ISD::SETOGE:
5117   case ISD::SETGT:
5118   case ISD::SETGE:
5119   case ISD::SETUGT:
5120   case ISD::SETUGE: {
5121     unsigned Opcode = (LHS == True) ? ISD::FMAXNUM : ISD::FMINNUM;
5122     if (TLI.isOperationLegal(Opcode, VT))
5123       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
5124     return SDValue();
5125   }
5126   default:
5127     return SDValue();
5128   }
5129 }
5130 
5131 // TODO: We should handle other cases of selecting between {-1,0,1} here.
5132 SDValue DAGCombiner::foldSelectOfConstants(SDNode *N) {
5133   SDValue Cond = N->getOperand(0);
5134   SDValue N1 = N->getOperand(1);
5135   SDValue N2 = N->getOperand(2);
5136   EVT VT = N->getValueType(0);
5137   EVT CondVT = Cond.getValueType();
5138   SDLoc DL(N);
5139 
5140   // fold (select Cond, 0, 1) -> (xor Cond, 1)
5141   // We can't do this reliably if integer based booleans have different contents
5142   // to floating point based booleans. This is because we can't tell whether we
5143   // have an integer-based boolean or a floating-point-based boolean unless we
5144   // can find the SETCC that produced it and inspect its operands. This is
5145   // fairly easy if C is the SETCC node, but it can potentially be
5146   // undiscoverable (or not reasonably discoverable). For example, it could be
5147   // in another basic block or it could require searching a complicated
5148   // expression.
5149   if (VT.isInteger() &&
5150       (CondVT == MVT::i1 || (CondVT.isInteger() &&
5151                              TLI.getBooleanContents(false, true) ==
5152                                  TargetLowering::ZeroOrOneBooleanContent &&
5153                              TLI.getBooleanContents(false, false) ==
5154                                  TargetLowering::ZeroOrOneBooleanContent)) &&
5155       isNullConstant(N1) && isOneConstant(N2)) {
5156     SDValue NotCond = DAG.getNode(ISD::XOR, DL, CondVT, Cond,
5157                                   DAG.getConstant(1, DL, CondVT));
5158     if (VT.bitsEq(CondVT))
5159       return NotCond;
5160     return DAG.getZExtOrTrunc(NotCond, DL, VT);
5161   }
5162 
5163   return SDValue();
5164 }
5165 
5166 SDValue DAGCombiner::visitSELECT(SDNode *N) {
5167   SDValue N0 = N->getOperand(0);
5168   SDValue N1 = N->getOperand(1);
5169   SDValue N2 = N->getOperand(2);
5170   EVT VT = N->getValueType(0);
5171   EVT VT0 = N0.getValueType();
5172 
5173   // fold (select C, X, X) -> X
5174   if (N1 == N2)
5175     return N1;
5176   if (const ConstantSDNode *N0C = dyn_cast<const ConstantSDNode>(N0)) {
5177     // fold (select true, X, Y) -> X
5178     // fold (select false, X, Y) -> Y
5179     return !N0C->isNullValue() ? N1 : N2;
5180   }
5181   // fold (select C, 1, X) -> (or C, X)
5182   if (VT == MVT::i1 && isOneConstant(N1))
5183     return DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N2);
5184 
5185   if (SDValue V = foldSelectOfConstants(N))
5186     return V;
5187 
5188   // fold (select C, 0, X) -> (and (not C), X)
5189   if (VT == VT0 && VT == MVT::i1 && isNullConstant(N1)) {
5190     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
5191     AddToWorklist(NOTNode.getNode());
5192     return DAG.getNode(ISD::AND, SDLoc(N), VT, NOTNode, N2);
5193   }
5194   // fold (select C, X, 1) -> (or (not C), X)
5195   if (VT == VT0 && VT == MVT::i1 && isOneConstant(N2)) {
5196     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
5197     AddToWorklist(NOTNode.getNode());
5198     return DAG.getNode(ISD::OR, SDLoc(N), VT, NOTNode, N1);
5199   }
5200   // fold (select C, X, 0) -> (and C, X)
5201   if (VT == MVT::i1 && isNullConstant(N2))
5202     return DAG.getNode(ISD::AND, SDLoc(N), VT, N0, N1);
5203   // fold (select X, X, Y) -> (or X, Y)
5204   // fold (select X, 1, Y) -> (or X, Y)
5205   if (VT == MVT::i1 && (N0 == N1 || isOneConstant(N1)))
5206     return DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N2);
5207   // fold (select X, Y, X) -> (and X, Y)
5208   // fold (select X, Y, 0) -> (and X, Y)
5209   if (VT == MVT::i1 && (N0 == N2 || isNullConstant(N2)))
5210     return DAG.getNode(ISD::AND, SDLoc(N), VT, N0, N1);
5211 
5212   // If we can fold this based on the true/false value, do so.
5213   if (SimplifySelectOps(N, N1, N2))
5214     return SDValue(N, 0);  // Don't revisit N.
5215 
5216   if (VT0 == MVT::i1) {
5217     // The code in this block deals with the following 2 equivalences:
5218     //    select(C0|C1, x, y) <=> select(C0, x, select(C1, x, y))
5219     //    select(C0&C1, x, y) <=> select(C0, select(C1, x, y), y)
5220     // The target can specify its prefered form with the
5221     // shouldNormalizeToSelectSequence() callback. However we always transform
5222     // to the right anyway if we find the inner select exists in the DAG anyway
5223     // and we always transform to the left side if we know that we can further
5224     // optimize the combination of the conditions.
5225     bool normalizeToSequence
5226       = TLI.shouldNormalizeToSelectSequence(*DAG.getContext(), VT);
5227     // select (and Cond0, Cond1), X, Y
5228     //   -> select Cond0, (select Cond1, X, Y), Y
5229     if (N0->getOpcode() == ISD::AND && N0->hasOneUse()) {
5230       SDValue Cond0 = N0->getOperand(0);
5231       SDValue Cond1 = N0->getOperand(1);
5232       SDValue InnerSelect = DAG.getNode(ISD::SELECT, SDLoc(N),
5233                                         N1.getValueType(), Cond1, N1, N2);
5234       if (normalizeToSequence || !InnerSelect.use_empty())
5235         return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Cond0,
5236                            InnerSelect, N2);
5237     }
5238     // select (or Cond0, Cond1), X, Y -> select Cond0, X, (select Cond1, X, Y)
5239     if (N0->getOpcode() == ISD::OR && N0->hasOneUse()) {
5240       SDValue Cond0 = N0->getOperand(0);
5241       SDValue Cond1 = N0->getOperand(1);
5242       SDValue InnerSelect = DAG.getNode(ISD::SELECT, SDLoc(N),
5243                                         N1.getValueType(), Cond1, N1, N2);
5244       if (normalizeToSequence || !InnerSelect.use_empty())
5245         return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Cond0, N1,
5246                            InnerSelect);
5247     }
5248 
5249     // select Cond0, (select Cond1, X, Y), Y -> select (and Cond0, Cond1), X, Y
5250     if (N1->getOpcode() == ISD::SELECT && N1->hasOneUse()) {
5251       SDValue N1_0 = N1->getOperand(0);
5252       SDValue N1_1 = N1->getOperand(1);
5253       SDValue N1_2 = N1->getOperand(2);
5254       if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
5255         // Create the actual and node if we can generate good code for it.
5256         if (!normalizeToSequence) {
5257           SDValue And = DAG.getNode(ISD::AND, SDLoc(N), N0.getValueType(),
5258                                     N0, N1_0);
5259           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), And,
5260                              N1_1, N2);
5261         }
5262         // Otherwise see if we can optimize the "and" to a better pattern.
5263         if (SDValue Combined = visitANDLike(N0, N1_0, N))
5264           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Combined,
5265                              N1_1, N2);
5266       }
5267     }
5268     // select Cond0, X, (select Cond1, X, Y) -> select (or Cond0, Cond1), X, Y
5269     if (N2->getOpcode() == ISD::SELECT && N2->hasOneUse()) {
5270       SDValue N2_0 = N2->getOperand(0);
5271       SDValue N2_1 = N2->getOperand(1);
5272       SDValue N2_2 = N2->getOperand(2);
5273       if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
5274         // Create the actual or node if we can generate good code for it.
5275         if (!normalizeToSequence) {
5276           SDValue Or = DAG.getNode(ISD::OR, SDLoc(N), N0.getValueType(),
5277                                    N0, N2_0);
5278           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Or,
5279                              N1, N2_2);
5280         }
5281         // Otherwise see if we can optimize to a better pattern.
5282         if (SDValue Combined = visitORLike(N0, N2_0, N))
5283           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Combined,
5284                              N1, N2_2);
5285       }
5286     }
5287   }
5288 
5289   // select (xor Cond, 1), X, Y -> select Cond, Y, X
5290   // select (xor Cond, 0), X, Y -> selext Cond, X, Y
5291   if (VT0 == MVT::i1) {
5292     if (N0->getOpcode() == ISD::XOR) {
5293       if (auto *C = dyn_cast<ConstantSDNode>(N0->getOperand(1))) {
5294         SDValue Cond0 = N0->getOperand(0);
5295         if (C->isOne())
5296           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(),
5297                              Cond0, N2, N1);
5298         else
5299           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(),
5300                              Cond0, N1, N2);
5301       }
5302     }
5303   }
5304 
5305   // fold selects based on a setcc into other things, such as min/max/abs
5306   if (N0.getOpcode() == ISD::SETCC) {
5307     // select x, y (fcmp lt x, y) -> fminnum x, y
5308     // select x, y (fcmp gt x, y) -> fmaxnum x, y
5309     //
5310     // This is OK if we don't care about what happens if either operand is a
5311     // NaN.
5312     //
5313 
5314     // FIXME: Instead of testing for UnsafeFPMath, this should be checking for
5315     // no signed zeros as well as no nans.
5316     const TargetOptions &Options = DAG.getTarget().Options;
5317     if (Options.UnsafeFPMath &&
5318         VT.isFloatingPoint() && N0.hasOneUse() &&
5319         DAG.isKnownNeverNaN(N1) && DAG.isKnownNeverNaN(N2)) {
5320       ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
5321 
5322       if (SDValue FMinMax = combineMinNumMaxNum(SDLoc(N), VT, N0.getOperand(0),
5323                                                 N0.getOperand(1), N1, N2, CC,
5324                                                 TLI, DAG))
5325         return FMinMax;
5326     }
5327 
5328     if ((!LegalOperations &&
5329          TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT)) ||
5330         TLI.isOperationLegal(ISD::SELECT_CC, VT))
5331       return DAG.getNode(ISD::SELECT_CC, SDLoc(N), VT,
5332                          N0.getOperand(0), N0.getOperand(1),
5333                          N1, N2, N0.getOperand(2));
5334     return SimplifySelect(SDLoc(N), N0, N1, N2);
5335   }
5336 
5337   return SDValue();
5338 }
5339 
5340 static
5341 std::pair<SDValue, SDValue> SplitVSETCC(const SDNode *N, SelectionDAG &DAG) {
5342   SDLoc DL(N);
5343   EVT LoVT, HiVT;
5344   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(N->getValueType(0));
5345 
5346   // Split the inputs.
5347   SDValue Lo, Hi, LL, LH, RL, RH;
5348   std::tie(LL, LH) = DAG.SplitVectorOperand(N, 0);
5349   std::tie(RL, RH) = DAG.SplitVectorOperand(N, 1);
5350 
5351   Lo = DAG.getNode(N->getOpcode(), DL, LoVT, LL, RL, N->getOperand(2));
5352   Hi = DAG.getNode(N->getOpcode(), DL, HiVT, LH, RH, N->getOperand(2));
5353 
5354   return std::make_pair(Lo, Hi);
5355 }
5356 
5357 // This function assumes all the vselect's arguments are CONCAT_VECTOR
5358 // nodes and that the condition is a BV of ConstantSDNodes (or undefs).
5359 static SDValue ConvertSelectToConcatVector(SDNode *N, SelectionDAG &DAG) {
5360   SDLoc DL(N);
5361   SDValue Cond = N->getOperand(0);
5362   SDValue LHS = N->getOperand(1);
5363   SDValue RHS = N->getOperand(2);
5364   EVT VT = N->getValueType(0);
5365   int NumElems = VT.getVectorNumElements();
5366   assert(LHS.getOpcode() == ISD::CONCAT_VECTORS &&
5367          RHS.getOpcode() == ISD::CONCAT_VECTORS &&
5368          Cond.getOpcode() == ISD::BUILD_VECTOR);
5369 
5370   // CONCAT_VECTOR can take an arbitrary number of arguments. We only care about
5371   // binary ones here.
5372   if (LHS->getNumOperands() != 2 || RHS->getNumOperands() != 2)
5373     return SDValue();
5374 
5375   // We're sure we have an even number of elements due to the
5376   // concat_vectors we have as arguments to vselect.
5377   // Skip BV elements until we find one that's not an UNDEF
5378   // After we find an UNDEF element, keep looping until we get to half the
5379   // length of the BV and see if all the non-undef nodes are the same.
5380   ConstantSDNode *BottomHalf = nullptr;
5381   for (int i = 0; i < NumElems / 2; ++i) {
5382     if (Cond->getOperand(i)->isUndef())
5383       continue;
5384 
5385     if (BottomHalf == nullptr)
5386       BottomHalf = cast<ConstantSDNode>(Cond.getOperand(i));
5387     else if (Cond->getOperand(i).getNode() != BottomHalf)
5388       return SDValue();
5389   }
5390 
5391   // Do the same for the second half of the BuildVector
5392   ConstantSDNode *TopHalf = nullptr;
5393   for (int i = NumElems / 2; i < NumElems; ++i) {
5394     if (Cond->getOperand(i)->isUndef())
5395       continue;
5396 
5397     if (TopHalf == nullptr)
5398       TopHalf = cast<ConstantSDNode>(Cond.getOperand(i));
5399     else if (Cond->getOperand(i).getNode() != TopHalf)
5400       return SDValue();
5401   }
5402 
5403   assert(TopHalf && BottomHalf &&
5404          "One half of the selector was all UNDEFs and the other was all the "
5405          "same value. This should have been addressed before this function.");
5406   return DAG.getNode(
5407       ISD::CONCAT_VECTORS, DL, VT,
5408       BottomHalf->isNullValue() ? RHS->getOperand(0) : LHS->getOperand(0),
5409       TopHalf->isNullValue() ? RHS->getOperand(1) : LHS->getOperand(1));
5410 }
5411 
5412 SDValue DAGCombiner::visitMSCATTER(SDNode *N) {
5413 
5414   if (Level >= AfterLegalizeTypes)
5415     return SDValue();
5416 
5417   MaskedScatterSDNode *MSC = cast<MaskedScatterSDNode>(N);
5418   SDValue Mask = MSC->getMask();
5419   SDValue Data  = MSC->getValue();
5420   SDLoc DL(N);
5421 
5422   // If the MSCATTER data type requires splitting and the mask is provided by a
5423   // SETCC, then split both nodes and its operands before legalization. This
5424   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5425   // and enables future optimizations (e.g. min/max pattern matching on X86).
5426   if (Mask.getOpcode() != ISD::SETCC)
5427     return SDValue();
5428 
5429   // Check if any splitting is required.
5430   if (TLI.getTypeAction(*DAG.getContext(), Data.getValueType()) !=
5431       TargetLowering::TypeSplitVector)
5432     return SDValue();
5433   SDValue MaskLo, MaskHi, Lo, Hi;
5434   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5435 
5436   EVT LoVT, HiVT;
5437   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MSC->getValueType(0));
5438 
5439   SDValue Chain = MSC->getChain();
5440 
5441   EVT MemoryVT = MSC->getMemoryVT();
5442   unsigned Alignment = MSC->getOriginalAlignment();
5443 
5444   EVT LoMemVT, HiMemVT;
5445   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5446 
5447   SDValue DataLo, DataHi;
5448   std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
5449 
5450   SDValue BasePtr = MSC->getBasePtr();
5451   SDValue IndexLo, IndexHi;
5452   std::tie(IndexLo, IndexHi) = DAG.SplitVector(MSC->getIndex(), DL);
5453 
5454   MachineMemOperand *MMO = DAG.getMachineFunction().
5455     getMachineMemOperand(MSC->getPointerInfo(),
5456                           MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
5457                           Alignment, MSC->getAAInfo(), MSC->getRanges());
5458 
5459   SDValue OpsLo[] = { Chain, DataLo, MaskLo, BasePtr, IndexLo };
5460   Lo = DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataLo.getValueType(),
5461                             DL, OpsLo, MMO);
5462 
5463   SDValue OpsHi[] = {Chain, DataHi, MaskHi, BasePtr, IndexHi};
5464   Hi = DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataHi.getValueType(),
5465                             DL, OpsHi, MMO);
5466 
5467   AddToWorklist(Lo.getNode());
5468   AddToWorklist(Hi.getNode());
5469 
5470   return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi);
5471 }
5472 
5473 SDValue DAGCombiner::visitMSTORE(SDNode *N) {
5474 
5475   if (Level >= AfterLegalizeTypes)
5476     return SDValue();
5477 
5478   MaskedStoreSDNode *MST = dyn_cast<MaskedStoreSDNode>(N);
5479   SDValue Mask = MST->getMask();
5480   SDValue Data  = MST->getValue();
5481   SDLoc DL(N);
5482 
5483   // If the MSTORE data type requires splitting and the mask is provided by a
5484   // SETCC, then split both nodes and its operands before legalization. This
5485   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5486   // and enables future optimizations (e.g. min/max pattern matching on X86).
5487   if (Mask.getOpcode() == ISD::SETCC) {
5488 
5489     // Check if any splitting is required.
5490     if (TLI.getTypeAction(*DAG.getContext(), Data.getValueType()) !=
5491         TargetLowering::TypeSplitVector)
5492       return SDValue();
5493 
5494     SDValue MaskLo, MaskHi, Lo, Hi;
5495     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5496 
5497     EVT LoVT, HiVT;
5498     std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MST->getValueType(0));
5499 
5500     SDValue Chain = MST->getChain();
5501     SDValue Ptr   = MST->getBasePtr();
5502 
5503     EVT MemoryVT = MST->getMemoryVT();
5504     unsigned Alignment = MST->getOriginalAlignment();
5505 
5506     // if Alignment is equal to the vector size,
5507     // take the half of it for the second part
5508     unsigned SecondHalfAlignment =
5509       (Alignment == Data->getValueType(0).getSizeInBits()/8) ?
5510          Alignment/2 : Alignment;
5511 
5512     EVT LoMemVT, HiMemVT;
5513     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5514 
5515     SDValue DataLo, DataHi;
5516     std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
5517 
5518     MachineMemOperand *MMO = DAG.getMachineFunction().
5519       getMachineMemOperand(MST->getPointerInfo(),
5520                            MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
5521                            Alignment, MST->getAAInfo(), MST->getRanges());
5522 
5523     Lo = DAG.getMaskedStore(Chain, DL, DataLo, Ptr, MaskLo, LoMemVT, MMO,
5524                             MST->isTruncatingStore());
5525 
5526     unsigned IncrementSize = LoMemVT.getSizeInBits()/8;
5527     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
5528                       DAG.getConstant(IncrementSize, DL, Ptr.getValueType()));
5529 
5530     MMO = DAG.getMachineFunction().
5531       getMachineMemOperand(MST->getPointerInfo(),
5532                            MachineMemOperand::MOStore,  HiMemVT.getStoreSize(),
5533                            SecondHalfAlignment, MST->getAAInfo(),
5534                            MST->getRanges());
5535 
5536     Hi = DAG.getMaskedStore(Chain, DL, DataHi, Ptr, MaskHi, HiMemVT, MMO,
5537                             MST->isTruncatingStore());
5538 
5539     AddToWorklist(Lo.getNode());
5540     AddToWorklist(Hi.getNode());
5541 
5542     return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi);
5543   }
5544   return SDValue();
5545 }
5546 
5547 SDValue DAGCombiner::visitMGATHER(SDNode *N) {
5548 
5549   if (Level >= AfterLegalizeTypes)
5550     return SDValue();
5551 
5552   MaskedGatherSDNode *MGT = dyn_cast<MaskedGatherSDNode>(N);
5553   SDValue Mask = MGT->getMask();
5554   SDLoc DL(N);
5555 
5556   // If the MGATHER result requires splitting and the mask is provided by a
5557   // SETCC, then split both nodes and its operands before legalization. This
5558   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5559   // and enables future optimizations (e.g. min/max pattern matching on X86).
5560 
5561   if (Mask.getOpcode() != ISD::SETCC)
5562     return SDValue();
5563 
5564   EVT VT = N->getValueType(0);
5565 
5566   // Check if any splitting is required.
5567   if (TLI.getTypeAction(*DAG.getContext(), VT) !=
5568       TargetLowering::TypeSplitVector)
5569     return SDValue();
5570 
5571   SDValue MaskLo, MaskHi, Lo, Hi;
5572   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5573 
5574   SDValue Src0 = MGT->getValue();
5575   SDValue Src0Lo, Src0Hi;
5576   std::tie(Src0Lo, Src0Hi) = DAG.SplitVector(Src0, DL);
5577 
5578   EVT LoVT, HiVT;
5579   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
5580 
5581   SDValue Chain = MGT->getChain();
5582   EVT MemoryVT = MGT->getMemoryVT();
5583   unsigned Alignment = MGT->getOriginalAlignment();
5584 
5585   EVT LoMemVT, HiMemVT;
5586   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5587 
5588   SDValue BasePtr = MGT->getBasePtr();
5589   SDValue Index = MGT->getIndex();
5590   SDValue IndexLo, IndexHi;
5591   std::tie(IndexLo, IndexHi) = DAG.SplitVector(Index, DL);
5592 
5593   MachineMemOperand *MMO = DAG.getMachineFunction().
5594     getMachineMemOperand(MGT->getPointerInfo(),
5595                           MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
5596                           Alignment, MGT->getAAInfo(), MGT->getRanges());
5597 
5598   SDValue OpsLo[] = { Chain, Src0Lo, MaskLo, BasePtr, IndexLo };
5599   Lo = DAG.getMaskedGather(DAG.getVTList(LoVT, MVT::Other), LoVT, DL, OpsLo,
5600                             MMO);
5601 
5602   SDValue OpsHi[] = {Chain, Src0Hi, MaskHi, BasePtr, IndexHi};
5603   Hi = DAG.getMaskedGather(DAG.getVTList(HiVT, MVT::Other), HiVT, DL, OpsHi,
5604                             MMO);
5605 
5606   AddToWorklist(Lo.getNode());
5607   AddToWorklist(Hi.getNode());
5608 
5609   // Build a factor node to remember that this load is independent of the
5610   // other one.
5611   Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
5612                       Hi.getValue(1));
5613 
5614   // Legalized the chain result - switch anything that used the old chain to
5615   // use the new one.
5616   DAG.ReplaceAllUsesOfValueWith(SDValue(MGT, 1), Chain);
5617 
5618   SDValue GatherRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
5619 
5620   SDValue RetOps[] = { GatherRes, Chain };
5621   return DAG.getMergeValues(RetOps, DL);
5622 }
5623 
5624 SDValue DAGCombiner::visitMLOAD(SDNode *N) {
5625 
5626   if (Level >= AfterLegalizeTypes)
5627     return SDValue();
5628 
5629   MaskedLoadSDNode *MLD = dyn_cast<MaskedLoadSDNode>(N);
5630   SDValue Mask = MLD->getMask();
5631   SDLoc DL(N);
5632 
5633   // If the MLOAD result requires splitting and the mask is provided by a
5634   // SETCC, then split both nodes and its operands before legalization. This
5635   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5636   // and enables future optimizations (e.g. min/max pattern matching on X86).
5637 
5638   if (Mask.getOpcode() == ISD::SETCC) {
5639     EVT VT = N->getValueType(0);
5640 
5641     // Check if any splitting is required.
5642     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
5643         TargetLowering::TypeSplitVector)
5644       return SDValue();
5645 
5646     SDValue MaskLo, MaskHi, Lo, Hi;
5647     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5648 
5649     SDValue Src0 = MLD->getSrc0();
5650     SDValue Src0Lo, Src0Hi;
5651     std::tie(Src0Lo, Src0Hi) = DAG.SplitVector(Src0, DL);
5652 
5653     EVT LoVT, HiVT;
5654     std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MLD->getValueType(0));
5655 
5656     SDValue Chain = MLD->getChain();
5657     SDValue Ptr   = MLD->getBasePtr();
5658     EVT MemoryVT = MLD->getMemoryVT();
5659     unsigned Alignment = MLD->getOriginalAlignment();
5660 
5661     // if Alignment is equal to the vector size,
5662     // take the half of it for the second part
5663     unsigned SecondHalfAlignment =
5664       (Alignment == MLD->getValueType(0).getSizeInBits()/8) ?
5665          Alignment/2 : Alignment;
5666 
5667     EVT LoMemVT, HiMemVT;
5668     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5669 
5670     MachineMemOperand *MMO = DAG.getMachineFunction().
5671     getMachineMemOperand(MLD->getPointerInfo(),
5672                          MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
5673                          Alignment, MLD->getAAInfo(), MLD->getRanges());
5674 
5675     Lo = DAG.getMaskedLoad(LoVT, DL, Chain, Ptr, MaskLo, Src0Lo, LoMemVT, MMO,
5676                            ISD::NON_EXTLOAD);
5677 
5678     unsigned IncrementSize = LoMemVT.getSizeInBits()/8;
5679     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
5680                       DAG.getConstant(IncrementSize, DL, Ptr.getValueType()));
5681 
5682     MMO = DAG.getMachineFunction().
5683     getMachineMemOperand(MLD->getPointerInfo(),
5684                          MachineMemOperand::MOLoad,  HiMemVT.getStoreSize(),
5685                          SecondHalfAlignment, MLD->getAAInfo(), MLD->getRanges());
5686 
5687     Hi = DAG.getMaskedLoad(HiVT, DL, Chain, Ptr, MaskHi, Src0Hi, HiMemVT, MMO,
5688                            ISD::NON_EXTLOAD);
5689 
5690     AddToWorklist(Lo.getNode());
5691     AddToWorklist(Hi.getNode());
5692 
5693     // Build a factor node to remember that this load is independent of the
5694     // other one.
5695     Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
5696                         Hi.getValue(1));
5697 
5698     // Legalized the chain result - switch anything that used the old chain to
5699     // use the new one.
5700     DAG.ReplaceAllUsesOfValueWith(SDValue(MLD, 1), Chain);
5701 
5702     SDValue LoadRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
5703 
5704     SDValue RetOps[] = { LoadRes, Chain };
5705     return DAG.getMergeValues(RetOps, DL);
5706   }
5707   return SDValue();
5708 }
5709 
5710 SDValue DAGCombiner::visitVSELECT(SDNode *N) {
5711   SDValue N0 = N->getOperand(0);
5712   SDValue N1 = N->getOperand(1);
5713   SDValue N2 = N->getOperand(2);
5714   SDLoc DL(N);
5715 
5716   // Canonicalize integer abs.
5717   // vselect (setg[te] X,  0),  X, -X ->
5718   // vselect (setgt    X, -1),  X, -X ->
5719   // vselect (setl[te] X,  0), -X,  X ->
5720   // Y = sra (X, size(X)-1); xor (add (X, Y), Y)
5721   if (N0.getOpcode() == ISD::SETCC) {
5722     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
5723     ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
5724     bool isAbs = false;
5725     bool RHSIsAllZeros = ISD::isBuildVectorAllZeros(RHS.getNode());
5726 
5727     if (((RHSIsAllZeros && (CC == ISD::SETGT || CC == ISD::SETGE)) ||
5728          (ISD::isBuildVectorAllOnes(RHS.getNode()) && CC == ISD::SETGT)) &&
5729         N1 == LHS && N2.getOpcode() == ISD::SUB && N1 == N2.getOperand(1))
5730       isAbs = ISD::isBuildVectorAllZeros(N2.getOperand(0).getNode());
5731     else if ((RHSIsAllZeros && (CC == ISD::SETLT || CC == ISD::SETLE)) &&
5732              N2 == LHS && N1.getOpcode() == ISD::SUB && N2 == N1.getOperand(1))
5733       isAbs = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
5734 
5735     if (isAbs) {
5736       EVT VT = LHS.getValueType();
5737       SDValue Shift = DAG.getNode(
5738           ISD::SRA, DL, VT, LHS,
5739           DAG.getConstant(VT.getScalarSizeInBits() - 1, DL, VT));
5740       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, LHS, Shift);
5741       AddToWorklist(Shift.getNode());
5742       AddToWorklist(Add.getNode());
5743       return DAG.getNode(ISD::XOR, DL, VT, Add, Shift);
5744     }
5745   }
5746 
5747   if (SimplifySelectOps(N, N1, N2))
5748     return SDValue(N, 0);  // Don't revisit N.
5749 
5750   // If the VSELECT result requires splitting and the mask is provided by a
5751   // SETCC, then split both nodes and its operands before legalization. This
5752   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5753   // and enables future optimizations (e.g. min/max pattern matching on X86).
5754   if (N0.getOpcode() == ISD::SETCC) {
5755     EVT VT = N->getValueType(0);
5756 
5757     // Check if any splitting is required.
5758     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
5759         TargetLowering::TypeSplitVector)
5760       return SDValue();
5761 
5762     SDValue Lo, Hi, CCLo, CCHi, LL, LH, RL, RH;
5763     std::tie(CCLo, CCHi) = SplitVSETCC(N0.getNode(), DAG);
5764     std::tie(LL, LH) = DAG.SplitVectorOperand(N, 1);
5765     std::tie(RL, RH) = DAG.SplitVectorOperand(N, 2);
5766 
5767     Lo = DAG.getNode(N->getOpcode(), DL, LL.getValueType(), CCLo, LL, RL);
5768     Hi = DAG.getNode(N->getOpcode(), DL, LH.getValueType(), CCHi, LH, RH);
5769 
5770     // Add the new VSELECT nodes to the work list in case they need to be split
5771     // again.
5772     AddToWorklist(Lo.getNode());
5773     AddToWorklist(Hi.getNode());
5774 
5775     return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
5776   }
5777 
5778   // Fold (vselect (build_vector all_ones), N1, N2) -> N1
5779   if (ISD::isBuildVectorAllOnes(N0.getNode()))
5780     return N1;
5781   // Fold (vselect (build_vector all_zeros), N1, N2) -> N2
5782   if (ISD::isBuildVectorAllZeros(N0.getNode()))
5783     return N2;
5784 
5785   // The ConvertSelectToConcatVector function is assuming both the above
5786   // checks for (vselect (build_vector all{ones,zeros) ...) have been made
5787   // and addressed.
5788   if (N1.getOpcode() == ISD::CONCAT_VECTORS &&
5789       N2.getOpcode() == ISD::CONCAT_VECTORS &&
5790       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())) {
5791     if (SDValue CV = ConvertSelectToConcatVector(N, DAG))
5792       return CV;
5793   }
5794 
5795   return SDValue();
5796 }
5797 
5798 SDValue DAGCombiner::visitSELECT_CC(SDNode *N) {
5799   SDValue N0 = N->getOperand(0);
5800   SDValue N1 = N->getOperand(1);
5801   SDValue N2 = N->getOperand(2);
5802   SDValue N3 = N->getOperand(3);
5803   SDValue N4 = N->getOperand(4);
5804   ISD::CondCode CC = cast<CondCodeSDNode>(N4)->get();
5805 
5806   // fold select_cc lhs, rhs, x, x, cc -> x
5807   if (N2 == N3)
5808     return N2;
5809 
5810   // Determine if the condition we're dealing with is constant
5811   if (SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()), N0, N1,
5812                                   CC, SDLoc(N), false)) {
5813     AddToWorklist(SCC.getNode());
5814 
5815     if (ConstantSDNode *SCCC = dyn_cast<ConstantSDNode>(SCC.getNode())) {
5816       if (!SCCC->isNullValue())
5817         return N2;    // cond always true -> true val
5818       else
5819         return N3;    // cond always false -> false val
5820     } else if (SCC->isUndef()) {
5821       // When the condition is UNDEF, just return the first operand. This is
5822       // coherent the DAG creation, no setcc node is created in this case
5823       return N2;
5824     } else if (SCC.getOpcode() == ISD::SETCC) {
5825       // Fold to a simpler select_cc
5826       return DAG.getNode(ISD::SELECT_CC, SDLoc(N), N2.getValueType(),
5827                          SCC.getOperand(0), SCC.getOperand(1), N2, N3,
5828                          SCC.getOperand(2));
5829     }
5830   }
5831 
5832   // If we can fold this based on the true/false value, do so.
5833   if (SimplifySelectOps(N, N2, N3))
5834     return SDValue(N, 0);  // Don't revisit N.
5835 
5836   // fold select_cc into other things, such as min/max/abs
5837   return SimplifySelectCC(SDLoc(N), N0, N1, N2, N3, CC);
5838 }
5839 
5840 SDValue DAGCombiner::visitSETCC(SDNode *N) {
5841   return SimplifySetCC(N->getValueType(0), N->getOperand(0), N->getOperand(1),
5842                        cast<CondCodeSDNode>(N->getOperand(2))->get(),
5843                        SDLoc(N));
5844 }
5845 
5846 SDValue DAGCombiner::visitSETCCE(SDNode *N) {
5847   SDValue LHS = N->getOperand(0);
5848   SDValue RHS = N->getOperand(1);
5849   SDValue Carry = N->getOperand(2);
5850   SDValue Cond = N->getOperand(3);
5851 
5852   // If Carry is false, fold to a regular SETCC.
5853   if (Carry.getOpcode() == ISD::CARRY_FALSE)
5854     return DAG.getNode(ISD::SETCC, SDLoc(N), N->getVTList(), LHS, RHS, Cond);
5855 
5856   return SDValue();
5857 }
5858 
5859 /// Try to fold a sext/zext/aext dag node into a ConstantSDNode or
5860 /// a build_vector of constants.
5861 /// This function is called by the DAGCombiner when visiting sext/zext/aext
5862 /// dag nodes (see for example method DAGCombiner::visitSIGN_EXTEND).
5863 /// Vector extends are not folded if operations are legal; this is to
5864 /// avoid introducing illegal build_vector dag nodes.
5865 static SDNode *tryToFoldExtendOfConstant(SDNode *N, const TargetLowering &TLI,
5866                                          SelectionDAG &DAG, bool LegalTypes,
5867                                          bool LegalOperations) {
5868   unsigned Opcode = N->getOpcode();
5869   SDValue N0 = N->getOperand(0);
5870   EVT VT = N->getValueType(0);
5871 
5872   assert((Opcode == ISD::SIGN_EXTEND || Opcode == ISD::ZERO_EXTEND ||
5873          Opcode == ISD::ANY_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG ||
5874          Opcode == ISD::ZERO_EXTEND_VECTOR_INREG)
5875          && "Expected EXTEND dag node in input!");
5876 
5877   // fold (sext c1) -> c1
5878   // fold (zext c1) -> c1
5879   // fold (aext c1) -> c1
5880   if (isa<ConstantSDNode>(N0))
5881     return DAG.getNode(Opcode, SDLoc(N), VT, N0).getNode();
5882 
5883   // fold (sext (build_vector AllConstants) -> (build_vector AllConstants)
5884   // fold (zext (build_vector AllConstants) -> (build_vector AllConstants)
5885   // fold (aext (build_vector AllConstants) -> (build_vector AllConstants)
5886   EVT SVT = VT.getScalarType();
5887   if (!(VT.isVector() &&
5888       (!LegalTypes || (!LegalOperations && TLI.isTypeLegal(SVT))) &&
5889       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())))
5890     return nullptr;
5891 
5892   // We can fold this node into a build_vector.
5893   unsigned VTBits = SVT.getSizeInBits();
5894   unsigned EVTBits = N0->getValueType(0).getScalarSizeInBits();
5895   SmallVector<SDValue, 8> Elts;
5896   unsigned NumElts = VT.getVectorNumElements();
5897   SDLoc DL(N);
5898 
5899   for (unsigned i=0; i != NumElts; ++i) {
5900     SDValue Op = N0->getOperand(i);
5901     if (Op->isUndef()) {
5902       Elts.push_back(DAG.getUNDEF(SVT));
5903       continue;
5904     }
5905 
5906     SDLoc DL(Op);
5907     // Get the constant value and if needed trunc it to the size of the type.
5908     // Nodes like build_vector might have constants wider than the scalar type.
5909     APInt C = cast<ConstantSDNode>(Op)->getAPIntValue().zextOrTrunc(EVTBits);
5910     if (Opcode == ISD::SIGN_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG)
5911       Elts.push_back(DAG.getConstant(C.sext(VTBits), DL, SVT));
5912     else
5913       Elts.push_back(DAG.getConstant(C.zext(VTBits), DL, SVT));
5914   }
5915 
5916   return DAG.getBuildVector(VT, DL, Elts).getNode();
5917 }
5918 
5919 // ExtendUsesToFormExtLoad - Trying to extend uses of a load to enable this:
5920 // "fold ({s|z|a}ext (load x)) -> ({s|z|a}ext (truncate ({s|z|a}extload x)))"
5921 // transformation. Returns true if extension are possible and the above
5922 // mentioned transformation is profitable.
5923 static bool ExtendUsesToFormExtLoad(SDNode *N, SDValue N0,
5924                                     unsigned ExtOpc,
5925                                     SmallVectorImpl<SDNode *> &ExtendNodes,
5926                                     const TargetLowering &TLI) {
5927   bool HasCopyToRegUses = false;
5928   bool isTruncFree = TLI.isTruncateFree(N->getValueType(0), N0.getValueType());
5929   for (SDNode::use_iterator UI = N0.getNode()->use_begin(),
5930                             UE = N0.getNode()->use_end();
5931        UI != UE; ++UI) {
5932     SDNode *User = *UI;
5933     if (User == N)
5934       continue;
5935     if (UI.getUse().getResNo() != N0.getResNo())
5936       continue;
5937     // FIXME: Only extend SETCC N, N and SETCC N, c for now.
5938     if (ExtOpc != ISD::ANY_EXTEND && User->getOpcode() == ISD::SETCC) {
5939       ISD::CondCode CC = cast<CondCodeSDNode>(User->getOperand(2))->get();
5940       if (ExtOpc == ISD::ZERO_EXTEND && ISD::isSignedIntSetCC(CC))
5941         // Sign bits will be lost after a zext.
5942         return false;
5943       bool Add = false;
5944       for (unsigned i = 0; i != 2; ++i) {
5945         SDValue UseOp = User->getOperand(i);
5946         if (UseOp == N0)
5947           continue;
5948         if (!isa<ConstantSDNode>(UseOp))
5949           return false;
5950         Add = true;
5951       }
5952       if (Add)
5953         ExtendNodes.push_back(User);
5954       continue;
5955     }
5956     // If truncates aren't free and there are users we can't
5957     // extend, it isn't worthwhile.
5958     if (!isTruncFree)
5959       return false;
5960     // Remember if this value is live-out.
5961     if (User->getOpcode() == ISD::CopyToReg)
5962       HasCopyToRegUses = true;
5963   }
5964 
5965   if (HasCopyToRegUses) {
5966     bool BothLiveOut = false;
5967     for (SDNode::use_iterator UI = N->use_begin(), UE = N->use_end();
5968          UI != UE; ++UI) {
5969       SDUse &Use = UI.getUse();
5970       if (Use.getResNo() == 0 && Use.getUser()->getOpcode() == ISD::CopyToReg) {
5971         BothLiveOut = true;
5972         break;
5973       }
5974     }
5975     if (BothLiveOut)
5976       // Both unextended and extended values are live out. There had better be
5977       // a good reason for the transformation.
5978       return ExtendNodes.size();
5979   }
5980   return true;
5981 }
5982 
5983 void DAGCombiner::ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs,
5984                                   SDValue Trunc, SDValue ExtLoad,
5985                                   const SDLoc &DL, ISD::NodeType ExtType) {
5986   // Extend SetCC uses if necessary.
5987   for (unsigned i = 0, e = SetCCs.size(); i != e; ++i) {
5988     SDNode *SetCC = SetCCs[i];
5989     SmallVector<SDValue, 4> Ops;
5990 
5991     for (unsigned j = 0; j != 2; ++j) {
5992       SDValue SOp = SetCC->getOperand(j);
5993       if (SOp == Trunc)
5994         Ops.push_back(ExtLoad);
5995       else
5996         Ops.push_back(DAG.getNode(ExtType, DL, ExtLoad->getValueType(0), SOp));
5997     }
5998 
5999     Ops.push_back(SetCC->getOperand(2));
6000     CombineTo(SetCC, DAG.getNode(ISD::SETCC, DL, SetCC->getValueType(0), Ops));
6001   }
6002 }
6003 
6004 // FIXME: Bring more similar combines here, common to sext/zext (maybe aext?).
6005 SDValue DAGCombiner::CombineExtLoad(SDNode *N) {
6006   SDValue N0 = N->getOperand(0);
6007   EVT DstVT = N->getValueType(0);
6008   EVT SrcVT = N0.getValueType();
6009 
6010   assert((N->getOpcode() == ISD::SIGN_EXTEND ||
6011           N->getOpcode() == ISD::ZERO_EXTEND) &&
6012          "Unexpected node type (not an extend)!");
6013 
6014   // fold (sext (load x)) to multiple smaller sextloads; same for zext.
6015   // For example, on a target with legal v4i32, but illegal v8i32, turn:
6016   //   (v8i32 (sext (v8i16 (load x))))
6017   // into:
6018   //   (v8i32 (concat_vectors (v4i32 (sextload x)),
6019   //                          (v4i32 (sextload (x + 16)))))
6020   // Where uses of the original load, i.e.:
6021   //   (v8i16 (load x))
6022   // are replaced with:
6023   //   (v8i16 (truncate
6024   //     (v8i32 (concat_vectors (v4i32 (sextload x)),
6025   //                            (v4i32 (sextload (x + 16)))))))
6026   //
6027   // This combine is only applicable to illegal, but splittable, vectors.
6028   // All legal types, and illegal non-vector types, are handled elsewhere.
6029   // This combine is controlled by TargetLowering::isVectorLoadExtDesirable.
6030   //
6031   if (N0->getOpcode() != ISD::LOAD)
6032     return SDValue();
6033 
6034   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6035 
6036   if (!ISD::isNON_EXTLoad(LN0) || !ISD::isUNINDEXEDLoad(LN0) ||
6037       !N0.hasOneUse() || LN0->isVolatile() || !DstVT.isVector() ||
6038       !DstVT.isPow2VectorType() || !TLI.isVectorLoadExtDesirable(SDValue(N, 0)))
6039     return SDValue();
6040 
6041   SmallVector<SDNode *, 4> SetCCs;
6042   if (!ExtendUsesToFormExtLoad(N, N0, N->getOpcode(), SetCCs, TLI))
6043     return SDValue();
6044 
6045   ISD::LoadExtType ExtType =
6046       N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
6047 
6048   // Try to split the vector types to get down to legal types.
6049   EVT SplitSrcVT = SrcVT;
6050   EVT SplitDstVT = DstVT;
6051   while (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT) &&
6052          SplitSrcVT.getVectorNumElements() > 1) {
6053     SplitDstVT = DAG.GetSplitDestVTs(SplitDstVT).first;
6054     SplitSrcVT = DAG.GetSplitDestVTs(SplitSrcVT).first;
6055   }
6056 
6057   if (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT))
6058     return SDValue();
6059 
6060   SDLoc DL(N);
6061   const unsigned NumSplits =
6062       DstVT.getVectorNumElements() / SplitDstVT.getVectorNumElements();
6063   const unsigned Stride = SplitSrcVT.getStoreSize();
6064   SmallVector<SDValue, 4> Loads;
6065   SmallVector<SDValue, 4> Chains;
6066 
6067   SDValue BasePtr = LN0->getBasePtr();
6068   for (unsigned Idx = 0; Idx < NumSplits; Idx++) {
6069     const unsigned Offset = Idx * Stride;
6070     const unsigned Align = MinAlign(LN0->getAlignment(), Offset);
6071 
6072     SDValue SplitLoad = DAG.getExtLoad(
6073         ExtType, DL, SplitDstVT, LN0->getChain(), BasePtr,
6074         LN0->getPointerInfo().getWithOffset(Offset), SplitSrcVT, Align,
6075         LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
6076 
6077     BasePtr = DAG.getNode(ISD::ADD, DL, BasePtr.getValueType(), BasePtr,
6078                           DAG.getConstant(Stride, DL, BasePtr.getValueType()));
6079 
6080     Loads.push_back(SplitLoad.getValue(0));
6081     Chains.push_back(SplitLoad.getValue(1));
6082   }
6083 
6084   SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
6085   SDValue NewValue = DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Loads);
6086 
6087   CombineTo(N, NewValue);
6088 
6089   // Replace uses of the original load (before extension)
6090   // with a truncate of the concatenated sextloaded vectors.
6091   SDValue Trunc =
6092       DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), NewValue);
6093   CombineTo(N0.getNode(), Trunc, NewChain);
6094   ExtendSetCCUses(SetCCs, Trunc, NewValue, DL,
6095                   (ISD::NodeType)N->getOpcode());
6096   return SDValue(N, 0); // Return N so it doesn't get rechecked!
6097 }
6098 
6099 SDValue DAGCombiner::visitSIGN_EXTEND(SDNode *N) {
6100   SDValue N0 = N->getOperand(0);
6101   EVT VT = N->getValueType(0);
6102 
6103   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6104                                               LegalOperations))
6105     return SDValue(Res, 0);
6106 
6107   // fold (sext (sext x)) -> (sext x)
6108   // fold (sext (aext x)) -> (sext x)
6109   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
6110     return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT,
6111                        N0.getOperand(0));
6112 
6113   if (N0.getOpcode() == ISD::TRUNCATE) {
6114     // fold (sext (truncate (load x))) -> (sext (smaller load x))
6115     // fold (sext (truncate (srl (load x), c))) -> (sext (smaller load (x+c/n)))
6116     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6117       SDNode* oye = N0.getNode()->getOperand(0).getNode();
6118       if (NarrowLoad.getNode() != N0.getNode()) {
6119         CombineTo(N0.getNode(), NarrowLoad);
6120         // CombineTo deleted the truncate, if needed, but not what's under it.
6121         AddToWorklist(oye);
6122       }
6123       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6124     }
6125 
6126     // See if the value being truncated is already sign extended.  If so, just
6127     // eliminate the trunc/sext pair.
6128     SDValue Op = N0.getOperand(0);
6129     unsigned OpBits   = Op.getScalarValueSizeInBits();
6130     unsigned MidBits  = N0.getScalarValueSizeInBits();
6131     unsigned DestBits = VT.getScalarSizeInBits();
6132     unsigned NumSignBits = DAG.ComputeNumSignBits(Op);
6133 
6134     if (OpBits == DestBits) {
6135       // Op is i32, Mid is i8, and Dest is i32.  If Op has more than 24 sign
6136       // bits, it is already ready.
6137       if (NumSignBits > DestBits-MidBits)
6138         return Op;
6139     } else if (OpBits < DestBits) {
6140       // Op is i32, Mid is i8, and Dest is i64.  If Op has more than 24 sign
6141       // bits, just sext from i32.
6142       if (NumSignBits > OpBits-MidBits)
6143         return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, Op);
6144     } else {
6145       // Op is i64, Mid is i8, and Dest is i32.  If Op has more than 56 sign
6146       // bits, just truncate to i32.
6147       if (NumSignBits > OpBits-MidBits)
6148         return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6149     }
6150 
6151     // fold (sext (truncate x)) -> (sextinreg x).
6152     if (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG,
6153                                                  N0.getValueType())) {
6154       if (OpBits < DestBits)
6155         Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N0), VT, Op);
6156       else if (OpBits > DestBits)
6157         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), VT, Op);
6158       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, Op,
6159                          DAG.getValueType(N0.getValueType()));
6160     }
6161   }
6162 
6163   // fold (sext (load x)) -> (sext (truncate (sextload x)))
6164   // Only generate vector extloads when 1) they're legal, and 2) they are
6165   // deemed desirable by the target.
6166   if (ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6167       ((!LegalOperations && !VT.isVector() &&
6168         !cast<LoadSDNode>(N0)->isVolatile()) ||
6169        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, N0.getValueType()))) {
6170     bool DoXform = true;
6171     SmallVector<SDNode*, 4> SetCCs;
6172     if (!N0.hasOneUse())
6173       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::SIGN_EXTEND, SetCCs, TLI);
6174     if (VT.isVector())
6175       DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0));
6176     if (DoXform) {
6177       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6178       SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
6179                                        LN0->getChain(),
6180                                        LN0->getBasePtr(), N0.getValueType(),
6181                                        LN0->getMemOperand());
6182       CombineTo(N, ExtLoad);
6183       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6184                                   N0.getValueType(), ExtLoad);
6185       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6186       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
6187                       ISD::SIGN_EXTEND);
6188       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6189     }
6190   }
6191 
6192   // fold (sext (load x)) to multiple smaller sextloads.
6193   // Only on illegal but splittable vectors.
6194   if (SDValue ExtLoad = CombineExtLoad(N))
6195     return ExtLoad;
6196 
6197   // fold (sext (sextload x)) -> (sext (truncate (sextload x)))
6198   // fold (sext ( extload x)) -> (sext (truncate (sextload x)))
6199   if ((ISD::isSEXTLoad(N0.getNode()) || ISD::isEXTLoad(N0.getNode())) &&
6200       ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) {
6201     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6202     EVT MemVT = LN0->getMemoryVT();
6203     if ((!LegalOperations && !LN0->isVolatile()) ||
6204         TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, MemVT)) {
6205       SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
6206                                        LN0->getChain(),
6207                                        LN0->getBasePtr(), MemVT,
6208                                        LN0->getMemOperand());
6209       CombineTo(N, ExtLoad);
6210       CombineTo(N0.getNode(),
6211                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6212                             N0.getValueType(), ExtLoad),
6213                 ExtLoad.getValue(1));
6214       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6215     }
6216   }
6217 
6218   // fold (sext (and/or/xor (load x), cst)) ->
6219   //      (and/or/xor (sextload x), (sext cst))
6220   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
6221        N0.getOpcode() == ISD::XOR) &&
6222       isa<LoadSDNode>(N0.getOperand(0)) &&
6223       N0.getOperand(1).getOpcode() == ISD::Constant &&
6224       TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, N0.getValueType()) &&
6225       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
6226     LoadSDNode *LN0 = cast<LoadSDNode>(N0.getOperand(0));
6227     if (LN0->getExtensionType() != ISD::ZEXTLOAD && LN0->isUnindexed()) {
6228       bool DoXform = true;
6229       SmallVector<SDNode*, 4> SetCCs;
6230       if (!N0.hasOneUse())
6231         DoXform = ExtendUsesToFormExtLoad(N, N0.getOperand(0), ISD::SIGN_EXTEND,
6232                                           SetCCs, TLI);
6233       if (DoXform) {
6234         SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(LN0), VT,
6235                                          LN0->getChain(), LN0->getBasePtr(),
6236                                          LN0->getMemoryVT(),
6237                                          LN0->getMemOperand());
6238         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6239         Mask = Mask.sext(VT.getSizeInBits());
6240         SDLoc DL(N);
6241         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
6242                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
6243         SDValue Trunc = DAG.getNode(ISD::TRUNCATE,
6244                                     SDLoc(N0.getOperand(0)),
6245                                     N0.getOperand(0).getValueType(), ExtLoad);
6246         CombineTo(N, And);
6247         CombineTo(N0.getOperand(0).getNode(), Trunc, ExtLoad.getValue(1));
6248         ExtendSetCCUses(SetCCs, Trunc, ExtLoad, DL,
6249                         ISD::SIGN_EXTEND);
6250         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6251       }
6252     }
6253   }
6254 
6255   if (N0.getOpcode() == ISD::SETCC) {
6256     EVT N0VT = N0.getOperand(0).getValueType();
6257     // sext(setcc) -> sext_in_reg(vsetcc) for vectors.
6258     // Only do this before legalize for now.
6259     if (VT.isVector() && !LegalOperations &&
6260         TLI.getBooleanContents(N0VT) ==
6261             TargetLowering::ZeroOrNegativeOneBooleanContent) {
6262       // On some architectures (such as SSE/NEON/etc) the SETCC result type is
6263       // of the same size as the compared operands. Only optimize sext(setcc())
6264       // if this is the case.
6265       EVT SVT = getSetCCResultType(N0VT);
6266 
6267       // We know that the # elements of the results is the same as the
6268       // # elements of the compare (and the # elements of the compare result
6269       // for that matter).  Check to see that they are the same size.  If so,
6270       // we know that the element size of the sext'd result matches the
6271       // element size of the compare operands.
6272       if (VT.getSizeInBits() == SVT.getSizeInBits())
6273         return DAG.getSetCC(SDLoc(N), VT, N0.getOperand(0),
6274                              N0.getOperand(1),
6275                              cast<CondCodeSDNode>(N0.getOperand(2))->get());
6276 
6277       // If the desired elements are smaller or larger than the source
6278       // elements we can use a matching integer vector type and then
6279       // truncate/sign extend
6280       EVT MatchingVectorType = N0VT.changeVectorElementTypeToInteger();
6281       if (SVT == MatchingVectorType) {
6282         SDValue VsetCC = DAG.getSetCC(SDLoc(N), MatchingVectorType,
6283                                N0.getOperand(0), N0.getOperand(1),
6284                                cast<CondCodeSDNode>(N0.getOperand(2))->get());
6285         return DAG.getSExtOrTrunc(VsetCC, SDLoc(N), VT);
6286       }
6287     }
6288 
6289     // sext(setcc x, y, cc) -> (select (setcc x, y, cc), T, 0)
6290     // Here, T can be 1 or -1, depending on the type of the setcc and
6291     // getBooleanContents().
6292     unsigned SetCCWidth = N0.getScalarValueSizeInBits();
6293 
6294     SDLoc DL(N);
6295     // To determine the "true" side of the select, we need to know the high bit
6296     // of the value returned by the setcc if it evaluates to true.
6297     // If the type of the setcc is i1, then the true case of the select is just
6298     // sext(i1 1), that is, -1.
6299     // If the type of the setcc is larger (say, i8) then the value of the high
6300     // bit depends on getBooleanContents(). So, ask TLI for a real "true" value
6301     // of the appropriate width.
6302     SDValue ExtTrueVal =
6303         (SetCCWidth == 1)
6304             ? DAG.getConstant(APInt::getAllOnesValue(VT.getScalarSizeInBits()),
6305                               DL, VT)
6306             : TLI.getConstTrueVal(DAG, VT, DL);
6307 
6308     if (SDValue SCC = SimplifySelectCC(
6309             DL, N0.getOperand(0), N0.getOperand(1), ExtTrueVal,
6310             DAG.getConstant(0, DL, VT),
6311             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
6312       return SCC;
6313 
6314     if (!VT.isVector()) {
6315       EVT SetCCVT = getSetCCResultType(N0.getOperand(0).getValueType());
6316       if (!LegalOperations ||
6317           TLI.isOperationLegal(ISD::SETCC, N0.getOperand(0).getValueType())) {
6318         SDLoc DL(N);
6319         ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
6320         SDValue SetCC =
6321             DAG.getSetCC(DL, SetCCVT, N0.getOperand(0), N0.getOperand(1), CC);
6322         return DAG.getSelect(DL, VT, SetCC, ExtTrueVal,
6323                              DAG.getConstant(0, DL, VT));
6324       }
6325     }
6326   }
6327 
6328   // fold (sext x) -> (zext x) if the sign bit is known zero.
6329   if ((!LegalOperations || TLI.isOperationLegal(ISD::ZERO_EXTEND, VT)) &&
6330       DAG.SignBitIsZero(N0))
6331     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, N0);
6332 
6333   return SDValue();
6334 }
6335 
6336 // isTruncateOf - If N is a truncate of some other value, return true, record
6337 // the value being truncated in Op and which of Op's bits are zero in KnownZero.
6338 // This function computes KnownZero to avoid a duplicated call to
6339 // computeKnownBits in the caller.
6340 static bool isTruncateOf(SelectionDAG &DAG, SDValue N, SDValue &Op,
6341                          APInt &KnownZero) {
6342   APInt KnownOne;
6343   if (N->getOpcode() == ISD::TRUNCATE) {
6344     Op = N->getOperand(0);
6345     DAG.computeKnownBits(Op, KnownZero, KnownOne);
6346     return true;
6347   }
6348 
6349   if (N->getOpcode() != ISD::SETCC || N->getValueType(0) != MVT::i1 ||
6350       cast<CondCodeSDNode>(N->getOperand(2))->get() != ISD::SETNE)
6351     return false;
6352 
6353   SDValue Op0 = N->getOperand(0);
6354   SDValue Op1 = N->getOperand(1);
6355   assert(Op0.getValueType() == Op1.getValueType());
6356 
6357   if (isNullConstant(Op0))
6358     Op = Op1;
6359   else if (isNullConstant(Op1))
6360     Op = Op0;
6361   else
6362     return false;
6363 
6364   DAG.computeKnownBits(Op, KnownZero, KnownOne);
6365 
6366   if (!(KnownZero | APInt(Op.getValueSizeInBits(), 1)).isAllOnesValue())
6367     return false;
6368 
6369   return true;
6370 }
6371 
6372 SDValue DAGCombiner::visitZERO_EXTEND(SDNode *N) {
6373   SDValue N0 = N->getOperand(0);
6374   EVT VT = N->getValueType(0);
6375 
6376   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6377                                               LegalOperations))
6378     return SDValue(Res, 0);
6379 
6380   // fold (zext (zext x)) -> (zext x)
6381   // fold (zext (aext x)) -> (zext x)
6382   if (N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
6383     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT,
6384                        N0.getOperand(0));
6385 
6386   // fold (zext (truncate x)) -> (zext x) or
6387   //      (zext (truncate x)) -> (truncate x)
6388   // This is valid when the truncated bits of x are already zero.
6389   // FIXME: We should extend this to work for vectors too.
6390   SDValue Op;
6391   APInt KnownZero;
6392   if (!VT.isVector() && isTruncateOf(DAG, N0, Op, KnownZero)) {
6393     APInt TruncatedBits =
6394       (Op.getValueSizeInBits() == N0.getValueSizeInBits()) ?
6395       APInt(Op.getValueSizeInBits(), 0) :
6396       APInt::getBitsSet(Op.getValueSizeInBits(),
6397                         N0.getValueSizeInBits(),
6398                         std::min(Op.getValueSizeInBits(),
6399                                  VT.getSizeInBits()));
6400     if (TruncatedBits == (KnownZero & TruncatedBits)) {
6401       if (VT.bitsGT(Op.getValueType()))
6402         return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, Op);
6403       if (VT.bitsLT(Op.getValueType()))
6404         return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6405 
6406       return Op;
6407     }
6408   }
6409 
6410   // fold (zext (truncate (load x))) -> (zext (smaller load x))
6411   // fold (zext (truncate (srl (load x), c))) -> (zext (small load (x+c/n)))
6412   if (N0.getOpcode() == ISD::TRUNCATE) {
6413     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6414       SDNode* oye = N0.getNode()->getOperand(0).getNode();
6415       if (NarrowLoad.getNode() != N0.getNode()) {
6416         CombineTo(N0.getNode(), NarrowLoad);
6417         // CombineTo deleted the truncate, if needed, but not what's under it.
6418         AddToWorklist(oye);
6419       }
6420       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6421     }
6422   }
6423 
6424   // fold (zext (truncate x)) -> (and x, mask)
6425   if (N0.getOpcode() == ISD::TRUNCATE) {
6426     // fold (zext (truncate (load x))) -> (zext (smaller load x))
6427     // fold (zext (truncate (srl (load x), c))) -> (zext (smaller load (x+c/n)))
6428     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6429       SDNode *oye = N0.getNode()->getOperand(0).getNode();
6430       if (NarrowLoad.getNode() != N0.getNode()) {
6431         CombineTo(N0.getNode(), NarrowLoad);
6432         // CombineTo deleted the truncate, if needed, but not what's under it.
6433         AddToWorklist(oye);
6434       }
6435       return SDValue(N, 0); // Return N so it doesn't get rechecked!
6436     }
6437 
6438     EVT SrcVT = N0.getOperand(0).getValueType();
6439     EVT MinVT = N0.getValueType();
6440 
6441     // Try to mask before the extension to avoid having to generate a larger mask,
6442     // possibly over several sub-vectors.
6443     if (SrcVT.bitsLT(VT)) {
6444       if (!LegalOperations || (TLI.isOperationLegal(ISD::AND, SrcVT) &&
6445                                TLI.isOperationLegal(ISD::ZERO_EXTEND, VT))) {
6446         SDValue Op = N0.getOperand(0);
6447         Op = DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
6448         AddToWorklist(Op.getNode());
6449         return DAG.getZExtOrTrunc(Op, SDLoc(N), VT);
6450       }
6451     }
6452 
6453     if (!LegalOperations || TLI.isOperationLegal(ISD::AND, VT)) {
6454       SDValue Op = N0.getOperand(0);
6455       if (SrcVT.bitsLT(VT)) {
6456         Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, Op);
6457         AddToWorklist(Op.getNode());
6458       } else if (SrcVT.bitsGT(VT)) {
6459         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6460         AddToWorklist(Op.getNode());
6461       }
6462       return DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
6463     }
6464   }
6465 
6466   // Fold (zext (and (trunc x), cst)) -> (and x, cst),
6467   // if either of the casts is not free.
6468   if (N0.getOpcode() == ISD::AND &&
6469       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
6470       N0.getOperand(1).getOpcode() == ISD::Constant &&
6471       (!TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
6472                            N0.getValueType()) ||
6473        !TLI.isZExtFree(N0.getValueType(), VT))) {
6474     SDValue X = N0.getOperand(0).getOperand(0);
6475     if (X.getValueType().bitsLT(VT)) {
6476       X = DAG.getNode(ISD::ANY_EXTEND, SDLoc(X), VT, X);
6477     } else if (X.getValueType().bitsGT(VT)) {
6478       X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
6479     }
6480     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6481     Mask = Mask.zext(VT.getSizeInBits());
6482     SDLoc DL(N);
6483     return DAG.getNode(ISD::AND, DL, VT,
6484                        X, DAG.getConstant(Mask, DL, VT));
6485   }
6486 
6487   // fold (zext (load x)) -> (zext (truncate (zextload x)))
6488   // Only generate vector extloads when 1) they're legal, and 2) they are
6489   // deemed desirable by the target.
6490   if (ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6491       ((!LegalOperations && !VT.isVector() &&
6492         !cast<LoadSDNode>(N0)->isVolatile()) ||
6493        TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, N0.getValueType()))) {
6494     bool DoXform = true;
6495     SmallVector<SDNode*, 4> SetCCs;
6496     if (!N0.hasOneUse())
6497       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::ZERO_EXTEND, SetCCs, TLI);
6498     if (VT.isVector())
6499       DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0));
6500     if (DoXform) {
6501       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6502       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N), VT,
6503                                        LN0->getChain(),
6504                                        LN0->getBasePtr(), N0.getValueType(),
6505                                        LN0->getMemOperand());
6506       CombineTo(N, ExtLoad);
6507       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6508                                   N0.getValueType(), ExtLoad);
6509       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6510 
6511       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
6512                       ISD::ZERO_EXTEND);
6513       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6514     }
6515   }
6516 
6517   // fold (zext (load x)) to multiple smaller zextloads.
6518   // Only on illegal but splittable vectors.
6519   if (SDValue ExtLoad = CombineExtLoad(N))
6520     return ExtLoad;
6521 
6522   // fold (zext (and/or/xor (load x), cst)) ->
6523   //      (and/or/xor (zextload x), (zext cst))
6524   // Unless (and (load x) cst) will match as a zextload already and has
6525   // additional users.
6526   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
6527        N0.getOpcode() == ISD::XOR) &&
6528       isa<LoadSDNode>(N0.getOperand(0)) &&
6529       N0.getOperand(1).getOpcode() == ISD::Constant &&
6530       TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, N0.getValueType()) &&
6531       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
6532     LoadSDNode *LN0 = cast<LoadSDNode>(N0.getOperand(0));
6533     if (LN0->getExtensionType() != ISD::SEXTLOAD && LN0->isUnindexed()) {
6534       bool DoXform = true;
6535       SmallVector<SDNode*, 4> SetCCs;
6536       if (!N0.hasOneUse()) {
6537         if (N0.getOpcode() == ISD::AND) {
6538           auto *AndC = cast<ConstantSDNode>(N0.getOperand(1));
6539           auto NarrowLoad = false;
6540           EVT LoadResultTy = AndC->getValueType(0);
6541           EVT ExtVT, LoadedVT;
6542           if (isAndLoadExtLoad(AndC, LN0, LoadResultTy, ExtVT, LoadedVT,
6543                                NarrowLoad))
6544             DoXform = false;
6545         }
6546         if (DoXform)
6547           DoXform = ExtendUsesToFormExtLoad(N, N0.getOperand(0),
6548                                             ISD::ZERO_EXTEND, SetCCs, TLI);
6549       }
6550       if (DoXform) {
6551         SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN0), VT,
6552                                          LN0->getChain(), LN0->getBasePtr(),
6553                                          LN0->getMemoryVT(),
6554                                          LN0->getMemOperand());
6555         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6556         Mask = Mask.zext(VT.getSizeInBits());
6557         SDLoc DL(N);
6558         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
6559                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
6560         SDValue Trunc = DAG.getNode(ISD::TRUNCATE,
6561                                     SDLoc(N0.getOperand(0)),
6562                                     N0.getOperand(0).getValueType(), ExtLoad);
6563         CombineTo(N, And);
6564         CombineTo(N0.getOperand(0).getNode(), Trunc, ExtLoad.getValue(1));
6565         ExtendSetCCUses(SetCCs, Trunc, ExtLoad, DL,
6566                         ISD::ZERO_EXTEND);
6567         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6568       }
6569     }
6570   }
6571 
6572   // fold (zext (zextload x)) -> (zext (truncate (zextload x)))
6573   // fold (zext ( extload x)) -> (zext (truncate (zextload x)))
6574   if ((ISD::isZEXTLoad(N0.getNode()) || ISD::isEXTLoad(N0.getNode())) &&
6575       ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) {
6576     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6577     EVT MemVT = LN0->getMemoryVT();
6578     if ((!LegalOperations && !LN0->isVolatile()) ||
6579         TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT)) {
6580       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N), VT,
6581                                        LN0->getChain(),
6582                                        LN0->getBasePtr(), MemVT,
6583                                        LN0->getMemOperand());
6584       CombineTo(N, ExtLoad);
6585       CombineTo(N0.getNode(),
6586                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(),
6587                             ExtLoad),
6588                 ExtLoad.getValue(1));
6589       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6590     }
6591   }
6592 
6593   if (N0.getOpcode() == ISD::SETCC) {
6594     // Only do this before legalize for now.
6595     if (!LegalOperations && VT.isVector() &&
6596         N0.getValueType().getVectorElementType() == MVT::i1) {
6597       EVT N00VT = N0.getOperand(0).getValueType();
6598       if (getSetCCResultType(N00VT) == N0.getValueType())
6599         return SDValue();
6600 
6601       // We know that the # elements of the results is the same as the #
6602       // elements of the compare (and the # elements of the compare result for
6603       // that matter). Check to see that they are the same size. If so, we know
6604       // that the element size of the sext'd result matches the element size of
6605       // the compare operands.
6606       SDLoc DL(N);
6607       SDValue VecOnes = DAG.getConstant(1, DL, VT);
6608       if (VT.getSizeInBits() == N00VT.getSizeInBits()) {
6609         // zext(setcc) -> (and (vsetcc), (1, 1, ...) for vectors.
6610         SDValue VSetCC = DAG.getNode(ISD::SETCC, DL, VT, N0.getOperand(0),
6611                                      N0.getOperand(1), N0.getOperand(2));
6612         return DAG.getNode(ISD::AND, DL, VT, VSetCC, VecOnes);
6613       }
6614 
6615       // If the desired elements are smaller or larger than the source
6616       // elements we can use a matching integer vector type and then
6617       // truncate/sign extend.
6618       EVT MatchingElementType = EVT::getIntegerVT(
6619           *DAG.getContext(), N00VT.getScalarSizeInBits());
6620       EVT MatchingVectorType = EVT::getVectorVT(
6621           *DAG.getContext(), MatchingElementType, N00VT.getVectorNumElements());
6622       SDValue VsetCC =
6623           DAG.getNode(ISD::SETCC, DL, MatchingVectorType, N0.getOperand(0),
6624                       N0.getOperand(1), N0.getOperand(2));
6625       return DAG.getNode(ISD::AND, DL, VT, DAG.getSExtOrTrunc(VsetCC, DL, VT),
6626                          VecOnes);
6627     }
6628 
6629     // zext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
6630     SDLoc DL(N);
6631     if (SDValue SCC = SimplifySelectCC(
6632             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
6633             DAG.getConstant(0, DL, VT),
6634             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
6635       return SCC;
6636   }
6637 
6638   // (zext (shl (zext x), cst)) -> (shl (zext x), cst)
6639   if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL) &&
6640       isa<ConstantSDNode>(N0.getOperand(1)) &&
6641       N0.getOperand(0).getOpcode() == ISD::ZERO_EXTEND &&
6642       N0.hasOneUse()) {
6643     SDValue ShAmt = N0.getOperand(1);
6644     unsigned ShAmtVal = cast<ConstantSDNode>(ShAmt)->getZExtValue();
6645     if (N0.getOpcode() == ISD::SHL) {
6646       SDValue InnerZExt = N0.getOperand(0);
6647       // If the original shl may be shifting out bits, do not perform this
6648       // transformation.
6649       unsigned KnownZeroBits = InnerZExt.getValueSizeInBits() -
6650         InnerZExt.getOperand(0).getValueSizeInBits();
6651       if (ShAmtVal > KnownZeroBits)
6652         return SDValue();
6653     }
6654 
6655     SDLoc DL(N);
6656 
6657     // Ensure that the shift amount is wide enough for the shifted value.
6658     if (VT.getSizeInBits() >= 256)
6659       ShAmt = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, ShAmt);
6660 
6661     return DAG.getNode(N0.getOpcode(), DL, VT,
6662                        DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0)),
6663                        ShAmt);
6664   }
6665 
6666   return SDValue();
6667 }
6668 
6669 SDValue DAGCombiner::visitANY_EXTEND(SDNode *N) {
6670   SDValue N0 = N->getOperand(0);
6671   EVT VT = N->getValueType(0);
6672 
6673   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6674                                               LegalOperations))
6675     return SDValue(Res, 0);
6676 
6677   // fold (aext (aext x)) -> (aext x)
6678   // fold (aext (zext x)) -> (zext x)
6679   // fold (aext (sext x)) -> (sext x)
6680   if (N0.getOpcode() == ISD::ANY_EXTEND  ||
6681       N0.getOpcode() == ISD::ZERO_EXTEND ||
6682       N0.getOpcode() == ISD::SIGN_EXTEND)
6683     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
6684 
6685   // fold (aext (truncate (load x))) -> (aext (smaller load x))
6686   // fold (aext (truncate (srl (load x), c))) -> (aext (small load (x+c/n)))
6687   if (N0.getOpcode() == ISD::TRUNCATE) {
6688     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6689       SDNode* oye = N0.getNode()->getOperand(0).getNode();
6690       if (NarrowLoad.getNode() != N0.getNode()) {
6691         CombineTo(N0.getNode(), NarrowLoad);
6692         // CombineTo deleted the truncate, if needed, but not what's under it.
6693         AddToWorklist(oye);
6694       }
6695       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6696     }
6697   }
6698 
6699   // fold (aext (truncate x))
6700   if (N0.getOpcode() == ISD::TRUNCATE) {
6701     SDValue TruncOp = N0.getOperand(0);
6702     if (TruncOp.getValueType() == VT)
6703       return TruncOp; // x iff x size == zext size.
6704     if (TruncOp.getValueType().bitsGT(VT))
6705       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, TruncOp);
6706     return DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, TruncOp);
6707   }
6708 
6709   // Fold (aext (and (trunc x), cst)) -> (and x, cst)
6710   // if the trunc is not free.
6711   if (N0.getOpcode() == ISD::AND &&
6712       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
6713       N0.getOperand(1).getOpcode() == ISD::Constant &&
6714       !TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
6715                           N0.getValueType())) {
6716     SDValue X = N0.getOperand(0).getOperand(0);
6717     if (X.getValueType().bitsLT(VT)) {
6718       X = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, X);
6719     } else if (X.getValueType().bitsGT(VT)) {
6720       X = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, X);
6721     }
6722     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6723     Mask = Mask.zext(VT.getSizeInBits());
6724     SDLoc DL(N);
6725     return DAG.getNode(ISD::AND, DL, VT,
6726                        X, DAG.getConstant(Mask, DL, VT));
6727   }
6728 
6729   // fold (aext (load x)) -> (aext (truncate (extload x)))
6730   // None of the supported targets knows how to perform load and any_ext
6731   // on vectors in one instruction.  We only perform this transformation on
6732   // scalars.
6733   if (ISD::isNON_EXTLoad(N0.getNode()) && !VT.isVector() &&
6734       ISD::isUNINDEXEDLoad(N0.getNode()) &&
6735       TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
6736     bool DoXform = true;
6737     SmallVector<SDNode*, 4> SetCCs;
6738     if (!N0.hasOneUse())
6739       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::ANY_EXTEND, SetCCs, TLI);
6740     if (DoXform) {
6741       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6742       SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
6743                                        LN0->getChain(),
6744                                        LN0->getBasePtr(), N0.getValueType(),
6745                                        LN0->getMemOperand());
6746       CombineTo(N, ExtLoad);
6747       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6748                                   N0.getValueType(), ExtLoad);
6749       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6750       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
6751                       ISD::ANY_EXTEND);
6752       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6753     }
6754   }
6755 
6756   // fold (aext (zextload x)) -> (aext (truncate (zextload x)))
6757   // fold (aext (sextload x)) -> (aext (truncate (sextload x)))
6758   // fold (aext ( extload x)) -> (aext (truncate (extload  x)))
6759   if (N0.getOpcode() == ISD::LOAD &&
6760       !ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6761       N0.hasOneUse()) {
6762     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6763     ISD::LoadExtType ExtType = LN0->getExtensionType();
6764     EVT MemVT = LN0->getMemoryVT();
6765     if (!LegalOperations || TLI.isLoadExtLegal(ExtType, VT, MemVT)) {
6766       SDValue ExtLoad = DAG.getExtLoad(ExtType, SDLoc(N),
6767                                        VT, LN0->getChain(), LN0->getBasePtr(),
6768                                        MemVT, LN0->getMemOperand());
6769       CombineTo(N, ExtLoad);
6770       CombineTo(N0.getNode(),
6771                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6772                             N0.getValueType(), ExtLoad),
6773                 ExtLoad.getValue(1));
6774       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6775     }
6776   }
6777 
6778   if (N0.getOpcode() == ISD::SETCC) {
6779     // For vectors:
6780     // aext(setcc) -> vsetcc
6781     // aext(setcc) -> truncate(vsetcc)
6782     // aext(setcc) -> aext(vsetcc)
6783     // Only do this before legalize for now.
6784     if (VT.isVector() && !LegalOperations) {
6785       EVT N0VT = N0.getOperand(0).getValueType();
6786         // We know that the # elements of the results is the same as the
6787         // # elements of the compare (and the # elements of the compare result
6788         // for that matter).  Check to see that they are the same size.  If so,
6789         // we know that the element size of the sext'd result matches the
6790         // element size of the compare operands.
6791       if (VT.getSizeInBits() == N0VT.getSizeInBits())
6792         return DAG.getSetCC(SDLoc(N), VT, N0.getOperand(0),
6793                              N0.getOperand(1),
6794                              cast<CondCodeSDNode>(N0.getOperand(2))->get());
6795       // If the desired elements are smaller or larger than the source
6796       // elements we can use a matching integer vector type and then
6797       // truncate/any extend
6798       else {
6799         EVT MatchingVectorType = N0VT.changeVectorElementTypeToInteger();
6800         SDValue VsetCC =
6801           DAG.getSetCC(SDLoc(N), MatchingVectorType, N0.getOperand(0),
6802                         N0.getOperand(1),
6803                         cast<CondCodeSDNode>(N0.getOperand(2))->get());
6804         return DAG.getAnyExtOrTrunc(VsetCC, SDLoc(N), VT);
6805       }
6806     }
6807 
6808     // aext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
6809     SDLoc DL(N);
6810     if (SDValue SCC = SimplifySelectCC(
6811             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
6812             DAG.getConstant(0, DL, VT),
6813             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
6814       return SCC;
6815   }
6816 
6817   return SDValue();
6818 }
6819 
6820 /// See if the specified operand can be simplified with the knowledge that only
6821 /// the bits specified by Mask are used.  If so, return the simpler operand,
6822 /// otherwise return a null SDValue.
6823 SDValue DAGCombiner::GetDemandedBits(SDValue V, const APInt &Mask) {
6824   switch (V.getOpcode()) {
6825   default: break;
6826   case ISD::Constant: {
6827     const ConstantSDNode *CV = cast<ConstantSDNode>(V.getNode());
6828     assert(CV && "Const value should be ConstSDNode.");
6829     const APInt &CVal = CV->getAPIntValue();
6830     APInt NewVal = CVal & Mask;
6831     if (NewVal != CVal)
6832       return DAG.getConstant(NewVal, SDLoc(V), V.getValueType());
6833     break;
6834   }
6835   case ISD::OR:
6836   case ISD::XOR:
6837     // If the LHS or RHS don't contribute bits to the or, drop them.
6838     if (DAG.MaskedValueIsZero(V.getOperand(0), Mask))
6839       return V.getOperand(1);
6840     if (DAG.MaskedValueIsZero(V.getOperand(1), Mask))
6841       return V.getOperand(0);
6842     break;
6843   case ISD::SRL:
6844     // Only look at single-use SRLs.
6845     if (!V.getNode()->hasOneUse())
6846       break;
6847     if (ConstantSDNode *RHSC = getAsNonOpaqueConstant(V.getOperand(1))) {
6848       // See if we can recursively simplify the LHS.
6849       unsigned Amt = RHSC->getZExtValue();
6850 
6851       // Watch out for shift count overflow though.
6852       if (Amt >= Mask.getBitWidth()) break;
6853       APInt NewMask = Mask << Amt;
6854       if (SDValue SimplifyLHS = GetDemandedBits(V.getOperand(0), NewMask))
6855         return DAG.getNode(ISD::SRL, SDLoc(V), V.getValueType(),
6856                            SimplifyLHS, V.getOperand(1));
6857     }
6858   }
6859   return SDValue();
6860 }
6861 
6862 /// If the result of a wider load is shifted to right of N  bits and then
6863 /// truncated to a narrower type and where N is a multiple of number of bits of
6864 /// the narrower type, transform it to a narrower load from address + N / num of
6865 /// bits of new type. If the result is to be extended, also fold the extension
6866 /// to form a extending load.
6867 SDValue DAGCombiner::ReduceLoadWidth(SDNode *N) {
6868   unsigned Opc = N->getOpcode();
6869 
6870   ISD::LoadExtType ExtType = ISD::NON_EXTLOAD;
6871   SDValue N0 = N->getOperand(0);
6872   EVT VT = N->getValueType(0);
6873   EVT ExtVT = VT;
6874 
6875   // This transformation isn't valid for vector loads.
6876   if (VT.isVector())
6877     return SDValue();
6878 
6879   // Special case: SIGN_EXTEND_INREG is basically truncating to ExtVT then
6880   // extended to VT.
6881   if (Opc == ISD::SIGN_EXTEND_INREG) {
6882     ExtType = ISD::SEXTLOAD;
6883     ExtVT = cast<VTSDNode>(N->getOperand(1))->getVT();
6884   } else if (Opc == ISD::SRL) {
6885     // Another special-case: SRL is basically zero-extending a narrower value.
6886     ExtType = ISD::ZEXTLOAD;
6887     N0 = SDValue(N, 0);
6888     ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
6889     if (!N01) return SDValue();
6890     ExtVT = EVT::getIntegerVT(*DAG.getContext(),
6891                               VT.getSizeInBits() - N01->getZExtValue());
6892   }
6893   if (LegalOperations && !TLI.isLoadExtLegal(ExtType, VT, ExtVT))
6894     return SDValue();
6895 
6896   unsigned EVTBits = ExtVT.getSizeInBits();
6897 
6898   // Do not generate loads of non-round integer types since these can
6899   // be expensive (and would be wrong if the type is not byte sized).
6900   if (!ExtVT.isRound())
6901     return SDValue();
6902 
6903   unsigned ShAmt = 0;
6904   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
6905     if (ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
6906       ShAmt = N01->getZExtValue();
6907       // Is the shift amount a multiple of size of VT?
6908       if ((ShAmt & (EVTBits-1)) == 0) {
6909         N0 = N0.getOperand(0);
6910         // Is the load width a multiple of size of VT?
6911         if ((N0.getValueSizeInBits() & (EVTBits-1)) != 0)
6912           return SDValue();
6913       }
6914 
6915       // At this point, we must have a load or else we can't do the transform.
6916       if (!isa<LoadSDNode>(N0)) return SDValue();
6917 
6918       // Because a SRL must be assumed to *need* to zero-extend the high bits
6919       // (as opposed to anyext the high bits), we can't combine the zextload
6920       // lowering of SRL and an sextload.
6921       if (cast<LoadSDNode>(N0)->getExtensionType() == ISD::SEXTLOAD)
6922         return SDValue();
6923 
6924       // If the shift amount is larger than the input type then we're not
6925       // accessing any of the loaded bytes.  If the load was a zextload/extload
6926       // then the result of the shift+trunc is zero/undef (handled elsewhere).
6927       if (ShAmt >= cast<LoadSDNode>(N0)->getMemoryVT().getSizeInBits())
6928         return SDValue();
6929     }
6930   }
6931 
6932   // If the load is shifted left (and the result isn't shifted back right),
6933   // we can fold the truncate through the shift.
6934   unsigned ShLeftAmt = 0;
6935   if (ShAmt == 0 && N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
6936       ExtVT == VT && TLI.isNarrowingProfitable(N0.getValueType(), VT)) {
6937     if (ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
6938       ShLeftAmt = N01->getZExtValue();
6939       N0 = N0.getOperand(0);
6940     }
6941   }
6942 
6943   // If we haven't found a load, we can't narrow it.  Don't transform one with
6944   // multiple uses, this would require adding a new load.
6945   if (!isa<LoadSDNode>(N0) || !N0.hasOneUse())
6946     return SDValue();
6947 
6948   // Don't change the width of a volatile load.
6949   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6950   if (LN0->isVolatile())
6951     return SDValue();
6952 
6953   // Verify that we are actually reducing a load width here.
6954   if (LN0->getMemoryVT().getSizeInBits() < EVTBits)
6955     return SDValue();
6956 
6957   // For the transform to be legal, the load must produce only two values
6958   // (the value loaded and the chain).  Don't transform a pre-increment
6959   // load, for example, which produces an extra value.  Otherwise the
6960   // transformation is not equivalent, and the downstream logic to replace
6961   // uses gets things wrong.
6962   if (LN0->getNumValues() > 2)
6963     return SDValue();
6964 
6965   // If the load that we're shrinking is an extload and we're not just
6966   // discarding the extension we can't simply shrink the load. Bail.
6967   // TODO: It would be possible to merge the extensions in some cases.
6968   if (LN0->getExtensionType() != ISD::NON_EXTLOAD &&
6969       LN0->getMemoryVT().getSizeInBits() < ExtVT.getSizeInBits() + ShAmt)
6970     return SDValue();
6971 
6972   if (!TLI.shouldReduceLoadWidth(LN0, ExtType, ExtVT))
6973     return SDValue();
6974 
6975   EVT PtrType = N0.getOperand(1).getValueType();
6976 
6977   if (PtrType == MVT::Untyped || PtrType.isExtended())
6978     // It's not possible to generate a constant of extended or untyped type.
6979     return SDValue();
6980 
6981   // For big endian targets, we need to adjust the offset to the pointer to
6982   // load the correct bytes.
6983   if (DAG.getDataLayout().isBigEndian()) {
6984     unsigned LVTStoreBits = LN0->getMemoryVT().getStoreSizeInBits();
6985     unsigned EVTStoreBits = ExtVT.getStoreSizeInBits();
6986     ShAmt = LVTStoreBits - EVTStoreBits - ShAmt;
6987   }
6988 
6989   uint64_t PtrOff = ShAmt / 8;
6990   unsigned NewAlign = MinAlign(LN0->getAlignment(), PtrOff);
6991   SDLoc DL(LN0);
6992   // The original load itself didn't wrap, so an offset within it doesn't.
6993   SDNodeFlags Flags;
6994   Flags.setNoUnsignedWrap(true);
6995   SDValue NewPtr = DAG.getNode(ISD::ADD, DL,
6996                                PtrType, LN0->getBasePtr(),
6997                                DAG.getConstant(PtrOff, DL, PtrType),
6998                                &Flags);
6999   AddToWorklist(NewPtr.getNode());
7000 
7001   SDValue Load;
7002   if (ExtType == ISD::NON_EXTLOAD)
7003     Load = DAG.getLoad(VT, SDLoc(N0), LN0->getChain(), NewPtr,
7004                        LN0->getPointerInfo().getWithOffset(PtrOff), NewAlign,
7005                        LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
7006   else
7007     Load = DAG.getExtLoad(ExtType, SDLoc(N0), VT, LN0->getChain(), NewPtr,
7008                           LN0->getPointerInfo().getWithOffset(PtrOff), ExtVT,
7009                           NewAlign, LN0->getMemOperand()->getFlags(),
7010                           LN0->getAAInfo());
7011 
7012   // Replace the old load's chain with the new load's chain.
7013   WorklistRemover DeadNodes(*this);
7014   DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
7015 
7016   // Shift the result left, if we've swallowed a left shift.
7017   SDValue Result = Load;
7018   if (ShLeftAmt != 0) {
7019     EVT ShImmTy = getShiftAmountTy(Result.getValueType());
7020     if (!isUIntN(ShImmTy.getSizeInBits(), ShLeftAmt))
7021       ShImmTy = VT;
7022     // If the shift amount is as large as the result size (but, presumably,
7023     // no larger than the source) then the useful bits of the result are
7024     // zero; we can't simply return the shortened shift, because the result
7025     // of that operation is undefined.
7026     SDLoc DL(N0);
7027     if (ShLeftAmt >= VT.getSizeInBits())
7028       Result = DAG.getConstant(0, DL, VT);
7029     else
7030       Result = DAG.getNode(ISD::SHL, DL, VT,
7031                           Result, DAG.getConstant(ShLeftAmt, DL, ShImmTy));
7032   }
7033 
7034   // Return the new loaded value.
7035   return Result;
7036 }
7037 
7038 SDValue DAGCombiner::visitSIGN_EXTEND_INREG(SDNode *N) {
7039   SDValue N0 = N->getOperand(0);
7040   SDValue N1 = N->getOperand(1);
7041   EVT VT = N->getValueType(0);
7042   EVT EVT = cast<VTSDNode>(N1)->getVT();
7043   unsigned VTBits = VT.getScalarSizeInBits();
7044   unsigned EVTBits = EVT.getScalarSizeInBits();
7045 
7046   if (N0.isUndef())
7047     return DAG.getUNDEF(VT);
7048 
7049   // fold (sext_in_reg c1) -> c1
7050   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7051     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, N0, N1);
7052 
7053   // If the input is already sign extended, just drop the extension.
7054   if (DAG.ComputeNumSignBits(N0) >= VTBits-EVTBits+1)
7055     return N0;
7056 
7057   // fold (sext_in_reg (sext_in_reg x, VT2), VT1) -> (sext_in_reg x, minVT) pt2
7058   if (N0.getOpcode() == ISD::SIGN_EXTEND_INREG &&
7059       EVT.bitsLT(cast<VTSDNode>(N0.getOperand(1))->getVT()))
7060     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
7061                        N0.getOperand(0), N1);
7062 
7063   // fold (sext_in_reg (sext x)) -> (sext x)
7064   // fold (sext_in_reg (aext x)) -> (sext x)
7065   // if x is small enough.
7066   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) {
7067     SDValue N00 = N0.getOperand(0);
7068     if (N00.getScalarValueSizeInBits() <= EVTBits &&
7069         (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT)))
7070       return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00, N1);
7071   }
7072 
7073   // fold (sext_in_reg x) -> (zext_in_reg x) if the sign bit is known zero.
7074   if (DAG.MaskedValueIsZero(N0, APInt::getBitsSet(VTBits, EVTBits-1, EVTBits)))
7075     return DAG.getZeroExtendInReg(N0, SDLoc(N), EVT.getScalarType());
7076 
7077   // fold operands of sext_in_reg based on knowledge that the top bits are not
7078   // demanded.
7079   if (SimplifyDemandedBits(SDValue(N, 0)))
7080     return SDValue(N, 0);
7081 
7082   // fold (sext_in_reg (load x)) -> (smaller sextload x)
7083   // fold (sext_in_reg (srl (load x), c)) -> (smaller sextload (x+c/evtbits))
7084   if (SDValue NarrowLoad = ReduceLoadWidth(N))
7085     return NarrowLoad;
7086 
7087   // fold (sext_in_reg (srl X, 24), i8) -> (sra X, 24)
7088   // fold (sext_in_reg (srl X, 23), i8) -> (sra X, 23) iff possible.
7089   // We already fold "(sext_in_reg (srl X, 25), i8) -> srl X, 25" above.
7090   if (N0.getOpcode() == ISD::SRL) {
7091     if (ConstantSDNode *ShAmt = dyn_cast<ConstantSDNode>(N0.getOperand(1)))
7092       if (ShAmt->getZExtValue()+EVTBits <= VTBits) {
7093         // We can turn this into an SRA iff the input to the SRL is already sign
7094         // extended enough.
7095         unsigned InSignBits = DAG.ComputeNumSignBits(N0.getOperand(0));
7096         if (VTBits-(ShAmt->getZExtValue()+EVTBits) < InSignBits)
7097           return DAG.getNode(ISD::SRA, SDLoc(N), VT,
7098                              N0.getOperand(0), N0.getOperand(1));
7099       }
7100   }
7101 
7102   // fold (sext_inreg (extload x)) -> (sextload x)
7103   if (ISD::isEXTLoad(N0.getNode()) &&
7104       ISD::isUNINDEXEDLoad(N0.getNode()) &&
7105       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
7106       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
7107        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
7108     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7109     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
7110                                      LN0->getChain(),
7111                                      LN0->getBasePtr(), EVT,
7112                                      LN0->getMemOperand());
7113     CombineTo(N, ExtLoad);
7114     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
7115     AddToWorklist(ExtLoad.getNode());
7116     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7117   }
7118   // fold (sext_inreg (zextload x)) -> (sextload x) iff load has one use
7119   if (ISD::isZEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
7120       N0.hasOneUse() &&
7121       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
7122       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
7123        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
7124     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7125     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
7126                                      LN0->getChain(),
7127                                      LN0->getBasePtr(), EVT,
7128                                      LN0->getMemOperand());
7129     CombineTo(N, ExtLoad);
7130     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
7131     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7132   }
7133 
7134   // Form (sext_inreg (bswap >> 16)) or (sext_inreg (rotl (bswap) 16))
7135   if (EVTBits <= 16 && N0.getOpcode() == ISD::OR) {
7136     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
7137                                            N0.getOperand(1), false))
7138       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
7139                          BSwap, N1);
7140   }
7141 
7142   return SDValue();
7143 }
7144 
7145 SDValue DAGCombiner::visitSIGN_EXTEND_VECTOR_INREG(SDNode *N) {
7146   SDValue N0 = N->getOperand(0);
7147   EVT VT = N->getValueType(0);
7148 
7149   if (N0.isUndef())
7150     return DAG.getUNDEF(VT);
7151 
7152   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
7153                                               LegalOperations))
7154     return SDValue(Res, 0);
7155 
7156   return SDValue();
7157 }
7158 
7159 SDValue DAGCombiner::visitZERO_EXTEND_VECTOR_INREG(SDNode *N) {
7160   SDValue N0 = N->getOperand(0);
7161   EVT VT = N->getValueType(0);
7162 
7163   if (N0.isUndef())
7164     return DAG.getUNDEF(VT);
7165 
7166   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
7167                                               LegalOperations))
7168     return SDValue(Res, 0);
7169 
7170   return SDValue();
7171 }
7172 
7173 SDValue DAGCombiner::visitTRUNCATE(SDNode *N) {
7174   SDValue N0 = N->getOperand(0);
7175   EVT VT = N->getValueType(0);
7176   bool isLE = DAG.getDataLayout().isLittleEndian();
7177 
7178   // noop truncate
7179   if (N0.getValueType() == N->getValueType(0))
7180     return N0;
7181   // fold (truncate c1) -> c1
7182   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7183     return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0);
7184   // fold (truncate (truncate x)) -> (truncate x)
7185   if (N0.getOpcode() == ISD::TRUNCATE)
7186     return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
7187   // fold (truncate (ext x)) -> (ext x) or (truncate x) or x
7188   if (N0.getOpcode() == ISD::ZERO_EXTEND ||
7189       N0.getOpcode() == ISD::SIGN_EXTEND ||
7190       N0.getOpcode() == ISD::ANY_EXTEND) {
7191     // if the source is smaller than the dest, we still need an extend.
7192     if (N0.getOperand(0).getValueType().bitsLT(VT))
7193       return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
7194     // if the source is larger than the dest, than we just need the truncate.
7195     if (N0.getOperand(0).getValueType().bitsGT(VT))
7196       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
7197     // if the source and dest are the same type, we can drop both the extend
7198     // and the truncate.
7199     return N0.getOperand(0);
7200   }
7201 
7202   // If this is anyext(trunc), don't fold it, allow ourselves to be folded.
7203   if (N->hasOneUse() && (N->use_begin()->getOpcode() == ISD::ANY_EXTEND))
7204     return SDValue();
7205 
7206   // Fold extract-and-trunc into a narrow extract. For example:
7207   //   i64 x = EXTRACT_VECTOR_ELT(v2i64 val, i32 1)
7208   //   i32 y = TRUNCATE(i64 x)
7209   //        -- becomes --
7210   //   v16i8 b = BITCAST (v2i64 val)
7211   //   i8 x = EXTRACT_VECTOR_ELT(v16i8 b, i32 8)
7212   //
7213   // Note: We only run this optimization after type legalization (which often
7214   // creates this pattern) and before operation legalization after which
7215   // we need to be more careful about the vector instructions that we generate.
7216   if (N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
7217       LegalTypes && !LegalOperations && N0->hasOneUse() && VT != MVT::i1) {
7218 
7219     EVT VecTy = N0.getOperand(0).getValueType();
7220     EVT ExTy = N0.getValueType();
7221     EVT TrTy = N->getValueType(0);
7222 
7223     unsigned NumElem = VecTy.getVectorNumElements();
7224     unsigned SizeRatio = ExTy.getSizeInBits()/TrTy.getSizeInBits();
7225 
7226     EVT NVT = EVT::getVectorVT(*DAG.getContext(), TrTy, SizeRatio * NumElem);
7227     assert(NVT.getSizeInBits() == VecTy.getSizeInBits() && "Invalid Size");
7228 
7229     SDValue EltNo = N0->getOperand(1);
7230     if (isa<ConstantSDNode>(EltNo) && isTypeLegal(NVT)) {
7231       int Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
7232       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
7233       int Index = isLE ? (Elt*SizeRatio) : (Elt*SizeRatio + (SizeRatio-1));
7234 
7235       SDLoc DL(N);
7236       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, TrTy,
7237                          DAG.getBitcast(NVT, N0.getOperand(0)),
7238                          DAG.getConstant(Index, DL, IndexTy));
7239     }
7240   }
7241 
7242   // trunc (select c, a, b) -> select c, (trunc a), (trunc b)
7243   if (N0.getOpcode() == ISD::SELECT) {
7244     EVT SrcVT = N0.getValueType();
7245     if ((!LegalOperations || TLI.isOperationLegal(ISD::SELECT, SrcVT)) &&
7246         TLI.isTruncateFree(SrcVT, VT)) {
7247       SDLoc SL(N0);
7248       SDValue Cond = N0.getOperand(0);
7249       SDValue TruncOp0 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1));
7250       SDValue TruncOp1 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(2));
7251       return DAG.getNode(ISD::SELECT, SDLoc(N), VT, Cond, TruncOp0, TruncOp1);
7252     }
7253   }
7254 
7255   // trunc (shl x, K) -> shl (trunc x), K => K < VT.getScalarSizeInBits()
7256   if (N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
7257       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::SHL, VT)) &&
7258       TLI.isTypeDesirableForOp(ISD::SHL, VT)) {
7259     if (const ConstantSDNode *CAmt = isConstOrConstSplat(N0.getOperand(1))) {
7260       uint64_t Amt = CAmt->getZExtValue();
7261       unsigned Size = VT.getScalarSizeInBits();
7262 
7263       if (Amt < Size) {
7264         SDLoc SL(N);
7265         EVT AmtVT = TLI.getShiftAmountTy(VT, DAG.getDataLayout());
7266 
7267         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0));
7268         return DAG.getNode(ISD::SHL, SL, VT, Trunc,
7269                            DAG.getConstant(Amt, SL, AmtVT));
7270       }
7271     }
7272   }
7273 
7274   // Fold a series of buildvector, bitcast, and truncate if possible.
7275   // For example fold
7276   //   (2xi32 trunc (bitcast ((4xi32)buildvector x, x, y, y) 2xi64)) to
7277   //   (2xi32 (buildvector x, y)).
7278   if (Level == AfterLegalizeVectorOps && VT.isVector() &&
7279       N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
7280       N0.getOperand(0).getOpcode() == ISD::BUILD_VECTOR &&
7281       N0.getOperand(0).hasOneUse()) {
7282 
7283     SDValue BuildVect = N0.getOperand(0);
7284     EVT BuildVectEltTy = BuildVect.getValueType().getVectorElementType();
7285     EVT TruncVecEltTy = VT.getVectorElementType();
7286 
7287     // Check that the element types match.
7288     if (BuildVectEltTy == TruncVecEltTy) {
7289       // Now we only need to compute the offset of the truncated elements.
7290       unsigned BuildVecNumElts =  BuildVect.getNumOperands();
7291       unsigned TruncVecNumElts = VT.getVectorNumElements();
7292       unsigned TruncEltOffset = BuildVecNumElts / TruncVecNumElts;
7293 
7294       assert((BuildVecNumElts % TruncVecNumElts) == 0 &&
7295              "Invalid number of elements");
7296 
7297       SmallVector<SDValue, 8> Opnds;
7298       for (unsigned i = 0, e = BuildVecNumElts; i != e; i += TruncEltOffset)
7299         Opnds.push_back(BuildVect.getOperand(i));
7300 
7301       return DAG.getBuildVector(VT, SDLoc(N), Opnds);
7302     }
7303   }
7304 
7305   // See if we can simplify the input to this truncate through knowledge that
7306   // only the low bits are being used.
7307   // For example "trunc (or (shl x, 8), y)" // -> trunc y
7308   // Currently we only perform this optimization on scalars because vectors
7309   // may have different active low bits.
7310   if (!VT.isVector()) {
7311     if (SDValue Shorter =
7312             GetDemandedBits(N0, APInt::getLowBitsSet(N0.getValueSizeInBits(),
7313                                                      VT.getSizeInBits())))
7314       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Shorter);
7315   }
7316   // fold (truncate (load x)) -> (smaller load x)
7317   // fold (truncate (srl (load x), c)) -> (smaller load (x+c/evtbits))
7318   if (!LegalTypes || TLI.isTypeDesirableForOp(N0.getOpcode(), VT)) {
7319     if (SDValue Reduced = ReduceLoadWidth(N))
7320       return Reduced;
7321 
7322     // Handle the case where the load remains an extending load even
7323     // after truncation.
7324     if (N0.hasOneUse() && ISD::isUNINDEXEDLoad(N0.getNode())) {
7325       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7326       if (!LN0->isVolatile() &&
7327           LN0->getMemoryVT().getStoreSizeInBits() < VT.getSizeInBits()) {
7328         SDValue NewLoad = DAG.getExtLoad(LN0->getExtensionType(), SDLoc(LN0),
7329                                          VT, LN0->getChain(), LN0->getBasePtr(),
7330                                          LN0->getMemoryVT(),
7331                                          LN0->getMemOperand());
7332         DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLoad.getValue(1));
7333         return NewLoad;
7334       }
7335     }
7336   }
7337   // fold (trunc (concat ... x ...)) -> (concat ..., (trunc x), ...)),
7338   // where ... are all 'undef'.
7339   if (N0.getOpcode() == ISD::CONCAT_VECTORS && !LegalTypes) {
7340     SmallVector<EVT, 8> VTs;
7341     SDValue V;
7342     unsigned Idx = 0;
7343     unsigned NumDefs = 0;
7344 
7345     for (unsigned i = 0, e = N0.getNumOperands(); i != e; ++i) {
7346       SDValue X = N0.getOperand(i);
7347       if (!X.isUndef()) {
7348         V = X;
7349         Idx = i;
7350         NumDefs++;
7351       }
7352       // Stop if more than one members are non-undef.
7353       if (NumDefs > 1)
7354         break;
7355       VTs.push_back(EVT::getVectorVT(*DAG.getContext(),
7356                                      VT.getVectorElementType(),
7357                                      X.getValueType().getVectorNumElements()));
7358     }
7359 
7360     if (NumDefs == 0)
7361       return DAG.getUNDEF(VT);
7362 
7363     if (NumDefs == 1) {
7364       assert(V.getNode() && "The single defined operand is empty!");
7365       SmallVector<SDValue, 8> Opnds;
7366       for (unsigned i = 0, e = VTs.size(); i != e; ++i) {
7367         if (i != Idx) {
7368           Opnds.push_back(DAG.getUNDEF(VTs[i]));
7369           continue;
7370         }
7371         SDValue NV = DAG.getNode(ISD::TRUNCATE, SDLoc(V), VTs[i], V);
7372         AddToWorklist(NV.getNode());
7373         Opnds.push_back(NV);
7374       }
7375       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Opnds);
7376     }
7377   }
7378 
7379   // Fold truncate of a bitcast of a vector to an extract of the low vector
7380   // element.
7381   //
7382   // e.g. trunc (i64 (bitcast v2i32:x)) -> extract_vector_elt v2i32:x, 0
7383   if (N0.getOpcode() == ISD::BITCAST && !VT.isVector()) {
7384     SDValue VecSrc = N0.getOperand(0);
7385     EVT SrcVT = VecSrc.getValueType();
7386     if (SrcVT.isVector() && SrcVT.getScalarType() == VT &&
7387         (!LegalOperations ||
7388          TLI.isOperationLegal(ISD::EXTRACT_VECTOR_ELT, SrcVT))) {
7389       SDLoc SL(N);
7390 
7391       EVT IdxVT = TLI.getVectorIdxTy(DAG.getDataLayout());
7392       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, VT,
7393                          VecSrc, DAG.getConstant(0, SL, IdxVT));
7394     }
7395   }
7396 
7397   // Simplify the operands using demanded-bits information.
7398   if (!VT.isVector() &&
7399       SimplifyDemandedBits(SDValue(N, 0)))
7400     return SDValue(N, 0);
7401 
7402   return SDValue();
7403 }
7404 
7405 static SDNode *getBuildPairElt(SDNode *N, unsigned i) {
7406   SDValue Elt = N->getOperand(i);
7407   if (Elt.getOpcode() != ISD::MERGE_VALUES)
7408     return Elt.getNode();
7409   return Elt.getOperand(Elt.getResNo()).getNode();
7410 }
7411 
7412 /// build_pair (load, load) -> load
7413 /// if load locations are consecutive.
7414 SDValue DAGCombiner::CombineConsecutiveLoads(SDNode *N, EVT VT) {
7415   assert(N->getOpcode() == ISD::BUILD_PAIR);
7416 
7417   LoadSDNode *LD1 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 0));
7418   LoadSDNode *LD2 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 1));
7419   if (!LD1 || !LD2 || !ISD::isNON_EXTLoad(LD1) || !LD1->hasOneUse() ||
7420       LD1->getAddressSpace() != LD2->getAddressSpace())
7421     return SDValue();
7422   EVT LD1VT = LD1->getValueType(0);
7423   unsigned LD1Bytes = LD1VT.getSizeInBits() / 8;
7424   if (ISD::isNON_EXTLoad(LD2) && LD2->hasOneUse() &&
7425       DAG.areNonVolatileConsecutiveLoads(LD2, LD1, LD1Bytes, 1)) {
7426     unsigned Align = LD1->getAlignment();
7427     unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
7428         VT.getTypeForEVT(*DAG.getContext()));
7429 
7430     if (NewAlign <= Align &&
7431         (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT)))
7432       return DAG.getLoad(VT, SDLoc(N), LD1->getChain(), LD1->getBasePtr(),
7433                          LD1->getPointerInfo(), Align);
7434   }
7435 
7436   return SDValue();
7437 }
7438 
7439 static unsigned getPPCf128HiElementSelector(const SelectionDAG &DAG) {
7440   // On little-endian machines, bitcasting from ppcf128 to i128 does swap the Hi
7441   // and Lo parts; on big-endian machines it doesn't.
7442   return DAG.getDataLayout().isBigEndian() ? 1 : 0;
7443 }
7444 
7445 static SDValue foldBitcastedFPLogic(SDNode *N, SelectionDAG &DAG,
7446                                     const TargetLowering &TLI) {
7447   // If this is not a bitcast to an FP type or if the target doesn't have
7448   // IEEE754-compliant FP logic, we're done.
7449   EVT VT = N->getValueType(0);
7450   if (!VT.isFloatingPoint() || !TLI.hasBitPreservingFPLogic(VT))
7451     return SDValue();
7452 
7453   // TODO: Use splat values for the constant-checking below and remove this
7454   // restriction.
7455   SDValue N0 = N->getOperand(0);
7456   EVT SourceVT = N0.getValueType();
7457   if (SourceVT.isVector())
7458     return SDValue();
7459 
7460   unsigned FPOpcode;
7461   APInt SignMask;
7462   switch (N0.getOpcode()) {
7463   case ISD::AND:
7464     FPOpcode = ISD::FABS;
7465     SignMask = ~APInt::getSignBit(SourceVT.getSizeInBits());
7466     break;
7467   case ISD::XOR:
7468     FPOpcode = ISD::FNEG;
7469     SignMask = APInt::getSignBit(SourceVT.getSizeInBits());
7470     break;
7471   // TODO: ISD::OR --> ISD::FNABS?
7472   default:
7473     return SDValue();
7474   }
7475 
7476   // Fold (bitcast int (and (bitcast fp X to int), 0x7fff...) to fp) -> fabs X
7477   // Fold (bitcast int (xor (bitcast fp X to int), 0x8000...) to fp) -> fneg X
7478   SDValue LogicOp0 = N0.getOperand(0);
7479   ConstantSDNode *LogicOp1 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7480   if (LogicOp1 && LogicOp1->getAPIntValue() == SignMask &&
7481       LogicOp0.getOpcode() == ISD::BITCAST &&
7482       LogicOp0->getOperand(0).getValueType() == VT)
7483     return DAG.getNode(FPOpcode, SDLoc(N), VT, LogicOp0->getOperand(0));
7484 
7485   return SDValue();
7486 }
7487 
7488 SDValue DAGCombiner::visitBITCAST(SDNode *N) {
7489   SDValue N0 = N->getOperand(0);
7490   EVT VT = N->getValueType(0);
7491 
7492   // If the input is a BUILD_VECTOR with all constant elements, fold this now.
7493   // Only do this before legalize, since afterward the target may be depending
7494   // on the bitconvert.
7495   // First check to see if this is all constant.
7496   if (!LegalTypes &&
7497       N0.getOpcode() == ISD::BUILD_VECTOR && N0.getNode()->hasOneUse() &&
7498       VT.isVector()) {
7499     bool isSimple = cast<BuildVectorSDNode>(N0)->isConstant();
7500 
7501     EVT DestEltVT = N->getValueType(0).getVectorElementType();
7502     assert(!DestEltVT.isVector() &&
7503            "Element type of vector ValueType must not be vector!");
7504     if (isSimple)
7505       return ConstantFoldBITCASTofBUILD_VECTOR(N0.getNode(), DestEltVT);
7506   }
7507 
7508   // If the input is a constant, let getNode fold it.
7509   if (isa<ConstantSDNode>(N0) || isa<ConstantFPSDNode>(N0)) {
7510     // If we can't allow illegal operations, we need to check that this is just
7511     // a fp -> int or int -> conversion and that the resulting operation will
7512     // be legal.
7513     if (!LegalOperations ||
7514         (isa<ConstantSDNode>(N0) && VT.isFloatingPoint() && !VT.isVector() &&
7515          TLI.isOperationLegal(ISD::ConstantFP, VT)) ||
7516         (isa<ConstantFPSDNode>(N0) && VT.isInteger() && !VT.isVector() &&
7517          TLI.isOperationLegal(ISD::Constant, VT)))
7518       return DAG.getBitcast(VT, N0);
7519   }
7520 
7521   // (conv (conv x, t1), t2) -> (conv x, t2)
7522   if (N0.getOpcode() == ISD::BITCAST)
7523     return DAG.getBitcast(VT, N0.getOperand(0));
7524 
7525   // fold (conv (load x)) -> (load (conv*)x)
7526   // If the resultant load doesn't need a higher alignment than the original!
7527   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
7528       // Do not change the width of a volatile load.
7529       !cast<LoadSDNode>(N0)->isVolatile() &&
7530       // Do not remove the cast if the types differ in endian layout.
7531       TLI.hasBigEndianPartOrdering(N0.getValueType(), DAG.getDataLayout()) ==
7532           TLI.hasBigEndianPartOrdering(VT, DAG.getDataLayout()) &&
7533       (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT)) &&
7534       TLI.isLoadBitCastBeneficial(N0.getValueType(), VT)) {
7535     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7536     unsigned OrigAlign = LN0->getAlignment();
7537 
7538     bool Fast = false;
7539     if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT,
7540                                LN0->getAddressSpace(), OrigAlign, &Fast) &&
7541         Fast) {
7542       SDValue Load =
7543           DAG.getLoad(VT, SDLoc(N), LN0->getChain(), LN0->getBasePtr(),
7544                       LN0->getPointerInfo(), OrigAlign,
7545                       LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
7546       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
7547       return Load;
7548     }
7549   }
7550 
7551   if (SDValue V = foldBitcastedFPLogic(N, DAG, TLI))
7552     return V;
7553 
7554   // fold (bitconvert (fneg x)) -> (xor (bitconvert x), signbit)
7555   // fold (bitconvert (fabs x)) -> (and (bitconvert x), (not signbit))
7556   //
7557   // For ppc_fp128:
7558   // fold (bitcast (fneg x)) ->
7559   //     flipbit = signbit
7560   //     (xor (bitcast x) (build_pair flipbit, flipbit))
7561   //
7562   // fold (bitcast (fabs x)) ->
7563   //     flipbit = (and (extract_element (bitcast x), 0), signbit)
7564   //     (xor (bitcast x) (build_pair flipbit, flipbit))
7565   // This often reduces constant pool loads.
7566   if (((N0.getOpcode() == ISD::FNEG && !TLI.isFNegFree(N0.getValueType())) ||
7567        (N0.getOpcode() == ISD::FABS && !TLI.isFAbsFree(N0.getValueType()))) &&
7568       N0.getNode()->hasOneUse() && VT.isInteger() &&
7569       !VT.isVector() && !N0.getValueType().isVector()) {
7570     SDValue NewConv = DAG.getBitcast(VT, N0.getOperand(0));
7571     AddToWorklist(NewConv.getNode());
7572 
7573     SDLoc DL(N);
7574     if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
7575       assert(VT.getSizeInBits() == 128);
7576       SDValue SignBit = DAG.getConstant(
7577           APInt::getSignBit(VT.getSizeInBits() / 2), SDLoc(N0), MVT::i64);
7578       SDValue FlipBit;
7579       if (N0.getOpcode() == ISD::FNEG) {
7580         FlipBit = SignBit;
7581         AddToWorklist(FlipBit.getNode());
7582       } else {
7583         assert(N0.getOpcode() == ISD::FABS);
7584         SDValue Hi =
7585             DAG.getNode(ISD::EXTRACT_ELEMENT, SDLoc(NewConv), MVT::i64, NewConv,
7586                         DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
7587                                               SDLoc(NewConv)));
7588         AddToWorklist(Hi.getNode());
7589         FlipBit = DAG.getNode(ISD::AND, SDLoc(N0), MVT::i64, Hi, SignBit);
7590         AddToWorklist(FlipBit.getNode());
7591       }
7592       SDValue FlipBits =
7593           DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
7594       AddToWorklist(FlipBits.getNode());
7595       return DAG.getNode(ISD::XOR, DL, VT, NewConv, FlipBits);
7596     }
7597     APInt SignBit = APInt::getSignBit(VT.getSizeInBits());
7598     if (N0.getOpcode() == ISD::FNEG)
7599       return DAG.getNode(ISD::XOR, DL, VT,
7600                          NewConv, DAG.getConstant(SignBit, DL, VT));
7601     assert(N0.getOpcode() == ISD::FABS);
7602     return DAG.getNode(ISD::AND, DL, VT,
7603                        NewConv, DAG.getConstant(~SignBit, DL, VT));
7604   }
7605 
7606   // fold (bitconvert (fcopysign cst, x)) ->
7607   //         (or (and (bitconvert x), sign), (and cst, (not sign)))
7608   // Note that we don't handle (copysign x, cst) because this can always be
7609   // folded to an fneg or fabs.
7610   //
7611   // For ppc_fp128:
7612   // fold (bitcast (fcopysign cst, x)) ->
7613   //     flipbit = (and (extract_element
7614   //                     (xor (bitcast cst), (bitcast x)), 0),
7615   //                    signbit)
7616   //     (xor (bitcast cst) (build_pair flipbit, flipbit))
7617   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse() &&
7618       isa<ConstantFPSDNode>(N0.getOperand(0)) &&
7619       VT.isInteger() && !VT.isVector()) {
7620     unsigned OrigXWidth = N0.getOperand(1).getValueSizeInBits();
7621     EVT IntXVT = EVT::getIntegerVT(*DAG.getContext(), OrigXWidth);
7622     if (isTypeLegal(IntXVT)) {
7623       SDValue X = DAG.getBitcast(IntXVT, N0.getOperand(1));
7624       AddToWorklist(X.getNode());
7625 
7626       // If X has a different width than the result/lhs, sext it or truncate it.
7627       unsigned VTWidth = VT.getSizeInBits();
7628       if (OrigXWidth < VTWidth) {
7629         X = DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, X);
7630         AddToWorklist(X.getNode());
7631       } else if (OrigXWidth > VTWidth) {
7632         // To get the sign bit in the right place, we have to shift it right
7633         // before truncating.
7634         SDLoc DL(X);
7635         X = DAG.getNode(ISD::SRL, DL,
7636                         X.getValueType(), X,
7637                         DAG.getConstant(OrigXWidth-VTWidth, DL,
7638                                         X.getValueType()));
7639         AddToWorklist(X.getNode());
7640         X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
7641         AddToWorklist(X.getNode());
7642       }
7643 
7644       if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
7645         APInt SignBit = APInt::getSignBit(VT.getSizeInBits() / 2);
7646         SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
7647         AddToWorklist(Cst.getNode());
7648         SDValue X = DAG.getBitcast(VT, N0.getOperand(1));
7649         AddToWorklist(X.getNode());
7650         SDValue XorResult = DAG.getNode(ISD::XOR, SDLoc(N0), VT, Cst, X);
7651         AddToWorklist(XorResult.getNode());
7652         SDValue XorResult64 = DAG.getNode(
7653             ISD::EXTRACT_ELEMENT, SDLoc(XorResult), MVT::i64, XorResult,
7654             DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
7655                                   SDLoc(XorResult)));
7656         AddToWorklist(XorResult64.getNode());
7657         SDValue FlipBit =
7658             DAG.getNode(ISD::AND, SDLoc(XorResult64), MVT::i64, XorResult64,
7659                         DAG.getConstant(SignBit, SDLoc(XorResult64), MVT::i64));
7660         AddToWorklist(FlipBit.getNode());
7661         SDValue FlipBits =
7662             DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
7663         AddToWorklist(FlipBits.getNode());
7664         return DAG.getNode(ISD::XOR, SDLoc(N), VT, Cst, FlipBits);
7665       }
7666       APInt SignBit = APInt::getSignBit(VT.getSizeInBits());
7667       X = DAG.getNode(ISD::AND, SDLoc(X), VT,
7668                       X, DAG.getConstant(SignBit, SDLoc(X), VT));
7669       AddToWorklist(X.getNode());
7670 
7671       SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
7672       Cst = DAG.getNode(ISD::AND, SDLoc(Cst), VT,
7673                         Cst, DAG.getConstant(~SignBit, SDLoc(Cst), VT));
7674       AddToWorklist(Cst.getNode());
7675 
7676       return DAG.getNode(ISD::OR, SDLoc(N), VT, X, Cst);
7677     }
7678   }
7679 
7680   // bitconvert(build_pair(ld, ld)) -> ld iff load locations are consecutive.
7681   if (N0.getOpcode() == ISD::BUILD_PAIR)
7682     if (SDValue CombineLD = CombineConsecutiveLoads(N0.getNode(), VT))
7683       return CombineLD;
7684 
7685   // Remove double bitcasts from shuffles - this is often a legacy of
7686   // XformToShuffleWithZero being used to combine bitmaskings (of
7687   // float vectors bitcast to integer vectors) into shuffles.
7688   // bitcast(shuffle(bitcast(s0),bitcast(s1))) -> shuffle(s0,s1)
7689   if (Level < AfterLegalizeDAG && TLI.isTypeLegal(VT) && VT.isVector() &&
7690       N0->getOpcode() == ISD::VECTOR_SHUFFLE &&
7691       VT.getVectorNumElements() >= N0.getValueType().getVectorNumElements() &&
7692       !(VT.getVectorNumElements() % N0.getValueType().getVectorNumElements())) {
7693     ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N0);
7694 
7695     // If operands are a bitcast, peek through if it casts the original VT.
7696     // If operands are a constant, just bitcast back to original VT.
7697     auto PeekThroughBitcast = [&](SDValue Op) {
7698       if (Op.getOpcode() == ISD::BITCAST &&
7699           Op.getOperand(0).getValueType() == VT)
7700         return SDValue(Op.getOperand(0));
7701       if (ISD::isBuildVectorOfConstantSDNodes(Op.getNode()) ||
7702           ISD::isBuildVectorOfConstantFPSDNodes(Op.getNode()))
7703         return DAG.getBitcast(VT, Op);
7704       return SDValue();
7705     };
7706 
7707     SDValue SV0 = PeekThroughBitcast(N0->getOperand(0));
7708     SDValue SV1 = PeekThroughBitcast(N0->getOperand(1));
7709     if (!(SV0 && SV1))
7710       return SDValue();
7711 
7712     int MaskScale =
7713         VT.getVectorNumElements() / N0.getValueType().getVectorNumElements();
7714     SmallVector<int, 8> NewMask;
7715     for (int M : SVN->getMask())
7716       for (int i = 0; i != MaskScale; ++i)
7717         NewMask.push_back(M < 0 ? -1 : M * MaskScale + i);
7718 
7719     bool LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
7720     if (!LegalMask) {
7721       std::swap(SV0, SV1);
7722       ShuffleVectorSDNode::commuteMask(NewMask);
7723       LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
7724     }
7725 
7726     if (LegalMask)
7727       return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, NewMask);
7728   }
7729 
7730   return SDValue();
7731 }
7732 
7733 SDValue DAGCombiner::visitBUILD_PAIR(SDNode *N) {
7734   EVT VT = N->getValueType(0);
7735   return CombineConsecutiveLoads(N, VT);
7736 }
7737 
7738 /// We know that BV is a build_vector node with Constant, ConstantFP or Undef
7739 /// operands. DstEltVT indicates the destination element value type.
7740 SDValue DAGCombiner::
7741 ConstantFoldBITCASTofBUILD_VECTOR(SDNode *BV, EVT DstEltVT) {
7742   EVT SrcEltVT = BV->getValueType(0).getVectorElementType();
7743 
7744   // If this is already the right type, we're done.
7745   if (SrcEltVT == DstEltVT) return SDValue(BV, 0);
7746 
7747   unsigned SrcBitSize = SrcEltVT.getSizeInBits();
7748   unsigned DstBitSize = DstEltVT.getSizeInBits();
7749 
7750   // If this is a conversion of N elements of one type to N elements of another
7751   // type, convert each element.  This handles FP<->INT cases.
7752   if (SrcBitSize == DstBitSize) {
7753     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
7754                               BV->getValueType(0).getVectorNumElements());
7755 
7756     // Due to the FP element handling below calling this routine recursively,
7757     // we can end up with a scalar-to-vector node here.
7758     if (BV->getOpcode() == ISD::SCALAR_TO_VECTOR)
7759       return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(BV), VT,
7760                          DAG.getBitcast(DstEltVT, BV->getOperand(0)));
7761 
7762     SmallVector<SDValue, 8> Ops;
7763     for (SDValue Op : BV->op_values()) {
7764       // If the vector element type is not legal, the BUILD_VECTOR operands
7765       // are promoted and implicitly truncated.  Make that explicit here.
7766       if (Op.getValueType() != SrcEltVT)
7767         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(BV), SrcEltVT, Op);
7768       Ops.push_back(DAG.getBitcast(DstEltVT, Op));
7769       AddToWorklist(Ops.back().getNode());
7770     }
7771     return DAG.getBuildVector(VT, SDLoc(BV), Ops);
7772   }
7773 
7774   // Otherwise, we're growing or shrinking the elements.  To avoid having to
7775   // handle annoying details of growing/shrinking FP values, we convert them to
7776   // int first.
7777   if (SrcEltVT.isFloatingPoint()) {
7778     // Convert the input float vector to a int vector where the elements are the
7779     // same sizes.
7780     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), SrcEltVT.getSizeInBits());
7781     BV = ConstantFoldBITCASTofBUILD_VECTOR(BV, IntVT).getNode();
7782     SrcEltVT = IntVT;
7783   }
7784 
7785   // Now we know the input is an integer vector.  If the output is a FP type,
7786   // convert to integer first, then to FP of the right size.
7787   if (DstEltVT.isFloatingPoint()) {
7788     EVT TmpVT = EVT::getIntegerVT(*DAG.getContext(), DstEltVT.getSizeInBits());
7789     SDNode *Tmp = ConstantFoldBITCASTofBUILD_VECTOR(BV, TmpVT).getNode();
7790 
7791     // Next, convert to FP elements of the same size.
7792     return ConstantFoldBITCASTofBUILD_VECTOR(Tmp, DstEltVT);
7793   }
7794 
7795   SDLoc DL(BV);
7796 
7797   // Okay, we know the src/dst types are both integers of differing types.
7798   // Handling growing first.
7799   assert(SrcEltVT.isInteger() && DstEltVT.isInteger());
7800   if (SrcBitSize < DstBitSize) {
7801     unsigned NumInputsPerOutput = DstBitSize/SrcBitSize;
7802 
7803     SmallVector<SDValue, 8> Ops;
7804     for (unsigned i = 0, e = BV->getNumOperands(); i != e;
7805          i += NumInputsPerOutput) {
7806       bool isLE = DAG.getDataLayout().isLittleEndian();
7807       APInt NewBits = APInt(DstBitSize, 0);
7808       bool EltIsUndef = true;
7809       for (unsigned j = 0; j != NumInputsPerOutput; ++j) {
7810         // Shift the previously computed bits over.
7811         NewBits <<= SrcBitSize;
7812         SDValue Op = BV->getOperand(i+ (isLE ? (NumInputsPerOutput-j-1) : j));
7813         if (Op.isUndef()) continue;
7814         EltIsUndef = false;
7815 
7816         NewBits |= cast<ConstantSDNode>(Op)->getAPIntValue().
7817                    zextOrTrunc(SrcBitSize).zext(DstBitSize);
7818       }
7819 
7820       if (EltIsUndef)
7821         Ops.push_back(DAG.getUNDEF(DstEltVT));
7822       else
7823         Ops.push_back(DAG.getConstant(NewBits, DL, DstEltVT));
7824     }
7825 
7826     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT, Ops.size());
7827     return DAG.getBuildVector(VT, DL, Ops);
7828   }
7829 
7830   // Finally, this must be the case where we are shrinking elements: each input
7831   // turns into multiple outputs.
7832   unsigned NumOutputsPerInput = SrcBitSize/DstBitSize;
7833   EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
7834                             NumOutputsPerInput*BV->getNumOperands());
7835   SmallVector<SDValue, 8> Ops;
7836 
7837   for (const SDValue &Op : BV->op_values()) {
7838     if (Op.isUndef()) {
7839       Ops.append(NumOutputsPerInput, DAG.getUNDEF(DstEltVT));
7840       continue;
7841     }
7842 
7843     APInt OpVal = cast<ConstantSDNode>(Op)->
7844                   getAPIntValue().zextOrTrunc(SrcBitSize);
7845 
7846     for (unsigned j = 0; j != NumOutputsPerInput; ++j) {
7847       APInt ThisVal = OpVal.trunc(DstBitSize);
7848       Ops.push_back(DAG.getConstant(ThisVal, DL, DstEltVT));
7849       OpVal = OpVal.lshr(DstBitSize);
7850     }
7851 
7852     // For big endian targets, swap the order of the pieces of each element.
7853     if (DAG.getDataLayout().isBigEndian())
7854       std::reverse(Ops.end()-NumOutputsPerInput, Ops.end());
7855   }
7856 
7857   return DAG.getBuildVector(VT, DL, Ops);
7858 }
7859 
7860 /// Try to perform FMA combining on a given FADD node.
7861 SDValue DAGCombiner::visitFADDForFMACombine(SDNode *N) {
7862   SDValue N0 = N->getOperand(0);
7863   SDValue N1 = N->getOperand(1);
7864   EVT VT = N->getValueType(0);
7865   SDLoc SL(N);
7866 
7867   const TargetOptions &Options = DAG.getTarget().Options;
7868   bool AllowFusion =
7869       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
7870 
7871   // Floating-point multiply-add with intermediate rounding.
7872   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
7873 
7874   // Floating-point multiply-add without intermediate rounding.
7875   bool HasFMA =
7876       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
7877       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
7878 
7879   // No valid opcode, do not combine.
7880   if (!HasFMAD && !HasFMA)
7881     return SDValue();
7882 
7883   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
7884   ;
7885   if (AllowFusion && STI && STI->generateFMAsInMachineCombiner(OptLevel))
7886     return SDValue();
7887 
7888   // Always prefer FMAD to FMA for precision.
7889   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
7890   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
7891   bool LookThroughFPExt = TLI.isFPExtFree(VT);
7892 
7893   // If we have two choices trying to fold (fadd (fmul u, v), (fmul x, y)),
7894   // prefer to fold the multiply with fewer uses.
7895   if (Aggressive && N0.getOpcode() == ISD::FMUL &&
7896       N1.getOpcode() == ISD::FMUL) {
7897     if (N0.getNode()->use_size() > N1.getNode()->use_size())
7898       std::swap(N0, N1);
7899   }
7900 
7901   // fold (fadd (fmul x, y), z) -> (fma x, y, z)
7902   if (N0.getOpcode() == ISD::FMUL &&
7903       (Aggressive || N0->hasOneUse())) {
7904     return DAG.getNode(PreferredFusedOpcode, SL, VT,
7905                        N0.getOperand(0), N0.getOperand(1), N1);
7906   }
7907 
7908   // fold (fadd x, (fmul y, z)) -> (fma y, z, x)
7909   // Note: Commutes FADD operands.
7910   if (N1.getOpcode() == ISD::FMUL &&
7911       (Aggressive || N1->hasOneUse())) {
7912     return DAG.getNode(PreferredFusedOpcode, SL, VT,
7913                        N1.getOperand(0), N1.getOperand(1), N0);
7914   }
7915 
7916   // Look through FP_EXTEND nodes to do more combining.
7917   if (AllowFusion && LookThroughFPExt) {
7918     // fold (fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext y), z)
7919     if (N0.getOpcode() == ISD::FP_EXTEND) {
7920       SDValue N00 = N0.getOperand(0);
7921       if (N00.getOpcode() == ISD::FMUL)
7922         return DAG.getNode(PreferredFusedOpcode, SL, VT,
7923                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7924                                        N00.getOperand(0)),
7925                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7926                                        N00.getOperand(1)), N1);
7927     }
7928 
7929     // fold (fadd x, (fpext (fmul y, z))) -> (fma (fpext y), (fpext z), x)
7930     // Note: Commutes FADD operands.
7931     if (N1.getOpcode() == ISD::FP_EXTEND) {
7932       SDValue N10 = N1.getOperand(0);
7933       if (N10.getOpcode() == ISD::FMUL)
7934         return DAG.getNode(PreferredFusedOpcode, SL, VT,
7935                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7936                                        N10.getOperand(0)),
7937                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
7938                                        N10.getOperand(1)), N0);
7939     }
7940   }
7941 
7942   // More folding opportunities when target permits.
7943   if ((AllowFusion || HasFMAD)  && Aggressive) {
7944     // fold (fadd (fma x, y, (fmul u, v)), z) -> (fma x, y (fma u, v, z))
7945     if (N0.getOpcode() == PreferredFusedOpcode &&
7946         N0.getOperand(2).getOpcode() == ISD::FMUL) {
7947       return DAG.getNode(PreferredFusedOpcode, SL, VT,
7948                          N0.getOperand(0), N0.getOperand(1),
7949                          DAG.getNode(PreferredFusedOpcode, SL, VT,
7950                                      N0.getOperand(2).getOperand(0),
7951                                      N0.getOperand(2).getOperand(1),
7952                                      N1));
7953     }
7954 
7955     // fold (fadd x, (fma y, z, (fmul u, v)) -> (fma y, z (fma u, v, x))
7956     if (N1->getOpcode() == PreferredFusedOpcode &&
7957         N1.getOperand(2).getOpcode() == ISD::FMUL) {
7958       return DAG.getNode(PreferredFusedOpcode, SL, VT,
7959                          N1.getOperand(0), N1.getOperand(1),
7960                          DAG.getNode(PreferredFusedOpcode, SL, VT,
7961                                      N1.getOperand(2).getOperand(0),
7962                                      N1.getOperand(2).getOperand(1),
7963                                      N0));
7964     }
7965 
7966     if (AllowFusion && LookThroughFPExt) {
7967       // fold (fadd (fma x, y, (fpext (fmul u, v))), z)
7968       //   -> (fma x, y, (fma (fpext u), (fpext v), z))
7969       auto FoldFAddFMAFPExtFMul = [&] (
7970           SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z) {
7971         return DAG.getNode(PreferredFusedOpcode, SL, VT, X, Y,
7972                            DAG.getNode(PreferredFusedOpcode, SL, VT,
7973                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
7974                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
7975                                        Z));
7976       };
7977       if (N0.getOpcode() == PreferredFusedOpcode) {
7978         SDValue N02 = N0.getOperand(2);
7979         if (N02.getOpcode() == ISD::FP_EXTEND) {
7980           SDValue N020 = N02.getOperand(0);
7981           if (N020.getOpcode() == ISD::FMUL)
7982             return FoldFAddFMAFPExtFMul(N0.getOperand(0), N0.getOperand(1),
7983                                         N020.getOperand(0), N020.getOperand(1),
7984                                         N1);
7985         }
7986       }
7987 
7988       // fold (fadd (fpext (fma x, y, (fmul u, v))), z)
7989       //   -> (fma (fpext x), (fpext y), (fma (fpext u), (fpext v), z))
7990       // FIXME: This turns two single-precision and one double-precision
7991       // operation into two double-precision operations, which might not be
7992       // interesting for all targets, especially GPUs.
7993       auto FoldFAddFPExtFMAFMul = [&] (
7994           SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z) {
7995         return DAG.getNode(PreferredFusedOpcode, SL, VT,
7996                            DAG.getNode(ISD::FP_EXTEND, SL, VT, X),
7997                            DAG.getNode(ISD::FP_EXTEND, SL, VT, Y),
7998                            DAG.getNode(PreferredFusedOpcode, SL, VT,
7999                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
8000                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
8001                                        Z));
8002       };
8003       if (N0.getOpcode() == ISD::FP_EXTEND) {
8004         SDValue N00 = N0.getOperand(0);
8005         if (N00.getOpcode() == PreferredFusedOpcode) {
8006           SDValue N002 = N00.getOperand(2);
8007           if (N002.getOpcode() == ISD::FMUL)
8008             return FoldFAddFPExtFMAFMul(N00.getOperand(0), N00.getOperand(1),
8009                                         N002.getOperand(0), N002.getOperand(1),
8010                                         N1);
8011         }
8012       }
8013 
8014       // fold (fadd x, (fma y, z, (fpext (fmul u, v)))
8015       //   -> (fma y, z, (fma (fpext u), (fpext v), x))
8016       if (N1.getOpcode() == PreferredFusedOpcode) {
8017         SDValue N12 = N1.getOperand(2);
8018         if (N12.getOpcode() == ISD::FP_EXTEND) {
8019           SDValue N120 = N12.getOperand(0);
8020           if (N120.getOpcode() == ISD::FMUL)
8021             return FoldFAddFMAFPExtFMul(N1.getOperand(0), N1.getOperand(1),
8022                                         N120.getOperand(0), N120.getOperand(1),
8023                                         N0);
8024         }
8025       }
8026 
8027       // fold (fadd x, (fpext (fma y, z, (fmul u, v)))
8028       //   -> (fma (fpext y), (fpext z), (fma (fpext u), (fpext v), x))
8029       // FIXME: This turns two single-precision and one double-precision
8030       // operation into two double-precision operations, which might not be
8031       // interesting for all targets, especially GPUs.
8032       if (N1.getOpcode() == ISD::FP_EXTEND) {
8033         SDValue N10 = N1.getOperand(0);
8034         if (N10.getOpcode() == PreferredFusedOpcode) {
8035           SDValue N102 = N10.getOperand(2);
8036           if (N102.getOpcode() == ISD::FMUL)
8037             return FoldFAddFPExtFMAFMul(N10.getOperand(0), N10.getOperand(1),
8038                                         N102.getOperand(0), N102.getOperand(1),
8039                                         N0);
8040         }
8041       }
8042     }
8043   }
8044 
8045   return SDValue();
8046 }
8047 
8048 /// Try to perform FMA combining on a given FSUB node.
8049 SDValue DAGCombiner::visitFSUBForFMACombine(SDNode *N) {
8050   SDValue N0 = N->getOperand(0);
8051   SDValue N1 = N->getOperand(1);
8052   EVT VT = N->getValueType(0);
8053   SDLoc SL(N);
8054 
8055   const TargetOptions &Options = DAG.getTarget().Options;
8056   bool AllowFusion =
8057       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
8058 
8059   // Floating-point multiply-add with intermediate rounding.
8060   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
8061 
8062   // Floating-point multiply-add without intermediate rounding.
8063   bool HasFMA =
8064       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
8065       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
8066 
8067   // No valid opcode, do not combine.
8068   if (!HasFMAD && !HasFMA)
8069     return SDValue();
8070 
8071   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
8072   if (AllowFusion && STI && STI->generateFMAsInMachineCombiner(OptLevel))
8073     return SDValue();
8074 
8075   // Always prefer FMAD to FMA for precision.
8076   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
8077   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
8078   bool LookThroughFPExt = TLI.isFPExtFree(VT);
8079 
8080   // fold (fsub (fmul x, y), z) -> (fma x, y, (fneg z))
8081   if (N0.getOpcode() == ISD::FMUL &&
8082       (Aggressive || N0->hasOneUse())) {
8083     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8084                        N0.getOperand(0), N0.getOperand(1),
8085                        DAG.getNode(ISD::FNEG, SL, VT, N1));
8086   }
8087 
8088   // fold (fsub x, (fmul y, z)) -> (fma (fneg y), z, x)
8089   // Note: Commutes FSUB operands.
8090   if (N1.getOpcode() == ISD::FMUL &&
8091       (Aggressive || N1->hasOneUse()))
8092     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8093                        DAG.getNode(ISD::FNEG, SL, VT,
8094                                    N1.getOperand(0)),
8095                        N1.getOperand(1), N0);
8096 
8097   // fold (fsub (fneg (fmul, x, y)), z) -> (fma (fneg x), y, (fneg z))
8098   if (N0.getOpcode() == ISD::FNEG &&
8099       N0.getOperand(0).getOpcode() == ISD::FMUL &&
8100       (Aggressive || (N0->hasOneUse() && N0.getOperand(0).hasOneUse()))) {
8101     SDValue N00 = N0.getOperand(0).getOperand(0);
8102     SDValue N01 = N0.getOperand(0).getOperand(1);
8103     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8104                        DAG.getNode(ISD::FNEG, SL, VT, N00), N01,
8105                        DAG.getNode(ISD::FNEG, SL, VT, N1));
8106   }
8107 
8108   // Look through FP_EXTEND nodes to do more combining.
8109   if (AllowFusion && LookThroughFPExt) {
8110     // fold (fsub (fpext (fmul x, y)), z)
8111     //   -> (fma (fpext x), (fpext y), (fneg z))
8112     if (N0.getOpcode() == ISD::FP_EXTEND) {
8113       SDValue N00 = N0.getOperand(0);
8114       if (N00.getOpcode() == ISD::FMUL)
8115         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8116                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8117                                        N00.getOperand(0)),
8118                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8119                                        N00.getOperand(1)),
8120                            DAG.getNode(ISD::FNEG, SL, VT, N1));
8121     }
8122 
8123     // fold (fsub x, (fpext (fmul y, z)))
8124     //   -> (fma (fneg (fpext y)), (fpext z), x)
8125     // Note: Commutes FSUB operands.
8126     if (N1.getOpcode() == ISD::FP_EXTEND) {
8127       SDValue N10 = N1.getOperand(0);
8128       if (N10.getOpcode() == ISD::FMUL)
8129         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8130                            DAG.getNode(ISD::FNEG, SL, VT,
8131                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
8132                                                    N10.getOperand(0))),
8133                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8134                                        N10.getOperand(1)),
8135                            N0);
8136     }
8137 
8138     // fold (fsub (fpext (fneg (fmul, x, y))), z)
8139     //   -> (fneg (fma (fpext x), (fpext y), z))
8140     // Note: This could be removed with appropriate canonicalization of the
8141     // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
8142     // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
8143     // from implementing the canonicalization in visitFSUB.
8144     if (N0.getOpcode() == ISD::FP_EXTEND) {
8145       SDValue N00 = N0.getOperand(0);
8146       if (N00.getOpcode() == ISD::FNEG) {
8147         SDValue N000 = N00.getOperand(0);
8148         if (N000.getOpcode() == ISD::FMUL) {
8149           return DAG.getNode(ISD::FNEG, SL, VT,
8150                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8151                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8152                                                      N000.getOperand(0)),
8153                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8154                                                      N000.getOperand(1)),
8155                                          N1));
8156         }
8157       }
8158     }
8159 
8160     // fold (fsub (fneg (fpext (fmul, x, y))), z)
8161     //   -> (fneg (fma (fpext x)), (fpext y), z)
8162     // Note: This could be removed with appropriate canonicalization of the
8163     // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
8164     // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
8165     // from implementing the canonicalization in visitFSUB.
8166     if (N0.getOpcode() == ISD::FNEG) {
8167       SDValue N00 = N0.getOperand(0);
8168       if (N00.getOpcode() == ISD::FP_EXTEND) {
8169         SDValue N000 = N00.getOperand(0);
8170         if (N000.getOpcode() == ISD::FMUL) {
8171           return DAG.getNode(ISD::FNEG, SL, VT,
8172                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8173                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8174                                                      N000.getOperand(0)),
8175                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8176                                                      N000.getOperand(1)),
8177                                          N1));
8178         }
8179       }
8180     }
8181 
8182   }
8183 
8184   // More folding opportunities when target permits.
8185   if ((AllowFusion || HasFMAD) && Aggressive) {
8186     // fold (fsub (fma x, y, (fmul u, v)), z)
8187     //   -> (fma x, y (fma u, v, (fneg z)))
8188     if (N0.getOpcode() == PreferredFusedOpcode &&
8189         N0.getOperand(2).getOpcode() == ISD::FMUL) {
8190       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8191                          N0.getOperand(0), N0.getOperand(1),
8192                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8193                                      N0.getOperand(2).getOperand(0),
8194                                      N0.getOperand(2).getOperand(1),
8195                                      DAG.getNode(ISD::FNEG, SL, VT,
8196                                                  N1)));
8197     }
8198 
8199     // fold (fsub x, (fma y, z, (fmul u, v)))
8200     //   -> (fma (fneg y), z, (fma (fneg u), v, x))
8201     if (N1.getOpcode() == PreferredFusedOpcode &&
8202         N1.getOperand(2).getOpcode() == ISD::FMUL) {
8203       SDValue N20 = N1.getOperand(2).getOperand(0);
8204       SDValue N21 = N1.getOperand(2).getOperand(1);
8205       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8206                          DAG.getNode(ISD::FNEG, SL, VT,
8207                                      N1.getOperand(0)),
8208                          N1.getOperand(1),
8209                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8210                                      DAG.getNode(ISD::FNEG, SL, VT, N20),
8211 
8212                                      N21, N0));
8213     }
8214 
8215     if (AllowFusion && LookThroughFPExt) {
8216       // fold (fsub (fma x, y, (fpext (fmul u, v))), z)
8217       //   -> (fma x, y (fma (fpext u), (fpext v), (fneg z)))
8218       if (N0.getOpcode() == PreferredFusedOpcode) {
8219         SDValue N02 = N0.getOperand(2);
8220         if (N02.getOpcode() == ISD::FP_EXTEND) {
8221           SDValue N020 = N02.getOperand(0);
8222           if (N020.getOpcode() == ISD::FMUL)
8223             return DAG.getNode(PreferredFusedOpcode, SL, VT,
8224                                N0.getOperand(0), N0.getOperand(1),
8225                                DAG.getNode(PreferredFusedOpcode, SL, VT,
8226                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8227                                                        N020.getOperand(0)),
8228                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8229                                                        N020.getOperand(1)),
8230                                            DAG.getNode(ISD::FNEG, SL, VT,
8231                                                        N1)));
8232         }
8233       }
8234 
8235       // fold (fsub (fpext (fma x, y, (fmul u, v))), z)
8236       //   -> (fma (fpext x), (fpext y),
8237       //           (fma (fpext u), (fpext v), (fneg z)))
8238       // FIXME: This turns two single-precision and one double-precision
8239       // operation into two double-precision operations, which might not be
8240       // interesting for all targets, especially GPUs.
8241       if (N0.getOpcode() == ISD::FP_EXTEND) {
8242         SDValue N00 = N0.getOperand(0);
8243         if (N00.getOpcode() == PreferredFusedOpcode) {
8244           SDValue N002 = N00.getOperand(2);
8245           if (N002.getOpcode() == ISD::FMUL)
8246             return DAG.getNode(PreferredFusedOpcode, SL, VT,
8247                                DAG.getNode(ISD::FP_EXTEND, SL, VT,
8248                                            N00.getOperand(0)),
8249                                DAG.getNode(ISD::FP_EXTEND, SL, VT,
8250                                            N00.getOperand(1)),
8251                                DAG.getNode(PreferredFusedOpcode, SL, VT,
8252                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8253                                                        N002.getOperand(0)),
8254                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8255                                                        N002.getOperand(1)),
8256                                            DAG.getNode(ISD::FNEG, SL, VT,
8257                                                        N1)));
8258         }
8259       }
8260 
8261       // fold (fsub x, (fma y, z, (fpext (fmul u, v))))
8262       //   -> (fma (fneg y), z, (fma (fneg (fpext u)), (fpext v), x))
8263       if (N1.getOpcode() == PreferredFusedOpcode &&
8264         N1.getOperand(2).getOpcode() == ISD::FP_EXTEND) {
8265         SDValue N120 = N1.getOperand(2).getOperand(0);
8266         if (N120.getOpcode() == ISD::FMUL) {
8267           SDValue N1200 = N120.getOperand(0);
8268           SDValue N1201 = N120.getOperand(1);
8269           return DAG.getNode(PreferredFusedOpcode, SL, VT,
8270                              DAG.getNode(ISD::FNEG, SL, VT, N1.getOperand(0)),
8271                              N1.getOperand(1),
8272                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8273                                          DAG.getNode(ISD::FNEG, SL, VT,
8274                                              DAG.getNode(ISD::FP_EXTEND, SL,
8275                                                          VT, N1200)),
8276                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8277                                                      N1201),
8278                                          N0));
8279         }
8280       }
8281 
8282       // fold (fsub x, (fpext (fma y, z, (fmul u, v))))
8283       //   -> (fma (fneg (fpext y)), (fpext z),
8284       //           (fma (fneg (fpext u)), (fpext v), x))
8285       // FIXME: This turns two single-precision and one double-precision
8286       // operation into two double-precision operations, which might not be
8287       // interesting for all targets, especially GPUs.
8288       if (N1.getOpcode() == ISD::FP_EXTEND &&
8289         N1.getOperand(0).getOpcode() == PreferredFusedOpcode) {
8290         SDValue N100 = N1.getOperand(0).getOperand(0);
8291         SDValue N101 = N1.getOperand(0).getOperand(1);
8292         SDValue N102 = N1.getOperand(0).getOperand(2);
8293         if (N102.getOpcode() == ISD::FMUL) {
8294           SDValue N1020 = N102.getOperand(0);
8295           SDValue N1021 = N102.getOperand(1);
8296           return DAG.getNode(PreferredFusedOpcode, SL, VT,
8297                              DAG.getNode(ISD::FNEG, SL, VT,
8298                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8299                                                      N100)),
8300                              DAG.getNode(ISD::FP_EXTEND, SL, VT, N101),
8301                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8302                                          DAG.getNode(ISD::FNEG, SL, VT,
8303                                              DAG.getNode(ISD::FP_EXTEND, SL,
8304                                                          VT, N1020)),
8305                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8306                                                      N1021),
8307                                          N0));
8308         }
8309       }
8310     }
8311   }
8312 
8313   return SDValue();
8314 }
8315 
8316 /// Try to perform FMA combining on a given FMUL node.
8317 SDValue DAGCombiner::visitFMULForFMACombine(SDNode *N) {
8318   SDValue N0 = N->getOperand(0);
8319   SDValue N1 = N->getOperand(1);
8320   EVT VT = N->getValueType(0);
8321   SDLoc SL(N);
8322 
8323   assert(N->getOpcode() == ISD::FMUL && "Expected FMUL Operation");
8324 
8325   const TargetOptions &Options = DAG.getTarget().Options;
8326   bool AllowFusion =
8327       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
8328 
8329   // Floating-point multiply-add with intermediate rounding.
8330   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
8331 
8332   // Floating-point multiply-add without intermediate rounding.
8333   bool HasFMA =
8334       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
8335       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
8336 
8337   // No valid opcode, do not combine.
8338   if (!HasFMAD && !HasFMA)
8339     return SDValue();
8340 
8341   // Always prefer FMAD to FMA for precision.
8342   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
8343   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
8344 
8345   // fold (fmul (fadd x, +1.0), y) -> (fma x, y, y)
8346   // fold (fmul (fadd x, -1.0), y) -> (fma x, y, (fneg y))
8347   auto FuseFADD = [&](SDValue X, SDValue Y) {
8348     if (X.getOpcode() == ISD::FADD && (Aggressive || X->hasOneUse())) {
8349       auto XC1 = isConstOrConstSplatFP(X.getOperand(1));
8350       if (XC1 && XC1->isExactlyValue(+1.0))
8351         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y);
8352       if (XC1 && XC1->isExactlyValue(-1.0))
8353         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
8354                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8355     }
8356     return SDValue();
8357   };
8358 
8359   if (SDValue FMA = FuseFADD(N0, N1))
8360     return FMA;
8361   if (SDValue FMA = FuseFADD(N1, N0))
8362     return FMA;
8363 
8364   // fold (fmul (fsub +1.0, x), y) -> (fma (fneg x), y, y)
8365   // fold (fmul (fsub -1.0, x), y) -> (fma (fneg x), y, (fneg y))
8366   // fold (fmul (fsub x, +1.0), y) -> (fma x, y, (fneg y))
8367   // fold (fmul (fsub x, -1.0), y) -> (fma x, y, y)
8368   auto FuseFSUB = [&](SDValue X, SDValue Y) {
8369     if (X.getOpcode() == ISD::FSUB && (Aggressive || X->hasOneUse())) {
8370       auto XC0 = isConstOrConstSplatFP(X.getOperand(0));
8371       if (XC0 && XC0->isExactlyValue(+1.0))
8372         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8373                            DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
8374                            Y);
8375       if (XC0 && XC0->isExactlyValue(-1.0))
8376         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8377                            DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
8378                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8379 
8380       auto XC1 = isConstOrConstSplatFP(X.getOperand(1));
8381       if (XC1 && XC1->isExactlyValue(+1.0))
8382         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
8383                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8384       if (XC1 && XC1->isExactlyValue(-1.0))
8385         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y);
8386     }
8387     return SDValue();
8388   };
8389 
8390   if (SDValue FMA = FuseFSUB(N0, N1))
8391     return FMA;
8392   if (SDValue FMA = FuseFSUB(N1, N0))
8393     return FMA;
8394 
8395   return SDValue();
8396 }
8397 
8398 SDValue DAGCombiner::visitFADD(SDNode *N) {
8399   SDValue N0 = N->getOperand(0);
8400   SDValue N1 = N->getOperand(1);
8401   bool N0CFP = isConstantFPBuildVectorOrConstantFP(N0);
8402   bool N1CFP = isConstantFPBuildVectorOrConstantFP(N1);
8403   EVT VT = N->getValueType(0);
8404   SDLoc DL(N);
8405   const TargetOptions &Options = DAG.getTarget().Options;
8406   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8407 
8408   // fold vector ops
8409   if (VT.isVector())
8410     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8411       return FoldedVOp;
8412 
8413   // fold (fadd c1, c2) -> c1 + c2
8414   if (N0CFP && N1CFP)
8415     return DAG.getNode(ISD::FADD, DL, VT, N0, N1, Flags);
8416 
8417   // canonicalize constant to RHS
8418   if (N0CFP && !N1CFP)
8419     return DAG.getNode(ISD::FADD, DL, VT, N1, N0, Flags);
8420 
8421   // fold (fadd A, (fneg B)) -> (fsub A, B)
8422   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
8423       isNegatibleForFree(N1, LegalOperations, TLI, &Options) == 2)
8424     return DAG.getNode(ISD::FSUB, DL, VT, N0,
8425                        GetNegatedExpression(N1, DAG, LegalOperations), Flags);
8426 
8427   // fold (fadd (fneg A), B) -> (fsub B, A)
8428   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
8429       isNegatibleForFree(N0, LegalOperations, TLI, &Options) == 2)
8430     return DAG.getNode(ISD::FSUB, DL, VT, N1,
8431                        GetNegatedExpression(N0, DAG, LegalOperations), Flags);
8432 
8433   // If 'unsafe math' is enabled, fold lots of things.
8434   if (Options.UnsafeFPMath) {
8435     // No FP constant should be created after legalization as Instruction
8436     // Selection pass has a hard time dealing with FP constants.
8437     bool AllowNewConst = (Level < AfterLegalizeDAG);
8438 
8439     // fold (fadd A, 0) -> A
8440     if (ConstantFPSDNode *N1C = isConstOrConstSplatFP(N1))
8441       if (N1C->isZero())
8442         return N0;
8443 
8444     // fold (fadd (fadd x, c1), c2) -> (fadd x, (fadd c1, c2))
8445     if (N1CFP && N0.getOpcode() == ISD::FADD && N0.getNode()->hasOneUse() &&
8446         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1)))
8447       return DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(0),
8448                          DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1), N1,
8449                                      Flags),
8450                          Flags);
8451 
8452     // If allowed, fold (fadd (fneg x), x) -> 0.0
8453     if (AllowNewConst && N0.getOpcode() == ISD::FNEG && N0.getOperand(0) == N1)
8454       return DAG.getConstantFP(0.0, DL, VT);
8455 
8456     // If allowed, fold (fadd x, (fneg x)) -> 0.0
8457     if (AllowNewConst && N1.getOpcode() == ISD::FNEG && N1.getOperand(0) == N0)
8458       return DAG.getConstantFP(0.0, DL, VT);
8459 
8460     // We can fold chains of FADD's of the same value into multiplications.
8461     // This transform is not safe in general because we are reducing the number
8462     // of rounding steps.
8463     if (TLI.isOperationLegalOrCustom(ISD::FMUL, VT) && !N0CFP && !N1CFP) {
8464       if (N0.getOpcode() == ISD::FMUL) {
8465         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
8466         bool CFP01 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(1));
8467 
8468         // (fadd (fmul x, c), x) -> (fmul x, c+1)
8469         if (CFP01 && !CFP00 && N0.getOperand(0) == N1) {
8470           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
8471                                        DAG.getConstantFP(1.0, DL, VT), Flags);
8472           return DAG.getNode(ISD::FMUL, DL, VT, N1, NewCFP, Flags);
8473         }
8474 
8475         // (fadd (fmul x, c), (fadd x, x)) -> (fmul x, c+2)
8476         if (CFP01 && !CFP00 && N1.getOpcode() == ISD::FADD &&
8477             N1.getOperand(0) == N1.getOperand(1) &&
8478             N0.getOperand(0) == N1.getOperand(0)) {
8479           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
8480                                        DAG.getConstantFP(2.0, DL, VT), Flags);
8481           return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), NewCFP, Flags);
8482         }
8483       }
8484 
8485       if (N1.getOpcode() == ISD::FMUL) {
8486         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
8487         bool CFP11 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(1));
8488 
8489         // (fadd x, (fmul x, c)) -> (fmul x, c+1)
8490         if (CFP11 && !CFP10 && N1.getOperand(0) == N0) {
8491           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
8492                                        DAG.getConstantFP(1.0, DL, VT), Flags);
8493           return DAG.getNode(ISD::FMUL, DL, VT, N0, NewCFP, Flags);
8494         }
8495 
8496         // (fadd (fadd x, x), (fmul x, c)) -> (fmul x, c+2)
8497         if (CFP11 && !CFP10 && N0.getOpcode() == ISD::FADD &&
8498             N0.getOperand(0) == N0.getOperand(1) &&
8499             N1.getOperand(0) == N0.getOperand(0)) {
8500           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
8501                                        DAG.getConstantFP(2.0, DL, VT), Flags);
8502           return DAG.getNode(ISD::FMUL, DL, VT, N1.getOperand(0), NewCFP, Flags);
8503         }
8504       }
8505 
8506       if (N0.getOpcode() == ISD::FADD && AllowNewConst) {
8507         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
8508         // (fadd (fadd x, x), x) -> (fmul x, 3.0)
8509         if (!CFP00 && N0.getOperand(0) == N0.getOperand(1) &&
8510             (N0.getOperand(0) == N1)) {
8511           return DAG.getNode(ISD::FMUL, DL, VT,
8512                              N1, DAG.getConstantFP(3.0, DL, VT), Flags);
8513         }
8514       }
8515 
8516       if (N1.getOpcode() == ISD::FADD && AllowNewConst) {
8517         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
8518         // (fadd x, (fadd x, x)) -> (fmul x, 3.0)
8519         if (!CFP10 && N1.getOperand(0) == N1.getOperand(1) &&
8520             N1.getOperand(0) == N0) {
8521           return DAG.getNode(ISD::FMUL, DL, VT,
8522                              N0, DAG.getConstantFP(3.0, DL, VT), Flags);
8523         }
8524       }
8525 
8526       // (fadd (fadd x, x), (fadd x, x)) -> (fmul x, 4.0)
8527       if (AllowNewConst &&
8528           N0.getOpcode() == ISD::FADD && N1.getOpcode() == ISD::FADD &&
8529           N0.getOperand(0) == N0.getOperand(1) &&
8530           N1.getOperand(0) == N1.getOperand(1) &&
8531           N0.getOperand(0) == N1.getOperand(0)) {
8532         return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0),
8533                            DAG.getConstantFP(4.0, DL, VT), Flags);
8534       }
8535     }
8536   } // enable-unsafe-fp-math
8537 
8538   // FADD -> FMA combines:
8539   if (SDValue Fused = visitFADDForFMACombine(N)) {
8540     AddToWorklist(Fused.getNode());
8541     return Fused;
8542   }
8543   return SDValue();
8544 }
8545 
8546 SDValue DAGCombiner::visitFSUB(SDNode *N) {
8547   SDValue N0 = N->getOperand(0);
8548   SDValue N1 = N->getOperand(1);
8549   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
8550   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
8551   EVT VT = N->getValueType(0);
8552   SDLoc DL(N);
8553   const TargetOptions &Options = DAG.getTarget().Options;
8554   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8555 
8556   // fold vector ops
8557   if (VT.isVector())
8558     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8559       return FoldedVOp;
8560 
8561   // fold (fsub c1, c2) -> c1-c2
8562   if (N0CFP && N1CFP)
8563     return DAG.getNode(ISD::FSUB, DL, VT, N0, N1, Flags);
8564 
8565   // fold (fsub A, (fneg B)) -> (fadd A, B)
8566   if (isNegatibleForFree(N1, LegalOperations, TLI, &Options))
8567     return DAG.getNode(ISD::FADD, DL, VT, N0,
8568                        GetNegatedExpression(N1, DAG, LegalOperations), Flags);
8569 
8570   // If 'unsafe math' is enabled, fold lots of things.
8571   if (Options.UnsafeFPMath) {
8572     // (fsub A, 0) -> A
8573     if (N1CFP && N1CFP->isZero())
8574       return N0;
8575 
8576     // (fsub 0, B) -> -B
8577     if (N0CFP && N0CFP->isZero()) {
8578       if (isNegatibleForFree(N1, LegalOperations, TLI, &Options))
8579         return GetNegatedExpression(N1, DAG, LegalOperations);
8580       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
8581         return DAG.getNode(ISD::FNEG, DL, VT, N1);
8582     }
8583 
8584     // (fsub x, x) -> 0.0
8585     if (N0 == N1)
8586       return DAG.getConstantFP(0.0f, DL, VT);
8587 
8588     // (fsub x, (fadd x, y)) -> (fneg y)
8589     // (fsub x, (fadd y, x)) -> (fneg y)
8590     if (N1.getOpcode() == ISD::FADD) {
8591       SDValue N10 = N1->getOperand(0);
8592       SDValue N11 = N1->getOperand(1);
8593 
8594       if (N10 == N0 && isNegatibleForFree(N11, LegalOperations, TLI, &Options))
8595         return GetNegatedExpression(N11, DAG, LegalOperations);
8596 
8597       if (N11 == N0 && isNegatibleForFree(N10, LegalOperations, TLI, &Options))
8598         return GetNegatedExpression(N10, DAG, LegalOperations);
8599     }
8600   }
8601 
8602   // FSUB -> FMA combines:
8603   if (SDValue Fused = visitFSUBForFMACombine(N)) {
8604     AddToWorklist(Fused.getNode());
8605     return Fused;
8606   }
8607 
8608   return SDValue();
8609 }
8610 
8611 SDValue DAGCombiner::visitFMUL(SDNode *N) {
8612   SDValue N0 = N->getOperand(0);
8613   SDValue N1 = N->getOperand(1);
8614   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
8615   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
8616   EVT VT = N->getValueType(0);
8617   SDLoc DL(N);
8618   const TargetOptions &Options = DAG.getTarget().Options;
8619   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8620 
8621   // fold vector ops
8622   if (VT.isVector()) {
8623     // This just handles C1 * C2 for vectors. Other vector folds are below.
8624     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8625       return FoldedVOp;
8626   }
8627 
8628   // fold (fmul c1, c2) -> c1*c2
8629   if (N0CFP && N1CFP)
8630     return DAG.getNode(ISD::FMUL, DL, VT, N0, N1, Flags);
8631 
8632   // canonicalize constant to RHS
8633   if (isConstantFPBuildVectorOrConstantFP(N0) &&
8634      !isConstantFPBuildVectorOrConstantFP(N1))
8635     return DAG.getNode(ISD::FMUL, DL, VT, N1, N0, Flags);
8636 
8637   // fold (fmul A, 1.0) -> A
8638   if (N1CFP && N1CFP->isExactlyValue(1.0))
8639     return N0;
8640 
8641   if (Options.UnsafeFPMath) {
8642     // fold (fmul A, 0) -> 0
8643     if (N1CFP && N1CFP->isZero())
8644       return N1;
8645 
8646     // fold (fmul (fmul x, c1), c2) -> (fmul x, (fmul c1, c2))
8647     if (N0.getOpcode() == ISD::FMUL) {
8648       // Fold scalars or any vector constants (not just splats).
8649       // This fold is done in general by InstCombine, but extra fmul insts
8650       // may have been generated during lowering.
8651       SDValue N00 = N0.getOperand(0);
8652       SDValue N01 = N0.getOperand(1);
8653       auto *BV1 = dyn_cast<BuildVectorSDNode>(N1);
8654       auto *BV00 = dyn_cast<BuildVectorSDNode>(N00);
8655       auto *BV01 = dyn_cast<BuildVectorSDNode>(N01);
8656 
8657       // Check 1: Make sure that the first operand of the inner multiply is NOT
8658       // a constant. Otherwise, we may induce infinite looping.
8659       if (!(isConstOrConstSplatFP(N00) || (BV00 && BV00->isConstant()))) {
8660         // Check 2: Make sure that the second operand of the inner multiply and
8661         // the second operand of the outer multiply are constants.
8662         if ((N1CFP && isConstOrConstSplatFP(N01)) ||
8663             (BV1 && BV01 && BV1->isConstant() && BV01->isConstant())) {
8664           SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, N01, N1, Flags);
8665           return DAG.getNode(ISD::FMUL, DL, VT, N00, MulConsts, Flags);
8666         }
8667       }
8668     }
8669 
8670     // fold (fmul (fadd x, x), c) -> (fmul x, (fmul 2.0, c))
8671     // Undo the fmul 2.0, x -> fadd x, x transformation, since if it occurs
8672     // during an early run of DAGCombiner can prevent folding with fmuls
8673     // inserted during lowering.
8674     if (N0.getOpcode() == ISD::FADD &&
8675         (N0.getOperand(0) == N0.getOperand(1)) &&
8676         N0.hasOneUse()) {
8677       const SDValue Two = DAG.getConstantFP(2.0, DL, VT);
8678       SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, Two, N1, Flags);
8679       return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), MulConsts, Flags);
8680     }
8681   }
8682 
8683   // fold (fmul X, 2.0) -> (fadd X, X)
8684   if (N1CFP && N1CFP->isExactlyValue(+2.0))
8685     return DAG.getNode(ISD::FADD, DL, VT, N0, N0, Flags);
8686 
8687   // fold (fmul X, -1.0) -> (fneg X)
8688   if (N1CFP && N1CFP->isExactlyValue(-1.0))
8689     if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
8690       return DAG.getNode(ISD::FNEG, DL, VT, N0);
8691 
8692   // fold (fmul (fneg X), (fneg Y)) -> (fmul X, Y)
8693   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) {
8694     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) {
8695       // Both can be negated for free, check to see if at least one is cheaper
8696       // negated.
8697       if (LHSNeg == 2 || RHSNeg == 2)
8698         return DAG.getNode(ISD::FMUL, DL, VT,
8699                            GetNegatedExpression(N0, DAG, LegalOperations),
8700                            GetNegatedExpression(N1, DAG, LegalOperations),
8701                            Flags);
8702     }
8703   }
8704 
8705   // FMUL -> FMA combines:
8706   if (SDValue Fused = visitFMULForFMACombine(N)) {
8707     AddToWorklist(Fused.getNode());
8708     return Fused;
8709   }
8710 
8711   return SDValue();
8712 }
8713 
8714 SDValue DAGCombiner::visitFMA(SDNode *N) {
8715   SDValue N0 = N->getOperand(0);
8716   SDValue N1 = N->getOperand(1);
8717   SDValue N2 = N->getOperand(2);
8718   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
8719   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
8720   EVT VT = N->getValueType(0);
8721   SDLoc DL(N);
8722   const TargetOptions &Options = DAG.getTarget().Options;
8723 
8724   // Constant fold FMA.
8725   if (isa<ConstantFPSDNode>(N0) &&
8726       isa<ConstantFPSDNode>(N1) &&
8727       isa<ConstantFPSDNode>(N2)) {
8728     return DAG.getNode(ISD::FMA, DL, VT, N0, N1, N2);
8729   }
8730 
8731   if (Options.UnsafeFPMath) {
8732     if (N0CFP && N0CFP->isZero())
8733       return N2;
8734     if (N1CFP && N1CFP->isZero())
8735       return N2;
8736   }
8737   // TODO: The FMA node should have flags that propagate to these nodes.
8738   if (N0CFP && N0CFP->isExactlyValue(1.0))
8739     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N1, N2);
8740   if (N1CFP && N1CFP->isExactlyValue(1.0))
8741     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N0, N2);
8742 
8743   // Canonicalize (fma c, x, y) -> (fma x, c, y)
8744   if (isConstantFPBuildVectorOrConstantFP(N0) &&
8745      !isConstantFPBuildVectorOrConstantFP(N1))
8746     return DAG.getNode(ISD::FMA, SDLoc(N), VT, N1, N0, N2);
8747 
8748   // TODO: FMA nodes should have flags that propagate to the created nodes.
8749   // For now, create a Flags object for use with all unsafe math transforms.
8750   SDNodeFlags Flags;
8751   Flags.setUnsafeAlgebra(true);
8752 
8753   if (Options.UnsafeFPMath) {
8754     // (fma x, c1, (fmul x, c2)) -> (fmul x, c1+c2)
8755     if (N2.getOpcode() == ISD::FMUL && N0 == N2.getOperand(0) &&
8756         isConstantFPBuildVectorOrConstantFP(N1) &&
8757         isConstantFPBuildVectorOrConstantFP(N2.getOperand(1))) {
8758       return DAG.getNode(ISD::FMUL, DL, VT, N0,
8759                          DAG.getNode(ISD::FADD, DL, VT, N1, N2.getOperand(1),
8760                                      &Flags), &Flags);
8761     }
8762 
8763     // (fma (fmul x, c1), c2, y) -> (fma x, c1*c2, y)
8764     if (N0.getOpcode() == ISD::FMUL &&
8765         isConstantFPBuildVectorOrConstantFP(N1) &&
8766         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1))) {
8767       return DAG.getNode(ISD::FMA, DL, VT,
8768                          N0.getOperand(0),
8769                          DAG.getNode(ISD::FMUL, DL, VT, N1, N0.getOperand(1),
8770                                      &Flags),
8771                          N2);
8772     }
8773   }
8774 
8775   // (fma x, 1, y) -> (fadd x, y)
8776   // (fma x, -1, y) -> (fadd (fneg x), y)
8777   if (N1CFP) {
8778     if (N1CFP->isExactlyValue(1.0))
8779       // TODO: The FMA node should have flags that propagate to this node.
8780       return DAG.getNode(ISD::FADD, DL, VT, N0, N2);
8781 
8782     if (N1CFP->isExactlyValue(-1.0) &&
8783         (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))) {
8784       SDValue RHSNeg = DAG.getNode(ISD::FNEG, DL, VT, N0);
8785       AddToWorklist(RHSNeg.getNode());
8786       // TODO: The FMA node should have flags that propagate to this node.
8787       return DAG.getNode(ISD::FADD, DL, VT, N2, RHSNeg);
8788     }
8789   }
8790 
8791   if (Options.UnsafeFPMath) {
8792     // (fma x, c, x) -> (fmul x, (c+1))
8793     if (N1CFP && N0 == N2) {
8794       return DAG.getNode(ISD::FMUL, DL, VT, N0,
8795                          DAG.getNode(ISD::FADD, DL, VT, N1,
8796                                      DAG.getConstantFP(1.0, DL, VT), &Flags),
8797                          &Flags);
8798     }
8799 
8800     // (fma x, c, (fneg x)) -> (fmul x, (c-1))
8801     if (N1CFP && N2.getOpcode() == ISD::FNEG && N2.getOperand(0) == N0) {
8802       return DAG.getNode(ISD::FMUL, DL, VT, N0,
8803                          DAG.getNode(ISD::FADD, DL, VT, N1,
8804                                      DAG.getConstantFP(-1.0, DL, VT), &Flags),
8805                          &Flags);
8806     }
8807   }
8808 
8809   return SDValue();
8810 }
8811 
8812 // Combine multiple FDIVs with the same divisor into multiple FMULs by the
8813 // reciprocal.
8814 // E.g., (a / D; b / D;) -> (recip = 1.0 / D; a * recip; b * recip)
8815 // Notice that this is not always beneficial. One reason is different target
8816 // may have different costs for FDIV and FMUL, so sometimes the cost of two
8817 // FDIVs may be lower than the cost of one FDIV and two FMULs. Another reason
8818 // is the critical path is increased from "one FDIV" to "one FDIV + one FMUL".
8819 SDValue DAGCombiner::combineRepeatedFPDivisors(SDNode *N) {
8820   bool UnsafeMath = DAG.getTarget().Options.UnsafeFPMath;
8821   const SDNodeFlags *Flags = N->getFlags();
8822   if (!UnsafeMath && !Flags->hasAllowReciprocal())
8823     return SDValue();
8824 
8825   // Skip if current node is a reciprocal.
8826   SDValue N0 = N->getOperand(0);
8827   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
8828   if (N0CFP && N0CFP->isExactlyValue(1.0))
8829     return SDValue();
8830 
8831   // Exit early if the target does not want this transform or if there can't
8832   // possibly be enough uses of the divisor to make the transform worthwhile.
8833   SDValue N1 = N->getOperand(1);
8834   unsigned MinUses = TLI.combineRepeatedFPDivisors();
8835   if (!MinUses || N1->use_size() < MinUses)
8836     return SDValue();
8837 
8838   // Find all FDIV users of the same divisor.
8839   // Use a set because duplicates may be present in the user list.
8840   SetVector<SDNode *> Users;
8841   for (auto *U : N1->uses()) {
8842     if (U->getOpcode() == ISD::FDIV && U->getOperand(1) == N1) {
8843       // This division is eligible for optimization only if global unsafe math
8844       // is enabled or if this division allows reciprocal formation.
8845       if (UnsafeMath || U->getFlags()->hasAllowReciprocal())
8846         Users.insert(U);
8847     }
8848   }
8849 
8850   // Now that we have the actual number of divisor uses, make sure it meets
8851   // the minimum threshold specified by the target.
8852   if (Users.size() < MinUses)
8853     return SDValue();
8854 
8855   EVT VT = N->getValueType(0);
8856   SDLoc DL(N);
8857   SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
8858   SDValue Reciprocal = DAG.getNode(ISD::FDIV, DL, VT, FPOne, N1, Flags);
8859 
8860   // Dividend / Divisor -> Dividend * Reciprocal
8861   for (auto *U : Users) {
8862     SDValue Dividend = U->getOperand(0);
8863     if (Dividend != FPOne) {
8864       SDValue NewNode = DAG.getNode(ISD::FMUL, SDLoc(U), VT, Dividend,
8865                                     Reciprocal, Flags);
8866       CombineTo(U, NewNode);
8867     } else if (U != Reciprocal.getNode()) {
8868       // In the absence of fast-math-flags, this user node is always the
8869       // same node as Reciprocal, but with FMF they may be different nodes.
8870       CombineTo(U, Reciprocal);
8871     }
8872   }
8873   return SDValue(N, 0);  // N was replaced.
8874 }
8875 
8876 SDValue DAGCombiner::visitFDIV(SDNode *N) {
8877   SDValue N0 = N->getOperand(0);
8878   SDValue N1 = N->getOperand(1);
8879   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
8880   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
8881   EVT VT = N->getValueType(0);
8882   SDLoc DL(N);
8883   const TargetOptions &Options = DAG.getTarget().Options;
8884   SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8885 
8886   // fold vector ops
8887   if (VT.isVector())
8888     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8889       return FoldedVOp;
8890 
8891   // fold (fdiv c1, c2) -> c1/c2
8892   if (N0CFP && N1CFP)
8893     return DAG.getNode(ISD::FDIV, SDLoc(N), VT, N0, N1, Flags);
8894 
8895   if (Options.UnsafeFPMath) {
8896     // fold (fdiv X, c2) -> fmul X, 1/c2 if losing precision is acceptable.
8897     if (N1CFP) {
8898       // Compute the reciprocal 1.0 / c2.
8899       const APFloat &N1APF = N1CFP->getValueAPF();
8900       APFloat Recip(N1APF.getSemantics(), 1); // 1.0
8901       APFloat::opStatus st = Recip.divide(N1APF, APFloat::rmNearestTiesToEven);
8902       // Only do the transform if the reciprocal is a legal fp immediate that
8903       // isn't too nasty (eg NaN, denormal, ...).
8904       if ((st == APFloat::opOK || st == APFloat::opInexact) && // Not too nasty
8905           (!LegalOperations ||
8906            // FIXME: custom lowering of ConstantFP might fail (see e.g. ARM
8907            // backend)... we should handle this gracefully after Legalize.
8908            // TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT) ||
8909            TLI.isOperationLegal(llvm::ISD::ConstantFP, VT) ||
8910            TLI.isFPImmLegal(Recip, VT)))
8911         return DAG.getNode(ISD::FMUL, DL, VT, N0,
8912                            DAG.getConstantFP(Recip, DL, VT), Flags);
8913     }
8914 
8915     // If this FDIV is part of a reciprocal square root, it may be folded
8916     // into a target-specific square root estimate instruction.
8917     if (N1.getOpcode() == ISD::FSQRT) {
8918       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0), Flags)) {
8919         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8920       }
8921     } else if (N1.getOpcode() == ISD::FP_EXTEND &&
8922                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
8923       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
8924                                           Flags)) {
8925         RV = DAG.getNode(ISD::FP_EXTEND, SDLoc(N1), VT, RV);
8926         AddToWorklist(RV.getNode());
8927         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8928       }
8929     } else if (N1.getOpcode() == ISD::FP_ROUND &&
8930                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
8931       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
8932                                           Flags)) {
8933         RV = DAG.getNode(ISD::FP_ROUND, SDLoc(N1), VT, RV, N1.getOperand(1));
8934         AddToWorklist(RV.getNode());
8935         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8936       }
8937     } else if (N1.getOpcode() == ISD::FMUL) {
8938       // Look through an FMUL. Even though this won't remove the FDIV directly,
8939       // it's still worthwhile to get rid of the FSQRT if possible.
8940       SDValue SqrtOp;
8941       SDValue OtherOp;
8942       if (N1.getOperand(0).getOpcode() == ISD::FSQRT) {
8943         SqrtOp = N1.getOperand(0);
8944         OtherOp = N1.getOperand(1);
8945       } else if (N1.getOperand(1).getOpcode() == ISD::FSQRT) {
8946         SqrtOp = N1.getOperand(1);
8947         OtherOp = N1.getOperand(0);
8948       }
8949       if (SqrtOp.getNode()) {
8950         // We found a FSQRT, so try to make this fold:
8951         // x / (y * sqrt(z)) -> x * (rsqrt(z) / y)
8952         if (SDValue RV = buildRsqrtEstimate(SqrtOp.getOperand(0), Flags)) {
8953           RV = DAG.getNode(ISD::FDIV, SDLoc(N1), VT, RV, OtherOp, Flags);
8954           AddToWorklist(RV.getNode());
8955           return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8956         }
8957       }
8958     }
8959 
8960     // Fold into a reciprocal estimate and multiply instead of a real divide.
8961     if (SDValue RV = BuildReciprocalEstimate(N1, Flags)) {
8962       AddToWorklist(RV.getNode());
8963       return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
8964     }
8965   }
8966 
8967   // (fdiv (fneg X), (fneg Y)) -> (fdiv X, Y)
8968   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) {
8969     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) {
8970       // Both can be negated for free, check to see if at least one is cheaper
8971       // negated.
8972       if (LHSNeg == 2 || RHSNeg == 2)
8973         return DAG.getNode(ISD::FDIV, SDLoc(N), VT,
8974                            GetNegatedExpression(N0, DAG, LegalOperations),
8975                            GetNegatedExpression(N1, DAG, LegalOperations),
8976                            Flags);
8977     }
8978   }
8979 
8980   if (SDValue CombineRepeatedDivisors = combineRepeatedFPDivisors(N))
8981     return CombineRepeatedDivisors;
8982 
8983   return SDValue();
8984 }
8985 
8986 SDValue DAGCombiner::visitFREM(SDNode *N) {
8987   SDValue N0 = N->getOperand(0);
8988   SDValue N1 = N->getOperand(1);
8989   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
8990   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
8991   EVT VT = N->getValueType(0);
8992 
8993   // fold (frem c1, c2) -> fmod(c1,c2)
8994   if (N0CFP && N1CFP)
8995     return DAG.getNode(ISD::FREM, SDLoc(N), VT, N0, N1,
8996                        &cast<BinaryWithFlagsSDNode>(N)->Flags);
8997 
8998   return SDValue();
8999 }
9000 
9001 SDValue DAGCombiner::visitFSQRT(SDNode *N) {
9002   if (!DAG.getTarget().Options.UnsafeFPMath)
9003     return SDValue();
9004 
9005   SDValue N0 = N->getOperand(0);
9006   if (TLI.isFsqrtCheap(N0, DAG))
9007     return SDValue();
9008 
9009   // TODO: FSQRT nodes should have flags that propagate to the created nodes.
9010   // For now, create a Flags object for use with all unsafe math transforms.
9011   SDNodeFlags Flags;
9012   Flags.setUnsafeAlgebra(true);
9013   return buildSqrtEstimate(N0, &Flags);
9014 }
9015 
9016 /// copysign(x, fp_extend(y)) -> copysign(x, y)
9017 /// copysign(x, fp_round(y)) -> copysign(x, y)
9018 static inline bool CanCombineFCOPYSIGN_EXTEND_ROUND(SDNode *N) {
9019   SDValue N1 = N->getOperand(1);
9020   if ((N1.getOpcode() == ISD::FP_EXTEND ||
9021        N1.getOpcode() == ISD::FP_ROUND)) {
9022     // Do not optimize out type conversion of f128 type yet.
9023     // For some targets like x86_64, configuration is changed to keep one f128
9024     // value in one SSE register, but instruction selection cannot handle
9025     // FCOPYSIGN on SSE registers yet.
9026     EVT N1VT = N1->getValueType(0);
9027     EVT N1Op0VT = N1->getOperand(0)->getValueType(0);
9028     return (N1VT == N1Op0VT || N1Op0VT != MVT::f128);
9029   }
9030   return false;
9031 }
9032 
9033 SDValue DAGCombiner::visitFCOPYSIGN(SDNode *N) {
9034   SDValue N0 = N->getOperand(0);
9035   SDValue N1 = N->getOperand(1);
9036   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9037   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
9038   EVT VT = N->getValueType(0);
9039 
9040   if (N0CFP && N1CFP) // Constant fold
9041     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1);
9042 
9043   if (N1CFP) {
9044     const APFloat &V = N1CFP->getValueAPF();
9045     // copysign(x, c1) -> fabs(x)       iff ispos(c1)
9046     // copysign(x, c1) -> fneg(fabs(x)) iff isneg(c1)
9047     if (!V.isNegative()) {
9048       if (!LegalOperations || TLI.isOperationLegal(ISD::FABS, VT))
9049         return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9050     } else {
9051       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
9052         return DAG.getNode(ISD::FNEG, SDLoc(N), VT,
9053                            DAG.getNode(ISD::FABS, SDLoc(N0), VT, N0));
9054     }
9055   }
9056 
9057   // copysign(fabs(x), y) -> copysign(x, y)
9058   // copysign(fneg(x), y) -> copysign(x, y)
9059   // copysign(copysign(x,z), y) -> copysign(x, y)
9060   if (N0.getOpcode() == ISD::FABS || N0.getOpcode() == ISD::FNEG ||
9061       N0.getOpcode() == ISD::FCOPYSIGN)
9062     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0.getOperand(0), N1);
9063 
9064   // copysign(x, abs(y)) -> abs(x)
9065   if (N1.getOpcode() == ISD::FABS)
9066     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9067 
9068   // copysign(x, copysign(y,z)) -> copysign(x, z)
9069   if (N1.getOpcode() == ISD::FCOPYSIGN)
9070     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(1));
9071 
9072   // copysign(x, fp_extend(y)) -> copysign(x, y)
9073   // copysign(x, fp_round(y)) -> copysign(x, y)
9074   if (CanCombineFCOPYSIGN_EXTEND_ROUND(N))
9075     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(0));
9076 
9077   return SDValue();
9078 }
9079 
9080 SDValue DAGCombiner::visitSINT_TO_FP(SDNode *N) {
9081   SDValue N0 = N->getOperand(0);
9082   EVT VT = N->getValueType(0);
9083   EVT OpVT = N0.getValueType();
9084 
9085   // fold (sint_to_fp c1) -> c1fp
9086   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
9087       // ...but only if the target supports immediate floating-point values
9088       (!LegalOperations ||
9089        TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT)))
9090     return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
9091 
9092   // If the input is a legal type, and SINT_TO_FP is not legal on this target,
9093   // but UINT_TO_FP is legal on this target, try to convert.
9094   if (!TLI.isOperationLegalOrCustom(ISD::SINT_TO_FP, OpVT) &&
9095       TLI.isOperationLegalOrCustom(ISD::UINT_TO_FP, OpVT)) {
9096     // If the sign bit is known to be zero, we can change this to UINT_TO_FP.
9097     if (DAG.SignBitIsZero(N0))
9098       return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
9099   }
9100 
9101   // The next optimizations are desirable only if SELECT_CC can be lowered.
9102   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
9103     // fold (sint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
9104     if (N0.getOpcode() == ISD::SETCC && N0.getValueType() == MVT::i1 &&
9105         !VT.isVector() &&
9106         (!LegalOperations ||
9107          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9108       SDLoc DL(N);
9109       SDValue Ops[] =
9110         { N0.getOperand(0), N0.getOperand(1),
9111           DAG.getConstantFP(-1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9112           N0.getOperand(2) };
9113       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9114     }
9115 
9116     // fold (sint_to_fp (zext (setcc x, y, cc))) ->
9117     //      (select_cc x, y, 1.0, 0.0,, cc)
9118     if (N0.getOpcode() == ISD::ZERO_EXTEND &&
9119         N0.getOperand(0).getOpcode() == ISD::SETCC &&!VT.isVector() &&
9120         (!LegalOperations ||
9121          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9122       SDLoc DL(N);
9123       SDValue Ops[] =
9124         { N0.getOperand(0).getOperand(0), N0.getOperand(0).getOperand(1),
9125           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9126           N0.getOperand(0).getOperand(2) };
9127       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9128     }
9129   }
9130 
9131   return SDValue();
9132 }
9133 
9134 SDValue DAGCombiner::visitUINT_TO_FP(SDNode *N) {
9135   SDValue N0 = N->getOperand(0);
9136   EVT VT = N->getValueType(0);
9137   EVT OpVT = N0.getValueType();
9138 
9139   // fold (uint_to_fp c1) -> c1fp
9140   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
9141       // ...but only if the target supports immediate floating-point values
9142       (!LegalOperations ||
9143        TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT)))
9144     return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
9145 
9146   // If the input is a legal type, and UINT_TO_FP is not legal on this target,
9147   // but SINT_TO_FP is legal on this target, try to convert.
9148   if (!TLI.isOperationLegalOrCustom(ISD::UINT_TO_FP, OpVT) &&
9149       TLI.isOperationLegalOrCustom(ISD::SINT_TO_FP, OpVT)) {
9150     // If the sign bit is known to be zero, we can change this to SINT_TO_FP.
9151     if (DAG.SignBitIsZero(N0))
9152       return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
9153   }
9154 
9155   // The next optimizations are desirable only if SELECT_CC can be lowered.
9156   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
9157     // fold (uint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
9158 
9159     if (N0.getOpcode() == ISD::SETCC && !VT.isVector() &&
9160         (!LegalOperations ||
9161          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9162       SDLoc DL(N);
9163       SDValue Ops[] =
9164         { N0.getOperand(0), N0.getOperand(1),
9165           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9166           N0.getOperand(2) };
9167       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9168     }
9169   }
9170 
9171   return SDValue();
9172 }
9173 
9174 // Fold (fp_to_{s/u}int ({s/u}int_to_fpx)) -> zext x, sext x, trunc x, or x
9175 static SDValue FoldIntToFPToInt(SDNode *N, SelectionDAG &DAG) {
9176   SDValue N0 = N->getOperand(0);
9177   EVT VT = N->getValueType(0);
9178 
9179   if (N0.getOpcode() != ISD::UINT_TO_FP && N0.getOpcode() != ISD::SINT_TO_FP)
9180     return SDValue();
9181 
9182   SDValue Src = N0.getOperand(0);
9183   EVT SrcVT = Src.getValueType();
9184   bool IsInputSigned = N0.getOpcode() == ISD::SINT_TO_FP;
9185   bool IsOutputSigned = N->getOpcode() == ISD::FP_TO_SINT;
9186 
9187   // We can safely assume the conversion won't overflow the output range,
9188   // because (for example) (uint8_t)18293.f is undefined behavior.
9189 
9190   // Since we can assume the conversion won't overflow, our decision as to
9191   // whether the input will fit in the float should depend on the minimum
9192   // of the input range and output range.
9193 
9194   // This means this is also safe for a signed input and unsigned output, since
9195   // a negative input would lead to undefined behavior.
9196   unsigned InputSize = (int)SrcVT.getScalarSizeInBits() - IsInputSigned;
9197   unsigned OutputSize = (int)VT.getScalarSizeInBits() - IsOutputSigned;
9198   unsigned ActualSize = std::min(InputSize, OutputSize);
9199   const fltSemantics &sem = DAG.EVTToAPFloatSemantics(N0.getValueType());
9200 
9201   // We can only fold away the float conversion if the input range can be
9202   // represented exactly in the float range.
9203   if (APFloat::semanticsPrecision(sem) >= ActualSize) {
9204     if (VT.getScalarSizeInBits() > SrcVT.getScalarSizeInBits()) {
9205       unsigned ExtOp = IsInputSigned && IsOutputSigned ? ISD::SIGN_EXTEND
9206                                                        : ISD::ZERO_EXTEND;
9207       return DAG.getNode(ExtOp, SDLoc(N), VT, Src);
9208     }
9209     if (VT.getScalarSizeInBits() < SrcVT.getScalarSizeInBits())
9210       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Src);
9211     return DAG.getBitcast(VT, Src);
9212   }
9213   return SDValue();
9214 }
9215 
9216 SDValue DAGCombiner::visitFP_TO_SINT(SDNode *N) {
9217   SDValue N0 = N->getOperand(0);
9218   EVT VT = N->getValueType(0);
9219 
9220   // fold (fp_to_sint c1fp) -> c1
9221   if (isConstantFPBuildVectorOrConstantFP(N0))
9222     return DAG.getNode(ISD::FP_TO_SINT, SDLoc(N), VT, N0);
9223 
9224   return FoldIntToFPToInt(N, DAG);
9225 }
9226 
9227 SDValue DAGCombiner::visitFP_TO_UINT(SDNode *N) {
9228   SDValue N0 = N->getOperand(0);
9229   EVT VT = N->getValueType(0);
9230 
9231   // fold (fp_to_uint c1fp) -> c1
9232   if (isConstantFPBuildVectorOrConstantFP(N0))
9233     return DAG.getNode(ISD::FP_TO_UINT, SDLoc(N), VT, N0);
9234 
9235   return FoldIntToFPToInt(N, DAG);
9236 }
9237 
9238 SDValue DAGCombiner::visitFP_ROUND(SDNode *N) {
9239   SDValue N0 = N->getOperand(0);
9240   SDValue N1 = N->getOperand(1);
9241   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9242   EVT VT = N->getValueType(0);
9243 
9244   // fold (fp_round c1fp) -> c1fp
9245   if (N0CFP)
9246     return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT, N0, N1);
9247 
9248   // fold (fp_round (fp_extend x)) -> x
9249   if (N0.getOpcode() == ISD::FP_EXTEND && VT == N0.getOperand(0).getValueType())
9250     return N0.getOperand(0);
9251 
9252   // fold (fp_round (fp_round x)) -> (fp_round x)
9253   if (N0.getOpcode() == ISD::FP_ROUND) {
9254     const bool NIsTrunc = N->getConstantOperandVal(1) == 1;
9255     const bool N0IsTrunc = N0.getNode()->getConstantOperandVal(1) == 1;
9256 
9257     // Skip this folding if it results in an fp_round from f80 to f16.
9258     //
9259     // f80 to f16 always generates an expensive (and as yet, unimplemented)
9260     // libcall to __truncxfhf2 instead of selecting native f16 conversion
9261     // instructions from f32 or f64.  Moreover, the first (value-preserving)
9262     // fp_round from f80 to either f32 or f64 may become a NOP in platforms like
9263     // x86.
9264     if (N0.getOperand(0).getValueType() == MVT::f80 && VT == MVT::f16)
9265       return SDValue();
9266 
9267     // If the first fp_round isn't a value preserving truncation, it might
9268     // introduce a tie in the second fp_round, that wouldn't occur in the
9269     // single-step fp_round we want to fold to.
9270     // In other words, double rounding isn't the same as rounding.
9271     // Also, this is a value preserving truncation iff both fp_round's are.
9272     if (DAG.getTarget().Options.UnsafeFPMath || N0IsTrunc) {
9273       SDLoc DL(N);
9274       return DAG.getNode(ISD::FP_ROUND, DL, VT, N0.getOperand(0),
9275                          DAG.getIntPtrConstant(NIsTrunc && N0IsTrunc, DL));
9276     }
9277   }
9278 
9279   // fold (fp_round (copysign X, Y)) -> (copysign (fp_round X), Y)
9280   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse()) {
9281     SDValue Tmp = DAG.getNode(ISD::FP_ROUND, SDLoc(N0), VT,
9282                               N0.getOperand(0), N1);
9283     AddToWorklist(Tmp.getNode());
9284     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT,
9285                        Tmp, N0.getOperand(1));
9286   }
9287 
9288   return SDValue();
9289 }
9290 
9291 SDValue DAGCombiner::visitFP_ROUND_INREG(SDNode *N) {
9292   SDValue N0 = N->getOperand(0);
9293   EVT VT = N->getValueType(0);
9294   EVT EVT = cast<VTSDNode>(N->getOperand(1))->getVT();
9295   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9296 
9297   // fold (fp_round_inreg c1fp) -> c1fp
9298   if (N0CFP && isTypeLegal(EVT)) {
9299     SDLoc DL(N);
9300     SDValue Round = DAG.getConstantFP(*N0CFP->getConstantFPValue(), DL, EVT);
9301     return DAG.getNode(ISD::FP_EXTEND, DL, VT, Round);
9302   }
9303 
9304   return SDValue();
9305 }
9306 
9307 SDValue DAGCombiner::visitFP_EXTEND(SDNode *N) {
9308   SDValue N0 = N->getOperand(0);
9309   EVT VT = N->getValueType(0);
9310 
9311   // If this is fp_round(fpextend), don't fold it, allow ourselves to be folded.
9312   if (N->hasOneUse() &&
9313       N->use_begin()->getOpcode() == ISD::FP_ROUND)
9314     return SDValue();
9315 
9316   // fold (fp_extend c1fp) -> c1fp
9317   if (isConstantFPBuildVectorOrConstantFP(N0))
9318     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, N0);
9319 
9320   // fold (fp_extend (fp16_to_fp op)) -> (fp16_to_fp op)
9321   if (N0.getOpcode() == ISD::FP16_TO_FP &&
9322       TLI.getOperationAction(ISD::FP16_TO_FP, VT) == TargetLowering::Legal)
9323     return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), VT, N0.getOperand(0));
9324 
9325   // Turn fp_extend(fp_round(X, 1)) -> x since the fp_round doesn't affect the
9326   // value of X.
9327   if (N0.getOpcode() == ISD::FP_ROUND
9328       && N0.getNode()->getConstantOperandVal(1) == 1) {
9329     SDValue In = N0.getOperand(0);
9330     if (In.getValueType() == VT) return In;
9331     if (VT.bitsLT(In.getValueType()))
9332       return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT,
9333                          In, N0.getOperand(1));
9334     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, In);
9335   }
9336 
9337   // fold (fpext (load x)) -> (fpext (fptrunc (extload x)))
9338   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
9339        TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
9340     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9341     SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
9342                                      LN0->getChain(),
9343                                      LN0->getBasePtr(), N0.getValueType(),
9344                                      LN0->getMemOperand());
9345     CombineTo(N, ExtLoad);
9346     CombineTo(N0.getNode(),
9347               DAG.getNode(ISD::FP_ROUND, SDLoc(N0),
9348                           N0.getValueType(), ExtLoad,
9349                           DAG.getIntPtrConstant(1, SDLoc(N0))),
9350               ExtLoad.getValue(1));
9351     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9352   }
9353 
9354   return SDValue();
9355 }
9356 
9357 SDValue DAGCombiner::visitFCEIL(SDNode *N) {
9358   SDValue N0 = N->getOperand(0);
9359   EVT VT = N->getValueType(0);
9360 
9361   // fold (fceil c1) -> fceil(c1)
9362   if (isConstantFPBuildVectorOrConstantFP(N0))
9363     return DAG.getNode(ISD::FCEIL, SDLoc(N), VT, N0);
9364 
9365   return SDValue();
9366 }
9367 
9368 SDValue DAGCombiner::visitFTRUNC(SDNode *N) {
9369   SDValue N0 = N->getOperand(0);
9370   EVT VT = N->getValueType(0);
9371 
9372   // fold (ftrunc c1) -> ftrunc(c1)
9373   if (isConstantFPBuildVectorOrConstantFP(N0))
9374     return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0);
9375 
9376   return SDValue();
9377 }
9378 
9379 SDValue DAGCombiner::visitFFLOOR(SDNode *N) {
9380   SDValue N0 = N->getOperand(0);
9381   EVT VT = N->getValueType(0);
9382 
9383   // fold (ffloor c1) -> ffloor(c1)
9384   if (isConstantFPBuildVectorOrConstantFP(N0))
9385     return DAG.getNode(ISD::FFLOOR, SDLoc(N), VT, N0);
9386 
9387   return SDValue();
9388 }
9389 
9390 // FIXME: FNEG and FABS have a lot in common; refactor.
9391 SDValue DAGCombiner::visitFNEG(SDNode *N) {
9392   SDValue N0 = N->getOperand(0);
9393   EVT VT = N->getValueType(0);
9394 
9395   // Constant fold FNEG.
9396   if (isConstantFPBuildVectorOrConstantFP(N0))
9397     return DAG.getNode(ISD::FNEG, SDLoc(N), VT, N0);
9398 
9399   if (isNegatibleForFree(N0, LegalOperations, DAG.getTargetLoweringInfo(),
9400                          &DAG.getTarget().Options))
9401     return GetNegatedExpression(N0, DAG, LegalOperations);
9402 
9403   // Transform fneg(bitconvert(x)) -> bitconvert(x ^ sign) to avoid loading
9404   // constant pool values.
9405   if (!TLI.isFNegFree(VT) &&
9406       N0.getOpcode() == ISD::BITCAST &&
9407       N0.getNode()->hasOneUse()) {
9408     SDValue Int = N0.getOperand(0);
9409     EVT IntVT = Int.getValueType();
9410     if (IntVT.isInteger() && !IntVT.isVector()) {
9411       APInt SignMask;
9412       if (N0.getValueType().isVector()) {
9413         // For a vector, get a mask such as 0x80... per scalar element
9414         // and splat it.
9415         SignMask = APInt::getSignBit(N0.getScalarValueSizeInBits());
9416         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
9417       } else {
9418         // For a scalar, just generate 0x80...
9419         SignMask = APInt::getSignBit(IntVT.getSizeInBits());
9420       }
9421       SDLoc DL0(N0);
9422       Int = DAG.getNode(ISD::XOR, DL0, IntVT, Int,
9423                         DAG.getConstant(SignMask, DL0, IntVT));
9424       AddToWorklist(Int.getNode());
9425       return DAG.getBitcast(VT, Int);
9426     }
9427   }
9428 
9429   // (fneg (fmul c, x)) -> (fmul -c, x)
9430   if (N0.getOpcode() == ISD::FMUL &&
9431       (N0.getNode()->hasOneUse() || !TLI.isFNegFree(VT))) {
9432     ConstantFPSDNode *CFP1 = dyn_cast<ConstantFPSDNode>(N0.getOperand(1));
9433     if (CFP1) {
9434       APFloat CVal = CFP1->getValueAPF();
9435       CVal.changeSign();
9436       if (Level >= AfterLegalizeDAG &&
9437           (TLI.isFPImmLegal(CVal, VT) ||
9438            TLI.isOperationLegal(ISD::ConstantFP, VT)))
9439         return DAG.getNode(ISD::FMUL, SDLoc(N), VT, N0.getOperand(0),
9440                            DAG.getNode(ISD::FNEG, SDLoc(N), VT,
9441                                        N0.getOperand(1)),
9442                            &cast<BinaryWithFlagsSDNode>(N0)->Flags);
9443     }
9444   }
9445 
9446   return SDValue();
9447 }
9448 
9449 SDValue DAGCombiner::visitFMINNUM(SDNode *N) {
9450   SDValue N0 = N->getOperand(0);
9451   SDValue N1 = N->getOperand(1);
9452   EVT VT = N->getValueType(0);
9453   const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9454   const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9455 
9456   if (N0CFP && N1CFP) {
9457     const APFloat &C0 = N0CFP->getValueAPF();
9458     const APFloat &C1 = N1CFP->getValueAPF();
9459     return DAG.getConstantFP(minnum(C0, C1), SDLoc(N), VT);
9460   }
9461 
9462   // Canonicalize to constant on RHS.
9463   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9464      !isConstantFPBuildVectorOrConstantFP(N1))
9465     return DAG.getNode(ISD::FMINNUM, SDLoc(N), VT, N1, N0);
9466 
9467   return SDValue();
9468 }
9469 
9470 SDValue DAGCombiner::visitFMAXNUM(SDNode *N) {
9471   SDValue N0 = N->getOperand(0);
9472   SDValue N1 = N->getOperand(1);
9473   EVT VT = N->getValueType(0);
9474   const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9475   const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9476 
9477   if (N0CFP && N1CFP) {
9478     const APFloat &C0 = N0CFP->getValueAPF();
9479     const APFloat &C1 = N1CFP->getValueAPF();
9480     return DAG.getConstantFP(maxnum(C0, C1), SDLoc(N), VT);
9481   }
9482 
9483   // Canonicalize to constant on RHS.
9484   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9485      !isConstantFPBuildVectorOrConstantFP(N1))
9486     return DAG.getNode(ISD::FMAXNUM, SDLoc(N), VT, N1, N0);
9487 
9488   return SDValue();
9489 }
9490 
9491 SDValue DAGCombiner::visitFABS(SDNode *N) {
9492   SDValue N0 = N->getOperand(0);
9493   EVT VT = N->getValueType(0);
9494 
9495   // fold (fabs c1) -> fabs(c1)
9496   if (isConstantFPBuildVectorOrConstantFP(N0))
9497     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9498 
9499   // fold (fabs (fabs x)) -> (fabs x)
9500   if (N0.getOpcode() == ISD::FABS)
9501     return N->getOperand(0);
9502 
9503   // fold (fabs (fneg x)) -> (fabs x)
9504   // fold (fabs (fcopysign x, y)) -> (fabs x)
9505   if (N0.getOpcode() == ISD::FNEG || N0.getOpcode() == ISD::FCOPYSIGN)
9506     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0.getOperand(0));
9507 
9508   // Transform fabs(bitconvert(x)) -> bitconvert(x & ~sign) to avoid loading
9509   // constant pool values.
9510   if (!TLI.isFAbsFree(VT) &&
9511       N0.getOpcode() == ISD::BITCAST &&
9512       N0.getNode()->hasOneUse()) {
9513     SDValue Int = N0.getOperand(0);
9514     EVT IntVT = Int.getValueType();
9515     if (IntVT.isInteger() && !IntVT.isVector()) {
9516       APInt SignMask;
9517       if (N0.getValueType().isVector()) {
9518         // For a vector, get a mask such as 0x7f... per scalar element
9519         // and splat it.
9520         SignMask = ~APInt::getSignBit(N0.getScalarValueSizeInBits());
9521         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
9522       } else {
9523         // For a scalar, just generate 0x7f...
9524         SignMask = ~APInt::getSignBit(IntVT.getSizeInBits());
9525       }
9526       SDLoc DL(N0);
9527       Int = DAG.getNode(ISD::AND, DL, IntVT, Int,
9528                         DAG.getConstant(SignMask, DL, IntVT));
9529       AddToWorklist(Int.getNode());
9530       return DAG.getBitcast(N->getValueType(0), Int);
9531     }
9532   }
9533 
9534   return SDValue();
9535 }
9536 
9537 SDValue DAGCombiner::visitBRCOND(SDNode *N) {
9538   SDValue Chain = N->getOperand(0);
9539   SDValue N1 = N->getOperand(1);
9540   SDValue N2 = N->getOperand(2);
9541 
9542   // If N is a constant we could fold this into a fallthrough or unconditional
9543   // branch. However that doesn't happen very often in normal code, because
9544   // Instcombine/SimplifyCFG should have handled the available opportunities.
9545   // If we did this folding here, it would be necessary to update the
9546   // MachineBasicBlock CFG, which is awkward.
9547 
9548   // fold a brcond with a setcc condition into a BR_CC node if BR_CC is legal
9549   // on the target.
9550   if (N1.getOpcode() == ISD::SETCC &&
9551       TLI.isOperationLegalOrCustom(ISD::BR_CC,
9552                                    N1.getOperand(0).getValueType())) {
9553     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
9554                        Chain, N1.getOperand(2),
9555                        N1.getOperand(0), N1.getOperand(1), N2);
9556   }
9557 
9558   if ((N1.hasOneUse() && N1.getOpcode() == ISD::SRL) ||
9559       ((N1.getOpcode() == ISD::TRUNCATE && N1.hasOneUse()) &&
9560        (N1.getOperand(0).hasOneUse() &&
9561         N1.getOperand(0).getOpcode() == ISD::SRL))) {
9562     SDNode *Trunc = nullptr;
9563     if (N1.getOpcode() == ISD::TRUNCATE) {
9564       // Look pass the truncate.
9565       Trunc = N1.getNode();
9566       N1 = N1.getOperand(0);
9567     }
9568 
9569     // Match this pattern so that we can generate simpler code:
9570     //
9571     //   %a = ...
9572     //   %b = and i32 %a, 2
9573     //   %c = srl i32 %b, 1
9574     //   brcond i32 %c ...
9575     //
9576     // into
9577     //
9578     //   %a = ...
9579     //   %b = and i32 %a, 2
9580     //   %c = setcc eq %b, 0
9581     //   brcond %c ...
9582     //
9583     // This applies only when the AND constant value has one bit set and the
9584     // SRL constant is equal to the log2 of the AND constant. The back-end is
9585     // smart enough to convert the result into a TEST/JMP sequence.
9586     SDValue Op0 = N1.getOperand(0);
9587     SDValue Op1 = N1.getOperand(1);
9588 
9589     if (Op0.getOpcode() == ISD::AND &&
9590         Op1.getOpcode() == ISD::Constant) {
9591       SDValue AndOp1 = Op0.getOperand(1);
9592 
9593       if (AndOp1.getOpcode() == ISD::Constant) {
9594         const APInt &AndConst = cast<ConstantSDNode>(AndOp1)->getAPIntValue();
9595 
9596         if (AndConst.isPowerOf2() &&
9597             cast<ConstantSDNode>(Op1)->getAPIntValue()==AndConst.logBase2()) {
9598           SDLoc DL(N);
9599           SDValue SetCC =
9600             DAG.getSetCC(DL,
9601                          getSetCCResultType(Op0.getValueType()),
9602                          Op0, DAG.getConstant(0, DL, Op0.getValueType()),
9603                          ISD::SETNE);
9604 
9605           SDValue NewBRCond = DAG.getNode(ISD::BRCOND, DL,
9606                                           MVT::Other, Chain, SetCC, N2);
9607           // Don't add the new BRCond into the worklist or else SimplifySelectCC
9608           // will convert it back to (X & C1) >> C2.
9609           CombineTo(N, NewBRCond, false);
9610           // Truncate is dead.
9611           if (Trunc)
9612             deleteAndRecombine(Trunc);
9613           // Replace the uses of SRL with SETCC
9614           WorklistRemover DeadNodes(*this);
9615           DAG.ReplaceAllUsesOfValueWith(N1, SetCC);
9616           deleteAndRecombine(N1.getNode());
9617           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9618         }
9619       }
9620     }
9621 
9622     if (Trunc)
9623       // Restore N1 if the above transformation doesn't match.
9624       N1 = N->getOperand(1);
9625   }
9626 
9627   // Transform br(xor(x, y)) -> br(x != y)
9628   // Transform br(xor(xor(x,y), 1)) -> br (x == y)
9629   if (N1.hasOneUse() && N1.getOpcode() == ISD::XOR) {
9630     SDNode *TheXor = N1.getNode();
9631     SDValue Op0 = TheXor->getOperand(0);
9632     SDValue Op1 = TheXor->getOperand(1);
9633     if (Op0.getOpcode() == Op1.getOpcode()) {
9634       // Avoid missing important xor optimizations.
9635       if (SDValue Tmp = visitXOR(TheXor)) {
9636         if (Tmp.getNode() != TheXor) {
9637           DEBUG(dbgs() << "\nReplacing.8 ";
9638                 TheXor->dump(&DAG);
9639                 dbgs() << "\nWith: ";
9640                 Tmp.getNode()->dump(&DAG);
9641                 dbgs() << '\n');
9642           WorklistRemover DeadNodes(*this);
9643           DAG.ReplaceAllUsesOfValueWith(N1, Tmp);
9644           deleteAndRecombine(TheXor);
9645           return DAG.getNode(ISD::BRCOND, SDLoc(N),
9646                              MVT::Other, Chain, Tmp, N2);
9647         }
9648 
9649         // visitXOR has changed XOR's operands or replaced the XOR completely,
9650         // bail out.
9651         return SDValue(N, 0);
9652       }
9653     }
9654 
9655     if (Op0.getOpcode() != ISD::SETCC && Op1.getOpcode() != ISD::SETCC) {
9656       bool Equal = false;
9657       if (isOneConstant(Op0) && Op0.hasOneUse() &&
9658           Op0.getOpcode() == ISD::XOR) {
9659         TheXor = Op0.getNode();
9660         Equal = true;
9661       }
9662 
9663       EVT SetCCVT = N1.getValueType();
9664       if (LegalTypes)
9665         SetCCVT = getSetCCResultType(SetCCVT);
9666       SDValue SetCC = DAG.getSetCC(SDLoc(TheXor),
9667                                    SetCCVT,
9668                                    Op0, Op1,
9669                                    Equal ? ISD::SETEQ : ISD::SETNE);
9670       // Replace the uses of XOR with SETCC
9671       WorklistRemover DeadNodes(*this);
9672       DAG.ReplaceAllUsesOfValueWith(N1, SetCC);
9673       deleteAndRecombine(N1.getNode());
9674       return DAG.getNode(ISD::BRCOND, SDLoc(N),
9675                          MVT::Other, Chain, SetCC, N2);
9676     }
9677   }
9678 
9679   return SDValue();
9680 }
9681 
9682 // Operand List for BR_CC: Chain, CondCC, CondLHS, CondRHS, DestBB.
9683 //
9684 SDValue DAGCombiner::visitBR_CC(SDNode *N) {
9685   CondCodeSDNode *CC = cast<CondCodeSDNode>(N->getOperand(1));
9686   SDValue CondLHS = N->getOperand(2), CondRHS = N->getOperand(3);
9687 
9688   // If N is a constant we could fold this into a fallthrough or unconditional
9689   // branch. However that doesn't happen very often in normal code, because
9690   // Instcombine/SimplifyCFG should have handled the available opportunities.
9691   // If we did this folding here, it would be necessary to update the
9692   // MachineBasicBlock CFG, which is awkward.
9693 
9694   // Use SimplifySetCC to simplify SETCC's.
9695   SDValue Simp = SimplifySetCC(getSetCCResultType(CondLHS.getValueType()),
9696                                CondLHS, CondRHS, CC->get(), SDLoc(N),
9697                                false);
9698   if (Simp.getNode()) AddToWorklist(Simp.getNode());
9699 
9700   // fold to a simpler setcc
9701   if (Simp.getNode() && Simp.getOpcode() == ISD::SETCC)
9702     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
9703                        N->getOperand(0), Simp.getOperand(2),
9704                        Simp.getOperand(0), Simp.getOperand(1),
9705                        N->getOperand(4));
9706 
9707   return SDValue();
9708 }
9709 
9710 /// Return true if 'Use' is a load or a store that uses N as its base pointer
9711 /// and that N may be folded in the load / store addressing mode.
9712 static bool canFoldInAddressingMode(SDNode *N, SDNode *Use,
9713                                     SelectionDAG &DAG,
9714                                     const TargetLowering &TLI) {
9715   EVT VT;
9716   unsigned AS;
9717 
9718   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(Use)) {
9719     if (LD->isIndexed() || LD->getBasePtr().getNode() != N)
9720       return false;
9721     VT = LD->getMemoryVT();
9722     AS = LD->getAddressSpace();
9723   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(Use)) {
9724     if (ST->isIndexed() || ST->getBasePtr().getNode() != N)
9725       return false;
9726     VT = ST->getMemoryVT();
9727     AS = ST->getAddressSpace();
9728   } else
9729     return false;
9730 
9731   TargetLowering::AddrMode AM;
9732   if (N->getOpcode() == ISD::ADD) {
9733     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
9734     if (Offset)
9735       // [reg +/- imm]
9736       AM.BaseOffs = Offset->getSExtValue();
9737     else
9738       // [reg +/- reg]
9739       AM.Scale = 1;
9740   } else if (N->getOpcode() == ISD::SUB) {
9741     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
9742     if (Offset)
9743       // [reg +/- imm]
9744       AM.BaseOffs = -Offset->getSExtValue();
9745     else
9746       // [reg +/- reg]
9747       AM.Scale = 1;
9748   } else
9749     return false;
9750 
9751   return TLI.isLegalAddressingMode(DAG.getDataLayout(), AM,
9752                                    VT.getTypeForEVT(*DAG.getContext()), AS);
9753 }
9754 
9755 /// Try turning a load/store into a pre-indexed load/store when the base
9756 /// pointer is an add or subtract and it has other uses besides the load/store.
9757 /// After the transformation, the new indexed load/store has effectively folded
9758 /// the add/subtract in and all of its other uses are redirected to the
9759 /// new load/store.
9760 bool DAGCombiner::CombineToPreIndexedLoadStore(SDNode *N) {
9761   if (Level < AfterLegalizeDAG)
9762     return false;
9763 
9764   bool isLoad = true;
9765   SDValue Ptr;
9766   EVT VT;
9767   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
9768     if (LD->isIndexed())
9769       return false;
9770     VT = LD->getMemoryVT();
9771     if (!TLI.isIndexedLoadLegal(ISD::PRE_INC, VT) &&
9772         !TLI.isIndexedLoadLegal(ISD::PRE_DEC, VT))
9773       return false;
9774     Ptr = LD->getBasePtr();
9775   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
9776     if (ST->isIndexed())
9777       return false;
9778     VT = ST->getMemoryVT();
9779     if (!TLI.isIndexedStoreLegal(ISD::PRE_INC, VT) &&
9780         !TLI.isIndexedStoreLegal(ISD::PRE_DEC, VT))
9781       return false;
9782     Ptr = ST->getBasePtr();
9783     isLoad = false;
9784   } else {
9785     return false;
9786   }
9787 
9788   // If the pointer is not an add/sub, or if it doesn't have multiple uses, bail
9789   // out.  There is no reason to make this a preinc/predec.
9790   if ((Ptr.getOpcode() != ISD::ADD && Ptr.getOpcode() != ISD::SUB) ||
9791       Ptr.getNode()->hasOneUse())
9792     return false;
9793 
9794   // Ask the target to do addressing mode selection.
9795   SDValue BasePtr;
9796   SDValue Offset;
9797   ISD::MemIndexedMode AM = ISD::UNINDEXED;
9798   if (!TLI.getPreIndexedAddressParts(N, BasePtr, Offset, AM, DAG))
9799     return false;
9800 
9801   // Backends without true r+i pre-indexed forms may need to pass a
9802   // constant base with a variable offset so that constant coercion
9803   // will work with the patterns in canonical form.
9804   bool Swapped = false;
9805   if (isa<ConstantSDNode>(BasePtr)) {
9806     std::swap(BasePtr, Offset);
9807     Swapped = true;
9808   }
9809 
9810   // Don't create a indexed load / store with zero offset.
9811   if (isNullConstant(Offset))
9812     return false;
9813 
9814   // Try turning it into a pre-indexed load / store except when:
9815   // 1) The new base ptr is a frame index.
9816   // 2) If N is a store and the new base ptr is either the same as or is a
9817   //    predecessor of the value being stored.
9818   // 3) Another use of old base ptr is a predecessor of N. If ptr is folded
9819   //    that would create a cycle.
9820   // 4) All uses are load / store ops that use it as old base ptr.
9821 
9822   // Check #1.  Preinc'ing a frame index would require copying the stack pointer
9823   // (plus the implicit offset) to a register to preinc anyway.
9824   if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
9825     return false;
9826 
9827   // Check #2.
9828   if (!isLoad) {
9829     SDValue Val = cast<StoreSDNode>(N)->getValue();
9830     if (Val == BasePtr || BasePtr.getNode()->isPredecessorOf(Val.getNode()))
9831       return false;
9832   }
9833 
9834   // Caches for hasPredecessorHelper.
9835   SmallPtrSet<const SDNode *, 32> Visited;
9836   SmallVector<const SDNode *, 16> Worklist;
9837   Worklist.push_back(N);
9838 
9839   // If the offset is a constant, there may be other adds of constants that
9840   // can be folded with this one. We should do this to avoid having to keep
9841   // a copy of the original base pointer.
9842   SmallVector<SDNode *, 16> OtherUses;
9843   if (isa<ConstantSDNode>(Offset))
9844     for (SDNode::use_iterator UI = BasePtr.getNode()->use_begin(),
9845                               UE = BasePtr.getNode()->use_end();
9846          UI != UE; ++UI) {
9847       SDUse &Use = UI.getUse();
9848       // Skip the use that is Ptr and uses of other results from BasePtr's
9849       // node (important for nodes that return multiple results).
9850       if (Use.getUser() == Ptr.getNode() || Use != BasePtr)
9851         continue;
9852 
9853       if (SDNode::hasPredecessorHelper(Use.getUser(), Visited, Worklist))
9854         continue;
9855 
9856       if (Use.getUser()->getOpcode() != ISD::ADD &&
9857           Use.getUser()->getOpcode() != ISD::SUB) {
9858         OtherUses.clear();
9859         break;
9860       }
9861 
9862       SDValue Op1 = Use.getUser()->getOperand((UI.getOperandNo() + 1) & 1);
9863       if (!isa<ConstantSDNode>(Op1)) {
9864         OtherUses.clear();
9865         break;
9866       }
9867 
9868       // FIXME: In some cases, we can be smarter about this.
9869       if (Op1.getValueType() != Offset.getValueType()) {
9870         OtherUses.clear();
9871         break;
9872       }
9873 
9874       OtherUses.push_back(Use.getUser());
9875     }
9876 
9877   if (Swapped)
9878     std::swap(BasePtr, Offset);
9879 
9880   // Now check for #3 and #4.
9881   bool RealUse = false;
9882 
9883   for (SDNode *Use : Ptr.getNode()->uses()) {
9884     if (Use == N)
9885       continue;
9886     if (SDNode::hasPredecessorHelper(Use, Visited, Worklist))
9887       return false;
9888 
9889     // If Ptr may be folded in addressing mode of other use, then it's
9890     // not profitable to do this transformation.
9891     if (!canFoldInAddressingMode(Ptr.getNode(), Use, DAG, TLI))
9892       RealUse = true;
9893   }
9894 
9895   if (!RealUse)
9896     return false;
9897 
9898   SDValue Result;
9899   if (isLoad)
9900     Result = DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
9901                                 BasePtr, Offset, AM);
9902   else
9903     Result = DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
9904                                  BasePtr, Offset, AM);
9905   ++PreIndexedNodes;
9906   ++NodesCombined;
9907   DEBUG(dbgs() << "\nReplacing.4 ";
9908         N->dump(&DAG);
9909         dbgs() << "\nWith: ";
9910         Result.getNode()->dump(&DAG);
9911         dbgs() << '\n');
9912   WorklistRemover DeadNodes(*this);
9913   if (isLoad) {
9914     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
9915     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
9916   } else {
9917     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
9918   }
9919 
9920   // Finally, since the node is now dead, remove it from the graph.
9921   deleteAndRecombine(N);
9922 
9923   if (Swapped)
9924     std::swap(BasePtr, Offset);
9925 
9926   // Replace other uses of BasePtr that can be updated to use Ptr
9927   for (unsigned i = 0, e = OtherUses.size(); i != e; ++i) {
9928     unsigned OffsetIdx = 1;
9929     if (OtherUses[i]->getOperand(OffsetIdx).getNode() == BasePtr.getNode())
9930       OffsetIdx = 0;
9931     assert(OtherUses[i]->getOperand(!OffsetIdx).getNode() ==
9932            BasePtr.getNode() && "Expected BasePtr operand");
9933 
9934     // We need to replace ptr0 in the following expression:
9935     //   x0 * offset0 + y0 * ptr0 = t0
9936     // knowing that
9937     //   x1 * offset1 + y1 * ptr0 = t1 (the indexed load/store)
9938     //
9939     // where x0, x1, y0 and y1 in {-1, 1} are given by the types of the
9940     // indexed load/store and the expresion that needs to be re-written.
9941     //
9942     // Therefore, we have:
9943     //   t0 = (x0 * offset0 - x1 * y0 * y1 *offset1) + (y0 * y1) * t1
9944 
9945     ConstantSDNode *CN =
9946       cast<ConstantSDNode>(OtherUses[i]->getOperand(OffsetIdx));
9947     int X0, X1, Y0, Y1;
9948     const APInt &Offset0 = CN->getAPIntValue();
9949     APInt Offset1 = cast<ConstantSDNode>(Offset)->getAPIntValue();
9950 
9951     X0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 1) ? -1 : 1;
9952     Y0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 0) ? -1 : 1;
9953     X1 = (AM == ISD::PRE_DEC && !Swapped) ? -1 : 1;
9954     Y1 = (AM == ISD::PRE_DEC && Swapped) ? -1 : 1;
9955 
9956     unsigned Opcode = (Y0 * Y1 < 0) ? ISD::SUB : ISD::ADD;
9957 
9958     APInt CNV = Offset0;
9959     if (X0 < 0) CNV = -CNV;
9960     if (X1 * Y0 * Y1 < 0) CNV = CNV + Offset1;
9961     else CNV = CNV - Offset1;
9962 
9963     SDLoc DL(OtherUses[i]);
9964 
9965     // We can now generate the new expression.
9966     SDValue NewOp1 = DAG.getConstant(CNV, DL, CN->getValueType(0));
9967     SDValue NewOp2 = Result.getValue(isLoad ? 1 : 0);
9968 
9969     SDValue NewUse = DAG.getNode(Opcode,
9970                                  DL,
9971                                  OtherUses[i]->getValueType(0), NewOp1, NewOp2);
9972     DAG.ReplaceAllUsesOfValueWith(SDValue(OtherUses[i], 0), NewUse);
9973     deleteAndRecombine(OtherUses[i]);
9974   }
9975 
9976   // Replace the uses of Ptr with uses of the updated base value.
9977   DAG.ReplaceAllUsesOfValueWith(Ptr, Result.getValue(isLoad ? 1 : 0));
9978   deleteAndRecombine(Ptr.getNode());
9979 
9980   return true;
9981 }
9982 
9983 /// Try to combine a load/store with a add/sub of the base pointer node into a
9984 /// post-indexed load/store. The transformation folded the add/subtract into the
9985 /// new indexed load/store effectively and all of its uses are redirected to the
9986 /// new load/store.
9987 bool DAGCombiner::CombineToPostIndexedLoadStore(SDNode *N) {
9988   if (Level < AfterLegalizeDAG)
9989     return false;
9990 
9991   bool isLoad = true;
9992   SDValue Ptr;
9993   EVT VT;
9994   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
9995     if (LD->isIndexed())
9996       return false;
9997     VT = LD->getMemoryVT();
9998     if (!TLI.isIndexedLoadLegal(ISD::POST_INC, VT) &&
9999         !TLI.isIndexedLoadLegal(ISD::POST_DEC, VT))
10000       return false;
10001     Ptr = LD->getBasePtr();
10002   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
10003     if (ST->isIndexed())
10004       return false;
10005     VT = ST->getMemoryVT();
10006     if (!TLI.isIndexedStoreLegal(ISD::POST_INC, VT) &&
10007         !TLI.isIndexedStoreLegal(ISD::POST_DEC, VT))
10008       return false;
10009     Ptr = ST->getBasePtr();
10010     isLoad = false;
10011   } else {
10012     return false;
10013   }
10014 
10015   if (Ptr.getNode()->hasOneUse())
10016     return false;
10017 
10018   for (SDNode *Op : Ptr.getNode()->uses()) {
10019     if (Op == N ||
10020         (Op->getOpcode() != ISD::ADD && Op->getOpcode() != ISD::SUB))
10021       continue;
10022 
10023     SDValue BasePtr;
10024     SDValue Offset;
10025     ISD::MemIndexedMode AM = ISD::UNINDEXED;
10026     if (TLI.getPostIndexedAddressParts(N, Op, BasePtr, Offset, AM, DAG)) {
10027       // Don't create a indexed load / store with zero offset.
10028       if (isNullConstant(Offset))
10029         continue;
10030 
10031       // Try turning it into a post-indexed load / store except when
10032       // 1) All uses are load / store ops that use it as base ptr (and
10033       //    it may be folded as addressing mmode).
10034       // 2) Op must be independent of N, i.e. Op is neither a predecessor
10035       //    nor a successor of N. Otherwise, if Op is folded that would
10036       //    create a cycle.
10037 
10038       if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
10039         continue;
10040 
10041       // Check for #1.
10042       bool TryNext = false;
10043       for (SDNode *Use : BasePtr.getNode()->uses()) {
10044         if (Use == Ptr.getNode())
10045           continue;
10046 
10047         // If all the uses are load / store addresses, then don't do the
10048         // transformation.
10049         if (Use->getOpcode() == ISD::ADD || Use->getOpcode() == ISD::SUB){
10050           bool RealUse = false;
10051           for (SDNode *UseUse : Use->uses()) {
10052             if (!canFoldInAddressingMode(Use, UseUse, DAG, TLI))
10053               RealUse = true;
10054           }
10055 
10056           if (!RealUse) {
10057             TryNext = true;
10058             break;
10059           }
10060         }
10061       }
10062 
10063       if (TryNext)
10064         continue;
10065 
10066       // Check for #2
10067       if (!Op->isPredecessorOf(N) && !N->isPredecessorOf(Op)) {
10068         SDValue Result = isLoad
10069           ? DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
10070                                BasePtr, Offset, AM)
10071           : DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
10072                                 BasePtr, Offset, AM);
10073         ++PostIndexedNodes;
10074         ++NodesCombined;
10075         DEBUG(dbgs() << "\nReplacing.5 ";
10076               N->dump(&DAG);
10077               dbgs() << "\nWith: ";
10078               Result.getNode()->dump(&DAG);
10079               dbgs() << '\n');
10080         WorklistRemover DeadNodes(*this);
10081         if (isLoad) {
10082           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
10083           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
10084         } else {
10085           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
10086         }
10087 
10088         // Finally, since the node is now dead, remove it from the graph.
10089         deleteAndRecombine(N);
10090 
10091         // Replace the uses of Use with uses of the updated base value.
10092         DAG.ReplaceAllUsesOfValueWith(SDValue(Op, 0),
10093                                       Result.getValue(isLoad ? 1 : 0));
10094         deleteAndRecombine(Op);
10095         return true;
10096       }
10097     }
10098   }
10099 
10100   return false;
10101 }
10102 
10103 /// \brief Return the base-pointer arithmetic from an indexed \p LD.
10104 SDValue DAGCombiner::SplitIndexingFromLoad(LoadSDNode *LD) {
10105   ISD::MemIndexedMode AM = LD->getAddressingMode();
10106   assert(AM != ISD::UNINDEXED);
10107   SDValue BP = LD->getOperand(1);
10108   SDValue Inc = LD->getOperand(2);
10109 
10110   // Some backends use TargetConstants for load offsets, but don't expect
10111   // TargetConstants in general ADD nodes. We can convert these constants into
10112   // regular Constants (if the constant is not opaque).
10113   assert((Inc.getOpcode() != ISD::TargetConstant ||
10114           !cast<ConstantSDNode>(Inc)->isOpaque()) &&
10115          "Cannot split out indexing using opaque target constants");
10116   if (Inc.getOpcode() == ISD::TargetConstant) {
10117     ConstantSDNode *ConstInc = cast<ConstantSDNode>(Inc);
10118     Inc = DAG.getConstant(*ConstInc->getConstantIntValue(), SDLoc(Inc),
10119                           ConstInc->getValueType(0));
10120   }
10121 
10122   unsigned Opc =
10123       (AM == ISD::PRE_INC || AM == ISD::POST_INC ? ISD::ADD : ISD::SUB);
10124   return DAG.getNode(Opc, SDLoc(LD), BP.getSimpleValueType(), BP, Inc);
10125 }
10126 
10127 SDValue DAGCombiner::visitLOAD(SDNode *N) {
10128   LoadSDNode *LD  = cast<LoadSDNode>(N);
10129   SDValue Chain = LD->getChain();
10130   SDValue Ptr   = LD->getBasePtr();
10131 
10132   // If load is not volatile and there are no uses of the loaded value (and
10133   // the updated indexed value in case of indexed loads), change uses of the
10134   // chain value into uses of the chain input (i.e. delete the dead load).
10135   if (!LD->isVolatile()) {
10136     if (N->getValueType(1) == MVT::Other) {
10137       // Unindexed loads.
10138       if (!N->hasAnyUseOfValue(0)) {
10139         // It's not safe to use the two value CombineTo variant here. e.g.
10140         // v1, chain2 = load chain1, loc
10141         // v2, chain3 = load chain2, loc
10142         // v3         = add v2, c
10143         // Now we replace use of chain2 with chain1.  This makes the second load
10144         // isomorphic to the one we are deleting, and thus makes this load live.
10145         DEBUG(dbgs() << "\nReplacing.6 ";
10146               N->dump(&DAG);
10147               dbgs() << "\nWith chain: ";
10148               Chain.getNode()->dump(&DAG);
10149               dbgs() << "\n");
10150         WorklistRemover DeadNodes(*this);
10151         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
10152 
10153         if (N->use_empty())
10154           deleteAndRecombine(N);
10155 
10156         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10157       }
10158     } else {
10159       // Indexed loads.
10160       assert(N->getValueType(2) == MVT::Other && "Malformed indexed loads?");
10161 
10162       // If this load has an opaque TargetConstant offset, then we cannot split
10163       // the indexing into an add/sub directly (that TargetConstant may not be
10164       // valid for a different type of node, and we cannot convert an opaque
10165       // target constant into a regular constant).
10166       bool HasOTCInc = LD->getOperand(2).getOpcode() == ISD::TargetConstant &&
10167                        cast<ConstantSDNode>(LD->getOperand(2))->isOpaque();
10168 
10169       if (!N->hasAnyUseOfValue(0) &&
10170           ((MaySplitLoadIndex && !HasOTCInc) || !N->hasAnyUseOfValue(1))) {
10171         SDValue Undef = DAG.getUNDEF(N->getValueType(0));
10172         SDValue Index;
10173         if (N->hasAnyUseOfValue(1) && MaySplitLoadIndex && !HasOTCInc) {
10174           Index = SplitIndexingFromLoad(LD);
10175           // Try to fold the base pointer arithmetic into subsequent loads and
10176           // stores.
10177           AddUsersToWorklist(N);
10178         } else
10179           Index = DAG.getUNDEF(N->getValueType(1));
10180         DEBUG(dbgs() << "\nReplacing.7 ";
10181               N->dump(&DAG);
10182               dbgs() << "\nWith: ";
10183               Undef.getNode()->dump(&DAG);
10184               dbgs() << " and 2 other values\n");
10185         WorklistRemover DeadNodes(*this);
10186         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Undef);
10187         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Index);
10188         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 2), Chain);
10189         deleteAndRecombine(N);
10190         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10191       }
10192     }
10193   }
10194 
10195   // If this load is directly stored, replace the load value with the stored
10196   // value.
10197   // TODO: Handle store large -> read small portion.
10198   // TODO: Handle TRUNCSTORE/LOADEXT
10199   if (ISD::isNormalLoad(N) && !LD->isVolatile()) {
10200     if (ISD::isNON_TRUNCStore(Chain.getNode())) {
10201       StoreSDNode *PrevST = cast<StoreSDNode>(Chain);
10202       if (PrevST->getBasePtr() == Ptr &&
10203           PrevST->getValue().getValueType() == N->getValueType(0))
10204       return CombineTo(N, Chain.getOperand(1), Chain);
10205     }
10206   }
10207 
10208   // Try to infer better alignment information than the load already has.
10209   if (OptLevel != CodeGenOpt::None && LD->isUnindexed()) {
10210     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
10211       if (Align > LD->getMemOperand()->getBaseAlignment()) {
10212         SDValue NewLoad = DAG.getExtLoad(
10213             LD->getExtensionType(), SDLoc(N), LD->getValueType(0), Chain, Ptr,
10214             LD->getPointerInfo(), LD->getMemoryVT(), Align,
10215             LD->getMemOperand()->getFlags(), LD->getAAInfo());
10216         if (NewLoad.getNode() != N)
10217           return CombineTo(N, NewLoad, SDValue(NewLoad.getNode(), 1), true);
10218       }
10219     }
10220   }
10221 
10222   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
10223                                                   : DAG.getSubtarget().useAA();
10224 #ifndef NDEBUG
10225   if (CombinerAAOnlyFunc.getNumOccurrences() &&
10226       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
10227     UseAA = false;
10228 #endif
10229   if (UseAA && LD->isUnindexed()) {
10230     // Walk up chain skipping non-aliasing memory nodes.
10231     SDValue BetterChain = FindBetterChain(N, Chain);
10232 
10233     // If there is a better chain.
10234     if (Chain != BetterChain) {
10235       SDValue ReplLoad;
10236 
10237       // Replace the chain to void dependency.
10238       if (LD->getExtensionType() == ISD::NON_EXTLOAD) {
10239         ReplLoad = DAG.getLoad(N->getValueType(0), SDLoc(LD),
10240                                BetterChain, Ptr, LD->getMemOperand());
10241       } else {
10242         ReplLoad = DAG.getExtLoad(LD->getExtensionType(), SDLoc(LD),
10243                                   LD->getValueType(0),
10244                                   BetterChain, Ptr, LD->getMemoryVT(),
10245                                   LD->getMemOperand());
10246       }
10247 
10248       // Create token factor to keep old chain connected.
10249       SDValue Token = DAG.getNode(ISD::TokenFactor, SDLoc(N),
10250                                   MVT::Other, Chain, ReplLoad.getValue(1));
10251 
10252       // Make sure the new and old chains are cleaned up.
10253       AddToWorklist(Token.getNode());
10254 
10255       // Replace uses with load result and token factor. Don't add users
10256       // to work list.
10257       return CombineTo(N, ReplLoad.getValue(0), Token, false);
10258     }
10259   }
10260 
10261   // Try transforming N to an indexed load.
10262   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
10263     return SDValue(N, 0);
10264 
10265   // Try to slice up N to more direct loads if the slices are mapped to
10266   // different register banks or pairing can take place.
10267   if (SliceUpLoad(N))
10268     return SDValue(N, 0);
10269 
10270   return SDValue();
10271 }
10272 
10273 namespace {
10274 /// \brief Helper structure used to slice a load in smaller loads.
10275 /// Basically a slice is obtained from the following sequence:
10276 /// Origin = load Ty1, Base
10277 /// Shift = srl Ty1 Origin, CstTy Amount
10278 /// Inst = trunc Shift to Ty2
10279 ///
10280 /// Then, it will be rewriten into:
10281 /// Slice = load SliceTy, Base + SliceOffset
10282 /// [Inst = zext Slice to Ty2], only if SliceTy <> Ty2
10283 ///
10284 /// SliceTy is deduced from the number of bits that are actually used to
10285 /// build Inst.
10286 struct LoadedSlice {
10287   /// \brief Helper structure used to compute the cost of a slice.
10288   struct Cost {
10289     /// Are we optimizing for code size.
10290     bool ForCodeSize;
10291     /// Various cost.
10292     unsigned Loads;
10293     unsigned Truncates;
10294     unsigned CrossRegisterBanksCopies;
10295     unsigned ZExts;
10296     unsigned Shift;
10297 
10298     Cost(bool ForCodeSize = false)
10299         : ForCodeSize(ForCodeSize), Loads(0), Truncates(0),
10300           CrossRegisterBanksCopies(0), ZExts(0), Shift(0) {}
10301 
10302     /// \brief Get the cost of one isolated slice.
10303     Cost(const LoadedSlice &LS, bool ForCodeSize = false)
10304         : ForCodeSize(ForCodeSize), Loads(1), Truncates(0),
10305           CrossRegisterBanksCopies(0), ZExts(0), Shift(0) {
10306       EVT TruncType = LS.Inst->getValueType(0);
10307       EVT LoadedType = LS.getLoadedType();
10308       if (TruncType != LoadedType &&
10309           !LS.DAG->getTargetLoweringInfo().isZExtFree(LoadedType, TruncType))
10310         ZExts = 1;
10311     }
10312 
10313     /// \brief Account for slicing gain in the current cost.
10314     /// Slicing provide a few gains like removing a shift or a
10315     /// truncate. This method allows to grow the cost of the original
10316     /// load with the gain from this slice.
10317     void addSliceGain(const LoadedSlice &LS) {
10318       // Each slice saves a truncate.
10319       const TargetLowering &TLI = LS.DAG->getTargetLoweringInfo();
10320       if (!TLI.isTruncateFree(LS.Inst->getOperand(0).getValueType(),
10321                               LS.Inst->getValueType(0)))
10322         ++Truncates;
10323       // If there is a shift amount, this slice gets rid of it.
10324       if (LS.Shift)
10325         ++Shift;
10326       // If this slice can merge a cross register bank copy, account for it.
10327       if (LS.canMergeExpensiveCrossRegisterBankCopy())
10328         ++CrossRegisterBanksCopies;
10329     }
10330 
10331     Cost &operator+=(const Cost &RHS) {
10332       Loads += RHS.Loads;
10333       Truncates += RHS.Truncates;
10334       CrossRegisterBanksCopies += RHS.CrossRegisterBanksCopies;
10335       ZExts += RHS.ZExts;
10336       Shift += RHS.Shift;
10337       return *this;
10338     }
10339 
10340     bool operator==(const Cost &RHS) const {
10341       return Loads == RHS.Loads && Truncates == RHS.Truncates &&
10342              CrossRegisterBanksCopies == RHS.CrossRegisterBanksCopies &&
10343              ZExts == RHS.ZExts && Shift == RHS.Shift;
10344     }
10345 
10346     bool operator!=(const Cost &RHS) const { return !(*this == RHS); }
10347 
10348     bool operator<(const Cost &RHS) const {
10349       // Assume cross register banks copies are as expensive as loads.
10350       // FIXME: Do we want some more target hooks?
10351       unsigned ExpensiveOpsLHS = Loads + CrossRegisterBanksCopies;
10352       unsigned ExpensiveOpsRHS = RHS.Loads + RHS.CrossRegisterBanksCopies;
10353       // Unless we are optimizing for code size, consider the
10354       // expensive operation first.
10355       if (!ForCodeSize && ExpensiveOpsLHS != ExpensiveOpsRHS)
10356         return ExpensiveOpsLHS < ExpensiveOpsRHS;
10357       return (Truncates + ZExts + Shift + ExpensiveOpsLHS) <
10358              (RHS.Truncates + RHS.ZExts + RHS.Shift + ExpensiveOpsRHS);
10359     }
10360 
10361     bool operator>(const Cost &RHS) const { return RHS < *this; }
10362 
10363     bool operator<=(const Cost &RHS) const { return !(RHS < *this); }
10364 
10365     bool operator>=(const Cost &RHS) const { return !(*this < RHS); }
10366   };
10367   // The last instruction that represent the slice. This should be a
10368   // truncate instruction.
10369   SDNode *Inst;
10370   // The original load instruction.
10371   LoadSDNode *Origin;
10372   // The right shift amount in bits from the original load.
10373   unsigned Shift;
10374   // The DAG from which Origin came from.
10375   // This is used to get some contextual information about legal types, etc.
10376   SelectionDAG *DAG;
10377 
10378   LoadedSlice(SDNode *Inst = nullptr, LoadSDNode *Origin = nullptr,
10379               unsigned Shift = 0, SelectionDAG *DAG = nullptr)
10380       : Inst(Inst), Origin(Origin), Shift(Shift), DAG(DAG) {}
10381 
10382   /// \brief Get the bits used in a chunk of bits \p BitWidth large.
10383   /// \return Result is \p BitWidth and has used bits set to 1 and
10384   ///         not used bits set to 0.
10385   APInt getUsedBits() const {
10386     // Reproduce the trunc(lshr) sequence:
10387     // - Start from the truncated value.
10388     // - Zero extend to the desired bit width.
10389     // - Shift left.
10390     assert(Origin && "No original load to compare against.");
10391     unsigned BitWidth = Origin->getValueSizeInBits(0);
10392     assert(Inst && "This slice is not bound to an instruction");
10393     assert(Inst->getValueSizeInBits(0) <= BitWidth &&
10394            "Extracted slice is bigger than the whole type!");
10395     APInt UsedBits(Inst->getValueSizeInBits(0), 0);
10396     UsedBits.setAllBits();
10397     UsedBits = UsedBits.zext(BitWidth);
10398     UsedBits <<= Shift;
10399     return UsedBits;
10400   }
10401 
10402   /// \brief Get the size of the slice to be loaded in bytes.
10403   unsigned getLoadedSize() const {
10404     unsigned SliceSize = getUsedBits().countPopulation();
10405     assert(!(SliceSize & 0x7) && "Size is not a multiple of a byte.");
10406     return SliceSize / 8;
10407   }
10408 
10409   /// \brief Get the type that will be loaded for this slice.
10410   /// Note: This may not be the final type for the slice.
10411   EVT getLoadedType() const {
10412     assert(DAG && "Missing context");
10413     LLVMContext &Ctxt = *DAG->getContext();
10414     return EVT::getIntegerVT(Ctxt, getLoadedSize() * 8);
10415   }
10416 
10417   /// \brief Get the alignment of the load used for this slice.
10418   unsigned getAlignment() const {
10419     unsigned Alignment = Origin->getAlignment();
10420     unsigned Offset = getOffsetFromBase();
10421     if (Offset != 0)
10422       Alignment = MinAlign(Alignment, Alignment + Offset);
10423     return Alignment;
10424   }
10425 
10426   /// \brief Check if this slice can be rewritten with legal operations.
10427   bool isLegal() const {
10428     // An invalid slice is not legal.
10429     if (!Origin || !Inst || !DAG)
10430       return false;
10431 
10432     // Offsets are for indexed load only, we do not handle that.
10433     if (!Origin->getOffset().isUndef())
10434       return false;
10435 
10436     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
10437 
10438     // Check that the type is legal.
10439     EVT SliceType = getLoadedType();
10440     if (!TLI.isTypeLegal(SliceType))
10441       return false;
10442 
10443     // Check that the load is legal for this type.
10444     if (!TLI.isOperationLegal(ISD::LOAD, SliceType))
10445       return false;
10446 
10447     // Check that the offset can be computed.
10448     // 1. Check its type.
10449     EVT PtrType = Origin->getBasePtr().getValueType();
10450     if (PtrType == MVT::Untyped || PtrType.isExtended())
10451       return false;
10452 
10453     // 2. Check that it fits in the immediate.
10454     if (!TLI.isLegalAddImmediate(getOffsetFromBase()))
10455       return false;
10456 
10457     // 3. Check that the computation is legal.
10458     if (!TLI.isOperationLegal(ISD::ADD, PtrType))
10459       return false;
10460 
10461     // Check that the zext is legal if it needs one.
10462     EVT TruncateType = Inst->getValueType(0);
10463     if (TruncateType != SliceType &&
10464         !TLI.isOperationLegal(ISD::ZERO_EXTEND, TruncateType))
10465       return false;
10466 
10467     return true;
10468   }
10469 
10470   /// \brief Get the offset in bytes of this slice in the original chunk of
10471   /// bits.
10472   /// \pre DAG != nullptr.
10473   uint64_t getOffsetFromBase() const {
10474     assert(DAG && "Missing context.");
10475     bool IsBigEndian = DAG->getDataLayout().isBigEndian();
10476     assert(!(Shift & 0x7) && "Shifts not aligned on Bytes are not supported.");
10477     uint64_t Offset = Shift / 8;
10478     unsigned TySizeInBytes = Origin->getValueSizeInBits(0) / 8;
10479     assert(!(Origin->getValueSizeInBits(0) & 0x7) &&
10480            "The size of the original loaded type is not a multiple of a"
10481            " byte.");
10482     // If Offset is bigger than TySizeInBytes, it means we are loading all
10483     // zeros. This should have been optimized before in the process.
10484     assert(TySizeInBytes > Offset &&
10485            "Invalid shift amount for given loaded size");
10486     if (IsBigEndian)
10487       Offset = TySizeInBytes - Offset - getLoadedSize();
10488     return Offset;
10489   }
10490 
10491   /// \brief Generate the sequence of instructions to load the slice
10492   /// represented by this object and redirect the uses of this slice to
10493   /// this new sequence of instructions.
10494   /// \pre this->Inst && this->Origin are valid Instructions and this
10495   /// object passed the legal check: LoadedSlice::isLegal returned true.
10496   /// \return The last instruction of the sequence used to load the slice.
10497   SDValue loadSlice() const {
10498     assert(Inst && Origin && "Unable to replace a non-existing slice.");
10499     const SDValue &OldBaseAddr = Origin->getBasePtr();
10500     SDValue BaseAddr = OldBaseAddr;
10501     // Get the offset in that chunk of bytes w.r.t. the endianess.
10502     int64_t Offset = static_cast<int64_t>(getOffsetFromBase());
10503     assert(Offset >= 0 && "Offset too big to fit in int64_t!");
10504     if (Offset) {
10505       // BaseAddr = BaseAddr + Offset.
10506       EVT ArithType = BaseAddr.getValueType();
10507       SDLoc DL(Origin);
10508       BaseAddr = DAG->getNode(ISD::ADD, DL, ArithType, BaseAddr,
10509                               DAG->getConstant(Offset, DL, ArithType));
10510     }
10511 
10512     // Create the type of the loaded slice according to its size.
10513     EVT SliceType = getLoadedType();
10514 
10515     // Create the load for the slice.
10516     SDValue LastInst =
10517         DAG->getLoad(SliceType, SDLoc(Origin), Origin->getChain(), BaseAddr,
10518                      Origin->getPointerInfo().getWithOffset(Offset),
10519                      getAlignment(), Origin->getMemOperand()->getFlags());
10520     // If the final type is not the same as the loaded type, this means that
10521     // we have to pad with zero. Create a zero extend for that.
10522     EVT FinalType = Inst->getValueType(0);
10523     if (SliceType != FinalType)
10524       LastInst =
10525           DAG->getNode(ISD::ZERO_EXTEND, SDLoc(LastInst), FinalType, LastInst);
10526     return LastInst;
10527   }
10528 
10529   /// \brief Check if this slice can be merged with an expensive cross register
10530   /// bank copy. E.g.,
10531   /// i = load i32
10532   /// f = bitcast i32 i to float
10533   bool canMergeExpensiveCrossRegisterBankCopy() const {
10534     if (!Inst || !Inst->hasOneUse())
10535       return false;
10536     SDNode *Use = *Inst->use_begin();
10537     if (Use->getOpcode() != ISD::BITCAST)
10538       return false;
10539     assert(DAG && "Missing context");
10540     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
10541     EVT ResVT = Use->getValueType(0);
10542     const TargetRegisterClass *ResRC = TLI.getRegClassFor(ResVT.getSimpleVT());
10543     const TargetRegisterClass *ArgRC =
10544         TLI.getRegClassFor(Use->getOperand(0).getValueType().getSimpleVT());
10545     if (ArgRC == ResRC || !TLI.isOperationLegal(ISD::LOAD, ResVT))
10546       return false;
10547 
10548     // At this point, we know that we perform a cross-register-bank copy.
10549     // Check if it is expensive.
10550     const TargetRegisterInfo *TRI = DAG->getSubtarget().getRegisterInfo();
10551     // Assume bitcasts are cheap, unless both register classes do not
10552     // explicitly share a common sub class.
10553     if (!TRI || TRI->getCommonSubClass(ArgRC, ResRC))
10554       return false;
10555 
10556     // Check if it will be merged with the load.
10557     // 1. Check the alignment constraint.
10558     unsigned RequiredAlignment = DAG->getDataLayout().getABITypeAlignment(
10559         ResVT.getTypeForEVT(*DAG->getContext()));
10560 
10561     if (RequiredAlignment > getAlignment())
10562       return false;
10563 
10564     // 2. Check that the load is a legal operation for that type.
10565     if (!TLI.isOperationLegal(ISD::LOAD, ResVT))
10566       return false;
10567 
10568     // 3. Check that we do not have a zext in the way.
10569     if (Inst->getValueType(0) != getLoadedType())
10570       return false;
10571 
10572     return true;
10573   }
10574 };
10575 }
10576 
10577 /// \brief Check that all bits set in \p UsedBits form a dense region, i.e.,
10578 /// \p UsedBits looks like 0..0 1..1 0..0.
10579 static bool areUsedBitsDense(const APInt &UsedBits) {
10580   // If all the bits are one, this is dense!
10581   if (UsedBits.isAllOnesValue())
10582     return true;
10583 
10584   // Get rid of the unused bits on the right.
10585   APInt NarrowedUsedBits = UsedBits.lshr(UsedBits.countTrailingZeros());
10586   // Get rid of the unused bits on the left.
10587   if (NarrowedUsedBits.countLeadingZeros())
10588     NarrowedUsedBits = NarrowedUsedBits.trunc(NarrowedUsedBits.getActiveBits());
10589   // Check that the chunk of bits is completely used.
10590   return NarrowedUsedBits.isAllOnesValue();
10591 }
10592 
10593 /// \brief Check whether or not \p First and \p Second are next to each other
10594 /// in memory. This means that there is no hole between the bits loaded
10595 /// by \p First and the bits loaded by \p Second.
10596 static bool areSlicesNextToEachOther(const LoadedSlice &First,
10597                                      const LoadedSlice &Second) {
10598   assert(First.Origin == Second.Origin && First.Origin &&
10599          "Unable to match different memory origins.");
10600   APInt UsedBits = First.getUsedBits();
10601   assert((UsedBits & Second.getUsedBits()) == 0 &&
10602          "Slices are not supposed to overlap.");
10603   UsedBits |= Second.getUsedBits();
10604   return areUsedBitsDense(UsedBits);
10605 }
10606 
10607 /// \brief Adjust the \p GlobalLSCost according to the target
10608 /// paring capabilities and the layout of the slices.
10609 /// \pre \p GlobalLSCost should account for at least as many loads as
10610 /// there is in the slices in \p LoadedSlices.
10611 static void adjustCostForPairing(SmallVectorImpl<LoadedSlice> &LoadedSlices,
10612                                  LoadedSlice::Cost &GlobalLSCost) {
10613   unsigned NumberOfSlices = LoadedSlices.size();
10614   // If there is less than 2 elements, no pairing is possible.
10615   if (NumberOfSlices < 2)
10616     return;
10617 
10618   // Sort the slices so that elements that are likely to be next to each
10619   // other in memory are next to each other in the list.
10620   std::sort(LoadedSlices.begin(), LoadedSlices.end(),
10621             [](const LoadedSlice &LHS, const LoadedSlice &RHS) {
10622     assert(LHS.Origin == RHS.Origin && "Different bases not implemented.");
10623     return LHS.getOffsetFromBase() < RHS.getOffsetFromBase();
10624   });
10625   const TargetLowering &TLI = LoadedSlices[0].DAG->getTargetLoweringInfo();
10626   // First (resp. Second) is the first (resp. Second) potentially candidate
10627   // to be placed in a paired load.
10628   const LoadedSlice *First = nullptr;
10629   const LoadedSlice *Second = nullptr;
10630   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice,
10631                 // Set the beginning of the pair.
10632                                                            First = Second) {
10633 
10634     Second = &LoadedSlices[CurrSlice];
10635 
10636     // If First is NULL, it means we start a new pair.
10637     // Get to the next slice.
10638     if (!First)
10639       continue;
10640 
10641     EVT LoadedType = First->getLoadedType();
10642 
10643     // If the types of the slices are different, we cannot pair them.
10644     if (LoadedType != Second->getLoadedType())
10645       continue;
10646 
10647     // Check if the target supplies paired loads for this type.
10648     unsigned RequiredAlignment = 0;
10649     if (!TLI.hasPairedLoad(LoadedType, RequiredAlignment)) {
10650       // move to the next pair, this type is hopeless.
10651       Second = nullptr;
10652       continue;
10653     }
10654     // Check if we meet the alignment requirement.
10655     if (RequiredAlignment > First->getAlignment())
10656       continue;
10657 
10658     // Check that both loads are next to each other in memory.
10659     if (!areSlicesNextToEachOther(*First, *Second))
10660       continue;
10661 
10662     assert(GlobalLSCost.Loads > 0 && "We save more loads than we created!");
10663     --GlobalLSCost.Loads;
10664     // Move to the next pair.
10665     Second = nullptr;
10666   }
10667 }
10668 
10669 /// \brief Check the profitability of all involved LoadedSlice.
10670 /// Currently, it is considered profitable if there is exactly two
10671 /// involved slices (1) which are (2) next to each other in memory, and
10672 /// whose cost (\see LoadedSlice::Cost) is smaller than the original load (3).
10673 ///
10674 /// Note: The order of the elements in \p LoadedSlices may be modified, but not
10675 /// the elements themselves.
10676 ///
10677 /// FIXME: When the cost model will be mature enough, we can relax
10678 /// constraints (1) and (2).
10679 static bool isSlicingProfitable(SmallVectorImpl<LoadedSlice> &LoadedSlices,
10680                                 const APInt &UsedBits, bool ForCodeSize) {
10681   unsigned NumberOfSlices = LoadedSlices.size();
10682   if (StressLoadSlicing)
10683     return NumberOfSlices > 1;
10684 
10685   // Check (1).
10686   if (NumberOfSlices != 2)
10687     return false;
10688 
10689   // Check (2).
10690   if (!areUsedBitsDense(UsedBits))
10691     return false;
10692 
10693   // Check (3).
10694   LoadedSlice::Cost OrigCost(ForCodeSize), GlobalSlicingCost(ForCodeSize);
10695   // The original code has one big load.
10696   OrigCost.Loads = 1;
10697   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice) {
10698     const LoadedSlice &LS = LoadedSlices[CurrSlice];
10699     // Accumulate the cost of all the slices.
10700     LoadedSlice::Cost SliceCost(LS, ForCodeSize);
10701     GlobalSlicingCost += SliceCost;
10702 
10703     // Account as cost in the original configuration the gain obtained
10704     // with the current slices.
10705     OrigCost.addSliceGain(LS);
10706   }
10707 
10708   // If the target supports paired load, adjust the cost accordingly.
10709   adjustCostForPairing(LoadedSlices, GlobalSlicingCost);
10710   return OrigCost > GlobalSlicingCost;
10711 }
10712 
10713 /// \brief If the given load, \p LI, is used only by trunc or trunc(lshr)
10714 /// operations, split it in the various pieces being extracted.
10715 ///
10716 /// This sort of thing is introduced by SROA.
10717 /// This slicing takes care not to insert overlapping loads.
10718 /// \pre LI is a simple load (i.e., not an atomic or volatile load).
10719 bool DAGCombiner::SliceUpLoad(SDNode *N) {
10720   if (Level < AfterLegalizeDAG)
10721     return false;
10722 
10723   LoadSDNode *LD = cast<LoadSDNode>(N);
10724   if (LD->isVolatile() || !ISD::isNormalLoad(LD) ||
10725       !LD->getValueType(0).isInteger())
10726     return false;
10727 
10728   // Keep track of already used bits to detect overlapping values.
10729   // In that case, we will just abort the transformation.
10730   APInt UsedBits(LD->getValueSizeInBits(0), 0);
10731 
10732   SmallVector<LoadedSlice, 4> LoadedSlices;
10733 
10734   // Check if this load is used as several smaller chunks of bits.
10735   // Basically, look for uses in trunc or trunc(lshr) and record a new chain
10736   // of computation for each trunc.
10737   for (SDNode::use_iterator UI = LD->use_begin(), UIEnd = LD->use_end();
10738        UI != UIEnd; ++UI) {
10739     // Skip the uses of the chain.
10740     if (UI.getUse().getResNo() != 0)
10741       continue;
10742 
10743     SDNode *User = *UI;
10744     unsigned Shift = 0;
10745 
10746     // Check if this is a trunc(lshr).
10747     if (User->getOpcode() == ISD::SRL && User->hasOneUse() &&
10748         isa<ConstantSDNode>(User->getOperand(1))) {
10749       Shift = cast<ConstantSDNode>(User->getOperand(1))->getZExtValue();
10750       User = *User->use_begin();
10751     }
10752 
10753     // At this point, User is a Truncate, iff we encountered, trunc or
10754     // trunc(lshr).
10755     if (User->getOpcode() != ISD::TRUNCATE)
10756       return false;
10757 
10758     // The width of the type must be a power of 2 and greater than 8-bits.
10759     // Otherwise the load cannot be represented in LLVM IR.
10760     // Moreover, if we shifted with a non-8-bits multiple, the slice
10761     // will be across several bytes. We do not support that.
10762     unsigned Width = User->getValueSizeInBits(0);
10763     if (Width < 8 || !isPowerOf2_32(Width) || (Shift & 0x7))
10764       return 0;
10765 
10766     // Build the slice for this chain of computations.
10767     LoadedSlice LS(User, LD, Shift, &DAG);
10768     APInt CurrentUsedBits = LS.getUsedBits();
10769 
10770     // Check if this slice overlaps with another.
10771     if ((CurrentUsedBits & UsedBits) != 0)
10772       return false;
10773     // Update the bits used globally.
10774     UsedBits |= CurrentUsedBits;
10775 
10776     // Check if the new slice would be legal.
10777     if (!LS.isLegal())
10778       return false;
10779 
10780     // Record the slice.
10781     LoadedSlices.push_back(LS);
10782   }
10783 
10784   // Abort slicing if it does not seem to be profitable.
10785   if (!isSlicingProfitable(LoadedSlices, UsedBits, ForCodeSize))
10786     return false;
10787 
10788   ++SlicedLoads;
10789 
10790   // Rewrite each chain to use an independent load.
10791   // By construction, each chain can be represented by a unique load.
10792 
10793   // Prepare the argument for the new token factor for all the slices.
10794   SmallVector<SDValue, 8> ArgChains;
10795   for (SmallVectorImpl<LoadedSlice>::const_iterator
10796            LSIt = LoadedSlices.begin(),
10797            LSItEnd = LoadedSlices.end();
10798        LSIt != LSItEnd; ++LSIt) {
10799     SDValue SliceInst = LSIt->loadSlice();
10800     CombineTo(LSIt->Inst, SliceInst, true);
10801     if (SliceInst.getOpcode() != ISD::LOAD)
10802       SliceInst = SliceInst.getOperand(0);
10803     assert(SliceInst->getOpcode() == ISD::LOAD &&
10804            "It takes more than a zext to get to the loaded slice!!");
10805     ArgChains.push_back(SliceInst.getValue(1));
10806   }
10807 
10808   SDValue Chain = DAG.getNode(ISD::TokenFactor, SDLoc(LD), MVT::Other,
10809                               ArgChains);
10810   DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
10811   return true;
10812 }
10813 
10814 /// Check to see if V is (and load (ptr), imm), where the load is having
10815 /// specific bytes cleared out.  If so, return the byte size being masked out
10816 /// and the shift amount.
10817 static std::pair<unsigned, unsigned>
10818 CheckForMaskedLoad(SDValue V, SDValue Ptr, SDValue Chain) {
10819   std::pair<unsigned, unsigned> Result(0, 0);
10820 
10821   // Check for the structure we're looking for.
10822   if (V->getOpcode() != ISD::AND ||
10823       !isa<ConstantSDNode>(V->getOperand(1)) ||
10824       !ISD::isNormalLoad(V->getOperand(0).getNode()))
10825     return Result;
10826 
10827   // Check the chain and pointer.
10828   LoadSDNode *LD = cast<LoadSDNode>(V->getOperand(0));
10829   if (LD->getBasePtr() != Ptr) return Result;  // Not from same pointer.
10830 
10831   // The store should be chained directly to the load or be an operand of a
10832   // tokenfactor.
10833   if (LD == Chain.getNode())
10834     ; // ok.
10835   else if (Chain->getOpcode() != ISD::TokenFactor)
10836     return Result; // Fail.
10837   else {
10838     bool isOk = false;
10839     for (const SDValue &ChainOp : Chain->op_values())
10840       if (ChainOp.getNode() == LD) {
10841         isOk = true;
10842         break;
10843       }
10844     if (!isOk) return Result;
10845   }
10846 
10847   // This only handles simple types.
10848   if (V.getValueType() != MVT::i16 &&
10849       V.getValueType() != MVT::i32 &&
10850       V.getValueType() != MVT::i64)
10851     return Result;
10852 
10853   // Check the constant mask.  Invert it so that the bits being masked out are
10854   // 0 and the bits being kept are 1.  Use getSExtValue so that leading bits
10855   // follow the sign bit for uniformity.
10856   uint64_t NotMask = ~cast<ConstantSDNode>(V->getOperand(1))->getSExtValue();
10857   unsigned NotMaskLZ = countLeadingZeros(NotMask);
10858   if (NotMaskLZ & 7) return Result;  // Must be multiple of a byte.
10859   unsigned NotMaskTZ = countTrailingZeros(NotMask);
10860   if (NotMaskTZ & 7) return Result;  // Must be multiple of a byte.
10861   if (NotMaskLZ == 64) return Result;  // All zero mask.
10862 
10863   // See if we have a continuous run of bits.  If so, we have 0*1+0*
10864   if (countTrailingOnes(NotMask >> NotMaskTZ) + NotMaskTZ + NotMaskLZ != 64)
10865     return Result;
10866 
10867   // Adjust NotMaskLZ down to be from the actual size of the int instead of i64.
10868   if (V.getValueType() != MVT::i64 && NotMaskLZ)
10869     NotMaskLZ -= 64-V.getValueSizeInBits();
10870 
10871   unsigned MaskedBytes = (V.getValueSizeInBits()-NotMaskLZ-NotMaskTZ)/8;
10872   switch (MaskedBytes) {
10873   case 1:
10874   case 2:
10875   case 4: break;
10876   default: return Result; // All one mask, or 5-byte mask.
10877   }
10878 
10879   // Verify that the first bit starts at a multiple of mask so that the access
10880   // is aligned the same as the access width.
10881   if (NotMaskTZ && NotMaskTZ/8 % MaskedBytes) return Result;
10882 
10883   Result.first = MaskedBytes;
10884   Result.second = NotMaskTZ/8;
10885   return Result;
10886 }
10887 
10888 
10889 /// Check to see if IVal is something that provides a value as specified by
10890 /// MaskInfo. If so, replace the specified store with a narrower store of
10891 /// truncated IVal.
10892 static SDNode *
10893 ShrinkLoadReplaceStoreWithStore(const std::pair<unsigned, unsigned> &MaskInfo,
10894                                 SDValue IVal, StoreSDNode *St,
10895                                 DAGCombiner *DC) {
10896   unsigned NumBytes = MaskInfo.first;
10897   unsigned ByteShift = MaskInfo.second;
10898   SelectionDAG &DAG = DC->getDAG();
10899 
10900   // Check to see if IVal is all zeros in the part being masked in by the 'or'
10901   // that uses this.  If not, this is not a replacement.
10902   APInt Mask = ~APInt::getBitsSet(IVal.getValueSizeInBits(),
10903                                   ByteShift*8, (ByteShift+NumBytes)*8);
10904   if (!DAG.MaskedValueIsZero(IVal, Mask)) return nullptr;
10905 
10906   // Check that it is legal on the target to do this.  It is legal if the new
10907   // VT we're shrinking to (i8/i16/i32) is legal or we're still before type
10908   // legalization.
10909   MVT VT = MVT::getIntegerVT(NumBytes*8);
10910   if (!DC->isTypeLegal(VT))
10911     return nullptr;
10912 
10913   // Okay, we can do this!  Replace the 'St' store with a store of IVal that is
10914   // shifted by ByteShift and truncated down to NumBytes.
10915   if (ByteShift) {
10916     SDLoc DL(IVal);
10917     IVal = DAG.getNode(ISD::SRL, DL, IVal.getValueType(), IVal,
10918                        DAG.getConstant(ByteShift*8, DL,
10919                                     DC->getShiftAmountTy(IVal.getValueType())));
10920   }
10921 
10922   // Figure out the offset for the store and the alignment of the access.
10923   unsigned StOffset;
10924   unsigned NewAlign = St->getAlignment();
10925 
10926   if (DAG.getDataLayout().isLittleEndian())
10927     StOffset = ByteShift;
10928   else
10929     StOffset = IVal.getValueType().getStoreSize() - ByteShift - NumBytes;
10930 
10931   SDValue Ptr = St->getBasePtr();
10932   if (StOffset) {
10933     SDLoc DL(IVal);
10934     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(),
10935                       Ptr, DAG.getConstant(StOffset, DL, Ptr.getValueType()));
10936     NewAlign = MinAlign(NewAlign, StOffset);
10937   }
10938 
10939   // Truncate down to the new size.
10940   IVal = DAG.getNode(ISD::TRUNCATE, SDLoc(IVal), VT, IVal);
10941 
10942   ++OpsNarrowed;
10943   return DAG
10944       .getStore(St->getChain(), SDLoc(St), IVal, Ptr,
10945                 St->getPointerInfo().getWithOffset(StOffset), NewAlign)
10946       .getNode();
10947 }
10948 
10949 
10950 /// Look for sequence of load / op / store where op is one of 'or', 'xor', and
10951 /// 'and' of immediates. If 'op' is only touching some of the loaded bits, try
10952 /// narrowing the load and store if it would end up being a win for performance
10953 /// or code size.
10954 SDValue DAGCombiner::ReduceLoadOpStoreWidth(SDNode *N) {
10955   StoreSDNode *ST  = cast<StoreSDNode>(N);
10956   if (ST->isVolatile())
10957     return SDValue();
10958 
10959   SDValue Chain = ST->getChain();
10960   SDValue Value = ST->getValue();
10961   SDValue Ptr   = ST->getBasePtr();
10962   EVT VT = Value.getValueType();
10963 
10964   if (ST->isTruncatingStore() || VT.isVector() || !Value.hasOneUse())
10965     return SDValue();
10966 
10967   unsigned Opc = Value.getOpcode();
10968 
10969   // If this is "store (or X, Y), P" and X is "(and (load P), cst)", where cst
10970   // is a byte mask indicating a consecutive number of bytes, check to see if
10971   // Y is known to provide just those bytes.  If so, we try to replace the
10972   // load + replace + store sequence with a single (narrower) store, which makes
10973   // the load dead.
10974   if (Opc == ISD::OR) {
10975     std::pair<unsigned, unsigned> MaskedLoad;
10976     MaskedLoad = CheckForMaskedLoad(Value.getOperand(0), Ptr, Chain);
10977     if (MaskedLoad.first)
10978       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
10979                                                   Value.getOperand(1), ST,this))
10980         return SDValue(NewST, 0);
10981 
10982     // Or is commutative, so try swapping X and Y.
10983     MaskedLoad = CheckForMaskedLoad(Value.getOperand(1), Ptr, Chain);
10984     if (MaskedLoad.first)
10985       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
10986                                                   Value.getOperand(0), ST,this))
10987         return SDValue(NewST, 0);
10988   }
10989 
10990   if ((Opc != ISD::OR && Opc != ISD::XOR && Opc != ISD::AND) ||
10991       Value.getOperand(1).getOpcode() != ISD::Constant)
10992     return SDValue();
10993 
10994   SDValue N0 = Value.getOperand(0);
10995   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
10996       Chain == SDValue(N0.getNode(), 1)) {
10997     LoadSDNode *LD = cast<LoadSDNode>(N0);
10998     if (LD->getBasePtr() != Ptr ||
10999         LD->getPointerInfo().getAddrSpace() !=
11000         ST->getPointerInfo().getAddrSpace())
11001       return SDValue();
11002 
11003     // Find the type to narrow it the load / op / store to.
11004     SDValue N1 = Value.getOperand(1);
11005     unsigned BitWidth = N1.getValueSizeInBits();
11006     APInt Imm = cast<ConstantSDNode>(N1)->getAPIntValue();
11007     if (Opc == ISD::AND)
11008       Imm ^= APInt::getAllOnesValue(BitWidth);
11009     if (Imm == 0 || Imm.isAllOnesValue())
11010       return SDValue();
11011     unsigned ShAmt = Imm.countTrailingZeros();
11012     unsigned MSB = BitWidth - Imm.countLeadingZeros() - 1;
11013     unsigned NewBW = NextPowerOf2(MSB - ShAmt);
11014     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
11015     // The narrowing should be profitable, the load/store operation should be
11016     // legal (or custom) and the store size should be equal to the NewVT width.
11017     while (NewBW < BitWidth &&
11018            (NewVT.getStoreSizeInBits() != NewBW ||
11019             !TLI.isOperationLegalOrCustom(Opc, NewVT) ||
11020             !TLI.isNarrowingProfitable(VT, NewVT))) {
11021       NewBW = NextPowerOf2(NewBW);
11022       NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
11023     }
11024     if (NewBW >= BitWidth)
11025       return SDValue();
11026 
11027     // If the lsb changed does not start at the type bitwidth boundary,
11028     // start at the previous one.
11029     if (ShAmt % NewBW)
11030       ShAmt = (((ShAmt + NewBW - 1) / NewBW) * NewBW) - NewBW;
11031     APInt Mask = APInt::getBitsSet(BitWidth, ShAmt,
11032                                    std::min(BitWidth, ShAmt + NewBW));
11033     if ((Imm & Mask) == Imm) {
11034       APInt NewImm = (Imm & Mask).lshr(ShAmt).trunc(NewBW);
11035       if (Opc == ISD::AND)
11036         NewImm ^= APInt::getAllOnesValue(NewBW);
11037       uint64_t PtrOff = ShAmt / 8;
11038       // For big endian targets, we need to adjust the offset to the pointer to
11039       // load the correct bytes.
11040       if (DAG.getDataLayout().isBigEndian())
11041         PtrOff = (BitWidth + 7 - NewBW) / 8 - PtrOff;
11042 
11043       unsigned NewAlign = MinAlign(LD->getAlignment(), PtrOff);
11044       Type *NewVTTy = NewVT.getTypeForEVT(*DAG.getContext());
11045       if (NewAlign < DAG.getDataLayout().getABITypeAlignment(NewVTTy))
11046         return SDValue();
11047 
11048       SDValue NewPtr = DAG.getNode(ISD::ADD, SDLoc(LD),
11049                                    Ptr.getValueType(), Ptr,
11050                                    DAG.getConstant(PtrOff, SDLoc(LD),
11051                                                    Ptr.getValueType()));
11052       SDValue NewLD =
11053           DAG.getLoad(NewVT, SDLoc(N0), LD->getChain(), NewPtr,
11054                       LD->getPointerInfo().getWithOffset(PtrOff), NewAlign,
11055                       LD->getMemOperand()->getFlags(), LD->getAAInfo());
11056       SDValue NewVal = DAG.getNode(Opc, SDLoc(Value), NewVT, NewLD,
11057                                    DAG.getConstant(NewImm, SDLoc(Value),
11058                                                    NewVT));
11059       SDValue NewST =
11060           DAG.getStore(Chain, SDLoc(N), NewVal, NewPtr,
11061                        ST->getPointerInfo().getWithOffset(PtrOff), NewAlign);
11062 
11063       AddToWorklist(NewPtr.getNode());
11064       AddToWorklist(NewLD.getNode());
11065       AddToWorklist(NewVal.getNode());
11066       WorklistRemover DeadNodes(*this);
11067       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLD.getValue(1));
11068       ++OpsNarrowed;
11069       return NewST;
11070     }
11071   }
11072 
11073   return SDValue();
11074 }
11075 
11076 /// For a given floating point load / store pair, if the load value isn't used
11077 /// by any other operations, then consider transforming the pair to integer
11078 /// load / store operations if the target deems the transformation profitable.
11079 SDValue DAGCombiner::TransformFPLoadStorePair(SDNode *N) {
11080   StoreSDNode *ST  = cast<StoreSDNode>(N);
11081   SDValue Chain = ST->getChain();
11082   SDValue Value = ST->getValue();
11083   if (ISD::isNormalStore(ST) && ISD::isNormalLoad(Value.getNode()) &&
11084       Value.hasOneUse() &&
11085       Chain == SDValue(Value.getNode(), 1)) {
11086     LoadSDNode *LD = cast<LoadSDNode>(Value);
11087     EVT VT = LD->getMemoryVT();
11088     if (!VT.isFloatingPoint() ||
11089         VT != ST->getMemoryVT() ||
11090         LD->isNonTemporal() ||
11091         ST->isNonTemporal() ||
11092         LD->getPointerInfo().getAddrSpace() != 0 ||
11093         ST->getPointerInfo().getAddrSpace() != 0)
11094       return SDValue();
11095 
11096     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits());
11097     if (!TLI.isOperationLegal(ISD::LOAD, IntVT) ||
11098         !TLI.isOperationLegal(ISD::STORE, IntVT) ||
11099         !TLI.isDesirableToTransformToIntegerOp(ISD::LOAD, VT) ||
11100         !TLI.isDesirableToTransformToIntegerOp(ISD::STORE, VT))
11101       return SDValue();
11102 
11103     unsigned LDAlign = LD->getAlignment();
11104     unsigned STAlign = ST->getAlignment();
11105     Type *IntVTTy = IntVT.getTypeForEVT(*DAG.getContext());
11106     unsigned ABIAlign = DAG.getDataLayout().getABITypeAlignment(IntVTTy);
11107     if (LDAlign < ABIAlign || STAlign < ABIAlign)
11108       return SDValue();
11109 
11110     SDValue NewLD =
11111         DAG.getLoad(IntVT, SDLoc(Value), LD->getChain(), LD->getBasePtr(),
11112                     LD->getPointerInfo(), LDAlign);
11113 
11114     SDValue NewST =
11115         DAG.getStore(NewLD.getValue(1), SDLoc(N), NewLD, ST->getBasePtr(),
11116                      ST->getPointerInfo(), STAlign);
11117 
11118     AddToWorklist(NewLD.getNode());
11119     AddToWorklist(NewST.getNode());
11120     WorklistRemover DeadNodes(*this);
11121     DAG.ReplaceAllUsesOfValueWith(Value.getValue(1), NewLD.getValue(1));
11122     ++LdStFP2Int;
11123     return NewST;
11124   }
11125 
11126   return SDValue();
11127 }
11128 
11129 namespace {
11130 /// Helper struct to parse and store a memory address as base + index + offset.
11131 /// We ignore sign extensions when it is safe to do so.
11132 /// The following two expressions are not equivalent. To differentiate we need
11133 /// to store whether there was a sign extension involved in the index
11134 /// computation.
11135 ///  (load (i64 add (i64 copyfromreg %c)
11136 ///                 (i64 signextend (add (i8 load %index)
11137 ///                                      (i8 1))))
11138 /// vs
11139 ///
11140 /// (load (i64 add (i64 copyfromreg %c)
11141 ///                (i64 signextend (i32 add (i32 signextend (i8 load %index))
11142 ///                                         (i32 1)))))
11143 struct BaseIndexOffset {
11144   SDValue Base;
11145   SDValue Index;
11146   int64_t Offset;
11147   bool IsIndexSignExt;
11148 
11149   BaseIndexOffset() : Offset(0), IsIndexSignExt(false) {}
11150 
11151   BaseIndexOffset(SDValue Base, SDValue Index, int64_t Offset,
11152                   bool IsIndexSignExt) :
11153     Base(Base), Index(Index), Offset(Offset), IsIndexSignExt(IsIndexSignExt) {}
11154 
11155   bool equalBaseIndex(const BaseIndexOffset &Other) {
11156     return Other.Base == Base && Other.Index == Index &&
11157       Other.IsIndexSignExt == IsIndexSignExt;
11158   }
11159 
11160   /// Parses tree in Ptr for base, index, offset addresses.
11161   static BaseIndexOffset match(SDValue Ptr, SelectionDAG &DAG) {
11162     bool IsIndexSignExt = false;
11163 
11164     // Split up a folded GlobalAddress+Offset into its component parts.
11165     if (GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(Ptr))
11166       if (GA->getOpcode() == ISD::GlobalAddress && GA->getOffset() != 0) {
11167         return BaseIndexOffset(DAG.getGlobalAddress(GA->getGlobal(),
11168                                                     SDLoc(GA),
11169                                                     GA->getValueType(0),
11170                                                     /*Offset=*/0,
11171                                                     /*isTargetGA=*/false,
11172                                                     GA->getTargetFlags()),
11173                                SDValue(),
11174                                GA->getOffset(),
11175                                IsIndexSignExt);
11176       }
11177 
11178     // We only can pattern match BASE + INDEX + OFFSET. If Ptr is not an ADD
11179     // instruction, then it could be just the BASE or everything else we don't
11180     // know how to handle. Just use Ptr as BASE and give up.
11181     if (Ptr->getOpcode() != ISD::ADD)
11182       return BaseIndexOffset(Ptr, SDValue(), 0, IsIndexSignExt);
11183 
11184     // We know that we have at least an ADD instruction. Try to pattern match
11185     // the simple case of BASE + OFFSET.
11186     if (isa<ConstantSDNode>(Ptr->getOperand(1))) {
11187       int64_t Offset = cast<ConstantSDNode>(Ptr->getOperand(1))->getSExtValue();
11188       return  BaseIndexOffset(Ptr->getOperand(0), SDValue(), Offset,
11189                               IsIndexSignExt);
11190     }
11191 
11192     // Inside a loop the current BASE pointer is calculated using an ADD and a
11193     // MUL instruction. In this case Ptr is the actual BASE pointer.
11194     // (i64 add (i64 %array_ptr)
11195     //          (i64 mul (i64 %induction_var)
11196     //                   (i64 %element_size)))
11197     if (Ptr->getOperand(1)->getOpcode() == ISD::MUL)
11198       return BaseIndexOffset(Ptr, SDValue(), 0, IsIndexSignExt);
11199 
11200     // Look at Base + Index + Offset cases.
11201     SDValue Base = Ptr->getOperand(0);
11202     SDValue IndexOffset = Ptr->getOperand(1);
11203 
11204     // Skip signextends.
11205     if (IndexOffset->getOpcode() == ISD::SIGN_EXTEND) {
11206       IndexOffset = IndexOffset->getOperand(0);
11207       IsIndexSignExt = true;
11208     }
11209 
11210     // Either the case of Base + Index (no offset) or something else.
11211     if (IndexOffset->getOpcode() != ISD::ADD)
11212       return BaseIndexOffset(Base, IndexOffset, 0, IsIndexSignExt);
11213 
11214     // Now we have the case of Base + Index + offset.
11215     SDValue Index = IndexOffset->getOperand(0);
11216     SDValue Offset = IndexOffset->getOperand(1);
11217 
11218     if (!isa<ConstantSDNode>(Offset))
11219       return BaseIndexOffset(Ptr, SDValue(), 0, IsIndexSignExt);
11220 
11221     // Ignore signextends.
11222     if (Index->getOpcode() == ISD::SIGN_EXTEND) {
11223       Index = Index->getOperand(0);
11224       IsIndexSignExt = true;
11225     } else IsIndexSignExt = false;
11226 
11227     int64_t Off = cast<ConstantSDNode>(Offset)->getSExtValue();
11228     return BaseIndexOffset(Base, Index, Off, IsIndexSignExt);
11229   }
11230 };
11231 } // namespace
11232 
11233 // This is a helper function for visitMUL to check the profitability
11234 // of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
11235 // MulNode is the original multiply, AddNode is (add x, c1),
11236 // and ConstNode is c2.
11237 //
11238 // If the (add x, c1) has multiple uses, we could increase
11239 // the number of adds if we make this transformation.
11240 // It would only be worth doing this if we can remove a
11241 // multiply in the process. Check for that here.
11242 // To illustrate:
11243 //     (A + c1) * c3
11244 //     (A + c2) * c3
11245 // We're checking for cases where we have common "c3 * A" expressions.
11246 bool DAGCombiner::isMulAddWithConstProfitable(SDNode *MulNode,
11247                                               SDValue &AddNode,
11248                                               SDValue &ConstNode) {
11249   APInt Val;
11250 
11251   // If the add only has one use, this would be OK to do.
11252   if (AddNode.getNode()->hasOneUse())
11253     return true;
11254 
11255   // Walk all the users of the constant with which we're multiplying.
11256   for (SDNode *Use : ConstNode->uses()) {
11257 
11258     if (Use == MulNode) // This use is the one we're on right now. Skip it.
11259       continue;
11260 
11261     if (Use->getOpcode() == ISD::MUL) { // We have another multiply use.
11262       SDNode *OtherOp;
11263       SDNode *MulVar = AddNode.getOperand(0).getNode();
11264 
11265       // OtherOp is what we're multiplying against the constant.
11266       if (Use->getOperand(0) == ConstNode)
11267         OtherOp = Use->getOperand(1).getNode();
11268       else
11269         OtherOp = Use->getOperand(0).getNode();
11270 
11271       // Check to see if multiply is with the same operand of our "add".
11272       //
11273       //     ConstNode  = CONST
11274       //     Use = ConstNode * A  <-- visiting Use. OtherOp is A.
11275       //     ...
11276       //     AddNode  = (A + c1)  <-- MulVar is A.
11277       //         = AddNode * ConstNode   <-- current visiting instruction.
11278       //
11279       // If we make this transformation, we will have a common
11280       // multiply (ConstNode * A) that we can save.
11281       if (OtherOp == MulVar)
11282         return true;
11283 
11284       // Now check to see if a future expansion will give us a common
11285       // multiply.
11286       //
11287       //     ConstNode  = CONST
11288       //     AddNode    = (A + c1)
11289       //     ...   = AddNode * ConstNode <-- current visiting instruction.
11290       //     ...
11291       //     OtherOp = (A + c2)
11292       //     Use     = OtherOp * ConstNode <-- visiting Use.
11293       //
11294       // If we make this transformation, we will have a common
11295       // multiply (CONST * A) after we also do the same transformation
11296       // to the "t2" instruction.
11297       if (OtherOp->getOpcode() == ISD::ADD &&
11298           DAG.isConstantIntBuildVectorOrConstantInt(OtherOp->getOperand(1)) &&
11299           OtherOp->getOperand(0).getNode() == MulVar)
11300         return true;
11301     }
11302   }
11303 
11304   // Didn't find a case where this would be profitable.
11305   return false;
11306 }
11307 
11308 SDValue DAGCombiner::getMergedConstantVectorStore(
11309     SelectionDAG &DAG, const SDLoc &SL, ArrayRef<MemOpLink> Stores,
11310     SmallVectorImpl<SDValue> &Chains, EVT Ty) const {
11311   SmallVector<SDValue, 8> BuildVector;
11312 
11313   for (unsigned I = 0, E = Ty.getVectorNumElements(); I != E; ++I) {
11314     StoreSDNode *St = cast<StoreSDNode>(Stores[I].MemNode);
11315     Chains.push_back(St->getChain());
11316     BuildVector.push_back(St->getValue());
11317   }
11318 
11319   return DAG.getBuildVector(Ty, SL, BuildVector);
11320 }
11321 
11322 bool DAGCombiner::MergeStoresOfConstantsOrVecElts(
11323                   SmallVectorImpl<MemOpLink> &StoreNodes, EVT MemVT,
11324                   unsigned NumStores, bool IsConstantSrc, bool UseVector) {
11325   // Make sure we have something to merge.
11326   if (NumStores < 2)
11327     return false;
11328 
11329   int64_t ElementSizeBytes = MemVT.getSizeInBits() / 8;
11330   LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
11331   unsigned LatestNodeUsed = 0;
11332 
11333   for (unsigned i=0; i < NumStores; ++i) {
11334     // Find a chain for the new wide-store operand. Notice that some
11335     // of the store nodes that we found may not be selected for inclusion
11336     // in the wide store. The chain we use needs to be the chain of the
11337     // latest store node which is *used* and replaced by the wide store.
11338     if (StoreNodes[i].SequenceNum < StoreNodes[LatestNodeUsed].SequenceNum)
11339       LatestNodeUsed = i;
11340   }
11341 
11342   SmallVector<SDValue, 8> Chains;
11343 
11344   // The latest Node in the DAG.
11345   LSBaseSDNode *LatestOp = StoreNodes[LatestNodeUsed].MemNode;
11346   SDLoc DL(StoreNodes[0].MemNode);
11347 
11348   SDValue StoredVal;
11349   if (UseVector) {
11350     bool IsVec = MemVT.isVector();
11351     unsigned Elts = NumStores;
11352     if (IsVec) {
11353       // When merging vector stores, get the total number of elements.
11354       Elts *= MemVT.getVectorNumElements();
11355     }
11356     // Get the type for the merged vector store.
11357     EVT Ty = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
11358     assert(TLI.isTypeLegal(Ty) && "Illegal vector store");
11359 
11360     if (IsConstantSrc) {
11361       StoredVal = getMergedConstantVectorStore(DAG, DL, StoreNodes, Chains, Ty);
11362     } else {
11363       SmallVector<SDValue, 8> Ops;
11364       for (unsigned i = 0; i < NumStores; ++i) {
11365         StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
11366         SDValue Val = St->getValue();
11367         // All operands of BUILD_VECTOR / CONCAT_VECTOR must have the same type.
11368         if (Val.getValueType() != MemVT)
11369           return false;
11370         Ops.push_back(Val);
11371         Chains.push_back(St->getChain());
11372       }
11373 
11374       // Build the extracted vector elements back into a vector.
11375       StoredVal = DAG.getNode(IsVec ? ISD::CONCAT_VECTORS : ISD::BUILD_VECTOR,
11376                               DL, Ty, Ops);    }
11377   } else {
11378     // We should always use a vector store when merging extracted vector
11379     // elements, so this path implies a store of constants.
11380     assert(IsConstantSrc && "Merged vector elements should use vector store");
11381 
11382     unsigned SizeInBits = NumStores * ElementSizeBytes * 8;
11383     APInt StoreInt(SizeInBits, 0);
11384 
11385     // Construct a single integer constant which is made of the smaller
11386     // constant inputs.
11387     bool IsLE = DAG.getDataLayout().isLittleEndian();
11388     for (unsigned i = 0; i < NumStores; ++i) {
11389       unsigned Idx = IsLE ? (NumStores - 1 - i) : i;
11390       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[Idx].MemNode);
11391       Chains.push_back(St->getChain());
11392 
11393       SDValue Val = St->getValue();
11394       StoreInt <<= ElementSizeBytes * 8;
11395       if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Val)) {
11396         StoreInt |= C->getAPIntValue().zext(SizeInBits);
11397       } else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(Val)) {
11398         StoreInt |= C->getValueAPF().bitcastToAPInt().zext(SizeInBits);
11399       } else {
11400         llvm_unreachable("Invalid constant element type");
11401       }
11402     }
11403 
11404     // Create the new Load and Store operations.
11405     EVT StoreTy = EVT::getIntegerVT(*DAG.getContext(), SizeInBits);
11406     StoredVal = DAG.getConstant(StoreInt, DL, StoreTy);
11407   }
11408 
11409   assert(!Chains.empty());
11410 
11411   SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
11412   SDValue NewStore = DAG.getStore(NewChain, DL, StoredVal,
11413                                   FirstInChain->getBasePtr(),
11414                                   FirstInChain->getPointerInfo(),
11415                                   FirstInChain->getAlignment());
11416 
11417   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11418                                                   : DAG.getSubtarget().useAA();
11419   if (UseAA) {
11420     // Replace all merged stores with the new store.
11421     for (unsigned i = 0; i < NumStores; ++i)
11422       CombineTo(StoreNodes[i].MemNode, NewStore);
11423   } else {
11424     // Replace the last store with the new store.
11425     CombineTo(LatestOp, NewStore);
11426     // Erase all other stores.
11427     for (unsigned i = 0; i < NumStores; ++i) {
11428       if (StoreNodes[i].MemNode == LatestOp)
11429         continue;
11430       StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
11431       // ReplaceAllUsesWith will replace all uses that existed when it was
11432       // called, but graph optimizations may cause new ones to appear. For
11433       // example, the case in pr14333 looks like
11434       //
11435       //  St's chain -> St -> another store -> X
11436       //
11437       // And the only difference from St to the other store is the chain.
11438       // When we change it's chain to be St's chain they become identical,
11439       // get CSEed and the net result is that X is now a use of St.
11440       // Since we know that St is redundant, just iterate.
11441       while (!St->use_empty())
11442         DAG.ReplaceAllUsesWith(SDValue(St, 0), St->getChain());
11443       deleteAndRecombine(St);
11444     }
11445   }
11446 
11447   return true;
11448 }
11449 
11450 void DAGCombiner::getStoreMergeAndAliasCandidates(
11451     StoreSDNode* St, SmallVectorImpl<MemOpLink> &StoreNodes,
11452     SmallVectorImpl<LSBaseSDNode*> &AliasLoadNodes) {
11453   // This holds the base pointer, index, and the offset in bytes from the base
11454   // pointer.
11455   BaseIndexOffset BasePtr = BaseIndexOffset::match(St->getBasePtr(), DAG);
11456 
11457   // We must have a base and an offset.
11458   if (!BasePtr.Base.getNode())
11459     return;
11460 
11461   // Do not handle stores to undef base pointers.
11462   if (BasePtr.Base.isUndef())
11463     return;
11464 
11465   // Walk up the chain and look for nodes with offsets from the same
11466   // base pointer. Stop when reaching an instruction with a different kind
11467   // or instruction which has a different base pointer.
11468   EVT MemVT = St->getMemoryVT();
11469   unsigned Seq = 0;
11470   StoreSDNode *Index = St;
11471 
11472 
11473   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11474                                                   : DAG.getSubtarget().useAA();
11475 
11476   if (UseAA) {
11477     // Look at other users of the same chain. Stores on the same chain do not
11478     // alias. If combiner-aa is enabled, non-aliasing stores are canonicalized
11479     // to be on the same chain, so don't bother looking at adjacent chains.
11480 
11481     SDValue Chain = St->getChain();
11482     for (auto I = Chain->use_begin(), E = Chain->use_end(); I != E; ++I) {
11483       if (StoreSDNode *OtherST = dyn_cast<StoreSDNode>(*I)) {
11484         if (I.getOperandNo() != 0)
11485           continue;
11486 
11487         if (OtherST->isVolatile() || OtherST->isIndexed())
11488           continue;
11489 
11490         if (OtherST->getMemoryVT() != MemVT)
11491           continue;
11492 
11493         BaseIndexOffset Ptr = BaseIndexOffset::match(OtherST->getBasePtr(), DAG);
11494 
11495         if (Ptr.equalBaseIndex(BasePtr))
11496           StoreNodes.push_back(MemOpLink(OtherST, Ptr.Offset, Seq++));
11497       }
11498     }
11499 
11500     return;
11501   }
11502 
11503   while (Index) {
11504     // If the chain has more than one use, then we can't reorder the mem ops.
11505     if (Index != St && !SDValue(Index, 0)->hasOneUse())
11506       break;
11507 
11508     // Find the base pointer and offset for this memory node.
11509     BaseIndexOffset Ptr = BaseIndexOffset::match(Index->getBasePtr(), DAG);
11510 
11511     // Check that the base pointer is the same as the original one.
11512     if (!Ptr.equalBaseIndex(BasePtr))
11513       break;
11514 
11515     // The memory operands must not be volatile.
11516     if (Index->isVolatile() || Index->isIndexed())
11517       break;
11518 
11519     // No truncation.
11520     if (Index->isTruncatingStore())
11521       break;
11522 
11523     // The stored memory type must be the same.
11524     if (Index->getMemoryVT() != MemVT)
11525       break;
11526 
11527     // We do not allow under-aligned stores in order to prevent
11528     // overriding stores. NOTE: this is a bad hack. Alignment SHOULD
11529     // be irrelevant here; what MATTERS is that we not move memory
11530     // operations that potentially overlap past each-other.
11531     if (Index->getAlignment() < MemVT.getStoreSize())
11532       break;
11533 
11534     // We found a potential memory operand to merge.
11535     StoreNodes.push_back(MemOpLink(Index, Ptr.Offset, Seq++));
11536 
11537     // Find the next memory operand in the chain. If the next operand in the
11538     // chain is a store then move up and continue the scan with the next
11539     // memory operand. If the next operand is a load save it and use alias
11540     // information to check if it interferes with anything.
11541     SDNode *NextInChain = Index->getChain().getNode();
11542     while (1) {
11543       if (StoreSDNode *STn = dyn_cast<StoreSDNode>(NextInChain)) {
11544         // We found a store node. Use it for the next iteration.
11545         Index = STn;
11546         break;
11547       } else if (LoadSDNode *Ldn = dyn_cast<LoadSDNode>(NextInChain)) {
11548         if (Ldn->isVolatile()) {
11549           Index = nullptr;
11550           break;
11551         }
11552 
11553         // Save the load node for later. Continue the scan.
11554         AliasLoadNodes.push_back(Ldn);
11555         NextInChain = Ldn->getChain().getNode();
11556         continue;
11557       } else {
11558         Index = nullptr;
11559         break;
11560       }
11561     }
11562   }
11563 }
11564 
11565 // We need to check that merging these stores does not cause a loop
11566 // in the DAG. Any store candidate may depend on another candidate
11567 // indirectly through its operand (we already consider dependencies
11568 // through the chain). Check in parallel by searching up from
11569 // non-chain operands of candidates.
11570 bool DAGCombiner::checkMergeStoreCandidatesForDependencies(
11571     SmallVectorImpl<MemOpLink> &StoreNodes) {
11572   SmallPtrSet<const SDNode *, 16> Visited;
11573   SmallVector<const SDNode *, 8> Worklist;
11574   // search ops of store candidates
11575   for (unsigned i = 0; i < StoreNodes.size(); ++i) {
11576     SDNode *n = StoreNodes[i].MemNode;
11577     // Potential loops may happen only through non-chain operands
11578     for (unsigned j = 1; j < n->getNumOperands(); ++j)
11579       Worklist.push_back(n->getOperand(j).getNode());
11580   }
11581   // search through DAG. We can stop early if we find a storenode
11582   for (unsigned i = 0; i < StoreNodes.size(); ++i) {
11583     if (SDNode::hasPredecessorHelper(StoreNodes[i].MemNode, Visited, Worklist))
11584       return false;
11585   }
11586   return true;
11587 }
11588 
11589 bool DAGCombiner::MergeConsecutiveStores(StoreSDNode* St) {
11590   if (OptLevel == CodeGenOpt::None)
11591     return false;
11592 
11593   EVT MemVT = St->getMemoryVT();
11594   int64_t ElementSizeBytes = MemVT.getSizeInBits() / 8;
11595   bool NoVectors = DAG.getMachineFunction().getFunction()->hasFnAttribute(
11596       Attribute::NoImplicitFloat);
11597 
11598   // This function cannot currently deal with non-byte-sized memory sizes.
11599   if (ElementSizeBytes * 8 != MemVT.getSizeInBits())
11600     return false;
11601 
11602   if (!MemVT.isSimple())
11603     return false;
11604 
11605   // Perform an early exit check. Do not bother looking at stored values that
11606   // are not constants, loads, or extracted vector elements.
11607   SDValue StoredVal = St->getValue();
11608   bool IsLoadSrc = isa<LoadSDNode>(StoredVal);
11609   bool IsConstantSrc = isa<ConstantSDNode>(StoredVal) ||
11610                        isa<ConstantFPSDNode>(StoredVal);
11611   bool IsExtractVecSrc = (StoredVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT ||
11612                           StoredVal.getOpcode() == ISD::EXTRACT_SUBVECTOR);
11613 
11614   if (!IsConstantSrc && !IsLoadSrc && !IsExtractVecSrc)
11615     return false;
11616 
11617   // Don't merge vectors into wider vectors if the source data comes from loads.
11618   // TODO: This restriction can be lifted by using logic similar to the
11619   // ExtractVecSrc case.
11620   if (MemVT.isVector() && IsLoadSrc)
11621     return false;
11622 
11623   // Only look at ends of store sequences.
11624   SDValue Chain = SDValue(St, 0);
11625   if (Chain->hasOneUse() && Chain->use_begin()->getOpcode() == ISD::STORE)
11626     return false;
11627 
11628   // Save the LoadSDNodes that we find in the chain.
11629   // We need to make sure that these nodes do not interfere with
11630   // any of the store nodes.
11631   SmallVector<LSBaseSDNode*, 8> AliasLoadNodes;
11632 
11633   // Save the StoreSDNodes that we find in the chain.
11634   SmallVector<MemOpLink, 8> StoreNodes;
11635 
11636   getStoreMergeAndAliasCandidates(St, StoreNodes, AliasLoadNodes);
11637 
11638   // Check if there is anything to merge.
11639   if (StoreNodes.size() < 2)
11640     return false;
11641 
11642   // only do dependence check in AA case
11643   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11644                                                   : DAG.getSubtarget().useAA();
11645   if (UseAA && !checkMergeStoreCandidatesForDependencies(StoreNodes))
11646     return false;
11647 
11648   // Sort the memory operands according to their distance from the
11649   // base pointer.  As a secondary criteria: make sure stores coming
11650   // later in the code come first in the list. This is important for
11651   // the non-UseAA case, because we're merging stores into the FINAL
11652   // store along a chain which potentially contains aliasing stores.
11653   // Thus, if there are multiple stores to the same address, the last
11654   // one can be considered for merging but not the others.
11655   std::sort(StoreNodes.begin(), StoreNodes.end(),
11656             [](MemOpLink LHS, MemOpLink RHS) {
11657     return LHS.OffsetFromBase < RHS.OffsetFromBase ||
11658            (LHS.OffsetFromBase == RHS.OffsetFromBase &&
11659             LHS.SequenceNum < RHS.SequenceNum);
11660   });
11661 
11662   // Scan the memory operations on the chain and find the first non-consecutive
11663   // store memory address.
11664   unsigned LastConsecutiveStore = 0;
11665   int64_t StartAddress = StoreNodes[0].OffsetFromBase;
11666   for (unsigned i = 0, e = StoreNodes.size(); i < e; ++i) {
11667 
11668     // Check that the addresses are consecutive starting from the second
11669     // element in the list of stores.
11670     if (i > 0) {
11671       int64_t CurrAddress = StoreNodes[i].OffsetFromBase;
11672       if (CurrAddress - StartAddress != (ElementSizeBytes * i))
11673         break;
11674     }
11675 
11676     // Check if this store interferes with any of the loads that we found.
11677     // If we find a load that alias with this store. Stop the sequence.
11678     if (any_of(AliasLoadNodes, [&](LSBaseSDNode *Ldn) {
11679           return isAlias(Ldn, StoreNodes[i].MemNode);
11680         }))
11681       break;
11682 
11683     // Mark this node as useful.
11684     LastConsecutiveStore = i;
11685   }
11686 
11687   // The node with the lowest store address.
11688   LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
11689   unsigned FirstStoreAS = FirstInChain->getAddressSpace();
11690   unsigned FirstStoreAlign = FirstInChain->getAlignment();
11691   LLVMContext &Context = *DAG.getContext();
11692   const DataLayout &DL = DAG.getDataLayout();
11693 
11694   // Store the constants into memory as one consecutive store.
11695   if (IsConstantSrc) {
11696     unsigned LastLegalType = 0;
11697     unsigned LastLegalVectorType = 0;
11698     bool NonZero = false;
11699     for (unsigned i=0; i<LastConsecutiveStore+1; ++i) {
11700       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
11701       SDValue StoredVal = St->getValue();
11702 
11703       if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(StoredVal)) {
11704         NonZero |= !C->isNullValue();
11705       } else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(StoredVal)) {
11706         NonZero |= !C->getConstantFPValue()->isNullValue();
11707       } else {
11708         // Non-constant.
11709         break;
11710       }
11711 
11712       // Find a legal type for the constant store.
11713       unsigned SizeInBits = (i+1) * ElementSizeBytes * 8;
11714       EVT StoreTy = EVT::getIntegerVT(Context, SizeInBits);
11715       bool IsFast;
11716       if (TLI.isTypeLegal(StoreTy) &&
11717           TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
11718                                  FirstStoreAlign, &IsFast) && IsFast) {
11719         LastLegalType = i+1;
11720       // Or check whether a truncstore is legal.
11721       } else if (TLI.getTypeAction(Context, StoreTy) ==
11722                  TargetLowering::TypePromoteInteger) {
11723         EVT LegalizedStoredValueTy =
11724           TLI.getTypeToTransformTo(Context, StoredVal.getValueType());
11725         if (TLI.isTruncStoreLegal(LegalizedStoredValueTy, StoreTy) &&
11726             TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
11727                                    FirstStoreAS, FirstStoreAlign, &IsFast) &&
11728             IsFast) {
11729           LastLegalType = i + 1;
11730         }
11731       }
11732 
11733       // We only use vectors if the constant is known to be zero or the target
11734       // allows it and the function is not marked with the noimplicitfloat
11735       // attribute.
11736       if ((!NonZero || TLI.storeOfVectorConstantIsCheap(MemVT, i+1,
11737                                                         FirstStoreAS)) &&
11738           !NoVectors) {
11739         // Find a legal type for the vector store.
11740         EVT Ty = EVT::getVectorVT(Context, MemVT, i+1);
11741         if (TLI.isTypeLegal(Ty) &&
11742             TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS,
11743                                    FirstStoreAlign, &IsFast) && IsFast)
11744           LastLegalVectorType = i + 1;
11745       }
11746     }
11747 
11748     // Check if we found a legal integer type to store.
11749     if (LastLegalType == 0 && LastLegalVectorType == 0)
11750       return false;
11751 
11752     bool UseVector = (LastLegalVectorType > LastLegalType) && !NoVectors;
11753     unsigned NumElem = UseVector ? LastLegalVectorType : LastLegalType;
11754 
11755     return MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumElem,
11756                                            true, UseVector);
11757   }
11758 
11759   // When extracting multiple vector elements, try to store them
11760   // in one vector store rather than a sequence of scalar stores.
11761   if (IsExtractVecSrc) {
11762     unsigned NumStoresToMerge = 0;
11763     bool IsVec = MemVT.isVector();
11764     for (unsigned i = 0; i < LastConsecutiveStore + 1; ++i) {
11765       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
11766       unsigned StoreValOpcode = St->getValue().getOpcode();
11767       // This restriction could be loosened.
11768       // Bail out if any stored values are not elements extracted from a vector.
11769       // It should be possible to handle mixed sources, but load sources need
11770       // more careful handling (see the block of code below that handles
11771       // consecutive loads).
11772       if (StoreValOpcode != ISD::EXTRACT_VECTOR_ELT &&
11773           StoreValOpcode != ISD::EXTRACT_SUBVECTOR)
11774         return false;
11775 
11776       // Find a legal type for the vector store.
11777       unsigned Elts = i + 1;
11778       if (IsVec) {
11779         // When merging vector stores, get the total number of elements.
11780         Elts *= MemVT.getVectorNumElements();
11781       }
11782       EVT Ty = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
11783       bool IsFast;
11784       if (TLI.isTypeLegal(Ty) &&
11785           TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS,
11786                                  FirstStoreAlign, &IsFast) && IsFast)
11787         NumStoresToMerge = i + 1;
11788     }
11789 
11790     return MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumStoresToMerge,
11791                                            false, true);
11792   }
11793 
11794   // Below we handle the case of multiple consecutive stores that
11795   // come from multiple consecutive loads. We merge them into a single
11796   // wide load and a single wide store.
11797 
11798   // Look for load nodes which are used by the stored values.
11799   SmallVector<MemOpLink, 8> LoadNodes;
11800 
11801   // Find acceptable loads. Loads need to have the same chain (token factor),
11802   // must not be zext, volatile, indexed, and they must be consecutive.
11803   BaseIndexOffset LdBasePtr;
11804   for (unsigned i=0; i<LastConsecutiveStore+1; ++i) {
11805     StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
11806     LoadSDNode *Ld = dyn_cast<LoadSDNode>(St->getValue());
11807     if (!Ld) break;
11808 
11809     // Loads must only have one use.
11810     if (!Ld->hasNUsesOfValue(1, 0))
11811       break;
11812 
11813     // The memory operands must not be volatile.
11814     if (Ld->isVolatile() || Ld->isIndexed())
11815       break;
11816 
11817     // We do not accept ext loads.
11818     if (Ld->getExtensionType() != ISD::NON_EXTLOAD)
11819       break;
11820 
11821     // The stored memory type must be the same.
11822     if (Ld->getMemoryVT() != MemVT)
11823       break;
11824 
11825     BaseIndexOffset LdPtr = BaseIndexOffset::match(Ld->getBasePtr(), DAG);
11826     // If this is not the first ptr that we check.
11827     if (LdBasePtr.Base.getNode()) {
11828       // The base ptr must be the same.
11829       if (!LdPtr.equalBaseIndex(LdBasePtr))
11830         break;
11831     } else {
11832       // Check that all other base pointers are the same as this one.
11833       LdBasePtr = LdPtr;
11834     }
11835 
11836     // We found a potential memory operand to merge.
11837     LoadNodes.push_back(MemOpLink(Ld, LdPtr.Offset, 0));
11838   }
11839 
11840   if (LoadNodes.size() < 2)
11841     return false;
11842 
11843   // If we have load/store pair instructions and we only have two values,
11844   // don't bother.
11845   unsigned RequiredAlignment;
11846   if (LoadNodes.size() == 2 && TLI.hasPairedLoad(MemVT, RequiredAlignment) &&
11847       St->getAlignment() >= RequiredAlignment)
11848     return false;
11849 
11850   LoadSDNode *FirstLoad = cast<LoadSDNode>(LoadNodes[0].MemNode);
11851   unsigned FirstLoadAS = FirstLoad->getAddressSpace();
11852   unsigned FirstLoadAlign = FirstLoad->getAlignment();
11853 
11854   // Scan the memory operations on the chain and find the first non-consecutive
11855   // load memory address. These variables hold the index in the store node
11856   // array.
11857   unsigned LastConsecutiveLoad = 0;
11858   // This variable refers to the size and not index in the array.
11859   unsigned LastLegalVectorType = 0;
11860   unsigned LastLegalIntegerType = 0;
11861   StartAddress = LoadNodes[0].OffsetFromBase;
11862   SDValue FirstChain = FirstLoad->getChain();
11863   for (unsigned i = 1; i < LoadNodes.size(); ++i) {
11864     // All loads must share the same chain.
11865     if (LoadNodes[i].MemNode->getChain() != FirstChain)
11866       break;
11867 
11868     int64_t CurrAddress = LoadNodes[i].OffsetFromBase;
11869     if (CurrAddress - StartAddress != (ElementSizeBytes * i))
11870       break;
11871     LastConsecutiveLoad = i;
11872     // Find a legal type for the vector store.
11873     EVT StoreTy = EVT::getVectorVT(Context, MemVT, i+1);
11874     bool IsFastSt, IsFastLd;
11875     if (TLI.isTypeLegal(StoreTy) &&
11876         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
11877                                FirstStoreAlign, &IsFastSt) && IsFastSt &&
11878         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS,
11879                                FirstLoadAlign, &IsFastLd) && IsFastLd) {
11880       LastLegalVectorType = i + 1;
11881     }
11882 
11883     // Find a legal type for the integer store.
11884     unsigned SizeInBits = (i+1) * ElementSizeBytes * 8;
11885     StoreTy = EVT::getIntegerVT(Context, SizeInBits);
11886     if (TLI.isTypeLegal(StoreTy) &&
11887         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
11888                                FirstStoreAlign, &IsFastSt) && IsFastSt &&
11889         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS,
11890                                FirstLoadAlign, &IsFastLd) && IsFastLd)
11891       LastLegalIntegerType = i + 1;
11892     // Or check whether a truncstore and extload is legal.
11893     else if (TLI.getTypeAction(Context, StoreTy) ==
11894              TargetLowering::TypePromoteInteger) {
11895       EVT LegalizedStoredValueTy =
11896         TLI.getTypeToTransformTo(Context, StoreTy);
11897       if (TLI.isTruncStoreLegal(LegalizedStoredValueTy, StoreTy) &&
11898           TLI.isLoadExtLegal(ISD::ZEXTLOAD, LegalizedStoredValueTy, StoreTy) &&
11899           TLI.isLoadExtLegal(ISD::SEXTLOAD, LegalizedStoredValueTy, StoreTy) &&
11900           TLI.isLoadExtLegal(ISD::EXTLOAD, LegalizedStoredValueTy, StoreTy) &&
11901           TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
11902                                  FirstStoreAS, FirstStoreAlign, &IsFastSt) &&
11903           IsFastSt &&
11904           TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
11905                                  FirstLoadAS, FirstLoadAlign, &IsFastLd) &&
11906           IsFastLd)
11907         LastLegalIntegerType = i+1;
11908     }
11909   }
11910 
11911   // Only use vector types if the vector type is larger than the integer type.
11912   // If they are the same, use integers.
11913   bool UseVectorTy = LastLegalVectorType > LastLegalIntegerType && !NoVectors;
11914   unsigned LastLegalType = std::max(LastLegalVectorType, LastLegalIntegerType);
11915 
11916   // We add +1 here because the LastXXX variables refer to location while
11917   // the NumElem refers to array/index size.
11918   unsigned NumElem = std::min(LastConsecutiveStore, LastConsecutiveLoad) + 1;
11919   NumElem = std::min(LastLegalType, NumElem);
11920 
11921   if (NumElem < 2)
11922     return false;
11923 
11924   // Collect the chains from all merged stores.
11925   SmallVector<SDValue, 8> MergeStoreChains;
11926   MergeStoreChains.push_back(StoreNodes[0].MemNode->getChain());
11927 
11928   // The latest Node in the DAG.
11929   unsigned LatestNodeUsed = 0;
11930   for (unsigned i=1; i<NumElem; ++i) {
11931     // Find a chain for the new wide-store operand. Notice that some
11932     // of the store nodes that we found may not be selected for inclusion
11933     // in the wide store. The chain we use needs to be the chain of the
11934     // latest store node which is *used* and replaced by the wide store.
11935     if (StoreNodes[i].SequenceNum < StoreNodes[LatestNodeUsed].SequenceNum)
11936       LatestNodeUsed = i;
11937 
11938     MergeStoreChains.push_back(StoreNodes[i].MemNode->getChain());
11939   }
11940 
11941   LSBaseSDNode *LatestOp = StoreNodes[LatestNodeUsed].MemNode;
11942 
11943   // Find if it is better to use vectors or integers to load and store
11944   // to memory.
11945   EVT JointMemOpVT;
11946   if (UseVectorTy) {
11947     JointMemOpVT = EVT::getVectorVT(Context, MemVT, NumElem);
11948   } else {
11949     unsigned SizeInBits = NumElem * ElementSizeBytes * 8;
11950     JointMemOpVT = EVT::getIntegerVT(Context, SizeInBits);
11951   }
11952 
11953   SDLoc LoadDL(LoadNodes[0].MemNode);
11954   SDLoc StoreDL(StoreNodes[0].MemNode);
11955 
11956   // The merged loads are required to have the same incoming chain, so
11957   // using the first's chain is acceptable.
11958   SDValue NewLoad = DAG.getLoad(JointMemOpVT, LoadDL, FirstLoad->getChain(),
11959                                 FirstLoad->getBasePtr(),
11960                                 FirstLoad->getPointerInfo(), FirstLoadAlign);
11961 
11962   SDValue NewStoreChain =
11963     DAG.getNode(ISD::TokenFactor, StoreDL, MVT::Other, MergeStoreChains);
11964 
11965   SDValue NewStore =
11966       DAG.getStore(NewStoreChain, StoreDL, NewLoad, FirstInChain->getBasePtr(),
11967                    FirstInChain->getPointerInfo(), FirstStoreAlign);
11968 
11969   // Transfer chain users from old loads to the new load.
11970   for (unsigned i = 0; i < NumElem; ++i) {
11971     LoadSDNode *Ld = cast<LoadSDNode>(LoadNodes[i].MemNode);
11972     DAG.ReplaceAllUsesOfValueWith(SDValue(Ld, 1),
11973                                   SDValue(NewLoad.getNode(), 1));
11974   }
11975 
11976   if (UseAA) {
11977     // Replace the all stores with the new store.
11978     for (unsigned i = 0; i < NumElem; ++i)
11979       CombineTo(StoreNodes[i].MemNode, NewStore);
11980   } else {
11981     // Replace the last store with the new store.
11982     CombineTo(LatestOp, NewStore);
11983     // Erase all other stores.
11984     for (unsigned i = 0; i < NumElem; ++i) {
11985       // Remove all Store nodes.
11986       if (StoreNodes[i].MemNode == LatestOp)
11987         continue;
11988       StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
11989       DAG.ReplaceAllUsesOfValueWith(SDValue(St, 0), St->getChain());
11990       deleteAndRecombine(St);
11991     }
11992   }
11993 
11994   return true;
11995 }
11996 
11997 SDValue DAGCombiner::replaceStoreChain(StoreSDNode *ST, SDValue BetterChain) {
11998   SDLoc SL(ST);
11999   SDValue ReplStore;
12000 
12001   // Replace the chain to avoid dependency.
12002   if (ST->isTruncatingStore()) {
12003     ReplStore = DAG.getTruncStore(BetterChain, SL, ST->getValue(),
12004                                   ST->getBasePtr(), ST->getMemoryVT(),
12005                                   ST->getMemOperand());
12006   } else {
12007     ReplStore = DAG.getStore(BetterChain, SL, ST->getValue(), ST->getBasePtr(),
12008                              ST->getMemOperand());
12009   }
12010 
12011   // Create token to keep both nodes around.
12012   SDValue Token = DAG.getNode(ISD::TokenFactor, SL,
12013                               MVT::Other, ST->getChain(), ReplStore);
12014 
12015   // Make sure the new and old chains are cleaned up.
12016   AddToWorklist(Token.getNode());
12017 
12018   // Don't add users to work list.
12019   return CombineTo(ST, Token, false);
12020 }
12021 
12022 SDValue DAGCombiner::replaceStoreOfFPConstant(StoreSDNode *ST) {
12023   SDValue Value = ST->getValue();
12024   if (Value.getOpcode() == ISD::TargetConstantFP)
12025     return SDValue();
12026 
12027   SDLoc DL(ST);
12028 
12029   SDValue Chain = ST->getChain();
12030   SDValue Ptr = ST->getBasePtr();
12031 
12032   const ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Value);
12033 
12034   // NOTE: If the original store is volatile, this transform must not increase
12035   // the number of stores.  For example, on x86-32 an f64 can be stored in one
12036   // processor operation but an i64 (which is not legal) requires two.  So the
12037   // transform should not be done in this case.
12038 
12039   SDValue Tmp;
12040   switch (CFP->getSimpleValueType(0).SimpleTy) {
12041   default:
12042     llvm_unreachable("Unknown FP type");
12043   case MVT::f16:    // We don't do this for these yet.
12044   case MVT::f80:
12045   case MVT::f128:
12046   case MVT::ppcf128:
12047     return SDValue();
12048   case MVT::f32:
12049     if ((isTypeLegal(MVT::i32) && !LegalOperations && !ST->isVolatile()) ||
12050         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
12051       ;
12052       Tmp = DAG.getConstant((uint32_t)CFP->getValueAPF().
12053                             bitcastToAPInt().getZExtValue(), SDLoc(CFP),
12054                             MVT::i32);
12055       return DAG.getStore(Chain, DL, Tmp, Ptr, ST->getMemOperand());
12056     }
12057 
12058     return SDValue();
12059   case MVT::f64:
12060     if ((TLI.isTypeLegal(MVT::i64) && !LegalOperations &&
12061          !ST->isVolatile()) ||
12062         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i64)) {
12063       ;
12064       Tmp = DAG.getConstant(CFP->getValueAPF().bitcastToAPInt().
12065                             getZExtValue(), SDLoc(CFP), MVT::i64);
12066       return DAG.getStore(Chain, DL, Tmp,
12067                           Ptr, ST->getMemOperand());
12068     }
12069 
12070     if (!ST->isVolatile() &&
12071         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
12072       // Many FP stores are not made apparent until after legalize, e.g. for
12073       // argument passing.  Since this is so common, custom legalize the
12074       // 64-bit integer store into two 32-bit stores.
12075       uint64_t Val = CFP->getValueAPF().bitcastToAPInt().getZExtValue();
12076       SDValue Lo = DAG.getConstant(Val & 0xFFFFFFFF, SDLoc(CFP), MVT::i32);
12077       SDValue Hi = DAG.getConstant(Val >> 32, SDLoc(CFP), MVT::i32);
12078       if (DAG.getDataLayout().isBigEndian())
12079         std::swap(Lo, Hi);
12080 
12081       unsigned Alignment = ST->getAlignment();
12082       MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
12083       AAMDNodes AAInfo = ST->getAAInfo();
12084 
12085       SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
12086                                  ST->getAlignment(), MMOFlags, AAInfo);
12087       Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
12088                         DAG.getConstant(4, DL, Ptr.getValueType()));
12089       Alignment = MinAlign(Alignment, 4U);
12090       SDValue St1 = DAG.getStore(Chain, DL, Hi, Ptr,
12091                                  ST->getPointerInfo().getWithOffset(4),
12092                                  Alignment, MMOFlags, AAInfo);
12093       return DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
12094                          St0, St1);
12095     }
12096 
12097     return SDValue();
12098   }
12099 }
12100 
12101 SDValue DAGCombiner::visitSTORE(SDNode *N) {
12102   StoreSDNode *ST  = cast<StoreSDNode>(N);
12103   SDValue Chain = ST->getChain();
12104   SDValue Value = ST->getValue();
12105   SDValue Ptr   = ST->getBasePtr();
12106 
12107   // If this is a store of a bit convert, store the input value if the
12108   // resultant store does not need a higher alignment than the original.
12109   if (Value.getOpcode() == ISD::BITCAST && !ST->isTruncatingStore() &&
12110       ST->isUnindexed()) {
12111     EVT SVT = Value.getOperand(0).getValueType();
12112     if (((!LegalOperations && !ST->isVolatile()) ||
12113          TLI.isOperationLegalOrCustom(ISD::STORE, SVT)) &&
12114         TLI.isStoreBitCastBeneficial(Value.getValueType(), SVT)) {
12115       unsigned OrigAlign = ST->getAlignment();
12116       bool Fast = false;
12117       if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), SVT,
12118                                  ST->getAddressSpace(), OrigAlign, &Fast) &&
12119           Fast) {
12120         return DAG.getStore(Chain, SDLoc(N), Value.getOperand(0), Ptr,
12121                             ST->getPointerInfo(), OrigAlign,
12122                             ST->getMemOperand()->getFlags(), ST->getAAInfo());
12123       }
12124     }
12125   }
12126 
12127   // Turn 'store undef, Ptr' -> nothing.
12128   if (Value.isUndef() && ST->isUnindexed())
12129     return Chain;
12130 
12131   // Try to infer better alignment information than the store already has.
12132   if (OptLevel != CodeGenOpt::None && ST->isUnindexed()) {
12133     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
12134       if (Align > ST->getAlignment()) {
12135         SDValue NewStore =
12136             DAG.getTruncStore(Chain, SDLoc(N), Value, Ptr, ST->getPointerInfo(),
12137                               ST->getMemoryVT(), Align,
12138                               ST->getMemOperand()->getFlags(), ST->getAAInfo());
12139         if (NewStore.getNode() != N)
12140           return CombineTo(ST, NewStore, true);
12141       }
12142     }
12143   }
12144 
12145   // Try transforming a pair floating point load / store ops to integer
12146   // load / store ops.
12147   if (SDValue NewST = TransformFPLoadStorePair(N))
12148     return NewST;
12149 
12150   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
12151                                                   : DAG.getSubtarget().useAA();
12152 #ifndef NDEBUG
12153   if (CombinerAAOnlyFunc.getNumOccurrences() &&
12154       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
12155     UseAA = false;
12156 #endif
12157   if (UseAA && ST->isUnindexed()) {
12158     // FIXME: We should do this even without AA enabled. AA will just allow
12159     // FindBetterChain to work in more situations. The problem with this is that
12160     // any combine that expects memory operations to be on consecutive chains
12161     // first needs to be updated to look for users of the same chain.
12162 
12163     // Walk up chain skipping non-aliasing memory nodes, on this store and any
12164     // adjacent stores.
12165     if (findBetterNeighborChains(ST)) {
12166       // replaceStoreChain uses CombineTo, which handled all of the worklist
12167       // manipulation. Return the original node to not do anything else.
12168       return SDValue(ST, 0);
12169     }
12170     Chain = ST->getChain();
12171   }
12172 
12173   // Try transforming N to an indexed store.
12174   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
12175     return SDValue(N, 0);
12176 
12177   // FIXME: is there such a thing as a truncating indexed store?
12178   if (ST->isTruncatingStore() && ST->isUnindexed() &&
12179       Value.getValueType().isInteger()) {
12180     // See if we can simplify the input to this truncstore with knowledge that
12181     // only the low bits are being used.  For example:
12182     // "truncstore (or (shl x, 8), y), i8"  -> "truncstore y, i8"
12183     SDValue Shorter = GetDemandedBits(
12184         Value, APInt::getLowBitsSet(Value.getScalarValueSizeInBits(),
12185                                     ST->getMemoryVT().getScalarSizeInBits()));
12186     AddToWorklist(Value.getNode());
12187     if (Shorter.getNode())
12188       return DAG.getTruncStore(Chain, SDLoc(N), Shorter,
12189                                Ptr, ST->getMemoryVT(), ST->getMemOperand());
12190 
12191     // Otherwise, see if we can simplify the operation with
12192     // SimplifyDemandedBits, which only works if the value has a single use.
12193     if (SimplifyDemandedBits(
12194             Value,
12195             APInt::getLowBitsSet(Value.getScalarValueSizeInBits(),
12196                                  ST->getMemoryVT().getScalarSizeInBits())))
12197       return SDValue(N, 0);
12198   }
12199 
12200   // If this is a load followed by a store to the same location, then the store
12201   // is dead/noop.
12202   if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Value)) {
12203     if (Ld->getBasePtr() == Ptr && ST->getMemoryVT() == Ld->getMemoryVT() &&
12204         ST->isUnindexed() && !ST->isVolatile() &&
12205         // There can't be any side effects between the load and store, such as
12206         // a call or store.
12207         Chain.reachesChainWithoutSideEffects(SDValue(Ld, 1))) {
12208       // The store is dead, remove it.
12209       return Chain;
12210     }
12211   }
12212 
12213   // If this is a store followed by a store with the same value to the same
12214   // location, then the store is dead/noop.
12215   if (StoreSDNode *ST1 = dyn_cast<StoreSDNode>(Chain)) {
12216     if (ST1->getBasePtr() == Ptr && ST->getMemoryVT() == ST1->getMemoryVT() &&
12217         ST1->getValue() == Value && ST->isUnindexed() && !ST->isVolatile() &&
12218         ST1->isUnindexed() && !ST1->isVolatile()) {
12219       // The store is dead, remove it.
12220       return Chain;
12221     }
12222   }
12223 
12224   // If this is an FP_ROUND or TRUNC followed by a store, fold this into a
12225   // truncating store.  We can do this even if this is already a truncstore.
12226   if ((Value.getOpcode() == ISD::FP_ROUND || Value.getOpcode() == ISD::TRUNCATE)
12227       && Value.getNode()->hasOneUse() && ST->isUnindexed() &&
12228       TLI.isTruncStoreLegal(Value.getOperand(0).getValueType(),
12229                             ST->getMemoryVT())) {
12230     return DAG.getTruncStore(Chain, SDLoc(N), Value.getOperand(0),
12231                              Ptr, ST->getMemoryVT(), ST->getMemOperand());
12232   }
12233 
12234   // Only perform this optimization before the types are legal, because we
12235   // don't want to perform this optimization on every DAGCombine invocation.
12236   if (!LegalTypes) {
12237     bool EverChanged = false;
12238 
12239     do {
12240       // There can be multiple store sequences on the same chain.
12241       // Keep trying to merge store sequences until we are unable to do so
12242       // or until we merge the last store on the chain.
12243       bool Changed = MergeConsecutiveStores(ST);
12244       EverChanged |= Changed;
12245       if (!Changed) break;
12246     } while (ST->getOpcode() != ISD::DELETED_NODE);
12247 
12248     if (EverChanged)
12249       return SDValue(N, 0);
12250   }
12251 
12252   // Turn 'store float 1.0, Ptr' -> 'store int 0x12345678, Ptr'
12253   //
12254   // Make sure to do this only after attempting to merge stores in order to
12255   //  avoid changing the types of some subset of stores due to visit order,
12256   //  preventing their merging.
12257   if (isa<ConstantFPSDNode>(Value)) {
12258     if (SDValue NewSt = replaceStoreOfFPConstant(ST))
12259       return NewSt;
12260   }
12261 
12262   if (SDValue NewSt = splitMergedValStore(ST))
12263     return NewSt;
12264 
12265   return ReduceLoadOpStoreWidth(N);
12266 }
12267 
12268 /// For the instruction sequence of store below, F and I values
12269 /// are bundled together as an i64 value before being stored into memory.
12270 /// Sometimes it is more efficent to generate separate stores for F and I,
12271 /// which can remove the bitwise instructions or sink them to colder places.
12272 ///
12273 ///   (store (or (zext (bitcast F to i32) to i64),
12274 ///              (shl (zext I to i64), 32)), addr)  -->
12275 ///   (store F, addr) and (store I, addr+4)
12276 ///
12277 /// Similarly, splitting for other merged store can also be beneficial, like:
12278 /// For pair of {i32, i32}, i64 store --> two i32 stores.
12279 /// For pair of {i32, i16}, i64 store --> two i32 stores.
12280 /// For pair of {i16, i16}, i32 store --> two i16 stores.
12281 /// For pair of {i16, i8},  i32 store --> two i16 stores.
12282 /// For pair of {i8, i8},   i16 store --> two i8 stores.
12283 ///
12284 /// We allow each target to determine specifically which kind of splitting is
12285 /// supported.
12286 ///
12287 /// The store patterns are commonly seen from the simple code snippet below
12288 /// if only std::make_pair(...) is sroa transformed before inlined into hoo.
12289 ///   void goo(const std::pair<int, float> &);
12290 ///   hoo() {
12291 ///     ...
12292 ///     goo(std::make_pair(tmp, ftmp));
12293 ///     ...
12294 ///   }
12295 ///
12296 SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
12297   if (OptLevel == CodeGenOpt::None)
12298     return SDValue();
12299 
12300   SDValue Val = ST->getValue();
12301   SDLoc DL(ST);
12302 
12303   // Match OR operand.
12304   if (!Val.getValueType().isScalarInteger() || Val.getOpcode() != ISD::OR)
12305     return SDValue();
12306 
12307   // Match SHL operand and get Lower and Higher parts of Val.
12308   SDValue Op1 = Val.getOperand(0);
12309   SDValue Op2 = Val.getOperand(1);
12310   SDValue Lo, Hi;
12311   if (Op1.getOpcode() != ISD::SHL) {
12312     std::swap(Op1, Op2);
12313     if (Op1.getOpcode() != ISD::SHL)
12314       return SDValue();
12315   }
12316   Lo = Op2;
12317   Hi = Op1.getOperand(0);
12318   if (!Op1.hasOneUse())
12319     return SDValue();
12320 
12321   // Match shift amount to HalfValBitSize.
12322   unsigned HalfValBitSize = Val.getValueSizeInBits() / 2;
12323   ConstantSDNode *ShAmt = dyn_cast<ConstantSDNode>(Op1.getOperand(1));
12324   if (!ShAmt || ShAmt->getAPIntValue() != HalfValBitSize)
12325     return SDValue();
12326 
12327   // Lo and Hi are zero-extended from int with size less equal than 32
12328   // to i64.
12329   if (Lo.getOpcode() != ISD::ZERO_EXTEND || !Lo.hasOneUse() ||
12330       !Lo.getOperand(0).getValueType().isScalarInteger() ||
12331       Lo.getOperand(0).getValueSizeInBits() > HalfValBitSize ||
12332       Hi.getOpcode() != ISD::ZERO_EXTEND || !Hi.hasOneUse() ||
12333       !Hi.getOperand(0).getValueType().isScalarInteger() ||
12334       Hi.getOperand(0).getValueSizeInBits() > HalfValBitSize)
12335     return SDValue();
12336 
12337   if (!TLI.isMultiStoresCheaperThanBitsMerge(Lo.getOperand(0),
12338                                              Hi.getOperand(0)))
12339     return SDValue();
12340 
12341   // Start to split store.
12342   unsigned Alignment = ST->getAlignment();
12343   MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
12344   AAMDNodes AAInfo = ST->getAAInfo();
12345 
12346   // Change the sizes of Lo and Hi's value types to HalfValBitSize.
12347   EVT VT = EVT::getIntegerVT(*DAG.getContext(), HalfValBitSize);
12348   Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Lo.getOperand(0));
12349   Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Hi.getOperand(0));
12350 
12351   SDValue Chain = ST->getChain();
12352   SDValue Ptr = ST->getBasePtr();
12353   // Lower value store.
12354   SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
12355                              ST->getAlignment(), MMOFlags, AAInfo);
12356   Ptr =
12357       DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
12358                   DAG.getConstant(HalfValBitSize / 8, DL, Ptr.getValueType()));
12359   // Higher value store.
12360   SDValue St1 =
12361       DAG.getStore(St0, DL, Hi, Ptr,
12362                    ST->getPointerInfo().getWithOffset(HalfValBitSize / 8),
12363                    Alignment / 2, MMOFlags, AAInfo);
12364   return St1;
12365 }
12366 
12367 SDValue DAGCombiner::visitINSERT_VECTOR_ELT(SDNode *N) {
12368   SDValue InVec = N->getOperand(0);
12369   SDValue InVal = N->getOperand(1);
12370   SDValue EltNo = N->getOperand(2);
12371   SDLoc DL(N);
12372 
12373   // If the inserted element is an UNDEF, just use the input vector.
12374   if (InVal.isUndef())
12375     return InVec;
12376 
12377   EVT VT = InVec.getValueType();
12378 
12379   // If we can't generate a legal BUILD_VECTOR, exit
12380   if (LegalOperations && !TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
12381     return SDValue();
12382 
12383   // Check that we know which element is being inserted
12384   if (!isa<ConstantSDNode>(EltNo))
12385     return SDValue();
12386   unsigned Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
12387 
12388   // Canonicalize insert_vector_elt dag nodes.
12389   // Example:
12390   // (insert_vector_elt (insert_vector_elt A, Idx0), Idx1)
12391   // -> (insert_vector_elt (insert_vector_elt A, Idx1), Idx0)
12392   //
12393   // Do this only if the child insert_vector node has one use; also
12394   // do this only if indices are both constants and Idx1 < Idx0.
12395   if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT && InVec.hasOneUse()
12396       && isa<ConstantSDNode>(InVec.getOperand(2))) {
12397     unsigned OtherElt =
12398       cast<ConstantSDNode>(InVec.getOperand(2))->getZExtValue();
12399     if (Elt < OtherElt) {
12400       // Swap nodes.
12401       SDValue NewOp = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT,
12402                                   InVec.getOperand(0), InVal, EltNo);
12403       AddToWorklist(NewOp.getNode());
12404       return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(InVec.getNode()),
12405                          VT, NewOp, InVec.getOperand(1), InVec.getOperand(2));
12406     }
12407   }
12408 
12409   // Check that the operand is a BUILD_VECTOR (or UNDEF, which can essentially
12410   // be converted to a BUILD_VECTOR).  Fill in the Ops vector with the
12411   // vector elements.
12412   SmallVector<SDValue, 8> Ops;
12413   // Do not combine these two vectors if the output vector will not replace
12414   // the input vector.
12415   if (InVec.getOpcode() == ISD::BUILD_VECTOR && InVec.hasOneUse()) {
12416     Ops.append(InVec.getNode()->op_begin(),
12417                InVec.getNode()->op_end());
12418   } else if (InVec.isUndef()) {
12419     unsigned NElts = VT.getVectorNumElements();
12420     Ops.append(NElts, DAG.getUNDEF(InVal.getValueType()));
12421   } else {
12422     return SDValue();
12423   }
12424 
12425   // Insert the element
12426   if (Elt < Ops.size()) {
12427     // All the operands of BUILD_VECTOR must have the same type;
12428     // we enforce that here.
12429     EVT OpVT = Ops[0].getValueType();
12430     if (InVal.getValueType() != OpVT)
12431       InVal = OpVT.bitsGT(InVal.getValueType()) ?
12432                 DAG.getNode(ISD::ANY_EXTEND, DL, OpVT, InVal) :
12433                 DAG.getNode(ISD::TRUNCATE, DL, OpVT, InVal);
12434     Ops[Elt] = InVal;
12435   }
12436 
12437   // Return the new vector
12438   return DAG.getBuildVector(VT, DL, Ops);
12439 }
12440 
12441 SDValue DAGCombiner::ReplaceExtractVectorEltOfLoadWithNarrowedLoad(
12442     SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad) {
12443   assert(!OriginalLoad->isVolatile());
12444 
12445   EVT ResultVT = EVE->getValueType(0);
12446   EVT VecEltVT = InVecVT.getVectorElementType();
12447   unsigned Align = OriginalLoad->getAlignment();
12448   unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
12449       VecEltVT.getTypeForEVT(*DAG.getContext()));
12450 
12451   if (NewAlign > Align || !TLI.isOperationLegalOrCustom(ISD::LOAD, VecEltVT))
12452     return SDValue();
12453 
12454   Align = NewAlign;
12455 
12456   SDValue NewPtr = OriginalLoad->getBasePtr();
12457   SDValue Offset;
12458   EVT PtrType = NewPtr.getValueType();
12459   MachinePointerInfo MPI;
12460   SDLoc DL(EVE);
12461   if (auto *ConstEltNo = dyn_cast<ConstantSDNode>(EltNo)) {
12462     int Elt = ConstEltNo->getZExtValue();
12463     unsigned PtrOff = VecEltVT.getSizeInBits() * Elt / 8;
12464     Offset = DAG.getConstant(PtrOff, DL, PtrType);
12465     MPI = OriginalLoad->getPointerInfo().getWithOffset(PtrOff);
12466   } else {
12467     Offset = DAG.getZExtOrTrunc(EltNo, DL, PtrType);
12468     Offset = DAG.getNode(
12469         ISD::MUL, DL, PtrType, Offset,
12470         DAG.getConstant(VecEltVT.getStoreSize(), DL, PtrType));
12471     MPI = OriginalLoad->getPointerInfo();
12472   }
12473   NewPtr = DAG.getNode(ISD::ADD, DL, PtrType, NewPtr, Offset);
12474 
12475   // The replacement we need to do here is a little tricky: we need to
12476   // replace an extractelement of a load with a load.
12477   // Use ReplaceAllUsesOfValuesWith to do the replacement.
12478   // Note that this replacement assumes that the extractvalue is the only
12479   // use of the load; that's okay because we don't want to perform this
12480   // transformation in other cases anyway.
12481   SDValue Load;
12482   SDValue Chain;
12483   if (ResultVT.bitsGT(VecEltVT)) {
12484     // If the result type of vextract is wider than the load, then issue an
12485     // extending load instead.
12486     ISD::LoadExtType ExtType = TLI.isLoadExtLegal(ISD::ZEXTLOAD, ResultVT,
12487                                                   VecEltVT)
12488                                    ? ISD::ZEXTLOAD
12489                                    : ISD::EXTLOAD;
12490     Load = DAG.getExtLoad(ExtType, SDLoc(EVE), ResultVT,
12491                           OriginalLoad->getChain(), NewPtr, MPI, VecEltVT,
12492                           Align, OriginalLoad->getMemOperand()->getFlags(),
12493                           OriginalLoad->getAAInfo());
12494     Chain = Load.getValue(1);
12495   } else {
12496     Load = DAG.getLoad(VecEltVT, SDLoc(EVE), OriginalLoad->getChain(), NewPtr,
12497                        MPI, Align, OriginalLoad->getMemOperand()->getFlags(),
12498                        OriginalLoad->getAAInfo());
12499     Chain = Load.getValue(1);
12500     if (ResultVT.bitsLT(VecEltVT))
12501       Load = DAG.getNode(ISD::TRUNCATE, SDLoc(EVE), ResultVT, Load);
12502     else
12503       Load = DAG.getBitcast(ResultVT, Load);
12504   }
12505   WorklistRemover DeadNodes(*this);
12506   SDValue From[] = { SDValue(EVE, 0), SDValue(OriginalLoad, 1) };
12507   SDValue To[] = { Load, Chain };
12508   DAG.ReplaceAllUsesOfValuesWith(From, To, 2);
12509   // Since we're explicitly calling ReplaceAllUses, add the new node to the
12510   // worklist explicitly as well.
12511   AddToWorklist(Load.getNode());
12512   AddUsersToWorklist(Load.getNode()); // Add users too
12513   // Make sure to revisit this node to clean it up; it will usually be dead.
12514   AddToWorklist(EVE);
12515   ++OpsNarrowed;
12516   return SDValue(EVE, 0);
12517 }
12518 
12519 SDValue DAGCombiner::visitEXTRACT_VECTOR_ELT(SDNode *N) {
12520   // (vextract (scalar_to_vector val, 0) -> val
12521   SDValue InVec = N->getOperand(0);
12522   EVT VT = InVec.getValueType();
12523   EVT NVT = N->getValueType(0);
12524 
12525   if (InVec.getOpcode() == ISD::SCALAR_TO_VECTOR) {
12526     // Check if the result type doesn't match the inserted element type. A
12527     // SCALAR_TO_VECTOR may truncate the inserted element and the
12528     // EXTRACT_VECTOR_ELT may widen the extracted vector.
12529     SDValue InOp = InVec.getOperand(0);
12530     if (InOp.getValueType() != NVT) {
12531       assert(InOp.getValueType().isInteger() && NVT.isInteger());
12532       return DAG.getSExtOrTrunc(InOp, SDLoc(InVec), NVT);
12533     }
12534     return InOp;
12535   }
12536 
12537   SDValue EltNo = N->getOperand(1);
12538   ConstantSDNode *ConstEltNo = dyn_cast<ConstantSDNode>(EltNo);
12539 
12540   // extract_vector_elt (build_vector x, y), 1 -> y
12541   if (ConstEltNo &&
12542       InVec.getOpcode() == ISD::BUILD_VECTOR &&
12543       TLI.isTypeLegal(VT) &&
12544       (InVec.hasOneUse() ||
12545        TLI.aggressivelyPreferBuildVectorSources(VT))) {
12546     SDValue Elt = InVec.getOperand(ConstEltNo->getZExtValue());
12547     EVT InEltVT = Elt.getValueType();
12548 
12549     // Sometimes build_vector's scalar input types do not match result type.
12550     if (NVT == InEltVT)
12551       return Elt;
12552 
12553     // TODO: It may be useful to truncate if free if the build_vector implicitly
12554     // converts.
12555   }
12556 
12557   // extract_vector_elt (v2i32 (bitcast i64:x)), 0 -> i32 (trunc i64:x)
12558   if (ConstEltNo && InVec.getOpcode() == ISD::BITCAST && InVec.hasOneUse() &&
12559       ConstEltNo->isNullValue() && VT.isInteger()) {
12560     SDValue BCSrc = InVec.getOperand(0);
12561     if (BCSrc.getValueType().isScalarInteger())
12562       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), NVT, BCSrc);
12563   }
12564 
12565   // extract_vector_elt (insert_vector_elt vec, val, idx), idx) -> val
12566   //
12567   // This only really matters if the index is non-constant since other combines
12568   // on the constant elements already work.
12569   if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT &&
12570       EltNo == InVec.getOperand(2)) {
12571     SDValue Elt = InVec.getOperand(1);
12572     return VT.isInteger() ? DAG.getAnyExtOrTrunc(Elt, SDLoc(N), NVT) : Elt;
12573   }
12574 
12575   // Transform: (EXTRACT_VECTOR_ELT( VECTOR_SHUFFLE )) -> EXTRACT_VECTOR_ELT.
12576   // We only perform this optimization before the op legalization phase because
12577   // we may introduce new vector instructions which are not backed by TD
12578   // patterns. For example on AVX, extracting elements from a wide vector
12579   // without using extract_subvector. However, if we can find an underlying
12580   // scalar value, then we can always use that.
12581   if (ConstEltNo && InVec.getOpcode() == ISD::VECTOR_SHUFFLE) {
12582     int NumElem = VT.getVectorNumElements();
12583     ShuffleVectorSDNode *SVOp = cast<ShuffleVectorSDNode>(InVec);
12584     // Find the new index to extract from.
12585     int OrigElt = SVOp->getMaskElt(ConstEltNo->getZExtValue());
12586 
12587     // Extracting an undef index is undef.
12588     if (OrigElt == -1)
12589       return DAG.getUNDEF(NVT);
12590 
12591     // Select the right vector half to extract from.
12592     SDValue SVInVec;
12593     if (OrigElt < NumElem) {
12594       SVInVec = InVec->getOperand(0);
12595     } else {
12596       SVInVec = InVec->getOperand(1);
12597       OrigElt -= NumElem;
12598     }
12599 
12600     if (SVInVec.getOpcode() == ISD::BUILD_VECTOR) {
12601       SDValue InOp = SVInVec.getOperand(OrigElt);
12602       if (InOp.getValueType() != NVT) {
12603         assert(InOp.getValueType().isInteger() && NVT.isInteger());
12604         InOp = DAG.getSExtOrTrunc(InOp, SDLoc(SVInVec), NVT);
12605       }
12606 
12607       return InOp;
12608     }
12609 
12610     // FIXME: We should handle recursing on other vector shuffles and
12611     // scalar_to_vector here as well.
12612 
12613     if (!LegalOperations) {
12614       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
12615       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(N), NVT, SVInVec,
12616                          DAG.getConstant(OrigElt, SDLoc(SVOp), IndexTy));
12617     }
12618   }
12619 
12620   bool BCNumEltsChanged = false;
12621   EVT ExtVT = VT.getVectorElementType();
12622   EVT LVT = ExtVT;
12623 
12624   // If the result of load has to be truncated, then it's not necessarily
12625   // profitable.
12626   if (NVT.bitsLT(LVT) && !TLI.isTruncateFree(LVT, NVT))
12627     return SDValue();
12628 
12629   if (InVec.getOpcode() == ISD::BITCAST) {
12630     // Don't duplicate a load with other uses.
12631     if (!InVec.hasOneUse())
12632       return SDValue();
12633 
12634     EVT BCVT = InVec.getOperand(0).getValueType();
12635     if (!BCVT.isVector() || ExtVT.bitsGT(BCVT.getVectorElementType()))
12636       return SDValue();
12637     if (VT.getVectorNumElements() != BCVT.getVectorNumElements())
12638       BCNumEltsChanged = true;
12639     InVec = InVec.getOperand(0);
12640     ExtVT = BCVT.getVectorElementType();
12641   }
12642 
12643   // (vextract (vN[if]M load $addr), i) -> ([if]M load $addr + i * size)
12644   if (!LegalOperations && !ConstEltNo && InVec.hasOneUse() &&
12645       ISD::isNormalLoad(InVec.getNode()) &&
12646       !N->getOperand(1)->hasPredecessor(InVec.getNode())) {
12647     SDValue Index = N->getOperand(1);
12648     if (LoadSDNode *OrigLoad = dyn_cast<LoadSDNode>(InVec)) {
12649       if (!OrigLoad->isVolatile()) {
12650         return ReplaceExtractVectorEltOfLoadWithNarrowedLoad(N, VT, Index,
12651                                                              OrigLoad);
12652       }
12653     }
12654   }
12655 
12656   // Perform only after legalization to ensure build_vector / vector_shuffle
12657   // optimizations have already been done.
12658   if (!LegalOperations) return SDValue();
12659 
12660   // (vextract (v4f32 load $addr), c) -> (f32 load $addr+c*size)
12661   // (vextract (v4f32 s2v (f32 load $addr)), c) -> (f32 load $addr+c*size)
12662   // (vextract (v4f32 shuffle (load $addr), <1,u,u,u>), 0) -> (f32 load $addr)
12663 
12664   if (ConstEltNo) {
12665     int Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
12666 
12667     LoadSDNode *LN0 = nullptr;
12668     const ShuffleVectorSDNode *SVN = nullptr;
12669     if (ISD::isNormalLoad(InVec.getNode())) {
12670       LN0 = cast<LoadSDNode>(InVec);
12671     } else if (InVec.getOpcode() == ISD::SCALAR_TO_VECTOR &&
12672                InVec.getOperand(0).getValueType() == ExtVT &&
12673                ISD::isNormalLoad(InVec.getOperand(0).getNode())) {
12674       // Don't duplicate a load with other uses.
12675       if (!InVec.hasOneUse())
12676         return SDValue();
12677 
12678       LN0 = cast<LoadSDNode>(InVec.getOperand(0));
12679     } else if ((SVN = dyn_cast<ShuffleVectorSDNode>(InVec))) {
12680       // (vextract (vector_shuffle (load $addr), v2, <1, u, u, u>), 1)
12681       // =>
12682       // (load $addr+1*size)
12683 
12684       // Don't duplicate a load with other uses.
12685       if (!InVec.hasOneUse())
12686         return SDValue();
12687 
12688       // If the bit convert changed the number of elements, it is unsafe
12689       // to examine the mask.
12690       if (BCNumEltsChanged)
12691         return SDValue();
12692 
12693       // Select the input vector, guarding against out of range extract vector.
12694       unsigned NumElems = VT.getVectorNumElements();
12695       int Idx = (Elt > (int)NumElems) ? -1 : SVN->getMaskElt(Elt);
12696       InVec = (Idx < (int)NumElems) ? InVec.getOperand(0) : InVec.getOperand(1);
12697 
12698       if (InVec.getOpcode() == ISD::BITCAST) {
12699         // Don't duplicate a load with other uses.
12700         if (!InVec.hasOneUse())
12701           return SDValue();
12702 
12703         InVec = InVec.getOperand(0);
12704       }
12705       if (ISD::isNormalLoad(InVec.getNode())) {
12706         LN0 = cast<LoadSDNode>(InVec);
12707         Elt = (Idx < (int)NumElems) ? Idx : Idx - (int)NumElems;
12708         EltNo = DAG.getConstant(Elt, SDLoc(EltNo), EltNo.getValueType());
12709       }
12710     }
12711 
12712     // Make sure we found a non-volatile load and the extractelement is
12713     // the only use.
12714     if (!LN0 || !LN0->hasNUsesOfValue(1,0) || LN0->isVolatile())
12715       return SDValue();
12716 
12717     // If Idx was -1 above, Elt is going to be -1, so just return undef.
12718     if (Elt == -1)
12719       return DAG.getUNDEF(LVT);
12720 
12721     return ReplaceExtractVectorEltOfLoadWithNarrowedLoad(N, VT, EltNo, LN0);
12722   }
12723 
12724   return SDValue();
12725 }
12726 
12727 // Simplify (build_vec (ext )) to (bitcast (build_vec ))
12728 SDValue DAGCombiner::reduceBuildVecExtToExtBuildVec(SDNode *N) {
12729   // We perform this optimization post type-legalization because
12730   // the type-legalizer often scalarizes integer-promoted vectors.
12731   // Performing this optimization before may create bit-casts which
12732   // will be type-legalized to complex code sequences.
12733   // We perform this optimization only before the operation legalizer because we
12734   // may introduce illegal operations.
12735   if (Level != AfterLegalizeVectorOps && Level != AfterLegalizeTypes)
12736     return SDValue();
12737 
12738   unsigned NumInScalars = N->getNumOperands();
12739   SDLoc DL(N);
12740   EVT VT = N->getValueType(0);
12741 
12742   // Check to see if this is a BUILD_VECTOR of a bunch of values
12743   // which come from any_extend or zero_extend nodes. If so, we can create
12744   // a new BUILD_VECTOR using bit-casts which may enable other BUILD_VECTOR
12745   // optimizations. We do not handle sign-extend because we can't fill the sign
12746   // using shuffles.
12747   EVT SourceType = MVT::Other;
12748   bool AllAnyExt = true;
12749 
12750   for (unsigned i = 0; i != NumInScalars; ++i) {
12751     SDValue In = N->getOperand(i);
12752     // Ignore undef inputs.
12753     if (In.isUndef()) continue;
12754 
12755     bool AnyExt  = In.getOpcode() == ISD::ANY_EXTEND;
12756     bool ZeroExt = In.getOpcode() == ISD::ZERO_EXTEND;
12757 
12758     // Abort if the element is not an extension.
12759     if (!ZeroExt && !AnyExt) {
12760       SourceType = MVT::Other;
12761       break;
12762     }
12763 
12764     // The input is a ZeroExt or AnyExt. Check the original type.
12765     EVT InTy = In.getOperand(0).getValueType();
12766 
12767     // Check that all of the widened source types are the same.
12768     if (SourceType == MVT::Other)
12769       // First time.
12770       SourceType = InTy;
12771     else if (InTy != SourceType) {
12772       // Multiple income types. Abort.
12773       SourceType = MVT::Other;
12774       break;
12775     }
12776 
12777     // Check if all of the extends are ANY_EXTENDs.
12778     AllAnyExt &= AnyExt;
12779   }
12780 
12781   // In order to have valid types, all of the inputs must be extended from the
12782   // same source type and all of the inputs must be any or zero extend.
12783   // Scalar sizes must be a power of two.
12784   EVT OutScalarTy = VT.getScalarType();
12785   bool ValidTypes = SourceType != MVT::Other &&
12786                  isPowerOf2_32(OutScalarTy.getSizeInBits()) &&
12787                  isPowerOf2_32(SourceType.getSizeInBits());
12788 
12789   // Create a new simpler BUILD_VECTOR sequence which other optimizations can
12790   // turn into a single shuffle instruction.
12791   if (!ValidTypes)
12792     return SDValue();
12793 
12794   bool isLE = DAG.getDataLayout().isLittleEndian();
12795   unsigned ElemRatio = OutScalarTy.getSizeInBits()/SourceType.getSizeInBits();
12796   assert(ElemRatio > 1 && "Invalid element size ratio");
12797   SDValue Filler = AllAnyExt ? DAG.getUNDEF(SourceType):
12798                                DAG.getConstant(0, DL, SourceType);
12799 
12800   unsigned NewBVElems = ElemRatio * VT.getVectorNumElements();
12801   SmallVector<SDValue, 8> Ops(NewBVElems, Filler);
12802 
12803   // Populate the new build_vector
12804   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
12805     SDValue Cast = N->getOperand(i);
12806     assert((Cast.getOpcode() == ISD::ANY_EXTEND ||
12807             Cast.getOpcode() == ISD::ZERO_EXTEND ||
12808             Cast.isUndef()) && "Invalid cast opcode");
12809     SDValue In;
12810     if (Cast.isUndef())
12811       In = DAG.getUNDEF(SourceType);
12812     else
12813       In = Cast->getOperand(0);
12814     unsigned Index = isLE ? (i * ElemRatio) :
12815                             (i * ElemRatio + (ElemRatio - 1));
12816 
12817     assert(Index < Ops.size() && "Invalid index");
12818     Ops[Index] = In;
12819   }
12820 
12821   // The type of the new BUILD_VECTOR node.
12822   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SourceType, NewBVElems);
12823   assert(VecVT.getSizeInBits() == VT.getSizeInBits() &&
12824          "Invalid vector size");
12825   // Check if the new vector type is legal.
12826   if (!isTypeLegal(VecVT)) return SDValue();
12827 
12828   // Make the new BUILD_VECTOR.
12829   SDValue BV = DAG.getBuildVector(VecVT, DL, Ops);
12830 
12831   // The new BUILD_VECTOR node has the potential to be further optimized.
12832   AddToWorklist(BV.getNode());
12833   // Bitcast to the desired type.
12834   return DAG.getBitcast(VT, BV);
12835 }
12836 
12837 SDValue DAGCombiner::reduceBuildVecConvertToConvertBuildVec(SDNode *N) {
12838   EVT VT = N->getValueType(0);
12839 
12840   unsigned NumInScalars = N->getNumOperands();
12841   SDLoc DL(N);
12842 
12843   EVT SrcVT = MVT::Other;
12844   unsigned Opcode = ISD::DELETED_NODE;
12845   unsigned NumDefs = 0;
12846 
12847   for (unsigned i = 0; i != NumInScalars; ++i) {
12848     SDValue In = N->getOperand(i);
12849     unsigned Opc = In.getOpcode();
12850 
12851     if (Opc == ISD::UNDEF)
12852       continue;
12853 
12854     // If all scalar values are floats and converted from integers.
12855     if (Opcode == ISD::DELETED_NODE &&
12856         (Opc == ISD::UINT_TO_FP || Opc == ISD::SINT_TO_FP)) {
12857       Opcode = Opc;
12858     }
12859 
12860     if (Opc != Opcode)
12861       return SDValue();
12862 
12863     EVT InVT = In.getOperand(0).getValueType();
12864 
12865     // If all scalar values are typed differently, bail out. It's chosen to
12866     // simplify BUILD_VECTOR of integer types.
12867     if (SrcVT == MVT::Other)
12868       SrcVT = InVT;
12869     if (SrcVT != InVT)
12870       return SDValue();
12871     NumDefs++;
12872   }
12873 
12874   // If the vector has just one element defined, it's not worth to fold it into
12875   // a vectorized one.
12876   if (NumDefs < 2)
12877     return SDValue();
12878 
12879   assert((Opcode == ISD::UINT_TO_FP || Opcode == ISD::SINT_TO_FP)
12880          && "Should only handle conversion from integer to float.");
12881   assert(SrcVT != MVT::Other && "Cannot determine source type!");
12882 
12883   EVT NVT = EVT::getVectorVT(*DAG.getContext(), SrcVT, NumInScalars);
12884 
12885   if (!TLI.isOperationLegalOrCustom(Opcode, NVT))
12886     return SDValue();
12887 
12888   // Just because the floating-point vector type is legal does not necessarily
12889   // mean that the corresponding integer vector type is.
12890   if (!isTypeLegal(NVT))
12891     return SDValue();
12892 
12893   SmallVector<SDValue, 8> Opnds;
12894   for (unsigned i = 0; i != NumInScalars; ++i) {
12895     SDValue In = N->getOperand(i);
12896 
12897     if (In.isUndef())
12898       Opnds.push_back(DAG.getUNDEF(SrcVT));
12899     else
12900       Opnds.push_back(In.getOperand(0));
12901   }
12902   SDValue BV = DAG.getBuildVector(NVT, DL, Opnds);
12903   AddToWorklist(BV.getNode());
12904 
12905   return DAG.getNode(Opcode, DL, VT, BV);
12906 }
12907 
12908 SDValue DAGCombiner::createBuildVecShuffle(SDLoc DL, SDNode *N,
12909                                            ArrayRef<int> VectorMask,
12910                                            SDValue VecIn1, SDValue VecIn2,
12911                                            unsigned LeftIdx) {
12912   MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout());
12913   SDValue ZeroIdx = DAG.getConstant(0, DL, IdxTy);
12914 
12915   EVT VT = N->getValueType(0);
12916   EVT InVT1 = VecIn1.getValueType();
12917   EVT InVT2 = VecIn2.getNode() ? VecIn2.getValueType() : InVT1;
12918 
12919   unsigned Vec2Offset = InVT1.getVectorNumElements();
12920   unsigned NumElems = VT.getVectorNumElements();
12921   unsigned ShuffleNumElems = NumElems;
12922 
12923   // We can't generate a shuffle node with mismatched input and output types.
12924   // Try to make the types match the type of the output.
12925   if (InVT1 != VT || InVT2 != VT) {
12926     if (InVT1.getSizeInBits() * 2 == VT.getSizeInBits() && InVT1 == InVT2) {
12927       // If both input vectors are exactly half the size of the output, concat
12928       // them. If we have only one (non-zero) input, concat it with undef.
12929       VecIn1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, VecIn1,
12930                            VecIn2.getNode() ? VecIn2 : DAG.getUNDEF(InVT1));
12931       VecIn2 = SDValue();
12932     } else if (InVT1.getSizeInBits() == VT.getSizeInBits() * 2) {
12933       if (!TLI.isExtractSubvectorCheap(VT, NumElems))
12934         return SDValue();
12935 
12936       if (!VecIn2.getNode()) {
12937         // If we only have one input vector, and it's twice the size of the
12938         // output, split it in two.
12939         VecIn2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1,
12940                              DAG.getConstant(NumElems, DL, IdxTy));
12941         VecIn1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1, ZeroIdx);
12942         // Since we now have shorter input vectors, adjust the offset of the
12943         // second vector's start.
12944         Vec2Offset = NumElems;
12945       } else if (InVT2.getSizeInBits() <= InVT1.getSizeInBits()) {
12946         // VecIn1 is wider than the output, and we have another, possibly
12947         // smaller input. Pad the smaller input with undefs, shuffle at the
12948         // input vector width, and extract the output.
12949         // The shuffle type is different than VT, so check legality again.
12950         if (LegalOperations &&
12951             !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, InVT1))
12952           return SDValue();
12953 
12954         if (InVT1 != InVT2)
12955           VecIn2 = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, InVT1,
12956                                DAG.getUNDEF(InVT1), VecIn2, ZeroIdx);
12957         ShuffleNumElems = NumElems * 2;
12958       } else {
12959         // Both VecIn1 and VecIn2 are wider than the output, and VecIn2 is wider
12960         // than VecIn1. We can't handle this for now - this case will disappear
12961         // when we start sorting the vectors by type.
12962         return SDValue();
12963       }
12964     } else {
12965       // TODO: Support cases where the length mismatch isn't exactly by a
12966       // factor of 2.
12967       // TODO: Move this check upwards, so that if we have bad type
12968       // mismatches, we don't create any DAG nodes.
12969       return SDValue();
12970     }
12971   }
12972 
12973   // Initialize mask to undef.
12974   SmallVector<int, 8> Mask(ShuffleNumElems, -1);
12975 
12976   // Only need to run up to the number of elements actually used, not the
12977   // total number of elements in the shuffle - if we are shuffling a wider
12978   // vector, the high lanes should be set to undef.
12979   for (unsigned i = 0; i != NumElems; ++i) {
12980     if (VectorMask[i] <= 0)
12981       continue;
12982 
12983     SDValue Extract = N->getOperand(i);
12984     unsigned ExtIndex =
12985         cast<ConstantSDNode>(Extract.getOperand(1))->getZExtValue();
12986 
12987     if (VectorMask[i] == (int)LeftIdx) {
12988       Mask[i] = ExtIndex;
12989     } else if (VectorMask[i] == (int)LeftIdx + 1) {
12990       Mask[i] = Vec2Offset + ExtIndex;
12991     }
12992   }
12993 
12994   // The type the input vectors may have changed above.
12995   InVT1 = VecIn1.getValueType();
12996 
12997   // If we already have a VecIn2, it should have the same type as VecIn1.
12998   // If we don't, get an undef/zero vector of the appropriate type.
12999   VecIn2 = VecIn2.getNode() ? VecIn2 : DAG.getUNDEF(InVT1);
13000   assert(InVT1 == VecIn2.getValueType() && "Unexpected second input type.");
13001 
13002   SDValue Shuffle = DAG.getVectorShuffle(InVT1, DL, VecIn1, VecIn2, Mask);
13003   if (ShuffleNumElems > NumElems)
13004     Shuffle = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Shuffle, ZeroIdx);
13005 
13006   return Shuffle;
13007 }
13008 
13009 // Check to see if this is a BUILD_VECTOR of a bunch of EXTRACT_VECTOR_ELT
13010 // operations. If the types of the vectors we're extracting from allow it,
13011 // turn this into a vector_shuffle node.
13012 SDValue DAGCombiner::reduceBuildVecToShuffle(SDNode *N) {
13013   SDLoc DL(N);
13014   EVT VT = N->getValueType(0);
13015 
13016   // Only type-legal BUILD_VECTOR nodes are converted to shuffle nodes.
13017   if (!isTypeLegal(VT))
13018     return SDValue();
13019 
13020   // May only combine to shuffle after legalize if shuffle is legal.
13021   if (LegalOperations && !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, VT))
13022     return SDValue();
13023 
13024   bool UsesZeroVector = false;
13025   unsigned NumElems = N->getNumOperands();
13026 
13027   // Record, for each element of the newly built vector, which input vector
13028   // that element comes from. -1 stands for undef, 0 for the zero vector,
13029   // and positive values for the input vectors.
13030   // VectorMask maps each element to its vector number, and VecIn maps vector
13031   // numbers to their initial SDValues.
13032 
13033   SmallVector<int, 8> VectorMask(NumElems, -1);
13034   SmallVector<SDValue, 8> VecIn;
13035   VecIn.push_back(SDValue());
13036 
13037   for (unsigned i = 0; i != NumElems; ++i) {
13038     SDValue Op = N->getOperand(i);
13039 
13040     if (Op.isUndef())
13041       continue;
13042 
13043     // See if we can use a blend with a zero vector.
13044     // TODO: Should we generalize this to a blend with an arbitrary constant
13045     // vector?
13046     if (isNullConstant(Op) || isNullFPConstant(Op)) {
13047       UsesZeroVector = true;
13048       VectorMask[i] = 0;
13049       continue;
13050     }
13051 
13052     // Not an undef or zero. If the input is something other than an
13053     // EXTRACT_VECTOR_ELT with a constant index, bail out.
13054     if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
13055         !isa<ConstantSDNode>(Op.getOperand(1)))
13056       return SDValue();
13057 
13058     SDValue ExtractedFromVec = Op.getOperand(0);
13059 
13060     // All inputs must have the same element type as the output.
13061     if (VT.getVectorElementType() !=
13062         ExtractedFromVec.getValueType().getVectorElementType())
13063       return SDValue();
13064 
13065     // Have we seen this input vector before?
13066     // The vectors are expected to be tiny (usually 1 or 2 elements), so using
13067     // a map back from SDValues to numbers isn't worth it.
13068     unsigned Idx = std::distance(
13069         VecIn.begin(), std::find(VecIn.begin(), VecIn.end(), ExtractedFromVec));
13070     if (Idx == VecIn.size())
13071       VecIn.push_back(ExtractedFromVec);
13072 
13073     VectorMask[i] = Idx;
13074   }
13075 
13076   // If we didn't find at least one input vector, bail out.
13077   if (VecIn.size() < 2)
13078     return SDValue();
13079 
13080   // TODO: We want to sort the vectors by descending length, so that adjacent
13081   // pairs have similar length, and the longer vector is always first in the
13082   // pair.
13083 
13084   // TODO: Should this fire if some of the input vectors has illegal type (like
13085   // it does now), or should we let legalization run its course first?
13086 
13087   // Shuffle phase:
13088   // Take pairs of vectors, and shuffle them so that the result has elements
13089   // from these vectors in the correct places.
13090   // For example, given:
13091   // t10: i32 = extract_vector_elt t1, Constant:i64<0>
13092   // t11: i32 = extract_vector_elt t2, Constant:i64<0>
13093   // t12: i32 = extract_vector_elt t3, Constant:i64<0>
13094   // t13: i32 = extract_vector_elt t1, Constant:i64<1>
13095   // t14: v4i32 = BUILD_VECTOR t10, t11, t12, t13
13096   // We will generate:
13097   // t20: v4i32 = vector_shuffle<0,4,u,1> t1, t2
13098   // t21: v4i32 = vector_shuffle<u,u,0,u> t3, undef
13099   SmallVector<SDValue, 4> Shuffles;
13100   for (unsigned In = 0, Len = (VecIn.size() / 2); In < Len; ++In) {
13101     unsigned LeftIdx = 2 * In + 1;
13102     SDValue VecLeft = VecIn[LeftIdx];
13103     SDValue VecRight =
13104         (LeftIdx + 1) < VecIn.size() ? VecIn[LeftIdx + 1] : SDValue();
13105 
13106     if (SDValue Shuffle = createBuildVecShuffle(DL, N, VectorMask, VecLeft,
13107                                                 VecRight, LeftIdx))
13108       Shuffles.push_back(Shuffle);
13109     else
13110       return SDValue();
13111   }
13112 
13113   // If we need the zero vector as an "ingredient" in the blend tree, add it
13114   // to the list of shuffles.
13115   if (UsesZeroVector)
13116     Shuffles.push_back(VT.isInteger() ? DAG.getConstant(0, DL, VT)
13117                                       : DAG.getConstantFP(0.0, DL, VT));
13118 
13119   // If we only have one shuffle, we're done.
13120   if (Shuffles.size() == 1)
13121     return Shuffles[0];
13122 
13123   // Update the vector mask to point to the post-shuffle vectors.
13124   for (int &Vec : VectorMask)
13125     if (Vec == 0)
13126       Vec = Shuffles.size() - 1;
13127     else
13128       Vec = (Vec - 1) / 2;
13129 
13130   // More than one shuffle. Generate a binary tree of blends, e.g. if from
13131   // the previous step we got the set of shuffles t10, t11, t12, t13, we will
13132   // generate:
13133   // t10: v8i32 = vector_shuffle<0,8,u,u,u,u,u,u> t1, t2
13134   // t11: v8i32 = vector_shuffle<u,u,0,8,u,u,u,u> t3, t4
13135   // t12: v8i32 = vector_shuffle<u,u,u,u,0,8,u,u> t5, t6
13136   // t13: v8i32 = vector_shuffle<u,u,u,u,u,u,0,8> t7, t8
13137   // t20: v8i32 = vector_shuffle<0,1,10,11,u,u,u,u> t10, t11
13138   // t21: v8i32 = vector_shuffle<u,u,u,u,4,5,14,15> t12, t13
13139   // t30: v8i32 = vector_shuffle<0,1,2,3,12,13,14,15> t20, t21
13140 
13141   // Make sure the initial size of the shuffle list is even.
13142   if (Shuffles.size() % 2)
13143     Shuffles.push_back(DAG.getUNDEF(VT));
13144 
13145   for (unsigned CurSize = Shuffles.size(); CurSize > 1; CurSize /= 2) {
13146     if (CurSize % 2) {
13147       Shuffles[CurSize] = DAG.getUNDEF(VT);
13148       CurSize++;
13149     }
13150     for (unsigned In = 0, Len = CurSize / 2; In < Len; ++In) {
13151       int Left = 2 * In;
13152       int Right = 2 * In + 1;
13153       SmallVector<int, 8> Mask(NumElems, -1);
13154       for (unsigned i = 0; i != NumElems; ++i) {
13155         if (VectorMask[i] == Left) {
13156           Mask[i] = i;
13157           VectorMask[i] = In;
13158         } else if (VectorMask[i] == Right) {
13159           Mask[i] = i + NumElems;
13160           VectorMask[i] = In;
13161         }
13162       }
13163 
13164       Shuffles[In] =
13165           DAG.getVectorShuffle(VT, DL, Shuffles[Left], Shuffles[Right], Mask);
13166     }
13167   }
13168 
13169   return Shuffles[0];
13170 }
13171 
13172 SDValue DAGCombiner::visitBUILD_VECTOR(SDNode *N) {
13173   EVT VT = N->getValueType(0);
13174 
13175   // A vector built entirely of undefs is undef.
13176   if (ISD::allOperandsUndef(N))
13177     return DAG.getUNDEF(VT);
13178 
13179   if (SDValue V = reduceBuildVecExtToExtBuildVec(N))
13180     return V;
13181 
13182   if (SDValue V = reduceBuildVecConvertToConvertBuildVec(N))
13183     return V;
13184 
13185   if (SDValue V = reduceBuildVecToShuffle(N))
13186     return V;
13187 
13188   return SDValue();
13189 }
13190 
13191 static SDValue combineConcatVectorOfScalars(SDNode *N, SelectionDAG &DAG) {
13192   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
13193   EVT OpVT = N->getOperand(0).getValueType();
13194 
13195   // If the operands are legal vectors, leave them alone.
13196   if (TLI.isTypeLegal(OpVT))
13197     return SDValue();
13198 
13199   SDLoc DL(N);
13200   EVT VT = N->getValueType(0);
13201   SmallVector<SDValue, 8> Ops;
13202 
13203   EVT SVT = EVT::getIntegerVT(*DAG.getContext(), OpVT.getSizeInBits());
13204   SDValue ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
13205 
13206   // Keep track of what we encounter.
13207   bool AnyInteger = false;
13208   bool AnyFP = false;
13209   for (const SDValue &Op : N->ops()) {
13210     if (ISD::BITCAST == Op.getOpcode() &&
13211         !Op.getOperand(0).getValueType().isVector())
13212       Ops.push_back(Op.getOperand(0));
13213     else if (ISD::UNDEF == Op.getOpcode())
13214       Ops.push_back(ScalarUndef);
13215     else
13216       return SDValue();
13217 
13218     // Note whether we encounter an integer or floating point scalar.
13219     // If it's neither, bail out, it could be something weird like x86mmx.
13220     EVT LastOpVT = Ops.back().getValueType();
13221     if (LastOpVT.isFloatingPoint())
13222       AnyFP = true;
13223     else if (LastOpVT.isInteger())
13224       AnyInteger = true;
13225     else
13226       return SDValue();
13227   }
13228 
13229   // If any of the operands is a floating point scalar bitcast to a vector,
13230   // use floating point types throughout, and bitcast everything.
13231   // Replace UNDEFs by another scalar UNDEF node, of the final desired type.
13232   if (AnyFP) {
13233     SVT = EVT::getFloatingPointVT(OpVT.getSizeInBits());
13234     ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
13235     if (AnyInteger) {
13236       for (SDValue &Op : Ops) {
13237         if (Op.getValueType() == SVT)
13238           continue;
13239         if (Op.isUndef())
13240           Op = ScalarUndef;
13241         else
13242           Op = DAG.getBitcast(SVT, Op);
13243       }
13244     }
13245   }
13246 
13247   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SVT,
13248                                VT.getSizeInBits() / SVT.getSizeInBits());
13249   return DAG.getBitcast(VT, DAG.getBuildVector(VecVT, DL, Ops));
13250 }
13251 
13252 // Check to see if this is a CONCAT_VECTORS of a bunch of EXTRACT_SUBVECTOR
13253 // operations. If so, and if the EXTRACT_SUBVECTOR vector inputs come from at
13254 // most two distinct vectors the same size as the result, attempt to turn this
13255 // into a legal shuffle.
13256 static SDValue combineConcatVectorOfExtracts(SDNode *N, SelectionDAG &DAG) {
13257   EVT VT = N->getValueType(0);
13258   EVT OpVT = N->getOperand(0).getValueType();
13259   int NumElts = VT.getVectorNumElements();
13260   int NumOpElts = OpVT.getVectorNumElements();
13261 
13262   SDValue SV0 = DAG.getUNDEF(VT), SV1 = DAG.getUNDEF(VT);
13263   SmallVector<int, 8> Mask;
13264 
13265   for (SDValue Op : N->ops()) {
13266     // Peek through any bitcast.
13267     while (Op.getOpcode() == ISD::BITCAST)
13268       Op = Op.getOperand(0);
13269 
13270     // UNDEF nodes convert to UNDEF shuffle mask values.
13271     if (Op.isUndef()) {
13272       Mask.append((unsigned)NumOpElts, -1);
13273       continue;
13274     }
13275 
13276     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
13277       return SDValue();
13278 
13279     // What vector are we extracting the subvector from and at what index?
13280     SDValue ExtVec = Op.getOperand(0);
13281 
13282     // We want the EVT of the original extraction to correctly scale the
13283     // extraction index.
13284     EVT ExtVT = ExtVec.getValueType();
13285 
13286     // Peek through any bitcast.
13287     while (ExtVec.getOpcode() == ISD::BITCAST)
13288       ExtVec = ExtVec.getOperand(0);
13289 
13290     // UNDEF nodes convert to UNDEF shuffle mask values.
13291     if (ExtVec.isUndef()) {
13292       Mask.append((unsigned)NumOpElts, -1);
13293       continue;
13294     }
13295 
13296     if (!isa<ConstantSDNode>(Op.getOperand(1)))
13297       return SDValue();
13298     int ExtIdx = cast<ConstantSDNode>(Op.getOperand(1))->getZExtValue();
13299 
13300     // Ensure that we are extracting a subvector from a vector the same
13301     // size as the result.
13302     if (ExtVT.getSizeInBits() != VT.getSizeInBits())
13303       return SDValue();
13304 
13305     // Scale the subvector index to account for any bitcast.
13306     int NumExtElts = ExtVT.getVectorNumElements();
13307     if (0 == (NumExtElts % NumElts))
13308       ExtIdx /= (NumExtElts / NumElts);
13309     else if (0 == (NumElts % NumExtElts))
13310       ExtIdx *= (NumElts / NumExtElts);
13311     else
13312       return SDValue();
13313 
13314     // At most we can reference 2 inputs in the final shuffle.
13315     if (SV0.isUndef() || SV0 == ExtVec) {
13316       SV0 = ExtVec;
13317       for (int i = 0; i != NumOpElts; ++i)
13318         Mask.push_back(i + ExtIdx);
13319     } else if (SV1.isUndef() || SV1 == ExtVec) {
13320       SV1 = ExtVec;
13321       for (int i = 0; i != NumOpElts; ++i)
13322         Mask.push_back(i + ExtIdx + NumElts);
13323     } else {
13324       return SDValue();
13325     }
13326   }
13327 
13328   if (!DAG.getTargetLoweringInfo().isShuffleMaskLegal(Mask, VT))
13329     return SDValue();
13330 
13331   return DAG.getVectorShuffle(VT, SDLoc(N), DAG.getBitcast(VT, SV0),
13332                               DAG.getBitcast(VT, SV1), Mask);
13333 }
13334 
13335 SDValue DAGCombiner::visitCONCAT_VECTORS(SDNode *N) {
13336   // If we only have one input vector, we don't need to do any concatenation.
13337   if (N->getNumOperands() == 1)
13338     return N->getOperand(0);
13339 
13340   // Check if all of the operands are undefs.
13341   EVT VT = N->getValueType(0);
13342   if (ISD::allOperandsUndef(N))
13343     return DAG.getUNDEF(VT);
13344 
13345   // Optimize concat_vectors where all but the first of the vectors are undef.
13346   if (std::all_of(std::next(N->op_begin()), N->op_end(), [](const SDValue &Op) {
13347         return Op.isUndef();
13348       })) {
13349     SDValue In = N->getOperand(0);
13350     assert(In.getValueType().isVector() && "Must concat vectors");
13351 
13352     // Transform: concat_vectors(scalar, undef) -> scalar_to_vector(sclr).
13353     if (In->getOpcode() == ISD::BITCAST &&
13354         !In->getOperand(0)->getValueType(0).isVector()) {
13355       SDValue Scalar = In->getOperand(0);
13356 
13357       // If the bitcast type isn't legal, it might be a trunc of a legal type;
13358       // look through the trunc so we can still do the transform:
13359       //   concat_vectors(trunc(scalar), undef) -> scalar_to_vector(scalar)
13360       if (Scalar->getOpcode() == ISD::TRUNCATE &&
13361           !TLI.isTypeLegal(Scalar.getValueType()) &&
13362           TLI.isTypeLegal(Scalar->getOperand(0).getValueType()))
13363         Scalar = Scalar->getOperand(0);
13364 
13365       EVT SclTy = Scalar->getValueType(0);
13366 
13367       if (!SclTy.isFloatingPoint() && !SclTy.isInteger())
13368         return SDValue();
13369 
13370       EVT NVT = EVT::getVectorVT(*DAG.getContext(), SclTy,
13371                                  VT.getSizeInBits() / SclTy.getSizeInBits());
13372       if (!TLI.isTypeLegal(NVT) || !TLI.isTypeLegal(Scalar.getValueType()))
13373         return SDValue();
13374 
13375       SDValue Res = DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), NVT, Scalar);
13376       return DAG.getBitcast(VT, Res);
13377     }
13378   }
13379 
13380   // Fold any combination of BUILD_VECTOR or UNDEF nodes into one BUILD_VECTOR.
13381   // We have already tested above for an UNDEF only concatenation.
13382   // fold (concat_vectors (BUILD_VECTOR A, B, ...), (BUILD_VECTOR C, D, ...))
13383   // -> (BUILD_VECTOR A, B, ..., C, D, ...)
13384   auto IsBuildVectorOrUndef = [](const SDValue &Op) {
13385     return ISD::UNDEF == Op.getOpcode() || ISD::BUILD_VECTOR == Op.getOpcode();
13386   };
13387   if (llvm::all_of(N->ops(), IsBuildVectorOrUndef)) {
13388     SmallVector<SDValue, 8> Opnds;
13389     EVT SVT = VT.getScalarType();
13390 
13391     EVT MinVT = SVT;
13392     if (!SVT.isFloatingPoint()) {
13393       // If BUILD_VECTOR are from built from integer, they may have different
13394       // operand types. Get the smallest type and truncate all operands to it.
13395       bool FoundMinVT = false;
13396       for (const SDValue &Op : N->ops())
13397         if (ISD::BUILD_VECTOR == Op.getOpcode()) {
13398           EVT OpSVT = Op.getOperand(0)->getValueType(0);
13399           MinVT = (!FoundMinVT || OpSVT.bitsLE(MinVT)) ? OpSVT : MinVT;
13400           FoundMinVT = true;
13401         }
13402       assert(FoundMinVT && "Concat vector type mismatch");
13403     }
13404 
13405     for (const SDValue &Op : N->ops()) {
13406       EVT OpVT = Op.getValueType();
13407       unsigned NumElts = OpVT.getVectorNumElements();
13408 
13409       if (ISD::UNDEF == Op.getOpcode())
13410         Opnds.append(NumElts, DAG.getUNDEF(MinVT));
13411 
13412       if (ISD::BUILD_VECTOR == Op.getOpcode()) {
13413         if (SVT.isFloatingPoint()) {
13414           assert(SVT == OpVT.getScalarType() && "Concat vector type mismatch");
13415           Opnds.append(Op->op_begin(), Op->op_begin() + NumElts);
13416         } else {
13417           for (unsigned i = 0; i != NumElts; ++i)
13418             Opnds.push_back(
13419                 DAG.getNode(ISD::TRUNCATE, SDLoc(N), MinVT, Op.getOperand(i)));
13420         }
13421       }
13422     }
13423 
13424     assert(VT.getVectorNumElements() == Opnds.size() &&
13425            "Concat vector type mismatch");
13426     return DAG.getBuildVector(VT, SDLoc(N), Opnds);
13427   }
13428 
13429   // Fold CONCAT_VECTORS of only bitcast scalars (or undef) to BUILD_VECTOR.
13430   if (SDValue V = combineConcatVectorOfScalars(N, DAG))
13431     return V;
13432 
13433   // Fold CONCAT_VECTORS of EXTRACT_SUBVECTOR (or undef) to VECTOR_SHUFFLE.
13434   if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT))
13435     if (SDValue V = combineConcatVectorOfExtracts(N, DAG))
13436       return V;
13437 
13438   // Type legalization of vectors and DAG canonicalization of SHUFFLE_VECTOR
13439   // nodes often generate nop CONCAT_VECTOR nodes.
13440   // Scan the CONCAT_VECTOR operands and look for a CONCAT operations that
13441   // place the incoming vectors at the exact same location.
13442   SDValue SingleSource = SDValue();
13443   unsigned PartNumElem = N->getOperand(0).getValueType().getVectorNumElements();
13444 
13445   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
13446     SDValue Op = N->getOperand(i);
13447 
13448     if (Op.isUndef())
13449       continue;
13450 
13451     // Check if this is the identity extract:
13452     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
13453       return SDValue();
13454 
13455     // Find the single incoming vector for the extract_subvector.
13456     if (SingleSource.getNode()) {
13457       if (Op.getOperand(0) != SingleSource)
13458         return SDValue();
13459     } else {
13460       SingleSource = Op.getOperand(0);
13461 
13462       // Check the source type is the same as the type of the result.
13463       // If not, this concat may extend the vector, so we can not
13464       // optimize it away.
13465       if (SingleSource.getValueType() != N->getValueType(0))
13466         return SDValue();
13467     }
13468 
13469     unsigned IdentityIndex = i * PartNumElem;
13470     ConstantSDNode *CS = dyn_cast<ConstantSDNode>(Op.getOperand(1));
13471     // The extract index must be constant.
13472     if (!CS)
13473       return SDValue();
13474 
13475     // Check that we are reading from the identity index.
13476     if (CS->getZExtValue() != IdentityIndex)
13477       return SDValue();
13478   }
13479 
13480   if (SingleSource.getNode())
13481     return SingleSource;
13482 
13483   return SDValue();
13484 }
13485 
13486 SDValue DAGCombiner::visitEXTRACT_SUBVECTOR(SDNode* N) {
13487   EVT NVT = N->getValueType(0);
13488   SDValue V = N->getOperand(0);
13489 
13490   if (V->getOpcode() == ISD::CONCAT_VECTORS) {
13491     // Combine:
13492     //    (extract_subvec (concat V1, V2, ...), i)
13493     // Into:
13494     //    Vi if possible
13495     // Only operand 0 is checked as 'concat' assumes all inputs of the same
13496     // type.
13497     if (V->getOperand(0).getValueType() != NVT)
13498       return SDValue();
13499     unsigned Idx = N->getConstantOperandVal(1);
13500     unsigned NumElems = NVT.getVectorNumElements();
13501     assert((Idx % NumElems) == 0 &&
13502            "IDX in concat is not a multiple of the result vector length.");
13503     return V->getOperand(Idx / NumElems);
13504   }
13505 
13506   // Skip bitcasting
13507   if (V->getOpcode() == ISD::BITCAST)
13508     V = V.getOperand(0);
13509 
13510   if (V->getOpcode() == ISD::INSERT_SUBVECTOR) {
13511     // Handle only simple case where vector being inserted and vector
13512     // being extracted are of same type, and are half size of larger vectors.
13513     EVT BigVT = V->getOperand(0).getValueType();
13514     EVT SmallVT = V->getOperand(1).getValueType();
13515     if (!NVT.bitsEq(SmallVT) || NVT.getSizeInBits()*2 != BigVT.getSizeInBits())
13516       return SDValue();
13517 
13518     // Only handle cases where both indexes are constants with the same type.
13519     ConstantSDNode *ExtIdx = dyn_cast<ConstantSDNode>(N->getOperand(1));
13520     ConstantSDNode *InsIdx = dyn_cast<ConstantSDNode>(V->getOperand(2));
13521 
13522     if (InsIdx && ExtIdx &&
13523         InsIdx->getValueType(0).getSizeInBits() <= 64 &&
13524         ExtIdx->getValueType(0).getSizeInBits() <= 64) {
13525       // Combine:
13526       //    (extract_subvec (insert_subvec V1, V2, InsIdx), ExtIdx)
13527       // Into:
13528       //    indices are equal or bit offsets are equal => V1
13529       //    otherwise => (extract_subvec V1, ExtIdx)
13530       if (InsIdx->getZExtValue() * SmallVT.getScalarSizeInBits() ==
13531           ExtIdx->getZExtValue() * NVT.getScalarSizeInBits())
13532         return DAG.getBitcast(NVT, V->getOperand(1));
13533       return DAG.getNode(
13534           ISD::EXTRACT_SUBVECTOR, SDLoc(N), NVT,
13535           DAG.getBitcast(N->getOperand(0).getValueType(), V->getOperand(0)),
13536           N->getOperand(1));
13537     }
13538   }
13539 
13540   return SDValue();
13541 }
13542 
13543 static SDValue simplifyShuffleOperandRecursively(SmallBitVector &UsedElements,
13544                                                  SDValue V, SelectionDAG &DAG) {
13545   SDLoc DL(V);
13546   EVT VT = V.getValueType();
13547 
13548   switch (V.getOpcode()) {
13549   default:
13550     return V;
13551 
13552   case ISD::CONCAT_VECTORS: {
13553     EVT OpVT = V->getOperand(0).getValueType();
13554     int OpSize = OpVT.getVectorNumElements();
13555     SmallBitVector OpUsedElements(OpSize, false);
13556     bool FoundSimplification = false;
13557     SmallVector<SDValue, 4> NewOps;
13558     NewOps.reserve(V->getNumOperands());
13559     for (int i = 0, NumOps = V->getNumOperands(); i < NumOps; ++i) {
13560       SDValue Op = V->getOperand(i);
13561       bool OpUsed = false;
13562       for (int j = 0; j < OpSize; ++j)
13563         if (UsedElements[i * OpSize + j]) {
13564           OpUsedElements[j] = true;
13565           OpUsed = true;
13566         }
13567       NewOps.push_back(
13568           OpUsed ? simplifyShuffleOperandRecursively(OpUsedElements, Op, DAG)
13569                  : DAG.getUNDEF(OpVT));
13570       FoundSimplification |= Op == NewOps.back();
13571       OpUsedElements.reset();
13572     }
13573     if (FoundSimplification)
13574       V = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, NewOps);
13575     return V;
13576   }
13577 
13578   case ISD::INSERT_SUBVECTOR: {
13579     SDValue BaseV = V->getOperand(0);
13580     SDValue SubV = V->getOperand(1);
13581     auto *IdxN = dyn_cast<ConstantSDNode>(V->getOperand(2));
13582     if (!IdxN)
13583       return V;
13584 
13585     int SubSize = SubV.getValueType().getVectorNumElements();
13586     int Idx = IdxN->getZExtValue();
13587     bool SubVectorUsed = false;
13588     SmallBitVector SubUsedElements(SubSize, false);
13589     for (int i = 0; i < SubSize; ++i)
13590       if (UsedElements[i + Idx]) {
13591         SubVectorUsed = true;
13592         SubUsedElements[i] = true;
13593         UsedElements[i + Idx] = false;
13594       }
13595 
13596     // Now recurse on both the base and sub vectors.
13597     SDValue SimplifiedSubV =
13598         SubVectorUsed
13599             ? simplifyShuffleOperandRecursively(SubUsedElements, SubV, DAG)
13600             : DAG.getUNDEF(SubV.getValueType());
13601     SDValue SimplifiedBaseV = simplifyShuffleOperandRecursively(UsedElements, BaseV, DAG);
13602     if (SimplifiedSubV != SubV || SimplifiedBaseV != BaseV)
13603       V = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VT,
13604                       SimplifiedBaseV, SimplifiedSubV, V->getOperand(2));
13605     return V;
13606   }
13607   }
13608 }
13609 
13610 static SDValue simplifyShuffleOperands(ShuffleVectorSDNode *SVN, SDValue N0,
13611                                        SDValue N1, SelectionDAG &DAG) {
13612   EVT VT = SVN->getValueType(0);
13613   int NumElts = VT.getVectorNumElements();
13614   SmallBitVector N0UsedElements(NumElts, false), N1UsedElements(NumElts, false);
13615   for (int M : SVN->getMask())
13616     if (M >= 0 && M < NumElts)
13617       N0UsedElements[M] = true;
13618     else if (M >= NumElts)
13619       N1UsedElements[M - NumElts] = true;
13620 
13621   SDValue S0 = simplifyShuffleOperandRecursively(N0UsedElements, N0, DAG);
13622   SDValue S1 = simplifyShuffleOperandRecursively(N1UsedElements, N1, DAG);
13623   if (S0 == N0 && S1 == N1)
13624     return SDValue();
13625 
13626   return DAG.getVectorShuffle(VT, SDLoc(SVN), S0, S1, SVN->getMask());
13627 }
13628 
13629 // Tries to turn a shuffle of two CONCAT_VECTORS into a single concat,
13630 // or turn a shuffle of a single concat into simpler shuffle then concat.
13631 static SDValue partitionShuffleOfConcats(SDNode *N, SelectionDAG &DAG) {
13632   EVT VT = N->getValueType(0);
13633   unsigned NumElts = VT.getVectorNumElements();
13634 
13635   SDValue N0 = N->getOperand(0);
13636   SDValue N1 = N->getOperand(1);
13637   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
13638 
13639   SmallVector<SDValue, 4> Ops;
13640   EVT ConcatVT = N0.getOperand(0).getValueType();
13641   unsigned NumElemsPerConcat = ConcatVT.getVectorNumElements();
13642   unsigned NumConcats = NumElts / NumElemsPerConcat;
13643 
13644   // Special case: shuffle(concat(A,B)) can be more efficiently represented
13645   // as concat(shuffle(A,B),UNDEF) if the shuffle doesn't set any of the high
13646   // half vector elements.
13647   if (NumElemsPerConcat * 2 == NumElts && N1.isUndef() &&
13648       std::all_of(SVN->getMask().begin() + NumElemsPerConcat,
13649                   SVN->getMask().end(), [](int i) { return i == -1; })) {
13650     N0 = DAG.getVectorShuffle(ConcatVT, SDLoc(N), N0.getOperand(0), N0.getOperand(1),
13651                               makeArrayRef(SVN->getMask().begin(), NumElemsPerConcat));
13652     N1 = DAG.getUNDEF(ConcatVT);
13653     return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N0, N1);
13654   }
13655 
13656   // Look at every vector that's inserted. We're looking for exact
13657   // subvector-sized copies from a concatenated vector
13658   for (unsigned I = 0; I != NumConcats; ++I) {
13659     // Make sure we're dealing with a copy.
13660     unsigned Begin = I * NumElemsPerConcat;
13661     bool AllUndef = true, NoUndef = true;
13662     for (unsigned J = Begin; J != Begin + NumElemsPerConcat; ++J) {
13663       if (SVN->getMaskElt(J) >= 0)
13664         AllUndef = false;
13665       else
13666         NoUndef = false;
13667     }
13668 
13669     if (NoUndef) {
13670       if (SVN->getMaskElt(Begin) % NumElemsPerConcat != 0)
13671         return SDValue();
13672 
13673       for (unsigned J = 1; J != NumElemsPerConcat; ++J)
13674         if (SVN->getMaskElt(Begin + J - 1) + 1 != SVN->getMaskElt(Begin + J))
13675           return SDValue();
13676 
13677       unsigned FirstElt = SVN->getMaskElt(Begin) / NumElemsPerConcat;
13678       if (FirstElt < N0.getNumOperands())
13679         Ops.push_back(N0.getOperand(FirstElt));
13680       else
13681         Ops.push_back(N1.getOperand(FirstElt - N0.getNumOperands()));
13682 
13683     } else if (AllUndef) {
13684       Ops.push_back(DAG.getUNDEF(N0.getOperand(0).getValueType()));
13685     } else { // Mixed with general masks and undefs, can't do optimization.
13686       return SDValue();
13687     }
13688   }
13689 
13690   return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops);
13691 }
13692 
13693 SDValue DAGCombiner::visitVECTOR_SHUFFLE(SDNode *N) {
13694   EVT VT = N->getValueType(0);
13695   unsigned NumElts = VT.getVectorNumElements();
13696 
13697   SDValue N0 = N->getOperand(0);
13698   SDValue N1 = N->getOperand(1);
13699 
13700   assert(N0.getValueType() == VT && "Vector shuffle must be normalized in DAG");
13701 
13702   // Canonicalize shuffle undef, undef -> undef
13703   if (N0.isUndef() && N1.isUndef())
13704     return DAG.getUNDEF(VT);
13705 
13706   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
13707 
13708   // Canonicalize shuffle v, v -> v, undef
13709   if (N0 == N1) {
13710     SmallVector<int, 8> NewMask;
13711     for (unsigned i = 0; i != NumElts; ++i) {
13712       int Idx = SVN->getMaskElt(i);
13713       if (Idx >= (int)NumElts) Idx -= NumElts;
13714       NewMask.push_back(Idx);
13715     }
13716     return DAG.getVectorShuffle(VT, SDLoc(N), N0, DAG.getUNDEF(VT), NewMask);
13717   }
13718 
13719   // Canonicalize shuffle undef, v -> v, undef.  Commute the shuffle mask.
13720   if (N0.isUndef())
13721     return DAG.getCommutedVectorShuffle(*SVN);
13722 
13723   // Remove references to rhs if it is undef
13724   if (N1.isUndef()) {
13725     bool Changed = false;
13726     SmallVector<int, 8> NewMask;
13727     for (unsigned i = 0; i != NumElts; ++i) {
13728       int Idx = SVN->getMaskElt(i);
13729       if (Idx >= (int)NumElts) {
13730         Idx = -1;
13731         Changed = true;
13732       }
13733       NewMask.push_back(Idx);
13734     }
13735     if (Changed)
13736       return DAG.getVectorShuffle(VT, SDLoc(N), N0, N1, NewMask);
13737   }
13738 
13739   // If it is a splat, check if the argument vector is another splat or a
13740   // build_vector.
13741   if (SVN->isSplat() && SVN->getSplatIndex() < (int)NumElts) {
13742     SDNode *V = N0.getNode();
13743 
13744     // If this is a bit convert that changes the element type of the vector but
13745     // not the number of vector elements, look through it.  Be careful not to
13746     // look though conversions that change things like v4f32 to v2f64.
13747     if (V->getOpcode() == ISD::BITCAST) {
13748       SDValue ConvInput = V->getOperand(0);
13749       if (ConvInput.getValueType().isVector() &&
13750           ConvInput.getValueType().getVectorNumElements() == NumElts)
13751         V = ConvInput.getNode();
13752     }
13753 
13754     if (V->getOpcode() == ISD::BUILD_VECTOR) {
13755       assert(V->getNumOperands() == NumElts &&
13756              "BUILD_VECTOR has wrong number of operands");
13757       SDValue Base;
13758       bool AllSame = true;
13759       for (unsigned i = 0; i != NumElts; ++i) {
13760         if (!V->getOperand(i).isUndef()) {
13761           Base = V->getOperand(i);
13762           break;
13763         }
13764       }
13765       // Splat of <u, u, u, u>, return <u, u, u, u>
13766       if (!Base.getNode())
13767         return N0;
13768       for (unsigned i = 0; i != NumElts; ++i) {
13769         if (V->getOperand(i) != Base) {
13770           AllSame = false;
13771           break;
13772         }
13773       }
13774       // Splat of <x, x, x, x>, return <x, x, x, x>
13775       if (AllSame)
13776         return N0;
13777 
13778       // Canonicalize any other splat as a build_vector.
13779       const SDValue &Splatted = V->getOperand(SVN->getSplatIndex());
13780       SmallVector<SDValue, 8> Ops(NumElts, Splatted);
13781       SDValue NewBV = DAG.getBuildVector(V->getValueType(0), SDLoc(N), Ops);
13782 
13783       // We may have jumped through bitcasts, so the type of the
13784       // BUILD_VECTOR may not match the type of the shuffle.
13785       if (V->getValueType(0) != VT)
13786         NewBV = DAG.getBitcast(VT, NewBV);
13787       return NewBV;
13788     }
13789   }
13790 
13791   // There are various patterns used to build up a vector from smaller vectors,
13792   // subvectors, or elements. Scan chains of these and replace unused insertions
13793   // or components with undef.
13794   if (SDValue S = simplifyShuffleOperands(SVN, N0, N1, DAG))
13795     return S;
13796 
13797   if (N0.getOpcode() == ISD::CONCAT_VECTORS &&
13798       Level < AfterLegalizeVectorOps &&
13799       (N1.isUndef() ||
13800       (N1.getOpcode() == ISD::CONCAT_VECTORS &&
13801        N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType()))) {
13802     if (SDValue V = partitionShuffleOfConcats(N, DAG))
13803       return V;
13804   }
13805 
13806   // Attempt to combine a shuffle of 2 inputs of 'scalar sources' -
13807   // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR.
13808   if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT)) {
13809     SmallVector<SDValue, 8> Ops;
13810     for (int M : SVN->getMask()) {
13811       SDValue Op = DAG.getUNDEF(VT.getScalarType());
13812       if (M >= 0) {
13813         int Idx = M % NumElts;
13814         SDValue &S = (M < (int)NumElts ? N0 : N1);
13815         if (S.getOpcode() == ISD::BUILD_VECTOR && S.hasOneUse()) {
13816           Op = S.getOperand(Idx);
13817         } else if (S.getOpcode() == ISD::SCALAR_TO_VECTOR && S.hasOneUse()) {
13818           if (Idx == 0)
13819             Op = S.getOperand(0);
13820         } else {
13821           // Operand can't be combined - bail out.
13822           break;
13823         }
13824       }
13825       Ops.push_back(Op);
13826     }
13827     if (Ops.size() == VT.getVectorNumElements()) {
13828       // BUILD_VECTOR requires all inputs to be of the same type, find the
13829       // maximum type and extend them all.
13830       EVT SVT = VT.getScalarType();
13831       if (SVT.isInteger())
13832         for (SDValue &Op : Ops)
13833           SVT = (SVT.bitsLT(Op.getValueType()) ? Op.getValueType() : SVT);
13834       if (SVT != VT.getScalarType())
13835         for (SDValue &Op : Ops)
13836           Op = TLI.isZExtFree(Op.getValueType(), SVT)
13837                    ? DAG.getZExtOrTrunc(Op, SDLoc(N), SVT)
13838                    : DAG.getSExtOrTrunc(Op, SDLoc(N), SVT);
13839       return DAG.getBuildVector(VT, SDLoc(N), Ops);
13840     }
13841   }
13842 
13843   // If this shuffle only has a single input that is a bitcasted shuffle,
13844   // attempt to merge the 2 shuffles and suitably bitcast the inputs/output
13845   // back to their original types.
13846   if (N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
13847       N1.isUndef() && Level < AfterLegalizeVectorOps &&
13848       TLI.isTypeLegal(VT)) {
13849 
13850     // Peek through the bitcast only if there is one user.
13851     SDValue BC0 = N0;
13852     while (BC0.getOpcode() == ISD::BITCAST) {
13853       if (!BC0.hasOneUse())
13854         break;
13855       BC0 = BC0.getOperand(0);
13856     }
13857 
13858     auto ScaleShuffleMask = [](ArrayRef<int> Mask, int Scale) {
13859       if (Scale == 1)
13860         return SmallVector<int, 8>(Mask.begin(), Mask.end());
13861 
13862       SmallVector<int, 8> NewMask;
13863       for (int M : Mask)
13864         for (int s = 0; s != Scale; ++s)
13865           NewMask.push_back(M < 0 ? -1 : Scale * M + s);
13866       return NewMask;
13867     };
13868 
13869     if (BC0.getOpcode() == ISD::VECTOR_SHUFFLE && BC0.hasOneUse()) {
13870       EVT SVT = VT.getScalarType();
13871       EVT InnerVT = BC0->getValueType(0);
13872       EVT InnerSVT = InnerVT.getScalarType();
13873 
13874       // Determine which shuffle works with the smaller scalar type.
13875       EVT ScaleVT = SVT.bitsLT(InnerSVT) ? VT : InnerVT;
13876       EVT ScaleSVT = ScaleVT.getScalarType();
13877 
13878       if (TLI.isTypeLegal(ScaleVT) &&
13879           0 == (InnerSVT.getSizeInBits() % ScaleSVT.getSizeInBits()) &&
13880           0 == (SVT.getSizeInBits() % ScaleSVT.getSizeInBits())) {
13881 
13882         int InnerScale = InnerSVT.getSizeInBits() / ScaleSVT.getSizeInBits();
13883         int OuterScale = SVT.getSizeInBits() / ScaleSVT.getSizeInBits();
13884 
13885         // Scale the shuffle masks to the smaller scalar type.
13886         ShuffleVectorSDNode *InnerSVN = cast<ShuffleVectorSDNode>(BC0);
13887         SmallVector<int, 8> InnerMask =
13888             ScaleShuffleMask(InnerSVN->getMask(), InnerScale);
13889         SmallVector<int, 8> OuterMask =
13890             ScaleShuffleMask(SVN->getMask(), OuterScale);
13891 
13892         // Merge the shuffle masks.
13893         SmallVector<int, 8> NewMask;
13894         for (int M : OuterMask)
13895           NewMask.push_back(M < 0 ? -1 : InnerMask[M]);
13896 
13897         // Test for shuffle mask legality over both commutations.
13898         SDValue SV0 = BC0->getOperand(0);
13899         SDValue SV1 = BC0->getOperand(1);
13900         bool LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
13901         if (!LegalMask) {
13902           std::swap(SV0, SV1);
13903           ShuffleVectorSDNode::commuteMask(NewMask);
13904           LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
13905         }
13906 
13907         if (LegalMask) {
13908           SV0 = DAG.getBitcast(ScaleVT, SV0);
13909           SV1 = DAG.getBitcast(ScaleVT, SV1);
13910           return DAG.getBitcast(
13911               VT, DAG.getVectorShuffle(ScaleVT, SDLoc(N), SV0, SV1, NewMask));
13912         }
13913       }
13914     }
13915   }
13916 
13917   // Canonicalize shuffles according to rules:
13918   //  shuffle(A, shuffle(A, B)) -> shuffle(shuffle(A,B), A)
13919   //  shuffle(B, shuffle(A, B)) -> shuffle(shuffle(A,B), B)
13920   //  shuffle(B, shuffle(A, Undef)) -> shuffle(shuffle(A, Undef), B)
13921   if (N1.getOpcode() == ISD::VECTOR_SHUFFLE &&
13922       N0.getOpcode() != ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG &&
13923       TLI.isTypeLegal(VT)) {
13924     // The incoming shuffle must be of the same type as the result of the
13925     // current shuffle.
13926     assert(N1->getOperand(0).getValueType() == VT &&
13927            "Shuffle types don't match");
13928 
13929     SDValue SV0 = N1->getOperand(0);
13930     SDValue SV1 = N1->getOperand(1);
13931     bool HasSameOp0 = N0 == SV0;
13932     bool IsSV1Undef = SV1.isUndef();
13933     if (HasSameOp0 || IsSV1Undef || N0 == SV1)
13934       // Commute the operands of this shuffle so that next rule
13935       // will trigger.
13936       return DAG.getCommutedVectorShuffle(*SVN);
13937   }
13938 
13939   // Try to fold according to rules:
13940   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
13941   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
13942   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
13943   // Don't try to fold shuffles with illegal type.
13944   // Only fold if this shuffle is the only user of the other shuffle.
13945   if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && N->isOnlyUserOf(N0.getNode()) &&
13946       Level < AfterLegalizeDAG && TLI.isTypeLegal(VT)) {
13947     ShuffleVectorSDNode *OtherSV = cast<ShuffleVectorSDNode>(N0);
13948 
13949     // The incoming shuffle must be of the same type as the result of the
13950     // current shuffle.
13951     assert(OtherSV->getOperand(0).getValueType() == VT &&
13952            "Shuffle types don't match");
13953 
13954     SDValue SV0, SV1;
13955     SmallVector<int, 4> Mask;
13956     // Compute the combined shuffle mask for a shuffle with SV0 as the first
13957     // operand, and SV1 as the second operand.
13958     for (unsigned i = 0; i != NumElts; ++i) {
13959       int Idx = SVN->getMaskElt(i);
13960       if (Idx < 0) {
13961         // Propagate Undef.
13962         Mask.push_back(Idx);
13963         continue;
13964       }
13965 
13966       SDValue CurrentVec;
13967       if (Idx < (int)NumElts) {
13968         // This shuffle index refers to the inner shuffle N0. Lookup the inner
13969         // shuffle mask to identify which vector is actually referenced.
13970         Idx = OtherSV->getMaskElt(Idx);
13971         if (Idx < 0) {
13972           // Propagate Undef.
13973           Mask.push_back(Idx);
13974           continue;
13975         }
13976 
13977         CurrentVec = (Idx < (int) NumElts) ? OtherSV->getOperand(0)
13978                                            : OtherSV->getOperand(1);
13979       } else {
13980         // This shuffle index references an element within N1.
13981         CurrentVec = N1;
13982       }
13983 
13984       // Simple case where 'CurrentVec' is UNDEF.
13985       if (CurrentVec.isUndef()) {
13986         Mask.push_back(-1);
13987         continue;
13988       }
13989 
13990       // Canonicalize the shuffle index. We don't know yet if CurrentVec
13991       // will be the first or second operand of the combined shuffle.
13992       Idx = Idx % NumElts;
13993       if (!SV0.getNode() || SV0 == CurrentVec) {
13994         // Ok. CurrentVec is the left hand side.
13995         // Update the mask accordingly.
13996         SV0 = CurrentVec;
13997         Mask.push_back(Idx);
13998         continue;
13999       }
14000 
14001       // Bail out if we cannot convert the shuffle pair into a single shuffle.
14002       if (SV1.getNode() && SV1 != CurrentVec)
14003         return SDValue();
14004 
14005       // Ok. CurrentVec is the right hand side.
14006       // Update the mask accordingly.
14007       SV1 = CurrentVec;
14008       Mask.push_back(Idx + NumElts);
14009     }
14010 
14011     // Check if all indices in Mask are Undef. In case, propagate Undef.
14012     bool isUndefMask = true;
14013     for (unsigned i = 0; i != NumElts && isUndefMask; ++i)
14014       isUndefMask &= Mask[i] < 0;
14015 
14016     if (isUndefMask)
14017       return DAG.getUNDEF(VT);
14018 
14019     if (!SV0.getNode())
14020       SV0 = DAG.getUNDEF(VT);
14021     if (!SV1.getNode())
14022       SV1 = DAG.getUNDEF(VT);
14023 
14024     // Avoid introducing shuffles with illegal mask.
14025     if (!TLI.isShuffleMaskLegal(Mask, VT)) {
14026       ShuffleVectorSDNode::commuteMask(Mask);
14027 
14028       if (!TLI.isShuffleMaskLegal(Mask, VT))
14029         return SDValue();
14030 
14031       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, A, M2)
14032       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, A, M2)
14033       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, B, M2)
14034       std::swap(SV0, SV1);
14035     }
14036 
14037     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
14038     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
14039     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
14040     return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, Mask);
14041   }
14042 
14043   return SDValue();
14044 }
14045 
14046 SDValue DAGCombiner::visitSCALAR_TO_VECTOR(SDNode *N) {
14047   SDValue InVal = N->getOperand(0);
14048   EVT VT = N->getValueType(0);
14049 
14050   // Replace a SCALAR_TO_VECTOR(EXTRACT_VECTOR_ELT(V,C0)) pattern
14051   // with a VECTOR_SHUFFLE.
14052   if (InVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
14053     SDValue InVec = InVal->getOperand(0);
14054     SDValue EltNo = InVal->getOperand(1);
14055 
14056     // FIXME: We could support implicit truncation if the shuffle can be
14057     // scaled to a smaller vector scalar type.
14058     ConstantSDNode *C0 = dyn_cast<ConstantSDNode>(EltNo);
14059     if (C0 && VT == InVec.getValueType() &&
14060         VT.getScalarType() == InVal.getValueType()) {
14061       SmallVector<int, 8> NewMask(VT.getVectorNumElements(), -1);
14062       int Elt = C0->getZExtValue();
14063       NewMask[0] = Elt;
14064 
14065       if (TLI.isShuffleMaskLegal(NewMask, VT))
14066         return DAG.getVectorShuffle(VT, SDLoc(N), InVec, DAG.getUNDEF(VT),
14067                                     NewMask);
14068     }
14069   }
14070 
14071   return SDValue();
14072 }
14073 
14074 SDValue DAGCombiner::visitINSERT_SUBVECTOR(SDNode *N) {
14075   EVT VT = N->getValueType(0);
14076   SDValue N0 = N->getOperand(0);
14077   SDValue N1 = N->getOperand(1);
14078   SDValue N2 = N->getOperand(2);
14079 
14080   // Combine INSERT_SUBVECTORs where we are inserting to the same index.
14081   // INSERT_SUBVECTOR( INSERT_SUBVECTOR( Vec, SubOld, Idx ), SubNew, Idx )
14082   // --> INSERT_SUBVECTOR( Vec, SubNew, Idx )
14083   if (N0.getOpcode() == ISD::INSERT_SUBVECTOR &&
14084       N0.getOperand(1).getValueType() == N1.getValueType() &&
14085       N0.getOperand(2) == N2)
14086     return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0.getOperand(0),
14087                        N1, N2);
14088 
14089   if (N0.getValueType() != N1.getValueType())
14090     return SDValue();
14091 
14092   // If the input vector is a concatenation, and the insert replaces
14093   // one of the halves, we can optimize into a single concat_vectors.
14094   if (N0.getOpcode() == ISD::CONCAT_VECTORS && N0->getNumOperands() == 2 &&
14095       N2.getOpcode() == ISD::Constant) {
14096     APInt InsIdx = cast<ConstantSDNode>(N2)->getAPIntValue();
14097 
14098     // Lower half: fold (insert_subvector (concat_vectors X, Y), Z) ->
14099     // (concat_vectors Z, Y)
14100     if (InsIdx == 0)
14101       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N1,
14102                          N0.getOperand(1));
14103 
14104     // Upper half: fold (insert_subvector (concat_vectors X, Y), Z) ->
14105     // (concat_vectors X, Z)
14106     if (InsIdx == VT.getVectorNumElements() / 2)
14107       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N0.getOperand(0),
14108                          N1);
14109   }
14110 
14111   return SDValue();
14112 }
14113 
14114 SDValue DAGCombiner::visitFP_TO_FP16(SDNode *N) {
14115   SDValue N0 = N->getOperand(0);
14116 
14117   // fold (fp_to_fp16 (fp16_to_fp op)) -> op
14118   if (N0->getOpcode() == ISD::FP16_TO_FP)
14119     return N0->getOperand(0);
14120 
14121   return SDValue();
14122 }
14123 
14124 SDValue DAGCombiner::visitFP16_TO_FP(SDNode *N) {
14125   SDValue N0 = N->getOperand(0);
14126 
14127   // fold fp16_to_fp(op & 0xffff) -> fp16_to_fp(op)
14128   if (N0->getOpcode() == ISD::AND) {
14129     ConstantSDNode *AndConst = getAsNonOpaqueConstant(N0.getOperand(1));
14130     if (AndConst && AndConst->getAPIntValue() == 0xffff) {
14131       return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), N->getValueType(0),
14132                          N0.getOperand(0));
14133     }
14134   }
14135 
14136   return SDValue();
14137 }
14138 
14139 /// Returns a vector_shuffle if it able to transform an AND to a vector_shuffle
14140 /// with the destination vector and a zero vector.
14141 /// e.g. AND V, <0xffffffff, 0, 0xffffffff, 0>. ==>
14142 ///      vector_shuffle V, Zero, <0, 4, 2, 4>
14143 SDValue DAGCombiner::XformToShuffleWithZero(SDNode *N) {
14144   EVT VT = N->getValueType(0);
14145   SDValue LHS = N->getOperand(0);
14146   SDValue RHS = N->getOperand(1);
14147   SDLoc DL(N);
14148 
14149   // Make sure we're not running after operation legalization where it
14150   // may have custom lowered the vector shuffles.
14151   if (LegalOperations)
14152     return SDValue();
14153 
14154   if (N->getOpcode() != ISD::AND)
14155     return SDValue();
14156 
14157   if (RHS.getOpcode() == ISD::BITCAST)
14158     RHS = RHS.getOperand(0);
14159 
14160   if (RHS.getOpcode() != ISD::BUILD_VECTOR)
14161     return SDValue();
14162 
14163   EVT RVT = RHS.getValueType();
14164   unsigned NumElts = RHS.getNumOperands();
14165 
14166   // Attempt to create a valid clear mask, splitting the mask into
14167   // sub elements and checking to see if each is
14168   // all zeros or all ones - suitable for shuffle masking.
14169   auto BuildClearMask = [&](int Split) {
14170     int NumSubElts = NumElts * Split;
14171     int NumSubBits = RVT.getScalarSizeInBits() / Split;
14172 
14173     SmallVector<int, 8> Indices;
14174     for (int i = 0; i != NumSubElts; ++i) {
14175       int EltIdx = i / Split;
14176       int SubIdx = i % Split;
14177       SDValue Elt = RHS.getOperand(EltIdx);
14178       if (Elt.isUndef()) {
14179         Indices.push_back(-1);
14180         continue;
14181       }
14182 
14183       APInt Bits;
14184       if (isa<ConstantSDNode>(Elt))
14185         Bits = cast<ConstantSDNode>(Elt)->getAPIntValue();
14186       else if (isa<ConstantFPSDNode>(Elt))
14187         Bits = cast<ConstantFPSDNode>(Elt)->getValueAPF().bitcastToAPInt();
14188       else
14189         return SDValue();
14190 
14191       // Extract the sub element from the constant bit mask.
14192       if (DAG.getDataLayout().isBigEndian()) {
14193         Bits = Bits.lshr((Split - SubIdx - 1) * NumSubBits);
14194       } else {
14195         Bits = Bits.lshr(SubIdx * NumSubBits);
14196       }
14197 
14198       if (Split > 1)
14199         Bits = Bits.trunc(NumSubBits);
14200 
14201       if (Bits.isAllOnesValue())
14202         Indices.push_back(i);
14203       else if (Bits == 0)
14204         Indices.push_back(i + NumSubElts);
14205       else
14206         return SDValue();
14207     }
14208 
14209     // Let's see if the target supports this vector_shuffle.
14210     EVT ClearSVT = EVT::getIntegerVT(*DAG.getContext(), NumSubBits);
14211     EVT ClearVT = EVT::getVectorVT(*DAG.getContext(), ClearSVT, NumSubElts);
14212     if (!TLI.isVectorClearMaskLegal(Indices, ClearVT))
14213       return SDValue();
14214 
14215     SDValue Zero = DAG.getConstant(0, DL, ClearVT);
14216     return DAG.getBitcast(VT, DAG.getVectorShuffle(ClearVT, DL,
14217                                                    DAG.getBitcast(ClearVT, LHS),
14218                                                    Zero, Indices));
14219   };
14220 
14221   // Determine maximum split level (byte level masking).
14222   int MaxSplit = 1;
14223   if (RVT.getScalarSizeInBits() % 8 == 0)
14224     MaxSplit = RVT.getScalarSizeInBits() / 8;
14225 
14226   for (int Split = 1; Split <= MaxSplit; ++Split)
14227     if (RVT.getScalarSizeInBits() % Split == 0)
14228       if (SDValue S = BuildClearMask(Split))
14229         return S;
14230 
14231   return SDValue();
14232 }
14233 
14234 /// Visit a binary vector operation, like ADD.
14235 SDValue DAGCombiner::SimplifyVBinOp(SDNode *N) {
14236   assert(N->getValueType(0).isVector() &&
14237          "SimplifyVBinOp only works on vectors!");
14238 
14239   SDValue LHS = N->getOperand(0);
14240   SDValue RHS = N->getOperand(1);
14241   SDValue Ops[] = {LHS, RHS};
14242 
14243   // See if we can constant fold the vector operation.
14244   if (SDValue Fold = DAG.FoldConstantVectorArithmetic(
14245           N->getOpcode(), SDLoc(LHS), LHS.getValueType(), Ops, N->getFlags()))
14246     return Fold;
14247 
14248   // Try to convert a constant mask AND into a shuffle clear mask.
14249   if (SDValue Shuffle = XformToShuffleWithZero(N))
14250     return Shuffle;
14251 
14252   // Type legalization might introduce new shuffles in the DAG.
14253   // Fold (VBinOp (shuffle (A, Undef, Mask)), (shuffle (B, Undef, Mask)))
14254   //   -> (shuffle (VBinOp (A, B)), Undef, Mask).
14255   if (LegalTypes && isa<ShuffleVectorSDNode>(LHS) &&
14256       isa<ShuffleVectorSDNode>(RHS) && LHS.hasOneUse() && RHS.hasOneUse() &&
14257       LHS.getOperand(1).isUndef() &&
14258       RHS.getOperand(1).isUndef()) {
14259     ShuffleVectorSDNode *SVN0 = cast<ShuffleVectorSDNode>(LHS);
14260     ShuffleVectorSDNode *SVN1 = cast<ShuffleVectorSDNode>(RHS);
14261 
14262     if (SVN0->getMask().equals(SVN1->getMask())) {
14263       EVT VT = N->getValueType(0);
14264       SDValue UndefVector = LHS.getOperand(1);
14265       SDValue NewBinOp = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
14266                                      LHS.getOperand(0), RHS.getOperand(0),
14267                                      N->getFlags());
14268       AddUsersToWorklist(N);
14269       return DAG.getVectorShuffle(VT, SDLoc(N), NewBinOp, UndefVector,
14270                                   SVN0->getMask());
14271     }
14272   }
14273 
14274   return SDValue();
14275 }
14276 
14277 SDValue DAGCombiner::SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1,
14278                                     SDValue N2) {
14279   assert(N0.getOpcode() ==ISD::SETCC && "First argument must be a SetCC node!");
14280 
14281   SDValue SCC = SimplifySelectCC(DL, N0.getOperand(0), N0.getOperand(1), N1, N2,
14282                                  cast<CondCodeSDNode>(N0.getOperand(2))->get());
14283 
14284   // If we got a simplified select_cc node back from SimplifySelectCC, then
14285   // break it down into a new SETCC node, and a new SELECT node, and then return
14286   // the SELECT node, since we were called with a SELECT node.
14287   if (SCC.getNode()) {
14288     // Check to see if we got a select_cc back (to turn into setcc/select).
14289     // Otherwise, just return whatever node we got back, like fabs.
14290     if (SCC.getOpcode() == ISD::SELECT_CC) {
14291       SDValue SETCC = DAG.getNode(ISD::SETCC, SDLoc(N0),
14292                                   N0.getValueType(),
14293                                   SCC.getOperand(0), SCC.getOperand(1),
14294                                   SCC.getOperand(4));
14295       AddToWorklist(SETCC.getNode());
14296       return DAG.getSelect(SDLoc(SCC), SCC.getValueType(), SETCC,
14297                            SCC.getOperand(2), SCC.getOperand(3));
14298     }
14299 
14300     return SCC;
14301   }
14302   return SDValue();
14303 }
14304 
14305 /// Given a SELECT or a SELECT_CC node, where LHS and RHS are the two values
14306 /// being selected between, see if we can simplify the select.  Callers of this
14307 /// should assume that TheSelect is deleted if this returns true.  As such, they
14308 /// should return the appropriate thing (e.g. the node) back to the top-level of
14309 /// the DAG combiner loop to avoid it being looked at.
14310 bool DAGCombiner::SimplifySelectOps(SDNode *TheSelect, SDValue LHS,
14311                                     SDValue RHS) {
14312 
14313   // fold (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
14314   // The select + setcc is redundant, because fsqrt returns NaN for X < 0.
14315   if (const ConstantFPSDNode *NaN = isConstOrConstSplatFP(LHS)) {
14316     if (NaN->isNaN() && RHS.getOpcode() == ISD::FSQRT) {
14317       // We have: (select (setcc ?, ?, ?), NaN, (fsqrt ?))
14318       SDValue Sqrt = RHS;
14319       ISD::CondCode CC;
14320       SDValue CmpLHS;
14321       const ConstantFPSDNode *Zero = nullptr;
14322 
14323       if (TheSelect->getOpcode() == ISD::SELECT_CC) {
14324         CC = dyn_cast<CondCodeSDNode>(TheSelect->getOperand(4))->get();
14325         CmpLHS = TheSelect->getOperand(0);
14326         Zero = isConstOrConstSplatFP(TheSelect->getOperand(1));
14327       } else {
14328         // SELECT or VSELECT
14329         SDValue Cmp = TheSelect->getOperand(0);
14330         if (Cmp.getOpcode() == ISD::SETCC) {
14331           CC = dyn_cast<CondCodeSDNode>(Cmp.getOperand(2))->get();
14332           CmpLHS = Cmp.getOperand(0);
14333           Zero = isConstOrConstSplatFP(Cmp.getOperand(1));
14334         }
14335       }
14336       if (Zero && Zero->isZero() &&
14337           Sqrt.getOperand(0) == CmpLHS && (CC == ISD::SETOLT ||
14338           CC == ISD::SETULT || CC == ISD::SETLT)) {
14339         // We have: (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
14340         CombineTo(TheSelect, Sqrt);
14341         return true;
14342       }
14343     }
14344   }
14345   // Cannot simplify select with vector condition
14346   if (TheSelect->getOperand(0).getValueType().isVector()) return false;
14347 
14348   // If this is a select from two identical things, try to pull the operation
14349   // through the select.
14350   if (LHS.getOpcode() != RHS.getOpcode() ||
14351       !LHS.hasOneUse() || !RHS.hasOneUse())
14352     return false;
14353 
14354   // If this is a load and the token chain is identical, replace the select
14355   // of two loads with a load through a select of the address to load from.
14356   // This triggers in things like "select bool X, 10.0, 123.0" after the FP
14357   // constants have been dropped into the constant pool.
14358   if (LHS.getOpcode() == ISD::LOAD) {
14359     LoadSDNode *LLD = cast<LoadSDNode>(LHS);
14360     LoadSDNode *RLD = cast<LoadSDNode>(RHS);
14361 
14362     // Token chains must be identical.
14363     if (LHS.getOperand(0) != RHS.getOperand(0) ||
14364         // Do not let this transformation reduce the number of volatile loads.
14365         LLD->isVolatile() || RLD->isVolatile() ||
14366         // FIXME: If either is a pre/post inc/dec load,
14367         // we'd need to split out the address adjustment.
14368         LLD->isIndexed() || RLD->isIndexed() ||
14369         // If this is an EXTLOAD, the VT's must match.
14370         LLD->getMemoryVT() != RLD->getMemoryVT() ||
14371         // If this is an EXTLOAD, the kind of extension must match.
14372         (LLD->getExtensionType() != RLD->getExtensionType() &&
14373          // The only exception is if one of the extensions is anyext.
14374          LLD->getExtensionType() != ISD::EXTLOAD &&
14375          RLD->getExtensionType() != ISD::EXTLOAD) ||
14376         // FIXME: this discards src value information.  This is
14377         // over-conservative. It would be beneficial to be able to remember
14378         // both potential memory locations.  Since we are discarding
14379         // src value info, don't do the transformation if the memory
14380         // locations are not in the default address space.
14381         LLD->getPointerInfo().getAddrSpace() != 0 ||
14382         RLD->getPointerInfo().getAddrSpace() != 0 ||
14383         !TLI.isOperationLegalOrCustom(TheSelect->getOpcode(),
14384                                       LLD->getBasePtr().getValueType()))
14385       return false;
14386 
14387     // Check that the select condition doesn't reach either load.  If so,
14388     // folding this will induce a cycle into the DAG.  If not, this is safe to
14389     // xform, so create a select of the addresses.
14390     SDValue Addr;
14391     if (TheSelect->getOpcode() == ISD::SELECT) {
14392       SDNode *CondNode = TheSelect->getOperand(0).getNode();
14393       if ((LLD->hasAnyUseOfValue(1) && LLD->isPredecessorOf(CondNode)) ||
14394           (RLD->hasAnyUseOfValue(1) && RLD->isPredecessorOf(CondNode)))
14395         return false;
14396       // The loads must not depend on one another.
14397       if (LLD->isPredecessorOf(RLD) ||
14398           RLD->isPredecessorOf(LLD))
14399         return false;
14400       Addr = DAG.getSelect(SDLoc(TheSelect),
14401                            LLD->getBasePtr().getValueType(),
14402                            TheSelect->getOperand(0), LLD->getBasePtr(),
14403                            RLD->getBasePtr());
14404     } else {  // Otherwise SELECT_CC
14405       SDNode *CondLHS = TheSelect->getOperand(0).getNode();
14406       SDNode *CondRHS = TheSelect->getOperand(1).getNode();
14407 
14408       if ((LLD->hasAnyUseOfValue(1) &&
14409            (LLD->isPredecessorOf(CondLHS) || LLD->isPredecessorOf(CondRHS))) ||
14410           (RLD->hasAnyUseOfValue(1) &&
14411            (RLD->isPredecessorOf(CondLHS) || RLD->isPredecessorOf(CondRHS))))
14412         return false;
14413 
14414       Addr = DAG.getNode(ISD::SELECT_CC, SDLoc(TheSelect),
14415                          LLD->getBasePtr().getValueType(),
14416                          TheSelect->getOperand(0),
14417                          TheSelect->getOperand(1),
14418                          LLD->getBasePtr(), RLD->getBasePtr(),
14419                          TheSelect->getOperand(4));
14420     }
14421 
14422     SDValue Load;
14423     // It is safe to replace the two loads if they have different alignments,
14424     // but the new load must be the minimum (most restrictive) alignment of the
14425     // inputs.
14426     unsigned Alignment = std::min(LLD->getAlignment(), RLD->getAlignment());
14427     MachineMemOperand::Flags MMOFlags = LLD->getMemOperand()->getFlags();
14428     if (!RLD->isInvariant())
14429       MMOFlags &= ~MachineMemOperand::MOInvariant;
14430     if (!RLD->isDereferenceable())
14431       MMOFlags &= ~MachineMemOperand::MODereferenceable;
14432     if (LLD->getExtensionType() == ISD::NON_EXTLOAD) {
14433       // FIXME: Discards pointer and AA info.
14434       Load = DAG.getLoad(TheSelect->getValueType(0), SDLoc(TheSelect),
14435                          LLD->getChain(), Addr, MachinePointerInfo(), Alignment,
14436                          MMOFlags);
14437     } else {
14438       // FIXME: Discards pointer and AA info.
14439       Load = DAG.getExtLoad(
14440           LLD->getExtensionType() == ISD::EXTLOAD ? RLD->getExtensionType()
14441                                                   : LLD->getExtensionType(),
14442           SDLoc(TheSelect), TheSelect->getValueType(0), LLD->getChain(), Addr,
14443           MachinePointerInfo(), LLD->getMemoryVT(), Alignment, MMOFlags);
14444     }
14445 
14446     // Users of the select now use the result of the load.
14447     CombineTo(TheSelect, Load);
14448 
14449     // Users of the old loads now use the new load's chain.  We know the
14450     // old-load value is dead now.
14451     CombineTo(LHS.getNode(), Load.getValue(0), Load.getValue(1));
14452     CombineTo(RHS.getNode(), Load.getValue(0), Load.getValue(1));
14453     return true;
14454   }
14455 
14456   return false;
14457 }
14458 
14459 /// Simplify an expression of the form (N0 cond N1) ? N2 : N3
14460 /// where 'cond' is the comparison specified by CC.
14461 SDValue DAGCombiner::SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
14462                                       SDValue N2, SDValue N3, ISD::CondCode CC,
14463                                       bool NotExtCompare) {
14464   // (x ? y : y) -> y.
14465   if (N2 == N3) return N2;
14466 
14467   EVT VT = N2.getValueType();
14468   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1.getNode());
14469   ConstantSDNode *N2C = dyn_cast<ConstantSDNode>(N2.getNode());
14470 
14471   // Determine if the condition we're dealing with is constant
14472   SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()),
14473                               N0, N1, CC, DL, false);
14474   if (SCC.getNode()) AddToWorklist(SCC.getNode());
14475 
14476   if (ConstantSDNode *SCCC = dyn_cast_or_null<ConstantSDNode>(SCC.getNode())) {
14477     // fold select_cc true, x, y -> x
14478     // fold select_cc false, x, y -> y
14479     return !SCCC->isNullValue() ? N2 : N3;
14480   }
14481 
14482   // Check to see if we can simplify the select into an fabs node
14483   if (ConstantFPSDNode *CFP = dyn_cast<ConstantFPSDNode>(N1)) {
14484     // Allow either -0.0 or 0.0
14485     if (CFP->isZero()) {
14486       // select (setg[te] X, +/-0.0), X, fneg(X) -> fabs
14487       if ((CC == ISD::SETGE || CC == ISD::SETGT) &&
14488           N0 == N2 && N3.getOpcode() == ISD::FNEG &&
14489           N2 == N3.getOperand(0))
14490         return DAG.getNode(ISD::FABS, DL, VT, N0);
14491 
14492       // select (setl[te] X, +/-0.0), fneg(X), X -> fabs
14493       if ((CC == ISD::SETLT || CC == ISD::SETLE) &&
14494           N0 == N3 && N2.getOpcode() == ISD::FNEG &&
14495           N2.getOperand(0) == N3)
14496         return DAG.getNode(ISD::FABS, DL, VT, N3);
14497     }
14498   }
14499 
14500   // Turn "(a cond b) ? 1.0f : 2.0f" into "load (tmp + ((a cond b) ? 0 : 4)"
14501   // where "tmp" is a constant pool entry containing an array with 1.0 and 2.0
14502   // in it.  This is a win when the constant is not otherwise available because
14503   // it replaces two constant pool loads with one.  We only do this if the FP
14504   // type is known to be legal, because if it isn't, then we are before legalize
14505   // types an we want the other legalization to happen first (e.g. to avoid
14506   // messing with soft float) and if the ConstantFP is not legal, because if
14507   // it is legal, we may not need to store the FP constant in a constant pool.
14508   if (ConstantFPSDNode *TV = dyn_cast<ConstantFPSDNode>(N2))
14509     if (ConstantFPSDNode *FV = dyn_cast<ConstantFPSDNode>(N3)) {
14510       if (TLI.isTypeLegal(N2.getValueType()) &&
14511           (TLI.getOperationAction(ISD::ConstantFP, N2.getValueType()) !=
14512                TargetLowering::Legal &&
14513            !TLI.isFPImmLegal(TV->getValueAPF(), TV->getValueType(0)) &&
14514            !TLI.isFPImmLegal(FV->getValueAPF(), FV->getValueType(0))) &&
14515           // If both constants have multiple uses, then we won't need to do an
14516           // extra load, they are likely around in registers for other users.
14517           (TV->hasOneUse() || FV->hasOneUse())) {
14518         Constant *Elts[] = {
14519           const_cast<ConstantFP*>(FV->getConstantFPValue()),
14520           const_cast<ConstantFP*>(TV->getConstantFPValue())
14521         };
14522         Type *FPTy = Elts[0]->getType();
14523         const DataLayout &TD = DAG.getDataLayout();
14524 
14525         // Create a ConstantArray of the two constants.
14526         Constant *CA = ConstantArray::get(ArrayType::get(FPTy, 2), Elts);
14527         SDValue CPIdx =
14528             DAG.getConstantPool(CA, TLI.getPointerTy(DAG.getDataLayout()),
14529                                 TD.getPrefTypeAlignment(FPTy));
14530         unsigned Alignment = cast<ConstantPoolSDNode>(CPIdx)->getAlignment();
14531 
14532         // Get the offsets to the 0 and 1 element of the array so that we can
14533         // select between them.
14534         SDValue Zero = DAG.getIntPtrConstant(0, DL);
14535         unsigned EltSize = (unsigned)TD.getTypeAllocSize(Elts[0]->getType());
14536         SDValue One = DAG.getIntPtrConstant(EltSize, SDLoc(FV));
14537 
14538         SDValue Cond = DAG.getSetCC(DL,
14539                                     getSetCCResultType(N0.getValueType()),
14540                                     N0, N1, CC);
14541         AddToWorklist(Cond.getNode());
14542         SDValue CstOffset = DAG.getSelect(DL, Zero.getValueType(),
14543                                           Cond, One, Zero);
14544         AddToWorklist(CstOffset.getNode());
14545         CPIdx = DAG.getNode(ISD::ADD, DL, CPIdx.getValueType(), CPIdx,
14546                             CstOffset);
14547         AddToWorklist(CPIdx.getNode());
14548         return DAG.getLoad(
14549             TV->getValueType(0), DL, DAG.getEntryNode(), CPIdx,
14550             MachinePointerInfo::getConstantPool(DAG.getMachineFunction()),
14551             Alignment);
14552       }
14553     }
14554 
14555   // Check to see if we can perform the "gzip trick", transforming
14556   // (select_cc setlt X, 0, A, 0) -> (and (sra X, (sub size(X), 1), A)
14557   if (isNullConstant(N3) && CC == ISD::SETLT &&
14558       (isNullConstant(N1) ||                 // (a < 0) ? b : 0
14559        (isOneConstant(N1) && N0 == N2))) {   // (a < 1) ? a : 0
14560     EVT XType = N0.getValueType();
14561     EVT AType = N2.getValueType();
14562     if (XType.bitsGE(AType)) {
14563       // and (sra X, size(X)-1, A) -> "and (srl X, C2), A" iff A is a
14564       // single-bit constant.
14565       if (N2C && ((N2C->getAPIntValue() & (N2C->getAPIntValue() - 1)) == 0)) {
14566         unsigned ShCtV = N2C->getAPIntValue().logBase2();
14567         ShCtV = XType.getSizeInBits() - ShCtV - 1;
14568         SDValue ShCt = DAG.getConstant(ShCtV, SDLoc(N0),
14569                                        getShiftAmountTy(N0.getValueType()));
14570         SDValue Shift = DAG.getNode(ISD::SRL, SDLoc(N0),
14571                                     XType, N0, ShCt);
14572         AddToWorklist(Shift.getNode());
14573 
14574         if (XType.bitsGT(AType)) {
14575           Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
14576           AddToWorklist(Shift.getNode());
14577         }
14578 
14579         return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
14580       }
14581 
14582       SDValue Shift = DAG.getNode(ISD::SRA, SDLoc(N0),
14583                                   XType, N0,
14584                                   DAG.getConstant(XType.getSizeInBits() - 1,
14585                                                   SDLoc(N0),
14586                                          getShiftAmountTy(N0.getValueType())));
14587       AddToWorklist(Shift.getNode());
14588 
14589       if (XType.bitsGT(AType)) {
14590         Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
14591         AddToWorklist(Shift.getNode());
14592       }
14593 
14594       return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
14595     }
14596   }
14597 
14598   // fold (select_cc seteq (and x, y), 0, 0, A) -> (and (shr (shl x)) A)
14599   // where y is has a single bit set.
14600   // A plaintext description would be, we can turn the SELECT_CC into an AND
14601   // when the condition can be materialized as an all-ones register.  Any
14602   // single bit-test can be materialized as an all-ones register with
14603   // shift-left and shift-right-arith.
14604   if (CC == ISD::SETEQ && N0->getOpcode() == ISD::AND &&
14605       N0->getValueType(0) == VT && isNullConstant(N1) && isNullConstant(N2)) {
14606     SDValue AndLHS = N0->getOperand(0);
14607     ConstantSDNode *ConstAndRHS = dyn_cast<ConstantSDNode>(N0->getOperand(1));
14608     if (ConstAndRHS && ConstAndRHS->getAPIntValue().countPopulation() == 1) {
14609       // Shift the tested bit over the sign bit.
14610       const APInt &AndMask = ConstAndRHS->getAPIntValue();
14611       SDValue ShlAmt =
14612         DAG.getConstant(AndMask.countLeadingZeros(), SDLoc(AndLHS),
14613                         getShiftAmountTy(AndLHS.getValueType()));
14614       SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N0), VT, AndLHS, ShlAmt);
14615 
14616       // Now arithmetic right shift it all the way over, so the result is either
14617       // all-ones, or zero.
14618       SDValue ShrAmt =
14619         DAG.getConstant(AndMask.getBitWidth() - 1, SDLoc(Shl),
14620                         getShiftAmountTy(Shl.getValueType()));
14621       SDValue Shr = DAG.getNode(ISD::SRA, SDLoc(N0), VT, Shl, ShrAmt);
14622 
14623       return DAG.getNode(ISD::AND, DL, VT, Shr, N3);
14624     }
14625   }
14626 
14627   // fold select C, 16, 0 -> shl C, 4
14628   if (N2C && isNullConstant(N3) && N2C->getAPIntValue().isPowerOf2() &&
14629       TLI.getBooleanContents(N0.getValueType()) ==
14630           TargetLowering::ZeroOrOneBooleanContent) {
14631 
14632     // If the caller doesn't want us to simplify this into a zext of a compare,
14633     // don't do it.
14634     if (NotExtCompare && N2C->isOne())
14635       return SDValue();
14636 
14637     // Get a SetCC of the condition
14638     // NOTE: Don't create a SETCC if it's not legal on this target.
14639     if (!LegalOperations ||
14640         TLI.isOperationLegal(ISD::SETCC, N0.getValueType())) {
14641       SDValue Temp, SCC;
14642       // cast from setcc result type to select result type
14643       if (LegalTypes) {
14644         SCC  = DAG.getSetCC(DL, getSetCCResultType(N0.getValueType()),
14645                             N0, N1, CC);
14646         if (N2.getValueType().bitsLT(SCC.getValueType()))
14647           Temp = DAG.getZeroExtendInReg(SCC, SDLoc(N2),
14648                                         N2.getValueType());
14649         else
14650           Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2),
14651                              N2.getValueType(), SCC);
14652       } else {
14653         SCC  = DAG.getSetCC(SDLoc(N0), MVT::i1, N0, N1, CC);
14654         Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2),
14655                            N2.getValueType(), SCC);
14656       }
14657 
14658       AddToWorklist(SCC.getNode());
14659       AddToWorklist(Temp.getNode());
14660 
14661       if (N2C->isOne())
14662         return Temp;
14663 
14664       // shl setcc result by log2 n2c
14665       return DAG.getNode(
14666           ISD::SHL, DL, N2.getValueType(), Temp,
14667           DAG.getConstant(N2C->getAPIntValue().logBase2(), SDLoc(Temp),
14668                           getShiftAmountTy(Temp.getValueType())));
14669     }
14670   }
14671 
14672   // Check to see if this is an integer abs.
14673   // select_cc setg[te] X,  0,  X, -X ->
14674   // select_cc setgt    X, -1,  X, -X ->
14675   // select_cc setl[te] X,  0, -X,  X ->
14676   // select_cc setlt    X,  1, -X,  X ->
14677   // Y = sra (X, size(X)-1); xor (add (X, Y), Y)
14678   if (N1C) {
14679     ConstantSDNode *SubC = nullptr;
14680     if (((N1C->isNullValue() && (CC == ISD::SETGT || CC == ISD::SETGE)) ||
14681          (N1C->isAllOnesValue() && CC == ISD::SETGT)) &&
14682         N0 == N2 && N3.getOpcode() == ISD::SUB && N0 == N3.getOperand(1))
14683       SubC = dyn_cast<ConstantSDNode>(N3.getOperand(0));
14684     else if (((N1C->isNullValue() && (CC == ISD::SETLT || CC == ISD::SETLE)) ||
14685               (N1C->isOne() && CC == ISD::SETLT)) &&
14686              N0 == N3 && N2.getOpcode() == ISD::SUB && N0 == N2.getOperand(1))
14687       SubC = dyn_cast<ConstantSDNode>(N2.getOperand(0));
14688 
14689     EVT XType = N0.getValueType();
14690     if (SubC && SubC->isNullValue() && XType.isInteger()) {
14691       SDLoc DL(N0);
14692       SDValue Shift = DAG.getNode(ISD::SRA, DL, XType,
14693                                   N0,
14694                                   DAG.getConstant(XType.getSizeInBits() - 1, DL,
14695                                          getShiftAmountTy(N0.getValueType())));
14696       SDValue Add = DAG.getNode(ISD::ADD, DL,
14697                                 XType, N0, Shift);
14698       AddToWorklist(Shift.getNode());
14699       AddToWorklist(Add.getNode());
14700       return DAG.getNode(ISD::XOR, DL, XType, Add, Shift);
14701     }
14702   }
14703 
14704   // select_cc seteq X, 0, sizeof(X), ctlz(X) -> ctlz(X)
14705   // select_cc seteq X, 0, sizeof(X), ctlz_zero_undef(X) -> ctlz(X)
14706   // select_cc seteq X, 0, sizeof(X), cttz(X) -> cttz(X)
14707   // select_cc seteq X, 0, sizeof(X), cttz_zero_undef(X) -> cttz(X)
14708   // select_cc setne X, 0, ctlz(X), sizeof(X) -> ctlz(X)
14709   // select_cc setne X, 0, ctlz_zero_undef(X), sizeof(X) -> ctlz(X)
14710   // select_cc setne X, 0, cttz(X), sizeof(X) -> cttz(X)
14711   // select_cc setne X, 0, cttz_zero_undef(X), sizeof(X) -> cttz(X)
14712   if (N1C && N1C->isNullValue() && (CC == ISD::SETEQ || CC == ISD::SETNE)) {
14713     SDValue ValueOnZero = N2;
14714     SDValue Count = N3;
14715     // If the condition is NE instead of E, swap the operands.
14716     if (CC == ISD::SETNE)
14717       std::swap(ValueOnZero, Count);
14718     // Check if the value on zero is a constant equal to the bits in the type.
14719     if (auto *ValueOnZeroC = dyn_cast<ConstantSDNode>(ValueOnZero)) {
14720       if (ValueOnZeroC->getAPIntValue() == VT.getSizeInBits()) {
14721         // If the other operand is cttz/cttz_zero_undef of N0, and cttz is
14722         // legal, combine to just cttz.
14723         if ((Count.getOpcode() == ISD::CTTZ ||
14724              Count.getOpcode() == ISD::CTTZ_ZERO_UNDEF) &&
14725             N0 == Count.getOperand(0) &&
14726             (!LegalOperations || TLI.isOperationLegal(ISD::CTTZ, VT)))
14727           return DAG.getNode(ISD::CTTZ, DL, VT, N0);
14728         // If the other operand is ctlz/ctlz_zero_undef of N0, and ctlz is
14729         // legal, combine to just ctlz.
14730         if ((Count.getOpcode() == ISD::CTLZ ||
14731              Count.getOpcode() == ISD::CTLZ_ZERO_UNDEF) &&
14732             N0 == Count.getOperand(0) &&
14733             (!LegalOperations || TLI.isOperationLegal(ISD::CTLZ, VT)))
14734           return DAG.getNode(ISD::CTLZ, DL, VT, N0);
14735       }
14736     }
14737   }
14738 
14739   return SDValue();
14740 }
14741 
14742 /// This is a stub for TargetLowering::SimplifySetCC.
14743 SDValue DAGCombiner::SimplifySetCC(EVT VT, SDValue N0, SDValue N1,
14744                                    ISD::CondCode Cond, const SDLoc &DL,
14745                                    bool foldBooleans) {
14746   TargetLowering::DAGCombinerInfo
14747     DagCombineInfo(DAG, Level, false, this);
14748   return TLI.SimplifySetCC(VT, N0, N1, Cond, foldBooleans, DagCombineInfo, DL);
14749 }
14750 
14751 /// Given an ISD::SDIV node expressing a divide by constant, return
14752 /// a DAG expression to select that will generate the same value by multiplying
14753 /// by a magic number.
14754 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
14755 SDValue DAGCombiner::BuildSDIV(SDNode *N) {
14756   // when optimising for minimum size, we don't want to expand a div to a mul
14757   // and a shift.
14758   if (DAG.getMachineFunction().getFunction()->optForMinSize())
14759     return SDValue();
14760 
14761   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
14762   if (!C)
14763     return SDValue();
14764 
14765   // Avoid division by zero.
14766   if (C->isNullValue())
14767     return SDValue();
14768 
14769   std::vector<SDNode*> Built;
14770   SDValue S =
14771       TLI.BuildSDIV(N, C->getAPIntValue(), DAG, LegalOperations, &Built);
14772 
14773   for (SDNode *N : Built)
14774     AddToWorklist(N);
14775   return S;
14776 }
14777 
14778 /// Given an ISD::SDIV node expressing a divide by constant power of 2, return a
14779 /// DAG expression that will generate the same value by right shifting.
14780 SDValue DAGCombiner::BuildSDIVPow2(SDNode *N) {
14781   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
14782   if (!C)
14783     return SDValue();
14784 
14785   // Avoid division by zero.
14786   if (C->isNullValue())
14787     return SDValue();
14788 
14789   std::vector<SDNode *> Built;
14790   SDValue S = TLI.BuildSDIVPow2(N, C->getAPIntValue(), DAG, &Built);
14791 
14792   for (SDNode *N : Built)
14793     AddToWorklist(N);
14794   return S;
14795 }
14796 
14797 /// Given an ISD::UDIV node expressing a divide by constant, return a DAG
14798 /// expression that will generate the same value by multiplying by a magic
14799 /// number.
14800 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
14801 SDValue DAGCombiner::BuildUDIV(SDNode *N) {
14802   // when optimising for minimum size, we don't want to expand a div to a mul
14803   // and a shift.
14804   if (DAG.getMachineFunction().getFunction()->optForMinSize())
14805     return SDValue();
14806 
14807   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
14808   if (!C)
14809     return SDValue();
14810 
14811   // Avoid division by zero.
14812   if (C->isNullValue())
14813     return SDValue();
14814 
14815   std::vector<SDNode*> Built;
14816   SDValue S =
14817       TLI.BuildUDIV(N, C->getAPIntValue(), DAG, LegalOperations, &Built);
14818 
14819   for (SDNode *N : Built)
14820     AddToWorklist(N);
14821   return S;
14822 }
14823 
14824 SDValue DAGCombiner::BuildReciprocalEstimate(SDValue Op, SDNodeFlags *Flags) {
14825   if (Level >= AfterLegalizeDAG)
14826     return SDValue();
14827 
14828   // Expose the DAG combiner to the target combiner implementations.
14829   TargetLowering::DAGCombinerInfo DCI(DAG, Level, false, this);
14830 
14831   unsigned Iterations = 0;
14832   if (SDValue Est = TLI.getRecipEstimate(Op, DCI, Iterations)) {
14833     if (Iterations) {
14834       // Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
14835       // For the reciprocal, we need to find the zero of the function:
14836       //   F(X) = A X - 1 [which has a zero at X = 1/A]
14837       //     =>
14838       //   X_{i+1} = X_i (2 - A X_i) = X_i + X_i (1 - A X_i) [this second form
14839       //     does not require additional intermediate precision]
14840       EVT VT = Op.getValueType();
14841       SDLoc DL(Op);
14842       SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
14843 
14844       AddToWorklist(Est.getNode());
14845 
14846       // Newton iterations: Est = Est + Est (1 - Arg * Est)
14847       for (unsigned i = 0; i < Iterations; ++i) {
14848         SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Op, Est, Flags);
14849         AddToWorklist(NewEst.getNode());
14850 
14851         NewEst = DAG.getNode(ISD::FSUB, DL, VT, FPOne, NewEst, Flags);
14852         AddToWorklist(NewEst.getNode());
14853 
14854         NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
14855         AddToWorklist(NewEst.getNode());
14856 
14857         Est = DAG.getNode(ISD::FADD, DL, VT, Est, NewEst, Flags);
14858         AddToWorklist(Est.getNode());
14859       }
14860     }
14861     return Est;
14862   }
14863 
14864   return SDValue();
14865 }
14866 
14867 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
14868 /// For the reciprocal sqrt, we need to find the zero of the function:
14869 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
14870 ///     =>
14871 ///   X_{i+1} = X_i (1.5 - A X_i^2 / 2)
14872 /// As a result, we precompute A/2 prior to the iteration loop.
14873 SDValue DAGCombiner::buildSqrtNROneConst(SDValue Arg, SDValue Est,
14874                                          unsigned Iterations,
14875                                          SDNodeFlags *Flags, bool Reciprocal) {
14876   EVT VT = Arg.getValueType();
14877   SDLoc DL(Arg);
14878   SDValue ThreeHalves = DAG.getConstantFP(1.5, DL, VT);
14879 
14880   // We now need 0.5 * Arg which we can write as (1.5 * Arg - Arg) so that
14881   // this entire sequence requires only one FP constant.
14882   SDValue HalfArg = DAG.getNode(ISD::FMUL, DL, VT, ThreeHalves, Arg, Flags);
14883   AddToWorklist(HalfArg.getNode());
14884 
14885   HalfArg = DAG.getNode(ISD::FSUB, DL, VT, HalfArg, Arg, Flags);
14886   AddToWorklist(HalfArg.getNode());
14887 
14888   // Newton iterations: Est = Est * (1.5 - HalfArg * Est * Est)
14889   for (unsigned i = 0; i < Iterations; ++i) {
14890     SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, Est, Flags);
14891     AddToWorklist(NewEst.getNode());
14892 
14893     NewEst = DAG.getNode(ISD::FMUL, DL, VT, HalfArg, NewEst, Flags);
14894     AddToWorklist(NewEst.getNode());
14895 
14896     NewEst = DAG.getNode(ISD::FSUB, DL, VT, ThreeHalves, NewEst, Flags);
14897     AddToWorklist(NewEst.getNode());
14898 
14899     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
14900     AddToWorklist(Est.getNode());
14901   }
14902 
14903   // If non-reciprocal square root is requested, multiply the result by Arg.
14904   if (!Reciprocal) {
14905     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, Arg, Flags);
14906     AddToWorklist(Est.getNode());
14907   }
14908 
14909   return Est;
14910 }
14911 
14912 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
14913 /// For the reciprocal sqrt, we need to find the zero of the function:
14914 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
14915 ///     =>
14916 ///   X_{i+1} = (-0.5 * X_i) * (A * X_i * X_i + (-3.0))
14917 SDValue DAGCombiner::buildSqrtNRTwoConst(SDValue Arg, SDValue Est,
14918                                          unsigned Iterations,
14919                                          SDNodeFlags *Flags, bool Reciprocal) {
14920   EVT VT = Arg.getValueType();
14921   SDLoc DL(Arg);
14922   SDValue MinusThree = DAG.getConstantFP(-3.0, DL, VT);
14923   SDValue MinusHalf = DAG.getConstantFP(-0.5, DL, VT);
14924 
14925   // This routine must enter the loop below to work correctly
14926   // when (Reciprocal == false).
14927   assert(Iterations > 0);
14928 
14929   // Newton iterations for reciprocal square root:
14930   // E = (E * -0.5) * ((A * E) * E + -3.0)
14931   for (unsigned i = 0; i < Iterations; ++i) {
14932     SDValue AE = DAG.getNode(ISD::FMUL, DL, VT, Arg, Est, Flags);
14933     AddToWorklist(AE.getNode());
14934 
14935     SDValue AEE = DAG.getNode(ISD::FMUL, DL, VT, AE, Est, Flags);
14936     AddToWorklist(AEE.getNode());
14937 
14938     SDValue RHS = DAG.getNode(ISD::FADD, DL, VT, AEE, MinusThree, Flags);
14939     AddToWorklist(RHS.getNode());
14940 
14941     // When calculating a square root at the last iteration build:
14942     // S = ((A * E) * -0.5) * ((A * E) * E + -3.0)
14943     // (notice a common subexpression)
14944     SDValue LHS;
14945     if (Reciprocal || (i + 1) < Iterations) {
14946       // RSQRT: LHS = (E * -0.5)
14947       LHS = DAG.getNode(ISD::FMUL, DL, VT, Est, MinusHalf, Flags);
14948     } else {
14949       // SQRT: LHS = (A * E) * -0.5
14950       LHS = DAG.getNode(ISD::FMUL, DL, VT, AE, MinusHalf, Flags);
14951     }
14952     AddToWorklist(LHS.getNode());
14953 
14954     Est = DAG.getNode(ISD::FMUL, DL, VT, LHS, RHS, Flags);
14955     AddToWorklist(Est.getNode());
14956   }
14957 
14958   return Est;
14959 }
14960 
14961 /// Build code to calculate either rsqrt(Op) or sqrt(Op). In the latter case
14962 /// Op*rsqrt(Op) is actually computed, so additional postprocessing is needed if
14963 /// Op can be zero.
14964 SDValue DAGCombiner::buildSqrtEstimateImpl(SDValue Op, SDNodeFlags *Flags,
14965                                            bool Reciprocal) {
14966   if (Level >= AfterLegalizeDAG)
14967     return SDValue();
14968 
14969   // Expose the DAG combiner to the target combiner implementations.
14970   TargetLowering::DAGCombinerInfo DCI(DAG, Level, false, this);
14971   unsigned Iterations = 0;
14972   bool UseOneConstNR = false;
14973   if (SDValue Est = TLI.getRsqrtEstimate(Op, DCI, Iterations, UseOneConstNR)) {
14974     AddToWorklist(Est.getNode());
14975     if (Iterations) {
14976       Est = UseOneConstNR
14977                 ? buildSqrtNROneConst(Op, Est, Iterations, Flags, Reciprocal)
14978                 : buildSqrtNRTwoConst(Op, Est, Iterations, Flags, Reciprocal);
14979     }
14980     return Est;
14981   }
14982 
14983   return SDValue();
14984 }
14985 
14986 SDValue DAGCombiner::buildRsqrtEstimate(SDValue Op, SDNodeFlags *Flags) {
14987   return buildSqrtEstimateImpl(Op, Flags, true);
14988 }
14989 
14990 SDValue DAGCombiner::buildSqrtEstimate(SDValue Op, SDNodeFlags *Flags) {
14991   SDValue Est = buildSqrtEstimateImpl(Op, Flags, false);
14992   if (!Est)
14993     return SDValue();
14994 
14995   // Unfortunately, Est is now NaN if the input was exactly 0.
14996   // Select out this case and force the answer to 0.
14997   EVT VT = Est.getValueType();
14998   SDLoc DL(Op);
14999   SDValue Zero = DAG.getConstantFP(0.0, DL, VT);
15000   EVT CCVT = getSetCCResultType(VT);
15001   SDValue ZeroCmp = DAG.getSetCC(DL, CCVT, Op, Zero, ISD::SETEQ);
15002   AddToWorklist(ZeroCmp.getNode());
15003 
15004   Est = DAG.getNode(VT.isVector() ? ISD::VSELECT : ISD::SELECT, DL, VT, ZeroCmp,
15005                     Zero, Est);
15006   AddToWorklist(Est.getNode());
15007   return Est;
15008 }
15009 
15010 /// Return true if base is a frame index, which is known not to alias with
15011 /// anything but itself.  Provides base object and offset as results.
15012 static bool FindBaseOffset(SDValue Ptr, SDValue &Base, int64_t &Offset,
15013                            const GlobalValue *&GV, const void *&CV) {
15014   // Assume it is a primitive operation.
15015   Base = Ptr; Offset = 0; GV = nullptr; CV = nullptr;
15016 
15017   // If it's an adding a simple constant then integrate the offset.
15018   if (Base.getOpcode() == ISD::ADD) {
15019     if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Base.getOperand(1))) {
15020       Base = Base.getOperand(0);
15021       Offset += C->getZExtValue();
15022     }
15023   }
15024 
15025   // Return the underlying GlobalValue, and update the Offset.  Return false
15026   // for GlobalAddressSDNode since the same GlobalAddress may be represented
15027   // by multiple nodes with different offsets.
15028   if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Base)) {
15029     GV = G->getGlobal();
15030     Offset += G->getOffset();
15031     return false;
15032   }
15033 
15034   // Return the underlying Constant value, and update the Offset.  Return false
15035   // for ConstantSDNodes since the same constant pool entry may be represented
15036   // by multiple nodes with different offsets.
15037   if (ConstantPoolSDNode *C = dyn_cast<ConstantPoolSDNode>(Base)) {
15038     CV = C->isMachineConstantPoolEntry() ? (const void *)C->getMachineCPVal()
15039                                          : (const void *)C->getConstVal();
15040     Offset += C->getOffset();
15041     return false;
15042   }
15043   // If it's any of the following then it can't alias with anything but itself.
15044   return isa<FrameIndexSDNode>(Base);
15045 }
15046 
15047 /// Return true if there is any possibility that the two addresses overlap.
15048 bool DAGCombiner::isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const {
15049   // If they are the same then they must be aliases.
15050   if (Op0->getBasePtr() == Op1->getBasePtr()) return true;
15051 
15052   // If they are both volatile then they cannot be reordered.
15053   if (Op0->isVolatile() && Op1->isVolatile()) return true;
15054 
15055   // If one operation reads from invariant memory, and the other may store, they
15056   // cannot alias. These should really be checking the equivalent of mayWrite,
15057   // but it only matters for memory nodes other than load /store.
15058   if (Op0->isInvariant() && Op1->writeMem())
15059     return false;
15060 
15061   if (Op1->isInvariant() && Op0->writeMem())
15062     return false;
15063 
15064   // Gather base node and offset information.
15065   SDValue Base1, Base2;
15066   int64_t Offset1, Offset2;
15067   const GlobalValue *GV1, *GV2;
15068   const void *CV1, *CV2;
15069   bool isFrameIndex1 = FindBaseOffset(Op0->getBasePtr(),
15070                                       Base1, Offset1, GV1, CV1);
15071   bool isFrameIndex2 = FindBaseOffset(Op1->getBasePtr(),
15072                                       Base2, Offset2, GV2, CV2);
15073 
15074   // If they have a same base address then check to see if they overlap.
15075   if (Base1 == Base2 || (GV1 && (GV1 == GV2)) || (CV1 && (CV1 == CV2)))
15076     return !((Offset1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= Offset2 ||
15077              (Offset2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= Offset1);
15078 
15079   // It is possible for different frame indices to alias each other, mostly
15080   // when tail call optimization reuses return address slots for arguments.
15081   // To catch this case, look up the actual index of frame indices to compute
15082   // the real alias relationship.
15083   if (isFrameIndex1 && isFrameIndex2) {
15084     MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
15085     Offset1 += MFI.getObjectOffset(cast<FrameIndexSDNode>(Base1)->getIndex());
15086     Offset2 += MFI.getObjectOffset(cast<FrameIndexSDNode>(Base2)->getIndex());
15087     return !((Offset1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= Offset2 ||
15088              (Offset2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= Offset1);
15089   }
15090 
15091   // Otherwise, if we know what the bases are, and they aren't identical, then
15092   // we know they cannot alias.
15093   if ((isFrameIndex1 || CV1 || GV1) && (isFrameIndex2 || CV2 || GV2))
15094     return false;
15095 
15096   // If we know required SrcValue1 and SrcValue2 have relatively large alignment
15097   // compared to the size and offset of the access, we may be able to prove they
15098   // do not alias.  This check is conservative for now to catch cases created by
15099   // splitting vector types.
15100   if ((Op0->getOriginalAlignment() == Op1->getOriginalAlignment()) &&
15101       (Op0->getSrcValueOffset() != Op1->getSrcValueOffset()) &&
15102       (Op0->getMemoryVT().getSizeInBits() >> 3 ==
15103        Op1->getMemoryVT().getSizeInBits() >> 3) &&
15104       (Op0->getOriginalAlignment() > (Op0->getMemoryVT().getSizeInBits() >> 3))) {
15105     int64_t OffAlign1 = Op0->getSrcValueOffset() % Op0->getOriginalAlignment();
15106     int64_t OffAlign2 = Op1->getSrcValueOffset() % Op1->getOriginalAlignment();
15107 
15108     // There is no overlap between these relatively aligned accesses of similar
15109     // size, return no alias.
15110     if ((OffAlign1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= OffAlign2 ||
15111         (OffAlign2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= OffAlign1)
15112       return false;
15113   }
15114 
15115   bool UseAA = CombinerGlobalAA.getNumOccurrences() > 0
15116                    ? CombinerGlobalAA
15117                    : DAG.getSubtarget().useAA();
15118 #ifndef NDEBUG
15119   if (CombinerAAOnlyFunc.getNumOccurrences() &&
15120       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
15121     UseAA = false;
15122 #endif
15123   if (UseAA &&
15124       Op0->getMemOperand()->getValue() && Op1->getMemOperand()->getValue()) {
15125     // Use alias analysis information.
15126     int64_t MinOffset = std::min(Op0->getSrcValueOffset(),
15127                                  Op1->getSrcValueOffset());
15128     int64_t Overlap1 = (Op0->getMemoryVT().getSizeInBits() >> 3) +
15129         Op0->getSrcValueOffset() - MinOffset;
15130     int64_t Overlap2 = (Op1->getMemoryVT().getSizeInBits() >> 3) +
15131         Op1->getSrcValueOffset() - MinOffset;
15132     AliasResult AAResult =
15133         AA.alias(MemoryLocation(Op0->getMemOperand()->getValue(), Overlap1,
15134                                 UseTBAA ? Op0->getAAInfo() : AAMDNodes()),
15135                  MemoryLocation(Op1->getMemOperand()->getValue(), Overlap2,
15136                                 UseTBAA ? Op1->getAAInfo() : AAMDNodes()));
15137     if (AAResult == NoAlias)
15138       return false;
15139   }
15140 
15141   // Otherwise we have to assume they alias.
15142   return true;
15143 }
15144 
15145 /// Walk up chain skipping non-aliasing memory nodes,
15146 /// looking for aliasing nodes and adding them to the Aliases vector.
15147 void DAGCombiner::GatherAllAliases(SDNode *N, SDValue OriginalChain,
15148                                    SmallVectorImpl<SDValue> &Aliases) {
15149   SmallVector<SDValue, 8> Chains;     // List of chains to visit.
15150   SmallPtrSet<SDNode *, 16> Visited;  // Visited node set.
15151 
15152   // Get alias information for node.
15153   bool IsLoad = isa<LoadSDNode>(N) && !cast<LSBaseSDNode>(N)->isVolatile();
15154 
15155   // Starting off.
15156   Chains.push_back(OriginalChain);
15157   unsigned Depth = 0;
15158 
15159   // Look at each chain and determine if it is an alias.  If so, add it to the
15160   // aliases list.  If not, then continue up the chain looking for the next
15161   // candidate.
15162   while (!Chains.empty()) {
15163     SDValue Chain = Chains.pop_back_val();
15164 
15165     // For TokenFactor nodes, look at each operand and only continue up the
15166     // chain until we reach the depth limit.
15167     //
15168     // FIXME: The depth check could be made to return the last non-aliasing
15169     // chain we found before we hit a tokenfactor rather than the original
15170     // chain.
15171     if (Depth > TLI.getGatherAllAliasesMaxDepth()) {
15172       Aliases.clear();
15173       Aliases.push_back(OriginalChain);
15174       return;
15175     }
15176 
15177     // Don't bother if we've been before.
15178     if (!Visited.insert(Chain.getNode()).second)
15179       continue;
15180 
15181     switch (Chain.getOpcode()) {
15182     case ISD::EntryToken:
15183       // Entry token is ideal chain operand, but handled in FindBetterChain.
15184       break;
15185 
15186     case ISD::LOAD:
15187     case ISD::STORE: {
15188       // Get alias information for Chain.
15189       bool IsOpLoad = isa<LoadSDNode>(Chain.getNode()) &&
15190           !cast<LSBaseSDNode>(Chain.getNode())->isVolatile();
15191 
15192       // If chain is alias then stop here.
15193       if (!(IsLoad && IsOpLoad) &&
15194           isAlias(cast<LSBaseSDNode>(N), cast<LSBaseSDNode>(Chain.getNode()))) {
15195         Aliases.push_back(Chain);
15196       } else {
15197         // Look further up the chain.
15198         Chains.push_back(Chain.getOperand(0));
15199         ++Depth;
15200       }
15201       break;
15202     }
15203 
15204     case ISD::TokenFactor:
15205       // We have to check each of the operands of the token factor for "small"
15206       // token factors, so we queue them up.  Adding the operands to the queue
15207       // (stack) in reverse order maintains the original order and increases the
15208       // likelihood that getNode will find a matching token factor (CSE.)
15209       if (Chain.getNumOperands() > 16) {
15210         Aliases.push_back(Chain);
15211         break;
15212       }
15213       for (unsigned n = Chain.getNumOperands(); n;)
15214         Chains.push_back(Chain.getOperand(--n));
15215       ++Depth;
15216       break;
15217 
15218     default:
15219       // For all other instructions we will just have to take what we can get.
15220       Aliases.push_back(Chain);
15221       break;
15222     }
15223   }
15224 }
15225 
15226 /// Walk up chain skipping non-aliasing memory nodes, looking for a better chain
15227 /// (aliasing node.)
15228 SDValue DAGCombiner::FindBetterChain(SDNode *N, SDValue OldChain) {
15229   SmallVector<SDValue, 8> Aliases;  // Ops for replacing token factor.
15230 
15231   // Accumulate all the aliases to this node.
15232   GatherAllAliases(N, OldChain, Aliases);
15233 
15234   // If no operands then chain to entry token.
15235   if (Aliases.size() == 0)
15236     return DAG.getEntryNode();
15237 
15238   // If a single operand then chain to it.  We don't need to revisit it.
15239   if (Aliases.size() == 1)
15240     return Aliases[0];
15241 
15242   // Construct a custom tailored token factor.
15243   return DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Aliases);
15244 }
15245 
15246 bool DAGCombiner::findBetterNeighborChains(StoreSDNode *St) {
15247   // This holds the base pointer, index, and the offset in bytes from the base
15248   // pointer.
15249   BaseIndexOffset BasePtr = BaseIndexOffset::match(St->getBasePtr(), DAG);
15250 
15251   // We must have a base and an offset.
15252   if (!BasePtr.Base.getNode())
15253     return false;
15254 
15255   // Do not handle stores to undef base pointers.
15256   if (BasePtr.Base.isUndef())
15257     return false;
15258 
15259   SmallVector<StoreSDNode *, 8> ChainedStores;
15260   ChainedStores.push_back(St);
15261 
15262   // Walk up the chain and look for nodes with offsets from the same
15263   // base pointer. Stop when reaching an instruction with a different kind
15264   // or instruction which has a different base pointer.
15265   StoreSDNode *Index = St;
15266   while (Index) {
15267     // If the chain has more than one use, then we can't reorder the mem ops.
15268     if (Index != St && !SDValue(Index, 0)->hasOneUse())
15269       break;
15270 
15271     if (Index->isVolatile() || Index->isIndexed())
15272       break;
15273 
15274     // Find the base pointer and offset for this memory node.
15275     BaseIndexOffset Ptr = BaseIndexOffset::match(Index->getBasePtr(), DAG);
15276 
15277     // Check that the base pointer is the same as the original one.
15278     if (!Ptr.equalBaseIndex(BasePtr))
15279       break;
15280 
15281     // Find the next memory operand in the chain. If the next operand in the
15282     // chain is a store then move up and continue the scan with the next
15283     // memory operand. If the next operand is a load save it and use alias
15284     // information to check if it interferes with anything.
15285     SDNode *NextInChain = Index->getChain().getNode();
15286     while (true) {
15287       if (StoreSDNode *STn = dyn_cast<StoreSDNode>(NextInChain)) {
15288         // We found a store node. Use it for the next iteration.
15289         if (STn->isVolatile() || STn->isIndexed()) {
15290           Index = nullptr;
15291           break;
15292         }
15293         ChainedStores.push_back(STn);
15294         Index = STn;
15295         break;
15296       } else if (LoadSDNode *Ldn = dyn_cast<LoadSDNode>(NextInChain)) {
15297         NextInChain = Ldn->getChain().getNode();
15298         continue;
15299       } else {
15300         Index = nullptr;
15301         break;
15302       }
15303     }
15304   }
15305 
15306   bool MadeChangeToSt = false;
15307   SmallVector<std::pair<StoreSDNode *, SDValue>, 8> BetterChains;
15308 
15309   for (StoreSDNode *ChainedStore : ChainedStores) {
15310     SDValue Chain = ChainedStore->getChain();
15311     SDValue BetterChain = FindBetterChain(ChainedStore, Chain);
15312 
15313     if (Chain != BetterChain) {
15314       if (ChainedStore == St)
15315         MadeChangeToSt = true;
15316       BetterChains.push_back(std::make_pair(ChainedStore, BetterChain));
15317     }
15318   }
15319 
15320   // Do all replacements after finding the replacements to make to avoid making
15321   // the chains more complicated by introducing new TokenFactors.
15322   for (auto Replacement : BetterChains)
15323     replaceStoreChain(Replacement.first, Replacement.second);
15324 
15325   return MadeChangeToSt;
15326 }
15327 
15328 /// This is the entry point for the file.
15329 void SelectionDAG::Combine(CombineLevel Level, AliasAnalysis &AA,
15330                            CodeGenOpt::Level OptLevel) {
15331   /// This is the main entry point to this class.
15332   DAGCombiner(*this, AA, OptLevel).Run(Level);
15333 }
15334