1 //===-- DAGCombiner.cpp - Implement a DAG node combiner -------------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 // This pass combines dag nodes to form fewer, simpler DAG nodes.  It can be run
11 // both before and after the DAG is legalized.
12 //
13 // This pass is not a substitute for the LLVM IR instcombine pass. This pass is
14 // primarily intended to handle simplification opportunities that are implicit
15 // in the LLVM IR and exposed by the various codegen lowering phases.
16 //
17 //===----------------------------------------------------------------------===//
18 
19 #include "llvm/ADT/SetVector.h"
20 #include "llvm/ADT/SmallBitVector.h"
21 #include "llvm/ADT/SmallPtrSet.h"
22 #include "llvm/ADT/SmallSet.h"
23 #include "llvm/ADT/Statistic.h"
24 #include "llvm/Analysis/AliasAnalysis.h"
25 #include "llvm/CodeGen/MachineFrameInfo.h"
26 #include "llvm/CodeGen/MachineFunction.h"
27 #include "llvm/CodeGen/SelectionDAG.h"
28 #include "llvm/CodeGen/SelectionDAGTargetInfo.h"
29 #include "llvm/IR/DataLayout.h"
30 #include "llvm/IR/DerivedTypes.h"
31 #include "llvm/IR/Function.h"
32 #include "llvm/IR/LLVMContext.h"
33 #include "llvm/Support/CommandLine.h"
34 #include "llvm/Support/Debug.h"
35 #include "llvm/Support/ErrorHandling.h"
36 #include "llvm/Support/MathExtras.h"
37 #include "llvm/Support/raw_ostream.h"
38 #include "llvm/Target/TargetLowering.h"
39 #include "llvm/Target/TargetOptions.h"
40 #include "llvm/Target/TargetRegisterInfo.h"
41 #include "llvm/Target/TargetSubtargetInfo.h"
42 #include <algorithm>
43 using namespace llvm;
44 
45 #define DEBUG_TYPE "dagcombine"
46 
47 STATISTIC(NodesCombined   , "Number of dag nodes combined");
48 STATISTIC(PreIndexedNodes , "Number of pre-indexed nodes created");
49 STATISTIC(PostIndexedNodes, "Number of post-indexed nodes created");
50 STATISTIC(OpsNarrowed     , "Number of load/op/store narrowed");
51 STATISTIC(LdStFP2Int      , "Number of fp load/store pairs transformed to int");
52 STATISTIC(SlicedLoads, "Number of load sliced");
53 
54 namespace {
55   static cl::opt<bool>
56     CombinerAA("combiner-alias-analysis", cl::Hidden,
57                cl::desc("Enable DAG combiner alias-analysis heuristics"));
58 
59   static cl::opt<bool>
60     CombinerGlobalAA("combiner-global-alias-analysis", cl::Hidden,
61                cl::desc("Enable DAG combiner's use of IR alias analysis"));
62 
63   static cl::opt<bool>
64     UseTBAA("combiner-use-tbaa", cl::Hidden, cl::init(true),
65                cl::desc("Enable DAG combiner's use of TBAA"));
66 
67 #ifndef NDEBUG
68   static cl::opt<std::string>
69     CombinerAAOnlyFunc("combiner-aa-only-func", cl::Hidden,
70                cl::desc("Only use DAG-combiner alias analysis in this"
71                         " function"));
72 #endif
73 
74   /// Hidden option to stress test load slicing, i.e., when this option
75   /// is enabled, load slicing bypasses most of its profitability guards.
76   static cl::opt<bool>
77   StressLoadSlicing("combiner-stress-load-slicing", cl::Hidden,
78                     cl::desc("Bypass the profitability model of load "
79                              "slicing"),
80                     cl::init(false));
81 
82   static cl::opt<bool>
83     MaySplitLoadIndex("combiner-split-load-index", cl::Hidden, cl::init(true),
84                       cl::desc("DAG combiner may split indexing from loads"));
85 
86 //------------------------------ DAGCombiner ---------------------------------//
87 
88   class DAGCombiner {
89     SelectionDAG &DAG;
90     const TargetLowering &TLI;
91     CombineLevel Level;
92     CodeGenOpt::Level OptLevel;
93     bool LegalOperations;
94     bool LegalTypes;
95     bool ForCodeSize;
96 
97     /// \brief Worklist of all of the nodes that need to be simplified.
98     ///
99     /// This must behave as a stack -- new nodes to process are pushed onto the
100     /// back and when processing we pop off of the back.
101     ///
102     /// The worklist will not contain duplicates but may contain null entries
103     /// due to nodes being deleted from the underlying DAG.
104     SmallVector<SDNode *, 64> Worklist;
105 
106     /// \brief Mapping from an SDNode to its position on the worklist.
107     ///
108     /// This is used to find and remove nodes from the worklist (by nulling
109     /// them) when they are deleted from the underlying DAG. It relies on
110     /// stable indices of nodes within the worklist.
111     DenseMap<SDNode *, unsigned> WorklistMap;
112 
113     /// \brief Set of nodes which have been combined (at least once).
114     ///
115     /// This is used to allow us to reliably add any operands of a DAG node
116     /// which have not yet been combined to the worklist.
117     SmallPtrSet<SDNode *, 32> CombinedNodes;
118 
119     // AA - Used for DAG load/store alias analysis.
120     AliasAnalysis &AA;
121 
122     /// When an instruction is simplified, add all users of the instruction to
123     /// the work lists because they might get more simplified now.
124     void AddUsersToWorklist(SDNode *N) {
125       for (SDNode *Node : N->uses())
126         AddToWorklist(Node);
127     }
128 
129     /// Call the node-specific routine that folds each particular type of node.
130     SDValue visit(SDNode *N);
131 
132   public:
133     /// Add to the worklist making sure its instance is at the back (next to be
134     /// processed.)
135     void AddToWorklist(SDNode *N) {
136       // Skip handle nodes as they can't usefully be combined and confuse the
137       // zero-use deletion strategy.
138       if (N->getOpcode() == ISD::HANDLENODE)
139         return;
140 
141       if (WorklistMap.insert(std::make_pair(N, Worklist.size())).second)
142         Worklist.push_back(N);
143     }
144 
145     /// Remove all instances of N from the worklist.
146     void removeFromWorklist(SDNode *N) {
147       CombinedNodes.erase(N);
148 
149       auto It = WorklistMap.find(N);
150       if (It == WorklistMap.end())
151         return; // Not in the worklist.
152 
153       // Null out the entry rather than erasing it to avoid a linear operation.
154       Worklist[It->second] = nullptr;
155       WorklistMap.erase(It);
156     }
157 
158     void deleteAndRecombine(SDNode *N);
159     bool recursivelyDeleteUnusedNodes(SDNode *N);
160 
161     /// Replaces all uses of the results of one DAG node with new values.
162     SDValue CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
163                       bool AddTo = true);
164 
165     /// Replaces all uses of the results of one DAG node with new values.
166     SDValue CombineTo(SDNode *N, SDValue Res, bool AddTo = true) {
167       return CombineTo(N, &Res, 1, AddTo);
168     }
169 
170     /// Replaces all uses of the results of one DAG node with new values.
171     SDValue CombineTo(SDNode *N, SDValue Res0, SDValue Res1,
172                       bool AddTo = true) {
173       SDValue To[] = { Res0, Res1 };
174       return CombineTo(N, To, 2, AddTo);
175     }
176 
177     void CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO);
178 
179   private:
180 
181     /// Check the specified integer node value to see if it can be simplified or
182     /// if things it uses can be simplified by bit propagation.
183     /// If so, return true.
184     bool SimplifyDemandedBits(SDValue Op) {
185       unsigned BitWidth = Op.getScalarValueSizeInBits();
186       APInt Demanded = APInt::getAllOnesValue(BitWidth);
187       return SimplifyDemandedBits(Op, Demanded);
188     }
189 
190     bool SimplifyDemandedBits(SDValue Op, const APInt &Demanded);
191 
192     bool CombineToPreIndexedLoadStore(SDNode *N);
193     bool CombineToPostIndexedLoadStore(SDNode *N);
194     SDValue SplitIndexingFromLoad(LoadSDNode *LD);
195     bool SliceUpLoad(SDNode *N);
196 
197     /// \brief Replace an ISD::EXTRACT_VECTOR_ELT of a load with a narrowed
198     ///   load.
199     ///
200     /// \param EVE ISD::EXTRACT_VECTOR_ELT to be replaced.
201     /// \param InVecVT type of the input vector to EVE with bitcasts resolved.
202     /// \param EltNo index of the vector element to load.
203     /// \param OriginalLoad load that EVE came from to be replaced.
204     /// \returns EVE on success SDValue() on failure.
205     SDValue ReplaceExtractVectorEltOfLoadWithNarrowedLoad(
206         SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad);
207     void ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad);
208     SDValue PromoteOperand(SDValue Op, EVT PVT, bool &Replace);
209     SDValue SExtPromoteOperand(SDValue Op, EVT PVT);
210     SDValue ZExtPromoteOperand(SDValue Op, EVT PVT);
211     SDValue PromoteIntBinOp(SDValue Op);
212     SDValue PromoteIntShiftOp(SDValue Op);
213     SDValue PromoteExtend(SDValue Op);
214     bool PromoteLoad(SDValue Op);
215 
216     void ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs, SDValue Trunc,
217                          SDValue ExtLoad, const SDLoc &DL,
218                          ISD::NodeType ExtType);
219 
220     /// Call the node-specific routine that knows how to fold each
221     /// particular type of node. If that doesn't do anything, try the
222     /// target-specific DAG combines.
223     SDValue combine(SDNode *N);
224 
225     // Visitation implementation - Implement dag node combining for different
226     // node types.  The semantics are as follows:
227     // Return Value:
228     //   SDValue.getNode() == 0 - No change was made
229     //   SDValue.getNode() == N - N was replaced, is dead and has been handled.
230     //   otherwise              - N should be replaced by the returned Operand.
231     //
232     SDValue visitTokenFactor(SDNode *N);
233     SDValue visitMERGE_VALUES(SDNode *N);
234     SDValue visitADD(SDNode *N);
235     SDValue visitADDLike(SDValue N0, SDValue N1, SDNode *LocReference);
236     SDValue visitSUB(SDNode *N);
237     SDValue visitADDC(SDNode *N);
238     SDValue visitSUBC(SDNode *N);
239     SDValue visitADDE(SDNode *N);
240     SDValue visitSUBE(SDNode *N);
241     SDValue visitMUL(SDNode *N);
242     SDValue useDivRem(SDNode *N);
243     SDValue visitSDIV(SDNode *N);
244     SDValue visitUDIV(SDNode *N);
245     SDValue visitREM(SDNode *N);
246     SDValue visitMULHU(SDNode *N);
247     SDValue visitMULHS(SDNode *N);
248     SDValue visitSMUL_LOHI(SDNode *N);
249     SDValue visitUMUL_LOHI(SDNode *N);
250     SDValue visitSMULO(SDNode *N);
251     SDValue visitUMULO(SDNode *N);
252     SDValue visitIMINMAX(SDNode *N);
253     SDValue visitAND(SDNode *N);
254     SDValue visitANDLike(SDValue N0, SDValue N1, SDNode *LocReference);
255     SDValue visitOR(SDNode *N);
256     SDValue visitORLike(SDValue N0, SDValue N1, SDNode *LocReference);
257     SDValue visitXOR(SDNode *N);
258     SDValue SimplifyVBinOp(SDNode *N);
259     SDValue visitSHL(SDNode *N);
260     SDValue visitSRA(SDNode *N);
261     SDValue visitSRL(SDNode *N);
262     SDValue visitRotate(SDNode *N);
263     SDValue visitBSWAP(SDNode *N);
264     SDValue visitBITREVERSE(SDNode *N);
265     SDValue visitCTLZ(SDNode *N);
266     SDValue visitCTLZ_ZERO_UNDEF(SDNode *N);
267     SDValue visitCTTZ(SDNode *N);
268     SDValue visitCTTZ_ZERO_UNDEF(SDNode *N);
269     SDValue visitCTPOP(SDNode *N);
270     SDValue visitSELECT(SDNode *N);
271     SDValue visitVSELECT(SDNode *N);
272     SDValue visitSELECT_CC(SDNode *N);
273     SDValue visitSETCC(SDNode *N);
274     SDValue visitSETCCE(SDNode *N);
275     SDValue visitSIGN_EXTEND(SDNode *N);
276     SDValue visitZERO_EXTEND(SDNode *N);
277     SDValue visitANY_EXTEND(SDNode *N);
278     SDValue visitSIGN_EXTEND_INREG(SDNode *N);
279     SDValue visitSIGN_EXTEND_VECTOR_INREG(SDNode *N);
280     SDValue visitZERO_EXTEND_VECTOR_INREG(SDNode *N);
281     SDValue visitTRUNCATE(SDNode *N);
282     SDValue visitBITCAST(SDNode *N);
283     SDValue visitBUILD_PAIR(SDNode *N);
284     SDValue visitFADD(SDNode *N);
285     SDValue visitFSUB(SDNode *N);
286     SDValue visitFMUL(SDNode *N);
287     SDValue visitFMA(SDNode *N);
288     SDValue visitFDIV(SDNode *N);
289     SDValue visitFREM(SDNode *N);
290     SDValue visitFSQRT(SDNode *N);
291     SDValue visitFCOPYSIGN(SDNode *N);
292     SDValue visitSINT_TO_FP(SDNode *N);
293     SDValue visitUINT_TO_FP(SDNode *N);
294     SDValue visitFP_TO_SINT(SDNode *N);
295     SDValue visitFP_TO_UINT(SDNode *N);
296     SDValue visitFP_ROUND(SDNode *N);
297     SDValue visitFP_ROUND_INREG(SDNode *N);
298     SDValue visitFP_EXTEND(SDNode *N);
299     SDValue visitFNEG(SDNode *N);
300     SDValue visitFABS(SDNode *N);
301     SDValue visitFCEIL(SDNode *N);
302     SDValue visitFTRUNC(SDNode *N);
303     SDValue visitFFLOOR(SDNode *N);
304     SDValue visitFMINNUM(SDNode *N);
305     SDValue visitFMAXNUM(SDNode *N);
306     SDValue visitBRCOND(SDNode *N);
307     SDValue visitBR_CC(SDNode *N);
308     SDValue visitLOAD(SDNode *N);
309 
310     SDValue replaceStoreChain(StoreSDNode *ST, SDValue BetterChain);
311     SDValue replaceStoreOfFPConstant(StoreSDNode *ST);
312 
313     SDValue visitSTORE(SDNode *N);
314     SDValue visitINSERT_VECTOR_ELT(SDNode *N);
315     SDValue visitEXTRACT_VECTOR_ELT(SDNode *N);
316     SDValue visitBUILD_VECTOR(SDNode *N);
317     SDValue visitCONCAT_VECTORS(SDNode *N);
318     SDValue visitEXTRACT_SUBVECTOR(SDNode *N);
319     SDValue visitVECTOR_SHUFFLE(SDNode *N);
320     SDValue visitSCALAR_TO_VECTOR(SDNode *N);
321     SDValue visitINSERT_SUBVECTOR(SDNode *N);
322     SDValue visitMLOAD(SDNode *N);
323     SDValue visitMSTORE(SDNode *N);
324     SDValue visitMGATHER(SDNode *N);
325     SDValue visitMSCATTER(SDNode *N);
326     SDValue visitFP_TO_FP16(SDNode *N);
327     SDValue visitFP16_TO_FP(SDNode *N);
328 
329     SDValue visitFADDForFMACombine(SDNode *N);
330     SDValue visitFSUBForFMACombine(SDNode *N);
331     SDValue visitFMULForFMADistributiveCombine(SDNode *N);
332 
333     SDValue XformToShuffleWithZero(SDNode *N);
334     SDValue ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue LHS,
335                            SDValue RHS);
336 
337     SDValue visitShiftByConstant(SDNode *N, ConstantSDNode *Amt);
338 
339     SDValue foldSelectOfConstants(SDNode *N);
340     bool SimplifySelectOps(SDNode *SELECT, SDValue LHS, SDValue RHS);
341     SDValue SimplifyBinOpWithSameOpcodeHands(SDNode *N);
342     SDValue SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2);
343     SDValue SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
344                              SDValue N2, SDValue N3, ISD::CondCode CC,
345                              bool NotExtCompare = false);
346     SDValue foldSelectCCToShiftAnd(const SDLoc &DL, SDValue N0, SDValue N1,
347                                    SDValue N2, SDValue N3, ISD::CondCode CC);
348     SDValue SimplifySetCC(EVT VT, SDValue N0, SDValue N1, ISD::CondCode Cond,
349                           const SDLoc &DL, bool foldBooleans = true);
350 
351     bool isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
352                            SDValue &CC) const;
353     bool isOneUseSetCC(SDValue N) const;
354 
355     SDValue SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
356                                          unsigned HiOp);
357     SDValue CombineConsecutiveLoads(SDNode *N, EVT VT);
358     SDValue CombineExtLoad(SDNode *N);
359     SDValue combineRepeatedFPDivisors(SDNode *N);
360     SDValue ConstantFoldBITCASTofBUILD_VECTOR(SDNode *, EVT);
361     SDValue BuildSDIV(SDNode *N);
362     SDValue BuildSDIVPow2(SDNode *N);
363     SDValue BuildUDIV(SDNode *N);
364     SDValue BuildLogBase2(SDValue Op, const SDLoc &DL);
365     SDValue BuildReciprocalEstimate(SDValue Op, SDNodeFlags *Flags);
366     SDValue buildRsqrtEstimate(SDValue Op, SDNodeFlags *Flags);
367     SDValue buildSqrtEstimate(SDValue Op, SDNodeFlags *Flags);
368     SDValue buildSqrtEstimateImpl(SDValue Op, SDNodeFlags *Flags, bool Recip);
369     SDValue buildSqrtNROneConst(SDValue Op, SDValue Est, unsigned Iterations,
370                                 SDNodeFlags *Flags, bool Reciprocal);
371     SDValue buildSqrtNRTwoConst(SDValue Op, SDValue Est, unsigned Iterations,
372                                 SDNodeFlags *Flags, bool Reciprocal);
373     SDValue MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
374                                bool DemandHighBits = true);
375     SDValue MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1);
376     SDNode *MatchRotatePosNeg(SDValue Shifted, SDValue Pos, SDValue Neg,
377                               SDValue InnerPos, SDValue InnerNeg,
378                               unsigned PosOpcode, unsigned NegOpcode,
379                               const SDLoc &DL);
380     SDNode *MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL);
381     SDValue MatchLoadCombine(SDNode *N);
382     SDValue ReduceLoadWidth(SDNode *N);
383     SDValue ReduceLoadOpStoreWidth(SDNode *N);
384     SDValue splitMergedValStore(StoreSDNode *ST);
385     SDValue TransformFPLoadStorePair(SDNode *N);
386     SDValue reduceBuildVecExtToExtBuildVec(SDNode *N);
387     SDValue reduceBuildVecConvertToConvertBuildVec(SDNode *N);
388     SDValue reduceBuildVecToShuffle(SDNode *N);
389     SDValue createBuildVecShuffle(const SDLoc &DL, SDNode *N,
390                                   ArrayRef<int> VectorMask, SDValue VecIn1,
391                                   SDValue VecIn2, unsigned LeftIdx);
392 
393     SDValue GetDemandedBits(SDValue V, const APInt &Mask);
394 
395     /// Walk up chain skipping non-aliasing memory nodes,
396     /// looking for aliasing nodes and adding them to the Aliases vector.
397     void GatherAllAliases(SDNode *N, SDValue OriginalChain,
398                           SmallVectorImpl<SDValue> &Aliases);
399 
400     /// Return true if there is any possibility that the two addresses overlap.
401     bool isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const;
402 
403     /// Walk up chain skipping non-aliasing memory nodes, looking for a better
404     /// chain (aliasing node.)
405     SDValue FindBetterChain(SDNode *N, SDValue Chain);
406 
407     /// Try to replace a store and any possibly adjacent stores on
408     /// consecutive chains with better chains. Return true only if St is
409     /// replaced.
410     ///
411     /// Notice that other chains may still be replaced even if the function
412     /// returns false.
413     bool findBetterNeighborChains(StoreSDNode *St);
414 
415     /// Match "(X shl/srl V1) & V2" where V2 may not be present.
416     bool MatchRotateHalf(SDValue Op, SDValue &Shift, SDValue &Mask);
417 
418     /// Holds a pointer to an LSBaseSDNode as well as information on where it
419     /// is located in a sequence of memory operations connected by a chain.
420     struct MemOpLink {
421       MemOpLink (LSBaseSDNode *N, int64_t Offset, unsigned Seq):
422       MemNode(N), OffsetFromBase(Offset), SequenceNum(Seq) { }
423       // Ptr to the mem node.
424       LSBaseSDNode *MemNode;
425       // Offset from the base ptr.
426       int64_t OffsetFromBase;
427       // What is the sequence number of this mem node.
428       // Lowest mem operand in the DAG starts at zero.
429       unsigned SequenceNum;
430     };
431 
432     /// This is a helper function for visitMUL to check the profitability
433     /// of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
434     /// MulNode is the original multiply, AddNode is (add x, c1),
435     /// and ConstNode is c2.
436     bool isMulAddWithConstProfitable(SDNode *MulNode,
437                                      SDValue &AddNode,
438                                      SDValue &ConstNode);
439 
440     /// This is a helper function for MergeStoresOfConstantsOrVecElts. Returns a
441     /// constant build_vector of the stored constant values in Stores.
442     SDValue getMergedConstantVectorStore(SelectionDAG &DAG, const SDLoc &SL,
443                                          ArrayRef<MemOpLink> Stores,
444                                          SmallVectorImpl<SDValue> &Chains,
445                                          EVT Ty) const;
446 
447     /// This is a helper function for visitAND and visitZERO_EXTEND.  Returns
448     /// true if the (and (load x) c) pattern matches an extload.  ExtVT returns
449     /// the type of the loaded value to be extended.  LoadedVT returns the type
450     /// of the original loaded value.  NarrowLoad returns whether the load would
451     /// need to be narrowed in order to match.
452     bool isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
453                           EVT LoadResultTy, EVT &ExtVT, EVT &LoadedVT,
454                           bool &NarrowLoad);
455 
456     /// This is a helper function for MergeConsecutiveStores. When the source
457     /// elements of the consecutive stores are all constants or all extracted
458     /// vector elements, try to merge them into one larger store.
459     /// \return number of stores that were merged into a merged store (always
460     /// a prefix of \p StoreNode).
461     bool MergeStoresOfConstantsOrVecElts(
462         SmallVectorImpl<MemOpLink> &StoreNodes, EVT MemVT, unsigned NumStores,
463         bool IsConstantSrc, bool UseVector);
464 
465     /// This is a helper function for MergeConsecutiveStores.
466     /// Stores that may be merged are placed in StoreNodes.
467     /// Loads that may alias with those stores are placed in AliasLoadNodes.
468     void getStoreMergeAndAliasCandidates(
469         StoreSDNode* St, SmallVectorImpl<MemOpLink> &StoreNodes,
470         SmallVectorImpl<LSBaseSDNode*> &AliasLoadNodes);
471 
472     /// Helper function for MergeConsecutiveStores. Checks if
473     /// Candidate stores have indirect dependency through their
474     /// operands. \return True if safe to merge
475     bool checkMergeStoreCandidatesForDependencies(
476         SmallVectorImpl<MemOpLink> &StoreNodes);
477 
478     /// Merge consecutive store operations into a wide store.
479     /// This optimization uses wide integers or vectors when possible.
480     /// \return number of stores that were merged into a merged store (the
481     /// affected nodes are stored as a prefix in \p StoreNodes).
482     bool MergeConsecutiveStores(StoreSDNode *N,
483                                 SmallVectorImpl<MemOpLink> &StoreNodes);
484 
485     /// \brief Try to transform a truncation where C is a constant:
486     ///     (trunc (and X, C)) -> (and (trunc X), (trunc C))
487     ///
488     /// \p N needs to be a truncation and its first operand an AND. Other
489     /// requirements are checked by the function (e.g. that trunc is
490     /// single-use) and if missed an empty SDValue is returned.
491     SDValue distributeTruncateThroughAnd(SDNode *N);
492 
493   public:
494     DAGCombiner(SelectionDAG &D, AliasAnalysis &A, CodeGenOpt::Level OL)
495         : DAG(D), TLI(D.getTargetLoweringInfo()), Level(BeforeLegalizeTypes),
496           OptLevel(OL), LegalOperations(false), LegalTypes(false), AA(A) {
497       ForCodeSize = DAG.getMachineFunction().getFunction()->optForSize();
498     }
499 
500     /// Runs the dag combiner on all nodes in the work list
501     void Run(CombineLevel AtLevel);
502 
503     SelectionDAG &getDAG() const { return DAG; }
504 
505     /// Returns a type large enough to hold any valid shift amount - before type
506     /// legalization these can be huge.
507     EVT getShiftAmountTy(EVT LHSTy) {
508       assert(LHSTy.isInteger() && "Shift amount is not an integer type!");
509       if (LHSTy.isVector())
510         return LHSTy;
511       auto &DL = DAG.getDataLayout();
512       return LegalTypes ? TLI.getScalarShiftAmountTy(DL, LHSTy)
513                         : TLI.getPointerTy(DL);
514     }
515 
516     /// This method returns true if we are running before type legalization or
517     /// if the specified VT is legal.
518     bool isTypeLegal(const EVT &VT) {
519       if (!LegalTypes) return true;
520       return TLI.isTypeLegal(VT);
521     }
522 
523     /// Convenience wrapper around TargetLowering::getSetCCResultType
524     EVT getSetCCResultType(EVT VT) const {
525       return TLI.getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
526     }
527   };
528 }
529 
530 
531 namespace {
532 /// This class is a DAGUpdateListener that removes any deleted
533 /// nodes from the worklist.
534 class WorklistRemover : public SelectionDAG::DAGUpdateListener {
535   DAGCombiner &DC;
536 public:
537   explicit WorklistRemover(DAGCombiner &dc)
538     : SelectionDAG::DAGUpdateListener(dc.getDAG()), DC(dc) {}
539 
540   void NodeDeleted(SDNode *N, SDNode *E) override {
541     DC.removeFromWorklist(N);
542   }
543 };
544 }
545 
546 //===----------------------------------------------------------------------===//
547 //  TargetLowering::DAGCombinerInfo implementation
548 //===----------------------------------------------------------------------===//
549 
550 void TargetLowering::DAGCombinerInfo::AddToWorklist(SDNode *N) {
551   ((DAGCombiner*)DC)->AddToWorklist(N);
552 }
553 
554 SDValue TargetLowering::DAGCombinerInfo::
555 CombineTo(SDNode *N, ArrayRef<SDValue> To, bool AddTo) {
556   return ((DAGCombiner*)DC)->CombineTo(N, &To[0], To.size(), AddTo);
557 }
558 
559 SDValue TargetLowering::DAGCombinerInfo::
560 CombineTo(SDNode *N, SDValue Res, bool AddTo) {
561   return ((DAGCombiner*)DC)->CombineTo(N, Res, AddTo);
562 }
563 
564 
565 SDValue TargetLowering::DAGCombinerInfo::
566 CombineTo(SDNode *N, SDValue Res0, SDValue Res1, bool AddTo) {
567   return ((DAGCombiner*)DC)->CombineTo(N, Res0, Res1, AddTo);
568 }
569 
570 void TargetLowering::DAGCombinerInfo::
571 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
572   return ((DAGCombiner*)DC)->CommitTargetLoweringOpt(TLO);
573 }
574 
575 //===----------------------------------------------------------------------===//
576 // Helper Functions
577 //===----------------------------------------------------------------------===//
578 
579 void DAGCombiner::deleteAndRecombine(SDNode *N) {
580   removeFromWorklist(N);
581 
582   // If the operands of this node are only used by the node, they will now be
583   // dead. Make sure to re-visit them and recursively delete dead nodes.
584   for (const SDValue &Op : N->ops())
585     // For an operand generating multiple values, one of the values may
586     // become dead allowing further simplification (e.g. split index
587     // arithmetic from an indexed load).
588     if (Op->hasOneUse() || Op->getNumValues() > 1)
589       AddToWorklist(Op.getNode());
590 
591   DAG.DeleteNode(N);
592 }
593 
594 /// Return 1 if we can compute the negated form of the specified expression for
595 /// the same cost as the expression itself, or 2 if we can compute the negated
596 /// form more cheaply than the expression itself.
597 static char isNegatibleForFree(SDValue Op, bool LegalOperations,
598                                const TargetLowering &TLI,
599                                const TargetOptions *Options,
600                                unsigned Depth = 0) {
601   // fneg is removable even if it has multiple uses.
602   if (Op.getOpcode() == ISD::FNEG) return 2;
603 
604   // Don't allow anything with multiple uses.
605   if (!Op.hasOneUse()) return 0;
606 
607   // Don't recurse exponentially.
608   if (Depth > 6) return 0;
609 
610   switch (Op.getOpcode()) {
611   default: return false;
612   case ISD::ConstantFP: {
613     if (!LegalOperations)
614       return 1;
615 
616     // Don't invert constant FP values after legalization unless the target says
617     // the negated constant is legal.
618     EVT VT = Op.getValueType();
619     return TLI.isOperationLegal(ISD::ConstantFP, VT) ||
620       TLI.isFPImmLegal(neg(cast<ConstantFPSDNode>(Op)->getValueAPF()), VT);
621   }
622   case ISD::FADD:
623     // FIXME: determine better conditions for this xform.
624     if (!Options->UnsafeFPMath) return 0;
625 
626     // After operation legalization, it might not be legal to create new FSUBs.
627     if (LegalOperations &&
628         !TLI.isOperationLegalOrCustom(ISD::FSUB,  Op.getValueType()))
629       return 0;
630 
631     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
632     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
633                                     Options, Depth + 1))
634       return V;
635     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
636     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
637                               Depth + 1);
638   case ISD::FSUB:
639     // We can't turn -(A-B) into B-A when we honor signed zeros.
640     if (!Options->NoSignedZerosFPMath &&
641         !Op.getNode()->getFlags()->hasNoSignedZeros())
642       return 0;
643 
644     // fold (fneg (fsub A, B)) -> (fsub B, A)
645     return 1;
646 
647   case ISD::FMUL:
648   case ISD::FDIV:
649     if (Options->HonorSignDependentRoundingFPMath()) return 0;
650 
651     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y) or (fmul X, (fneg Y))
652     if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI,
653                                     Options, Depth + 1))
654       return V;
655 
656     return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options,
657                               Depth + 1);
658 
659   case ISD::FP_EXTEND:
660   case ISD::FP_ROUND:
661   case ISD::FSIN:
662     return isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI, Options,
663                               Depth + 1);
664   }
665 }
666 
667 /// If isNegatibleForFree returns true, return the newly negated expression.
668 static SDValue GetNegatedExpression(SDValue Op, SelectionDAG &DAG,
669                                     bool LegalOperations, unsigned Depth = 0) {
670   const TargetOptions &Options = DAG.getTarget().Options;
671   // fneg is removable even if it has multiple uses.
672   if (Op.getOpcode() == ISD::FNEG) return Op.getOperand(0);
673 
674   // Don't allow anything with multiple uses.
675   assert(Op.hasOneUse() && "Unknown reuse!");
676 
677   assert(Depth <= 6 && "GetNegatedExpression doesn't match isNegatibleForFree");
678 
679   const SDNodeFlags *Flags = Op.getNode()->getFlags();
680 
681   switch (Op.getOpcode()) {
682   default: llvm_unreachable("Unknown code");
683   case ISD::ConstantFP: {
684     APFloat V = cast<ConstantFPSDNode>(Op)->getValueAPF();
685     V.changeSign();
686     return DAG.getConstantFP(V, SDLoc(Op), Op.getValueType());
687   }
688   case ISD::FADD:
689     // FIXME: determine better conditions for this xform.
690     assert(Options.UnsafeFPMath);
691 
692     // fold (fneg (fadd A, B)) -> (fsub (fneg A), B)
693     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
694                            DAG.getTargetLoweringInfo(), &Options, Depth+1))
695       return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
696                          GetNegatedExpression(Op.getOperand(0), DAG,
697                                               LegalOperations, Depth+1),
698                          Op.getOperand(1), Flags);
699     // fold (fneg (fadd A, B)) -> (fsub (fneg B), A)
700     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
701                        GetNegatedExpression(Op.getOperand(1), DAG,
702                                             LegalOperations, Depth+1),
703                        Op.getOperand(0), Flags);
704   case ISD::FSUB:
705     // fold (fneg (fsub 0, B)) -> B
706     if (ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(Op.getOperand(0)))
707       if (N0CFP->isZero())
708         return Op.getOperand(1);
709 
710     // fold (fneg (fsub A, B)) -> (fsub B, A)
711     return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(),
712                        Op.getOperand(1), Op.getOperand(0), Flags);
713 
714   case ISD::FMUL:
715   case ISD::FDIV:
716     assert(!Options.HonorSignDependentRoundingFPMath());
717 
718     // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y)
719     if (isNegatibleForFree(Op.getOperand(0), LegalOperations,
720                            DAG.getTargetLoweringInfo(), &Options, Depth+1))
721       return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
722                          GetNegatedExpression(Op.getOperand(0), DAG,
723                                               LegalOperations, Depth+1),
724                          Op.getOperand(1), Flags);
725 
726     // fold (fneg (fmul X, Y)) -> (fmul X, (fneg Y))
727     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
728                        Op.getOperand(0),
729                        GetNegatedExpression(Op.getOperand(1), DAG,
730                                             LegalOperations, Depth+1), Flags);
731 
732   case ISD::FP_EXTEND:
733   case ISD::FSIN:
734     return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(),
735                        GetNegatedExpression(Op.getOperand(0), DAG,
736                                             LegalOperations, Depth+1));
737   case ISD::FP_ROUND:
738       return DAG.getNode(ISD::FP_ROUND, SDLoc(Op), Op.getValueType(),
739                          GetNegatedExpression(Op.getOperand(0), DAG,
740                                               LegalOperations, Depth+1),
741                          Op.getOperand(1));
742   }
743 }
744 
745 // APInts must be the same size for most operations, this helper
746 // function zero extends the shorter of the pair so that they match.
747 // We provide an Offset so that we can create bitwidths that won't overflow.
748 static void zeroExtendToMatch(APInt &LHS, APInt &RHS, unsigned Offset = 0) {
749   unsigned Bits = Offset + std::max(LHS.getBitWidth(), RHS.getBitWidth());
750   LHS = LHS.zextOrSelf(Bits);
751   RHS = RHS.zextOrSelf(Bits);
752 }
753 
754 // Return true if this node is a setcc, or is a select_cc
755 // that selects between the target values used for true and false, making it
756 // equivalent to a setcc. Also, set the incoming LHS, RHS, and CC references to
757 // the appropriate nodes based on the type of node we are checking. This
758 // simplifies life a bit for the callers.
759 bool DAGCombiner::isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS,
760                                     SDValue &CC) const {
761   if (N.getOpcode() == ISD::SETCC) {
762     LHS = N.getOperand(0);
763     RHS = N.getOperand(1);
764     CC  = N.getOperand(2);
765     return true;
766   }
767 
768   if (N.getOpcode() != ISD::SELECT_CC ||
769       !TLI.isConstTrueVal(N.getOperand(2).getNode()) ||
770       !TLI.isConstFalseVal(N.getOperand(3).getNode()))
771     return false;
772 
773   if (TLI.getBooleanContents(N.getValueType()) ==
774       TargetLowering::UndefinedBooleanContent)
775     return false;
776 
777   LHS = N.getOperand(0);
778   RHS = N.getOperand(1);
779   CC  = N.getOperand(4);
780   return true;
781 }
782 
783 /// Return true if this is a SetCC-equivalent operation with only one use.
784 /// If this is true, it allows the users to invert the operation for free when
785 /// it is profitable to do so.
786 bool DAGCombiner::isOneUseSetCC(SDValue N) const {
787   SDValue N0, N1, N2;
788   if (isSetCCEquivalent(N, N0, N1, N2) && N.getNode()->hasOneUse())
789     return true;
790   return false;
791 }
792 
793 // \brief Returns the SDNode if it is a constant float BuildVector
794 // or constant float.
795 static SDNode *isConstantFPBuildVectorOrConstantFP(SDValue N) {
796   if (isa<ConstantFPSDNode>(N))
797     return N.getNode();
798   if (ISD::isBuildVectorOfConstantFPSDNodes(N.getNode()))
799     return N.getNode();
800   return nullptr;
801 }
802 
803 // Determines if it is a constant integer or a build vector of constant
804 // integers (and undefs).
805 // Do not permit build vector implicit truncation.
806 static bool isConstantOrConstantVector(SDValue N, bool NoOpaques = false) {
807   if (ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N))
808     return !(Const->isOpaque() && NoOpaques);
809   if (N.getOpcode() != ISD::BUILD_VECTOR)
810     return false;
811   unsigned BitWidth = N.getScalarValueSizeInBits();
812   for (const SDValue &Op : N->op_values()) {
813     if (Op.isUndef())
814       continue;
815     ConstantSDNode *Const = dyn_cast<ConstantSDNode>(Op);
816     if (!Const || Const->getAPIntValue().getBitWidth() != BitWidth ||
817         (Const->isOpaque() && NoOpaques))
818       return false;
819   }
820   return true;
821 }
822 
823 // Determines if it is a constant null integer or a splatted vector of a
824 // constant null integer (with no undefs).
825 // Build vector implicit truncation is not an issue for null values.
826 static bool isNullConstantOrNullSplatConstant(SDValue N) {
827   if (ConstantSDNode *Splat = isConstOrConstSplat(N))
828     return Splat->isNullValue();
829   return false;
830 }
831 
832 // Determines if it is a constant integer of one or a splatted vector of a
833 // constant integer of one (with no undefs).
834 // Do not permit build vector implicit truncation.
835 static bool isOneConstantOrOneSplatConstant(SDValue N) {
836   unsigned BitWidth = N.getScalarValueSizeInBits();
837   if (ConstantSDNode *Splat = isConstOrConstSplat(N))
838     return Splat->isOne() && Splat->getAPIntValue().getBitWidth() == BitWidth;
839   return false;
840 }
841 
842 // Determines if it is a constant integer of all ones or a splatted vector of a
843 // constant integer of all ones (with no undefs).
844 // Do not permit build vector implicit truncation.
845 static bool isAllOnesConstantOrAllOnesSplatConstant(SDValue N) {
846   unsigned BitWidth = N.getScalarValueSizeInBits();
847   if (ConstantSDNode *Splat = isConstOrConstSplat(N))
848     return Splat->isAllOnesValue() &&
849            Splat->getAPIntValue().getBitWidth() == BitWidth;
850   return false;
851 }
852 
853 // Determines if a BUILD_VECTOR is composed of all-constants possibly mixed with
854 // undef's.
855 static bool isAnyConstantBuildVector(const SDNode *N) {
856   return ISD::isBuildVectorOfConstantSDNodes(N) ||
857          ISD::isBuildVectorOfConstantFPSDNodes(N);
858 }
859 
860 SDValue DAGCombiner::ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue N0,
861                                     SDValue N1) {
862   EVT VT = N0.getValueType();
863   if (N0.getOpcode() == Opc) {
864     if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1))) {
865       if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
866         // reassoc. (op (op x, c1), c2) -> (op x, (op c1, c2))
867         if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, L, R))
868           return DAG.getNode(Opc, DL, VT, N0.getOperand(0), OpNode);
869         return SDValue();
870       }
871       if (N0.hasOneUse()) {
872         // reassoc. (op (op x, c1), y) -> (op (op x, y), c1) iff x+c1 has one
873         // use
874         SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0.getOperand(0), N1);
875         if (!OpNode.getNode())
876           return SDValue();
877         AddToWorklist(OpNode.getNode());
878         return DAG.getNode(Opc, DL, VT, OpNode, N0.getOperand(1));
879       }
880     }
881   }
882 
883   if (N1.getOpcode() == Opc) {
884     if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1.getOperand(1))) {
885       if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
886         // reassoc. (op c2, (op x, c1)) -> (op x, (op c1, c2))
887         if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, R, L))
888           return DAG.getNode(Opc, DL, VT, N1.getOperand(0), OpNode);
889         return SDValue();
890       }
891       if (N1.hasOneUse()) {
892         // reassoc. (op x, (op y, c1)) -> (op (op x, y), c1) iff x+c1 has one
893         // use
894         SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0, N1.getOperand(0));
895         if (!OpNode.getNode())
896           return SDValue();
897         AddToWorklist(OpNode.getNode());
898         return DAG.getNode(Opc, DL, VT, OpNode, N1.getOperand(1));
899       }
900     }
901   }
902 
903   return SDValue();
904 }
905 
906 SDValue DAGCombiner::CombineTo(SDNode *N, const SDValue *To, unsigned NumTo,
907                                bool AddTo) {
908   assert(N->getNumValues() == NumTo && "Broken CombineTo call!");
909   ++NodesCombined;
910   DEBUG(dbgs() << "\nReplacing.1 ";
911         N->dump(&DAG);
912         dbgs() << "\nWith: ";
913         To[0].getNode()->dump(&DAG);
914         dbgs() << " and " << NumTo-1 << " other values\n");
915   for (unsigned i = 0, e = NumTo; i != e; ++i)
916     assert((!To[i].getNode() ||
917             N->getValueType(i) == To[i].getValueType()) &&
918            "Cannot combine value to value of different type!");
919 
920   WorklistRemover DeadNodes(*this);
921   DAG.ReplaceAllUsesWith(N, To);
922   if (AddTo) {
923     // Push the new nodes and any users onto the worklist
924     for (unsigned i = 0, e = NumTo; i != e; ++i) {
925       if (To[i].getNode()) {
926         AddToWorklist(To[i].getNode());
927         AddUsersToWorklist(To[i].getNode());
928       }
929     }
930   }
931 
932   // Finally, if the node is now dead, remove it from the graph.  The node
933   // may not be dead if the replacement process recursively simplified to
934   // something else needing this node.
935   if (N->use_empty())
936     deleteAndRecombine(N);
937   return SDValue(N, 0);
938 }
939 
940 void DAGCombiner::
941 CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) {
942   // Replace all uses.  If any nodes become isomorphic to other nodes and
943   // are deleted, make sure to remove them from our worklist.
944   WorklistRemover DeadNodes(*this);
945   DAG.ReplaceAllUsesOfValueWith(TLO.Old, TLO.New);
946 
947   // Push the new node and any (possibly new) users onto the worklist.
948   AddToWorklist(TLO.New.getNode());
949   AddUsersToWorklist(TLO.New.getNode());
950 
951   // Finally, if the node is now dead, remove it from the graph.  The node
952   // may not be dead if the replacement process recursively simplified to
953   // something else needing this node.
954   if (TLO.Old.getNode()->use_empty())
955     deleteAndRecombine(TLO.Old.getNode());
956 }
957 
958 /// Check the specified integer node value to see if it can be simplified or if
959 /// things it uses can be simplified by bit propagation. If so, return true.
960 bool DAGCombiner::SimplifyDemandedBits(SDValue Op, const APInt &Demanded) {
961   TargetLowering::TargetLoweringOpt TLO(DAG, LegalTypes, LegalOperations);
962   APInt KnownZero, KnownOne;
963   if (!TLI.SimplifyDemandedBits(Op, Demanded, KnownZero, KnownOne, TLO))
964     return false;
965 
966   // Revisit the node.
967   AddToWorklist(Op.getNode());
968 
969   // Replace the old value with the new one.
970   ++NodesCombined;
971   DEBUG(dbgs() << "\nReplacing.2 ";
972         TLO.Old.getNode()->dump(&DAG);
973         dbgs() << "\nWith: ";
974         TLO.New.getNode()->dump(&DAG);
975         dbgs() << '\n');
976 
977   CommitTargetLoweringOpt(TLO);
978   return true;
979 }
980 
981 void DAGCombiner::ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad) {
982   SDLoc DL(Load);
983   EVT VT = Load->getValueType(0);
984   SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, VT, SDValue(ExtLoad, 0));
985 
986   DEBUG(dbgs() << "\nReplacing.9 ";
987         Load->dump(&DAG);
988         dbgs() << "\nWith: ";
989         Trunc.getNode()->dump(&DAG);
990         dbgs() << '\n');
991   WorklistRemover DeadNodes(*this);
992   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 0), Trunc);
993   DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), SDValue(ExtLoad, 1));
994   deleteAndRecombine(Load);
995   AddToWorklist(Trunc.getNode());
996 }
997 
998 SDValue DAGCombiner::PromoteOperand(SDValue Op, EVT PVT, bool &Replace) {
999   Replace = false;
1000   SDLoc DL(Op);
1001   if (ISD::isUNINDEXEDLoad(Op.getNode())) {
1002     LoadSDNode *LD = cast<LoadSDNode>(Op);
1003     EVT MemVT = LD->getMemoryVT();
1004     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD)
1005       ? (TLI.isLoadExtLegal(ISD::ZEXTLOAD, PVT, MemVT) ? ISD::ZEXTLOAD
1006                                                        : ISD::EXTLOAD)
1007       : LD->getExtensionType();
1008     Replace = true;
1009     return DAG.getExtLoad(ExtType, DL, PVT,
1010                           LD->getChain(), LD->getBasePtr(),
1011                           MemVT, LD->getMemOperand());
1012   }
1013 
1014   unsigned Opc = Op.getOpcode();
1015   switch (Opc) {
1016   default: break;
1017   case ISD::AssertSext:
1018     return DAG.getNode(ISD::AssertSext, DL, PVT,
1019                        SExtPromoteOperand(Op.getOperand(0), PVT),
1020                        Op.getOperand(1));
1021   case ISD::AssertZext:
1022     return DAG.getNode(ISD::AssertZext, DL, PVT,
1023                        ZExtPromoteOperand(Op.getOperand(0), PVT),
1024                        Op.getOperand(1));
1025   case ISD::Constant: {
1026     unsigned ExtOpc =
1027       Op.getValueType().isByteSized() ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
1028     return DAG.getNode(ExtOpc, DL, PVT, Op);
1029   }
1030   }
1031 
1032   if (!TLI.isOperationLegal(ISD::ANY_EXTEND, PVT))
1033     return SDValue();
1034   return DAG.getNode(ISD::ANY_EXTEND, DL, PVT, Op);
1035 }
1036 
1037 SDValue DAGCombiner::SExtPromoteOperand(SDValue Op, EVT PVT) {
1038   if (!TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, PVT))
1039     return SDValue();
1040   EVT OldVT = Op.getValueType();
1041   SDLoc DL(Op);
1042   bool Replace = false;
1043   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1044   if (!NewOp.getNode())
1045     return SDValue();
1046   AddToWorklist(NewOp.getNode());
1047 
1048   if (Replace)
1049     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1050   return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, NewOp.getValueType(), NewOp,
1051                      DAG.getValueType(OldVT));
1052 }
1053 
1054 SDValue DAGCombiner::ZExtPromoteOperand(SDValue Op, EVT PVT) {
1055   EVT OldVT = Op.getValueType();
1056   SDLoc DL(Op);
1057   bool Replace = false;
1058   SDValue NewOp = PromoteOperand(Op, PVT, Replace);
1059   if (!NewOp.getNode())
1060     return SDValue();
1061   AddToWorklist(NewOp.getNode());
1062 
1063   if (Replace)
1064     ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode());
1065   return DAG.getZeroExtendInReg(NewOp, DL, OldVT);
1066 }
1067 
1068 /// Promote the specified integer binary operation if the target indicates it is
1069 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1070 /// i32 since i16 instructions are longer.
1071 SDValue DAGCombiner::PromoteIntBinOp(SDValue Op) {
1072   if (!LegalOperations)
1073     return SDValue();
1074 
1075   EVT VT = Op.getValueType();
1076   if (VT.isVector() || !VT.isInteger())
1077     return SDValue();
1078 
1079   // If operation type is 'undesirable', e.g. i16 on x86, consider
1080   // promoting it.
1081   unsigned Opc = Op.getOpcode();
1082   if (TLI.isTypeDesirableForOp(Opc, VT))
1083     return SDValue();
1084 
1085   EVT PVT = VT;
1086   // Consult target whether it is a good idea to promote this operation and
1087   // what's the right type to promote it to.
1088   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1089     assert(PVT != VT && "Don't know what type to promote to!");
1090 
1091     bool Replace0 = false;
1092     SDValue N0 = Op.getOperand(0);
1093     SDValue NN0 = PromoteOperand(N0, PVT, Replace0);
1094     if (!NN0.getNode())
1095       return SDValue();
1096 
1097     bool Replace1 = false;
1098     SDValue N1 = Op.getOperand(1);
1099     SDValue NN1;
1100     if (N0 == N1)
1101       NN1 = NN0;
1102     else {
1103       NN1 = PromoteOperand(N1, PVT, Replace1);
1104       if (!NN1.getNode())
1105         return SDValue();
1106     }
1107 
1108     AddToWorklist(NN0.getNode());
1109     if (NN1.getNode())
1110       AddToWorklist(NN1.getNode());
1111 
1112     if (Replace0)
1113       ReplaceLoadWithPromotedLoad(N0.getNode(), NN0.getNode());
1114     if (Replace1)
1115       ReplaceLoadWithPromotedLoad(N1.getNode(), NN1.getNode());
1116 
1117     DEBUG(dbgs() << "\nPromoting ";
1118           Op.getNode()->dump(&DAG));
1119     SDLoc DL(Op);
1120     return DAG.getNode(ISD::TRUNCATE, DL, VT,
1121                        DAG.getNode(Opc, DL, PVT, NN0, NN1));
1122   }
1123   return SDValue();
1124 }
1125 
1126 /// Promote the specified integer shift operation if the target indicates it is
1127 /// beneficial. e.g. On x86, it's usually better to promote i16 operations to
1128 /// i32 since i16 instructions are longer.
1129 SDValue DAGCombiner::PromoteIntShiftOp(SDValue Op) {
1130   if (!LegalOperations)
1131     return SDValue();
1132 
1133   EVT VT = Op.getValueType();
1134   if (VT.isVector() || !VT.isInteger())
1135     return SDValue();
1136 
1137   // If operation type is 'undesirable', e.g. i16 on x86, consider
1138   // promoting it.
1139   unsigned Opc = Op.getOpcode();
1140   if (TLI.isTypeDesirableForOp(Opc, VT))
1141     return SDValue();
1142 
1143   EVT PVT = VT;
1144   // Consult target whether it is a good idea to promote this operation and
1145   // what's the right type to promote it to.
1146   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1147     assert(PVT != VT && "Don't know what type to promote to!");
1148 
1149     bool Replace = false;
1150     SDValue N0 = Op.getOperand(0);
1151     if (Opc == ISD::SRA)
1152       N0 = SExtPromoteOperand(Op.getOperand(0), PVT);
1153     else if (Opc == ISD::SRL)
1154       N0 = ZExtPromoteOperand(Op.getOperand(0), PVT);
1155     else
1156       N0 = PromoteOperand(N0, PVT, Replace);
1157     if (!N0.getNode())
1158       return SDValue();
1159 
1160     AddToWorklist(N0.getNode());
1161     if (Replace)
1162       ReplaceLoadWithPromotedLoad(Op.getOperand(0).getNode(), N0.getNode());
1163 
1164     DEBUG(dbgs() << "\nPromoting ";
1165           Op.getNode()->dump(&DAG));
1166     SDLoc DL(Op);
1167     return DAG.getNode(ISD::TRUNCATE, DL, VT,
1168                        DAG.getNode(Opc, DL, PVT, N0, Op.getOperand(1)));
1169   }
1170   return SDValue();
1171 }
1172 
1173 SDValue DAGCombiner::PromoteExtend(SDValue Op) {
1174   if (!LegalOperations)
1175     return SDValue();
1176 
1177   EVT VT = Op.getValueType();
1178   if (VT.isVector() || !VT.isInteger())
1179     return SDValue();
1180 
1181   // If operation type is 'undesirable', e.g. i16 on x86, consider
1182   // promoting it.
1183   unsigned Opc = Op.getOpcode();
1184   if (TLI.isTypeDesirableForOp(Opc, VT))
1185     return SDValue();
1186 
1187   EVT PVT = VT;
1188   // Consult target whether it is a good idea to promote this operation and
1189   // what's the right type to promote it to.
1190   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1191     assert(PVT != VT && "Don't know what type to promote to!");
1192     // fold (aext (aext x)) -> (aext x)
1193     // fold (aext (zext x)) -> (zext x)
1194     // fold (aext (sext x)) -> (sext x)
1195     DEBUG(dbgs() << "\nPromoting ";
1196           Op.getNode()->dump(&DAG));
1197     return DAG.getNode(Op.getOpcode(), SDLoc(Op), VT, Op.getOperand(0));
1198   }
1199   return SDValue();
1200 }
1201 
1202 bool DAGCombiner::PromoteLoad(SDValue Op) {
1203   if (!LegalOperations)
1204     return false;
1205 
1206   if (!ISD::isUNINDEXEDLoad(Op.getNode()))
1207     return false;
1208 
1209   EVT VT = Op.getValueType();
1210   if (VT.isVector() || !VT.isInteger())
1211     return false;
1212 
1213   // If operation type is 'undesirable', e.g. i16 on x86, consider
1214   // promoting it.
1215   unsigned Opc = Op.getOpcode();
1216   if (TLI.isTypeDesirableForOp(Opc, VT))
1217     return false;
1218 
1219   EVT PVT = VT;
1220   // Consult target whether it is a good idea to promote this operation and
1221   // what's the right type to promote it to.
1222   if (TLI.IsDesirableToPromoteOp(Op, PVT)) {
1223     assert(PVT != VT && "Don't know what type to promote to!");
1224 
1225     SDLoc DL(Op);
1226     SDNode *N = Op.getNode();
1227     LoadSDNode *LD = cast<LoadSDNode>(N);
1228     EVT MemVT = LD->getMemoryVT();
1229     ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD)
1230       ? (TLI.isLoadExtLegal(ISD::ZEXTLOAD, PVT, MemVT) ? ISD::ZEXTLOAD
1231                                                        : ISD::EXTLOAD)
1232       : LD->getExtensionType();
1233     SDValue NewLD = DAG.getExtLoad(ExtType, DL, PVT,
1234                                    LD->getChain(), LD->getBasePtr(),
1235                                    MemVT, LD->getMemOperand());
1236     SDValue Result = DAG.getNode(ISD::TRUNCATE, DL, VT, NewLD);
1237 
1238     DEBUG(dbgs() << "\nPromoting ";
1239           N->dump(&DAG);
1240           dbgs() << "\nTo: ";
1241           Result.getNode()->dump(&DAG);
1242           dbgs() << '\n');
1243     WorklistRemover DeadNodes(*this);
1244     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result);
1245     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), NewLD.getValue(1));
1246     deleteAndRecombine(N);
1247     AddToWorklist(Result.getNode());
1248     return true;
1249   }
1250   return false;
1251 }
1252 
1253 /// \brief Recursively delete a node which has no uses and any operands for
1254 /// which it is the only use.
1255 ///
1256 /// Note that this both deletes the nodes and removes them from the worklist.
1257 /// It also adds any nodes who have had a user deleted to the worklist as they
1258 /// may now have only one use and subject to other combines.
1259 bool DAGCombiner::recursivelyDeleteUnusedNodes(SDNode *N) {
1260   if (!N->use_empty())
1261     return false;
1262 
1263   SmallSetVector<SDNode *, 16> Nodes;
1264   Nodes.insert(N);
1265   do {
1266     N = Nodes.pop_back_val();
1267     if (!N)
1268       continue;
1269 
1270     if (N->use_empty()) {
1271       for (const SDValue &ChildN : N->op_values())
1272         Nodes.insert(ChildN.getNode());
1273 
1274       removeFromWorklist(N);
1275       DAG.DeleteNode(N);
1276     } else {
1277       AddToWorklist(N);
1278     }
1279   } while (!Nodes.empty());
1280   return true;
1281 }
1282 
1283 //===----------------------------------------------------------------------===//
1284 //  Main DAG Combiner implementation
1285 //===----------------------------------------------------------------------===//
1286 
1287 void DAGCombiner::Run(CombineLevel AtLevel) {
1288   // set the instance variables, so that the various visit routines may use it.
1289   Level = AtLevel;
1290   LegalOperations = Level >= AfterLegalizeVectorOps;
1291   LegalTypes = Level >= AfterLegalizeTypes;
1292 
1293   // Add all the dag nodes to the worklist.
1294   for (SDNode &Node : DAG.allnodes())
1295     AddToWorklist(&Node);
1296 
1297   // Create a dummy node (which is not added to allnodes), that adds a reference
1298   // to the root node, preventing it from being deleted, and tracking any
1299   // changes of the root.
1300   HandleSDNode Dummy(DAG.getRoot());
1301 
1302   // While the worklist isn't empty, find a node and try to combine it.
1303   while (!WorklistMap.empty()) {
1304     SDNode *N;
1305     // The Worklist holds the SDNodes in order, but it may contain null entries.
1306     do {
1307       N = Worklist.pop_back_val();
1308     } while (!N);
1309 
1310     bool GoodWorklistEntry = WorklistMap.erase(N);
1311     (void)GoodWorklistEntry;
1312     assert(GoodWorklistEntry &&
1313            "Found a worklist entry without a corresponding map entry!");
1314 
1315     // If N has no uses, it is dead.  Make sure to revisit all N's operands once
1316     // N is deleted from the DAG, since they too may now be dead or may have a
1317     // reduced number of uses, allowing other xforms.
1318     if (recursivelyDeleteUnusedNodes(N))
1319       continue;
1320 
1321     WorklistRemover DeadNodes(*this);
1322 
1323     // If this combine is running after legalizing the DAG, re-legalize any
1324     // nodes pulled off the worklist.
1325     if (Level == AfterLegalizeDAG) {
1326       SmallSetVector<SDNode *, 16> UpdatedNodes;
1327       bool NIsValid = DAG.LegalizeOp(N, UpdatedNodes);
1328 
1329       for (SDNode *LN : UpdatedNodes) {
1330         AddToWorklist(LN);
1331         AddUsersToWorklist(LN);
1332       }
1333       if (!NIsValid)
1334         continue;
1335     }
1336 
1337     DEBUG(dbgs() << "\nCombining: "; N->dump(&DAG));
1338 
1339     // Add any operands of the new node which have not yet been combined to the
1340     // worklist as well. Because the worklist uniques things already, this
1341     // won't repeatedly process the same operand.
1342     CombinedNodes.insert(N);
1343     for (const SDValue &ChildN : N->op_values())
1344       if (!CombinedNodes.count(ChildN.getNode()))
1345         AddToWorklist(ChildN.getNode());
1346 
1347     SDValue RV = combine(N);
1348 
1349     if (!RV.getNode())
1350       continue;
1351 
1352     ++NodesCombined;
1353 
1354     // If we get back the same node we passed in, rather than a new node or
1355     // zero, we know that the node must have defined multiple values and
1356     // CombineTo was used.  Since CombineTo takes care of the worklist
1357     // mechanics for us, we have no work to do in this case.
1358     if (RV.getNode() == N)
1359       continue;
1360 
1361     assert(N->getOpcode() != ISD::DELETED_NODE &&
1362            RV.getOpcode() != ISD::DELETED_NODE &&
1363            "Node was deleted but visit returned new node!");
1364 
1365     DEBUG(dbgs() << " ... into: ";
1366           RV.getNode()->dump(&DAG));
1367 
1368     if (N->getNumValues() == RV.getNode()->getNumValues())
1369       DAG.ReplaceAllUsesWith(N, RV.getNode());
1370     else {
1371       assert(N->getValueType(0) == RV.getValueType() &&
1372              N->getNumValues() == 1 && "Type mismatch");
1373       SDValue OpV = RV;
1374       DAG.ReplaceAllUsesWith(N, &OpV);
1375     }
1376 
1377     // Push the new node and any users onto the worklist
1378     AddToWorklist(RV.getNode());
1379     AddUsersToWorklist(RV.getNode());
1380 
1381     // Finally, if the node is now dead, remove it from the graph.  The node
1382     // may not be dead if the replacement process recursively simplified to
1383     // something else needing this node. This will also take care of adding any
1384     // operands which have lost a user to the worklist.
1385     recursivelyDeleteUnusedNodes(N);
1386   }
1387 
1388   // If the root changed (e.g. it was a dead load, update the root).
1389   DAG.setRoot(Dummy.getValue());
1390   DAG.RemoveDeadNodes();
1391 }
1392 
1393 SDValue DAGCombiner::visit(SDNode *N) {
1394   switch (N->getOpcode()) {
1395   default: break;
1396   case ISD::TokenFactor:        return visitTokenFactor(N);
1397   case ISD::MERGE_VALUES:       return visitMERGE_VALUES(N);
1398   case ISD::ADD:                return visitADD(N);
1399   case ISD::SUB:                return visitSUB(N);
1400   case ISD::ADDC:               return visitADDC(N);
1401   case ISD::SUBC:               return visitSUBC(N);
1402   case ISD::ADDE:               return visitADDE(N);
1403   case ISD::SUBE:               return visitSUBE(N);
1404   case ISD::MUL:                return visitMUL(N);
1405   case ISD::SDIV:               return visitSDIV(N);
1406   case ISD::UDIV:               return visitUDIV(N);
1407   case ISD::SREM:
1408   case ISD::UREM:               return visitREM(N);
1409   case ISD::MULHU:              return visitMULHU(N);
1410   case ISD::MULHS:              return visitMULHS(N);
1411   case ISD::SMUL_LOHI:          return visitSMUL_LOHI(N);
1412   case ISD::UMUL_LOHI:          return visitUMUL_LOHI(N);
1413   case ISD::SMULO:              return visitSMULO(N);
1414   case ISD::UMULO:              return visitUMULO(N);
1415   case ISD::SMIN:
1416   case ISD::SMAX:
1417   case ISD::UMIN:
1418   case ISD::UMAX:               return visitIMINMAX(N);
1419   case ISD::AND:                return visitAND(N);
1420   case ISD::OR:                 return visitOR(N);
1421   case ISD::XOR:                return visitXOR(N);
1422   case ISD::SHL:                return visitSHL(N);
1423   case ISD::SRA:                return visitSRA(N);
1424   case ISD::SRL:                return visitSRL(N);
1425   case ISD::ROTR:
1426   case ISD::ROTL:               return visitRotate(N);
1427   case ISD::BSWAP:              return visitBSWAP(N);
1428   case ISD::BITREVERSE:         return visitBITREVERSE(N);
1429   case ISD::CTLZ:               return visitCTLZ(N);
1430   case ISD::CTLZ_ZERO_UNDEF:    return visitCTLZ_ZERO_UNDEF(N);
1431   case ISD::CTTZ:               return visitCTTZ(N);
1432   case ISD::CTTZ_ZERO_UNDEF:    return visitCTTZ_ZERO_UNDEF(N);
1433   case ISD::CTPOP:              return visitCTPOP(N);
1434   case ISD::SELECT:             return visitSELECT(N);
1435   case ISD::VSELECT:            return visitVSELECT(N);
1436   case ISD::SELECT_CC:          return visitSELECT_CC(N);
1437   case ISD::SETCC:              return visitSETCC(N);
1438   case ISD::SETCCE:             return visitSETCCE(N);
1439   case ISD::SIGN_EXTEND:        return visitSIGN_EXTEND(N);
1440   case ISD::ZERO_EXTEND:        return visitZERO_EXTEND(N);
1441   case ISD::ANY_EXTEND:         return visitANY_EXTEND(N);
1442   case ISD::SIGN_EXTEND_INREG:  return visitSIGN_EXTEND_INREG(N);
1443   case ISD::SIGN_EXTEND_VECTOR_INREG: return visitSIGN_EXTEND_VECTOR_INREG(N);
1444   case ISD::ZERO_EXTEND_VECTOR_INREG: return visitZERO_EXTEND_VECTOR_INREG(N);
1445   case ISD::TRUNCATE:           return visitTRUNCATE(N);
1446   case ISD::BITCAST:            return visitBITCAST(N);
1447   case ISD::BUILD_PAIR:         return visitBUILD_PAIR(N);
1448   case ISD::FADD:               return visitFADD(N);
1449   case ISD::FSUB:               return visitFSUB(N);
1450   case ISD::FMUL:               return visitFMUL(N);
1451   case ISD::FMA:                return visitFMA(N);
1452   case ISD::FDIV:               return visitFDIV(N);
1453   case ISD::FREM:               return visitFREM(N);
1454   case ISD::FSQRT:              return visitFSQRT(N);
1455   case ISD::FCOPYSIGN:          return visitFCOPYSIGN(N);
1456   case ISD::SINT_TO_FP:         return visitSINT_TO_FP(N);
1457   case ISD::UINT_TO_FP:         return visitUINT_TO_FP(N);
1458   case ISD::FP_TO_SINT:         return visitFP_TO_SINT(N);
1459   case ISD::FP_TO_UINT:         return visitFP_TO_UINT(N);
1460   case ISD::FP_ROUND:           return visitFP_ROUND(N);
1461   case ISD::FP_ROUND_INREG:     return visitFP_ROUND_INREG(N);
1462   case ISD::FP_EXTEND:          return visitFP_EXTEND(N);
1463   case ISD::FNEG:               return visitFNEG(N);
1464   case ISD::FABS:               return visitFABS(N);
1465   case ISD::FFLOOR:             return visitFFLOOR(N);
1466   case ISD::FMINNUM:            return visitFMINNUM(N);
1467   case ISD::FMAXNUM:            return visitFMAXNUM(N);
1468   case ISD::FCEIL:              return visitFCEIL(N);
1469   case ISD::FTRUNC:             return visitFTRUNC(N);
1470   case ISD::BRCOND:             return visitBRCOND(N);
1471   case ISD::BR_CC:              return visitBR_CC(N);
1472   case ISD::LOAD:               return visitLOAD(N);
1473   case ISD::STORE:              return visitSTORE(N);
1474   case ISD::INSERT_VECTOR_ELT:  return visitINSERT_VECTOR_ELT(N);
1475   case ISD::EXTRACT_VECTOR_ELT: return visitEXTRACT_VECTOR_ELT(N);
1476   case ISD::BUILD_VECTOR:       return visitBUILD_VECTOR(N);
1477   case ISD::CONCAT_VECTORS:     return visitCONCAT_VECTORS(N);
1478   case ISD::EXTRACT_SUBVECTOR:  return visitEXTRACT_SUBVECTOR(N);
1479   case ISD::VECTOR_SHUFFLE:     return visitVECTOR_SHUFFLE(N);
1480   case ISD::SCALAR_TO_VECTOR:   return visitSCALAR_TO_VECTOR(N);
1481   case ISD::INSERT_SUBVECTOR:   return visitINSERT_SUBVECTOR(N);
1482   case ISD::MGATHER:            return visitMGATHER(N);
1483   case ISD::MLOAD:              return visitMLOAD(N);
1484   case ISD::MSCATTER:           return visitMSCATTER(N);
1485   case ISD::MSTORE:             return visitMSTORE(N);
1486   case ISD::FP_TO_FP16:         return visitFP_TO_FP16(N);
1487   case ISD::FP16_TO_FP:         return visitFP16_TO_FP(N);
1488   }
1489   return SDValue();
1490 }
1491 
1492 SDValue DAGCombiner::combine(SDNode *N) {
1493   SDValue RV = visit(N);
1494 
1495   // If nothing happened, try a target-specific DAG combine.
1496   if (!RV.getNode()) {
1497     assert(N->getOpcode() != ISD::DELETED_NODE &&
1498            "Node was deleted but visit returned NULL!");
1499 
1500     if (N->getOpcode() >= ISD::BUILTIN_OP_END ||
1501         TLI.hasTargetDAGCombine((ISD::NodeType)N->getOpcode())) {
1502 
1503       // Expose the DAG combiner to the target combiner impls.
1504       TargetLowering::DAGCombinerInfo
1505         DagCombineInfo(DAG, Level, false, this);
1506 
1507       RV = TLI.PerformDAGCombine(N, DagCombineInfo);
1508     }
1509   }
1510 
1511   // If nothing happened still, try promoting the operation.
1512   if (!RV.getNode()) {
1513     switch (N->getOpcode()) {
1514     default: break;
1515     case ISD::ADD:
1516     case ISD::SUB:
1517     case ISD::MUL:
1518     case ISD::AND:
1519     case ISD::OR:
1520     case ISD::XOR:
1521       RV = PromoteIntBinOp(SDValue(N, 0));
1522       break;
1523     case ISD::SHL:
1524     case ISD::SRA:
1525     case ISD::SRL:
1526       RV = PromoteIntShiftOp(SDValue(N, 0));
1527       break;
1528     case ISD::SIGN_EXTEND:
1529     case ISD::ZERO_EXTEND:
1530     case ISD::ANY_EXTEND:
1531       RV = PromoteExtend(SDValue(N, 0));
1532       break;
1533     case ISD::LOAD:
1534       if (PromoteLoad(SDValue(N, 0)))
1535         RV = SDValue(N, 0);
1536       break;
1537     }
1538   }
1539 
1540   // If N is a commutative binary node, try commuting it to enable more
1541   // sdisel CSE.
1542   if (!RV.getNode() && SelectionDAG::isCommutativeBinOp(N->getOpcode()) &&
1543       N->getNumValues() == 1) {
1544     SDValue N0 = N->getOperand(0);
1545     SDValue N1 = N->getOperand(1);
1546 
1547     // Constant operands are canonicalized to RHS.
1548     if (isa<ConstantSDNode>(N0) || !isa<ConstantSDNode>(N1)) {
1549       SDValue Ops[] = {N1, N0};
1550       SDNode *CSENode = DAG.getNodeIfExists(N->getOpcode(), N->getVTList(), Ops,
1551                                             N->getFlags());
1552       if (CSENode)
1553         return SDValue(CSENode, 0);
1554     }
1555   }
1556 
1557   return RV;
1558 }
1559 
1560 /// Given a node, return its input chain if it has one, otherwise return a null
1561 /// sd operand.
1562 static SDValue getInputChainForNode(SDNode *N) {
1563   if (unsigned NumOps = N->getNumOperands()) {
1564     if (N->getOperand(0).getValueType() == MVT::Other)
1565       return N->getOperand(0);
1566     if (N->getOperand(NumOps-1).getValueType() == MVT::Other)
1567       return N->getOperand(NumOps-1);
1568     for (unsigned i = 1; i < NumOps-1; ++i)
1569       if (N->getOperand(i).getValueType() == MVT::Other)
1570         return N->getOperand(i);
1571   }
1572   return SDValue();
1573 }
1574 
1575 SDValue DAGCombiner::visitTokenFactor(SDNode *N) {
1576   // If N has two operands, where one has an input chain equal to the other,
1577   // the 'other' chain is redundant.
1578   if (N->getNumOperands() == 2) {
1579     if (getInputChainForNode(N->getOperand(0).getNode()) == N->getOperand(1))
1580       return N->getOperand(0);
1581     if (getInputChainForNode(N->getOperand(1).getNode()) == N->getOperand(0))
1582       return N->getOperand(1);
1583   }
1584 
1585   SmallVector<SDNode *, 8> TFs;     // List of token factors to visit.
1586   SmallVector<SDValue, 8> Ops;    // Ops for replacing token factor.
1587   SmallPtrSet<SDNode*, 16> SeenOps;
1588   bool Changed = false;             // If we should replace this token factor.
1589 
1590   // Start out with this token factor.
1591   TFs.push_back(N);
1592 
1593   // Iterate through token factors.  The TFs grows when new token factors are
1594   // encountered.
1595   for (unsigned i = 0; i < TFs.size(); ++i) {
1596     SDNode *TF = TFs[i];
1597 
1598     // Check each of the operands.
1599     for (const SDValue &Op : TF->op_values()) {
1600 
1601       switch (Op.getOpcode()) {
1602       case ISD::EntryToken:
1603         // Entry tokens don't need to be added to the list. They are
1604         // redundant.
1605         Changed = true;
1606         break;
1607 
1608       case ISD::TokenFactor:
1609         if (Op.hasOneUse() && !is_contained(TFs, Op.getNode())) {
1610           // Queue up for processing.
1611           TFs.push_back(Op.getNode());
1612           // Clean up in case the token factor is removed.
1613           AddToWorklist(Op.getNode());
1614           Changed = true;
1615           break;
1616         }
1617         LLVM_FALLTHROUGH;
1618 
1619       default:
1620         // Only add if it isn't already in the list.
1621         if (SeenOps.insert(Op.getNode()).second)
1622           Ops.push_back(Op);
1623         else
1624           Changed = true;
1625         break;
1626       }
1627     }
1628   }
1629 
1630   SDValue Result;
1631 
1632   // If we've changed things around then replace token factor.
1633   if (Changed) {
1634     if (Ops.empty()) {
1635       // The entry token is the only possible outcome.
1636       Result = DAG.getEntryNode();
1637     } else {
1638       // New and improved token factor.
1639       Result = DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Ops);
1640     }
1641 
1642     // Add users to worklist if AA is enabled, since it may introduce
1643     // a lot of new chained token factors while removing memory deps.
1644     bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
1645       : DAG.getSubtarget().useAA();
1646     return CombineTo(N, Result, UseAA /*add to worklist*/);
1647   }
1648 
1649   return Result;
1650 }
1651 
1652 /// MERGE_VALUES can always be eliminated.
1653 SDValue DAGCombiner::visitMERGE_VALUES(SDNode *N) {
1654   WorklistRemover DeadNodes(*this);
1655   // Replacing results may cause a different MERGE_VALUES to suddenly
1656   // be CSE'd with N, and carry its uses with it. Iterate until no
1657   // uses remain, to ensure that the node can be safely deleted.
1658   // First add the users of this node to the work list so that they
1659   // can be tried again once they have new operands.
1660   AddUsersToWorklist(N);
1661   do {
1662     for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i)
1663       DAG.ReplaceAllUsesOfValueWith(SDValue(N, i), N->getOperand(i));
1664   } while (!N->use_empty());
1665   deleteAndRecombine(N);
1666   return SDValue(N, 0);   // Return N so it doesn't get rechecked!
1667 }
1668 
1669 /// If \p N is a ConstantSDNode with isOpaque() == false return it casted to a
1670 /// ConstantSDNode pointer else nullptr.
1671 static ConstantSDNode *getAsNonOpaqueConstant(SDValue N) {
1672   ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N);
1673   return Const != nullptr && !Const->isOpaque() ? Const : nullptr;
1674 }
1675 
1676 SDValue DAGCombiner::visitADD(SDNode *N) {
1677   SDValue N0 = N->getOperand(0);
1678   SDValue N1 = N->getOperand(1);
1679   EVT VT = N0.getValueType();
1680   SDLoc DL(N);
1681 
1682   // fold vector ops
1683   if (VT.isVector()) {
1684     if (SDValue FoldedVOp = SimplifyVBinOp(N))
1685       return FoldedVOp;
1686 
1687     // fold (add x, 0) -> x, vector edition
1688     if (ISD::isBuildVectorAllZeros(N1.getNode()))
1689       return N0;
1690     if (ISD::isBuildVectorAllZeros(N0.getNode()))
1691       return N1;
1692   }
1693 
1694   // fold (add x, undef) -> undef
1695   if (N0.isUndef())
1696     return N0;
1697 
1698   if (N1.isUndef())
1699     return N1;
1700 
1701   if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) {
1702     // canonicalize constant to RHS
1703     if (!DAG.isConstantIntBuildVectorOrConstantInt(N1))
1704       return DAG.getNode(ISD::ADD, DL, VT, N1, N0);
1705     // fold (add c1, c2) -> c1+c2
1706     return DAG.FoldConstantArithmetic(ISD::ADD, DL, VT, N0.getNode(),
1707                                       N1.getNode());
1708   }
1709 
1710   // fold (add x, 0) -> x
1711   if (isNullConstant(N1))
1712     return N0;
1713 
1714   // fold ((c1-A)+c2) -> (c1+c2)-A
1715   if (isConstantOrConstantVector(N1, /* NoOpaque */ true)) {
1716     if (N0.getOpcode() == ISD::SUB)
1717       if (isConstantOrConstantVector(N0.getOperand(0), /* NoOpaque */ true)) {
1718         return DAG.getNode(ISD::SUB, DL, VT,
1719                            DAG.getNode(ISD::ADD, DL, VT, N1, N0.getOperand(0)),
1720                            N0.getOperand(1));
1721       }
1722   }
1723 
1724   // reassociate add
1725   if (SDValue RADD = ReassociateOps(ISD::ADD, DL, N0, N1))
1726     return RADD;
1727 
1728   // fold ((0-A) + B) -> B-A
1729   if (N0.getOpcode() == ISD::SUB &&
1730       isNullConstantOrNullSplatConstant(N0.getOperand(0)))
1731     return DAG.getNode(ISD::SUB, DL, VT, N1, N0.getOperand(1));
1732 
1733   // fold (A + (0-B)) -> A-B
1734   if (N1.getOpcode() == ISD::SUB &&
1735       isNullConstantOrNullSplatConstant(N1.getOperand(0)))
1736     return DAG.getNode(ISD::SUB, DL, VT, N0, N1.getOperand(1));
1737 
1738   // fold (A+(B-A)) -> B
1739   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(1))
1740     return N1.getOperand(0);
1741 
1742   // fold ((B-A)+A) -> B
1743   if (N0.getOpcode() == ISD::SUB && N1 == N0.getOperand(1))
1744     return N0.getOperand(0);
1745 
1746   // fold (A+(B-(A+C))) to (B-C)
1747   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
1748       N0 == N1.getOperand(1).getOperand(0))
1749     return DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(0),
1750                        N1.getOperand(1).getOperand(1));
1751 
1752   // fold (A+(B-(C+A))) to (B-C)
1753   if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD &&
1754       N0 == N1.getOperand(1).getOperand(1))
1755     return DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(0),
1756                        N1.getOperand(1).getOperand(0));
1757 
1758   // fold (A+((B-A)+or-C)) to (B+or-C)
1759   if ((N1.getOpcode() == ISD::SUB || N1.getOpcode() == ISD::ADD) &&
1760       N1.getOperand(0).getOpcode() == ISD::SUB &&
1761       N0 == N1.getOperand(0).getOperand(1))
1762     return DAG.getNode(N1.getOpcode(), DL, VT, N1.getOperand(0).getOperand(0),
1763                        N1.getOperand(1));
1764 
1765   // fold (A-B)+(C-D) to (A+C)-(B+D) when A or C is constant
1766   if (N0.getOpcode() == ISD::SUB && N1.getOpcode() == ISD::SUB) {
1767     SDValue N00 = N0.getOperand(0);
1768     SDValue N01 = N0.getOperand(1);
1769     SDValue N10 = N1.getOperand(0);
1770     SDValue N11 = N1.getOperand(1);
1771 
1772     if (isConstantOrConstantVector(N00) || isConstantOrConstantVector(N10))
1773       return DAG.getNode(ISD::SUB, DL, VT,
1774                          DAG.getNode(ISD::ADD, SDLoc(N0), VT, N00, N10),
1775                          DAG.getNode(ISD::ADD, SDLoc(N1), VT, N01, N11));
1776   }
1777 
1778   if (SimplifyDemandedBits(SDValue(N, 0)))
1779     return SDValue(N, 0);
1780 
1781   // fold (a+b) -> (a|b) iff a and b share no bits.
1782   if ((!LegalOperations || TLI.isOperationLegal(ISD::OR, VT)) &&
1783       VT.isInteger() && DAG.haveNoCommonBitsSet(N0, N1))
1784     return DAG.getNode(ISD::OR, DL, VT, N0, N1);
1785 
1786   if (SDValue Combined = visitADDLike(N0, N1, N))
1787     return Combined;
1788 
1789   if (SDValue Combined = visitADDLike(N1, N0, N))
1790     return Combined;
1791 
1792   return SDValue();
1793 }
1794 
1795 SDValue DAGCombiner::visitADDLike(SDValue N0, SDValue N1, SDNode *LocReference) {
1796   EVT VT = N0.getValueType();
1797   SDLoc DL(LocReference);
1798 
1799   // fold (add x, shl(0 - y, n)) -> sub(x, shl(y, n))
1800   if (N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::SUB &&
1801       isNullConstantOrNullSplatConstant(N1.getOperand(0).getOperand(0)))
1802     return DAG.getNode(ISD::SUB, DL, VT, N0,
1803                        DAG.getNode(ISD::SHL, DL, VT,
1804                                    N1.getOperand(0).getOperand(1),
1805                                    N1.getOperand(1)));
1806 
1807   if (N1.getOpcode() == ISD::AND) {
1808     SDValue AndOp0 = N1.getOperand(0);
1809     unsigned NumSignBits = DAG.ComputeNumSignBits(AndOp0);
1810     unsigned DestBits = VT.getScalarSizeInBits();
1811 
1812     // (add z, (and (sbbl x, x), 1)) -> (sub z, (sbbl x, x))
1813     // and similar xforms where the inner op is either ~0 or 0.
1814     if (NumSignBits == DestBits &&
1815         isOneConstantOrOneSplatConstant(N1->getOperand(1)))
1816       return DAG.getNode(ISD::SUB, DL, VT, N0, AndOp0);
1817   }
1818 
1819   // add (sext i1), X -> sub X, (zext i1)
1820   if (N0.getOpcode() == ISD::SIGN_EXTEND &&
1821       N0.getOperand(0).getValueType() == MVT::i1 &&
1822       !TLI.isOperationLegal(ISD::SIGN_EXTEND, MVT::i1)) {
1823     SDValue ZExt = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0));
1824     return DAG.getNode(ISD::SUB, DL, VT, N1, ZExt);
1825   }
1826 
1827   // add X, (sextinreg Y i1) -> sub X, (and Y 1)
1828   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
1829     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
1830     if (TN->getVT() == MVT::i1) {
1831       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
1832                                  DAG.getConstant(1, DL, VT));
1833       return DAG.getNode(ISD::SUB, DL, VT, N0, ZExt);
1834     }
1835   }
1836 
1837   return SDValue();
1838 }
1839 
1840 SDValue DAGCombiner::visitADDC(SDNode *N) {
1841   SDValue N0 = N->getOperand(0);
1842   SDValue N1 = N->getOperand(1);
1843   EVT VT = N0.getValueType();
1844   SDLoc DL(N);
1845 
1846   // If the flag result is dead, turn this into an ADD.
1847   if (!N->hasAnyUseOfValue(1))
1848     return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1),
1849                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
1850 
1851   // canonicalize constant to RHS.
1852   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
1853   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
1854   if (N0C && !N1C)
1855     return DAG.getNode(ISD::ADDC, DL, N->getVTList(), N1, N0);
1856 
1857   // fold (addc x, 0) -> x + no carry out
1858   if (isNullConstant(N1))
1859     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE,
1860                                         DL, MVT::Glue));
1861 
1862   // If it cannot overflow, transform into an add.
1863   if (DAG.computeOverflowKind(N0, N1) == SelectionDAG::OFK_Never)
1864     return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1),
1865                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
1866 
1867   return SDValue();
1868 }
1869 
1870 SDValue DAGCombiner::visitADDE(SDNode *N) {
1871   SDValue N0 = N->getOperand(0);
1872   SDValue N1 = N->getOperand(1);
1873   SDValue CarryIn = N->getOperand(2);
1874 
1875   // canonicalize constant to RHS
1876   ConstantSDNode *N0C = dyn_cast<ConstantSDNode>(N0);
1877   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
1878   if (N0C && !N1C)
1879     return DAG.getNode(ISD::ADDE, SDLoc(N), N->getVTList(),
1880                        N1, N0, CarryIn);
1881 
1882   // fold (adde x, y, false) -> (addc x, y)
1883   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
1884     return DAG.getNode(ISD::ADDC, SDLoc(N), N->getVTList(), N0, N1);
1885 
1886   return SDValue();
1887 }
1888 
1889 // Since it may not be valid to emit a fold to zero for vector initializers
1890 // check if we can before folding.
1891 static SDValue tryFoldToZero(const SDLoc &DL, const TargetLowering &TLI, EVT VT,
1892                              SelectionDAG &DAG, bool LegalOperations,
1893                              bool LegalTypes) {
1894   if (!VT.isVector())
1895     return DAG.getConstant(0, DL, VT);
1896   if (!LegalOperations || TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
1897     return DAG.getConstant(0, DL, VT);
1898   return SDValue();
1899 }
1900 
1901 SDValue DAGCombiner::visitSUB(SDNode *N) {
1902   SDValue N0 = N->getOperand(0);
1903   SDValue N1 = N->getOperand(1);
1904   EVT VT = N0.getValueType();
1905   SDLoc DL(N);
1906 
1907   // fold vector ops
1908   if (VT.isVector()) {
1909     if (SDValue FoldedVOp = SimplifyVBinOp(N))
1910       return FoldedVOp;
1911 
1912     // fold (sub x, 0) -> x, vector edition
1913     if (ISD::isBuildVectorAllZeros(N1.getNode()))
1914       return N0;
1915   }
1916 
1917   // fold (sub x, x) -> 0
1918   // FIXME: Refactor this and xor and other similar operations together.
1919   if (N0 == N1)
1920     return tryFoldToZero(DL, TLI, VT, DAG, LegalOperations, LegalTypes);
1921   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
1922       DAG.isConstantIntBuildVectorOrConstantInt(N1)) {
1923     // fold (sub c1, c2) -> c1-c2
1924     return DAG.FoldConstantArithmetic(ISD::SUB, DL, VT, N0.getNode(),
1925                                       N1.getNode());
1926   }
1927 
1928   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
1929 
1930   // fold (sub x, c) -> (add x, -c)
1931   if (N1C) {
1932     return DAG.getNode(ISD::ADD, DL, VT, N0,
1933                        DAG.getConstant(-N1C->getAPIntValue(), DL, VT));
1934   }
1935 
1936   if (isNullConstantOrNullSplatConstant(N0)) {
1937     unsigned BitWidth = VT.getScalarSizeInBits();
1938     // Right-shifting everything out but the sign bit followed by negation is
1939     // the same as flipping arithmetic/logical shift type without the negation:
1940     // -(X >>u 31) -> (X >>s 31)
1941     // -(X >>s 31) -> (X >>u 31)
1942     if (N1->getOpcode() == ISD::SRA || N1->getOpcode() == ISD::SRL) {
1943       ConstantSDNode *ShiftAmt = isConstOrConstSplat(N1.getOperand(1));
1944       if (ShiftAmt && ShiftAmt->getZExtValue() == BitWidth - 1) {
1945         auto NewSh = N1->getOpcode() == ISD::SRA ? ISD::SRL : ISD::SRA;
1946         if (!LegalOperations || TLI.isOperationLegal(NewSh, VT))
1947           return DAG.getNode(NewSh, DL, VT, N1.getOperand(0), N1.getOperand(1));
1948       }
1949     }
1950 
1951     // 0 - X --> 0 if the sub is NUW.
1952     if (N->getFlags()->hasNoUnsignedWrap())
1953       return N0;
1954 
1955     if (DAG.MaskedValueIsZero(N1, ~APInt::getSignBit(BitWidth))) {
1956       // N1 is either 0 or the minimum signed value. If the sub is NSW, then
1957       // N1 must be 0 because negating the minimum signed value is undefined.
1958       if (N->getFlags()->hasNoSignedWrap())
1959         return N0;
1960 
1961       // 0 - X --> X if X is 0 or the minimum signed value.
1962       return N1;
1963     }
1964   }
1965 
1966   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1)
1967   if (isAllOnesConstantOrAllOnesSplatConstant(N0))
1968     return DAG.getNode(ISD::XOR, DL, VT, N1, N0);
1969 
1970   // fold A-(A-B) -> B
1971   if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(0))
1972     return N1.getOperand(1);
1973 
1974   // fold (A+B)-A -> B
1975   if (N0.getOpcode() == ISD::ADD && N0.getOperand(0) == N1)
1976     return N0.getOperand(1);
1977 
1978   // fold (A+B)-B -> A
1979   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1) == N1)
1980     return N0.getOperand(0);
1981 
1982   // fold C2-(A+C1) -> (C2-C1)-A
1983   if (N1.getOpcode() == ISD::ADD) {
1984     SDValue N11 = N1.getOperand(1);
1985     if (isConstantOrConstantVector(N0, /* NoOpaques */ true) &&
1986         isConstantOrConstantVector(N11, /* NoOpaques */ true)) {
1987       SDValue NewC = DAG.getNode(ISD::SUB, DL, VT, N0, N11);
1988       return DAG.getNode(ISD::SUB, DL, VT, NewC, N1.getOperand(0));
1989     }
1990   }
1991 
1992   // fold ((A+(B+or-C))-B) -> A+or-C
1993   if (N0.getOpcode() == ISD::ADD &&
1994       (N0.getOperand(1).getOpcode() == ISD::SUB ||
1995        N0.getOperand(1).getOpcode() == ISD::ADD) &&
1996       N0.getOperand(1).getOperand(0) == N1)
1997     return DAG.getNode(N0.getOperand(1).getOpcode(), DL, VT, N0.getOperand(0),
1998                        N0.getOperand(1).getOperand(1));
1999 
2000   // fold ((A+(C+B))-B) -> A+C
2001   if (N0.getOpcode() == ISD::ADD && N0.getOperand(1).getOpcode() == ISD::ADD &&
2002       N0.getOperand(1).getOperand(1) == N1)
2003     return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0),
2004                        N0.getOperand(1).getOperand(0));
2005 
2006   // fold ((A-(B-C))-C) -> A-B
2007   if (N0.getOpcode() == ISD::SUB && N0.getOperand(1).getOpcode() == ISD::SUB &&
2008       N0.getOperand(1).getOperand(1) == N1)
2009     return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0),
2010                        N0.getOperand(1).getOperand(0));
2011 
2012   // If either operand of a sub is undef, the result is undef
2013   if (N0.isUndef())
2014     return N0;
2015   if (N1.isUndef())
2016     return N1;
2017 
2018   // If the relocation model supports it, consider symbol offsets.
2019   if (GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(N0))
2020     if (!LegalOperations && TLI.isOffsetFoldingLegal(GA)) {
2021       // fold (sub Sym, c) -> Sym-c
2022       if (N1C && GA->getOpcode() == ISD::GlobalAddress)
2023         return DAG.getGlobalAddress(GA->getGlobal(), SDLoc(N1C), VT,
2024                                     GA->getOffset() -
2025                                         (uint64_t)N1C->getSExtValue());
2026       // fold (sub Sym+c1, Sym+c2) -> c1-c2
2027       if (GlobalAddressSDNode *GB = dyn_cast<GlobalAddressSDNode>(N1))
2028         if (GA->getGlobal() == GB->getGlobal())
2029           return DAG.getConstant((uint64_t)GA->getOffset() - GB->getOffset(),
2030                                  DL, VT);
2031     }
2032 
2033   // sub X, (sextinreg Y i1) -> add X, (and Y 1)
2034   if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) {
2035     VTSDNode *TN = cast<VTSDNode>(N1.getOperand(1));
2036     if (TN->getVT() == MVT::i1) {
2037       SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0),
2038                                  DAG.getConstant(1, DL, VT));
2039       return DAG.getNode(ISD::ADD, DL, VT, N0, ZExt);
2040     }
2041   }
2042 
2043   return SDValue();
2044 }
2045 
2046 SDValue DAGCombiner::visitSUBC(SDNode *N) {
2047   SDValue N0 = N->getOperand(0);
2048   SDValue N1 = N->getOperand(1);
2049   EVT VT = N0.getValueType();
2050   SDLoc DL(N);
2051 
2052   // If the flag result is dead, turn this into an SUB.
2053   if (!N->hasAnyUseOfValue(1))
2054     return CombineTo(N, DAG.getNode(ISD::SUB, DL, VT, N0, N1),
2055                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2056 
2057   // fold (subc x, x) -> 0 + no borrow
2058   if (N0 == N1)
2059     return CombineTo(N, DAG.getConstant(0, DL, VT),
2060                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2061 
2062   // fold (subc x, 0) -> x + no borrow
2063   if (isNullConstant(N1))
2064     return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2065 
2066   // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1) + no borrow
2067   if (isAllOnesConstant(N0))
2068     return CombineTo(N, DAG.getNode(ISD::XOR, DL, VT, N1, N0),
2069                      DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue));
2070 
2071   return SDValue();
2072 }
2073 
2074 SDValue DAGCombiner::visitSUBE(SDNode *N) {
2075   SDValue N0 = N->getOperand(0);
2076   SDValue N1 = N->getOperand(1);
2077   SDValue CarryIn = N->getOperand(2);
2078 
2079   // fold (sube x, y, false) -> (subc x, y)
2080   if (CarryIn.getOpcode() == ISD::CARRY_FALSE)
2081     return DAG.getNode(ISD::SUBC, SDLoc(N), N->getVTList(), N0, N1);
2082 
2083   return SDValue();
2084 }
2085 
2086 SDValue DAGCombiner::visitMUL(SDNode *N) {
2087   SDValue N0 = N->getOperand(0);
2088   SDValue N1 = N->getOperand(1);
2089   EVT VT = N0.getValueType();
2090 
2091   // fold (mul x, undef) -> 0
2092   if (N0.isUndef() || N1.isUndef())
2093     return DAG.getConstant(0, SDLoc(N), VT);
2094 
2095   bool N0IsConst = false;
2096   bool N1IsConst = false;
2097   bool N1IsOpaqueConst = false;
2098   bool N0IsOpaqueConst = false;
2099   APInt ConstValue0, ConstValue1;
2100   // fold vector ops
2101   if (VT.isVector()) {
2102     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2103       return FoldedVOp;
2104 
2105     N0IsConst = ISD::isConstantSplatVector(N0.getNode(), ConstValue0);
2106     N1IsConst = ISD::isConstantSplatVector(N1.getNode(), ConstValue1);
2107   } else {
2108     N0IsConst = isa<ConstantSDNode>(N0);
2109     if (N0IsConst) {
2110       ConstValue0 = cast<ConstantSDNode>(N0)->getAPIntValue();
2111       N0IsOpaqueConst = cast<ConstantSDNode>(N0)->isOpaque();
2112     }
2113     N1IsConst = isa<ConstantSDNode>(N1);
2114     if (N1IsConst) {
2115       ConstValue1 = cast<ConstantSDNode>(N1)->getAPIntValue();
2116       N1IsOpaqueConst = cast<ConstantSDNode>(N1)->isOpaque();
2117     }
2118   }
2119 
2120   // fold (mul c1, c2) -> c1*c2
2121   if (N0IsConst && N1IsConst && !N0IsOpaqueConst && !N1IsOpaqueConst)
2122     return DAG.FoldConstantArithmetic(ISD::MUL, SDLoc(N), VT,
2123                                       N0.getNode(), N1.getNode());
2124 
2125   // canonicalize constant to RHS (vector doesn't have to splat)
2126   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2127      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
2128     return DAG.getNode(ISD::MUL, SDLoc(N), VT, N1, N0);
2129   // fold (mul x, 0) -> 0
2130   if (N1IsConst && ConstValue1 == 0)
2131     return N1;
2132   // We require a splat of the entire scalar bit width for non-contiguous
2133   // bit patterns.
2134   bool IsFullSplat =
2135     ConstValue1.getBitWidth() == VT.getScalarSizeInBits();
2136   // fold (mul x, 1) -> x
2137   if (N1IsConst && ConstValue1 == 1 && IsFullSplat)
2138     return N0;
2139   // fold (mul x, -1) -> 0-x
2140   if (N1IsConst && ConstValue1.isAllOnesValue()) {
2141     SDLoc DL(N);
2142     return DAG.getNode(ISD::SUB, DL, VT,
2143                        DAG.getConstant(0, DL, VT), N0);
2144   }
2145   // fold (mul x, (1 << c)) -> x << c
2146   if (N1IsConst && !N1IsOpaqueConst && ConstValue1.isPowerOf2() &&
2147       IsFullSplat) {
2148     SDLoc DL(N);
2149     return DAG.getNode(ISD::SHL, DL, VT, N0,
2150                        DAG.getConstant(ConstValue1.logBase2(), DL,
2151                                        getShiftAmountTy(N0.getValueType())));
2152   }
2153   // fold (mul x, -(1 << c)) -> -(x << c) or (-x) << c
2154   if (N1IsConst && !N1IsOpaqueConst && (-ConstValue1).isPowerOf2() &&
2155       IsFullSplat) {
2156     unsigned Log2Val = (-ConstValue1).logBase2();
2157     SDLoc DL(N);
2158     // FIXME: If the input is something that is easily negated (e.g. a
2159     // single-use add), we should put the negate there.
2160     return DAG.getNode(ISD::SUB, DL, VT,
2161                        DAG.getConstant(0, DL, VT),
2162                        DAG.getNode(ISD::SHL, DL, VT, N0,
2163                             DAG.getConstant(Log2Val, DL,
2164                                       getShiftAmountTy(N0.getValueType()))));
2165   }
2166 
2167   // (mul (shl X, c1), c2) -> (mul X, c2 << c1)
2168   if (N0.getOpcode() == ISD::SHL &&
2169       isConstantOrConstantVector(N1, /* NoOpaques */ true) &&
2170       isConstantOrConstantVector(N0.getOperand(1), /* NoOpaques */ true)) {
2171     SDValue C3 = DAG.getNode(ISD::SHL, SDLoc(N), VT, N1, N0.getOperand(1));
2172     if (isConstantOrConstantVector(C3))
2173       return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), C3);
2174   }
2175 
2176   // Change (mul (shl X, C), Y) -> (shl (mul X, Y), C) when the shift has one
2177   // use.
2178   {
2179     SDValue Sh(nullptr, 0), Y(nullptr, 0);
2180 
2181     // Check for both (mul (shl X, C), Y)  and  (mul Y, (shl X, C)).
2182     if (N0.getOpcode() == ISD::SHL &&
2183         isConstantOrConstantVector(N0.getOperand(1)) &&
2184         N0.getNode()->hasOneUse()) {
2185       Sh = N0; Y = N1;
2186     } else if (N1.getOpcode() == ISD::SHL &&
2187                isConstantOrConstantVector(N1.getOperand(1)) &&
2188                N1.getNode()->hasOneUse()) {
2189       Sh = N1; Y = N0;
2190     }
2191 
2192     if (Sh.getNode()) {
2193       SDValue Mul = DAG.getNode(ISD::MUL, SDLoc(N), VT, Sh.getOperand(0), Y);
2194       return DAG.getNode(ISD::SHL, SDLoc(N), VT, Mul, Sh.getOperand(1));
2195     }
2196   }
2197 
2198   // fold (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2)
2199   if (DAG.isConstantIntBuildVectorOrConstantInt(N1) &&
2200       N0.getOpcode() == ISD::ADD &&
2201       DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1)) &&
2202       isMulAddWithConstProfitable(N, N0, N1))
2203       return DAG.getNode(ISD::ADD, SDLoc(N), VT,
2204                          DAG.getNode(ISD::MUL, SDLoc(N0), VT,
2205                                      N0.getOperand(0), N1),
2206                          DAG.getNode(ISD::MUL, SDLoc(N1), VT,
2207                                      N0.getOperand(1), N1));
2208 
2209   // reassociate mul
2210   if (SDValue RMUL = ReassociateOps(ISD::MUL, SDLoc(N), N0, N1))
2211     return RMUL;
2212 
2213   return SDValue();
2214 }
2215 
2216 /// Return true if divmod libcall is available.
2217 static bool isDivRemLibcallAvailable(SDNode *Node, bool isSigned,
2218                                      const TargetLowering &TLI) {
2219   RTLIB::Libcall LC;
2220   EVT NodeType = Node->getValueType(0);
2221   if (!NodeType.isSimple())
2222     return false;
2223   switch (NodeType.getSimpleVT().SimpleTy) {
2224   default: return false; // No libcall for vector types.
2225   case MVT::i8:   LC= isSigned ? RTLIB::SDIVREM_I8  : RTLIB::UDIVREM_I8;  break;
2226   case MVT::i16:  LC= isSigned ? RTLIB::SDIVREM_I16 : RTLIB::UDIVREM_I16; break;
2227   case MVT::i32:  LC= isSigned ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32; break;
2228   case MVT::i64:  LC= isSigned ? RTLIB::SDIVREM_I64 : RTLIB::UDIVREM_I64; break;
2229   case MVT::i128: LC= isSigned ? RTLIB::SDIVREM_I128:RTLIB::UDIVREM_I128; break;
2230   }
2231 
2232   return TLI.getLibcallName(LC) != nullptr;
2233 }
2234 
2235 /// Issue divrem if both quotient and remainder are needed.
2236 SDValue DAGCombiner::useDivRem(SDNode *Node) {
2237   if (Node->use_empty())
2238     return SDValue(); // This is a dead node, leave it alone.
2239 
2240   unsigned Opcode = Node->getOpcode();
2241   bool isSigned = (Opcode == ISD::SDIV) || (Opcode == ISD::SREM);
2242   unsigned DivRemOpc = isSigned ? ISD::SDIVREM : ISD::UDIVREM;
2243 
2244   // DivMod lib calls can still work on non-legal types if using lib-calls.
2245   EVT VT = Node->getValueType(0);
2246   if (VT.isVector() || !VT.isInteger())
2247     return SDValue();
2248 
2249   if (!TLI.isTypeLegal(VT) && !TLI.isOperationCustom(DivRemOpc, VT))
2250     return SDValue();
2251 
2252   // If DIVREM is going to get expanded into a libcall,
2253   // but there is no libcall available, then don't combine.
2254   if (!TLI.isOperationLegalOrCustom(DivRemOpc, VT) &&
2255       !isDivRemLibcallAvailable(Node, isSigned, TLI))
2256     return SDValue();
2257 
2258   // If div is legal, it's better to do the normal expansion
2259   unsigned OtherOpcode = 0;
2260   if ((Opcode == ISD::SDIV) || (Opcode == ISD::UDIV)) {
2261     OtherOpcode = isSigned ? ISD::SREM : ISD::UREM;
2262     if (TLI.isOperationLegalOrCustom(Opcode, VT))
2263       return SDValue();
2264   } else {
2265     OtherOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
2266     if (TLI.isOperationLegalOrCustom(OtherOpcode, VT))
2267       return SDValue();
2268   }
2269 
2270   SDValue Op0 = Node->getOperand(0);
2271   SDValue Op1 = Node->getOperand(1);
2272   SDValue combined;
2273   for (SDNode::use_iterator UI = Op0.getNode()->use_begin(),
2274          UE = Op0.getNode()->use_end(); UI != UE;) {
2275     SDNode *User = *UI++;
2276     if (User == Node || User->use_empty())
2277       continue;
2278     // Convert the other matching node(s), too;
2279     // otherwise, the DIVREM may get target-legalized into something
2280     // target-specific that we won't be able to recognize.
2281     unsigned UserOpc = User->getOpcode();
2282     if ((UserOpc == Opcode || UserOpc == OtherOpcode || UserOpc == DivRemOpc) &&
2283         User->getOperand(0) == Op0 &&
2284         User->getOperand(1) == Op1) {
2285       if (!combined) {
2286         if (UserOpc == OtherOpcode) {
2287           SDVTList VTs = DAG.getVTList(VT, VT);
2288           combined = DAG.getNode(DivRemOpc, SDLoc(Node), VTs, Op0, Op1);
2289         } else if (UserOpc == DivRemOpc) {
2290           combined = SDValue(User, 0);
2291         } else {
2292           assert(UserOpc == Opcode);
2293           continue;
2294         }
2295       }
2296       if (UserOpc == ISD::SDIV || UserOpc == ISD::UDIV)
2297         CombineTo(User, combined);
2298       else if (UserOpc == ISD::SREM || UserOpc == ISD::UREM)
2299         CombineTo(User, combined.getValue(1));
2300     }
2301   }
2302   return combined;
2303 }
2304 
2305 SDValue DAGCombiner::visitSDIV(SDNode *N) {
2306   SDValue N0 = N->getOperand(0);
2307   SDValue N1 = N->getOperand(1);
2308   EVT VT = N->getValueType(0);
2309 
2310   // fold vector ops
2311   if (VT.isVector())
2312     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2313       return FoldedVOp;
2314 
2315   SDLoc DL(N);
2316 
2317   // fold (sdiv c1, c2) -> c1/c2
2318   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2319   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2320   if (N0C && N1C && !N0C->isOpaque() && !N1C->isOpaque())
2321     return DAG.FoldConstantArithmetic(ISD::SDIV, DL, VT, N0C, N1C);
2322   // fold (sdiv X, 1) -> X
2323   if (N1C && N1C->isOne())
2324     return N0;
2325   // fold (sdiv X, -1) -> 0-X
2326   if (N1C && N1C->isAllOnesValue())
2327     return DAG.getNode(ISD::SUB, DL, VT,
2328                        DAG.getConstant(0, DL, VT), N0);
2329 
2330   // If we know the sign bits of both operands are zero, strength reduce to a
2331   // udiv instead.  Handles (X&15) /s 4 -> X&15 >> 2
2332   if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
2333     return DAG.getNode(ISD::UDIV, DL, N1.getValueType(), N0, N1);
2334 
2335   // fold (sdiv X, pow2) -> simple ops after legalize
2336   // FIXME: We check for the exact bit here because the generic lowering gives
2337   // better results in that case. The target-specific lowering should learn how
2338   // to handle exact sdivs efficiently.
2339   if (N1C && !N1C->isNullValue() && !N1C->isOpaque() &&
2340       !cast<BinaryWithFlagsSDNode>(N)->Flags.hasExact() &&
2341       (N1C->getAPIntValue().isPowerOf2() ||
2342        (-N1C->getAPIntValue()).isPowerOf2())) {
2343     // Target-specific implementation of sdiv x, pow2.
2344     if (SDValue Res = BuildSDIVPow2(N))
2345       return Res;
2346 
2347     unsigned lg2 = N1C->getAPIntValue().countTrailingZeros();
2348 
2349     // Splat the sign bit into the register
2350     SDValue SGN =
2351         DAG.getNode(ISD::SRA, DL, VT, N0,
2352                     DAG.getConstant(VT.getScalarSizeInBits() - 1, DL,
2353                                     getShiftAmountTy(N0.getValueType())));
2354     AddToWorklist(SGN.getNode());
2355 
2356     // Add (N0 < 0) ? abs2 - 1 : 0;
2357     SDValue SRL =
2358         DAG.getNode(ISD::SRL, DL, VT, SGN,
2359                     DAG.getConstant(VT.getScalarSizeInBits() - lg2, DL,
2360                                     getShiftAmountTy(SGN.getValueType())));
2361     SDValue ADD = DAG.getNode(ISD::ADD, DL, VT, N0, SRL);
2362     AddToWorklist(SRL.getNode());
2363     AddToWorklist(ADD.getNode());    // Divide by pow2
2364     SDValue SRA = DAG.getNode(ISD::SRA, DL, VT, ADD,
2365                   DAG.getConstant(lg2, DL,
2366                                   getShiftAmountTy(ADD.getValueType())));
2367 
2368     // If we're dividing by a positive value, we're done.  Otherwise, we must
2369     // negate the result.
2370     if (N1C->getAPIntValue().isNonNegative())
2371       return SRA;
2372 
2373     AddToWorklist(SRA.getNode());
2374     return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), SRA);
2375   }
2376 
2377   // If integer divide is expensive and we satisfy the requirements, emit an
2378   // alternate sequence.  Targets may check function attributes for size/speed
2379   // trade-offs.
2380   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2381   if (N1C && !TLI.isIntDivCheap(N->getValueType(0), Attr))
2382     if (SDValue Op = BuildSDIV(N))
2383       return Op;
2384 
2385   // sdiv, srem -> sdivrem
2386   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is
2387   // true.  Otherwise, we break the simplification logic in visitREM().
2388   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
2389     if (SDValue DivRem = useDivRem(N))
2390         return DivRem;
2391 
2392   // undef / X -> 0
2393   if (N0.isUndef())
2394     return DAG.getConstant(0, DL, VT);
2395   // X / undef -> undef
2396   if (N1.isUndef())
2397     return N1;
2398 
2399   return SDValue();
2400 }
2401 
2402 SDValue DAGCombiner::visitUDIV(SDNode *N) {
2403   SDValue N0 = N->getOperand(0);
2404   SDValue N1 = N->getOperand(1);
2405   EVT VT = N->getValueType(0);
2406 
2407   // fold vector ops
2408   if (VT.isVector())
2409     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2410       return FoldedVOp;
2411 
2412   SDLoc DL(N);
2413 
2414   // fold (udiv c1, c2) -> c1/c2
2415   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2416   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2417   if (N0C && N1C)
2418     if (SDValue Folded = DAG.FoldConstantArithmetic(ISD::UDIV, DL, VT,
2419                                                     N0C, N1C))
2420       return Folded;
2421 
2422   // fold (udiv x, (1 << c)) -> x >>u c
2423   if (isConstantOrConstantVector(N1, /*NoOpaques*/ true) &&
2424       DAG.isKnownToBeAPowerOfTwo(N1)) {
2425     SDValue LogBase2 = BuildLogBase2(N1, DL);
2426     AddToWorklist(LogBase2.getNode());
2427 
2428     EVT ShiftVT = getShiftAmountTy(N0.getValueType());
2429     SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ShiftVT);
2430     AddToWorklist(Trunc.getNode());
2431     return DAG.getNode(ISD::SRL, DL, VT, N0, Trunc);
2432   }
2433 
2434   // fold (udiv x, (shl c, y)) -> x >>u (log2(c)+y) iff c is power of 2
2435   if (N1.getOpcode() == ISD::SHL) {
2436     SDValue N10 = N1.getOperand(0);
2437     if (isConstantOrConstantVector(N10, /*NoOpaques*/ true) &&
2438         DAG.isKnownToBeAPowerOfTwo(N10)) {
2439       SDValue LogBase2 = BuildLogBase2(N10, DL);
2440       AddToWorklist(LogBase2.getNode());
2441 
2442       EVT ADDVT = N1.getOperand(1).getValueType();
2443       SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ADDVT);
2444       AddToWorklist(Trunc.getNode());
2445       SDValue Add = DAG.getNode(ISD::ADD, DL, ADDVT, N1.getOperand(1), Trunc);
2446       AddToWorklist(Add.getNode());
2447       return DAG.getNode(ISD::SRL, DL, VT, N0, Add);
2448     }
2449   }
2450 
2451   // fold (udiv x, c) -> alternate
2452   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2453   if (N1C && !TLI.isIntDivCheap(N->getValueType(0), Attr))
2454     if (SDValue Op = BuildUDIV(N))
2455       return Op;
2456 
2457   // sdiv, srem -> sdivrem
2458   // If the divisor is constant, then return DIVREM only if isIntDivCheap() is
2459   // true.  Otherwise, we break the simplification logic in visitREM().
2460   if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr))
2461     if (SDValue DivRem = useDivRem(N))
2462         return DivRem;
2463 
2464   // undef / X -> 0
2465   if (N0.isUndef())
2466     return DAG.getConstant(0, DL, VT);
2467   // X / undef -> undef
2468   if (N1.isUndef())
2469     return N1;
2470 
2471   return SDValue();
2472 }
2473 
2474 // handles ISD::SREM and ISD::UREM
2475 SDValue DAGCombiner::visitREM(SDNode *N) {
2476   unsigned Opcode = N->getOpcode();
2477   SDValue N0 = N->getOperand(0);
2478   SDValue N1 = N->getOperand(1);
2479   EVT VT = N->getValueType(0);
2480   bool isSigned = (Opcode == ISD::SREM);
2481   SDLoc DL(N);
2482 
2483   // fold (rem c1, c2) -> c1%c2
2484   ConstantSDNode *N0C = isConstOrConstSplat(N0);
2485   ConstantSDNode *N1C = isConstOrConstSplat(N1);
2486   if (N0C && N1C)
2487     if (SDValue Folded = DAG.FoldConstantArithmetic(Opcode, DL, VT, N0C, N1C))
2488       return Folded;
2489 
2490   if (isSigned) {
2491     // If we know the sign bits of both operands are zero, strength reduce to a
2492     // urem instead.  Handles (X & 0x0FFFFFFF) %s 16 -> X&15
2493     if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0))
2494       return DAG.getNode(ISD::UREM, DL, VT, N0, N1);
2495   } else {
2496     SDValue NegOne = DAG.getAllOnesConstant(DL, VT);
2497     if (DAG.isKnownToBeAPowerOfTwo(N1)) {
2498       // fold (urem x, pow2) -> (and x, pow2-1)
2499       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N1, NegOne);
2500       AddToWorklist(Add.getNode());
2501       return DAG.getNode(ISD::AND, DL, VT, N0, Add);
2502     }
2503     if (N1.getOpcode() == ISD::SHL &&
2504         DAG.isKnownToBeAPowerOfTwo(N1.getOperand(0))) {
2505       // fold (urem x, (shl pow2, y)) -> (and x, (add (shl pow2, y), -1))
2506       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N1, NegOne);
2507       AddToWorklist(Add.getNode());
2508       return DAG.getNode(ISD::AND, DL, VT, N0, Add);
2509     }
2510   }
2511 
2512   AttributeSet Attr = DAG.getMachineFunction().getFunction()->getAttributes();
2513 
2514   // If X/C can be simplified by the division-by-constant logic, lower
2515   // X%C to the equivalent of X-X/C*C.
2516   // To avoid mangling nodes, this simplification requires that the combine()
2517   // call for the speculative DIV must not cause a DIVREM conversion.  We guard
2518   // against this by skipping the simplification if isIntDivCheap().  When
2519   // div is not cheap, combine will not return a DIVREM.  Regardless,
2520   // checking cheapness here makes sense since the simplification results in
2521   // fatter code.
2522   if (N1C && !N1C->isNullValue() && !TLI.isIntDivCheap(VT, Attr)) {
2523     unsigned DivOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
2524     SDValue Div = DAG.getNode(DivOpcode, DL, VT, N0, N1);
2525     AddToWorklist(Div.getNode());
2526     SDValue OptimizedDiv = combine(Div.getNode());
2527     if (OptimizedDiv.getNode() && OptimizedDiv.getNode() != Div.getNode()) {
2528       assert((OptimizedDiv.getOpcode() != ISD::UDIVREM) &&
2529              (OptimizedDiv.getOpcode() != ISD::SDIVREM));
2530       SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, OptimizedDiv, N1);
2531       SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul);
2532       AddToWorklist(Mul.getNode());
2533       return Sub;
2534     }
2535   }
2536 
2537   // sdiv, srem -> sdivrem
2538   if (SDValue DivRem = useDivRem(N))
2539     return DivRem.getValue(1);
2540 
2541   // undef % X -> 0
2542   if (N0.isUndef())
2543     return DAG.getConstant(0, DL, VT);
2544   // X % undef -> undef
2545   if (N1.isUndef())
2546     return N1;
2547 
2548   return SDValue();
2549 }
2550 
2551 SDValue DAGCombiner::visitMULHS(SDNode *N) {
2552   SDValue N0 = N->getOperand(0);
2553   SDValue N1 = N->getOperand(1);
2554   EVT VT = N->getValueType(0);
2555   SDLoc DL(N);
2556 
2557   // fold (mulhs x, 0) -> 0
2558   if (isNullConstant(N1))
2559     return N1;
2560   // fold (mulhs x, 1) -> (sra x, size(x)-1)
2561   if (isOneConstant(N1)) {
2562     SDLoc DL(N);
2563     return DAG.getNode(ISD::SRA, DL, N0.getValueType(), N0,
2564                        DAG.getConstant(N0.getValueSizeInBits() - 1, DL,
2565                                        getShiftAmountTy(N0.getValueType())));
2566   }
2567   // fold (mulhs x, undef) -> 0
2568   if (N0.isUndef() || N1.isUndef())
2569     return DAG.getConstant(0, SDLoc(N), VT);
2570 
2571   // If the type twice as wide is legal, transform the mulhs to a wider multiply
2572   // plus a shift.
2573   if (VT.isSimple() && !VT.isVector()) {
2574     MVT Simple = VT.getSimpleVT();
2575     unsigned SimpleSize = Simple.getSizeInBits();
2576     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2577     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2578       N0 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N0);
2579       N1 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N1);
2580       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
2581       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
2582             DAG.getConstant(SimpleSize, DL,
2583                             getShiftAmountTy(N1.getValueType())));
2584       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
2585     }
2586   }
2587 
2588   return SDValue();
2589 }
2590 
2591 SDValue DAGCombiner::visitMULHU(SDNode *N) {
2592   SDValue N0 = N->getOperand(0);
2593   SDValue N1 = N->getOperand(1);
2594   EVT VT = N->getValueType(0);
2595   SDLoc DL(N);
2596 
2597   // fold (mulhu x, 0) -> 0
2598   if (isNullConstant(N1))
2599     return N1;
2600   // fold (mulhu x, 1) -> 0
2601   if (isOneConstant(N1))
2602     return DAG.getConstant(0, DL, N0.getValueType());
2603   // fold (mulhu x, undef) -> 0
2604   if (N0.isUndef() || N1.isUndef())
2605     return DAG.getConstant(0, DL, VT);
2606 
2607   // If the type twice as wide is legal, transform the mulhu to a wider multiply
2608   // plus a shift.
2609   if (VT.isSimple() && !VT.isVector()) {
2610     MVT Simple = VT.getSimpleVT();
2611     unsigned SimpleSize = Simple.getSizeInBits();
2612     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2613     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2614       N0 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N0);
2615       N1 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N1);
2616       N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1);
2617       N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1,
2618             DAG.getConstant(SimpleSize, DL,
2619                             getShiftAmountTy(N1.getValueType())));
2620       return DAG.getNode(ISD::TRUNCATE, DL, VT, N1);
2621     }
2622   }
2623 
2624   return SDValue();
2625 }
2626 
2627 /// Perform optimizations common to nodes that compute two values. LoOp and HiOp
2628 /// give the opcodes for the two computations that are being performed. Return
2629 /// true if a simplification was made.
2630 SDValue DAGCombiner::SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp,
2631                                                 unsigned HiOp) {
2632   // If the high half is not needed, just compute the low half.
2633   bool HiExists = N->hasAnyUseOfValue(1);
2634   if (!HiExists &&
2635       (!LegalOperations ||
2636        TLI.isOperationLegalOrCustom(LoOp, N->getValueType(0)))) {
2637     SDValue Res = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
2638     return CombineTo(N, Res, Res);
2639   }
2640 
2641   // If the low half is not needed, just compute the high half.
2642   bool LoExists = N->hasAnyUseOfValue(0);
2643   if (!LoExists &&
2644       (!LegalOperations ||
2645        TLI.isOperationLegal(HiOp, N->getValueType(1)))) {
2646     SDValue Res = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
2647     return CombineTo(N, Res, Res);
2648   }
2649 
2650   // If both halves are used, return as it is.
2651   if (LoExists && HiExists)
2652     return SDValue();
2653 
2654   // If the two computed results can be simplified separately, separate them.
2655   if (LoExists) {
2656     SDValue Lo = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops());
2657     AddToWorklist(Lo.getNode());
2658     SDValue LoOpt = combine(Lo.getNode());
2659     if (LoOpt.getNode() && LoOpt.getNode() != Lo.getNode() &&
2660         (!LegalOperations ||
2661          TLI.isOperationLegal(LoOpt.getOpcode(), LoOpt.getValueType())))
2662       return CombineTo(N, LoOpt, LoOpt);
2663   }
2664 
2665   if (HiExists) {
2666     SDValue Hi = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops());
2667     AddToWorklist(Hi.getNode());
2668     SDValue HiOpt = combine(Hi.getNode());
2669     if (HiOpt.getNode() && HiOpt != Hi &&
2670         (!LegalOperations ||
2671          TLI.isOperationLegal(HiOpt.getOpcode(), HiOpt.getValueType())))
2672       return CombineTo(N, HiOpt, HiOpt);
2673   }
2674 
2675   return SDValue();
2676 }
2677 
2678 SDValue DAGCombiner::visitSMUL_LOHI(SDNode *N) {
2679   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHS))
2680     return Res;
2681 
2682   EVT VT = N->getValueType(0);
2683   SDLoc DL(N);
2684 
2685   // If the type is twice as wide is legal, transform the mulhu to a wider
2686   // multiply plus a shift.
2687   if (VT.isSimple() && !VT.isVector()) {
2688     MVT Simple = VT.getSimpleVT();
2689     unsigned SimpleSize = Simple.getSizeInBits();
2690     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2691     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2692       SDValue Lo = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(0));
2693       SDValue Hi = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(1));
2694       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
2695       // Compute the high part as N1.
2696       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
2697             DAG.getConstant(SimpleSize, DL,
2698                             getShiftAmountTy(Lo.getValueType())));
2699       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
2700       // Compute the low part as N0.
2701       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
2702       return CombineTo(N, Lo, Hi);
2703     }
2704   }
2705 
2706   return SDValue();
2707 }
2708 
2709 SDValue DAGCombiner::visitUMUL_LOHI(SDNode *N) {
2710   if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHU))
2711     return Res;
2712 
2713   EVT VT = N->getValueType(0);
2714   SDLoc DL(N);
2715 
2716   // If the type is twice as wide is legal, transform the mulhu to a wider
2717   // multiply plus a shift.
2718   if (VT.isSimple() && !VT.isVector()) {
2719     MVT Simple = VT.getSimpleVT();
2720     unsigned SimpleSize = Simple.getSizeInBits();
2721     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2);
2722     if (TLI.isOperationLegal(ISD::MUL, NewVT)) {
2723       SDValue Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(0));
2724       SDValue Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(1));
2725       Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi);
2726       // Compute the high part as N1.
2727       Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo,
2728             DAG.getConstant(SimpleSize, DL,
2729                             getShiftAmountTy(Lo.getValueType())));
2730       Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi);
2731       // Compute the low part as N0.
2732       Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo);
2733       return CombineTo(N, Lo, Hi);
2734     }
2735   }
2736 
2737   return SDValue();
2738 }
2739 
2740 SDValue DAGCombiner::visitSMULO(SDNode *N) {
2741   // (smulo x, 2) -> (saddo x, x)
2742   if (ConstantSDNode *C2 = dyn_cast<ConstantSDNode>(N->getOperand(1)))
2743     if (C2->getAPIntValue() == 2)
2744       return DAG.getNode(ISD::SADDO, SDLoc(N), N->getVTList(),
2745                          N->getOperand(0), N->getOperand(0));
2746 
2747   return SDValue();
2748 }
2749 
2750 SDValue DAGCombiner::visitUMULO(SDNode *N) {
2751   // (umulo x, 2) -> (uaddo x, x)
2752   if (ConstantSDNode *C2 = dyn_cast<ConstantSDNode>(N->getOperand(1)))
2753     if (C2->getAPIntValue() == 2)
2754       return DAG.getNode(ISD::UADDO, SDLoc(N), N->getVTList(),
2755                          N->getOperand(0), N->getOperand(0));
2756 
2757   return SDValue();
2758 }
2759 
2760 SDValue DAGCombiner::visitIMINMAX(SDNode *N) {
2761   SDValue N0 = N->getOperand(0);
2762   SDValue N1 = N->getOperand(1);
2763   EVT VT = N0.getValueType();
2764 
2765   // fold vector ops
2766   if (VT.isVector())
2767     if (SDValue FoldedVOp = SimplifyVBinOp(N))
2768       return FoldedVOp;
2769 
2770   // fold (add c1, c2) -> c1+c2
2771   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
2772   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
2773   if (N0C && N1C)
2774     return DAG.FoldConstantArithmetic(N->getOpcode(), SDLoc(N), VT, N0C, N1C);
2775 
2776   // canonicalize constant to RHS
2777   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
2778      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
2779     return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N1, N0);
2780 
2781   return SDValue();
2782 }
2783 
2784 /// If this is a binary operator with two operands of the same opcode, try to
2785 /// simplify it.
2786 SDValue DAGCombiner::SimplifyBinOpWithSameOpcodeHands(SDNode *N) {
2787   SDValue N0 = N->getOperand(0), N1 = N->getOperand(1);
2788   EVT VT = N0.getValueType();
2789   assert(N0.getOpcode() == N1.getOpcode() && "Bad input!");
2790 
2791   // Bail early if none of these transforms apply.
2792   if (N0.getNumOperands() == 0) return SDValue();
2793 
2794   // For each of OP in AND/OR/XOR:
2795   // fold (OP (zext x), (zext y)) -> (zext (OP x, y))
2796   // fold (OP (sext x), (sext y)) -> (sext (OP x, y))
2797   // fold (OP (aext x), (aext y)) -> (aext (OP x, y))
2798   // fold (OP (bswap x), (bswap y)) -> (bswap (OP x, y))
2799   // fold (OP (trunc x), (trunc y)) -> (trunc (OP x, y)) (if trunc isn't free)
2800   //
2801   // do not sink logical op inside of a vector extend, since it may combine
2802   // into a vsetcc.
2803   EVT Op0VT = N0.getOperand(0).getValueType();
2804   if ((N0.getOpcode() == ISD::ZERO_EXTEND ||
2805        N0.getOpcode() == ISD::SIGN_EXTEND ||
2806        N0.getOpcode() == ISD::BSWAP ||
2807        // Avoid infinite looping with PromoteIntBinOp.
2808        (N0.getOpcode() == ISD::ANY_EXTEND &&
2809         (!LegalTypes || TLI.isTypeDesirableForOp(N->getOpcode(), Op0VT))) ||
2810        (N0.getOpcode() == ISD::TRUNCATE &&
2811         (!TLI.isZExtFree(VT, Op0VT) ||
2812          !TLI.isTruncateFree(Op0VT, VT)) &&
2813         TLI.isTypeLegal(Op0VT))) &&
2814       !VT.isVector() &&
2815       Op0VT == N1.getOperand(0).getValueType() &&
2816       (!LegalOperations || TLI.isOperationLegal(N->getOpcode(), Op0VT))) {
2817     SDValue ORNode = DAG.getNode(N->getOpcode(), SDLoc(N0),
2818                                  N0.getOperand(0).getValueType(),
2819                                  N0.getOperand(0), N1.getOperand(0));
2820     AddToWorklist(ORNode.getNode());
2821     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, ORNode);
2822   }
2823 
2824   // For each of OP in SHL/SRL/SRA/AND...
2825   //   fold (and (OP x, z), (OP y, z)) -> (OP (and x, y), z)
2826   //   fold (or  (OP x, z), (OP y, z)) -> (OP (or  x, y), z)
2827   //   fold (xor (OP x, z), (OP y, z)) -> (OP (xor x, y), z)
2828   if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL ||
2829        N0.getOpcode() == ISD::SRA || N0.getOpcode() == ISD::AND) &&
2830       N0.getOperand(1) == N1.getOperand(1)) {
2831     SDValue ORNode = DAG.getNode(N->getOpcode(), SDLoc(N0),
2832                                  N0.getOperand(0).getValueType(),
2833                                  N0.getOperand(0), N1.getOperand(0));
2834     AddToWorklist(ORNode.getNode());
2835     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT,
2836                        ORNode, N0.getOperand(1));
2837   }
2838 
2839   // Simplify xor/and/or (bitcast(A), bitcast(B)) -> bitcast(op (A,B))
2840   // Only perform this optimization up until type legalization, before
2841   // LegalizeVectorOprs. LegalizeVectorOprs promotes vector operations by
2842   // adding bitcasts. For example (xor v4i32) is promoted to (v2i64), and
2843   // we don't want to undo this promotion.
2844   // We also handle SCALAR_TO_VECTOR because xor/or/and operations are cheaper
2845   // on scalars.
2846   if ((N0.getOpcode() == ISD::BITCAST ||
2847        N0.getOpcode() == ISD::SCALAR_TO_VECTOR) &&
2848        Level <= AfterLegalizeTypes) {
2849     SDValue In0 = N0.getOperand(0);
2850     SDValue In1 = N1.getOperand(0);
2851     EVT In0Ty = In0.getValueType();
2852     EVT In1Ty = In1.getValueType();
2853     SDLoc DL(N);
2854     // If both incoming values are integers, and the original types are the
2855     // same.
2856     if (In0Ty.isInteger() && In1Ty.isInteger() && In0Ty == In1Ty) {
2857       SDValue Op = DAG.getNode(N->getOpcode(), DL, In0Ty, In0, In1);
2858       SDValue BC = DAG.getNode(N0.getOpcode(), DL, VT, Op);
2859       AddToWorklist(Op.getNode());
2860       return BC;
2861     }
2862   }
2863 
2864   // Xor/and/or are indifferent to the swizzle operation (shuffle of one value).
2865   // Simplify xor/and/or (shuff(A), shuff(B)) -> shuff(op (A,B))
2866   // If both shuffles use the same mask, and both shuffle within a single
2867   // vector, then it is worthwhile to move the swizzle after the operation.
2868   // The type-legalizer generates this pattern when loading illegal
2869   // vector types from memory. In many cases this allows additional shuffle
2870   // optimizations.
2871   // There are other cases where moving the shuffle after the xor/and/or
2872   // is profitable even if shuffles don't perform a swizzle.
2873   // If both shuffles use the same mask, and both shuffles have the same first
2874   // or second operand, then it might still be profitable to move the shuffle
2875   // after the xor/and/or operation.
2876   if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG) {
2877     ShuffleVectorSDNode *SVN0 = cast<ShuffleVectorSDNode>(N0);
2878     ShuffleVectorSDNode *SVN1 = cast<ShuffleVectorSDNode>(N1);
2879 
2880     assert(N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType() &&
2881            "Inputs to shuffles are not the same type");
2882 
2883     // Check that both shuffles use the same mask. The masks are known to be of
2884     // the same length because the result vector type is the same.
2885     // Check also that shuffles have only one use to avoid introducing extra
2886     // instructions.
2887     if (SVN0->hasOneUse() && SVN1->hasOneUse() &&
2888         SVN0->getMask().equals(SVN1->getMask())) {
2889       SDValue ShOp = N0->getOperand(1);
2890 
2891       // Don't try to fold this node if it requires introducing a
2892       // build vector of all zeros that might be illegal at this stage.
2893       if (N->getOpcode() == ISD::XOR && !ShOp.isUndef()) {
2894         if (!LegalTypes)
2895           ShOp = DAG.getConstant(0, SDLoc(N), VT);
2896         else
2897           ShOp = SDValue();
2898       }
2899 
2900       // (AND (shuf (A, C), shuf (B, C)) -> shuf (AND (A, B), C)
2901       // (OR  (shuf (A, C), shuf (B, C)) -> shuf (OR  (A, B), C)
2902       // (XOR (shuf (A, C), shuf (B, C)) -> shuf (XOR (A, B), V_0)
2903       if (N0.getOperand(1) == N1.getOperand(1) && ShOp.getNode()) {
2904         SDValue NewNode = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
2905                                       N0->getOperand(0), N1->getOperand(0));
2906         AddToWorklist(NewNode.getNode());
2907         return DAG.getVectorShuffle(VT, SDLoc(N), NewNode, ShOp,
2908                                     SVN0->getMask());
2909       }
2910 
2911       // Don't try to fold this node if it requires introducing a
2912       // build vector of all zeros that might be illegal at this stage.
2913       ShOp = N0->getOperand(0);
2914       if (N->getOpcode() == ISD::XOR && !ShOp.isUndef()) {
2915         if (!LegalTypes)
2916           ShOp = DAG.getConstant(0, SDLoc(N), VT);
2917         else
2918           ShOp = SDValue();
2919       }
2920 
2921       // (AND (shuf (C, A), shuf (C, B)) -> shuf (C, AND (A, B))
2922       // (OR  (shuf (C, A), shuf (C, B)) -> shuf (C, OR  (A, B))
2923       // (XOR (shuf (C, A), shuf (C, B)) -> shuf (V_0, XOR (A, B))
2924       if (N0->getOperand(0) == N1->getOperand(0) && ShOp.getNode()) {
2925         SDValue NewNode = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
2926                                       N0->getOperand(1), N1->getOperand(1));
2927         AddToWorklist(NewNode.getNode());
2928         return DAG.getVectorShuffle(VT, SDLoc(N), ShOp, NewNode,
2929                                     SVN0->getMask());
2930       }
2931     }
2932   }
2933 
2934   return SDValue();
2935 }
2936 
2937 /// This contains all DAGCombine rules which reduce two values combined by
2938 /// an And operation to a single value. This makes them reusable in the context
2939 /// of visitSELECT(). Rules involving constants are not included as
2940 /// visitSELECT() already handles those cases.
2941 SDValue DAGCombiner::visitANDLike(SDValue N0, SDValue N1,
2942                                   SDNode *LocReference) {
2943   EVT VT = N1.getValueType();
2944 
2945   // fold (and x, undef) -> 0
2946   if (N0.isUndef() || N1.isUndef())
2947     return DAG.getConstant(0, SDLoc(LocReference), VT);
2948   // fold (and (setcc x), (setcc y)) -> (setcc (and x, y))
2949   SDValue LL, LR, RL, RR, CC0, CC1;
2950   if (isSetCCEquivalent(N0, LL, LR, CC0) && isSetCCEquivalent(N1, RL, RR, CC1)){
2951     ISD::CondCode Op0 = cast<CondCodeSDNode>(CC0)->get();
2952     ISD::CondCode Op1 = cast<CondCodeSDNode>(CC1)->get();
2953 
2954     if (LR == RR && isa<ConstantSDNode>(LR) && Op0 == Op1 &&
2955         LL.getValueType().isInteger()) {
2956       // fold (and (seteq X, 0), (seteq Y, 0)) -> (seteq (or X, Y), 0)
2957       if (isNullConstant(LR) && Op1 == ISD::SETEQ) {
2958         EVT CCVT = getSetCCResultType(LR.getValueType());
2959         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2960           SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(N0),
2961                                        LR.getValueType(), LL, RL);
2962           AddToWorklist(ORNode.getNode());
2963           return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
2964         }
2965       }
2966       if (isAllOnesConstant(LR)) {
2967         // fold (and (seteq X, -1), (seteq Y, -1)) -> (seteq (and X, Y), -1)
2968         if (Op1 == ISD::SETEQ) {
2969           EVT CCVT = getSetCCResultType(LR.getValueType());
2970           if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2971             SDValue ANDNode = DAG.getNode(ISD::AND, SDLoc(N0),
2972                                           LR.getValueType(), LL, RL);
2973             AddToWorklist(ANDNode.getNode());
2974             return DAG.getSetCC(SDLoc(LocReference), VT, ANDNode, LR, Op1);
2975           }
2976         }
2977         // fold (and (setgt X, -1), (setgt Y, -1)) -> (setgt (or X, Y), -1)
2978         if (Op1 == ISD::SETGT) {
2979           EVT CCVT = getSetCCResultType(LR.getValueType());
2980           if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2981             SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(N0),
2982                                          LR.getValueType(), LL, RL);
2983             AddToWorklist(ORNode.getNode());
2984             return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
2985           }
2986         }
2987       }
2988     }
2989     // Simplify (and (setne X, 0), (setne X, -1)) -> (setuge (add X, 1), 2)
2990     if (LL == RL && isa<ConstantSDNode>(LR) && isa<ConstantSDNode>(RR) &&
2991         Op0 == Op1 && LL.getValueType().isInteger() &&
2992       Op0 == ISD::SETNE && ((isNullConstant(LR) && isAllOnesConstant(RR)) ||
2993                             (isAllOnesConstant(LR) && isNullConstant(RR)))) {
2994       EVT CCVT = getSetCCResultType(LL.getValueType());
2995       if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
2996         SDLoc DL(N0);
2997         SDValue ADDNode = DAG.getNode(ISD::ADD, DL, LL.getValueType(),
2998                                       LL, DAG.getConstant(1, DL,
2999                                                           LL.getValueType()));
3000         AddToWorklist(ADDNode.getNode());
3001         return DAG.getSetCC(SDLoc(LocReference), VT, ADDNode,
3002                             DAG.getConstant(2, DL, LL.getValueType()),
3003                             ISD::SETUGE);
3004       }
3005     }
3006     // canonicalize equivalent to ll == rl
3007     if (LL == RR && LR == RL) {
3008       Op1 = ISD::getSetCCSwappedOperands(Op1);
3009       std::swap(RL, RR);
3010     }
3011     if (LL == RL && LR == RR) {
3012       bool isInteger = LL.getValueType().isInteger();
3013       ISD::CondCode Result = ISD::getSetCCAndOperation(Op0, Op1, isInteger);
3014       if (Result != ISD::SETCC_INVALID &&
3015           (!LegalOperations ||
3016            (TLI.isCondCodeLegal(Result, LL.getSimpleValueType()) &&
3017             TLI.isOperationLegal(ISD::SETCC, LL.getValueType())))) {
3018         EVT CCVT = getSetCCResultType(LL.getValueType());
3019         if (N0.getValueType() == CCVT ||
3020             (!LegalOperations && N0.getValueType() == MVT::i1))
3021           return DAG.getSetCC(SDLoc(LocReference), N0.getValueType(),
3022                               LL, LR, Result);
3023       }
3024     }
3025   }
3026 
3027   if (N0.getOpcode() == ISD::ADD && N1.getOpcode() == ISD::SRL &&
3028       VT.getSizeInBits() <= 64) {
3029     if (ConstantSDNode *ADDI = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
3030       APInt ADDC = ADDI->getAPIntValue();
3031       if (!TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
3032         // Look for (and (add x, c1), (lshr y, c2)). If C1 wasn't a legal
3033         // immediate for an add, but it is legal if its top c2 bits are set,
3034         // transform the ADD so the immediate doesn't need to be materialized
3035         // in a register.
3036         if (ConstantSDNode *SRLI = dyn_cast<ConstantSDNode>(N1.getOperand(1))) {
3037           APInt Mask = APInt::getHighBitsSet(VT.getSizeInBits(),
3038                                              SRLI->getZExtValue());
3039           if (DAG.MaskedValueIsZero(N0.getOperand(1), Mask)) {
3040             ADDC |= Mask;
3041             if (TLI.isLegalAddImmediate(ADDC.getSExtValue())) {
3042               SDLoc DL(N0);
3043               SDValue NewAdd =
3044                 DAG.getNode(ISD::ADD, DL, VT,
3045                             N0.getOperand(0), DAG.getConstant(ADDC, DL, VT));
3046               CombineTo(N0.getNode(), NewAdd);
3047               // Return N so it doesn't get rechecked!
3048               return SDValue(LocReference, 0);
3049             }
3050           }
3051         }
3052       }
3053     }
3054   }
3055 
3056   // Reduce bit extract of low half of an integer to the narrower type.
3057   // (and (srl i64:x, K), KMask) ->
3058   //   (i64 zero_extend (and (srl (i32 (trunc i64:x)), K)), KMask)
3059   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
3060     if (ConstantSDNode *CAnd = dyn_cast<ConstantSDNode>(N1)) {
3061       if (ConstantSDNode *CShift = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
3062         unsigned Size = VT.getSizeInBits();
3063         const APInt &AndMask = CAnd->getAPIntValue();
3064         unsigned ShiftBits = CShift->getZExtValue();
3065 
3066         // Bail out, this node will probably disappear anyway.
3067         if (ShiftBits == 0)
3068           return SDValue();
3069 
3070         unsigned MaskBits = AndMask.countTrailingOnes();
3071         EVT HalfVT = EVT::getIntegerVT(*DAG.getContext(), Size / 2);
3072 
3073         if (APIntOps::isMask(AndMask) &&
3074             // Required bits must not span the two halves of the integer and
3075             // must fit in the half size type.
3076             (ShiftBits + MaskBits <= Size / 2) &&
3077             TLI.isNarrowingProfitable(VT, HalfVT) &&
3078             TLI.isTypeDesirableForOp(ISD::AND, HalfVT) &&
3079             TLI.isTypeDesirableForOp(ISD::SRL, HalfVT) &&
3080             TLI.isTruncateFree(VT, HalfVT) &&
3081             TLI.isZExtFree(HalfVT, VT)) {
3082           // The isNarrowingProfitable is to avoid regressions on PPC and
3083           // AArch64 which match a few 64-bit bit insert / bit extract patterns
3084           // on downstream users of this. Those patterns could probably be
3085           // extended to handle extensions mixed in.
3086 
3087           SDValue SL(N0);
3088           assert(MaskBits <= Size);
3089 
3090           // Extracting the highest bit of the low half.
3091           EVT ShiftVT = TLI.getShiftAmountTy(HalfVT, DAG.getDataLayout());
3092           SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, HalfVT,
3093                                       N0.getOperand(0));
3094 
3095           SDValue NewMask = DAG.getConstant(AndMask.trunc(Size / 2), SL, HalfVT);
3096           SDValue ShiftK = DAG.getConstant(ShiftBits, SL, ShiftVT);
3097           SDValue Shift = DAG.getNode(ISD::SRL, SL, HalfVT, Trunc, ShiftK);
3098           SDValue And = DAG.getNode(ISD::AND, SL, HalfVT, Shift, NewMask);
3099           return DAG.getNode(ISD::ZERO_EXTEND, SL, VT, And);
3100         }
3101       }
3102     }
3103   }
3104 
3105   return SDValue();
3106 }
3107 
3108 bool DAGCombiner::isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN,
3109                                    EVT LoadResultTy, EVT &ExtVT, EVT &LoadedVT,
3110                                    bool &NarrowLoad) {
3111   uint32_t ActiveBits = AndC->getAPIntValue().getActiveBits();
3112 
3113   if (ActiveBits == 0 || !APIntOps::isMask(ActiveBits, AndC->getAPIntValue()))
3114     return false;
3115 
3116   ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits);
3117   LoadedVT = LoadN->getMemoryVT();
3118 
3119   if (ExtVT == LoadedVT &&
3120       (!LegalOperations ||
3121        TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))) {
3122     // ZEXTLOAD will match without needing to change the size of the value being
3123     // loaded.
3124     NarrowLoad = false;
3125     return true;
3126   }
3127 
3128   // Do not change the width of a volatile load.
3129   if (LoadN->isVolatile())
3130     return false;
3131 
3132   // Do not generate loads of non-round integer types since these can
3133   // be expensive (and would be wrong if the type is not byte sized).
3134   if (!LoadedVT.bitsGT(ExtVT) || !ExtVT.isRound())
3135     return false;
3136 
3137   if (LegalOperations &&
3138       !TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))
3139     return false;
3140 
3141   if (!TLI.shouldReduceLoadWidth(LoadN, ISD::ZEXTLOAD, ExtVT))
3142     return false;
3143 
3144   NarrowLoad = true;
3145   return true;
3146 }
3147 
3148 SDValue DAGCombiner::visitAND(SDNode *N) {
3149   SDValue N0 = N->getOperand(0);
3150   SDValue N1 = N->getOperand(1);
3151   EVT VT = N1.getValueType();
3152 
3153   // x & x --> x
3154   if (N0 == N1)
3155     return N0;
3156 
3157   // fold vector ops
3158   if (VT.isVector()) {
3159     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3160       return FoldedVOp;
3161 
3162     // fold (and x, 0) -> 0, vector edition
3163     if (ISD::isBuildVectorAllZeros(N0.getNode()))
3164       // do not return N0, because undef node may exist in N0
3165       return DAG.getConstant(APInt::getNullValue(N0.getScalarValueSizeInBits()),
3166                              SDLoc(N), N0.getValueType());
3167     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3168       // do not return N1, because undef node may exist in N1
3169       return DAG.getConstant(APInt::getNullValue(N1.getScalarValueSizeInBits()),
3170                              SDLoc(N), N1.getValueType());
3171 
3172     // fold (and x, -1) -> x, vector edition
3173     if (ISD::isBuildVectorAllOnes(N0.getNode()))
3174       return N1;
3175     if (ISD::isBuildVectorAllOnes(N1.getNode()))
3176       return N0;
3177   }
3178 
3179   // fold (and c1, c2) -> c1&c2
3180   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
3181   ConstantSDNode *N1C = isConstOrConstSplat(N1);
3182   if (N0C && N1C && !N1C->isOpaque())
3183     return DAG.FoldConstantArithmetic(ISD::AND, SDLoc(N), VT, N0C, N1C);
3184   // canonicalize constant to RHS
3185   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3186      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
3187     return DAG.getNode(ISD::AND, SDLoc(N), VT, N1, N0);
3188   // fold (and x, -1) -> x
3189   if (isAllOnesConstant(N1))
3190     return N0;
3191   // if (and x, c) is known to be zero, return 0
3192   unsigned BitWidth = VT.getScalarSizeInBits();
3193   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
3194                                    APInt::getAllOnesValue(BitWidth)))
3195     return DAG.getConstant(0, SDLoc(N), VT);
3196   // reassociate and
3197   if (SDValue RAND = ReassociateOps(ISD::AND, SDLoc(N), N0, N1))
3198     return RAND;
3199   // fold (and (or x, C), D) -> D if (C & D) == D
3200   if (N1C && N0.getOpcode() == ISD::OR)
3201     if (ConstantSDNode *ORI = isConstOrConstSplat(N0.getOperand(1)))
3202       if ((ORI->getAPIntValue() & N1C->getAPIntValue()) == N1C->getAPIntValue())
3203         return N1;
3204   // fold (and (any_ext V), c) -> (zero_ext V) if 'and' only clears top bits.
3205   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
3206     SDValue N0Op0 = N0.getOperand(0);
3207     APInt Mask = ~N1C->getAPIntValue();
3208     Mask = Mask.trunc(N0Op0.getScalarValueSizeInBits());
3209     if (DAG.MaskedValueIsZero(N0Op0, Mask)) {
3210       SDValue Zext = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N),
3211                                  N0.getValueType(), N0Op0);
3212 
3213       // Replace uses of the AND with uses of the Zero extend node.
3214       CombineTo(N, Zext);
3215 
3216       // We actually want to replace all uses of the any_extend with the
3217       // zero_extend, to avoid duplicating things.  This will later cause this
3218       // AND to be folded.
3219       CombineTo(N0.getNode(), Zext);
3220       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3221     }
3222   }
3223   // similarly fold (and (X (load ([non_ext|any_ext|zero_ext] V))), c) ->
3224   // (X (load ([non_ext|zero_ext] V))) if 'and' only clears top bits which must
3225   // already be zero by virtue of the width of the base type of the load.
3226   //
3227   // the 'X' node here can either be nothing or an extract_vector_elt to catch
3228   // more cases.
3229   if ((N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
3230        N0.getValueSizeInBits() == N0.getOperand(0).getScalarValueSizeInBits() &&
3231        N0.getOperand(0).getOpcode() == ISD::LOAD &&
3232        N0.getOperand(0).getResNo() == 0) ||
3233       (N0.getOpcode() == ISD::LOAD && N0.getResNo() == 0)) {
3234     LoadSDNode *Load = cast<LoadSDNode>( (N0.getOpcode() == ISD::LOAD) ?
3235                                          N0 : N0.getOperand(0) );
3236 
3237     // Get the constant (if applicable) the zero'th operand is being ANDed with.
3238     // This can be a pure constant or a vector splat, in which case we treat the
3239     // vector as a scalar and use the splat value.
3240     APInt Constant = APInt::getNullValue(1);
3241     if (const ConstantSDNode *C = dyn_cast<ConstantSDNode>(N1)) {
3242       Constant = C->getAPIntValue();
3243     } else if (BuildVectorSDNode *Vector = dyn_cast<BuildVectorSDNode>(N1)) {
3244       APInt SplatValue, SplatUndef;
3245       unsigned SplatBitSize;
3246       bool HasAnyUndefs;
3247       bool IsSplat = Vector->isConstantSplat(SplatValue, SplatUndef,
3248                                              SplatBitSize, HasAnyUndefs);
3249       if (IsSplat) {
3250         // Undef bits can contribute to a possible optimisation if set, so
3251         // set them.
3252         SplatValue |= SplatUndef;
3253 
3254         // The splat value may be something like "0x00FFFFFF", which means 0 for
3255         // the first vector value and FF for the rest, repeating. We need a mask
3256         // that will apply equally to all members of the vector, so AND all the
3257         // lanes of the constant together.
3258         EVT VT = Vector->getValueType(0);
3259         unsigned BitWidth = VT.getScalarSizeInBits();
3260 
3261         // If the splat value has been compressed to a bitlength lower
3262         // than the size of the vector lane, we need to re-expand it to
3263         // the lane size.
3264         if (BitWidth > SplatBitSize)
3265           for (SplatValue = SplatValue.zextOrTrunc(BitWidth);
3266                SplatBitSize < BitWidth;
3267                SplatBitSize = SplatBitSize * 2)
3268             SplatValue |= SplatValue.shl(SplatBitSize);
3269 
3270         // Make sure that variable 'Constant' is only set if 'SplatBitSize' is a
3271         // multiple of 'BitWidth'. Otherwise, we could propagate a wrong value.
3272         if (SplatBitSize % BitWidth == 0) {
3273           Constant = APInt::getAllOnesValue(BitWidth);
3274           for (unsigned i = 0, n = SplatBitSize/BitWidth; i < n; ++i)
3275             Constant &= SplatValue.lshr(i*BitWidth).zextOrTrunc(BitWidth);
3276         }
3277       }
3278     }
3279 
3280     // If we want to change an EXTLOAD to a ZEXTLOAD, ensure a ZEXTLOAD is
3281     // actually legal and isn't going to get expanded, else this is a false
3282     // optimisation.
3283     bool CanZextLoadProfitably = TLI.isLoadExtLegal(ISD::ZEXTLOAD,
3284                                                     Load->getValueType(0),
3285                                                     Load->getMemoryVT());
3286 
3287     // Resize the constant to the same size as the original memory access before
3288     // extension. If it is still the AllOnesValue then this AND is completely
3289     // unneeded.
3290     Constant = Constant.zextOrTrunc(Load->getMemoryVT().getScalarSizeInBits());
3291 
3292     bool B;
3293     switch (Load->getExtensionType()) {
3294     default: B = false; break;
3295     case ISD::EXTLOAD: B = CanZextLoadProfitably; break;
3296     case ISD::ZEXTLOAD:
3297     case ISD::NON_EXTLOAD: B = true; break;
3298     }
3299 
3300     if (B && Constant.isAllOnesValue()) {
3301       // If the load type was an EXTLOAD, convert to ZEXTLOAD in order to
3302       // preserve semantics once we get rid of the AND.
3303       SDValue NewLoad(Load, 0);
3304       if (Load->getExtensionType() == ISD::EXTLOAD) {
3305         NewLoad = DAG.getLoad(Load->getAddressingMode(), ISD::ZEXTLOAD,
3306                               Load->getValueType(0), SDLoc(Load),
3307                               Load->getChain(), Load->getBasePtr(),
3308                               Load->getOffset(), Load->getMemoryVT(),
3309                               Load->getMemOperand());
3310         // Replace uses of the EXTLOAD with the new ZEXTLOAD.
3311         if (Load->getNumValues() == 3) {
3312           // PRE/POST_INC loads have 3 values.
3313           SDValue To[] = { NewLoad.getValue(0), NewLoad.getValue(1),
3314                            NewLoad.getValue(2) };
3315           CombineTo(Load, To, 3, true);
3316         } else {
3317           CombineTo(Load, NewLoad.getValue(0), NewLoad.getValue(1));
3318         }
3319       }
3320 
3321       // Fold the AND away, taking care not to fold to the old load node if we
3322       // replaced it.
3323       CombineTo(N, (N0.getNode() == Load) ? NewLoad : N0);
3324 
3325       return SDValue(N, 0); // Return N so it doesn't get rechecked!
3326     }
3327   }
3328 
3329   // fold (and (load x), 255) -> (zextload x, i8)
3330   // fold (and (extload x, i16), 255) -> (zextload x, i8)
3331   // fold (and (any_ext (extload x, i16)), 255) -> (zextload x, i8)
3332   if (!VT.isVector() && N1C && (N0.getOpcode() == ISD::LOAD ||
3333                                 (N0.getOpcode() == ISD::ANY_EXTEND &&
3334                                  N0.getOperand(0).getOpcode() == ISD::LOAD))) {
3335     bool HasAnyExt = N0.getOpcode() == ISD::ANY_EXTEND;
3336     LoadSDNode *LN0 = HasAnyExt
3337       ? cast<LoadSDNode>(N0.getOperand(0))
3338       : cast<LoadSDNode>(N0);
3339     if (LN0->getExtensionType() != ISD::SEXTLOAD &&
3340         LN0->isUnindexed() && N0.hasOneUse() && SDValue(LN0, 0).hasOneUse()) {
3341       auto NarrowLoad = false;
3342       EVT LoadResultTy = HasAnyExt ? LN0->getValueType(0) : VT;
3343       EVT ExtVT, LoadedVT;
3344       if (isAndLoadExtLoad(N1C, LN0, LoadResultTy, ExtVT, LoadedVT,
3345                            NarrowLoad)) {
3346         if (!NarrowLoad) {
3347           SDValue NewLoad =
3348             DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN0), LoadResultTy,
3349                            LN0->getChain(), LN0->getBasePtr(), ExtVT,
3350                            LN0->getMemOperand());
3351           AddToWorklist(N);
3352           CombineTo(LN0, NewLoad, NewLoad.getValue(1));
3353           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3354         } else {
3355           EVT PtrType = LN0->getOperand(1).getValueType();
3356 
3357           unsigned Alignment = LN0->getAlignment();
3358           SDValue NewPtr = LN0->getBasePtr();
3359 
3360           // For big endian targets, we need to add an offset to the pointer
3361           // to load the correct bytes.  For little endian systems, we merely
3362           // need to read fewer bytes from the same pointer.
3363           if (DAG.getDataLayout().isBigEndian()) {
3364             unsigned LVTStoreBytes = LoadedVT.getStoreSize();
3365             unsigned EVTStoreBytes = ExtVT.getStoreSize();
3366             unsigned PtrOff = LVTStoreBytes - EVTStoreBytes;
3367             SDLoc DL(LN0);
3368             NewPtr = DAG.getNode(ISD::ADD, DL, PtrType,
3369                                  NewPtr, DAG.getConstant(PtrOff, DL, PtrType));
3370             Alignment = MinAlign(Alignment, PtrOff);
3371           }
3372 
3373           AddToWorklist(NewPtr.getNode());
3374 
3375           SDValue Load = DAG.getExtLoad(
3376               ISD::ZEXTLOAD, SDLoc(LN0), LoadResultTy, LN0->getChain(), NewPtr,
3377               LN0->getPointerInfo(), ExtVT, Alignment,
3378               LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
3379           AddToWorklist(N);
3380           CombineTo(LN0, Load, Load.getValue(1));
3381           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3382         }
3383       }
3384     }
3385   }
3386 
3387   if (SDValue Combined = visitANDLike(N0, N1, N))
3388     return Combined;
3389 
3390   // Simplify: (and (op x...), (op y...))  -> (op (and x, y))
3391   if (N0.getOpcode() == N1.getOpcode())
3392     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
3393       return Tmp;
3394 
3395   // Masking the negated extension of a boolean is just the zero-extended
3396   // boolean:
3397   // and (sub 0, zext(bool X)), 1 --> zext(bool X)
3398   // and (sub 0, sext(bool X)), 1 --> zext(bool X)
3399   //
3400   // Note: the SimplifyDemandedBits fold below can make an information-losing
3401   // transform, and then we have no way to find this better fold.
3402   if (N1C && N1C->isOne() && N0.getOpcode() == ISD::SUB) {
3403     ConstantSDNode *SubLHS = isConstOrConstSplat(N0.getOperand(0));
3404     SDValue SubRHS = N0.getOperand(1);
3405     if (SubLHS && SubLHS->isNullValue()) {
3406       if (SubRHS.getOpcode() == ISD::ZERO_EXTEND &&
3407           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
3408         return SubRHS;
3409       if (SubRHS.getOpcode() == ISD::SIGN_EXTEND &&
3410           SubRHS.getOperand(0).getScalarValueSizeInBits() == 1)
3411         return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, SubRHS.getOperand(0));
3412     }
3413   }
3414 
3415   // fold (and (sign_extend_inreg x, i16 to i32), 1) -> (and x, 1)
3416   // fold (and (sra)) -> (and (srl)) when possible.
3417   if (!VT.isVector() && SimplifyDemandedBits(SDValue(N, 0)))
3418     return SDValue(N, 0);
3419 
3420   // fold (zext_inreg (extload x)) -> (zextload x)
3421   if (ISD::isEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode())) {
3422     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
3423     EVT MemVT = LN0->getMemoryVT();
3424     // If we zero all the possible extended bits, then we can turn this into
3425     // a zextload if we are running before legalize or the operation is legal.
3426     unsigned BitWidth = N1.getScalarValueSizeInBits();
3427     if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth,
3428                            BitWidth - MemVT.getScalarSizeInBits())) &&
3429         ((!LegalOperations && !LN0->isVolatile()) ||
3430          TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) {
3431       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT,
3432                                        LN0->getChain(), LN0->getBasePtr(),
3433                                        MemVT, LN0->getMemOperand());
3434       AddToWorklist(N);
3435       CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
3436       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3437     }
3438   }
3439   // fold (zext_inreg (sextload x)) -> (zextload x) iff load has one use
3440   if (ISD::isSEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
3441       N0.hasOneUse()) {
3442     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
3443     EVT MemVT = LN0->getMemoryVT();
3444     // If we zero all the possible extended bits, then we can turn this into
3445     // a zextload if we are running before legalize or the operation is legal.
3446     unsigned BitWidth = N1.getScalarValueSizeInBits();
3447     if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth,
3448                            BitWidth - MemVT.getScalarSizeInBits())) &&
3449         ((!LegalOperations && !LN0->isVolatile()) ||
3450          TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) {
3451       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT,
3452                                        LN0->getChain(), LN0->getBasePtr(),
3453                                        MemVT, LN0->getMemOperand());
3454       AddToWorklist(N);
3455       CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
3456       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
3457     }
3458   }
3459   // fold (and (or (srl N, 8), (shl N, 8)), 0xffff) -> (srl (bswap N), const)
3460   if (N1C && N1C->getAPIntValue() == 0xffff && N0.getOpcode() == ISD::OR) {
3461     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
3462                                            N0.getOperand(1), false))
3463       return BSwap;
3464   }
3465 
3466   return SDValue();
3467 }
3468 
3469 /// Match (a >> 8) | (a << 8) as (bswap a) >> 16.
3470 SDValue DAGCombiner::MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1,
3471                                         bool DemandHighBits) {
3472   if (!LegalOperations)
3473     return SDValue();
3474 
3475   EVT VT = N->getValueType(0);
3476   if (VT != MVT::i64 && VT != MVT::i32 && VT != MVT::i16)
3477     return SDValue();
3478   if (!TLI.isOperationLegal(ISD::BSWAP, VT))
3479     return SDValue();
3480 
3481   // Recognize (and (shl a, 8), 0xff), (and (srl a, 8), 0xff00)
3482   bool LookPassAnd0 = false;
3483   bool LookPassAnd1 = false;
3484   if (N0.getOpcode() == ISD::AND && N0.getOperand(0).getOpcode() == ISD::SRL)
3485       std::swap(N0, N1);
3486   if (N1.getOpcode() == ISD::AND && N1.getOperand(0).getOpcode() == ISD::SHL)
3487       std::swap(N0, N1);
3488   if (N0.getOpcode() == ISD::AND) {
3489     if (!N0.getNode()->hasOneUse())
3490       return SDValue();
3491     ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3492     if (!N01C || N01C->getZExtValue() != 0xFF00)
3493       return SDValue();
3494     N0 = N0.getOperand(0);
3495     LookPassAnd0 = true;
3496   }
3497 
3498   if (N1.getOpcode() == ISD::AND) {
3499     if (!N1.getNode()->hasOneUse())
3500       return SDValue();
3501     ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
3502     if (!N11C || N11C->getZExtValue() != 0xFF)
3503       return SDValue();
3504     N1 = N1.getOperand(0);
3505     LookPassAnd1 = true;
3506   }
3507 
3508   if (N0.getOpcode() == ISD::SRL && N1.getOpcode() == ISD::SHL)
3509     std::swap(N0, N1);
3510   if (N0.getOpcode() != ISD::SHL || N1.getOpcode() != ISD::SRL)
3511     return SDValue();
3512   if (!N0.getNode()->hasOneUse() || !N1.getNode()->hasOneUse())
3513     return SDValue();
3514 
3515   ConstantSDNode *N01C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3516   ConstantSDNode *N11C = dyn_cast<ConstantSDNode>(N1.getOperand(1));
3517   if (!N01C || !N11C)
3518     return SDValue();
3519   if (N01C->getZExtValue() != 8 || N11C->getZExtValue() != 8)
3520     return SDValue();
3521 
3522   // Look for (shl (and a, 0xff), 8), (srl (and a, 0xff00), 8)
3523   SDValue N00 = N0->getOperand(0);
3524   if (!LookPassAnd0 && N00.getOpcode() == ISD::AND) {
3525     if (!N00.getNode()->hasOneUse())
3526       return SDValue();
3527     ConstantSDNode *N001C = dyn_cast<ConstantSDNode>(N00.getOperand(1));
3528     if (!N001C || N001C->getZExtValue() != 0xFF)
3529       return SDValue();
3530     N00 = N00.getOperand(0);
3531     LookPassAnd0 = true;
3532   }
3533 
3534   SDValue N10 = N1->getOperand(0);
3535   if (!LookPassAnd1 && N10.getOpcode() == ISD::AND) {
3536     if (!N10.getNode()->hasOneUse())
3537       return SDValue();
3538     ConstantSDNode *N101C = dyn_cast<ConstantSDNode>(N10.getOperand(1));
3539     if (!N101C || N101C->getZExtValue() != 0xFF00)
3540       return SDValue();
3541     N10 = N10.getOperand(0);
3542     LookPassAnd1 = true;
3543   }
3544 
3545   if (N00 != N10)
3546     return SDValue();
3547 
3548   // Make sure everything beyond the low halfword gets set to zero since the SRL
3549   // 16 will clear the top bits.
3550   unsigned OpSizeInBits = VT.getSizeInBits();
3551   if (DemandHighBits && OpSizeInBits > 16) {
3552     // If the left-shift isn't masked out then the only way this is a bswap is
3553     // if all bits beyond the low 8 are 0. In that case the entire pattern
3554     // reduces to a left shift anyway: leave it for other parts of the combiner.
3555     if (!LookPassAnd0)
3556       return SDValue();
3557 
3558     // However, if the right shift isn't masked out then it might be because
3559     // it's not needed. See if we can spot that too.
3560     if (!LookPassAnd1 &&
3561         !DAG.MaskedValueIsZero(
3562             N10, APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - 16)))
3563       return SDValue();
3564   }
3565 
3566   SDValue Res = DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N00);
3567   if (OpSizeInBits > 16) {
3568     SDLoc DL(N);
3569     Res = DAG.getNode(ISD::SRL, DL, VT, Res,
3570                       DAG.getConstant(OpSizeInBits - 16, DL,
3571                                       getShiftAmountTy(VT)));
3572   }
3573   return Res;
3574 }
3575 
3576 /// Return true if the specified node is an element that makes up a 32-bit
3577 /// packed halfword byteswap.
3578 /// ((x & 0x000000ff) << 8) |
3579 /// ((x & 0x0000ff00) >> 8) |
3580 /// ((x & 0x00ff0000) << 8) |
3581 /// ((x & 0xff000000) >> 8)
3582 static bool isBSwapHWordElement(SDValue N, MutableArrayRef<SDNode *> Parts) {
3583   if (!N.getNode()->hasOneUse())
3584     return false;
3585 
3586   unsigned Opc = N.getOpcode();
3587   if (Opc != ISD::AND && Opc != ISD::SHL && Opc != ISD::SRL)
3588     return false;
3589 
3590   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3591   if (!N1C)
3592     return false;
3593 
3594   unsigned Num;
3595   switch (N1C->getZExtValue()) {
3596   default:
3597     return false;
3598   case 0xFF:       Num = 0; break;
3599   case 0xFF00:     Num = 1; break;
3600   case 0xFF0000:   Num = 2; break;
3601   case 0xFF000000: Num = 3; break;
3602   }
3603 
3604   // Look for (x & 0xff) << 8 as well as ((x << 8) & 0xff00).
3605   SDValue N0 = N.getOperand(0);
3606   if (Opc == ISD::AND) {
3607     if (Num == 0 || Num == 2) {
3608       // (x >> 8) & 0xff
3609       // (x >> 8) & 0xff0000
3610       if (N0.getOpcode() != ISD::SRL)
3611         return false;
3612       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3613       if (!C || C->getZExtValue() != 8)
3614         return false;
3615     } else {
3616       // (x << 8) & 0xff00
3617       // (x << 8) & 0xff000000
3618       if (N0.getOpcode() != ISD::SHL)
3619         return false;
3620       ConstantSDNode *C = dyn_cast<ConstantSDNode>(N0.getOperand(1));
3621       if (!C || C->getZExtValue() != 8)
3622         return false;
3623     }
3624   } else if (Opc == ISD::SHL) {
3625     // (x & 0xff) << 8
3626     // (x & 0xff0000) << 8
3627     if (Num != 0 && Num != 2)
3628       return false;
3629     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3630     if (!C || C->getZExtValue() != 8)
3631       return false;
3632   } else { // Opc == ISD::SRL
3633     // (x & 0xff00) >> 8
3634     // (x & 0xff000000) >> 8
3635     if (Num != 1 && Num != 3)
3636       return false;
3637     ConstantSDNode *C = dyn_cast<ConstantSDNode>(N.getOperand(1));
3638     if (!C || C->getZExtValue() != 8)
3639       return false;
3640   }
3641 
3642   if (Parts[Num])
3643     return false;
3644 
3645   Parts[Num] = N0.getOperand(0).getNode();
3646   return true;
3647 }
3648 
3649 /// Match a 32-bit packed halfword bswap. That is
3650 /// ((x & 0x000000ff) << 8) |
3651 /// ((x & 0x0000ff00) >> 8) |
3652 /// ((x & 0x00ff0000) << 8) |
3653 /// ((x & 0xff000000) >> 8)
3654 /// => (rotl (bswap x), 16)
3655 SDValue DAGCombiner::MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1) {
3656   if (!LegalOperations)
3657     return SDValue();
3658 
3659   EVT VT = N->getValueType(0);
3660   if (VT != MVT::i32)
3661     return SDValue();
3662   if (!TLI.isOperationLegal(ISD::BSWAP, VT))
3663     return SDValue();
3664 
3665   // Look for either
3666   // (or (or (and), (and)), (or (and), (and)))
3667   // (or (or (or (and), (and)), (and)), (and))
3668   if (N0.getOpcode() != ISD::OR)
3669     return SDValue();
3670   SDValue N00 = N0.getOperand(0);
3671   SDValue N01 = N0.getOperand(1);
3672   SDNode *Parts[4] = {};
3673 
3674   if (N1.getOpcode() == ISD::OR &&
3675       N00.getNumOperands() == 2 && N01.getNumOperands() == 2) {
3676     // (or (or (and), (and)), (or (and), (and)))
3677     SDValue N000 = N00.getOperand(0);
3678     if (!isBSwapHWordElement(N000, Parts))
3679       return SDValue();
3680 
3681     SDValue N001 = N00.getOperand(1);
3682     if (!isBSwapHWordElement(N001, Parts))
3683       return SDValue();
3684     SDValue N010 = N01.getOperand(0);
3685     if (!isBSwapHWordElement(N010, Parts))
3686       return SDValue();
3687     SDValue N011 = N01.getOperand(1);
3688     if (!isBSwapHWordElement(N011, Parts))
3689       return SDValue();
3690   } else {
3691     // (or (or (or (and), (and)), (and)), (and))
3692     if (!isBSwapHWordElement(N1, Parts))
3693       return SDValue();
3694     if (!isBSwapHWordElement(N01, Parts))
3695       return SDValue();
3696     if (N00.getOpcode() != ISD::OR)
3697       return SDValue();
3698     SDValue N000 = N00.getOperand(0);
3699     if (!isBSwapHWordElement(N000, Parts))
3700       return SDValue();
3701     SDValue N001 = N00.getOperand(1);
3702     if (!isBSwapHWordElement(N001, Parts))
3703       return SDValue();
3704   }
3705 
3706   // Make sure the parts are all coming from the same node.
3707   if (Parts[0] != Parts[1] || Parts[0] != Parts[2] || Parts[0] != Parts[3])
3708     return SDValue();
3709 
3710   SDLoc DL(N);
3711   SDValue BSwap = DAG.getNode(ISD::BSWAP, DL, VT,
3712                               SDValue(Parts[0], 0));
3713 
3714   // Result of the bswap should be rotated by 16. If it's not legal, then
3715   // do  (x << 16) | (x >> 16).
3716   SDValue ShAmt = DAG.getConstant(16, DL, getShiftAmountTy(VT));
3717   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT))
3718     return DAG.getNode(ISD::ROTL, DL, VT, BSwap, ShAmt);
3719   if (TLI.isOperationLegalOrCustom(ISD::ROTR, VT))
3720     return DAG.getNode(ISD::ROTR, DL, VT, BSwap, ShAmt);
3721   return DAG.getNode(ISD::OR, DL, VT,
3722                      DAG.getNode(ISD::SHL, DL, VT, BSwap, ShAmt),
3723                      DAG.getNode(ISD::SRL, DL, VT, BSwap, ShAmt));
3724 }
3725 
3726 /// This contains all DAGCombine rules which reduce two values combined by
3727 /// an Or operation to a single value \see visitANDLike().
3728 SDValue DAGCombiner::visitORLike(SDValue N0, SDValue N1, SDNode *LocReference) {
3729   EVT VT = N1.getValueType();
3730   // fold (or x, undef) -> -1
3731   if (!LegalOperations && (N0.isUndef() || N1.isUndef()))
3732     return DAG.getAllOnesConstant(SDLoc(LocReference), VT);
3733 
3734   // fold (or (setcc x), (setcc y)) -> (setcc (or x, y))
3735   SDValue LL, LR, RL, RR, CC0, CC1;
3736   if (isSetCCEquivalent(N0, LL, LR, CC0) && isSetCCEquivalent(N1, RL, RR, CC1)){
3737     ISD::CondCode Op0 = cast<CondCodeSDNode>(CC0)->get();
3738     ISD::CondCode Op1 = cast<CondCodeSDNode>(CC1)->get();
3739 
3740     if (LR == RR && Op0 == Op1 && LL.getValueType().isInteger()) {
3741       // fold (or (setne X, 0), (setne Y, 0)) -> (setne (or X, Y), 0)
3742       // fold (or (setlt X, 0), (setlt Y, 0)) -> (setne (or X, Y), 0)
3743       if (isNullConstant(LR) && (Op1 == ISD::SETNE || Op1 == ISD::SETLT)) {
3744         EVT CCVT = getSetCCResultType(LR.getValueType());
3745         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
3746           SDValue ORNode = DAG.getNode(ISD::OR, SDLoc(LR),
3747                                        LR.getValueType(), LL, RL);
3748           AddToWorklist(ORNode.getNode());
3749           return DAG.getSetCC(SDLoc(LocReference), VT, ORNode, LR, Op1);
3750         }
3751       }
3752       // fold (or (setne X, -1), (setne Y, -1)) -> (setne (and X, Y), -1)
3753       // fold (or (setgt X, -1), (setgt Y  -1)) -> (setgt (and X, Y), -1)
3754       if (isAllOnesConstant(LR) && (Op1 == ISD::SETNE || Op1 == ISD::SETGT)) {
3755         EVT CCVT = getSetCCResultType(LR.getValueType());
3756         if (VT == CCVT || (!LegalOperations && VT == MVT::i1)) {
3757           SDValue ANDNode = DAG.getNode(ISD::AND, SDLoc(LR),
3758                                         LR.getValueType(), LL, RL);
3759           AddToWorklist(ANDNode.getNode());
3760           return DAG.getSetCC(SDLoc(LocReference), VT, ANDNode, LR, Op1);
3761         }
3762       }
3763     }
3764     // canonicalize equivalent to ll == rl
3765     if (LL == RR && LR == RL) {
3766       Op1 = ISD::getSetCCSwappedOperands(Op1);
3767       std::swap(RL, RR);
3768     }
3769     if (LL == RL && LR == RR) {
3770       bool isInteger = LL.getValueType().isInteger();
3771       ISD::CondCode Result = ISD::getSetCCOrOperation(Op0, Op1, isInteger);
3772       if (Result != ISD::SETCC_INVALID &&
3773           (!LegalOperations ||
3774            (TLI.isCondCodeLegal(Result, LL.getSimpleValueType()) &&
3775             TLI.isOperationLegal(ISD::SETCC, LL.getValueType())))) {
3776         EVT CCVT = getSetCCResultType(LL.getValueType());
3777         if (N0.getValueType() == CCVT ||
3778             (!LegalOperations && N0.getValueType() == MVT::i1))
3779           return DAG.getSetCC(SDLoc(LocReference), N0.getValueType(),
3780                               LL, LR, Result);
3781       }
3782     }
3783   }
3784 
3785   // (or (and X, C1), (and Y, C2))  -> (and (or X, Y), C3) if possible.
3786   if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND &&
3787       // Don't increase # computations.
3788       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
3789     // We can only do this xform if we know that bits from X that are set in C2
3790     // but not in C1 are already zero.  Likewise for Y.
3791     if (const ConstantSDNode *N0O1C =
3792         getAsNonOpaqueConstant(N0.getOperand(1))) {
3793       if (const ConstantSDNode *N1O1C =
3794           getAsNonOpaqueConstant(N1.getOperand(1))) {
3795         // We can only do this xform if we know that bits from X that are set in
3796         // C2 but not in C1 are already zero.  Likewise for Y.
3797         const APInt &LHSMask = N0O1C->getAPIntValue();
3798         const APInt &RHSMask = N1O1C->getAPIntValue();
3799 
3800         if (DAG.MaskedValueIsZero(N0.getOperand(0), RHSMask&~LHSMask) &&
3801             DAG.MaskedValueIsZero(N1.getOperand(0), LHSMask&~RHSMask)) {
3802           SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
3803                                   N0.getOperand(0), N1.getOperand(0));
3804           SDLoc DL(LocReference);
3805           return DAG.getNode(ISD::AND, DL, VT, X,
3806                              DAG.getConstant(LHSMask | RHSMask, DL, VT));
3807         }
3808       }
3809     }
3810   }
3811 
3812   // (or (and X, M), (and X, N)) -> (and X, (or M, N))
3813   if (N0.getOpcode() == ISD::AND &&
3814       N1.getOpcode() == ISD::AND &&
3815       N0.getOperand(0) == N1.getOperand(0) &&
3816       // Don't increase # computations.
3817       (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) {
3818     SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT,
3819                             N0.getOperand(1), N1.getOperand(1));
3820     return DAG.getNode(ISD::AND, SDLoc(LocReference), VT, N0.getOperand(0), X);
3821   }
3822 
3823   return SDValue();
3824 }
3825 
3826 SDValue DAGCombiner::visitOR(SDNode *N) {
3827   SDValue N0 = N->getOperand(0);
3828   SDValue N1 = N->getOperand(1);
3829   EVT VT = N1.getValueType();
3830 
3831   // x | x --> x
3832   if (N0 == N1)
3833     return N0;
3834 
3835   // fold vector ops
3836   if (VT.isVector()) {
3837     if (SDValue FoldedVOp = SimplifyVBinOp(N))
3838       return FoldedVOp;
3839 
3840     // fold (or x, 0) -> x, vector edition
3841     if (ISD::isBuildVectorAllZeros(N0.getNode()))
3842       return N1;
3843     if (ISD::isBuildVectorAllZeros(N1.getNode()))
3844       return N0;
3845 
3846     // fold (or x, -1) -> -1, vector edition
3847     if (ISD::isBuildVectorAllOnes(N0.getNode()))
3848       // do not return N0, because undef node may exist in N0
3849       return DAG.getAllOnesConstant(SDLoc(N), N0.getValueType());
3850     if (ISD::isBuildVectorAllOnes(N1.getNode()))
3851       // do not return N1, because undef node may exist in N1
3852       return DAG.getAllOnesConstant(SDLoc(N), N1.getValueType());
3853 
3854     // fold (or (shuf A, V_0, MA), (shuf B, V_0, MB)) -> (shuf A, B, Mask)
3855     // Do this only if the resulting shuffle is legal.
3856     if (isa<ShuffleVectorSDNode>(N0) &&
3857         isa<ShuffleVectorSDNode>(N1) &&
3858         // Avoid folding a node with illegal type.
3859         TLI.isTypeLegal(VT)) {
3860       bool ZeroN00 = ISD::isBuildVectorAllZeros(N0.getOperand(0).getNode());
3861       bool ZeroN01 = ISD::isBuildVectorAllZeros(N0.getOperand(1).getNode());
3862       bool ZeroN10 = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
3863       bool ZeroN11 = ISD::isBuildVectorAllZeros(N1.getOperand(1).getNode());
3864       // Ensure both shuffles have a zero input.
3865       if ((ZeroN00 || ZeroN01) && (ZeroN10 || ZeroN11)) {
3866         assert((!ZeroN00 || !ZeroN01) && "Both inputs zero!");
3867         assert((!ZeroN10 || !ZeroN11) && "Both inputs zero!");
3868         const ShuffleVectorSDNode *SV0 = cast<ShuffleVectorSDNode>(N0);
3869         const ShuffleVectorSDNode *SV1 = cast<ShuffleVectorSDNode>(N1);
3870         bool CanFold = true;
3871         int NumElts = VT.getVectorNumElements();
3872         SmallVector<int, 4> Mask(NumElts);
3873 
3874         for (int i = 0; i != NumElts; ++i) {
3875           int M0 = SV0->getMaskElt(i);
3876           int M1 = SV1->getMaskElt(i);
3877 
3878           // Determine if either index is pointing to a zero vector.
3879           bool M0Zero = M0 < 0 || (ZeroN00 == (M0 < NumElts));
3880           bool M1Zero = M1 < 0 || (ZeroN10 == (M1 < NumElts));
3881 
3882           // If one element is zero and the otherside is undef, keep undef.
3883           // This also handles the case that both are undef.
3884           if ((M0Zero && M1 < 0) || (M1Zero && M0 < 0)) {
3885             Mask[i] = -1;
3886             continue;
3887           }
3888 
3889           // Make sure only one of the elements is zero.
3890           if (M0Zero == M1Zero) {
3891             CanFold = false;
3892             break;
3893           }
3894 
3895           assert((M0 >= 0 || M1 >= 0) && "Undef index!");
3896 
3897           // We have a zero and non-zero element. If the non-zero came from
3898           // SV0 make the index a LHS index. If it came from SV1, make it
3899           // a RHS index. We need to mod by NumElts because we don't care
3900           // which operand it came from in the original shuffles.
3901           Mask[i] = M1Zero ? M0 % NumElts : (M1 % NumElts) + NumElts;
3902         }
3903 
3904         if (CanFold) {
3905           SDValue NewLHS = ZeroN00 ? N0.getOperand(1) : N0.getOperand(0);
3906           SDValue NewRHS = ZeroN10 ? N1.getOperand(1) : N1.getOperand(0);
3907 
3908           bool LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
3909           if (!LegalMask) {
3910             std::swap(NewLHS, NewRHS);
3911             ShuffleVectorSDNode::commuteMask(Mask);
3912             LegalMask = TLI.isShuffleMaskLegal(Mask, VT);
3913           }
3914 
3915           if (LegalMask)
3916             return DAG.getVectorShuffle(VT, SDLoc(N), NewLHS, NewRHS, Mask);
3917         }
3918       }
3919     }
3920   }
3921 
3922   // fold (or c1, c2) -> c1|c2
3923   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
3924   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1);
3925   if (N0C && N1C && !N1C->isOpaque())
3926     return DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N), VT, N0C, N1C);
3927   // canonicalize constant to RHS
3928   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
3929      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
3930     return DAG.getNode(ISD::OR, SDLoc(N), VT, N1, N0);
3931   // fold (or x, 0) -> x
3932   if (isNullConstant(N1))
3933     return N0;
3934   // fold (or x, -1) -> -1
3935   if (isAllOnesConstant(N1))
3936     return N1;
3937   // fold (or x, c) -> c iff (x & ~c) == 0
3938   if (N1C && DAG.MaskedValueIsZero(N0, ~N1C->getAPIntValue()))
3939     return N1;
3940 
3941   if (SDValue Combined = visitORLike(N0, N1, N))
3942     return Combined;
3943 
3944   // Recognize halfword bswaps as (bswap + rotl 16) or (bswap + shl 16)
3945   if (SDValue BSwap = MatchBSwapHWord(N, N0, N1))
3946     return BSwap;
3947   if (SDValue BSwap = MatchBSwapHWordLow(N, N0, N1))
3948     return BSwap;
3949 
3950   // reassociate or
3951   if (SDValue ROR = ReassociateOps(ISD::OR, SDLoc(N), N0, N1))
3952     return ROR;
3953   // Canonicalize (or (and X, c1), c2) -> (and (or X, c2), c1|c2)
3954   // iff (c1 & c2) == 0.
3955   if (N1C && N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
3956              isa<ConstantSDNode>(N0.getOperand(1))) {
3957     ConstantSDNode *C1 = cast<ConstantSDNode>(N0.getOperand(1));
3958     if ((C1->getAPIntValue() & N1C->getAPIntValue()) != 0) {
3959       if (SDValue COR = DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N1), VT,
3960                                                    N1C, C1))
3961         return DAG.getNode(
3962             ISD::AND, SDLoc(N), VT,
3963             DAG.getNode(ISD::OR, SDLoc(N0), VT, N0.getOperand(0), N1), COR);
3964       return SDValue();
3965     }
3966   }
3967   // Simplify: (or (op x...), (op y...))  -> (op (or x, y))
3968   if (N0.getOpcode() == N1.getOpcode())
3969     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
3970       return Tmp;
3971 
3972   // See if this is some rotate idiom.
3973   if (SDNode *Rot = MatchRotate(N0, N1, SDLoc(N)))
3974     return SDValue(Rot, 0);
3975 
3976   if (SDValue Load = MatchLoadCombine(N))
3977     return Load;
3978 
3979   // Simplify the operands using demanded-bits information.
3980   if (!VT.isVector() &&
3981       SimplifyDemandedBits(SDValue(N, 0)))
3982     return SDValue(N, 0);
3983 
3984   return SDValue();
3985 }
3986 
3987 /// Match "(X shl/srl V1) & V2" where V2 may not be present.
3988 bool DAGCombiner::MatchRotateHalf(SDValue Op, SDValue &Shift, SDValue &Mask) {
3989   if (Op.getOpcode() == ISD::AND) {
3990     if (DAG.isConstantIntBuildVectorOrConstantInt(Op.getOperand(1))) {
3991       Mask = Op.getOperand(1);
3992       Op = Op.getOperand(0);
3993     } else {
3994       return false;
3995     }
3996   }
3997 
3998   if (Op.getOpcode() == ISD::SRL || Op.getOpcode() == ISD::SHL) {
3999     Shift = Op;
4000     return true;
4001   }
4002 
4003   return false;
4004 }
4005 
4006 // Return true if we can prove that, whenever Neg and Pos are both in the
4007 // range [0, EltSize), Neg == (Pos == 0 ? 0 : EltSize - Pos).  This means that
4008 // for two opposing shifts shift1 and shift2 and a value X with OpBits bits:
4009 //
4010 //     (or (shift1 X, Neg), (shift2 X, Pos))
4011 //
4012 // reduces to a rotate in direction shift2 by Pos or (equivalently) a rotate
4013 // in direction shift1 by Neg.  The range [0, EltSize) means that we only need
4014 // to consider shift amounts with defined behavior.
4015 static bool matchRotateSub(SDValue Pos, SDValue Neg, unsigned EltSize) {
4016   // If EltSize is a power of 2 then:
4017   //
4018   //  (a) (Pos == 0 ? 0 : EltSize - Pos) == (EltSize - Pos) & (EltSize - 1)
4019   //  (b) Neg == Neg & (EltSize - 1) whenever Neg is in [0, EltSize).
4020   //
4021   // So if EltSize is a power of 2 and Neg is (and Neg', EltSize-1), we check
4022   // for the stronger condition:
4023   //
4024   //     Neg & (EltSize - 1) == (EltSize - Pos) & (EltSize - 1)    [A]
4025   //
4026   // for all Neg and Pos.  Since Neg & (EltSize - 1) == Neg' & (EltSize - 1)
4027   // we can just replace Neg with Neg' for the rest of the function.
4028   //
4029   // In other cases we check for the even stronger condition:
4030   //
4031   //     Neg == EltSize - Pos                                    [B]
4032   //
4033   // for all Neg and Pos.  Note that the (or ...) then invokes undefined
4034   // behavior if Pos == 0 (and consequently Neg == EltSize).
4035   //
4036   // We could actually use [A] whenever EltSize is a power of 2, but the
4037   // only extra cases that it would match are those uninteresting ones
4038   // where Neg and Pos are never in range at the same time.  E.g. for
4039   // EltSize == 32, using [A] would allow a Neg of the form (sub 64, Pos)
4040   // as well as (sub 32, Pos), but:
4041   //
4042   //     (or (shift1 X, (sub 64, Pos)), (shift2 X, Pos))
4043   //
4044   // always invokes undefined behavior for 32-bit X.
4045   //
4046   // Below, Mask == EltSize - 1 when using [A] and is all-ones otherwise.
4047   unsigned MaskLoBits = 0;
4048   if (Neg.getOpcode() == ISD::AND && isPowerOf2_64(EltSize)) {
4049     if (ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(1))) {
4050       if (NegC->getAPIntValue() == EltSize - 1) {
4051         Neg = Neg.getOperand(0);
4052         MaskLoBits = Log2_64(EltSize);
4053       }
4054     }
4055   }
4056 
4057   // Check whether Neg has the form (sub NegC, NegOp1) for some NegC and NegOp1.
4058   if (Neg.getOpcode() != ISD::SUB)
4059     return false;
4060   ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(0));
4061   if (!NegC)
4062     return false;
4063   SDValue NegOp1 = Neg.getOperand(1);
4064 
4065   // On the RHS of [A], if Pos is Pos' & (EltSize - 1), just replace Pos with
4066   // Pos'.  The truncation is redundant for the purpose of the equality.
4067   if (MaskLoBits && Pos.getOpcode() == ISD::AND)
4068     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1)))
4069       if (PosC->getAPIntValue() == EltSize - 1)
4070         Pos = Pos.getOperand(0);
4071 
4072   // The condition we need is now:
4073   //
4074   //     (NegC - NegOp1) & Mask == (EltSize - Pos) & Mask
4075   //
4076   // If NegOp1 == Pos then we need:
4077   //
4078   //              EltSize & Mask == NegC & Mask
4079   //
4080   // (because "x & Mask" is a truncation and distributes through subtraction).
4081   APInt Width;
4082   if (Pos == NegOp1)
4083     Width = NegC->getAPIntValue();
4084 
4085   // Check for cases where Pos has the form (add NegOp1, PosC) for some PosC.
4086   // Then the condition we want to prove becomes:
4087   //
4088   //     (NegC - NegOp1) & Mask == (EltSize - (NegOp1 + PosC)) & Mask
4089   //
4090   // which, again because "x & Mask" is a truncation, becomes:
4091   //
4092   //                NegC & Mask == (EltSize - PosC) & Mask
4093   //             EltSize & Mask == (NegC + PosC) & Mask
4094   else if (Pos.getOpcode() == ISD::ADD && Pos.getOperand(0) == NegOp1) {
4095     if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1)))
4096       Width = PosC->getAPIntValue() + NegC->getAPIntValue();
4097     else
4098       return false;
4099   } else
4100     return false;
4101 
4102   // Now we just need to check that EltSize & Mask == Width & Mask.
4103   if (MaskLoBits)
4104     // EltSize & Mask is 0 since Mask is EltSize - 1.
4105     return Width.getLoBits(MaskLoBits) == 0;
4106   return Width == EltSize;
4107 }
4108 
4109 // A subroutine of MatchRotate used once we have found an OR of two opposite
4110 // shifts of Shifted.  If Neg == <operand size> - Pos then the OR reduces
4111 // to both (PosOpcode Shifted, Pos) and (NegOpcode Shifted, Neg), with the
4112 // former being preferred if supported.  InnerPos and InnerNeg are Pos and
4113 // Neg with outer conversions stripped away.
4114 SDNode *DAGCombiner::MatchRotatePosNeg(SDValue Shifted, SDValue Pos,
4115                                        SDValue Neg, SDValue InnerPos,
4116                                        SDValue InnerNeg, unsigned PosOpcode,
4117                                        unsigned NegOpcode, const SDLoc &DL) {
4118   // fold (or (shl x, (*ext y)),
4119   //          (srl x, (*ext (sub 32, y)))) ->
4120   //   (rotl x, y) or (rotr x, (sub 32, y))
4121   //
4122   // fold (or (shl x, (*ext (sub 32, y))),
4123   //          (srl x, (*ext y))) ->
4124   //   (rotr x, y) or (rotl x, (sub 32, y))
4125   EVT VT = Shifted.getValueType();
4126   if (matchRotateSub(InnerPos, InnerNeg, VT.getScalarSizeInBits())) {
4127     bool HasPos = TLI.isOperationLegalOrCustom(PosOpcode, VT);
4128     return DAG.getNode(HasPos ? PosOpcode : NegOpcode, DL, VT, Shifted,
4129                        HasPos ? Pos : Neg).getNode();
4130   }
4131 
4132   return nullptr;
4133 }
4134 
4135 // MatchRotate - Handle an 'or' of two operands.  If this is one of the many
4136 // idioms for rotate, and if the target supports rotation instructions, generate
4137 // a rot[lr].
4138 SDNode *DAGCombiner::MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL) {
4139   // Must be a legal type.  Expanded 'n promoted things won't work with rotates.
4140   EVT VT = LHS.getValueType();
4141   if (!TLI.isTypeLegal(VT)) return nullptr;
4142 
4143   // The target must have at least one rotate flavor.
4144   bool HasROTL = TLI.isOperationLegalOrCustom(ISD::ROTL, VT);
4145   bool HasROTR = TLI.isOperationLegalOrCustom(ISD::ROTR, VT);
4146   if (!HasROTL && !HasROTR) return nullptr;
4147 
4148   // Match "(X shl/srl V1) & V2" where V2 may not be present.
4149   SDValue LHSShift;   // The shift.
4150   SDValue LHSMask;    // AND value if any.
4151   if (!MatchRotateHalf(LHS, LHSShift, LHSMask))
4152     return nullptr; // Not part of a rotate.
4153 
4154   SDValue RHSShift;   // The shift.
4155   SDValue RHSMask;    // AND value if any.
4156   if (!MatchRotateHalf(RHS, RHSShift, RHSMask))
4157     return nullptr; // Not part of a rotate.
4158 
4159   if (LHSShift.getOperand(0) != RHSShift.getOperand(0))
4160     return nullptr;   // Not shifting the same value.
4161 
4162   if (LHSShift.getOpcode() == RHSShift.getOpcode())
4163     return nullptr;   // Shifts must disagree.
4164 
4165   // Canonicalize shl to left side in a shl/srl pair.
4166   if (RHSShift.getOpcode() == ISD::SHL) {
4167     std::swap(LHS, RHS);
4168     std::swap(LHSShift, RHSShift);
4169     std::swap(LHSMask, RHSMask);
4170   }
4171 
4172   unsigned EltSizeInBits = VT.getScalarSizeInBits();
4173   SDValue LHSShiftArg = LHSShift.getOperand(0);
4174   SDValue LHSShiftAmt = LHSShift.getOperand(1);
4175   SDValue RHSShiftArg = RHSShift.getOperand(0);
4176   SDValue RHSShiftAmt = RHSShift.getOperand(1);
4177 
4178   // fold (or (shl x, C1), (srl x, C2)) -> (rotl x, C1)
4179   // fold (or (shl x, C1), (srl x, C2)) -> (rotr x, C2)
4180   if (isConstOrConstSplat(LHSShiftAmt) && isConstOrConstSplat(RHSShiftAmt)) {
4181     uint64_t LShVal = isConstOrConstSplat(LHSShiftAmt)->getZExtValue();
4182     uint64_t RShVal = isConstOrConstSplat(RHSShiftAmt)->getZExtValue();
4183     if ((LShVal + RShVal) != EltSizeInBits)
4184       return nullptr;
4185 
4186     SDValue Rot = DAG.getNode(HasROTL ? ISD::ROTL : ISD::ROTR, DL, VT,
4187                               LHSShiftArg, HasROTL ? LHSShiftAmt : RHSShiftAmt);
4188 
4189     // If there is an AND of either shifted operand, apply it to the result.
4190     if (LHSMask.getNode() || RHSMask.getNode()) {
4191       SDValue Mask = DAG.getAllOnesConstant(DL, VT);
4192 
4193       if (LHSMask.getNode()) {
4194         APInt RHSBits = APInt::getLowBitsSet(EltSizeInBits, LShVal);
4195         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
4196                            DAG.getNode(ISD::OR, DL, VT, LHSMask,
4197                                        DAG.getConstant(RHSBits, DL, VT)));
4198       }
4199       if (RHSMask.getNode()) {
4200         APInt LHSBits = APInt::getHighBitsSet(EltSizeInBits, RShVal);
4201         Mask = DAG.getNode(ISD::AND, DL, VT, Mask,
4202                            DAG.getNode(ISD::OR, DL, VT, RHSMask,
4203                                        DAG.getConstant(LHSBits, DL, VT)));
4204       }
4205 
4206       Rot = DAG.getNode(ISD::AND, DL, VT, Rot, Mask);
4207     }
4208 
4209     return Rot.getNode();
4210   }
4211 
4212   // If there is a mask here, and we have a variable shift, we can't be sure
4213   // that we're masking out the right stuff.
4214   if (LHSMask.getNode() || RHSMask.getNode())
4215     return nullptr;
4216 
4217   // If the shift amount is sign/zext/any-extended just peel it off.
4218   SDValue LExtOp0 = LHSShiftAmt;
4219   SDValue RExtOp0 = RHSShiftAmt;
4220   if ((LHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
4221        LHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
4222        LHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
4223        LHSShiftAmt.getOpcode() == ISD::TRUNCATE) &&
4224       (RHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND ||
4225        RHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND ||
4226        RHSShiftAmt.getOpcode() == ISD::ANY_EXTEND ||
4227        RHSShiftAmt.getOpcode() == ISD::TRUNCATE)) {
4228     LExtOp0 = LHSShiftAmt.getOperand(0);
4229     RExtOp0 = RHSShiftAmt.getOperand(0);
4230   }
4231 
4232   SDNode *TryL = MatchRotatePosNeg(LHSShiftArg, LHSShiftAmt, RHSShiftAmt,
4233                                    LExtOp0, RExtOp0, ISD::ROTL, ISD::ROTR, DL);
4234   if (TryL)
4235     return TryL;
4236 
4237   SDNode *TryR = MatchRotatePosNeg(RHSShiftArg, RHSShiftAmt, LHSShiftAmt,
4238                                    RExtOp0, LExtOp0, ISD::ROTR, ISD::ROTL, DL);
4239   if (TryR)
4240     return TryR;
4241 
4242   return nullptr;
4243 }
4244 
4245 namespace {
4246 /// Helper struct to parse and store a memory address as base + index + offset.
4247 /// We ignore sign extensions when it is safe to do so.
4248 /// The following two expressions are not equivalent. To differentiate we need
4249 /// to store whether there was a sign extension involved in the index
4250 /// computation.
4251 ///  (load (i64 add (i64 copyfromreg %c)
4252 ///                 (i64 signextend (add (i8 load %index)
4253 ///                                      (i8 1))))
4254 /// vs
4255 ///
4256 /// (load (i64 add (i64 copyfromreg %c)
4257 ///                (i64 signextend (i32 add (i32 signextend (i8 load %index))
4258 ///                                         (i32 1)))))
4259 struct BaseIndexOffset {
4260   SDValue Base;
4261   SDValue Index;
4262   int64_t Offset;
4263   bool IsIndexSignExt;
4264 
4265   BaseIndexOffset() : Offset(0), IsIndexSignExt(false) {}
4266 
4267   BaseIndexOffset(SDValue Base, SDValue Index, int64_t Offset,
4268                   bool IsIndexSignExt) :
4269     Base(Base), Index(Index), Offset(Offset), IsIndexSignExt(IsIndexSignExt) {}
4270 
4271   bool equalBaseIndex(const BaseIndexOffset &Other) {
4272     return Other.Base == Base && Other.Index == Index &&
4273       Other.IsIndexSignExt == IsIndexSignExt;
4274   }
4275 
4276   /// Parses tree in Ptr for base, index, offset addresses.
4277   static BaseIndexOffset match(SDValue Ptr, SelectionDAG &DAG,
4278                                int64_t PartialOffset = 0) {
4279     bool IsIndexSignExt = false;
4280 
4281     // Split up a folded GlobalAddress+Offset into its component parts.
4282     if (GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(Ptr))
4283       if (GA->getOpcode() == ISD::GlobalAddress && GA->getOffset() != 0) {
4284         return BaseIndexOffset(DAG.getGlobalAddress(GA->getGlobal(),
4285                                                     SDLoc(GA),
4286                                                     GA->getValueType(0),
4287                                                     /*Offset=*/PartialOffset,
4288                                                     /*isTargetGA=*/false,
4289                                                     GA->getTargetFlags()),
4290                                SDValue(),
4291                                GA->getOffset(),
4292                                IsIndexSignExt);
4293       }
4294 
4295     // We only can pattern match BASE + INDEX + OFFSET. If Ptr is not an ADD
4296     // instruction, then it could be just the BASE or everything else we don't
4297     // know how to handle. Just use Ptr as BASE and give up.
4298     if (Ptr->getOpcode() != ISD::ADD)
4299       return BaseIndexOffset(Ptr, SDValue(), PartialOffset, IsIndexSignExt);
4300 
4301     // We know that we have at least an ADD instruction. Try to pattern match
4302     // the simple case of BASE + OFFSET.
4303     if (isa<ConstantSDNode>(Ptr->getOperand(1))) {
4304       int64_t Offset = cast<ConstantSDNode>(Ptr->getOperand(1))->getSExtValue();
4305       return match(Ptr->getOperand(0), DAG, Offset + PartialOffset);
4306     }
4307 
4308     // Inside a loop the current BASE pointer is calculated using an ADD and a
4309     // MUL instruction. In this case Ptr is the actual BASE pointer.
4310     // (i64 add (i64 %array_ptr)
4311     //          (i64 mul (i64 %induction_var)
4312     //                   (i64 %element_size)))
4313     if (Ptr->getOperand(1)->getOpcode() == ISD::MUL)
4314       return BaseIndexOffset(Ptr, SDValue(), PartialOffset, IsIndexSignExt);
4315 
4316     // Look at Base + Index + Offset cases.
4317     SDValue Base = Ptr->getOperand(0);
4318     SDValue IndexOffset = Ptr->getOperand(1);
4319 
4320     // Skip signextends.
4321     if (IndexOffset->getOpcode() == ISD::SIGN_EXTEND) {
4322       IndexOffset = IndexOffset->getOperand(0);
4323       IsIndexSignExt = true;
4324     }
4325 
4326     // Either the case of Base + Index (no offset) or something else.
4327     if (IndexOffset->getOpcode() != ISD::ADD)
4328       return BaseIndexOffset(Base, IndexOffset, PartialOffset, IsIndexSignExt);
4329 
4330     // Now we have the case of Base + Index + offset.
4331     SDValue Index = IndexOffset->getOperand(0);
4332     SDValue Offset = IndexOffset->getOperand(1);
4333 
4334     if (!isa<ConstantSDNode>(Offset))
4335       return BaseIndexOffset(Ptr, SDValue(), PartialOffset, IsIndexSignExt);
4336 
4337     // Ignore signextends.
4338     if (Index->getOpcode() == ISD::SIGN_EXTEND) {
4339       Index = Index->getOperand(0);
4340       IsIndexSignExt = true;
4341     } else IsIndexSignExt = false;
4342 
4343     int64_t Off = cast<ConstantSDNode>(Offset)->getSExtValue();
4344     return BaseIndexOffset(Base, Index, Off + PartialOffset, IsIndexSignExt);
4345   }
4346 };
4347 } // namespace
4348 
4349 namespace {
4350 /// Represents known origin of an individual byte in load combine pattern. The
4351 /// value of the byte is either constant zero or comes from memory.
4352 struct ByteProvider {
4353   // For constant zero providers Load is set to nullptr. For memory providers
4354   // Load represents the node which loads the byte from memory.
4355   // ByteOffset is the offset of the byte in the value produced by the load.
4356   LoadSDNode *Load;
4357   unsigned ByteOffset;
4358 
4359   ByteProvider() : Load(nullptr), ByteOffset(0) {}
4360 
4361   static ByteProvider getMemory(LoadSDNode *Load, unsigned ByteOffset) {
4362     return ByteProvider(Load, ByteOffset);
4363   }
4364   static ByteProvider getConstantZero() { return ByteProvider(nullptr, 0); }
4365 
4366   bool isConstantZero() const { return !Load; }
4367   bool isMemory() const { return Load; }
4368 
4369   bool operator==(const ByteProvider &Other) const {
4370     return Other.Load == Load && Other.ByteOffset == ByteOffset;
4371   }
4372 
4373 private:
4374   ByteProvider(LoadSDNode *Load, unsigned ByteOffset)
4375       : Load(Load), ByteOffset(ByteOffset) {}
4376 };
4377 
4378 /// Recursively traverses the expression calculating the origin of the requested
4379 /// byte of the given value. Returns None if the provider can't be calculated.
4380 ///
4381 /// For all the values except the root of the expression verifies that the value
4382 /// has exactly one use and if it's not true return None. This way if the origin
4383 /// of the byte is returned it's guaranteed that the values which contribute to
4384 /// the byte are not used outside of this expression.
4385 ///
4386 /// Because the parts of the expression are not allowed to have more than one
4387 /// use this function iterates over trees, not DAGs. So it never visits the same
4388 /// node more than once.
4389 const Optional<ByteProvider> calculateByteProvider(SDValue Op, unsigned Index,
4390                                                    unsigned Depth,
4391                                                    bool Root = false) {
4392   // Typical i64 by i8 pattern requires recursion up to 8 calls depth
4393   if (Depth == 10)
4394     return None;
4395 
4396   if (!Root && !Op.hasOneUse())
4397     return None;
4398 
4399   assert(Op.getValueType().isScalarInteger() && "can't handle other types");
4400   unsigned BitWidth = Op.getValueSizeInBits();
4401   if (BitWidth % 8 != 0)
4402     return None;
4403   unsigned ByteWidth = BitWidth / 8;
4404   assert(Index < ByteWidth && "invalid index requested");
4405   (void) ByteWidth;
4406 
4407   switch (Op.getOpcode()) {
4408   case ISD::OR: {
4409     auto LHS = calculateByteProvider(Op->getOperand(0), Index, Depth + 1);
4410     if (!LHS)
4411       return None;
4412     auto RHS = calculateByteProvider(Op->getOperand(1), Index, Depth + 1);
4413     if (!RHS)
4414       return None;
4415 
4416     if (LHS->isConstantZero())
4417       return RHS;
4418     else if (RHS->isConstantZero())
4419       return LHS;
4420     else
4421       return None;
4422   }
4423   case ISD::SHL: {
4424     auto ShiftOp = dyn_cast<ConstantSDNode>(Op->getOperand(1));
4425     if (!ShiftOp)
4426       return None;
4427 
4428     uint64_t BitShift = ShiftOp->getZExtValue();
4429     if (BitShift % 8 != 0)
4430       return None;
4431     uint64_t ByteShift = BitShift / 8;
4432 
4433     return Index < ByteShift
4434                ? ByteProvider::getConstantZero()
4435                : calculateByteProvider(Op->getOperand(0), Index - ByteShift,
4436                                        Depth + 1);
4437   }
4438   case ISD::ZERO_EXTEND: {
4439     SDValue NarrowOp = Op->getOperand(0);
4440     unsigned NarrowBitWidth = NarrowOp.getScalarValueSizeInBits();
4441     if (NarrowBitWidth % 8 != 0)
4442       return None;
4443     uint64_t NarrowByteWidth = NarrowBitWidth / 8;
4444 
4445     return Index >= NarrowByteWidth
4446                ? ByteProvider::getConstantZero()
4447                : calculateByteProvider(NarrowOp, Index, Depth + 1);
4448   }
4449   case ISD::BSWAP:
4450     return calculateByteProvider(Op->getOperand(0), ByteWidth - Index - 1,
4451                                  Depth + 1);
4452   case ISD::LOAD: {
4453     auto L = cast<LoadSDNode>(Op.getNode());
4454 
4455     // TODO: support ext loads
4456     if (L->isVolatile() || L->isIndexed() ||
4457         L->getExtensionType() != ISD::NON_EXTLOAD)
4458       return None;
4459 
4460     return ByteProvider::getMemory(L, Index);
4461   }
4462   }
4463 
4464   return None;
4465 }
4466 } // namespace
4467 
4468 /// Match a pattern where a wide type scalar value is loaded by several narrow
4469 /// loads and combined by shifts and ors. Fold it into a single load or a load
4470 /// and a BSWAP if the targets supports it.
4471 ///
4472 /// Assuming little endian target:
4473 ///  i8 *a = ...
4474 ///  i32 val = a[0] | (a[1] << 8) | (a[2] << 16) | (a[3] << 24)
4475 /// =>
4476 ///  i32 val = *((i32)a)
4477 ///
4478 ///  i8 *a = ...
4479 ///  i32 val = (a[0] << 24) | (a[1] << 16) | (a[2] << 8) | a[3]
4480 /// =>
4481 ///  i32 val = BSWAP(*((i32)a))
4482 ///
4483 /// TODO: This rule matches complex patterns with OR node roots and doesn't
4484 /// interact well with the worklist mechanism. When a part of the pattern is
4485 /// updated (e.g. one of the loads) its direct users are put into the worklist,
4486 /// but the root node of the pattern which triggers the load combine is not
4487 /// necessarily a direct user of the changed node. For example, once the address
4488 /// of t28 load is reassociated load combine won't be triggered:
4489 ///             t25: i32 = add t4, Constant:i32<2>
4490 ///           t26: i64 = sign_extend t25
4491 ///        t27: i64 = add t2, t26
4492 ///       t28: i8,ch = load<LD1[%tmp9]> t0, t27, undef:i64
4493 ///     t29: i32 = zero_extend t28
4494 ///   t32: i32 = shl t29, Constant:i8<8>
4495 /// t33: i32 = or t23, t32
4496 /// As a possible fix visitLoad can check if the load can be a part of a load
4497 /// combine pattern and add corresponding OR roots to the worklist.
4498 SDValue DAGCombiner::MatchLoadCombine(SDNode *N) {
4499   assert(N->getOpcode() == ISD::OR &&
4500          "Can only match load combining against OR nodes");
4501 
4502   // Handles simple types only
4503   EVT VT = N->getValueType(0);
4504   if (VT != MVT::i16 && VT != MVT::i32 && VT != MVT::i64)
4505     return SDValue();
4506   unsigned ByteWidth = VT.getSizeInBits() / 8;
4507 
4508   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
4509   // Before legalize we can introduce too wide illegal loads which will be later
4510   // split into legal sized loads. This enables us to combine i64 load by i8
4511   // patterns to a couple of i32 loads on 32 bit targets.
4512   if (LegalOperations && !TLI.isOperationLegal(ISD::LOAD, VT))
4513     return SDValue();
4514 
4515   std::function<unsigned(unsigned, unsigned)> LittleEndianByteAt = [](
4516     unsigned BW, unsigned i) { return i; };
4517   std::function<unsigned(unsigned, unsigned)> BigEndianByteAt = [](
4518     unsigned BW, unsigned i) { return BW - i - 1; };
4519 
4520   Optional<BaseIndexOffset> Base;
4521   SDValue Chain;
4522 
4523   SmallSet<LoadSDNode *, 8> Loads;
4524   LoadSDNode *FirstLoad = nullptr;
4525   int64_t FirstOffset = INT64_MAX;
4526 
4527   bool IsBigEndianTarget = DAG.getDataLayout().isBigEndian();
4528   auto ByteAt = IsBigEndianTarget ? BigEndianByteAt : LittleEndianByteAt;
4529 
4530   // Check if all the bytes of the OR we are looking at are loaded from the same
4531   // base address. Collect bytes offsets from Base address in ByteOffsets.
4532   SmallVector<int64_t, 4> ByteOffsets(ByteWidth);
4533   for (unsigned i = 0; i < ByteWidth; i++) {
4534     auto P = calculateByteProvider(SDValue(N, 0), i, 0, /*Root=*/true);
4535     if (!P || !P->isMemory()) // All the bytes must be loaded from memory
4536       return SDValue();
4537 
4538     LoadSDNode *L = P->Load;
4539     assert(L->hasNUsesOfValue(1, 0) && !L->isVolatile() && !L->isIndexed() &&
4540            (L->getExtensionType() == ISD::NON_EXTLOAD) &&
4541            "Must be enforced by calculateByteProvider");
4542     assert(L->getOffset().isUndef() && "Unindexed load must have undef offset");
4543 
4544     // All loads must share the same chain
4545     SDValue LChain = L->getChain();
4546     if (!Chain)
4547       Chain = LChain;
4548     else if (Chain != LChain)
4549       return SDValue();
4550 
4551     // Loads must share the same base address
4552     BaseIndexOffset Ptr = BaseIndexOffset::match(L->getBasePtr(), DAG);
4553     if (!Base)
4554       Base = Ptr;
4555     else if (!Base->equalBaseIndex(Ptr))
4556       return SDValue();
4557 
4558     // Calculate the offset of the current byte from the base address
4559     unsigned LoadBitWidth = L->getMemoryVT().getSizeInBits();
4560     assert(LoadBitWidth % 8 == 0 &&
4561            "can only analyze providers for individual bytes not bit");
4562     unsigned LoadByteWidth = LoadBitWidth / 8;
4563     int64_t MemoryByteOffset = ByteAt(LoadByteWidth, P->ByteOffset);
4564     int64_t ByteOffsetFromBase = Ptr.Offset + MemoryByteOffset;
4565     ByteOffsets[i] = ByteOffsetFromBase;
4566 
4567     // Remember the first byte load
4568     if (ByteOffsetFromBase < FirstOffset) {
4569       FirstLoad = L;
4570       FirstOffset = ByteOffsetFromBase;
4571     }
4572 
4573     Loads.insert(L);
4574   }
4575   assert(Loads.size() > 0 && "All the bytes of the value must be loaded from "
4576          "memory, so there must be at least one load which produces the value");
4577   assert(Base && "Base address of the accessed memory location must be set");
4578   assert(FirstOffset != INT64_MAX && "First byte offset must be set");
4579 
4580   // Check if the bytes of the OR we are looking at match with either big or
4581   // little endian value load
4582   bool BigEndian = true, LittleEndian = true;
4583   for (unsigned i = 0; i < ByteWidth; i++) {
4584     int64_t CurrentByteOffset = ByteOffsets[i] - FirstOffset;
4585     LittleEndian &= CurrentByteOffset == LittleEndianByteAt(ByteWidth, i);
4586     BigEndian &= CurrentByteOffset == BigEndianByteAt(ByteWidth, i);
4587     if (!BigEndian && !LittleEndian)
4588       return SDValue();
4589   }
4590   assert((BigEndian != LittleEndian) && "should be either or");
4591   assert(FirstLoad && "must be set");
4592 
4593   // The node we are looking at matches with the pattern, check if we can
4594   // replace it with a single load and bswap if needed.
4595 
4596   // If the load needs byte swap check if the target supports it
4597   bool NeedsBswap = IsBigEndianTarget != BigEndian;
4598 
4599   // Before legalize we can introduce illegal bswaps which will be later
4600   // converted to an explicit bswap sequence. This way we end up with a single
4601   // load and byte shuffling instead of several loads and byte shuffling.
4602   if (NeedsBswap && LegalOperations && !TLI.isOperationLegal(ISD::BSWAP, VT))
4603     return SDValue();
4604 
4605   // Check that a load of the wide type is both allowed and fast on the target
4606   bool Fast = false;
4607   bool Allowed = TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(),
4608                                         VT, FirstLoad->getAddressSpace(),
4609                                         FirstLoad->getAlignment(), &Fast);
4610   if (!Allowed || !Fast)
4611     return SDValue();
4612 
4613   SDValue NewLoad =
4614       DAG.getLoad(VT, SDLoc(N), Chain, FirstLoad->getBasePtr(),
4615                   FirstLoad->getPointerInfo(), FirstLoad->getAlignment());
4616 
4617   // Transfer chain users from old loads to the new load.
4618   for (LoadSDNode *L : Loads)
4619     DAG.ReplaceAllUsesOfValueWith(SDValue(L, 1), SDValue(NewLoad.getNode(), 1));
4620 
4621   return NeedsBswap ? DAG.getNode(ISD::BSWAP, SDLoc(N), VT, NewLoad) : NewLoad;
4622 }
4623 
4624 SDValue DAGCombiner::visitXOR(SDNode *N) {
4625   SDValue N0 = N->getOperand(0);
4626   SDValue N1 = N->getOperand(1);
4627   EVT VT = N0.getValueType();
4628 
4629   // fold vector ops
4630   if (VT.isVector()) {
4631     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4632       return FoldedVOp;
4633 
4634     // fold (xor x, 0) -> x, vector edition
4635     if (ISD::isBuildVectorAllZeros(N0.getNode()))
4636       return N1;
4637     if (ISD::isBuildVectorAllZeros(N1.getNode()))
4638       return N0;
4639   }
4640 
4641   // fold (xor undef, undef) -> 0. This is a common idiom (misuse).
4642   if (N0.isUndef() && N1.isUndef())
4643     return DAG.getConstant(0, SDLoc(N), VT);
4644   // fold (xor x, undef) -> undef
4645   if (N0.isUndef())
4646     return N0;
4647   if (N1.isUndef())
4648     return N1;
4649   // fold (xor c1, c2) -> c1^c2
4650   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4651   ConstantSDNode *N1C = getAsNonOpaqueConstant(N1);
4652   if (N0C && N1C)
4653     return DAG.FoldConstantArithmetic(ISD::XOR, SDLoc(N), VT, N0C, N1C);
4654   // canonicalize constant to RHS
4655   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
4656      !DAG.isConstantIntBuildVectorOrConstantInt(N1))
4657     return DAG.getNode(ISD::XOR, SDLoc(N), VT, N1, N0);
4658   // fold (xor x, 0) -> x
4659   if (isNullConstant(N1))
4660     return N0;
4661   // reassociate xor
4662   if (SDValue RXOR = ReassociateOps(ISD::XOR, SDLoc(N), N0, N1))
4663     return RXOR;
4664 
4665   // fold !(x cc y) -> (x !cc y)
4666   SDValue LHS, RHS, CC;
4667   if (TLI.isConstTrueVal(N1.getNode()) && isSetCCEquivalent(N0, LHS, RHS, CC)) {
4668     bool isInt = LHS.getValueType().isInteger();
4669     ISD::CondCode NotCC = ISD::getSetCCInverse(cast<CondCodeSDNode>(CC)->get(),
4670                                                isInt);
4671 
4672     if (!LegalOperations ||
4673         TLI.isCondCodeLegal(NotCC, LHS.getSimpleValueType())) {
4674       switch (N0.getOpcode()) {
4675       default:
4676         llvm_unreachable("Unhandled SetCC Equivalent!");
4677       case ISD::SETCC:
4678         return DAG.getSetCC(SDLoc(N), VT, LHS, RHS, NotCC);
4679       case ISD::SELECT_CC:
4680         return DAG.getSelectCC(SDLoc(N), LHS, RHS, N0.getOperand(2),
4681                                N0.getOperand(3), NotCC);
4682       }
4683     }
4684   }
4685 
4686   // fold (not (zext (setcc x, y))) -> (zext (not (setcc x, y)))
4687   if (isOneConstant(N1) && N0.getOpcode() == ISD::ZERO_EXTEND &&
4688       N0.getNode()->hasOneUse() &&
4689       isSetCCEquivalent(N0.getOperand(0), LHS, RHS, CC)){
4690     SDValue V = N0.getOperand(0);
4691     SDLoc DL(N0);
4692     V = DAG.getNode(ISD::XOR, DL, V.getValueType(), V,
4693                     DAG.getConstant(1, DL, V.getValueType()));
4694     AddToWorklist(V.getNode());
4695     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, V);
4696   }
4697 
4698   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are setcc
4699   if (isOneConstant(N1) && VT == MVT::i1 &&
4700       (N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::AND)) {
4701     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
4702     if (isOneUseSetCC(RHS) || isOneUseSetCC(LHS)) {
4703       unsigned NewOpcode = N0.getOpcode() == ISD::AND ? ISD::OR : ISD::AND;
4704       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
4705       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
4706       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
4707       return DAG.getNode(NewOpcode, SDLoc(N), VT, LHS, RHS);
4708     }
4709   }
4710   // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are constants
4711   if (isAllOnesConstant(N1) &&
4712       (N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::AND)) {
4713     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
4714     if (isa<ConstantSDNode>(RHS) || isa<ConstantSDNode>(LHS)) {
4715       unsigned NewOpcode = N0.getOpcode() == ISD::AND ? ISD::OR : ISD::AND;
4716       LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS
4717       RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS
4718       AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode());
4719       return DAG.getNode(NewOpcode, SDLoc(N), VT, LHS, RHS);
4720     }
4721   }
4722   // fold (xor (and x, y), y) -> (and (not x), y)
4723   if (N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
4724       N0->getOperand(1) == N1) {
4725     SDValue X = N0->getOperand(0);
4726     SDValue NotX = DAG.getNOT(SDLoc(X), X, VT);
4727     AddToWorklist(NotX.getNode());
4728     return DAG.getNode(ISD::AND, SDLoc(N), VT, NotX, N1);
4729   }
4730   // fold (xor (xor x, c1), c2) -> (xor x, (xor c1, c2))
4731   if (N1C && N0.getOpcode() == ISD::XOR) {
4732     if (const ConstantSDNode *N00C = getAsNonOpaqueConstant(N0.getOperand(0))) {
4733       SDLoc DL(N);
4734       return DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(1),
4735                          DAG.getConstant(N1C->getAPIntValue() ^
4736                                          N00C->getAPIntValue(), DL, VT));
4737     }
4738     if (const ConstantSDNode *N01C = getAsNonOpaqueConstant(N0.getOperand(1))) {
4739       SDLoc DL(N);
4740       return DAG.getNode(ISD::XOR, DL, VT, N0.getOperand(0),
4741                          DAG.getConstant(N1C->getAPIntValue() ^
4742                                          N01C->getAPIntValue(), DL, VT));
4743     }
4744   }
4745   // fold (xor x, x) -> 0
4746   if (N0 == N1)
4747     return tryFoldToZero(SDLoc(N), TLI, VT, DAG, LegalOperations, LegalTypes);
4748 
4749   // fold (xor (shl 1, x), -1) -> (rotl ~1, x)
4750   // Here is a concrete example of this equivalence:
4751   // i16   x ==  14
4752   // i16 shl ==   1 << 14  == 16384 == 0b0100000000000000
4753   // i16 xor == ~(1 << 14) == 49151 == 0b1011111111111111
4754   //
4755   // =>
4756   //
4757   // i16     ~1      == 0b1111111111111110
4758   // i16 rol(~1, 14) == 0b1011111111111111
4759   //
4760   // Some additional tips to help conceptualize this transform:
4761   // - Try to see the operation as placing a single zero in a value of all ones.
4762   // - There exists no value for x which would allow the result to contain zero.
4763   // - Values of x larger than the bitwidth are undefined and do not require a
4764   //   consistent result.
4765   // - Pushing the zero left requires shifting one bits in from the right.
4766   // A rotate left of ~1 is a nice way of achieving the desired result.
4767   if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT) && N0.getOpcode() == ISD::SHL
4768       && isAllOnesConstant(N1) && isOneConstant(N0.getOperand(0))) {
4769     SDLoc DL(N);
4770     return DAG.getNode(ISD::ROTL, DL, VT, DAG.getConstant(~1, DL, VT),
4771                        N0.getOperand(1));
4772   }
4773 
4774   // Simplify: xor (op x...), (op y...)  -> (op (xor x, y))
4775   if (N0.getOpcode() == N1.getOpcode())
4776     if (SDValue Tmp = SimplifyBinOpWithSameOpcodeHands(N))
4777       return Tmp;
4778 
4779   // Simplify the expression using non-local knowledge.
4780   if (!VT.isVector() &&
4781       SimplifyDemandedBits(SDValue(N, 0)))
4782     return SDValue(N, 0);
4783 
4784   return SDValue();
4785 }
4786 
4787 /// Handle transforms common to the three shifts, when the shift amount is a
4788 /// constant.
4789 SDValue DAGCombiner::visitShiftByConstant(SDNode *N, ConstantSDNode *Amt) {
4790   SDNode *LHS = N->getOperand(0).getNode();
4791   if (!LHS->hasOneUse()) return SDValue();
4792 
4793   // We want to pull some binops through shifts, so that we have (and (shift))
4794   // instead of (shift (and)), likewise for add, or, xor, etc.  This sort of
4795   // thing happens with address calculations, so it's important to canonicalize
4796   // it.
4797   bool HighBitSet = false;  // Can we transform this if the high bit is set?
4798 
4799   switch (LHS->getOpcode()) {
4800   default: return SDValue();
4801   case ISD::OR:
4802   case ISD::XOR:
4803     HighBitSet = false; // We can only transform sra if the high bit is clear.
4804     break;
4805   case ISD::AND:
4806     HighBitSet = true;  // We can only transform sra if the high bit is set.
4807     break;
4808   case ISD::ADD:
4809     if (N->getOpcode() != ISD::SHL)
4810       return SDValue(); // only shl(add) not sr[al](add).
4811     HighBitSet = false; // We can only transform sra if the high bit is clear.
4812     break;
4813   }
4814 
4815   // We require the RHS of the binop to be a constant and not opaque as well.
4816   ConstantSDNode *BinOpCst = getAsNonOpaqueConstant(LHS->getOperand(1));
4817   if (!BinOpCst) return SDValue();
4818 
4819   // FIXME: disable this unless the input to the binop is a shift by a constant
4820   // or is copy/select.Enable this in other cases when figure out it's exactly profitable.
4821   SDNode *BinOpLHSVal = LHS->getOperand(0).getNode();
4822   bool isShift = BinOpLHSVal->getOpcode() == ISD::SHL ||
4823                  BinOpLHSVal->getOpcode() == ISD::SRA ||
4824                  BinOpLHSVal->getOpcode() == ISD::SRL;
4825   bool isCopyOrSelect = BinOpLHSVal->getOpcode() == ISD::CopyFromReg ||
4826                         BinOpLHSVal->getOpcode() == ISD::SELECT;
4827 
4828   if ((!isShift || !isa<ConstantSDNode>(BinOpLHSVal->getOperand(1))) &&
4829       !isCopyOrSelect)
4830     return SDValue();
4831 
4832   if (isCopyOrSelect && N->hasOneUse())
4833     return SDValue();
4834 
4835   EVT VT = N->getValueType(0);
4836 
4837   // If this is a signed shift right, and the high bit is modified by the
4838   // logical operation, do not perform the transformation. The highBitSet
4839   // boolean indicates the value of the high bit of the constant which would
4840   // cause it to be modified for this operation.
4841   if (N->getOpcode() == ISD::SRA) {
4842     bool BinOpRHSSignSet = BinOpCst->getAPIntValue().isNegative();
4843     if (BinOpRHSSignSet != HighBitSet)
4844       return SDValue();
4845   }
4846 
4847   if (!TLI.isDesirableToCommuteWithShift(LHS))
4848     return SDValue();
4849 
4850   // Fold the constants, shifting the binop RHS by the shift amount.
4851   SDValue NewRHS = DAG.getNode(N->getOpcode(), SDLoc(LHS->getOperand(1)),
4852                                N->getValueType(0),
4853                                LHS->getOperand(1), N->getOperand(1));
4854   assert(isa<ConstantSDNode>(NewRHS) && "Folding was not successful!");
4855 
4856   // Create the new shift.
4857   SDValue NewShift = DAG.getNode(N->getOpcode(),
4858                                  SDLoc(LHS->getOperand(0)),
4859                                  VT, LHS->getOperand(0), N->getOperand(1));
4860 
4861   // Create the new binop.
4862   return DAG.getNode(LHS->getOpcode(), SDLoc(N), VT, NewShift, NewRHS);
4863 }
4864 
4865 SDValue DAGCombiner::distributeTruncateThroughAnd(SDNode *N) {
4866   assert(N->getOpcode() == ISD::TRUNCATE);
4867   assert(N->getOperand(0).getOpcode() == ISD::AND);
4868 
4869   // (truncate:TruncVT (and N00, N01C)) -> (and (truncate:TruncVT N00), TruncC)
4870   if (N->hasOneUse() && N->getOperand(0).hasOneUse()) {
4871     SDValue N01 = N->getOperand(0).getOperand(1);
4872     if (isConstantOrConstantVector(N01, /* NoOpaques */ true)) {
4873       SDLoc DL(N);
4874       EVT TruncVT = N->getValueType(0);
4875       SDValue N00 = N->getOperand(0).getOperand(0);
4876       SDValue Trunc00 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N00);
4877       SDValue Trunc01 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N01);
4878       AddToWorklist(Trunc00.getNode());
4879       AddToWorklist(Trunc01.getNode());
4880       return DAG.getNode(ISD::AND, DL, TruncVT, Trunc00, Trunc01);
4881     }
4882   }
4883 
4884   return SDValue();
4885 }
4886 
4887 SDValue DAGCombiner::visitRotate(SDNode *N) {
4888   // fold (rot* x, (trunc (and y, c))) -> (rot* x, (and (trunc y), (trunc c))).
4889   if (N->getOperand(1).getOpcode() == ISD::TRUNCATE &&
4890       N->getOperand(1).getOperand(0).getOpcode() == ISD::AND) {
4891     if (SDValue NewOp1 =
4892             distributeTruncateThroughAnd(N->getOperand(1).getNode()))
4893       return DAG.getNode(N->getOpcode(), SDLoc(N), N->getValueType(0),
4894                          N->getOperand(0), NewOp1);
4895   }
4896   return SDValue();
4897 }
4898 
4899 SDValue DAGCombiner::visitSHL(SDNode *N) {
4900   SDValue N0 = N->getOperand(0);
4901   SDValue N1 = N->getOperand(1);
4902   EVT VT = N0.getValueType();
4903   unsigned OpSizeInBits = VT.getScalarSizeInBits();
4904 
4905   // fold vector ops
4906   if (VT.isVector()) {
4907     if (SDValue FoldedVOp = SimplifyVBinOp(N))
4908       return FoldedVOp;
4909 
4910     BuildVectorSDNode *N1CV = dyn_cast<BuildVectorSDNode>(N1);
4911     // If setcc produces all-one true value then:
4912     // (shl (and (setcc) N01CV) N1CV) -> (and (setcc) N01CV<<N1CV)
4913     if (N1CV && N1CV->isConstant()) {
4914       if (N0.getOpcode() == ISD::AND) {
4915         SDValue N00 = N0->getOperand(0);
4916         SDValue N01 = N0->getOperand(1);
4917         BuildVectorSDNode *N01CV = dyn_cast<BuildVectorSDNode>(N01);
4918 
4919         if (N01CV && N01CV->isConstant() && N00.getOpcode() == ISD::SETCC &&
4920             TLI.getBooleanContents(N00.getOperand(0).getValueType()) ==
4921                 TargetLowering::ZeroOrNegativeOneBooleanContent) {
4922           if (SDValue C = DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT,
4923                                                      N01CV, N1CV))
4924             return DAG.getNode(ISD::AND, SDLoc(N), VT, N00, C);
4925         }
4926       }
4927     }
4928   }
4929 
4930   ConstantSDNode *N1C = isConstOrConstSplat(N1);
4931 
4932   // fold (shl c1, c2) -> c1<<c2
4933   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
4934   if (N0C && N1C && !N1C->isOpaque())
4935     return DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT, N0C, N1C);
4936   // fold (shl 0, x) -> 0
4937   if (isNullConstant(N0))
4938     return N0;
4939   // fold (shl x, c >= size(x)) -> undef
4940   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
4941     return DAG.getUNDEF(VT);
4942   // fold (shl x, 0) -> x
4943   if (N1C && N1C->isNullValue())
4944     return N0;
4945   // fold (shl undef, x) -> 0
4946   if (N0.isUndef())
4947     return DAG.getConstant(0, SDLoc(N), VT);
4948   // if (shl x, c) is known to be zero, return 0
4949   if (DAG.MaskedValueIsZero(SDValue(N, 0),
4950                             APInt::getAllOnesValue(OpSizeInBits)))
4951     return DAG.getConstant(0, SDLoc(N), VT);
4952   // fold (shl x, (trunc (and y, c))) -> (shl x, (and (trunc y), (trunc c))).
4953   if (N1.getOpcode() == ISD::TRUNCATE &&
4954       N1.getOperand(0).getOpcode() == ISD::AND) {
4955     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
4956       return DAG.getNode(ISD::SHL, SDLoc(N), VT, N0, NewOp1);
4957   }
4958 
4959   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
4960     return SDValue(N, 0);
4961 
4962   // fold (shl (shl x, c1), c2) -> 0 or (shl x, (add c1, c2))
4963   if (N1C && N0.getOpcode() == ISD::SHL) {
4964     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
4965       SDLoc DL(N);
4966       APInt c1 = N0C1->getAPIntValue();
4967       APInt c2 = N1C->getAPIntValue();
4968       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4969 
4970       APInt Sum = c1 + c2;
4971       if (Sum.uge(OpSizeInBits))
4972         return DAG.getConstant(0, DL, VT);
4973 
4974       return DAG.getNode(
4975           ISD::SHL, DL, VT, N0.getOperand(0),
4976           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
4977     }
4978   }
4979 
4980   // fold (shl (ext (shl x, c1)), c2) -> (ext (shl x, (add c1, c2)))
4981   // For this to be valid, the second form must not preserve any of the bits
4982   // that are shifted out by the inner shift in the first form.  This means
4983   // the outer shift size must be >= the number of bits added by the ext.
4984   // As a corollary, we don't care what kind of ext it is.
4985   if (N1C && (N0.getOpcode() == ISD::ZERO_EXTEND ||
4986               N0.getOpcode() == ISD::ANY_EXTEND ||
4987               N0.getOpcode() == ISD::SIGN_EXTEND) &&
4988       N0.getOperand(0).getOpcode() == ISD::SHL) {
4989     SDValue N0Op0 = N0.getOperand(0);
4990     if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) {
4991       APInt c1 = N0Op0C1->getAPIntValue();
4992       APInt c2 = N1C->getAPIntValue();
4993       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
4994 
4995       EVT InnerShiftVT = N0Op0.getValueType();
4996       uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits();
4997       if (c2.uge(OpSizeInBits - InnerShiftSize)) {
4998         SDLoc DL(N0);
4999         APInt Sum = c1 + c2;
5000         if (Sum.uge(OpSizeInBits))
5001           return DAG.getConstant(0, DL, VT);
5002 
5003         return DAG.getNode(
5004             ISD::SHL, DL, VT,
5005             DAG.getNode(N0.getOpcode(), DL, VT, N0Op0->getOperand(0)),
5006             DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
5007       }
5008     }
5009   }
5010 
5011   // fold (shl (zext (srl x, C)), C) -> (zext (shl (srl x, C), C))
5012   // Only fold this if the inner zext has no other uses to avoid increasing
5013   // the total number of instructions.
5014   if (N1C && N0.getOpcode() == ISD::ZERO_EXTEND && N0.hasOneUse() &&
5015       N0.getOperand(0).getOpcode() == ISD::SRL) {
5016     SDValue N0Op0 = N0.getOperand(0);
5017     if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) {
5018       if (N0Op0C1->getAPIntValue().ult(VT.getScalarSizeInBits())) {
5019         uint64_t c1 = N0Op0C1->getZExtValue();
5020         uint64_t c2 = N1C->getZExtValue();
5021         if (c1 == c2) {
5022           SDValue NewOp0 = N0.getOperand(0);
5023           EVT CountVT = NewOp0.getOperand(1).getValueType();
5024           SDLoc DL(N);
5025           SDValue NewSHL = DAG.getNode(ISD::SHL, DL, NewOp0.getValueType(),
5026                                        NewOp0,
5027                                        DAG.getConstant(c2, DL, CountVT));
5028           AddToWorklist(NewSHL.getNode());
5029           return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N0), VT, NewSHL);
5030         }
5031       }
5032     }
5033   }
5034 
5035   // fold (shl (sr[la] exact X,  C1), C2) -> (shl    X, (C2-C1)) if C1 <= C2
5036   // fold (shl (sr[la] exact X,  C1), C2) -> (sr[la] X, (C2-C1)) if C1  > C2
5037   if (N1C && (N0.getOpcode() == ISD::SRL || N0.getOpcode() == ISD::SRA) &&
5038       cast<BinaryWithFlagsSDNode>(N0)->Flags.hasExact()) {
5039     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
5040       uint64_t C1 = N0C1->getZExtValue();
5041       uint64_t C2 = N1C->getZExtValue();
5042       SDLoc DL(N);
5043       if (C1 <= C2)
5044         return DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
5045                            DAG.getConstant(C2 - C1, DL, N1.getValueType()));
5046       return DAG.getNode(N0.getOpcode(), DL, VT, N0.getOperand(0),
5047                          DAG.getConstant(C1 - C2, DL, N1.getValueType()));
5048     }
5049   }
5050 
5051   // fold (shl (srl x, c1), c2) -> (and (shl x, (sub c2, c1), MASK) or
5052   //                               (and (srl x, (sub c1, c2), MASK)
5053   // Only fold this if the inner shift has no other uses -- if it does, folding
5054   // this will increase the total number of instructions.
5055   if (N1C && N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
5056     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
5057       uint64_t c1 = N0C1->getZExtValue();
5058       if (c1 < OpSizeInBits) {
5059         uint64_t c2 = N1C->getZExtValue();
5060         APInt Mask = APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - c1);
5061         SDValue Shift;
5062         if (c2 > c1) {
5063           Mask = Mask.shl(c2 - c1);
5064           SDLoc DL(N);
5065           Shift = DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0),
5066                               DAG.getConstant(c2 - c1, DL, N1.getValueType()));
5067         } else {
5068           Mask = Mask.lshr(c1 - c2);
5069           SDLoc DL(N);
5070           Shift = DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0),
5071                               DAG.getConstant(c1 - c2, DL, N1.getValueType()));
5072         }
5073         SDLoc DL(N0);
5074         return DAG.getNode(ISD::AND, DL, VT, Shift,
5075                            DAG.getConstant(Mask, DL, VT));
5076       }
5077     }
5078   }
5079 
5080   // fold (shl (sra x, c1), c1) -> (and x, (shl -1, c1))
5081   if (N0.getOpcode() == ISD::SRA && N1 == N0.getOperand(1) &&
5082       isConstantOrConstantVector(N1, /* No Opaques */ true)) {
5083     SDLoc DL(N);
5084     SDValue AllBits = DAG.getAllOnesConstant(DL, VT);
5085     SDValue HiBitsMask = DAG.getNode(ISD::SHL, DL, VT, AllBits, N1);
5086     return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), HiBitsMask);
5087   }
5088 
5089   // fold (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
5090   // Variant of version done on multiply, except mul by a power of 2 is turned
5091   // into a shift.
5092   if (N0.getOpcode() == ISD::ADD && N0.getNode()->hasOneUse() &&
5093       isConstantOrConstantVector(N1, /* No Opaques */ true) &&
5094       isConstantOrConstantVector(N0.getOperand(1), /* No Opaques */ true)) {
5095     SDValue Shl0 = DAG.getNode(ISD::SHL, SDLoc(N0), VT, N0.getOperand(0), N1);
5096     SDValue Shl1 = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1);
5097     AddToWorklist(Shl0.getNode());
5098     AddToWorklist(Shl1.getNode());
5099     return DAG.getNode(ISD::ADD, SDLoc(N), VT, Shl0, Shl1);
5100   }
5101 
5102   // fold (shl (mul x, c1), c2) -> (mul x, c1 << c2)
5103   if (N0.getOpcode() == ISD::MUL && N0.getNode()->hasOneUse() &&
5104       isConstantOrConstantVector(N1, /* No Opaques */ true) &&
5105       isConstantOrConstantVector(N0.getOperand(1), /* No Opaques */ true)) {
5106     SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1);
5107     if (isConstantOrConstantVector(Shl))
5108       return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), Shl);
5109   }
5110 
5111   if (N1C && !N1C->isOpaque())
5112     if (SDValue NewSHL = visitShiftByConstant(N, N1C))
5113       return NewSHL;
5114 
5115   return SDValue();
5116 }
5117 
5118 SDValue DAGCombiner::visitSRA(SDNode *N) {
5119   SDValue N0 = N->getOperand(0);
5120   SDValue N1 = N->getOperand(1);
5121   EVT VT = N0.getValueType();
5122   unsigned OpSizeInBits = VT.getScalarSizeInBits();
5123 
5124   // Arithmetic shifting an all-sign-bit value is a no-op.
5125   if (DAG.ComputeNumSignBits(N0) == OpSizeInBits)
5126     return N0;
5127 
5128   // fold vector ops
5129   if (VT.isVector())
5130     if (SDValue FoldedVOp = SimplifyVBinOp(N))
5131       return FoldedVOp;
5132 
5133   ConstantSDNode *N1C = isConstOrConstSplat(N1);
5134 
5135   // fold (sra c1, c2) -> (sra c1, c2)
5136   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
5137   if (N0C && N1C && !N1C->isOpaque())
5138     return DAG.FoldConstantArithmetic(ISD::SRA, SDLoc(N), VT, N0C, N1C);
5139   // fold (sra 0, x) -> 0
5140   if (isNullConstant(N0))
5141     return N0;
5142   // fold (sra -1, x) -> -1
5143   if (isAllOnesConstant(N0))
5144     return N0;
5145   // fold (sra x, c >= size(x)) -> undef
5146   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
5147     return DAG.getUNDEF(VT);
5148   // fold (sra x, 0) -> x
5149   if (N1C && N1C->isNullValue())
5150     return N0;
5151   // fold (sra (shl x, c1), c1) -> sext_inreg for some c1 and target supports
5152   // sext_inreg.
5153   if (N1C && N0.getOpcode() == ISD::SHL && N1 == N0.getOperand(1)) {
5154     unsigned LowBits = OpSizeInBits - (unsigned)N1C->getZExtValue();
5155     EVT ExtVT = EVT::getIntegerVT(*DAG.getContext(), LowBits);
5156     if (VT.isVector())
5157       ExtVT = EVT::getVectorVT(*DAG.getContext(),
5158                                ExtVT, VT.getVectorNumElements());
5159     if ((!LegalOperations ||
5160          TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, ExtVT)))
5161       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
5162                          N0.getOperand(0), DAG.getValueType(ExtVT));
5163   }
5164 
5165   // fold (sra (sra x, c1), c2) -> (sra x, (add c1, c2))
5166   if (N1C && N0.getOpcode() == ISD::SRA) {
5167     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
5168       SDLoc DL(N);
5169       APInt c1 = N0C1->getAPIntValue();
5170       APInt c2 = N1C->getAPIntValue();
5171       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
5172 
5173       APInt Sum = c1 + c2;
5174       if (Sum.uge(OpSizeInBits))
5175         Sum = APInt(OpSizeInBits, OpSizeInBits - 1);
5176 
5177       return DAG.getNode(
5178           ISD::SRA, DL, VT, N0.getOperand(0),
5179           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
5180     }
5181   }
5182 
5183   // fold (sra (shl X, m), (sub result_size, n))
5184   // -> (sign_extend (trunc (shl X, (sub (sub result_size, n), m)))) for
5185   // result_size - n != m.
5186   // If truncate is free for the target sext(shl) is likely to result in better
5187   // code.
5188   if (N0.getOpcode() == ISD::SHL && N1C) {
5189     // Get the two constanst of the shifts, CN0 = m, CN = n.
5190     const ConstantSDNode *N01C = isConstOrConstSplat(N0.getOperand(1));
5191     if (N01C) {
5192       LLVMContext &Ctx = *DAG.getContext();
5193       // Determine what the truncate's result bitsize and type would be.
5194       EVT TruncVT = EVT::getIntegerVT(Ctx, OpSizeInBits - N1C->getZExtValue());
5195 
5196       if (VT.isVector())
5197         TruncVT = EVT::getVectorVT(Ctx, TruncVT, VT.getVectorNumElements());
5198 
5199       // Determine the residual right-shift amount.
5200       int ShiftAmt = N1C->getZExtValue() - N01C->getZExtValue();
5201 
5202       // If the shift is not a no-op (in which case this should be just a sign
5203       // extend already), the truncated to type is legal, sign_extend is legal
5204       // on that type, and the truncate to that type is both legal and free,
5205       // perform the transform.
5206       if ((ShiftAmt > 0) &&
5207           TLI.isOperationLegalOrCustom(ISD::SIGN_EXTEND, TruncVT) &&
5208           TLI.isOperationLegalOrCustom(ISD::TRUNCATE, VT) &&
5209           TLI.isTruncateFree(VT, TruncVT)) {
5210 
5211         SDLoc DL(N);
5212         SDValue Amt = DAG.getConstant(ShiftAmt, DL,
5213             getShiftAmountTy(N0.getOperand(0).getValueType()));
5214         SDValue Shift = DAG.getNode(ISD::SRL, DL, VT,
5215                                     N0.getOperand(0), Amt);
5216         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, TruncVT,
5217                                     Shift);
5218         return DAG.getNode(ISD::SIGN_EXTEND, DL,
5219                            N->getValueType(0), Trunc);
5220       }
5221     }
5222   }
5223 
5224   // fold (sra x, (trunc (and y, c))) -> (sra x, (and (trunc y), (trunc c))).
5225   if (N1.getOpcode() == ISD::TRUNCATE &&
5226       N1.getOperand(0).getOpcode() == ISD::AND) {
5227     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
5228       return DAG.getNode(ISD::SRA, SDLoc(N), VT, N0, NewOp1);
5229   }
5230 
5231   // fold (sra (trunc (srl x, c1)), c2) -> (trunc (sra x, c1 + c2))
5232   //      if c1 is equal to the number of bits the trunc removes
5233   if (N0.getOpcode() == ISD::TRUNCATE &&
5234       (N0.getOperand(0).getOpcode() == ISD::SRL ||
5235        N0.getOperand(0).getOpcode() == ISD::SRA) &&
5236       N0.getOperand(0).hasOneUse() &&
5237       N0.getOperand(0).getOperand(1).hasOneUse() &&
5238       N1C) {
5239     SDValue N0Op0 = N0.getOperand(0);
5240     if (ConstantSDNode *LargeShift = isConstOrConstSplat(N0Op0.getOperand(1))) {
5241       unsigned LargeShiftVal = LargeShift->getZExtValue();
5242       EVT LargeVT = N0Op0.getValueType();
5243 
5244       if (LargeVT.getScalarSizeInBits() - OpSizeInBits == LargeShiftVal) {
5245         SDLoc DL(N);
5246         SDValue Amt =
5247           DAG.getConstant(LargeShiftVal + N1C->getZExtValue(), DL,
5248                           getShiftAmountTy(N0Op0.getOperand(0).getValueType()));
5249         SDValue SRA = DAG.getNode(ISD::SRA, DL, LargeVT,
5250                                   N0Op0.getOperand(0), Amt);
5251         return DAG.getNode(ISD::TRUNCATE, DL, VT, SRA);
5252       }
5253     }
5254   }
5255 
5256   // Simplify, based on bits shifted out of the LHS.
5257   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
5258     return SDValue(N, 0);
5259 
5260 
5261   // If the sign bit is known to be zero, switch this to a SRL.
5262   if (DAG.SignBitIsZero(N0))
5263     return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, N1);
5264 
5265   if (N1C && !N1C->isOpaque())
5266     if (SDValue NewSRA = visitShiftByConstant(N, N1C))
5267       return NewSRA;
5268 
5269   return SDValue();
5270 }
5271 
5272 SDValue DAGCombiner::visitSRL(SDNode *N) {
5273   SDValue N0 = N->getOperand(0);
5274   SDValue N1 = N->getOperand(1);
5275   EVT VT = N0.getValueType();
5276   unsigned OpSizeInBits = VT.getScalarSizeInBits();
5277 
5278   // fold vector ops
5279   if (VT.isVector())
5280     if (SDValue FoldedVOp = SimplifyVBinOp(N))
5281       return FoldedVOp;
5282 
5283   ConstantSDNode *N1C = isConstOrConstSplat(N1);
5284 
5285   // fold (srl c1, c2) -> c1 >>u c2
5286   ConstantSDNode *N0C = getAsNonOpaqueConstant(N0);
5287   if (N0C && N1C && !N1C->isOpaque())
5288     return DAG.FoldConstantArithmetic(ISD::SRL, SDLoc(N), VT, N0C, N1C);
5289   // fold (srl 0, x) -> 0
5290   if (isNullConstant(N0))
5291     return N0;
5292   // fold (srl x, c >= size(x)) -> undef
5293   if (N1C && N1C->getAPIntValue().uge(OpSizeInBits))
5294     return DAG.getUNDEF(VT);
5295   // fold (srl x, 0) -> x
5296   if (N1C && N1C->isNullValue())
5297     return N0;
5298   // if (srl x, c) is known to be zero, return 0
5299   if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0),
5300                                    APInt::getAllOnesValue(OpSizeInBits)))
5301     return DAG.getConstant(0, SDLoc(N), VT);
5302 
5303   // fold (srl (srl x, c1), c2) -> 0 or (srl x, (add c1, c2))
5304   if (N1C && N0.getOpcode() == ISD::SRL) {
5305     if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) {
5306       SDLoc DL(N);
5307       APInt c1 = N0C1->getAPIntValue();
5308       APInt c2 = N1C->getAPIntValue();
5309       zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */);
5310 
5311       APInt Sum = c1 + c2;
5312       if (Sum.uge(OpSizeInBits))
5313         return DAG.getConstant(0, DL, VT);
5314 
5315       return DAG.getNode(
5316           ISD::SRL, DL, VT, N0.getOperand(0),
5317           DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType()));
5318     }
5319   }
5320 
5321   // fold (srl (trunc (srl x, c1)), c2) -> 0 or (trunc (srl x, (add c1, c2)))
5322   if (N1C && N0.getOpcode() == ISD::TRUNCATE &&
5323       N0.getOperand(0).getOpcode() == ISD::SRL &&
5324       isa<ConstantSDNode>(N0.getOperand(0)->getOperand(1))) {
5325     uint64_t c1 =
5326       cast<ConstantSDNode>(N0.getOperand(0)->getOperand(1))->getZExtValue();
5327     uint64_t c2 = N1C->getZExtValue();
5328     EVT InnerShiftVT = N0.getOperand(0).getValueType();
5329     EVT ShiftCountVT = N0.getOperand(0)->getOperand(1).getValueType();
5330     uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits();
5331     // This is only valid if the OpSizeInBits + c1 = size of inner shift.
5332     if (c1 + OpSizeInBits == InnerShiftSize) {
5333       SDLoc DL(N0);
5334       if (c1 + c2 >= InnerShiftSize)
5335         return DAG.getConstant(0, DL, VT);
5336       return DAG.getNode(ISD::TRUNCATE, DL, VT,
5337                          DAG.getNode(ISD::SRL, DL, InnerShiftVT,
5338                                      N0.getOperand(0)->getOperand(0),
5339                                      DAG.getConstant(c1 + c2, DL,
5340                                                      ShiftCountVT)));
5341     }
5342   }
5343 
5344   // fold (srl (shl x, c), c) -> (and x, cst2)
5345   if (N0.getOpcode() == ISD::SHL && N0.getOperand(1) == N1 &&
5346       isConstantOrConstantVector(N1, /* NoOpaques */ true)) {
5347     SDLoc DL(N);
5348     SDValue Mask =
5349         DAG.getNode(ISD::SRL, DL, VT, DAG.getAllOnesConstant(DL, VT), N1);
5350     AddToWorklist(Mask.getNode());
5351     return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), Mask);
5352   }
5353 
5354   // fold (srl (anyextend x), c) -> (and (anyextend (srl x, c)), mask)
5355   if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) {
5356     // Shifting in all undef bits?
5357     EVT SmallVT = N0.getOperand(0).getValueType();
5358     unsigned BitSize = SmallVT.getScalarSizeInBits();
5359     if (N1C->getZExtValue() >= BitSize)
5360       return DAG.getUNDEF(VT);
5361 
5362     if (!LegalTypes || TLI.isTypeDesirableForOp(ISD::SRL, SmallVT)) {
5363       uint64_t ShiftAmt = N1C->getZExtValue();
5364       SDLoc DL0(N0);
5365       SDValue SmallShift = DAG.getNode(ISD::SRL, DL0, SmallVT,
5366                                        N0.getOperand(0),
5367                           DAG.getConstant(ShiftAmt, DL0,
5368                                           getShiftAmountTy(SmallVT)));
5369       AddToWorklist(SmallShift.getNode());
5370       APInt Mask = APInt::getAllOnesValue(OpSizeInBits).lshr(ShiftAmt);
5371       SDLoc DL(N);
5372       return DAG.getNode(ISD::AND, DL, VT,
5373                          DAG.getNode(ISD::ANY_EXTEND, DL, VT, SmallShift),
5374                          DAG.getConstant(Mask, DL, VT));
5375     }
5376   }
5377 
5378   // fold (srl (sra X, Y), 31) -> (srl X, 31).  This srl only looks at the sign
5379   // bit, which is unmodified by sra.
5380   if (N1C && N1C->getZExtValue() + 1 == OpSizeInBits) {
5381     if (N0.getOpcode() == ISD::SRA)
5382       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0.getOperand(0), N1);
5383   }
5384 
5385   // fold (srl (ctlz x), "5") -> x  iff x has one bit set (the low bit).
5386   if (N1C && N0.getOpcode() == ISD::CTLZ &&
5387       N1C->getAPIntValue() == Log2_32(OpSizeInBits)) {
5388     APInt KnownZero, KnownOne;
5389     DAG.computeKnownBits(N0.getOperand(0), KnownZero, KnownOne);
5390 
5391     // If any of the input bits are KnownOne, then the input couldn't be all
5392     // zeros, thus the result of the srl will always be zero.
5393     if (KnownOne.getBoolValue()) return DAG.getConstant(0, SDLoc(N0), VT);
5394 
5395     // If all of the bits input the to ctlz node are known to be zero, then
5396     // the result of the ctlz is "32" and the result of the shift is one.
5397     APInt UnknownBits = ~KnownZero;
5398     if (UnknownBits == 0) return DAG.getConstant(1, SDLoc(N0), VT);
5399 
5400     // Otherwise, check to see if there is exactly one bit input to the ctlz.
5401     if ((UnknownBits & (UnknownBits - 1)) == 0) {
5402       // Okay, we know that only that the single bit specified by UnknownBits
5403       // could be set on input to the CTLZ node. If this bit is set, the SRL
5404       // will return 0, if it is clear, it returns 1. Change the CTLZ/SRL pair
5405       // to an SRL/XOR pair, which is likely to simplify more.
5406       unsigned ShAmt = UnknownBits.countTrailingZeros();
5407       SDValue Op = N0.getOperand(0);
5408 
5409       if (ShAmt) {
5410         SDLoc DL(N0);
5411         Op = DAG.getNode(ISD::SRL, DL, VT, Op,
5412                   DAG.getConstant(ShAmt, DL,
5413                                   getShiftAmountTy(Op.getValueType())));
5414         AddToWorklist(Op.getNode());
5415       }
5416 
5417       SDLoc DL(N);
5418       return DAG.getNode(ISD::XOR, DL, VT,
5419                          Op, DAG.getConstant(1, DL, VT));
5420     }
5421   }
5422 
5423   // fold (srl x, (trunc (and y, c))) -> (srl x, (and (trunc y), (trunc c))).
5424   if (N1.getOpcode() == ISD::TRUNCATE &&
5425       N1.getOperand(0).getOpcode() == ISD::AND) {
5426     if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode()))
5427       return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, NewOp1);
5428   }
5429 
5430   // fold operands of srl based on knowledge that the low bits are not
5431   // demanded.
5432   if (N1C && SimplifyDemandedBits(SDValue(N, 0)))
5433     return SDValue(N, 0);
5434 
5435   if (N1C && !N1C->isOpaque())
5436     if (SDValue NewSRL = visitShiftByConstant(N, N1C))
5437       return NewSRL;
5438 
5439   // Attempt to convert a srl of a load into a narrower zero-extending load.
5440   if (SDValue NarrowLoad = ReduceLoadWidth(N))
5441     return NarrowLoad;
5442 
5443   // Here is a common situation. We want to optimize:
5444   //
5445   //   %a = ...
5446   //   %b = and i32 %a, 2
5447   //   %c = srl i32 %b, 1
5448   //   brcond i32 %c ...
5449   //
5450   // into
5451   //
5452   //   %a = ...
5453   //   %b = and %a, 2
5454   //   %c = setcc eq %b, 0
5455   //   brcond %c ...
5456   //
5457   // However when after the source operand of SRL is optimized into AND, the SRL
5458   // itself may not be optimized further. Look for it and add the BRCOND into
5459   // the worklist.
5460   if (N->hasOneUse()) {
5461     SDNode *Use = *N->use_begin();
5462     if (Use->getOpcode() == ISD::BRCOND)
5463       AddToWorklist(Use);
5464     else if (Use->getOpcode() == ISD::TRUNCATE && Use->hasOneUse()) {
5465       // Also look pass the truncate.
5466       Use = *Use->use_begin();
5467       if (Use->getOpcode() == ISD::BRCOND)
5468         AddToWorklist(Use);
5469     }
5470   }
5471 
5472   return SDValue();
5473 }
5474 
5475 SDValue DAGCombiner::visitBSWAP(SDNode *N) {
5476   SDValue N0 = N->getOperand(0);
5477   EVT VT = N->getValueType(0);
5478 
5479   // fold (bswap c1) -> c2
5480   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5481     return DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N0);
5482   // fold (bswap (bswap x)) -> x
5483   if (N0.getOpcode() == ISD::BSWAP)
5484     return N0->getOperand(0);
5485   return SDValue();
5486 }
5487 
5488 SDValue DAGCombiner::visitBITREVERSE(SDNode *N) {
5489   SDValue N0 = N->getOperand(0);
5490   EVT VT = N->getValueType(0);
5491 
5492   // fold (bitreverse c1) -> c2
5493   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5494     return DAG.getNode(ISD::BITREVERSE, SDLoc(N), VT, N0);
5495   // fold (bitreverse (bitreverse x)) -> x
5496   if (N0.getOpcode() == ISD::BITREVERSE)
5497     return N0.getOperand(0);
5498   return SDValue();
5499 }
5500 
5501 SDValue DAGCombiner::visitCTLZ(SDNode *N) {
5502   SDValue N0 = N->getOperand(0);
5503   EVT VT = N->getValueType(0);
5504 
5505   // fold (ctlz c1) -> c2
5506   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5507     return DAG.getNode(ISD::CTLZ, SDLoc(N), VT, N0);
5508   return SDValue();
5509 }
5510 
5511 SDValue DAGCombiner::visitCTLZ_ZERO_UNDEF(SDNode *N) {
5512   SDValue N0 = N->getOperand(0);
5513   EVT VT = N->getValueType(0);
5514 
5515   // fold (ctlz_zero_undef c1) -> c2
5516   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5517     return DAG.getNode(ISD::CTLZ_ZERO_UNDEF, SDLoc(N), VT, N0);
5518   return SDValue();
5519 }
5520 
5521 SDValue DAGCombiner::visitCTTZ(SDNode *N) {
5522   SDValue N0 = N->getOperand(0);
5523   EVT VT = N->getValueType(0);
5524 
5525   // fold (cttz c1) -> c2
5526   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5527     return DAG.getNode(ISD::CTTZ, SDLoc(N), VT, N0);
5528   return SDValue();
5529 }
5530 
5531 SDValue DAGCombiner::visitCTTZ_ZERO_UNDEF(SDNode *N) {
5532   SDValue N0 = N->getOperand(0);
5533   EVT VT = N->getValueType(0);
5534 
5535   // fold (cttz_zero_undef c1) -> c2
5536   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5537     return DAG.getNode(ISD::CTTZ_ZERO_UNDEF, SDLoc(N), VT, N0);
5538   return SDValue();
5539 }
5540 
5541 SDValue DAGCombiner::visitCTPOP(SDNode *N) {
5542   SDValue N0 = N->getOperand(0);
5543   EVT VT = N->getValueType(0);
5544 
5545   // fold (ctpop c1) -> c2
5546   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
5547     return DAG.getNode(ISD::CTPOP, SDLoc(N), VT, N0);
5548   return SDValue();
5549 }
5550 
5551 
5552 /// \brief Generate Min/Max node
5553 static SDValue combineMinNumMaxNum(const SDLoc &DL, EVT VT, SDValue LHS,
5554                                    SDValue RHS, SDValue True, SDValue False,
5555                                    ISD::CondCode CC, const TargetLowering &TLI,
5556                                    SelectionDAG &DAG) {
5557   if (!(LHS == True && RHS == False) && !(LHS == False && RHS == True))
5558     return SDValue();
5559 
5560   switch (CC) {
5561   case ISD::SETOLT:
5562   case ISD::SETOLE:
5563   case ISD::SETLT:
5564   case ISD::SETLE:
5565   case ISD::SETULT:
5566   case ISD::SETULE: {
5567     unsigned Opcode = (LHS == True) ? ISD::FMINNUM : ISD::FMAXNUM;
5568     if (TLI.isOperationLegal(Opcode, VT))
5569       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
5570     return SDValue();
5571   }
5572   case ISD::SETOGT:
5573   case ISD::SETOGE:
5574   case ISD::SETGT:
5575   case ISD::SETGE:
5576   case ISD::SETUGT:
5577   case ISD::SETUGE: {
5578     unsigned Opcode = (LHS == True) ? ISD::FMAXNUM : ISD::FMINNUM;
5579     if (TLI.isOperationLegal(Opcode, VT))
5580       return DAG.getNode(Opcode, DL, VT, LHS, RHS);
5581     return SDValue();
5582   }
5583   default:
5584     return SDValue();
5585   }
5586 }
5587 
5588 SDValue DAGCombiner::foldSelectOfConstants(SDNode *N) {
5589   SDValue Cond = N->getOperand(0);
5590   SDValue N1 = N->getOperand(1);
5591   SDValue N2 = N->getOperand(2);
5592   EVT VT = N->getValueType(0);
5593   EVT CondVT = Cond.getValueType();
5594   SDLoc DL(N);
5595 
5596   if (!VT.isInteger())
5597     return SDValue();
5598 
5599   if (!isa<ConstantSDNode>(N1) || !isa<ConstantSDNode>(N2))
5600     return SDValue();
5601 
5602   // TODO: We should handle other cases of selecting between {-1,0,1} here.
5603   if (CondVT == MVT::i1) {
5604     if (isNullConstant(N1) && isOneConstant(N2)) {
5605       // select Cond, 0, 1 --> zext (!Cond)
5606       SDValue NotCond = DAG.getNOT(DL, Cond, MVT::i1);
5607       if (VT != MVT::i1)
5608         NotCond = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, NotCond);
5609       return NotCond;
5610     }
5611     return SDValue();
5612   }
5613 
5614   // fold (select Cond, 0, 1) -> (xor Cond, 1)
5615   // We can't do this reliably if integer based booleans have different contents
5616   // to floating point based booleans. This is because we can't tell whether we
5617   // have an integer-based boolean or a floating-point-based boolean unless we
5618   // can find the SETCC that produced it and inspect its operands. This is
5619   // fairly easy if C is the SETCC node, but it can potentially be
5620   // undiscoverable (or not reasonably discoverable). For example, it could be
5621   // in another basic block or it could require searching a complicated
5622   // expression.
5623   if (CondVT.isInteger() &&
5624       TLI.getBooleanContents(false, true) ==
5625           TargetLowering::ZeroOrOneBooleanContent &&
5626       TLI.getBooleanContents(false, false) ==
5627           TargetLowering::ZeroOrOneBooleanContent &&
5628       isNullConstant(N1) && isOneConstant(N2)) {
5629     SDValue NotCond =
5630         DAG.getNode(ISD::XOR, DL, CondVT, Cond, DAG.getConstant(1, DL, CondVT));
5631     if (VT.bitsEq(CondVT))
5632       return NotCond;
5633     return DAG.getZExtOrTrunc(NotCond, DL, VT);
5634   }
5635 
5636   return SDValue();
5637 }
5638 
5639 SDValue DAGCombiner::visitSELECT(SDNode *N) {
5640   SDValue N0 = N->getOperand(0);
5641   SDValue N1 = N->getOperand(1);
5642   SDValue N2 = N->getOperand(2);
5643   EVT VT = N->getValueType(0);
5644   EVT VT0 = N0.getValueType();
5645 
5646   // fold (select C, X, X) -> X
5647   if (N1 == N2)
5648     return N1;
5649   if (const ConstantSDNode *N0C = dyn_cast<const ConstantSDNode>(N0)) {
5650     // fold (select true, X, Y) -> X
5651     // fold (select false, X, Y) -> Y
5652     return !N0C->isNullValue() ? N1 : N2;
5653   }
5654   // fold (select X, X, Y) -> (or X, Y)
5655   // fold (select X, 1, Y) -> (or C, Y)
5656   if (VT == VT0 && VT == MVT::i1 && (N0 == N1 || isOneConstant(N1)))
5657     return DAG.getNode(ISD::OR, SDLoc(N), VT, N0, N2);
5658 
5659   if (SDValue V = foldSelectOfConstants(N))
5660     return V;
5661 
5662   // fold (select C, 0, X) -> (and (not C), X)
5663   if (VT == VT0 && VT == MVT::i1 && isNullConstant(N1)) {
5664     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
5665     AddToWorklist(NOTNode.getNode());
5666     return DAG.getNode(ISD::AND, SDLoc(N), VT, NOTNode, N2);
5667   }
5668   // fold (select C, X, 1) -> (or (not C), X)
5669   if (VT == VT0 && VT == MVT::i1 && isOneConstant(N2)) {
5670     SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT);
5671     AddToWorklist(NOTNode.getNode());
5672     return DAG.getNode(ISD::OR, SDLoc(N), VT, NOTNode, N1);
5673   }
5674   // fold (select X, Y, X) -> (and X, Y)
5675   // fold (select X, Y, 0) -> (and X, Y)
5676   if (VT == VT0 && VT == MVT::i1 && (N0 == N2 || isNullConstant(N2)))
5677     return DAG.getNode(ISD::AND, SDLoc(N), VT, N0, N1);
5678 
5679   // If we can fold this based on the true/false value, do so.
5680   if (SimplifySelectOps(N, N1, N2))
5681     return SDValue(N, 0);  // Don't revisit N.
5682 
5683   if (VT0 == MVT::i1) {
5684     // The code in this block deals with the following 2 equivalences:
5685     //    select(C0|C1, x, y) <=> select(C0, x, select(C1, x, y))
5686     //    select(C0&C1, x, y) <=> select(C0, select(C1, x, y), y)
5687     // The target can specify its preferred form with the
5688     // shouldNormalizeToSelectSequence() callback. However we always transform
5689     // to the right anyway if we find the inner select exists in the DAG anyway
5690     // and we always transform to the left side if we know that we can further
5691     // optimize the combination of the conditions.
5692     bool normalizeToSequence
5693       = TLI.shouldNormalizeToSelectSequence(*DAG.getContext(), VT);
5694     // select (and Cond0, Cond1), X, Y
5695     //   -> select Cond0, (select Cond1, X, Y), Y
5696     if (N0->getOpcode() == ISD::AND && N0->hasOneUse()) {
5697       SDValue Cond0 = N0->getOperand(0);
5698       SDValue Cond1 = N0->getOperand(1);
5699       SDValue InnerSelect = DAG.getNode(ISD::SELECT, SDLoc(N),
5700                                         N1.getValueType(), Cond1, N1, N2);
5701       if (normalizeToSequence || !InnerSelect.use_empty())
5702         return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Cond0,
5703                            InnerSelect, N2);
5704     }
5705     // select (or Cond0, Cond1), X, Y -> select Cond0, X, (select Cond1, X, Y)
5706     if (N0->getOpcode() == ISD::OR && N0->hasOneUse()) {
5707       SDValue Cond0 = N0->getOperand(0);
5708       SDValue Cond1 = N0->getOperand(1);
5709       SDValue InnerSelect = DAG.getNode(ISD::SELECT, SDLoc(N),
5710                                         N1.getValueType(), Cond1, N1, N2);
5711       if (normalizeToSequence || !InnerSelect.use_empty())
5712         return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Cond0, N1,
5713                            InnerSelect);
5714     }
5715 
5716     // select Cond0, (select Cond1, X, Y), Y -> select (and Cond0, Cond1), X, Y
5717     if (N1->getOpcode() == ISD::SELECT && N1->hasOneUse()) {
5718       SDValue N1_0 = N1->getOperand(0);
5719       SDValue N1_1 = N1->getOperand(1);
5720       SDValue N1_2 = N1->getOperand(2);
5721       if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) {
5722         // Create the actual and node if we can generate good code for it.
5723         if (!normalizeToSequence) {
5724           SDValue And = DAG.getNode(ISD::AND, SDLoc(N), N0.getValueType(),
5725                                     N0, N1_0);
5726           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), And,
5727                              N1_1, N2);
5728         }
5729         // Otherwise see if we can optimize the "and" to a better pattern.
5730         if (SDValue Combined = visitANDLike(N0, N1_0, N))
5731           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Combined,
5732                              N1_1, N2);
5733       }
5734     }
5735     // select Cond0, X, (select Cond1, X, Y) -> select (or Cond0, Cond1), X, Y
5736     if (N2->getOpcode() == ISD::SELECT && N2->hasOneUse()) {
5737       SDValue N2_0 = N2->getOperand(0);
5738       SDValue N2_1 = N2->getOperand(1);
5739       SDValue N2_2 = N2->getOperand(2);
5740       if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) {
5741         // Create the actual or node if we can generate good code for it.
5742         if (!normalizeToSequence) {
5743           SDValue Or = DAG.getNode(ISD::OR, SDLoc(N), N0.getValueType(),
5744                                    N0, N2_0);
5745           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Or,
5746                              N1, N2_2);
5747         }
5748         // Otherwise see if we can optimize to a better pattern.
5749         if (SDValue Combined = visitORLike(N0, N2_0, N))
5750           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(), Combined,
5751                              N1, N2_2);
5752       }
5753     }
5754   }
5755 
5756   // select (xor Cond, 1), X, Y -> select Cond, Y, X
5757   if (VT0 == MVT::i1) {
5758     if (N0->getOpcode() == ISD::XOR) {
5759       if (auto *C = dyn_cast<ConstantSDNode>(N0->getOperand(1))) {
5760         SDValue Cond0 = N0->getOperand(0);
5761         if (C->isOne())
5762           return DAG.getNode(ISD::SELECT, SDLoc(N), N1.getValueType(),
5763                              Cond0, N2, N1);
5764       }
5765     }
5766   }
5767 
5768   // fold selects based on a setcc into other things, such as min/max/abs
5769   if (N0.getOpcode() == ISD::SETCC) {
5770     // select x, y (fcmp lt x, y) -> fminnum x, y
5771     // select x, y (fcmp gt x, y) -> fmaxnum x, y
5772     //
5773     // This is OK if we don't care about what happens if either operand is a
5774     // NaN.
5775     //
5776 
5777     // FIXME: Instead of testing for UnsafeFPMath, this should be checking for
5778     // no signed zeros as well as no nans.
5779     const TargetOptions &Options = DAG.getTarget().Options;
5780     if (Options.UnsafeFPMath &&
5781         VT.isFloatingPoint() && N0.hasOneUse() &&
5782         DAG.isKnownNeverNaN(N1) && DAG.isKnownNeverNaN(N2)) {
5783       ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
5784 
5785       if (SDValue FMinMax = combineMinNumMaxNum(SDLoc(N), VT, N0.getOperand(0),
5786                                                 N0.getOperand(1), N1, N2, CC,
5787                                                 TLI, DAG))
5788         return FMinMax;
5789     }
5790 
5791     if ((!LegalOperations &&
5792          TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT)) ||
5793         TLI.isOperationLegal(ISD::SELECT_CC, VT))
5794       return DAG.getNode(ISD::SELECT_CC, SDLoc(N), VT,
5795                          N0.getOperand(0), N0.getOperand(1),
5796                          N1, N2, N0.getOperand(2));
5797     return SimplifySelect(SDLoc(N), N0, N1, N2);
5798   }
5799 
5800   return SDValue();
5801 }
5802 
5803 static
5804 std::pair<SDValue, SDValue> SplitVSETCC(const SDNode *N, SelectionDAG &DAG) {
5805   SDLoc DL(N);
5806   EVT LoVT, HiVT;
5807   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(N->getValueType(0));
5808 
5809   // Split the inputs.
5810   SDValue Lo, Hi, LL, LH, RL, RH;
5811   std::tie(LL, LH) = DAG.SplitVectorOperand(N, 0);
5812   std::tie(RL, RH) = DAG.SplitVectorOperand(N, 1);
5813 
5814   Lo = DAG.getNode(N->getOpcode(), DL, LoVT, LL, RL, N->getOperand(2));
5815   Hi = DAG.getNode(N->getOpcode(), DL, HiVT, LH, RH, N->getOperand(2));
5816 
5817   return std::make_pair(Lo, Hi);
5818 }
5819 
5820 // This function assumes all the vselect's arguments are CONCAT_VECTOR
5821 // nodes and that the condition is a BV of ConstantSDNodes (or undefs).
5822 static SDValue ConvertSelectToConcatVector(SDNode *N, SelectionDAG &DAG) {
5823   SDLoc DL(N);
5824   SDValue Cond = N->getOperand(0);
5825   SDValue LHS = N->getOperand(1);
5826   SDValue RHS = N->getOperand(2);
5827   EVT VT = N->getValueType(0);
5828   int NumElems = VT.getVectorNumElements();
5829   assert(LHS.getOpcode() == ISD::CONCAT_VECTORS &&
5830          RHS.getOpcode() == ISD::CONCAT_VECTORS &&
5831          Cond.getOpcode() == ISD::BUILD_VECTOR);
5832 
5833   // CONCAT_VECTOR can take an arbitrary number of arguments. We only care about
5834   // binary ones here.
5835   if (LHS->getNumOperands() != 2 || RHS->getNumOperands() != 2)
5836     return SDValue();
5837 
5838   // We're sure we have an even number of elements due to the
5839   // concat_vectors we have as arguments to vselect.
5840   // Skip BV elements until we find one that's not an UNDEF
5841   // After we find an UNDEF element, keep looping until we get to half the
5842   // length of the BV and see if all the non-undef nodes are the same.
5843   ConstantSDNode *BottomHalf = nullptr;
5844   for (int i = 0; i < NumElems / 2; ++i) {
5845     if (Cond->getOperand(i)->isUndef())
5846       continue;
5847 
5848     if (BottomHalf == nullptr)
5849       BottomHalf = cast<ConstantSDNode>(Cond.getOperand(i));
5850     else if (Cond->getOperand(i).getNode() != BottomHalf)
5851       return SDValue();
5852   }
5853 
5854   // Do the same for the second half of the BuildVector
5855   ConstantSDNode *TopHalf = nullptr;
5856   for (int i = NumElems / 2; i < NumElems; ++i) {
5857     if (Cond->getOperand(i)->isUndef())
5858       continue;
5859 
5860     if (TopHalf == nullptr)
5861       TopHalf = cast<ConstantSDNode>(Cond.getOperand(i));
5862     else if (Cond->getOperand(i).getNode() != TopHalf)
5863       return SDValue();
5864   }
5865 
5866   assert(TopHalf && BottomHalf &&
5867          "One half of the selector was all UNDEFs and the other was all the "
5868          "same value. This should have been addressed before this function.");
5869   return DAG.getNode(
5870       ISD::CONCAT_VECTORS, DL, VT,
5871       BottomHalf->isNullValue() ? RHS->getOperand(0) : LHS->getOperand(0),
5872       TopHalf->isNullValue() ? RHS->getOperand(1) : LHS->getOperand(1));
5873 }
5874 
5875 SDValue DAGCombiner::visitMSCATTER(SDNode *N) {
5876 
5877   if (Level >= AfterLegalizeTypes)
5878     return SDValue();
5879 
5880   MaskedScatterSDNode *MSC = cast<MaskedScatterSDNode>(N);
5881   SDValue Mask = MSC->getMask();
5882   SDValue Data  = MSC->getValue();
5883   SDLoc DL(N);
5884 
5885   // If the MSCATTER data type requires splitting and the mask is provided by a
5886   // SETCC, then split both nodes and its operands before legalization. This
5887   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5888   // and enables future optimizations (e.g. min/max pattern matching on X86).
5889   if (Mask.getOpcode() != ISD::SETCC)
5890     return SDValue();
5891 
5892   // Check if any splitting is required.
5893   if (TLI.getTypeAction(*DAG.getContext(), Data.getValueType()) !=
5894       TargetLowering::TypeSplitVector)
5895     return SDValue();
5896   SDValue MaskLo, MaskHi, Lo, Hi;
5897   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5898 
5899   EVT LoVT, HiVT;
5900   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MSC->getValueType(0));
5901 
5902   SDValue Chain = MSC->getChain();
5903 
5904   EVT MemoryVT = MSC->getMemoryVT();
5905   unsigned Alignment = MSC->getOriginalAlignment();
5906 
5907   EVT LoMemVT, HiMemVT;
5908   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5909 
5910   SDValue DataLo, DataHi;
5911   std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
5912 
5913   SDValue BasePtr = MSC->getBasePtr();
5914   SDValue IndexLo, IndexHi;
5915   std::tie(IndexLo, IndexHi) = DAG.SplitVector(MSC->getIndex(), DL);
5916 
5917   MachineMemOperand *MMO = DAG.getMachineFunction().
5918     getMachineMemOperand(MSC->getPointerInfo(),
5919                           MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
5920                           Alignment, MSC->getAAInfo(), MSC->getRanges());
5921 
5922   SDValue OpsLo[] = { Chain, DataLo, MaskLo, BasePtr, IndexLo };
5923   Lo = DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataLo.getValueType(),
5924                             DL, OpsLo, MMO);
5925 
5926   SDValue OpsHi[] = {Chain, DataHi, MaskHi, BasePtr, IndexHi};
5927   Hi = DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataHi.getValueType(),
5928                             DL, OpsHi, MMO);
5929 
5930   AddToWorklist(Lo.getNode());
5931   AddToWorklist(Hi.getNode());
5932 
5933   return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi);
5934 }
5935 
5936 SDValue DAGCombiner::visitMSTORE(SDNode *N) {
5937 
5938   if (Level >= AfterLegalizeTypes)
5939     return SDValue();
5940 
5941   MaskedStoreSDNode *MST = dyn_cast<MaskedStoreSDNode>(N);
5942   SDValue Mask = MST->getMask();
5943   SDValue Data  = MST->getValue();
5944   EVT VT = Data.getValueType();
5945   SDLoc DL(N);
5946 
5947   // If the MSTORE data type requires splitting and the mask is provided by a
5948   // SETCC, then split both nodes and its operands before legalization. This
5949   // prevents the type legalizer from unrolling SETCC into scalar comparisons
5950   // and enables future optimizations (e.g. min/max pattern matching on X86).
5951   if (Mask.getOpcode() == ISD::SETCC) {
5952 
5953     // Check if any splitting is required.
5954     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
5955         TargetLowering::TypeSplitVector)
5956       return SDValue();
5957 
5958     SDValue MaskLo, MaskHi, Lo, Hi;
5959     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
5960 
5961     SDValue Chain = MST->getChain();
5962     SDValue Ptr   = MST->getBasePtr();
5963 
5964     EVT MemoryVT = MST->getMemoryVT();
5965     unsigned Alignment = MST->getOriginalAlignment();
5966 
5967     // if Alignment is equal to the vector size,
5968     // take the half of it for the second part
5969     unsigned SecondHalfAlignment =
5970       (Alignment == VT.getSizeInBits() / 8) ? Alignment / 2 : Alignment;
5971 
5972     EVT LoMemVT, HiMemVT;
5973     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
5974 
5975     SDValue DataLo, DataHi;
5976     std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL);
5977 
5978     MachineMemOperand *MMO = DAG.getMachineFunction().
5979       getMachineMemOperand(MST->getPointerInfo(),
5980                            MachineMemOperand::MOStore,  LoMemVT.getStoreSize(),
5981                            Alignment, MST->getAAInfo(), MST->getRanges());
5982 
5983     Lo = DAG.getMaskedStore(Chain, DL, DataLo, Ptr, MaskLo, LoMemVT, MMO,
5984                             MST->isTruncatingStore(),
5985                             MST->isCompressingStore());
5986 
5987     Ptr = TLI.IncrementMemoryAddress(Ptr, MaskLo, DL, LoMemVT, DAG,
5988                                      MST->isCompressingStore());
5989 
5990     MMO = DAG.getMachineFunction().
5991       getMachineMemOperand(MST->getPointerInfo(),
5992                            MachineMemOperand::MOStore,  HiMemVT.getStoreSize(),
5993                            SecondHalfAlignment, MST->getAAInfo(),
5994                            MST->getRanges());
5995 
5996     Hi = DAG.getMaskedStore(Chain, DL, DataHi, Ptr, MaskHi, HiMemVT, MMO,
5997                             MST->isTruncatingStore(),
5998                             MST->isCompressingStore());
5999 
6000     AddToWorklist(Lo.getNode());
6001     AddToWorklist(Hi.getNode());
6002 
6003     return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi);
6004   }
6005   return SDValue();
6006 }
6007 
6008 SDValue DAGCombiner::visitMGATHER(SDNode *N) {
6009 
6010   if (Level >= AfterLegalizeTypes)
6011     return SDValue();
6012 
6013   MaskedGatherSDNode *MGT = dyn_cast<MaskedGatherSDNode>(N);
6014   SDValue Mask = MGT->getMask();
6015   SDLoc DL(N);
6016 
6017   // If the MGATHER result requires splitting and the mask is provided by a
6018   // SETCC, then split both nodes and its operands before legalization. This
6019   // prevents the type legalizer from unrolling SETCC into scalar comparisons
6020   // and enables future optimizations (e.g. min/max pattern matching on X86).
6021 
6022   if (Mask.getOpcode() != ISD::SETCC)
6023     return SDValue();
6024 
6025   EVT VT = N->getValueType(0);
6026 
6027   // Check if any splitting is required.
6028   if (TLI.getTypeAction(*DAG.getContext(), VT) !=
6029       TargetLowering::TypeSplitVector)
6030     return SDValue();
6031 
6032   SDValue MaskLo, MaskHi, Lo, Hi;
6033   std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
6034 
6035   SDValue Src0 = MGT->getValue();
6036   SDValue Src0Lo, Src0Hi;
6037   std::tie(Src0Lo, Src0Hi) = DAG.SplitVector(Src0, DL);
6038 
6039   EVT LoVT, HiVT;
6040   std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT);
6041 
6042   SDValue Chain = MGT->getChain();
6043   EVT MemoryVT = MGT->getMemoryVT();
6044   unsigned Alignment = MGT->getOriginalAlignment();
6045 
6046   EVT LoMemVT, HiMemVT;
6047   std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
6048 
6049   SDValue BasePtr = MGT->getBasePtr();
6050   SDValue Index = MGT->getIndex();
6051   SDValue IndexLo, IndexHi;
6052   std::tie(IndexLo, IndexHi) = DAG.SplitVector(Index, DL);
6053 
6054   MachineMemOperand *MMO = DAG.getMachineFunction().
6055     getMachineMemOperand(MGT->getPointerInfo(),
6056                           MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
6057                           Alignment, MGT->getAAInfo(), MGT->getRanges());
6058 
6059   SDValue OpsLo[] = { Chain, Src0Lo, MaskLo, BasePtr, IndexLo };
6060   Lo = DAG.getMaskedGather(DAG.getVTList(LoVT, MVT::Other), LoVT, DL, OpsLo,
6061                             MMO);
6062 
6063   SDValue OpsHi[] = {Chain, Src0Hi, MaskHi, BasePtr, IndexHi};
6064   Hi = DAG.getMaskedGather(DAG.getVTList(HiVT, MVT::Other), HiVT, DL, OpsHi,
6065                             MMO);
6066 
6067   AddToWorklist(Lo.getNode());
6068   AddToWorklist(Hi.getNode());
6069 
6070   // Build a factor node to remember that this load is independent of the
6071   // other one.
6072   Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
6073                       Hi.getValue(1));
6074 
6075   // Legalized the chain result - switch anything that used the old chain to
6076   // use the new one.
6077   DAG.ReplaceAllUsesOfValueWith(SDValue(MGT, 1), Chain);
6078 
6079   SDValue GatherRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
6080 
6081   SDValue RetOps[] = { GatherRes, Chain };
6082   return DAG.getMergeValues(RetOps, DL);
6083 }
6084 
6085 SDValue DAGCombiner::visitMLOAD(SDNode *N) {
6086 
6087   if (Level >= AfterLegalizeTypes)
6088     return SDValue();
6089 
6090   MaskedLoadSDNode *MLD = dyn_cast<MaskedLoadSDNode>(N);
6091   SDValue Mask = MLD->getMask();
6092   SDLoc DL(N);
6093 
6094   // If the MLOAD result requires splitting and the mask is provided by a
6095   // SETCC, then split both nodes and its operands before legalization. This
6096   // prevents the type legalizer from unrolling SETCC into scalar comparisons
6097   // and enables future optimizations (e.g. min/max pattern matching on X86).
6098 
6099   if (Mask.getOpcode() == ISD::SETCC) {
6100     EVT VT = N->getValueType(0);
6101 
6102     // Check if any splitting is required.
6103     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
6104         TargetLowering::TypeSplitVector)
6105       return SDValue();
6106 
6107     SDValue MaskLo, MaskHi, Lo, Hi;
6108     std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG);
6109 
6110     SDValue Src0 = MLD->getSrc0();
6111     SDValue Src0Lo, Src0Hi;
6112     std::tie(Src0Lo, Src0Hi) = DAG.SplitVector(Src0, DL);
6113 
6114     EVT LoVT, HiVT;
6115     std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MLD->getValueType(0));
6116 
6117     SDValue Chain = MLD->getChain();
6118     SDValue Ptr   = MLD->getBasePtr();
6119     EVT MemoryVT = MLD->getMemoryVT();
6120     unsigned Alignment = MLD->getOriginalAlignment();
6121 
6122     // if Alignment is equal to the vector size,
6123     // take the half of it for the second part
6124     unsigned SecondHalfAlignment =
6125       (Alignment == MLD->getValueType(0).getSizeInBits()/8) ?
6126          Alignment/2 : Alignment;
6127 
6128     EVT LoMemVT, HiMemVT;
6129     std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT);
6130 
6131     MachineMemOperand *MMO = DAG.getMachineFunction().
6132     getMachineMemOperand(MLD->getPointerInfo(),
6133                          MachineMemOperand::MOLoad,  LoMemVT.getStoreSize(),
6134                          Alignment, MLD->getAAInfo(), MLD->getRanges());
6135 
6136     Lo = DAG.getMaskedLoad(LoVT, DL, Chain, Ptr, MaskLo, Src0Lo, LoMemVT, MMO,
6137                            ISD::NON_EXTLOAD, MLD->isExpandingLoad());
6138 
6139     Ptr = TLI.IncrementMemoryAddress(Ptr, MaskLo, DL, LoMemVT, DAG,
6140                                      MLD->isExpandingLoad());
6141 
6142     MMO = DAG.getMachineFunction().
6143     getMachineMemOperand(MLD->getPointerInfo(),
6144                          MachineMemOperand::MOLoad,  HiMemVT.getStoreSize(),
6145                          SecondHalfAlignment, MLD->getAAInfo(), MLD->getRanges());
6146 
6147     Hi = DAG.getMaskedLoad(HiVT, DL, Chain, Ptr, MaskHi, Src0Hi, HiMemVT, MMO,
6148                            ISD::NON_EXTLOAD, MLD->isExpandingLoad());
6149 
6150     AddToWorklist(Lo.getNode());
6151     AddToWorklist(Hi.getNode());
6152 
6153     // Build a factor node to remember that this load is independent of the
6154     // other one.
6155     Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1),
6156                         Hi.getValue(1));
6157 
6158     // Legalized the chain result - switch anything that used the old chain to
6159     // use the new one.
6160     DAG.ReplaceAllUsesOfValueWith(SDValue(MLD, 1), Chain);
6161 
6162     SDValue LoadRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
6163 
6164     SDValue RetOps[] = { LoadRes, Chain };
6165     return DAG.getMergeValues(RetOps, DL);
6166   }
6167   return SDValue();
6168 }
6169 
6170 SDValue DAGCombiner::visitVSELECT(SDNode *N) {
6171   SDValue N0 = N->getOperand(0);
6172   SDValue N1 = N->getOperand(1);
6173   SDValue N2 = N->getOperand(2);
6174   SDLoc DL(N);
6175 
6176   // fold (vselect C, X, X) -> X
6177   if (N1 == N2)
6178     return N1;
6179 
6180   // Canonicalize integer abs.
6181   // vselect (setg[te] X,  0),  X, -X ->
6182   // vselect (setgt    X, -1),  X, -X ->
6183   // vselect (setl[te] X,  0), -X,  X ->
6184   // Y = sra (X, size(X)-1); xor (add (X, Y), Y)
6185   if (N0.getOpcode() == ISD::SETCC) {
6186     SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1);
6187     ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
6188     bool isAbs = false;
6189     bool RHSIsAllZeros = ISD::isBuildVectorAllZeros(RHS.getNode());
6190 
6191     if (((RHSIsAllZeros && (CC == ISD::SETGT || CC == ISD::SETGE)) ||
6192          (ISD::isBuildVectorAllOnes(RHS.getNode()) && CC == ISD::SETGT)) &&
6193         N1 == LHS && N2.getOpcode() == ISD::SUB && N1 == N2.getOperand(1))
6194       isAbs = ISD::isBuildVectorAllZeros(N2.getOperand(0).getNode());
6195     else if ((RHSIsAllZeros && (CC == ISD::SETLT || CC == ISD::SETLE)) &&
6196              N2 == LHS && N1.getOpcode() == ISD::SUB && N2 == N1.getOperand(1))
6197       isAbs = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode());
6198 
6199     if (isAbs) {
6200       EVT VT = LHS.getValueType();
6201       SDValue Shift = DAG.getNode(
6202           ISD::SRA, DL, VT, LHS,
6203           DAG.getConstant(VT.getScalarSizeInBits() - 1, DL, VT));
6204       SDValue Add = DAG.getNode(ISD::ADD, DL, VT, LHS, Shift);
6205       AddToWorklist(Shift.getNode());
6206       AddToWorklist(Add.getNode());
6207       return DAG.getNode(ISD::XOR, DL, VT, Add, Shift);
6208     }
6209   }
6210 
6211   if (SimplifySelectOps(N, N1, N2))
6212     return SDValue(N, 0);  // Don't revisit N.
6213 
6214   // If the VSELECT result requires splitting and the mask is provided by a
6215   // SETCC, then split both nodes and its operands before legalization. This
6216   // prevents the type legalizer from unrolling SETCC into scalar comparisons
6217   // and enables future optimizations (e.g. min/max pattern matching on X86).
6218   if (N0.getOpcode() == ISD::SETCC) {
6219     EVT VT = N->getValueType(0);
6220 
6221     // Check if any splitting is required.
6222     if (TLI.getTypeAction(*DAG.getContext(), VT) !=
6223         TargetLowering::TypeSplitVector)
6224       return SDValue();
6225 
6226     SDValue Lo, Hi, CCLo, CCHi, LL, LH, RL, RH;
6227     std::tie(CCLo, CCHi) = SplitVSETCC(N0.getNode(), DAG);
6228     std::tie(LL, LH) = DAG.SplitVectorOperand(N, 1);
6229     std::tie(RL, RH) = DAG.SplitVectorOperand(N, 2);
6230 
6231     Lo = DAG.getNode(N->getOpcode(), DL, LL.getValueType(), CCLo, LL, RL);
6232     Hi = DAG.getNode(N->getOpcode(), DL, LH.getValueType(), CCHi, LH, RH);
6233 
6234     // Add the new VSELECT nodes to the work list in case they need to be split
6235     // again.
6236     AddToWorklist(Lo.getNode());
6237     AddToWorklist(Hi.getNode());
6238 
6239     return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
6240   }
6241 
6242   // Fold (vselect (build_vector all_ones), N1, N2) -> N1
6243   if (ISD::isBuildVectorAllOnes(N0.getNode()))
6244     return N1;
6245   // Fold (vselect (build_vector all_zeros), N1, N2) -> N2
6246   if (ISD::isBuildVectorAllZeros(N0.getNode()))
6247     return N2;
6248 
6249   // The ConvertSelectToConcatVector function is assuming both the above
6250   // checks for (vselect (build_vector all{ones,zeros) ...) have been made
6251   // and addressed.
6252   if (N1.getOpcode() == ISD::CONCAT_VECTORS &&
6253       N2.getOpcode() == ISD::CONCAT_VECTORS &&
6254       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())) {
6255     if (SDValue CV = ConvertSelectToConcatVector(N, DAG))
6256       return CV;
6257   }
6258 
6259   return SDValue();
6260 }
6261 
6262 SDValue DAGCombiner::visitSELECT_CC(SDNode *N) {
6263   SDValue N0 = N->getOperand(0);
6264   SDValue N1 = N->getOperand(1);
6265   SDValue N2 = N->getOperand(2);
6266   SDValue N3 = N->getOperand(3);
6267   SDValue N4 = N->getOperand(4);
6268   ISD::CondCode CC = cast<CondCodeSDNode>(N4)->get();
6269 
6270   // fold select_cc lhs, rhs, x, x, cc -> x
6271   if (N2 == N3)
6272     return N2;
6273 
6274   // Determine if the condition we're dealing with is constant
6275   if (SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()), N0, N1,
6276                                   CC, SDLoc(N), false)) {
6277     AddToWorklist(SCC.getNode());
6278 
6279     if (ConstantSDNode *SCCC = dyn_cast<ConstantSDNode>(SCC.getNode())) {
6280       if (!SCCC->isNullValue())
6281         return N2;    // cond always true -> true val
6282       else
6283         return N3;    // cond always false -> false val
6284     } else if (SCC->isUndef()) {
6285       // When the condition is UNDEF, just return the first operand. This is
6286       // coherent the DAG creation, no setcc node is created in this case
6287       return N2;
6288     } else if (SCC.getOpcode() == ISD::SETCC) {
6289       // Fold to a simpler select_cc
6290       return DAG.getNode(ISD::SELECT_CC, SDLoc(N), N2.getValueType(),
6291                          SCC.getOperand(0), SCC.getOperand(1), N2, N3,
6292                          SCC.getOperand(2));
6293     }
6294   }
6295 
6296   // If we can fold this based on the true/false value, do so.
6297   if (SimplifySelectOps(N, N2, N3))
6298     return SDValue(N, 0);  // Don't revisit N.
6299 
6300   // fold select_cc into other things, such as min/max/abs
6301   return SimplifySelectCC(SDLoc(N), N0, N1, N2, N3, CC);
6302 }
6303 
6304 SDValue DAGCombiner::visitSETCC(SDNode *N) {
6305   return SimplifySetCC(N->getValueType(0), N->getOperand(0), N->getOperand(1),
6306                        cast<CondCodeSDNode>(N->getOperand(2))->get(),
6307                        SDLoc(N));
6308 }
6309 
6310 SDValue DAGCombiner::visitSETCCE(SDNode *N) {
6311   SDValue LHS = N->getOperand(0);
6312   SDValue RHS = N->getOperand(1);
6313   SDValue Carry = N->getOperand(2);
6314   SDValue Cond = N->getOperand(3);
6315 
6316   // If Carry is false, fold to a regular SETCC.
6317   if (Carry.getOpcode() == ISD::CARRY_FALSE)
6318     return DAG.getNode(ISD::SETCC, SDLoc(N), N->getVTList(), LHS, RHS, Cond);
6319 
6320   return SDValue();
6321 }
6322 
6323 /// Try to fold a sext/zext/aext dag node into a ConstantSDNode or
6324 /// a build_vector of constants.
6325 /// This function is called by the DAGCombiner when visiting sext/zext/aext
6326 /// dag nodes (see for example method DAGCombiner::visitSIGN_EXTEND).
6327 /// Vector extends are not folded if operations are legal; this is to
6328 /// avoid introducing illegal build_vector dag nodes.
6329 static SDNode *tryToFoldExtendOfConstant(SDNode *N, const TargetLowering &TLI,
6330                                          SelectionDAG &DAG, bool LegalTypes,
6331                                          bool LegalOperations) {
6332   unsigned Opcode = N->getOpcode();
6333   SDValue N0 = N->getOperand(0);
6334   EVT VT = N->getValueType(0);
6335 
6336   assert((Opcode == ISD::SIGN_EXTEND || Opcode == ISD::ZERO_EXTEND ||
6337          Opcode == ISD::ANY_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG ||
6338          Opcode == ISD::ZERO_EXTEND_VECTOR_INREG)
6339          && "Expected EXTEND dag node in input!");
6340 
6341   // fold (sext c1) -> c1
6342   // fold (zext c1) -> c1
6343   // fold (aext c1) -> c1
6344   if (isa<ConstantSDNode>(N0))
6345     return DAG.getNode(Opcode, SDLoc(N), VT, N0).getNode();
6346 
6347   // fold (sext (build_vector AllConstants) -> (build_vector AllConstants)
6348   // fold (zext (build_vector AllConstants) -> (build_vector AllConstants)
6349   // fold (aext (build_vector AllConstants) -> (build_vector AllConstants)
6350   EVT SVT = VT.getScalarType();
6351   if (!(VT.isVector() &&
6352       (!LegalTypes || (!LegalOperations && TLI.isTypeLegal(SVT))) &&
6353       ISD::isBuildVectorOfConstantSDNodes(N0.getNode())))
6354     return nullptr;
6355 
6356   // We can fold this node into a build_vector.
6357   unsigned VTBits = SVT.getSizeInBits();
6358   unsigned EVTBits = N0->getValueType(0).getScalarSizeInBits();
6359   SmallVector<SDValue, 8> Elts;
6360   unsigned NumElts = VT.getVectorNumElements();
6361   SDLoc DL(N);
6362 
6363   for (unsigned i=0; i != NumElts; ++i) {
6364     SDValue Op = N0->getOperand(i);
6365     if (Op->isUndef()) {
6366       Elts.push_back(DAG.getUNDEF(SVT));
6367       continue;
6368     }
6369 
6370     SDLoc DL(Op);
6371     // Get the constant value and if needed trunc it to the size of the type.
6372     // Nodes like build_vector might have constants wider than the scalar type.
6373     APInt C = cast<ConstantSDNode>(Op)->getAPIntValue().zextOrTrunc(EVTBits);
6374     if (Opcode == ISD::SIGN_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG)
6375       Elts.push_back(DAG.getConstant(C.sext(VTBits), DL, SVT));
6376     else
6377       Elts.push_back(DAG.getConstant(C.zext(VTBits), DL, SVT));
6378   }
6379 
6380   return DAG.getBuildVector(VT, DL, Elts).getNode();
6381 }
6382 
6383 // ExtendUsesToFormExtLoad - Trying to extend uses of a load to enable this:
6384 // "fold ({s|z|a}ext (load x)) -> ({s|z|a}ext (truncate ({s|z|a}extload x)))"
6385 // transformation. Returns true if extension are possible and the above
6386 // mentioned transformation is profitable.
6387 static bool ExtendUsesToFormExtLoad(SDNode *N, SDValue N0,
6388                                     unsigned ExtOpc,
6389                                     SmallVectorImpl<SDNode *> &ExtendNodes,
6390                                     const TargetLowering &TLI) {
6391   bool HasCopyToRegUses = false;
6392   bool isTruncFree = TLI.isTruncateFree(N->getValueType(0), N0.getValueType());
6393   for (SDNode::use_iterator UI = N0.getNode()->use_begin(),
6394                             UE = N0.getNode()->use_end();
6395        UI != UE; ++UI) {
6396     SDNode *User = *UI;
6397     if (User == N)
6398       continue;
6399     if (UI.getUse().getResNo() != N0.getResNo())
6400       continue;
6401     // FIXME: Only extend SETCC N, N and SETCC N, c for now.
6402     if (ExtOpc != ISD::ANY_EXTEND && User->getOpcode() == ISD::SETCC) {
6403       ISD::CondCode CC = cast<CondCodeSDNode>(User->getOperand(2))->get();
6404       if (ExtOpc == ISD::ZERO_EXTEND && ISD::isSignedIntSetCC(CC))
6405         // Sign bits will be lost after a zext.
6406         return false;
6407       bool Add = false;
6408       for (unsigned i = 0; i != 2; ++i) {
6409         SDValue UseOp = User->getOperand(i);
6410         if (UseOp == N0)
6411           continue;
6412         if (!isa<ConstantSDNode>(UseOp))
6413           return false;
6414         Add = true;
6415       }
6416       if (Add)
6417         ExtendNodes.push_back(User);
6418       continue;
6419     }
6420     // If truncates aren't free and there are users we can't
6421     // extend, it isn't worthwhile.
6422     if (!isTruncFree)
6423       return false;
6424     // Remember if this value is live-out.
6425     if (User->getOpcode() == ISD::CopyToReg)
6426       HasCopyToRegUses = true;
6427   }
6428 
6429   if (HasCopyToRegUses) {
6430     bool BothLiveOut = false;
6431     for (SDNode::use_iterator UI = N->use_begin(), UE = N->use_end();
6432          UI != UE; ++UI) {
6433       SDUse &Use = UI.getUse();
6434       if (Use.getResNo() == 0 && Use.getUser()->getOpcode() == ISD::CopyToReg) {
6435         BothLiveOut = true;
6436         break;
6437       }
6438     }
6439     if (BothLiveOut)
6440       // Both unextended and extended values are live out. There had better be
6441       // a good reason for the transformation.
6442       return ExtendNodes.size();
6443   }
6444   return true;
6445 }
6446 
6447 void DAGCombiner::ExtendSetCCUses(const SmallVectorImpl<SDNode *> &SetCCs,
6448                                   SDValue Trunc, SDValue ExtLoad,
6449                                   const SDLoc &DL, ISD::NodeType ExtType) {
6450   // Extend SetCC uses if necessary.
6451   for (unsigned i = 0, e = SetCCs.size(); i != e; ++i) {
6452     SDNode *SetCC = SetCCs[i];
6453     SmallVector<SDValue, 4> Ops;
6454 
6455     for (unsigned j = 0; j != 2; ++j) {
6456       SDValue SOp = SetCC->getOperand(j);
6457       if (SOp == Trunc)
6458         Ops.push_back(ExtLoad);
6459       else
6460         Ops.push_back(DAG.getNode(ExtType, DL, ExtLoad->getValueType(0), SOp));
6461     }
6462 
6463     Ops.push_back(SetCC->getOperand(2));
6464     CombineTo(SetCC, DAG.getNode(ISD::SETCC, DL, SetCC->getValueType(0), Ops));
6465   }
6466 }
6467 
6468 // FIXME: Bring more similar combines here, common to sext/zext (maybe aext?).
6469 SDValue DAGCombiner::CombineExtLoad(SDNode *N) {
6470   SDValue N0 = N->getOperand(0);
6471   EVT DstVT = N->getValueType(0);
6472   EVT SrcVT = N0.getValueType();
6473 
6474   assert((N->getOpcode() == ISD::SIGN_EXTEND ||
6475           N->getOpcode() == ISD::ZERO_EXTEND) &&
6476          "Unexpected node type (not an extend)!");
6477 
6478   // fold (sext (load x)) to multiple smaller sextloads; same for zext.
6479   // For example, on a target with legal v4i32, but illegal v8i32, turn:
6480   //   (v8i32 (sext (v8i16 (load x))))
6481   // into:
6482   //   (v8i32 (concat_vectors (v4i32 (sextload x)),
6483   //                          (v4i32 (sextload (x + 16)))))
6484   // Where uses of the original load, i.e.:
6485   //   (v8i16 (load x))
6486   // are replaced with:
6487   //   (v8i16 (truncate
6488   //     (v8i32 (concat_vectors (v4i32 (sextload x)),
6489   //                            (v4i32 (sextload (x + 16)))))))
6490   //
6491   // This combine is only applicable to illegal, but splittable, vectors.
6492   // All legal types, and illegal non-vector types, are handled elsewhere.
6493   // This combine is controlled by TargetLowering::isVectorLoadExtDesirable.
6494   //
6495   if (N0->getOpcode() != ISD::LOAD)
6496     return SDValue();
6497 
6498   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6499 
6500   if (!ISD::isNON_EXTLoad(LN0) || !ISD::isUNINDEXEDLoad(LN0) ||
6501       !N0.hasOneUse() || LN0->isVolatile() || !DstVT.isVector() ||
6502       !DstVT.isPow2VectorType() || !TLI.isVectorLoadExtDesirable(SDValue(N, 0)))
6503     return SDValue();
6504 
6505   SmallVector<SDNode *, 4> SetCCs;
6506   if (!ExtendUsesToFormExtLoad(N, N0, N->getOpcode(), SetCCs, TLI))
6507     return SDValue();
6508 
6509   ISD::LoadExtType ExtType =
6510       N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
6511 
6512   // Try to split the vector types to get down to legal types.
6513   EVT SplitSrcVT = SrcVT;
6514   EVT SplitDstVT = DstVT;
6515   while (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT) &&
6516          SplitSrcVT.getVectorNumElements() > 1) {
6517     SplitDstVT = DAG.GetSplitDestVTs(SplitDstVT).first;
6518     SplitSrcVT = DAG.GetSplitDestVTs(SplitSrcVT).first;
6519   }
6520 
6521   if (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT))
6522     return SDValue();
6523 
6524   SDLoc DL(N);
6525   const unsigned NumSplits =
6526       DstVT.getVectorNumElements() / SplitDstVT.getVectorNumElements();
6527   const unsigned Stride = SplitSrcVT.getStoreSize();
6528   SmallVector<SDValue, 4> Loads;
6529   SmallVector<SDValue, 4> Chains;
6530 
6531   SDValue BasePtr = LN0->getBasePtr();
6532   for (unsigned Idx = 0; Idx < NumSplits; Idx++) {
6533     const unsigned Offset = Idx * Stride;
6534     const unsigned Align = MinAlign(LN0->getAlignment(), Offset);
6535 
6536     SDValue SplitLoad = DAG.getExtLoad(
6537         ExtType, DL, SplitDstVT, LN0->getChain(), BasePtr,
6538         LN0->getPointerInfo().getWithOffset(Offset), SplitSrcVT, Align,
6539         LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
6540 
6541     BasePtr = DAG.getNode(ISD::ADD, DL, BasePtr.getValueType(), BasePtr,
6542                           DAG.getConstant(Stride, DL, BasePtr.getValueType()));
6543 
6544     Loads.push_back(SplitLoad.getValue(0));
6545     Chains.push_back(SplitLoad.getValue(1));
6546   }
6547 
6548   SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
6549   SDValue NewValue = DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Loads);
6550 
6551   CombineTo(N, NewValue);
6552 
6553   // Replace uses of the original load (before extension)
6554   // with a truncate of the concatenated sextloaded vectors.
6555   SDValue Trunc =
6556       DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), NewValue);
6557   CombineTo(N0.getNode(), Trunc, NewChain);
6558   ExtendSetCCUses(SetCCs, Trunc, NewValue, DL,
6559                   (ISD::NodeType)N->getOpcode());
6560   return SDValue(N, 0); // Return N so it doesn't get rechecked!
6561 }
6562 
6563 SDValue DAGCombiner::visitSIGN_EXTEND(SDNode *N) {
6564   SDValue N0 = N->getOperand(0);
6565   EVT VT = N->getValueType(0);
6566   SDLoc DL(N);
6567 
6568   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6569                                               LegalOperations))
6570     return SDValue(Res, 0);
6571 
6572   // fold (sext (sext x)) -> (sext x)
6573   // fold (sext (aext x)) -> (sext x)
6574   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
6575     return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, N0.getOperand(0));
6576 
6577   if (N0.getOpcode() == ISD::TRUNCATE) {
6578     // fold (sext (truncate (load x))) -> (sext (smaller load x))
6579     // fold (sext (truncate (srl (load x), c))) -> (sext (smaller load (x+c/n)))
6580     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6581       SDNode *oye = N0.getOperand(0).getNode();
6582       if (NarrowLoad.getNode() != N0.getNode()) {
6583         CombineTo(N0.getNode(), NarrowLoad);
6584         // CombineTo deleted the truncate, if needed, but not what's under it.
6585         AddToWorklist(oye);
6586       }
6587       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6588     }
6589 
6590     // See if the value being truncated is already sign extended.  If so, just
6591     // eliminate the trunc/sext pair.
6592     SDValue Op = N0.getOperand(0);
6593     unsigned OpBits   = Op.getScalarValueSizeInBits();
6594     unsigned MidBits  = N0.getScalarValueSizeInBits();
6595     unsigned DestBits = VT.getScalarSizeInBits();
6596     unsigned NumSignBits = DAG.ComputeNumSignBits(Op);
6597 
6598     if (OpBits == DestBits) {
6599       // Op is i32, Mid is i8, and Dest is i32.  If Op has more than 24 sign
6600       // bits, it is already ready.
6601       if (NumSignBits > DestBits-MidBits)
6602         return Op;
6603     } else if (OpBits < DestBits) {
6604       // Op is i32, Mid is i8, and Dest is i64.  If Op has more than 24 sign
6605       // bits, just sext from i32.
6606       if (NumSignBits > OpBits-MidBits)
6607         return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Op);
6608     } else {
6609       // Op is i64, Mid is i8, and Dest is i32.  If Op has more than 56 sign
6610       // bits, just truncate to i32.
6611       if (NumSignBits > OpBits-MidBits)
6612         return DAG.getNode(ISD::TRUNCATE, DL, VT, Op);
6613     }
6614 
6615     // fold (sext (truncate x)) -> (sextinreg x).
6616     if (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG,
6617                                                  N0.getValueType())) {
6618       if (OpBits < DestBits)
6619         Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N0), VT, Op);
6620       else if (OpBits > DestBits)
6621         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), VT, Op);
6622       return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, Op,
6623                          DAG.getValueType(N0.getValueType()));
6624     }
6625   }
6626 
6627   // fold (sext (load x)) -> (sext (truncate (sextload x)))
6628   // Only generate vector extloads when 1) they're legal, and 2) they are
6629   // deemed desirable by the target.
6630   if (ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6631       ((!LegalOperations && !VT.isVector() &&
6632         !cast<LoadSDNode>(N0)->isVolatile()) ||
6633        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, N0.getValueType()))) {
6634     bool DoXform = true;
6635     SmallVector<SDNode*, 4> SetCCs;
6636     if (!N0.hasOneUse())
6637       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::SIGN_EXTEND, SetCCs, TLI);
6638     if (VT.isVector())
6639       DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0));
6640     if (DoXform) {
6641       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6642       SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, DL, VT, LN0->getChain(),
6643                                        LN0->getBasePtr(), N0.getValueType(),
6644                                        LN0->getMemOperand());
6645       CombineTo(N, ExtLoad);
6646       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6647                                   N0.getValueType(), ExtLoad);
6648       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6649       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, DL, ISD::SIGN_EXTEND);
6650       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6651     }
6652   }
6653 
6654   // fold (sext (load x)) to multiple smaller sextloads.
6655   // Only on illegal but splittable vectors.
6656   if (SDValue ExtLoad = CombineExtLoad(N))
6657     return ExtLoad;
6658 
6659   // fold (sext (sextload x)) -> (sext (truncate (sextload x)))
6660   // fold (sext ( extload x)) -> (sext (truncate (sextload x)))
6661   if ((ISD::isSEXTLoad(N0.getNode()) || ISD::isEXTLoad(N0.getNode())) &&
6662       ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) {
6663     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6664     EVT MemVT = LN0->getMemoryVT();
6665     if ((!LegalOperations && !LN0->isVolatile()) ||
6666         TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, MemVT)) {
6667       SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, DL, VT, LN0->getChain(),
6668                                        LN0->getBasePtr(), MemVT,
6669                                        LN0->getMemOperand());
6670       CombineTo(N, ExtLoad);
6671       CombineTo(N0.getNode(),
6672                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6673                             N0.getValueType(), ExtLoad),
6674                 ExtLoad.getValue(1));
6675       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6676     }
6677   }
6678 
6679   // fold (sext (and/or/xor (load x), cst)) ->
6680   //      (and/or/xor (sextload x), (sext cst))
6681   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
6682        N0.getOpcode() == ISD::XOR) &&
6683       isa<LoadSDNode>(N0.getOperand(0)) &&
6684       N0.getOperand(1).getOpcode() == ISD::Constant &&
6685       TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, N0.getValueType()) &&
6686       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
6687     LoadSDNode *LN0 = cast<LoadSDNode>(N0.getOperand(0));
6688     if (LN0->getExtensionType() != ISD::ZEXTLOAD && LN0->isUnindexed()) {
6689       bool DoXform = true;
6690       SmallVector<SDNode*, 4> SetCCs;
6691       if (!N0.hasOneUse())
6692         DoXform = ExtendUsesToFormExtLoad(N, N0.getOperand(0), ISD::SIGN_EXTEND,
6693                                           SetCCs, TLI);
6694       if (DoXform) {
6695         SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(LN0), VT,
6696                                          LN0->getChain(), LN0->getBasePtr(),
6697                                          LN0->getMemoryVT(),
6698                                          LN0->getMemOperand());
6699         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6700         Mask = Mask.sext(VT.getSizeInBits());
6701         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
6702                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
6703         SDValue Trunc = DAG.getNode(ISD::TRUNCATE,
6704                                     SDLoc(N0.getOperand(0)),
6705                                     N0.getOperand(0).getValueType(), ExtLoad);
6706         CombineTo(N, And);
6707         CombineTo(N0.getOperand(0).getNode(), Trunc, ExtLoad.getValue(1));
6708         ExtendSetCCUses(SetCCs, Trunc, ExtLoad, DL, ISD::SIGN_EXTEND);
6709         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6710       }
6711     }
6712   }
6713 
6714   if (N0.getOpcode() == ISD::SETCC) {
6715     SDValue N00 = N0.getOperand(0);
6716     SDValue N01 = N0.getOperand(1);
6717     ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
6718     EVT N00VT = N0.getOperand(0).getValueType();
6719 
6720     // sext(setcc) -> sext_in_reg(vsetcc) for vectors.
6721     // Only do this before legalize for now.
6722     if (VT.isVector() && !LegalOperations &&
6723         TLI.getBooleanContents(N00VT) ==
6724             TargetLowering::ZeroOrNegativeOneBooleanContent) {
6725       // On some architectures (such as SSE/NEON/etc) the SETCC result type is
6726       // of the same size as the compared operands. Only optimize sext(setcc())
6727       // if this is the case.
6728       EVT SVT = getSetCCResultType(N00VT);
6729 
6730       // We know that the # elements of the results is the same as the
6731       // # elements of the compare (and the # elements of the compare result
6732       // for that matter).  Check to see that they are the same size.  If so,
6733       // we know that the element size of the sext'd result matches the
6734       // element size of the compare operands.
6735       if (VT.getSizeInBits() == SVT.getSizeInBits())
6736         return DAG.getSetCC(DL, VT, N00, N01, CC);
6737 
6738       // If the desired elements are smaller or larger than the source
6739       // elements, we can use a matching integer vector type and then
6740       // truncate/sign extend.
6741       EVT MatchingVecType = N00VT.changeVectorElementTypeToInteger();
6742       if (SVT == MatchingVecType) {
6743         SDValue VsetCC = DAG.getSetCC(DL, MatchingVecType, N00, N01, CC);
6744         return DAG.getSExtOrTrunc(VsetCC, DL, VT);
6745       }
6746     }
6747 
6748     // sext(setcc x, y, cc) -> (select (setcc x, y, cc), T, 0)
6749     // Here, T can be 1 or -1, depending on the type of the setcc and
6750     // getBooleanContents().
6751     unsigned SetCCWidth = N0.getScalarValueSizeInBits();
6752 
6753     // To determine the "true" side of the select, we need to know the high bit
6754     // of the value returned by the setcc if it evaluates to true.
6755     // If the type of the setcc is i1, then the true case of the select is just
6756     // sext(i1 1), that is, -1.
6757     // If the type of the setcc is larger (say, i8) then the value of the high
6758     // bit depends on getBooleanContents(), so ask TLI for a real "true" value
6759     // of the appropriate width.
6760     SDValue ExtTrueVal = (SetCCWidth == 1) ? DAG.getAllOnesConstant(DL, VT)
6761                                            : TLI.getConstTrueVal(DAG, VT, DL);
6762     SDValue Zero = DAG.getConstant(0, DL, VT);
6763     if (SDValue SCC =
6764             SimplifySelectCC(DL, N00, N01, ExtTrueVal, Zero, CC, true))
6765       return SCC;
6766 
6767     if (!VT.isVector()) {
6768       EVT SetCCVT = getSetCCResultType(N00VT);
6769       if (!LegalOperations || TLI.isOperationLegal(ISD::SETCC, N00VT)) {
6770         SDValue SetCC = DAG.getSetCC(DL, SetCCVT, N00, N01, CC);
6771         return DAG.getSelect(DL, VT, SetCC, ExtTrueVal, Zero);
6772       }
6773     }
6774   }
6775 
6776   // fold (sext x) -> (zext x) if the sign bit is known zero.
6777   if ((!LegalOperations || TLI.isOperationLegal(ISD::ZERO_EXTEND, VT)) &&
6778       DAG.SignBitIsZero(N0))
6779     return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0);
6780 
6781   return SDValue();
6782 }
6783 
6784 // isTruncateOf - If N is a truncate of some other value, return true, record
6785 // the value being truncated in Op and which of Op's bits are zero in KnownZero.
6786 // This function computes KnownZero to avoid a duplicated call to
6787 // computeKnownBits in the caller.
6788 static bool isTruncateOf(SelectionDAG &DAG, SDValue N, SDValue &Op,
6789                          APInt &KnownZero) {
6790   APInt KnownOne;
6791   if (N->getOpcode() == ISD::TRUNCATE) {
6792     Op = N->getOperand(0);
6793     DAG.computeKnownBits(Op, KnownZero, KnownOne);
6794     return true;
6795   }
6796 
6797   if (N->getOpcode() != ISD::SETCC || N->getValueType(0) != MVT::i1 ||
6798       cast<CondCodeSDNode>(N->getOperand(2))->get() != ISD::SETNE)
6799     return false;
6800 
6801   SDValue Op0 = N->getOperand(0);
6802   SDValue Op1 = N->getOperand(1);
6803   assert(Op0.getValueType() == Op1.getValueType());
6804 
6805   if (isNullConstant(Op0))
6806     Op = Op1;
6807   else if (isNullConstant(Op1))
6808     Op = Op0;
6809   else
6810     return false;
6811 
6812   DAG.computeKnownBits(Op, KnownZero, KnownOne);
6813 
6814   if (!(KnownZero | APInt(Op.getValueSizeInBits(), 1)).isAllOnesValue())
6815     return false;
6816 
6817   return true;
6818 }
6819 
6820 SDValue DAGCombiner::visitZERO_EXTEND(SDNode *N) {
6821   SDValue N0 = N->getOperand(0);
6822   EVT VT = N->getValueType(0);
6823 
6824   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
6825                                               LegalOperations))
6826     return SDValue(Res, 0);
6827 
6828   // fold (zext (zext x)) -> (zext x)
6829   // fold (zext (aext x)) -> (zext x)
6830   if (N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND)
6831     return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT,
6832                        N0.getOperand(0));
6833 
6834   // fold (zext (truncate x)) -> (zext x) or
6835   //      (zext (truncate x)) -> (truncate x)
6836   // This is valid when the truncated bits of x are already zero.
6837   // FIXME: We should extend this to work for vectors too.
6838   SDValue Op;
6839   APInt KnownZero;
6840   if (!VT.isVector() && isTruncateOf(DAG, N0, Op, KnownZero)) {
6841     APInt TruncatedBits =
6842       (Op.getValueSizeInBits() == N0.getValueSizeInBits()) ?
6843       APInt(Op.getValueSizeInBits(), 0) :
6844       APInt::getBitsSet(Op.getValueSizeInBits(),
6845                         N0.getValueSizeInBits(),
6846                         std::min(Op.getValueSizeInBits(),
6847                                  VT.getSizeInBits()));
6848     if (TruncatedBits == (KnownZero & TruncatedBits)) {
6849       if (VT.bitsGT(Op.getValueType()))
6850         return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, Op);
6851       if (VT.bitsLT(Op.getValueType()))
6852         return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6853 
6854       return Op;
6855     }
6856   }
6857 
6858   // fold (zext (truncate (load x))) -> (zext (smaller load x))
6859   // fold (zext (truncate (srl (load x), c))) -> (zext (small load (x+c/n)))
6860   if (N0.getOpcode() == ISD::TRUNCATE) {
6861     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6862       SDNode *oye = N0.getOperand(0).getNode();
6863       if (NarrowLoad.getNode() != N0.getNode()) {
6864         CombineTo(N0.getNode(), NarrowLoad);
6865         // CombineTo deleted the truncate, if needed, but not what's under it.
6866         AddToWorklist(oye);
6867       }
6868       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6869     }
6870   }
6871 
6872   // fold (zext (truncate x)) -> (and x, mask)
6873   if (N0.getOpcode() == ISD::TRUNCATE) {
6874     // fold (zext (truncate (load x))) -> (zext (smaller load x))
6875     // fold (zext (truncate (srl (load x), c))) -> (zext (smaller load (x+c/n)))
6876     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
6877       SDNode *oye = N0.getOperand(0).getNode();
6878       if (NarrowLoad.getNode() != N0.getNode()) {
6879         CombineTo(N0.getNode(), NarrowLoad);
6880         // CombineTo deleted the truncate, if needed, but not what's under it.
6881         AddToWorklist(oye);
6882       }
6883       return SDValue(N, 0); // Return N so it doesn't get rechecked!
6884     }
6885 
6886     EVT SrcVT = N0.getOperand(0).getValueType();
6887     EVT MinVT = N0.getValueType();
6888 
6889     // Try to mask before the extension to avoid having to generate a larger mask,
6890     // possibly over several sub-vectors.
6891     if (SrcVT.bitsLT(VT)) {
6892       if (!LegalOperations || (TLI.isOperationLegal(ISD::AND, SrcVT) &&
6893                                TLI.isOperationLegal(ISD::ZERO_EXTEND, VT))) {
6894         SDValue Op = N0.getOperand(0);
6895         Op = DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
6896         AddToWorklist(Op.getNode());
6897         return DAG.getZExtOrTrunc(Op, SDLoc(N), VT);
6898       }
6899     }
6900 
6901     if (!LegalOperations || TLI.isOperationLegal(ISD::AND, VT)) {
6902       SDValue Op = N0.getOperand(0);
6903       if (SrcVT.bitsLT(VT)) {
6904         Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, Op);
6905         AddToWorklist(Op.getNode());
6906       } else if (SrcVT.bitsGT(VT)) {
6907         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Op);
6908         AddToWorklist(Op.getNode());
6909       }
6910       return DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType());
6911     }
6912   }
6913 
6914   // Fold (zext (and (trunc x), cst)) -> (and x, cst),
6915   // if either of the casts is not free.
6916   if (N0.getOpcode() == ISD::AND &&
6917       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
6918       N0.getOperand(1).getOpcode() == ISD::Constant &&
6919       (!TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
6920                            N0.getValueType()) ||
6921        !TLI.isZExtFree(N0.getValueType(), VT))) {
6922     SDValue X = N0.getOperand(0).getOperand(0);
6923     if (X.getValueType().bitsLT(VT)) {
6924       X = DAG.getNode(ISD::ANY_EXTEND, SDLoc(X), VT, X);
6925     } else if (X.getValueType().bitsGT(VT)) {
6926       X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
6927     }
6928     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
6929     Mask = Mask.zext(VT.getSizeInBits());
6930     SDLoc DL(N);
6931     return DAG.getNode(ISD::AND, DL, VT,
6932                        X, DAG.getConstant(Mask, DL, VT));
6933   }
6934 
6935   // fold (zext (load x)) -> (zext (truncate (zextload x)))
6936   // Only generate vector extloads when 1) they're legal, and 2) they are
6937   // deemed desirable by the target.
6938   if (ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
6939       ((!LegalOperations && !VT.isVector() &&
6940         !cast<LoadSDNode>(N0)->isVolatile()) ||
6941        TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, N0.getValueType()))) {
6942     bool DoXform = true;
6943     SmallVector<SDNode*, 4> SetCCs;
6944     if (!N0.hasOneUse())
6945       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::ZERO_EXTEND, SetCCs, TLI);
6946     if (VT.isVector())
6947       DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0));
6948     if (DoXform) {
6949       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
6950       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N), VT,
6951                                        LN0->getChain(),
6952                                        LN0->getBasePtr(), N0.getValueType(),
6953                                        LN0->getMemOperand());
6954       CombineTo(N, ExtLoad);
6955       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
6956                                   N0.getValueType(), ExtLoad);
6957       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
6958 
6959       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
6960                       ISD::ZERO_EXTEND);
6961       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
6962     }
6963   }
6964 
6965   // fold (zext (load x)) to multiple smaller zextloads.
6966   // Only on illegal but splittable vectors.
6967   if (SDValue ExtLoad = CombineExtLoad(N))
6968     return ExtLoad;
6969 
6970   // fold (zext (and/or/xor (load x), cst)) ->
6971   //      (and/or/xor (zextload x), (zext cst))
6972   // Unless (and (load x) cst) will match as a zextload already and has
6973   // additional users.
6974   if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR ||
6975        N0.getOpcode() == ISD::XOR) &&
6976       isa<LoadSDNode>(N0.getOperand(0)) &&
6977       N0.getOperand(1).getOpcode() == ISD::Constant &&
6978       TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, N0.getValueType()) &&
6979       (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) {
6980     LoadSDNode *LN0 = cast<LoadSDNode>(N0.getOperand(0));
6981     if (LN0->getExtensionType() != ISD::SEXTLOAD && LN0->isUnindexed()) {
6982       bool DoXform = true;
6983       SmallVector<SDNode*, 4> SetCCs;
6984       if (!N0.hasOneUse()) {
6985         if (N0.getOpcode() == ISD::AND) {
6986           auto *AndC = cast<ConstantSDNode>(N0.getOperand(1));
6987           auto NarrowLoad = false;
6988           EVT LoadResultTy = AndC->getValueType(0);
6989           EVT ExtVT, LoadedVT;
6990           if (isAndLoadExtLoad(AndC, LN0, LoadResultTy, ExtVT, LoadedVT,
6991                                NarrowLoad))
6992             DoXform = false;
6993         }
6994         if (DoXform)
6995           DoXform = ExtendUsesToFormExtLoad(N, N0.getOperand(0),
6996                                             ISD::ZERO_EXTEND, SetCCs, TLI);
6997       }
6998       if (DoXform) {
6999         SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN0), VT,
7000                                          LN0->getChain(), LN0->getBasePtr(),
7001                                          LN0->getMemoryVT(),
7002                                          LN0->getMemOperand());
7003         APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
7004         Mask = Mask.zext(VT.getSizeInBits());
7005         SDLoc DL(N);
7006         SDValue And = DAG.getNode(N0.getOpcode(), DL, VT,
7007                                   ExtLoad, DAG.getConstant(Mask, DL, VT));
7008         SDValue Trunc = DAG.getNode(ISD::TRUNCATE,
7009                                     SDLoc(N0.getOperand(0)),
7010                                     N0.getOperand(0).getValueType(), ExtLoad);
7011         CombineTo(N, And);
7012         CombineTo(N0.getOperand(0).getNode(), Trunc, ExtLoad.getValue(1));
7013         ExtendSetCCUses(SetCCs, Trunc, ExtLoad, DL,
7014                         ISD::ZERO_EXTEND);
7015         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7016       }
7017     }
7018   }
7019 
7020   // fold (zext (zextload x)) -> (zext (truncate (zextload x)))
7021   // fold (zext ( extload x)) -> (zext (truncate (zextload x)))
7022   if ((ISD::isZEXTLoad(N0.getNode()) || ISD::isEXTLoad(N0.getNode())) &&
7023       ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) {
7024     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7025     EVT MemVT = LN0->getMemoryVT();
7026     if ((!LegalOperations && !LN0->isVolatile()) ||
7027         TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT)) {
7028       SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N), VT,
7029                                        LN0->getChain(),
7030                                        LN0->getBasePtr(), MemVT,
7031                                        LN0->getMemOperand());
7032       CombineTo(N, ExtLoad);
7033       CombineTo(N0.getNode(),
7034                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(),
7035                             ExtLoad),
7036                 ExtLoad.getValue(1));
7037       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7038     }
7039   }
7040 
7041   if (N0.getOpcode() == ISD::SETCC) {
7042     // Only do this before legalize for now.
7043     if (!LegalOperations && VT.isVector() &&
7044         N0.getValueType().getVectorElementType() == MVT::i1) {
7045       EVT N00VT = N0.getOperand(0).getValueType();
7046       if (getSetCCResultType(N00VT) == N0.getValueType())
7047         return SDValue();
7048 
7049       // We know that the # elements of the results is the same as the #
7050       // elements of the compare (and the # elements of the compare result for
7051       // that matter). Check to see that they are the same size. If so, we know
7052       // that the element size of the sext'd result matches the element size of
7053       // the compare operands.
7054       SDLoc DL(N);
7055       SDValue VecOnes = DAG.getConstant(1, DL, VT);
7056       if (VT.getSizeInBits() == N00VT.getSizeInBits()) {
7057         // zext(setcc) -> (and (vsetcc), (1, 1, ...) for vectors.
7058         SDValue VSetCC = DAG.getNode(ISD::SETCC, DL, VT, N0.getOperand(0),
7059                                      N0.getOperand(1), N0.getOperand(2));
7060         return DAG.getNode(ISD::AND, DL, VT, VSetCC, VecOnes);
7061       }
7062 
7063       // If the desired elements are smaller or larger than the source
7064       // elements we can use a matching integer vector type and then
7065       // truncate/sign extend.
7066       EVT MatchingElementType = EVT::getIntegerVT(
7067           *DAG.getContext(), N00VT.getScalarSizeInBits());
7068       EVT MatchingVectorType = EVT::getVectorVT(
7069           *DAG.getContext(), MatchingElementType, N00VT.getVectorNumElements());
7070       SDValue VsetCC =
7071           DAG.getNode(ISD::SETCC, DL, MatchingVectorType, N0.getOperand(0),
7072                       N0.getOperand(1), N0.getOperand(2));
7073       return DAG.getNode(ISD::AND, DL, VT, DAG.getSExtOrTrunc(VsetCC, DL, VT),
7074                          VecOnes);
7075     }
7076 
7077     // zext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
7078     SDLoc DL(N);
7079     if (SDValue SCC = SimplifySelectCC(
7080             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
7081             DAG.getConstant(0, DL, VT),
7082             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
7083       return SCC;
7084   }
7085 
7086   // (zext (shl (zext x), cst)) -> (shl (zext x), cst)
7087   if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL) &&
7088       isa<ConstantSDNode>(N0.getOperand(1)) &&
7089       N0.getOperand(0).getOpcode() == ISD::ZERO_EXTEND &&
7090       N0.hasOneUse()) {
7091     SDValue ShAmt = N0.getOperand(1);
7092     unsigned ShAmtVal = cast<ConstantSDNode>(ShAmt)->getZExtValue();
7093     if (N0.getOpcode() == ISD::SHL) {
7094       SDValue InnerZExt = N0.getOperand(0);
7095       // If the original shl may be shifting out bits, do not perform this
7096       // transformation.
7097       unsigned KnownZeroBits = InnerZExt.getValueSizeInBits() -
7098         InnerZExt.getOperand(0).getValueSizeInBits();
7099       if (ShAmtVal > KnownZeroBits)
7100         return SDValue();
7101     }
7102 
7103     SDLoc DL(N);
7104 
7105     // Ensure that the shift amount is wide enough for the shifted value.
7106     if (VT.getSizeInBits() >= 256)
7107       ShAmt = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, ShAmt);
7108 
7109     return DAG.getNode(N0.getOpcode(), DL, VT,
7110                        DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0)),
7111                        ShAmt);
7112   }
7113 
7114   return SDValue();
7115 }
7116 
7117 SDValue DAGCombiner::visitANY_EXTEND(SDNode *N) {
7118   SDValue N0 = N->getOperand(0);
7119   EVT VT = N->getValueType(0);
7120 
7121   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
7122                                               LegalOperations))
7123     return SDValue(Res, 0);
7124 
7125   // fold (aext (aext x)) -> (aext x)
7126   // fold (aext (zext x)) -> (zext x)
7127   // fold (aext (sext x)) -> (sext x)
7128   if (N0.getOpcode() == ISD::ANY_EXTEND  ||
7129       N0.getOpcode() == ISD::ZERO_EXTEND ||
7130       N0.getOpcode() == ISD::SIGN_EXTEND)
7131     return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
7132 
7133   // fold (aext (truncate (load x))) -> (aext (smaller load x))
7134   // fold (aext (truncate (srl (load x), c))) -> (aext (small load (x+c/n)))
7135   if (N0.getOpcode() == ISD::TRUNCATE) {
7136     if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) {
7137       SDNode *oye = N0.getOperand(0).getNode();
7138       if (NarrowLoad.getNode() != N0.getNode()) {
7139         CombineTo(N0.getNode(), NarrowLoad);
7140         // CombineTo deleted the truncate, if needed, but not what's under it.
7141         AddToWorklist(oye);
7142       }
7143       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7144     }
7145   }
7146 
7147   // fold (aext (truncate x))
7148   if (N0.getOpcode() == ISD::TRUNCATE) {
7149     SDValue TruncOp = N0.getOperand(0);
7150     if (TruncOp.getValueType() == VT)
7151       return TruncOp; // x iff x size == zext size.
7152     if (TruncOp.getValueType().bitsGT(VT))
7153       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, TruncOp);
7154     return DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), VT, TruncOp);
7155   }
7156 
7157   // Fold (aext (and (trunc x), cst)) -> (and x, cst)
7158   // if the trunc is not free.
7159   if (N0.getOpcode() == ISD::AND &&
7160       N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
7161       N0.getOperand(1).getOpcode() == ISD::Constant &&
7162       !TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(),
7163                           N0.getValueType())) {
7164     SDLoc DL(N);
7165     SDValue X = N0.getOperand(0).getOperand(0);
7166     if (X.getValueType().bitsLT(VT)) {
7167       X = DAG.getNode(ISD::ANY_EXTEND, DL, VT, X);
7168     } else if (X.getValueType().bitsGT(VT)) {
7169       X = DAG.getNode(ISD::TRUNCATE, DL, VT, X);
7170     }
7171     APInt Mask = cast<ConstantSDNode>(N0.getOperand(1))->getAPIntValue();
7172     Mask = Mask.zext(VT.getSizeInBits());
7173     return DAG.getNode(ISD::AND, DL, VT,
7174                        X, DAG.getConstant(Mask, DL, VT));
7175   }
7176 
7177   // fold (aext (load x)) -> (aext (truncate (extload x)))
7178   // None of the supported targets knows how to perform load and any_ext
7179   // on vectors in one instruction.  We only perform this transformation on
7180   // scalars.
7181   if (ISD::isNON_EXTLoad(N0.getNode()) && !VT.isVector() &&
7182       ISD::isUNINDEXEDLoad(N0.getNode()) &&
7183       TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
7184     bool DoXform = true;
7185     SmallVector<SDNode*, 4> SetCCs;
7186     if (!N0.hasOneUse())
7187       DoXform = ExtendUsesToFormExtLoad(N, N0, ISD::ANY_EXTEND, SetCCs, TLI);
7188     if (DoXform) {
7189       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7190       SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
7191                                        LN0->getChain(),
7192                                        LN0->getBasePtr(), N0.getValueType(),
7193                                        LN0->getMemOperand());
7194       CombineTo(N, ExtLoad);
7195       SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
7196                                   N0.getValueType(), ExtLoad);
7197       CombineTo(N0.getNode(), Trunc, ExtLoad.getValue(1));
7198       ExtendSetCCUses(SetCCs, Trunc, ExtLoad, SDLoc(N),
7199                       ISD::ANY_EXTEND);
7200       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7201     }
7202   }
7203 
7204   // fold (aext (zextload x)) -> (aext (truncate (zextload x)))
7205   // fold (aext (sextload x)) -> (aext (truncate (sextload x)))
7206   // fold (aext ( extload x)) -> (aext (truncate (extload  x)))
7207   if (N0.getOpcode() == ISD::LOAD &&
7208       !ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
7209       N0.hasOneUse()) {
7210     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7211     ISD::LoadExtType ExtType = LN0->getExtensionType();
7212     EVT MemVT = LN0->getMemoryVT();
7213     if (!LegalOperations || TLI.isLoadExtLegal(ExtType, VT, MemVT)) {
7214       SDValue ExtLoad = DAG.getExtLoad(ExtType, SDLoc(N),
7215                                        VT, LN0->getChain(), LN0->getBasePtr(),
7216                                        MemVT, LN0->getMemOperand());
7217       CombineTo(N, ExtLoad);
7218       CombineTo(N0.getNode(),
7219                 DAG.getNode(ISD::TRUNCATE, SDLoc(N0),
7220                             N0.getValueType(), ExtLoad),
7221                 ExtLoad.getValue(1));
7222       return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7223     }
7224   }
7225 
7226   if (N0.getOpcode() == ISD::SETCC) {
7227     // For vectors:
7228     // aext(setcc) -> vsetcc
7229     // aext(setcc) -> truncate(vsetcc)
7230     // aext(setcc) -> aext(vsetcc)
7231     // Only do this before legalize for now.
7232     if (VT.isVector() && !LegalOperations) {
7233       EVT N0VT = N0.getOperand(0).getValueType();
7234         // We know that the # elements of the results is the same as the
7235         // # elements of the compare (and the # elements of the compare result
7236         // for that matter).  Check to see that they are the same size.  If so,
7237         // we know that the element size of the sext'd result matches the
7238         // element size of the compare operands.
7239       if (VT.getSizeInBits() == N0VT.getSizeInBits())
7240         return DAG.getSetCC(SDLoc(N), VT, N0.getOperand(0),
7241                              N0.getOperand(1),
7242                              cast<CondCodeSDNode>(N0.getOperand(2))->get());
7243       // If the desired elements are smaller or larger than the source
7244       // elements we can use a matching integer vector type and then
7245       // truncate/any extend
7246       else {
7247         EVT MatchingVectorType = N0VT.changeVectorElementTypeToInteger();
7248         SDValue VsetCC =
7249           DAG.getSetCC(SDLoc(N), MatchingVectorType, N0.getOperand(0),
7250                         N0.getOperand(1),
7251                         cast<CondCodeSDNode>(N0.getOperand(2))->get());
7252         return DAG.getAnyExtOrTrunc(VsetCC, SDLoc(N), VT);
7253       }
7254     }
7255 
7256     // aext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc
7257     SDLoc DL(N);
7258     if (SDValue SCC = SimplifySelectCC(
7259             DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT),
7260             DAG.getConstant(0, DL, VT),
7261             cast<CondCodeSDNode>(N0.getOperand(2))->get(), true))
7262       return SCC;
7263   }
7264 
7265   return SDValue();
7266 }
7267 
7268 /// See if the specified operand can be simplified with the knowledge that only
7269 /// the bits specified by Mask are used.  If so, return the simpler operand,
7270 /// otherwise return a null SDValue.
7271 SDValue DAGCombiner::GetDemandedBits(SDValue V, const APInt &Mask) {
7272   switch (V.getOpcode()) {
7273   default: break;
7274   case ISD::Constant: {
7275     const ConstantSDNode *CV = cast<ConstantSDNode>(V.getNode());
7276     assert(CV && "Const value should be ConstSDNode.");
7277     const APInt &CVal = CV->getAPIntValue();
7278     APInt NewVal = CVal & Mask;
7279     if (NewVal != CVal)
7280       return DAG.getConstant(NewVal, SDLoc(V), V.getValueType());
7281     break;
7282   }
7283   case ISD::OR:
7284   case ISD::XOR:
7285     // If the LHS or RHS don't contribute bits to the or, drop them.
7286     if (DAG.MaskedValueIsZero(V.getOperand(0), Mask))
7287       return V.getOperand(1);
7288     if (DAG.MaskedValueIsZero(V.getOperand(1), Mask))
7289       return V.getOperand(0);
7290     break;
7291   case ISD::SRL:
7292     // Only look at single-use SRLs.
7293     if (!V.getNode()->hasOneUse())
7294       break;
7295     if (ConstantSDNode *RHSC = getAsNonOpaqueConstant(V.getOperand(1))) {
7296       // See if we can recursively simplify the LHS.
7297       unsigned Amt = RHSC->getZExtValue();
7298 
7299       // Watch out for shift count overflow though.
7300       if (Amt >= Mask.getBitWidth()) break;
7301       APInt NewMask = Mask << Amt;
7302       if (SDValue SimplifyLHS = GetDemandedBits(V.getOperand(0), NewMask))
7303         return DAG.getNode(ISD::SRL, SDLoc(V), V.getValueType(),
7304                            SimplifyLHS, V.getOperand(1));
7305     }
7306   }
7307   return SDValue();
7308 }
7309 
7310 /// If the result of a wider load is shifted to right of N  bits and then
7311 /// truncated to a narrower type and where N is a multiple of number of bits of
7312 /// the narrower type, transform it to a narrower load from address + N / num of
7313 /// bits of new type. If the result is to be extended, also fold the extension
7314 /// to form a extending load.
7315 SDValue DAGCombiner::ReduceLoadWidth(SDNode *N) {
7316   unsigned Opc = N->getOpcode();
7317 
7318   ISD::LoadExtType ExtType = ISD::NON_EXTLOAD;
7319   SDValue N0 = N->getOperand(0);
7320   EVT VT = N->getValueType(0);
7321   EVT ExtVT = VT;
7322 
7323   // This transformation isn't valid for vector loads.
7324   if (VT.isVector())
7325     return SDValue();
7326 
7327   // Special case: SIGN_EXTEND_INREG is basically truncating to ExtVT then
7328   // extended to VT.
7329   if (Opc == ISD::SIGN_EXTEND_INREG) {
7330     ExtType = ISD::SEXTLOAD;
7331     ExtVT = cast<VTSDNode>(N->getOperand(1))->getVT();
7332   } else if (Opc == ISD::SRL) {
7333     // Another special-case: SRL is basically zero-extending a narrower value.
7334     ExtType = ISD::ZEXTLOAD;
7335     N0 = SDValue(N, 0);
7336     ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7337     if (!N01) return SDValue();
7338     ExtVT = EVT::getIntegerVT(*DAG.getContext(),
7339                               VT.getSizeInBits() - N01->getZExtValue());
7340   }
7341   if (LegalOperations && !TLI.isLoadExtLegal(ExtType, VT, ExtVT))
7342     return SDValue();
7343 
7344   unsigned EVTBits = ExtVT.getSizeInBits();
7345 
7346   // Do not generate loads of non-round integer types since these can
7347   // be expensive (and would be wrong if the type is not byte sized).
7348   if (!ExtVT.isRound())
7349     return SDValue();
7350 
7351   unsigned ShAmt = 0;
7352   if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) {
7353     if (ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
7354       ShAmt = N01->getZExtValue();
7355       // Is the shift amount a multiple of size of VT?
7356       if ((ShAmt & (EVTBits-1)) == 0) {
7357         N0 = N0.getOperand(0);
7358         // Is the load width a multiple of size of VT?
7359         if ((N0.getValueSizeInBits() & (EVTBits-1)) != 0)
7360           return SDValue();
7361       }
7362 
7363       // At this point, we must have a load or else we can't do the transform.
7364       if (!isa<LoadSDNode>(N0)) return SDValue();
7365 
7366       // Because a SRL must be assumed to *need* to zero-extend the high bits
7367       // (as opposed to anyext the high bits), we can't combine the zextload
7368       // lowering of SRL and an sextload.
7369       if (cast<LoadSDNode>(N0)->getExtensionType() == ISD::SEXTLOAD)
7370         return SDValue();
7371 
7372       // If the shift amount is larger than the input type then we're not
7373       // accessing any of the loaded bytes.  If the load was a zextload/extload
7374       // then the result of the shift+trunc is zero/undef (handled elsewhere).
7375       if (ShAmt >= cast<LoadSDNode>(N0)->getMemoryVT().getSizeInBits())
7376         return SDValue();
7377     }
7378   }
7379 
7380   // If the load is shifted left (and the result isn't shifted back right),
7381   // we can fold the truncate through the shift.
7382   unsigned ShLeftAmt = 0;
7383   if (ShAmt == 0 && N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
7384       ExtVT == VT && TLI.isNarrowingProfitable(N0.getValueType(), VT)) {
7385     if (ConstantSDNode *N01 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) {
7386       ShLeftAmt = N01->getZExtValue();
7387       N0 = N0.getOperand(0);
7388     }
7389   }
7390 
7391   // If we haven't found a load, we can't narrow it.  Don't transform one with
7392   // multiple uses, this would require adding a new load.
7393   if (!isa<LoadSDNode>(N0) || !N0.hasOneUse())
7394     return SDValue();
7395 
7396   // Don't change the width of a volatile load.
7397   LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7398   if (LN0->isVolatile())
7399     return SDValue();
7400 
7401   // Verify that we are actually reducing a load width here.
7402   if (LN0->getMemoryVT().getSizeInBits() < EVTBits)
7403     return SDValue();
7404 
7405   // For the transform to be legal, the load must produce only two values
7406   // (the value loaded and the chain).  Don't transform a pre-increment
7407   // load, for example, which produces an extra value.  Otherwise the
7408   // transformation is not equivalent, and the downstream logic to replace
7409   // uses gets things wrong.
7410   if (LN0->getNumValues() > 2)
7411     return SDValue();
7412 
7413   // If the load that we're shrinking is an extload and we're not just
7414   // discarding the extension we can't simply shrink the load. Bail.
7415   // TODO: It would be possible to merge the extensions in some cases.
7416   if (LN0->getExtensionType() != ISD::NON_EXTLOAD &&
7417       LN0->getMemoryVT().getSizeInBits() < ExtVT.getSizeInBits() + ShAmt)
7418     return SDValue();
7419 
7420   if (!TLI.shouldReduceLoadWidth(LN0, ExtType, ExtVT))
7421     return SDValue();
7422 
7423   EVT PtrType = N0.getOperand(1).getValueType();
7424 
7425   if (PtrType == MVT::Untyped || PtrType.isExtended())
7426     // It's not possible to generate a constant of extended or untyped type.
7427     return SDValue();
7428 
7429   // For big endian targets, we need to adjust the offset to the pointer to
7430   // load the correct bytes.
7431   if (DAG.getDataLayout().isBigEndian()) {
7432     unsigned LVTStoreBits = LN0->getMemoryVT().getStoreSizeInBits();
7433     unsigned EVTStoreBits = ExtVT.getStoreSizeInBits();
7434     ShAmt = LVTStoreBits - EVTStoreBits - ShAmt;
7435   }
7436 
7437   uint64_t PtrOff = ShAmt / 8;
7438   unsigned NewAlign = MinAlign(LN0->getAlignment(), PtrOff);
7439   SDLoc DL(LN0);
7440   // The original load itself didn't wrap, so an offset within it doesn't.
7441   SDNodeFlags Flags;
7442   Flags.setNoUnsignedWrap(true);
7443   SDValue NewPtr = DAG.getNode(ISD::ADD, DL,
7444                                PtrType, LN0->getBasePtr(),
7445                                DAG.getConstant(PtrOff, DL, PtrType),
7446                                &Flags);
7447   AddToWorklist(NewPtr.getNode());
7448 
7449   SDValue Load;
7450   if (ExtType == ISD::NON_EXTLOAD)
7451     Load = DAG.getLoad(VT, SDLoc(N0), LN0->getChain(), NewPtr,
7452                        LN0->getPointerInfo().getWithOffset(PtrOff), NewAlign,
7453                        LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
7454   else
7455     Load = DAG.getExtLoad(ExtType, SDLoc(N0), VT, LN0->getChain(), NewPtr,
7456                           LN0->getPointerInfo().getWithOffset(PtrOff), ExtVT,
7457                           NewAlign, LN0->getMemOperand()->getFlags(),
7458                           LN0->getAAInfo());
7459 
7460   // Replace the old load's chain with the new load's chain.
7461   WorklistRemover DeadNodes(*this);
7462   DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
7463 
7464   // Shift the result left, if we've swallowed a left shift.
7465   SDValue Result = Load;
7466   if (ShLeftAmt != 0) {
7467     EVT ShImmTy = getShiftAmountTy(Result.getValueType());
7468     if (!isUIntN(ShImmTy.getSizeInBits(), ShLeftAmt))
7469       ShImmTy = VT;
7470     // If the shift amount is as large as the result size (but, presumably,
7471     // no larger than the source) then the useful bits of the result are
7472     // zero; we can't simply return the shortened shift, because the result
7473     // of that operation is undefined.
7474     SDLoc DL(N0);
7475     if (ShLeftAmt >= VT.getSizeInBits())
7476       Result = DAG.getConstant(0, DL, VT);
7477     else
7478       Result = DAG.getNode(ISD::SHL, DL, VT,
7479                           Result, DAG.getConstant(ShLeftAmt, DL, ShImmTy));
7480   }
7481 
7482   // Return the new loaded value.
7483   return Result;
7484 }
7485 
7486 SDValue DAGCombiner::visitSIGN_EXTEND_INREG(SDNode *N) {
7487   SDValue N0 = N->getOperand(0);
7488   SDValue N1 = N->getOperand(1);
7489   EVT VT = N->getValueType(0);
7490   EVT EVT = cast<VTSDNode>(N1)->getVT();
7491   unsigned VTBits = VT.getScalarSizeInBits();
7492   unsigned EVTBits = EVT.getScalarSizeInBits();
7493 
7494   if (N0.isUndef())
7495     return DAG.getUNDEF(VT);
7496 
7497   // fold (sext_in_reg c1) -> c1
7498   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7499     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, N0, N1);
7500 
7501   // If the input is already sign extended, just drop the extension.
7502   if (DAG.ComputeNumSignBits(N0) >= VTBits-EVTBits+1)
7503     return N0;
7504 
7505   // fold (sext_in_reg (sext_in_reg x, VT2), VT1) -> (sext_in_reg x, minVT) pt2
7506   if (N0.getOpcode() == ISD::SIGN_EXTEND_INREG &&
7507       EVT.bitsLT(cast<VTSDNode>(N0.getOperand(1))->getVT()))
7508     return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
7509                        N0.getOperand(0), N1);
7510 
7511   // fold (sext_in_reg (sext x)) -> (sext x)
7512   // fold (sext_in_reg (aext x)) -> (sext x)
7513   // if x is small enough.
7514   if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) {
7515     SDValue N00 = N0.getOperand(0);
7516     if (N00.getScalarValueSizeInBits() <= EVTBits &&
7517         (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT)))
7518       return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00, N1);
7519   }
7520 
7521   // fold (sext_in_reg (*_extend_vector_inreg x)) -> (sext_vector_in_reg x)
7522   if ((N0.getOpcode() == ISD::ANY_EXTEND_VECTOR_INREG ||
7523        N0.getOpcode() == ISD::SIGN_EXTEND_VECTOR_INREG ||
7524        N0.getOpcode() == ISD::ZERO_EXTEND_VECTOR_INREG) &&
7525       N0.getOperand(0).getScalarValueSizeInBits() == EVTBits) {
7526     if (!LegalOperations ||
7527         TLI.isOperationLegal(ISD::SIGN_EXTEND_VECTOR_INREG, VT))
7528       return DAG.getSignExtendVectorInReg(N0.getOperand(0), SDLoc(N), VT);
7529   }
7530 
7531   // fold (sext_in_reg (zext x)) -> (sext x)
7532   // iff we are extending the source sign bit.
7533   if (N0.getOpcode() == ISD::ZERO_EXTEND) {
7534     SDValue N00 = N0.getOperand(0);
7535     if (N00.getScalarValueSizeInBits() == EVTBits &&
7536         (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT)))
7537       return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00, N1);
7538   }
7539 
7540   // fold (sext_in_reg x) -> (zext_in_reg x) if the sign bit is known zero.
7541   if (DAG.MaskedValueIsZero(N0, APInt::getBitsSet(VTBits, EVTBits-1, EVTBits)))
7542     return DAG.getZeroExtendInReg(N0, SDLoc(N), EVT.getScalarType());
7543 
7544   // fold operands of sext_in_reg based on knowledge that the top bits are not
7545   // demanded.
7546   if (SimplifyDemandedBits(SDValue(N, 0)))
7547     return SDValue(N, 0);
7548 
7549   // fold (sext_in_reg (load x)) -> (smaller sextload x)
7550   // fold (sext_in_reg (srl (load x), c)) -> (smaller sextload (x+c/evtbits))
7551   if (SDValue NarrowLoad = ReduceLoadWidth(N))
7552     return NarrowLoad;
7553 
7554   // fold (sext_in_reg (srl X, 24), i8) -> (sra X, 24)
7555   // fold (sext_in_reg (srl X, 23), i8) -> (sra X, 23) iff possible.
7556   // We already fold "(sext_in_reg (srl X, 25), i8) -> srl X, 25" above.
7557   if (N0.getOpcode() == ISD::SRL) {
7558     if (ConstantSDNode *ShAmt = dyn_cast<ConstantSDNode>(N0.getOperand(1)))
7559       if (ShAmt->getZExtValue()+EVTBits <= VTBits) {
7560         // We can turn this into an SRA iff the input to the SRL is already sign
7561         // extended enough.
7562         unsigned InSignBits = DAG.ComputeNumSignBits(N0.getOperand(0));
7563         if (VTBits-(ShAmt->getZExtValue()+EVTBits) < InSignBits)
7564           return DAG.getNode(ISD::SRA, SDLoc(N), VT,
7565                              N0.getOperand(0), N0.getOperand(1));
7566       }
7567   }
7568 
7569   // fold (sext_inreg (extload x)) -> (sextload x)
7570   if (ISD::isEXTLoad(N0.getNode()) &&
7571       ISD::isUNINDEXEDLoad(N0.getNode()) &&
7572       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
7573       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
7574        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
7575     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7576     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
7577                                      LN0->getChain(),
7578                                      LN0->getBasePtr(), EVT,
7579                                      LN0->getMemOperand());
7580     CombineTo(N, ExtLoad);
7581     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
7582     AddToWorklist(ExtLoad.getNode());
7583     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7584   }
7585   // fold (sext_inreg (zextload x)) -> (sextload x) iff load has one use
7586   if (ISD::isZEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) &&
7587       N0.hasOneUse() &&
7588       EVT == cast<LoadSDNode>(N0)->getMemoryVT() &&
7589       ((!LegalOperations && !cast<LoadSDNode>(N0)->isVolatile()) ||
7590        TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) {
7591     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7592     SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT,
7593                                      LN0->getChain(),
7594                                      LN0->getBasePtr(), EVT,
7595                                      LN0->getMemOperand());
7596     CombineTo(N, ExtLoad);
7597     CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1));
7598     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
7599   }
7600 
7601   // Form (sext_inreg (bswap >> 16)) or (sext_inreg (rotl (bswap) 16))
7602   if (EVTBits <= 16 && N0.getOpcode() == ISD::OR) {
7603     if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0),
7604                                            N0.getOperand(1), false))
7605       return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT,
7606                          BSwap, N1);
7607   }
7608 
7609   return SDValue();
7610 }
7611 
7612 SDValue DAGCombiner::visitSIGN_EXTEND_VECTOR_INREG(SDNode *N) {
7613   SDValue N0 = N->getOperand(0);
7614   EVT VT = N->getValueType(0);
7615 
7616   if (N0.isUndef())
7617     return DAG.getUNDEF(VT);
7618 
7619   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
7620                                               LegalOperations))
7621     return SDValue(Res, 0);
7622 
7623   return SDValue();
7624 }
7625 
7626 SDValue DAGCombiner::visitZERO_EXTEND_VECTOR_INREG(SDNode *N) {
7627   SDValue N0 = N->getOperand(0);
7628   EVT VT = N->getValueType(0);
7629 
7630   if (N0.isUndef())
7631     return DAG.getUNDEF(VT);
7632 
7633   if (SDNode *Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes,
7634                                               LegalOperations))
7635     return SDValue(Res, 0);
7636 
7637   return SDValue();
7638 }
7639 
7640 SDValue DAGCombiner::visitTRUNCATE(SDNode *N) {
7641   SDValue N0 = N->getOperand(0);
7642   EVT VT = N->getValueType(0);
7643   bool isLE = DAG.getDataLayout().isLittleEndian();
7644 
7645   // noop truncate
7646   if (N0.getValueType() == N->getValueType(0))
7647     return N0;
7648   // fold (truncate c1) -> c1
7649   if (DAG.isConstantIntBuildVectorOrConstantInt(N0))
7650     return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0);
7651   // fold (truncate (truncate x)) -> (truncate x)
7652   if (N0.getOpcode() == ISD::TRUNCATE)
7653     return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
7654   // fold (truncate (ext x)) -> (ext x) or (truncate x) or x
7655   if (N0.getOpcode() == ISD::ZERO_EXTEND ||
7656       N0.getOpcode() == ISD::SIGN_EXTEND ||
7657       N0.getOpcode() == ISD::ANY_EXTEND) {
7658     // if the source is smaller than the dest, we still need an extend.
7659     if (N0.getOperand(0).getValueType().bitsLT(VT))
7660       return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0));
7661     // if the source is larger than the dest, than we just need the truncate.
7662     if (N0.getOperand(0).getValueType().bitsGT(VT))
7663       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0));
7664     // if the source and dest are the same type, we can drop both the extend
7665     // and the truncate.
7666     return N0.getOperand(0);
7667   }
7668 
7669   // If this is anyext(trunc), don't fold it, allow ourselves to be folded.
7670   if (N->hasOneUse() && (N->use_begin()->getOpcode() == ISD::ANY_EXTEND))
7671     return SDValue();
7672 
7673   // Fold extract-and-trunc into a narrow extract. For example:
7674   //   i64 x = EXTRACT_VECTOR_ELT(v2i64 val, i32 1)
7675   //   i32 y = TRUNCATE(i64 x)
7676   //        -- becomes --
7677   //   v16i8 b = BITCAST (v2i64 val)
7678   //   i8 x = EXTRACT_VECTOR_ELT(v16i8 b, i32 8)
7679   //
7680   // Note: We only run this optimization after type legalization (which often
7681   // creates this pattern) and before operation legalization after which
7682   // we need to be more careful about the vector instructions that we generate.
7683   if (N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
7684       LegalTypes && !LegalOperations && N0->hasOneUse() && VT != MVT::i1) {
7685 
7686     EVT VecTy = N0.getOperand(0).getValueType();
7687     EVT ExTy = N0.getValueType();
7688     EVT TrTy = N->getValueType(0);
7689 
7690     unsigned NumElem = VecTy.getVectorNumElements();
7691     unsigned SizeRatio = ExTy.getSizeInBits()/TrTy.getSizeInBits();
7692 
7693     EVT NVT = EVT::getVectorVT(*DAG.getContext(), TrTy, SizeRatio * NumElem);
7694     assert(NVT.getSizeInBits() == VecTy.getSizeInBits() && "Invalid Size");
7695 
7696     SDValue EltNo = N0->getOperand(1);
7697     if (isa<ConstantSDNode>(EltNo) && isTypeLegal(NVT)) {
7698       int Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
7699       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
7700       int Index = isLE ? (Elt*SizeRatio) : (Elt*SizeRatio + (SizeRatio-1));
7701 
7702       SDLoc DL(N);
7703       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, TrTy,
7704                          DAG.getBitcast(NVT, N0.getOperand(0)),
7705                          DAG.getConstant(Index, DL, IndexTy));
7706     }
7707   }
7708 
7709   // trunc (select c, a, b) -> select c, (trunc a), (trunc b)
7710   if (N0.getOpcode() == ISD::SELECT && N0.hasOneUse()) {
7711     EVT SrcVT = N0.getValueType();
7712     if ((!LegalOperations || TLI.isOperationLegal(ISD::SELECT, SrcVT)) &&
7713         TLI.isTruncateFree(SrcVT, VT)) {
7714       SDLoc SL(N0);
7715       SDValue Cond = N0.getOperand(0);
7716       SDValue TruncOp0 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1));
7717       SDValue TruncOp1 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(2));
7718       return DAG.getNode(ISD::SELECT, SDLoc(N), VT, Cond, TruncOp0, TruncOp1);
7719     }
7720   }
7721 
7722   // trunc (shl x, K) -> shl (trunc x), K => K < VT.getScalarSizeInBits()
7723   if (N0.getOpcode() == ISD::SHL && N0.hasOneUse() &&
7724       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::SHL, VT)) &&
7725       TLI.isTypeDesirableForOp(ISD::SHL, VT)) {
7726     if (const ConstantSDNode *CAmt = isConstOrConstSplat(N0.getOperand(1))) {
7727       uint64_t Amt = CAmt->getZExtValue();
7728       unsigned Size = VT.getScalarSizeInBits();
7729 
7730       if (Amt < Size) {
7731         SDLoc SL(N);
7732         EVT AmtVT = TLI.getShiftAmountTy(VT, DAG.getDataLayout());
7733 
7734         SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0));
7735         return DAG.getNode(ISD::SHL, SL, VT, Trunc,
7736                            DAG.getConstant(Amt, SL, AmtVT));
7737       }
7738     }
7739   }
7740 
7741   // Fold a series of buildvector, bitcast, and truncate if possible.
7742   // For example fold
7743   //   (2xi32 trunc (bitcast ((4xi32)buildvector x, x, y, y) 2xi64)) to
7744   //   (2xi32 (buildvector x, y)).
7745   if (Level == AfterLegalizeVectorOps && VT.isVector() &&
7746       N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
7747       N0.getOperand(0).getOpcode() == ISD::BUILD_VECTOR &&
7748       N0.getOperand(0).hasOneUse()) {
7749 
7750     SDValue BuildVect = N0.getOperand(0);
7751     EVT BuildVectEltTy = BuildVect.getValueType().getVectorElementType();
7752     EVT TruncVecEltTy = VT.getVectorElementType();
7753 
7754     // Check that the element types match.
7755     if (BuildVectEltTy == TruncVecEltTy) {
7756       // Now we only need to compute the offset of the truncated elements.
7757       unsigned BuildVecNumElts =  BuildVect.getNumOperands();
7758       unsigned TruncVecNumElts = VT.getVectorNumElements();
7759       unsigned TruncEltOffset = BuildVecNumElts / TruncVecNumElts;
7760 
7761       assert((BuildVecNumElts % TruncVecNumElts) == 0 &&
7762              "Invalid number of elements");
7763 
7764       SmallVector<SDValue, 8> Opnds;
7765       for (unsigned i = 0, e = BuildVecNumElts; i != e; i += TruncEltOffset)
7766         Opnds.push_back(BuildVect.getOperand(i));
7767 
7768       return DAG.getBuildVector(VT, SDLoc(N), Opnds);
7769     }
7770   }
7771 
7772   // See if we can simplify the input to this truncate through knowledge that
7773   // only the low bits are being used.
7774   // For example "trunc (or (shl x, 8), y)" // -> trunc y
7775   // Currently we only perform this optimization on scalars because vectors
7776   // may have different active low bits.
7777   if (!VT.isVector()) {
7778     if (SDValue Shorter =
7779             GetDemandedBits(N0, APInt::getLowBitsSet(N0.getValueSizeInBits(),
7780                                                      VT.getSizeInBits())))
7781       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Shorter);
7782   }
7783 
7784   // fold (truncate (load x)) -> (smaller load x)
7785   // fold (truncate (srl (load x), c)) -> (smaller load (x+c/evtbits))
7786   if (!LegalTypes || TLI.isTypeDesirableForOp(N0.getOpcode(), VT)) {
7787     if (SDValue Reduced = ReduceLoadWidth(N))
7788       return Reduced;
7789 
7790     // Handle the case where the load remains an extending load even
7791     // after truncation.
7792     if (N0.hasOneUse() && ISD::isUNINDEXEDLoad(N0.getNode())) {
7793       LoadSDNode *LN0 = cast<LoadSDNode>(N0);
7794       if (!LN0->isVolatile() &&
7795           LN0->getMemoryVT().getStoreSizeInBits() < VT.getSizeInBits()) {
7796         SDValue NewLoad = DAG.getExtLoad(LN0->getExtensionType(), SDLoc(LN0),
7797                                          VT, LN0->getChain(), LN0->getBasePtr(),
7798                                          LN0->getMemoryVT(),
7799                                          LN0->getMemOperand());
7800         DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLoad.getValue(1));
7801         return NewLoad;
7802       }
7803     }
7804   }
7805 
7806   // fold (trunc (concat ... x ...)) -> (concat ..., (trunc x), ...)),
7807   // where ... are all 'undef'.
7808   if (N0.getOpcode() == ISD::CONCAT_VECTORS && !LegalTypes) {
7809     SmallVector<EVT, 8> VTs;
7810     SDValue V;
7811     unsigned Idx = 0;
7812     unsigned NumDefs = 0;
7813 
7814     for (unsigned i = 0, e = N0.getNumOperands(); i != e; ++i) {
7815       SDValue X = N0.getOperand(i);
7816       if (!X.isUndef()) {
7817         V = X;
7818         Idx = i;
7819         NumDefs++;
7820       }
7821       // Stop if more than one members are non-undef.
7822       if (NumDefs > 1)
7823         break;
7824       VTs.push_back(EVT::getVectorVT(*DAG.getContext(),
7825                                      VT.getVectorElementType(),
7826                                      X.getValueType().getVectorNumElements()));
7827     }
7828 
7829     if (NumDefs == 0)
7830       return DAG.getUNDEF(VT);
7831 
7832     if (NumDefs == 1) {
7833       assert(V.getNode() && "The single defined operand is empty!");
7834       SmallVector<SDValue, 8> Opnds;
7835       for (unsigned i = 0, e = VTs.size(); i != e; ++i) {
7836         if (i != Idx) {
7837           Opnds.push_back(DAG.getUNDEF(VTs[i]));
7838           continue;
7839         }
7840         SDValue NV = DAG.getNode(ISD::TRUNCATE, SDLoc(V), VTs[i], V);
7841         AddToWorklist(NV.getNode());
7842         Opnds.push_back(NV);
7843       }
7844       return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Opnds);
7845     }
7846   }
7847 
7848   // Fold truncate of a bitcast of a vector to an extract of the low vector
7849   // element.
7850   //
7851   // e.g. trunc (i64 (bitcast v2i32:x)) -> extract_vector_elt v2i32:x, 0
7852   if (N0.getOpcode() == ISD::BITCAST && !VT.isVector()) {
7853     SDValue VecSrc = N0.getOperand(0);
7854     EVT SrcVT = VecSrc.getValueType();
7855     if (SrcVT.isVector() && SrcVT.getScalarType() == VT &&
7856         (!LegalOperations ||
7857          TLI.isOperationLegalOrCustom(ISD::EXTRACT_VECTOR_ELT, SrcVT))) {
7858       SDLoc SL(N);
7859 
7860       EVT IdxVT = TLI.getVectorIdxTy(DAG.getDataLayout());
7861       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, VT,
7862                          VecSrc, DAG.getConstant(0, SL, IdxVT));
7863     }
7864   }
7865 
7866   // Simplify the operands using demanded-bits information.
7867   if (!VT.isVector() &&
7868       SimplifyDemandedBits(SDValue(N, 0)))
7869     return SDValue(N, 0);
7870 
7871   // (trunc adde(X, Y, Carry)) -> (adde trunc(X), trunc(Y), Carry)
7872   // When the adde's carry is not used.
7873   if (N0.getOpcode() == ISD::ADDE && N0.hasOneUse() &&
7874       !N0.getNode()->hasAnyUseOfValue(1) &&
7875       (!LegalOperations || TLI.isOperationLegal(ISD::ADDE, VT))) {
7876     SDLoc SL(N);
7877     auto X = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0));
7878     auto Y = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1));
7879     return DAG.getNode(ISD::ADDE, SL, DAG.getVTList(VT, MVT::Glue),
7880                        X, Y, N0.getOperand(2));
7881   }
7882 
7883   return SDValue();
7884 }
7885 
7886 static SDNode *getBuildPairElt(SDNode *N, unsigned i) {
7887   SDValue Elt = N->getOperand(i);
7888   if (Elt.getOpcode() != ISD::MERGE_VALUES)
7889     return Elt.getNode();
7890   return Elt.getOperand(Elt.getResNo()).getNode();
7891 }
7892 
7893 /// build_pair (load, load) -> load
7894 /// if load locations are consecutive.
7895 SDValue DAGCombiner::CombineConsecutiveLoads(SDNode *N, EVT VT) {
7896   assert(N->getOpcode() == ISD::BUILD_PAIR);
7897 
7898   LoadSDNode *LD1 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 0));
7899   LoadSDNode *LD2 = dyn_cast<LoadSDNode>(getBuildPairElt(N, 1));
7900   if (!LD1 || !LD2 || !ISD::isNON_EXTLoad(LD1) || !LD1->hasOneUse() ||
7901       LD1->getAddressSpace() != LD2->getAddressSpace())
7902     return SDValue();
7903   EVT LD1VT = LD1->getValueType(0);
7904   unsigned LD1Bytes = LD1VT.getSizeInBits() / 8;
7905   if (ISD::isNON_EXTLoad(LD2) && LD2->hasOneUse() &&
7906       DAG.areNonVolatileConsecutiveLoads(LD2, LD1, LD1Bytes, 1)) {
7907     unsigned Align = LD1->getAlignment();
7908     unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
7909         VT.getTypeForEVT(*DAG.getContext()));
7910 
7911     if (NewAlign <= Align &&
7912         (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT)))
7913       return DAG.getLoad(VT, SDLoc(N), LD1->getChain(), LD1->getBasePtr(),
7914                          LD1->getPointerInfo(), Align);
7915   }
7916 
7917   return SDValue();
7918 }
7919 
7920 static unsigned getPPCf128HiElementSelector(const SelectionDAG &DAG) {
7921   // On little-endian machines, bitcasting from ppcf128 to i128 does swap the Hi
7922   // and Lo parts; on big-endian machines it doesn't.
7923   return DAG.getDataLayout().isBigEndian() ? 1 : 0;
7924 }
7925 
7926 static SDValue foldBitcastedFPLogic(SDNode *N, SelectionDAG &DAG,
7927                                     const TargetLowering &TLI) {
7928   // If this is not a bitcast to an FP type or if the target doesn't have
7929   // IEEE754-compliant FP logic, we're done.
7930   EVT VT = N->getValueType(0);
7931   if (!VT.isFloatingPoint() || !TLI.hasBitPreservingFPLogic(VT))
7932     return SDValue();
7933 
7934   // TODO: Use splat values for the constant-checking below and remove this
7935   // restriction.
7936   SDValue N0 = N->getOperand(0);
7937   EVT SourceVT = N0.getValueType();
7938   if (SourceVT.isVector())
7939     return SDValue();
7940 
7941   unsigned FPOpcode;
7942   APInt SignMask;
7943   switch (N0.getOpcode()) {
7944   case ISD::AND:
7945     FPOpcode = ISD::FABS;
7946     SignMask = ~APInt::getSignBit(SourceVT.getSizeInBits());
7947     break;
7948   case ISD::XOR:
7949     FPOpcode = ISD::FNEG;
7950     SignMask = APInt::getSignBit(SourceVT.getSizeInBits());
7951     break;
7952   // TODO: ISD::OR --> ISD::FNABS?
7953   default:
7954     return SDValue();
7955   }
7956 
7957   // Fold (bitcast int (and (bitcast fp X to int), 0x7fff...) to fp) -> fabs X
7958   // Fold (bitcast int (xor (bitcast fp X to int), 0x8000...) to fp) -> fneg X
7959   SDValue LogicOp0 = N0.getOperand(0);
7960   ConstantSDNode *LogicOp1 = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7961   if (LogicOp1 && LogicOp1->getAPIntValue() == SignMask &&
7962       LogicOp0.getOpcode() == ISD::BITCAST &&
7963       LogicOp0->getOperand(0).getValueType() == VT)
7964     return DAG.getNode(FPOpcode, SDLoc(N), VT, LogicOp0->getOperand(0));
7965 
7966   return SDValue();
7967 }
7968 
7969 SDValue DAGCombiner::visitBITCAST(SDNode *N) {
7970   SDValue N0 = N->getOperand(0);
7971   EVT VT = N->getValueType(0);
7972 
7973   // If the input is a BUILD_VECTOR with all constant elements, fold this now.
7974   // Only do this before legalize, since afterward the target may be depending
7975   // on the bitconvert.
7976   // First check to see if this is all constant.
7977   if (!LegalTypes &&
7978       N0.getOpcode() == ISD::BUILD_VECTOR && N0.getNode()->hasOneUse() &&
7979       VT.isVector()) {
7980     bool isSimple = cast<BuildVectorSDNode>(N0)->isConstant();
7981 
7982     EVT DestEltVT = N->getValueType(0).getVectorElementType();
7983     assert(!DestEltVT.isVector() &&
7984            "Element type of vector ValueType must not be vector!");
7985     if (isSimple)
7986       return ConstantFoldBITCASTofBUILD_VECTOR(N0.getNode(), DestEltVT);
7987   }
7988 
7989   // If the input is a constant, let getNode fold it.
7990   if (isa<ConstantSDNode>(N0) || isa<ConstantFPSDNode>(N0)) {
7991     // If we can't allow illegal operations, we need to check that this is just
7992     // a fp -> int or int -> conversion and that the resulting operation will
7993     // be legal.
7994     if (!LegalOperations ||
7995         (isa<ConstantSDNode>(N0) && VT.isFloatingPoint() && !VT.isVector() &&
7996          TLI.isOperationLegal(ISD::ConstantFP, VT)) ||
7997         (isa<ConstantFPSDNode>(N0) && VT.isInteger() && !VT.isVector() &&
7998          TLI.isOperationLegal(ISD::Constant, VT)))
7999       return DAG.getBitcast(VT, N0);
8000   }
8001 
8002   // (conv (conv x, t1), t2) -> (conv x, t2)
8003   if (N0.getOpcode() == ISD::BITCAST)
8004     return DAG.getBitcast(VT, N0.getOperand(0));
8005 
8006   // fold (conv (load x)) -> (load (conv*)x)
8007   // If the resultant load doesn't need a higher alignment than the original!
8008   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
8009       // Do not change the width of a volatile load.
8010       !cast<LoadSDNode>(N0)->isVolatile() &&
8011       // Do not remove the cast if the types differ in endian layout.
8012       TLI.hasBigEndianPartOrdering(N0.getValueType(), DAG.getDataLayout()) ==
8013           TLI.hasBigEndianPartOrdering(VT, DAG.getDataLayout()) &&
8014       (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT)) &&
8015       TLI.isLoadBitCastBeneficial(N0.getValueType(), VT)) {
8016     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
8017     unsigned OrigAlign = LN0->getAlignment();
8018 
8019     bool Fast = false;
8020     if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT,
8021                                LN0->getAddressSpace(), OrigAlign, &Fast) &&
8022         Fast) {
8023       SDValue Load =
8024           DAG.getLoad(VT, SDLoc(N), LN0->getChain(), LN0->getBasePtr(),
8025                       LN0->getPointerInfo(), OrigAlign,
8026                       LN0->getMemOperand()->getFlags(), LN0->getAAInfo());
8027       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
8028       return Load;
8029     }
8030   }
8031 
8032   if (SDValue V = foldBitcastedFPLogic(N, DAG, TLI))
8033     return V;
8034 
8035   // fold (bitconvert (fneg x)) -> (xor (bitconvert x), signbit)
8036   // fold (bitconvert (fabs x)) -> (and (bitconvert x), (not signbit))
8037   //
8038   // For ppc_fp128:
8039   // fold (bitcast (fneg x)) ->
8040   //     flipbit = signbit
8041   //     (xor (bitcast x) (build_pair flipbit, flipbit))
8042   //
8043   // fold (bitcast (fabs x)) ->
8044   //     flipbit = (and (extract_element (bitcast x), 0), signbit)
8045   //     (xor (bitcast x) (build_pair flipbit, flipbit))
8046   // This often reduces constant pool loads.
8047   if (((N0.getOpcode() == ISD::FNEG && !TLI.isFNegFree(N0.getValueType())) ||
8048        (N0.getOpcode() == ISD::FABS && !TLI.isFAbsFree(N0.getValueType()))) &&
8049       N0.getNode()->hasOneUse() && VT.isInteger() &&
8050       !VT.isVector() && !N0.getValueType().isVector()) {
8051     SDValue NewConv = DAG.getBitcast(VT, N0.getOperand(0));
8052     AddToWorklist(NewConv.getNode());
8053 
8054     SDLoc DL(N);
8055     if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
8056       assert(VT.getSizeInBits() == 128);
8057       SDValue SignBit = DAG.getConstant(
8058           APInt::getSignBit(VT.getSizeInBits() / 2), SDLoc(N0), MVT::i64);
8059       SDValue FlipBit;
8060       if (N0.getOpcode() == ISD::FNEG) {
8061         FlipBit = SignBit;
8062         AddToWorklist(FlipBit.getNode());
8063       } else {
8064         assert(N0.getOpcode() == ISD::FABS);
8065         SDValue Hi =
8066             DAG.getNode(ISD::EXTRACT_ELEMENT, SDLoc(NewConv), MVT::i64, NewConv,
8067                         DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
8068                                               SDLoc(NewConv)));
8069         AddToWorklist(Hi.getNode());
8070         FlipBit = DAG.getNode(ISD::AND, SDLoc(N0), MVT::i64, Hi, SignBit);
8071         AddToWorklist(FlipBit.getNode());
8072       }
8073       SDValue FlipBits =
8074           DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
8075       AddToWorklist(FlipBits.getNode());
8076       return DAG.getNode(ISD::XOR, DL, VT, NewConv, FlipBits);
8077     }
8078     APInt SignBit = APInt::getSignBit(VT.getSizeInBits());
8079     if (N0.getOpcode() == ISD::FNEG)
8080       return DAG.getNode(ISD::XOR, DL, VT,
8081                          NewConv, DAG.getConstant(SignBit, DL, VT));
8082     assert(N0.getOpcode() == ISD::FABS);
8083     return DAG.getNode(ISD::AND, DL, VT,
8084                        NewConv, DAG.getConstant(~SignBit, DL, VT));
8085   }
8086 
8087   // fold (bitconvert (fcopysign cst, x)) ->
8088   //         (or (and (bitconvert x), sign), (and cst, (not sign)))
8089   // Note that we don't handle (copysign x, cst) because this can always be
8090   // folded to an fneg or fabs.
8091   //
8092   // For ppc_fp128:
8093   // fold (bitcast (fcopysign cst, x)) ->
8094   //     flipbit = (and (extract_element
8095   //                     (xor (bitcast cst), (bitcast x)), 0),
8096   //                    signbit)
8097   //     (xor (bitcast cst) (build_pair flipbit, flipbit))
8098   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse() &&
8099       isa<ConstantFPSDNode>(N0.getOperand(0)) &&
8100       VT.isInteger() && !VT.isVector()) {
8101     unsigned OrigXWidth = N0.getOperand(1).getValueSizeInBits();
8102     EVT IntXVT = EVT::getIntegerVT(*DAG.getContext(), OrigXWidth);
8103     if (isTypeLegal(IntXVT)) {
8104       SDValue X = DAG.getBitcast(IntXVT, N0.getOperand(1));
8105       AddToWorklist(X.getNode());
8106 
8107       // If X has a different width than the result/lhs, sext it or truncate it.
8108       unsigned VTWidth = VT.getSizeInBits();
8109       if (OrigXWidth < VTWidth) {
8110         X = DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, X);
8111         AddToWorklist(X.getNode());
8112       } else if (OrigXWidth > VTWidth) {
8113         // To get the sign bit in the right place, we have to shift it right
8114         // before truncating.
8115         SDLoc DL(X);
8116         X = DAG.getNode(ISD::SRL, DL,
8117                         X.getValueType(), X,
8118                         DAG.getConstant(OrigXWidth-VTWidth, DL,
8119                                         X.getValueType()));
8120         AddToWorklist(X.getNode());
8121         X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
8122         AddToWorklist(X.getNode());
8123       }
8124 
8125       if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) {
8126         APInt SignBit = APInt::getSignBit(VT.getSizeInBits() / 2);
8127         SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
8128         AddToWorklist(Cst.getNode());
8129         SDValue X = DAG.getBitcast(VT, N0.getOperand(1));
8130         AddToWorklist(X.getNode());
8131         SDValue XorResult = DAG.getNode(ISD::XOR, SDLoc(N0), VT, Cst, X);
8132         AddToWorklist(XorResult.getNode());
8133         SDValue XorResult64 = DAG.getNode(
8134             ISD::EXTRACT_ELEMENT, SDLoc(XorResult), MVT::i64, XorResult,
8135             DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG),
8136                                   SDLoc(XorResult)));
8137         AddToWorklist(XorResult64.getNode());
8138         SDValue FlipBit =
8139             DAG.getNode(ISD::AND, SDLoc(XorResult64), MVT::i64, XorResult64,
8140                         DAG.getConstant(SignBit, SDLoc(XorResult64), MVT::i64));
8141         AddToWorklist(FlipBit.getNode());
8142         SDValue FlipBits =
8143             DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit);
8144         AddToWorklist(FlipBits.getNode());
8145         return DAG.getNode(ISD::XOR, SDLoc(N), VT, Cst, FlipBits);
8146       }
8147       APInt SignBit = APInt::getSignBit(VT.getSizeInBits());
8148       X = DAG.getNode(ISD::AND, SDLoc(X), VT,
8149                       X, DAG.getConstant(SignBit, SDLoc(X), VT));
8150       AddToWorklist(X.getNode());
8151 
8152       SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0));
8153       Cst = DAG.getNode(ISD::AND, SDLoc(Cst), VT,
8154                         Cst, DAG.getConstant(~SignBit, SDLoc(Cst), VT));
8155       AddToWorklist(Cst.getNode());
8156 
8157       return DAG.getNode(ISD::OR, SDLoc(N), VT, X, Cst);
8158     }
8159   }
8160 
8161   // bitconvert(build_pair(ld, ld)) -> ld iff load locations are consecutive.
8162   if (N0.getOpcode() == ISD::BUILD_PAIR)
8163     if (SDValue CombineLD = CombineConsecutiveLoads(N0.getNode(), VT))
8164       return CombineLD;
8165 
8166   // Remove double bitcasts from shuffles - this is often a legacy of
8167   // XformToShuffleWithZero being used to combine bitmaskings (of
8168   // float vectors bitcast to integer vectors) into shuffles.
8169   // bitcast(shuffle(bitcast(s0),bitcast(s1))) -> shuffle(s0,s1)
8170   if (Level < AfterLegalizeDAG && TLI.isTypeLegal(VT) && VT.isVector() &&
8171       N0->getOpcode() == ISD::VECTOR_SHUFFLE &&
8172       VT.getVectorNumElements() >= N0.getValueType().getVectorNumElements() &&
8173       !(VT.getVectorNumElements() % N0.getValueType().getVectorNumElements())) {
8174     ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N0);
8175 
8176     // If operands are a bitcast, peek through if it casts the original VT.
8177     // If operands are a constant, just bitcast back to original VT.
8178     auto PeekThroughBitcast = [&](SDValue Op) {
8179       if (Op.getOpcode() == ISD::BITCAST &&
8180           Op.getOperand(0).getValueType() == VT)
8181         return SDValue(Op.getOperand(0));
8182       if (ISD::isBuildVectorOfConstantSDNodes(Op.getNode()) ||
8183           ISD::isBuildVectorOfConstantFPSDNodes(Op.getNode()))
8184         return DAG.getBitcast(VT, Op);
8185       return SDValue();
8186     };
8187 
8188     SDValue SV0 = PeekThroughBitcast(N0->getOperand(0));
8189     SDValue SV1 = PeekThroughBitcast(N0->getOperand(1));
8190     if (!(SV0 && SV1))
8191       return SDValue();
8192 
8193     int MaskScale =
8194         VT.getVectorNumElements() / N0.getValueType().getVectorNumElements();
8195     SmallVector<int, 8> NewMask;
8196     for (int M : SVN->getMask())
8197       for (int i = 0; i != MaskScale; ++i)
8198         NewMask.push_back(M < 0 ? -1 : M * MaskScale + i);
8199 
8200     bool LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
8201     if (!LegalMask) {
8202       std::swap(SV0, SV1);
8203       ShuffleVectorSDNode::commuteMask(NewMask);
8204       LegalMask = TLI.isShuffleMaskLegal(NewMask, VT);
8205     }
8206 
8207     if (LegalMask)
8208       return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, NewMask);
8209   }
8210 
8211   return SDValue();
8212 }
8213 
8214 SDValue DAGCombiner::visitBUILD_PAIR(SDNode *N) {
8215   EVT VT = N->getValueType(0);
8216   return CombineConsecutiveLoads(N, VT);
8217 }
8218 
8219 /// We know that BV is a build_vector node with Constant, ConstantFP or Undef
8220 /// operands. DstEltVT indicates the destination element value type.
8221 SDValue DAGCombiner::
8222 ConstantFoldBITCASTofBUILD_VECTOR(SDNode *BV, EVT DstEltVT) {
8223   EVT SrcEltVT = BV->getValueType(0).getVectorElementType();
8224 
8225   // If this is already the right type, we're done.
8226   if (SrcEltVT == DstEltVT) return SDValue(BV, 0);
8227 
8228   unsigned SrcBitSize = SrcEltVT.getSizeInBits();
8229   unsigned DstBitSize = DstEltVT.getSizeInBits();
8230 
8231   // If this is a conversion of N elements of one type to N elements of another
8232   // type, convert each element.  This handles FP<->INT cases.
8233   if (SrcBitSize == DstBitSize) {
8234     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
8235                               BV->getValueType(0).getVectorNumElements());
8236 
8237     // Due to the FP element handling below calling this routine recursively,
8238     // we can end up with a scalar-to-vector node here.
8239     if (BV->getOpcode() == ISD::SCALAR_TO_VECTOR)
8240       return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(BV), VT,
8241                          DAG.getBitcast(DstEltVT, BV->getOperand(0)));
8242 
8243     SmallVector<SDValue, 8> Ops;
8244     for (SDValue Op : BV->op_values()) {
8245       // If the vector element type is not legal, the BUILD_VECTOR operands
8246       // are promoted and implicitly truncated.  Make that explicit here.
8247       if (Op.getValueType() != SrcEltVT)
8248         Op = DAG.getNode(ISD::TRUNCATE, SDLoc(BV), SrcEltVT, Op);
8249       Ops.push_back(DAG.getBitcast(DstEltVT, Op));
8250       AddToWorklist(Ops.back().getNode());
8251     }
8252     return DAG.getBuildVector(VT, SDLoc(BV), Ops);
8253   }
8254 
8255   // Otherwise, we're growing or shrinking the elements.  To avoid having to
8256   // handle annoying details of growing/shrinking FP values, we convert them to
8257   // int first.
8258   if (SrcEltVT.isFloatingPoint()) {
8259     // Convert the input float vector to a int vector where the elements are the
8260     // same sizes.
8261     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), SrcEltVT.getSizeInBits());
8262     BV = ConstantFoldBITCASTofBUILD_VECTOR(BV, IntVT).getNode();
8263     SrcEltVT = IntVT;
8264   }
8265 
8266   // Now we know the input is an integer vector.  If the output is a FP type,
8267   // convert to integer first, then to FP of the right size.
8268   if (DstEltVT.isFloatingPoint()) {
8269     EVT TmpVT = EVT::getIntegerVT(*DAG.getContext(), DstEltVT.getSizeInBits());
8270     SDNode *Tmp = ConstantFoldBITCASTofBUILD_VECTOR(BV, TmpVT).getNode();
8271 
8272     // Next, convert to FP elements of the same size.
8273     return ConstantFoldBITCASTofBUILD_VECTOR(Tmp, DstEltVT);
8274   }
8275 
8276   SDLoc DL(BV);
8277 
8278   // Okay, we know the src/dst types are both integers of differing types.
8279   // Handling growing first.
8280   assert(SrcEltVT.isInteger() && DstEltVT.isInteger());
8281   if (SrcBitSize < DstBitSize) {
8282     unsigned NumInputsPerOutput = DstBitSize/SrcBitSize;
8283 
8284     SmallVector<SDValue, 8> Ops;
8285     for (unsigned i = 0, e = BV->getNumOperands(); i != e;
8286          i += NumInputsPerOutput) {
8287       bool isLE = DAG.getDataLayout().isLittleEndian();
8288       APInt NewBits = APInt(DstBitSize, 0);
8289       bool EltIsUndef = true;
8290       for (unsigned j = 0; j != NumInputsPerOutput; ++j) {
8291         // Shift the previously computed bits over.
8292         NewBits <<= SrcBitSize;
8293         SDValue Op = BV->getOperand(i+ (isLE ? (NumInputsPerOutput-j-1) : j));
8294         if (Op.isUndef()) continue;
8295         EltIsUndef = false;
8296 
8297         NewBits |= cast<ConstantSDNode>(Op)->getAPIntValue().
8298                    zextOrTrunc(SrcBitSize).zext(DstBitSize);
8299       }
8300 
8301       if (EltIsUndef)
8302         Ops.push_back(DAG.getUNDEF(DstEltVT));
8303       else
8304         Ops.push_back(DAG.getConstant(NewBits, DL, DstEltVT));
8305     }
8306 
8307     EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT, Ops.size());
8308     return DAG.getBuildVector(VT, DL, Ops);
8309   }
8310 
8311   // Finally, this must be the case where we are shrinking elements: each input
8312   // turns into multiple outputs.
8313   unsigned NumOutputsPerInput = SrcBitSize/DstBitSize;
8314   EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT,
8315                             NumOutputsPerInput*BV->getNumOperands());
8316   SmallVector<SDValue, 8> Ops;
8317 
8318   for (const SDValue &Op : BV->op_values()) {
8319     if (Op.isUndef()) {
8320       Ops.append(NumOutputsPerInput, DAG.getUNDEF(DstEltVT));
8321       continue;
8322     }
8323 
8324     APInt OpVal = cast<ConstantSDNode>(Op)->
8325                   getAPIntValue().zextOrTrunc(SrcBitSize);
8326 
8327     for (unsigned j = 0; j != NumOutputsPerInput; ++j) {
8328       APInt ThisVal = OpVal.trunc(DstBitSize);
8329       Ops.push_back(DAG.getConstant(ThisVal, DL, DstEltVT));
8330       OpVal = OpVal.lshr(DstBitSize);
8331     }
8332 
8333     // For big endian targets, swap the order of the pieces of each element.
8334     if (DAG.getDataLayout().isBigEndian())
8335       std::reverse(Ops.end()-NumOutputsPerInput, Ops.end());
8336   }
8337 
8338   return DAG.getBuildVector(VT, DL, Ops);
8339 }
8340 
8341 /// Try to perform FMA combining on a given FADD node.
8342 SDValue DAGCombiner::visitFADDForFMACombine(SDNode *N) {
8343   SDValue N0 = N->getOperand(0);
8344   SDValue N1 = N->getOperand(1);
8345   EVT VT = N->getValueType(0);
8346   SDLoc SL(N);
8347 
8348   const TargetOptions &Options = DAG.getTarget().Options;
8349   bool AllowFusion =
8350       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
8351 
8352   // Floating-point multiply-add with intermediate rounding.
8353   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
8354 
8355   // Floating-point multiply-add without intermediate rounding.
8356   bool HasFMA =
8357       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
8358       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
8359 
8360   // No valid opcode, do not combine.
8361   if (!HasFMAD && !HasFMA)
8362     return SDValue();
8363 
8364   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
8365   ;
8366   if (AllowFusion && STI && STI->generateFMAsInMachineCombiner(OptLevel))
8367     return SDValue();
8368 
8369   // Always prefer FMAD to FMA for precision.
8370   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
8371   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
8372   bool LookThroughFPExt = TLI.isFPExtFree(VT);
8373 
8374   // If we have two choices trying to fold (fadd (fmul u, v), (fmul x, y)),
8375   // prefer to fold the multiply with fewer uses.
8376   if (Aggressive && N0.getOpcode() == ISD::FMUL &&
8377       N1.getOpcode() == ISD::FMUL) {
8378     if (N0.getNode()->use_size() > N1.getNode()->use_size())
8379       std::swap(N0, N1);
8380   }
8381 
8382   // fold (fadd (fmul x, y), z) -> (fma x, y, z)
8383   if (N0.getOpcode() == ISD::FMUL &&
8384       (Aggressive || N0->hasOneUse())) {
8385     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8386                        N0.getOperand(0), N0.getOperand(1), N1);
8387   }
8388 
8389   // fold (fadd x, (fmul y, z)) -> (fma y, z, x)
8390   // Note: Commutes FADD operands.
8391   if (N1.getOpcode() == ISD::FMUL &&
8392       (Aggressive || N1->hasOneUse())) {
8393     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8394                        N1.getOperand(0), N1.getOperand(1), N0);
8395   }
8396 
8397   // Look through FP_EXTEND nodes to do more combining.
8398   if (AllowFusion && LookThroughFPExt) {
8399     // fold (fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext y), z)
8400     if (N0.getOpcode() == ISD::FP_EXTEND) {
8401       SDValue N00 = N0.getOperand(0);
8402       if (N00.getOpcode() == ISD::FMUL)
8403         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8404                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8405                                        N00.getOperand(0)),
8406                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8407                                        N00.getOperand(1)), N1);
8408     }
8409 
8410     // fold (fadd x, (fpext (fmul y, z))) -> (fma (fpext y), (fpext z), x)
8411     // Note: Commutes FADD operands.
8412     if (N1.getOpcode() == ISD::FP_EXTEND) {
8413       SDValue N10 = N1.getOperand(0);
8414       if (N10.getOpcode() == ISD::FMUL)
8415         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8416                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8417                                        N10.getOperand(0)),
8418                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8419                                        N10.getOperand(1)), N0);
8420     }
8421   }
8422 
8423   // More folding opportunities when target permits.
8424   if (Aggressive) {
8425     // fold (fadd (fma x, y, (fmul u, v)), z) -> (fma x, y (fma u, v, z))
8426     // FIXME: The UnsafeAlgebra flag should be propagated to FMA/FMAD, but FMF
8427     // are currently only supported on binary nodes.
8428     if (Options.UnsafeFPMath &&
8429         N0.getOpcode() == PreferredFusedOpcode &&
8430         N0.getOperand(2).getOpcode() == ISD::FMUL &&
8431         N0->hasOneUse() && N0.getOperand(2)->hasOneUse()) {
8432       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8433                          N0.getOperand(0), N0.getOperand(1),
8434                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8435                                      N0.getOperand(2).getOperand(0),
8436                                      N0.getOperand(2).getOperand(1),
8437                                      N1));
8438     }
8439 
8440     // fold (fadd x, (fma y, z, (fmul u, v)) -> (fma y, z (fma u, v, x))
8441     // FIXME: The UnsafeAlgebra flag should be propagated to FMA/FMAD, but FMF
8442     // are currently only supported on binary nodes.
8443     if (Options.UnsafeFPMath &&
8444         N1->getOpcode() == PreferredFusedOpcode &&
8445         N1.getOperand(2).getOpcode() == ISD::FMUL &&
8446         N1->hasOneUse() && N1.getOperand(2)->hasOneUse()) {
8447       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8448                          N1.getOperand(0), N1.getOperand(1),
8449                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8450                                      N1.getOperand(2).getOperand(0),
8451                                      N1.getOperand(2).getOperand(1),
8452                                      N0));
8453     }
8454 
8455     if (AllowFusion && LookThroughFPExt) {
8456       // fold (fadd (fma x, y, (fpext (fmul u, v))), z)
8457       //   -> (fma x, y, (fma (fpext u), (fpext v), z))
8458       auto FoldFAddFMAFPExtFMul = [&] (
8459           SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z) {
8460         return DAG.getNode(PreferredFusedOpcode, SL, VT, X, Y,
8461                            DAG.getNode(PreferredFusedOpcode, SL, VT,
8462                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
8463                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
8464                                        Z));
8465       };
8466       if (N0.getOpcode() == PreferredFusedOpcode) {
8467         SDValue N02 = N0.getOperand(2);
8468         if (N02.getOpcode() == ISD::FP_EXTEND) {
8469           SDValue N020 = N02.getOperand(0);
8470           if (N020.getOpcode() == ISD::FMUL)
8471             return FoldFAddFMAFPExtFMul(N0.getOperand(0), N0.getOperand(1),
8472                                         N020.getOperand(0), N020.getOperand(1),
8473                                         N1);
8474         }
8475       }
8476 
8477       // fold (fadd (fpext (fma x, y, (fmul u, v))), z)
8478       //   -> (fma (fpext x), (fpext y), (fma (fpext u), (fpext v), z))
8479       // FIXME: This turns two single-precision and one double-precision
8480       // operation into two double-precision operations, which might not be
8481       // interesting for all targets, especially GPUs.
8482       auto FoldFAddFPExtFMAFMul = [&] (
8483           SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z) {
8484         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8485                            DAG.getNode(ISD::FP_EXTEND, SL, VT, X),
8486                            DAG.getNode(ISD::FP_EXTEND, SL, VT, Y),
8487                            DAG.getNode(PreferredFusedOpcode, SL, VT,
8488                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, U),
8489                                        DAG.getNode(ISD::FP_EXTEND, SL, VT, V),
8490                                        Z));
8491       };
8492       if (N0.getOpcode() == ISD::FP_EXTEND) {
8493         SDValue N00 = N0.getOperand(0);
8494         if (N00.getOpcode() == PreferredFusedOpcode) {
8495           SDValue N002 = N00.getOperand(2);
8496           if (N002.getOpcode() == ISD::FMUL)
8497             return FoldFAddFPExtFMAFMul(N00.getOperand(0), N00.getOperand(1),
8498                                         N002.getOperand(0), N002.getOperand(1),
8499                                         N1);
8500         }
8501       }
8502 
8503       // fold (fadd x, (fma y, z, (fpext (fmul u, v)))
8504       //   -> (fma y, z, (fma (fpext u), (fpext v), x))
8505       if (N1.getOpcode() == PreferredFusedOpcode) {
8506         SDValue N12 = N1.getOperand(2);
8507         if (N12.getOpcode() == ISD::FP_EXTEND) {
8508           SDValue N120 = N12.getOperand(0);
8509           if (N120.getOpcode() == ISD::FMUL)
8510             return FoldFAddFMAFPExtFMul(N1.getOperand(0), N1.getOperand(1),
8511                                         N120.getOperand(0), N120.getOperand(1),
8512                                         N0);
8513         }
8514       }
8515 
8516       // fold (fadd x, (fpext (fma y, z, (fmul u, v)))
8517       //   -> (fma (fpext y), (fpext z), (fma (fpext u), (fpext v), x))
8518       // FIXME: This turns two single-precision and one double-precision
8519       // operation into two double-precision operations, which might not be
8520       // interesting for all targets, especially GPUs.
8521       if (N1.getOpcode() == ISD::FP_EXTEND) {
8522         SDValue N10 = N1.getOperand(0);
8523         if (N10.getOpcode() == PreferredFusedOpcode) {
8524           SDValue N102 = N10.getOperand(2);
8525           if (N102.getOpcode() == ISD::FMUL)
8526             return FoldFAddFPExtFMAFMul(N10.getOperand(0), N10.getOperand(1),
8527                                         N102.getOperand(0), N102.getOperand(1),
8528                                         N0);
8529         }
8530       }
8531     }
8532   }
8533 
8534   return SDValue();
8535 }
8536 
8537 /// Try to perform FMA combining on a given FSUB node.
8538 SDValue DAGCombiner::visitFSUBForFMACombine(SDNode *N) {
8539   SDValue N0 = N->getOperand(0);
8540   SDValue N1 = N->getOperand(1);
8541   EVT VT = N->getValueType(0);
8542   SDLoc SL(N);
8543 
8544   const TargetOptions &Options = DAG.getTarget().Options;
8545   bool AllowFusion =
8546       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath);
8547 
8548   // Floating-point multiply-add with intermediate rounding.
8549   bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
8550 
8551   // Floating-point multiply-add without intermediate rounding.
8552   bool HasFMA =
8553       AllowFusion && TLI.isFMAFasterThanFMulAndFAdd(VT) &&
8554       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
8555 
8556   // No valid opcode, do not combine.
8557   if (!HasFMAD && !HasFMA)
8558     return SDValue();
8559 
8560   const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo();
8561   if (AllowFusion && STI && STI->generateFMAsInMachineCombiner(OptLevel))
8562     return SDValue();
8563 
8564   // Always prefer FMAD to FMA for precision.
8565   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
8566   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
8567   bool LookThroughFPExt = TLI.isFPExtFree(VT);
8568 
8569   // fold (fsub (fmul x, y), z) -> (fma x, y, (fneg z))
8570   if (N0.getOpcode() == ISD::FMUL &&
8571       (Aggressive || N0->hasOneUse())) {
8572     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8573                        N0.getOperand(0), N0.getOperand(1),
8574                        DAG.getNode(ISD::FNEG, SL, VT, N1));
8575   }
8576 
8577   // fold (fsub x, (fmul y, z)) -> (fma (fneg y), z, x)
8578   // Note: Commutes FSUB operands.
8579   if (N1.getOpcode() == ISD::FMUL &&
8580       (Aggressive || N1->hasOneUse()))
8581     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8582                        DAG.getNode(ISD::FNEG, SL, VT,
8583                                    N1.getOperand(0)),
8584                        N1.getOperand(1), N0);
8585 
8586   // fold (fsub (fneg (fmul, x, y)), z) -> (fma (fneg x), y, (fneg z))
8587   if (N0.getOpcode() == ISD::FNEG &&
8588       N0.getOperand(0).getOpcode() == ISD::FMUL &&
8589       (Aggressive || (N0->hasOneUse() && N0.getOperand(0).hasOneUse()))) {
8590     SDValue N00 = N0.getOperand(0).getOperand(0);
8591     SDValue N01 = N0.getOperand(0).getOperand(1);
8592     return DAG.getNode(PreferredFusedOpcode, SL, VT,
8593                        DAG.getNode(ISD::FNEG, SL, VT, N00), N01,
8594                        DAG.getNode(ISD::FNEG, SL, VT, N1));
8595   }
8596 
8597   // Look through FP_EXTEND nodes to do more combining.
8598   if (AllowFusion && LookThroughFPExt) {
8599     // fold (fsub (fpext (fmul x, y)), z)
8600     //   -> (fma (fpext x), (fpext y), (fneg z))
8601     if (N0.getOpcode() == ISD::FP_EXTEND) {
8602       SDValue N00 = N0.getOperand(0);
8603       if (N00.getOpcode() == ISD::FMUL)
8604         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8605                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8606                                        N00.getOperand(0)),
8607                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8608                                        N00.getOperand(1)),
8609                            DAG.getNode(ISD::FNEG, SL, VT, N1));
8610     }
8611 
8612     // fold (fsub x, (fpext (fmul y, z)))
8613     //   -> (fma (fneg (fpext y)), (fpext z), x)
8614     // Note: Commutes FSUB operands.
8615     if (N1.getOpcode() == ISD::FP_EXTEND) {
8616       SDValue N10 = N1.getOperand(0);
8617       if (N10.getOpcode() == ISD::FMUL)
8618         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8619                            DAG.getNode(ISD::FNEG, SL, VT,
8620                                        DAG.getNode(ISD::FP_EXTEND, SL, VT,
8621                                                    N10.getOperand(0))),
8622                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8623                                        N10.getOperand(1)),
8624                            N0);
8625     }
8626 
8627     // fold (fsub (fpext (fneg (fmul, x, y))), z)
8628     //   -> (fneg (fma (fpext x), (fpext y), z))
8629     // Note: This could be removed with appropriate canonicalization of the
8630     // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
8631     // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
8632     // from implementing the canonicalization in visitFSUB.
8633     if (N0.getOpcode() == ISD::FP_EXTEND) {
8634       SDValue N00 = N0.getOperand(0);
8635       if (N00.getOpcode() == ISD::FNEG) {
8636         SDValue N000 = N00.getOperand(0);
8637         if (N000.getOpcode() == ISD::FMUL) {
8638           return DAG.getNode(ISD::FNEG, SL, VT,
8639                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8640                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8641                                                      N000.getOperand(0)),
8642                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8643                                                      N000.getOperand(1)),
8644                                          N1));
8645         }
8646       }
8647     }
8648 
8649     // fold (fsub (fneg (fpext (fmul, x, y))), z)
8650     //   -> (fneg (fma (fpext x)), (fpext y), z)
8651     // Note: This could be removed with appropriate canonicalization of the
8652     // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the
8653     // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent
8654     // from implementing the canonicalization in visitFSUB.
8655     if (N0.getOpcode() == ISD::FNEG) {
8656       SDValue N00 = N0.getOperand(0);
8657       if (N00.getOpcode() == ISD::FP_EXTEND) {
8658         SDValue N000 = N00.getOperand(0);
8659         if (N000.getOpcode() == ISD::FMUL) {
8660           return DAG.getNode(ISD::FNEG, SL, VT,
8661                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8662                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8663                                                      N000.getOperand(0)),
8664                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8665                                                      N000.getOperand(1)),
8666                                          N1));
8667         }
8668       }
8669     }
8670 
8671   }
8672 
8673   // More folding opportunities when target permits.
8674   if (Aggressive) {
8675     // fold (fsub (fma x, y, (fmul u, v)), z)
8676     //   -> (fma x, y (fma u, v, (fneg z)))
8677     // FIXME: The UnsafeAlgebra flag should be propagated to FMA/FMAD, but FMF
8678     // are currently only supported on binary nodes.
8679     if (Options.UnsafeFPMath &&
8680         N0.getOpcode() == PreferredFusedOpcode &&
8681         N0.getOperand(2).getOpcode() == ISD::FMUL &&
8682         N0->hasOneUse() && N0.getOperand(2)->hasOneUse()) {
8683       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8684                          N0.getOperand(0), N0.getOperand(1),
8685                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8686                                      N0.getOperand(2).getOperand(0),
8687                                      N0.getOperand(2).getOperand(1),
8688                                      DAG.getNode(ISD::FNEG, SL, VT,
8689                                                  N1)));
8690     }
8691 
8692     // fold (fsub x, (fma y, z, (fmul u, v)))
8693     //   -> (fma (fneg y), z, (fma (fneg u), v, x))
8694     // FIXME: The UnsafeAlgebra flag should be propagated to FMA/FMAD, but FMF
8695     // are currently only supported on binary nodes.
8696     if (Options.UnsafeFPMath &&
8697         N1.getOpcode() == PreferredFusedOpcode &&
8698         N1.getOperand(2).getOpcode() == ISD::FMUL) {
8699       SDValue N20 = N1.getOperand(2).getOperand(0);
8700       SDValue N21 = N1.getOperand(2).getOperand(1);
8701       return DAG.getNode(PreferredFusedOpcode, SL, VT,
8702                          DAG.getNode(ISD::FNEG, SL, VT,
8703                                      N1.getOperand(0)),
8704                          N1.getOperand(1),
8705                          DAG.getNode(PreferredFusedOpcode, SL, VT,
8706                                      DAG.getNode(ISD::FNEG, SL, VT, N20),
8707 
8708                                      N21, N0));
8709     }
8710 
8711     if (AllowFusion && LookThroughFPExt) {
8712       // fold (fsub (fma x, y, (fpext (fmul u, v))), z)
8713       //   -> (fma x, y (fma (fpext u), (fpext v), (fneg z)))
8714       if (N0.getOpcode() == PreferredFusedOpcode) {
8715         SDValue N02 = N0.getOperand(2);
8716         if (N02.getOpcode() == ISD::FP_EXTEND) {
8717           SDValue N020 = N02.getOperand(0);
8718           if (N020.getOpcode() == ISD::FMUL)
8719             return DAG.getNode(PreferredFusedOpcode, SL, VT,
8720                                N0.getOperand(0), N0.getOperand(1),
8721                                DAG.getNode(PreferredFusedOpcode, SL, VT,
8722                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8723                                                        N020.getOperand(0)),
8724                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8725                                                        N020.getOperand(1)),
8726                                            DAG.getNode(ISD::FNEG, SL, VT,
8727                                                        N1)));
8728         }
8729       }
8730 
8731       // fold (fsub (fpext (fma x, y, (fmul u, v))), z)
8732       //   -> (fma (fpext x), (fpext y),
8733       //           (fma (fpext u), (fpext v), (fneg z)))
8734       // FIXME: This turns two single-precision and one double-precision
8735       // operation into two double-precision operations, which might not be
8736       // interesting for all targets, especially GPUs.
8737       if (N0.getOpcode() == ISD::FP_EXTEND) {
8738         SDValue N00 = N0.getOperand(0);
8739         if (N00.getOpcode() == PreferredFusedOpcode) {
8740           SDValue N002 = N00.getOperand(2);
8741           if (N002.getOpcode() == ISD::FMUL)
8742             return DAG.getNode(PreferredFusedOpcode, SL, VT,
8743                                DAG.getNode(ISD::FP_EXTEND, SL, VT,
8744                                            N00.getOperand(0)),
8745                                DAG.getNode(ISD::FP_EXTEND, SL, VT,
8746                                            N00.getOperand(1)),
8747                                DAG.getNode(PreferredFusedOpcode, SL, VT,
8748                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8749                                                        N002.getOperand(0)),
8750                                            DAG.getNode(ISD::FP_EXTEND, SL, VT,
8751                                                        N002.getOperand(1)),
8752                                            DAG.getNode(ISD::FNEG, SL, VT,
8753                                                        N1)));
8754         }
8755       }
8756 
8757       // fold (fsub x, (fma y, z, (fpext (fmul u, v))))
8758       //   -> (fma (fneg y), z, (fma (fneg (fpext u)), (fpext v), x))
8759       if (N1.getOpcode() == PreferredFusedOpcode &&
8760         N1.getOperand(2).getOpcode() == ISD::FP_EXTEND) {
8761         SDValue N120 = N1.getOperand(2).getOperand(0);
8762         if (N120.getOpcode() == ISD::FMUL) {
8763           SDValue N1200 = N120.getOperand(0);
8764           SDValue N1201 = N120.getOperand(1);
8765           return DAG.getNode(PreferredFusedOpcode, SL, VT,
8766                              DAG.getNode(ISD::FNEG, SL, VT, N1.getOperand(0)),
8767                              N1.getOperand(1),
8768                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8769                                          DAG.getNode(ISD::FNEG, SL, VT,
8770                                              DAG.getNode(ISD::FP_EXTEND, SL,
8771                                                          VT, N1200)),
8772                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8773                                                      N1201),
8774                                          N0));
8775         }
8776       }
8777 
8778       // fold (fsub x, (fpext (fma y, z, (fmul u, v))))
8779       //   -> (fma (fneg (fpext y)), (fpext z),
8780       //           (fma (fneg (fpext u)), (fpext v), x))
8781       // FIXME: This turns two single-precision and one double-precision
8782       // operation into two double-precision operations, which might not be
8783       // interesting for all targets, especially GPUs.
8784       if (N1.getOpcode() == ISD::FP_EXTEND &&
8785         N1.getOperand(0).getOpcode() == PreferredFusedOpcode) {
8786         SDValue N100 = N1.getOperand(0).getOperand(0);
8787         SDValue N101 = N1.getOperand(0).getOperand(1);
8788         SDValue N102 = N1.getOperand(0).getOperand(2);
8789         if (N102.getOpcode() == ISD::FMUL) {
8790           SDValue N1020 = N102.getOperand(0);
8791           SDValue N1021 = N102.getOperand(1);
8792           return DAG.getNode(PreferredFusedOpcode, SL, VT,
8793                              DAG.getNode(ISD::FNEG, SL, VT,
8794                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8795                                                      N100)),
8796                              DAG.getNode(ISD::FP_EXTEND, SL, VT, N101),
8797                              DAG.getNode(PreferredFusedOpcode, SL, VT,
8798                                          DAG.getNode(ISD::FNEG, SL, VT,
8799                                              DAG.getNode(ISD::FP_EXTEND, SL,
8800                                                          VT, N1020)),
8801                                          DAG.getNode(ISD::FP_EXTEND, SL, VT,
8802                                                      N1021),
8803                                          N0));
8804         }
8805       }
8806     }
8807   }
8808 
8809   return SDValue();
8810 }
8811 
8812 /// Try to perform FMA combining on a given FMUL node based on the distributive
8813 /// law x * (y + 1) = x * y + x and variants thereof (commuted versions,
8814 /// subtraction instead of addition).
8815 SDValue DAGCombiner::visitFMULForFMADistributiveCombine(SDNode *N) {
8816   SDValue N0 = N->getOperand(0);
8817   SDValue N1 = N->getOperand(1);
8818   EVT VT = N->getValueType(0);
8819   SDLoc SL(N);
8820 
8821   assert(N->getOpcode() == ISD::FMUL && "Expected FMUL Operation");
8822 
8823   const TargetOptions &Options = DAG.getTarget().Options;
8824 
8825   // The transforms below are incorrect when x == 0 and y == inf, because the
8826   // intermediate multiplication produces a nan.
8827   if (!Options.NoInfsFPMath)
8828     return SDValue();
8829 
8830   // Floating-point multiply-add without intermediate rounding.
8831   bool HasFMA =
8832       (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath) &&
8833       TLI.isFMAFasterThanFMulAndFAdd(VT) &&
8834       (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT));
8835 
8836   // Floating-point multiply-add with intermediate rounding. This can result
8837   // in a less precise result due to the changed rounding order.
8838   bool HasFMAD = Options.UnsafeFPMath &&
8839                  (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT));
8840 
8841   // No valid opcode, do not combine.
8842   if (!HasFMAD && !HasFMA)
8843     return SDValue();
8844 
8845   // Always prefer FMAD to FMA for precision.
8846   unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA;
8847   bool Aggressive = TLI.enableAggressiveFMAFusion(VT);
8848 
8849   // fold (fmul (fadd x, +1.0), y) -> (fma x, y, y)
8850   // fold (fmul (fadd x, -1.0), y) -> (fma x, y, (fneg y))
8851   auto FuseFADD = [&](SDValue X, SDValue Y) {
8852     if (X.getOpcode() == ISD::FADD && (Aggressive || X->hasOneUse())) {
8853       auto XC1 = isConstOrConstSplatFP(X.getOperand(1));
8854       if (XC1 && XC1->isExactlyValue(+1.0))
8855         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y);
8856       if (XC1 && XC1->isExactlyValue(-1.0))
8857         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
8858                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8859     }
8860     return SDValue();
8861   };
8862 
8863   if (SDValue FMA = FuseFADD(N0, N1))
8864     return FMA;
8865   if (SDValue FMA = FuseFADD(N1, N0))
8866     return FMA;
8867 
8868   // fold (fmul (fsub +1.0, x), y) -> (fma (fneg x), y, y)
8869   // fold (fmul (fsub -1.0, x), y) -> (fma (fneg x), y, (fneg y))
8870   // fold (fmul (fsub x, +1.0), y) -> (fma x, y, (fneg y))
8871   // fold (fmul (fsub x, -1.0), y) -> (fma x, y, y)
8872   auto FuseFSUB = [&](SDValue X, SDValue Y) {
8873     if (X.getOpcode() == ISD::FSUB && (Aggressive || X->hasOneUse())) {
8874       auto XC0 = isConstOrConstSplatFP(X.getOperand(0));
8875       if (XC0 && XC0->isExactlyValue(+1.0))
8876         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8877                            DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
8878                            Y);
8879       if (XC0 && XC0->isExactlyValue(-1.0))
8880         return DAG.getNode(PreferredFusedOpcode, SL, VT,
8881                            DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y,
8882                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8883 
8884       auto XC1 = isConstOrConstSplatFP(X.getOperand(1));
8885       if (XC1 && XC1->isExactlyValue(+1.0))
8886         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y,
8887                            DAG.getNode(ISD::FNEG, SL, VT, Y));
8888       if (XC1 && XC1->isExactlyValue(-1.0))
8889         return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y);
8890     }
8891     return SDValue();
8892   };
8893 
8894   if (SDValue FMA = FuseFSUB(N0, N1))
8895     return FMA;
8896   if (SDValue FMA = FuseFSUB(N1, N0))
8897     return FMA;
8898 
8899   return SDValue();
8900 }
8901 
8902 SDValue DAGCombiner::visitFADD(SDNode *N) {
8903   SDValue N0 = N->getOperand(0);
8904   SDValue N1 = N->getOperand(1);
8905   bool N0CFP = isConstantFPBuildVectorOrConstantFP(N0);
8906   bool N1CFP = isConstantFPBuildVectorOrConstantFP(N1);
8907   EVT VT = N->getValueType(0);
8908   SDLoc DL(N);
8909   const TargetOptions &Options = DAG.getTarget().Options;
8910   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
8911 
8912   // fold vector ops
8913   if (VT.isVector())
8914     if (SDValue FoldedVOp = SimplifyVBinOp(N))
8915       return FoldedVOp;
8916 
8917   // fold (fadd c1, c2) -> c1 + c2
8918   if (N0CFP && N1CFP)
8919     return DAG.getNode(ISD::FADD, DL, VT, N0, N1, Flags);
8920 
8921   // canonicalize constant to RHS
8922   if (N0CFP && !N1CFP)
8923     return DAG.getNode(ISD::FADD, DL, VT, N1, N0, Flags);
8924 
8925   // fold (fadd A, (fneg B)) -> (fsub A, B)
8926   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
8927       isNegatibleForFree(N1, LegalOperations, TLI, &Options) == 2)
8928     return DAG.getNode(ISD::FSUB, DL, VT, N0,
8929                        GetNegatedExpression(N1, DAG, LegalOperations), Flags);
8930 
8931   // fold (fadd (fneg A), B) -> (fsub B, A)
8932   if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) &&
8933       isNegatibleForFree(N0, LegalOperations, TLI, &Options) == 2)
8934     return DAG.getNode(ISD::FSUB, DL, VT, N1,
8935                        GetNegatedExpression(N0, DAG, LegalOperations), Flags);
8936 
8937   // FIXME: Auto-upgrade the target/function-level option.
8938   if (Options.UnsafeFPMath || N->getFlags()->hasNoSignedZeros()) {
8939     // fold (fadd A, 0) -> A
8940     if (ConstantFPSDNode *N1C = isConstOrConstSplatFP(N1))
8941       if (N1C->isZero())
8942         return N0;
8943   }
8944 
8945   // If 'unsafe math' is enabled, fold lots of things.
8946   if (Options.UnsafeFPMath) {
8947     // No FP constant should be created after legalization as Instruction
8948     // Selection pass has a hard time dealing with FP constants.
8949     bool AllowNewConst = (Level < AfterLegalizeDAG);
8950 
8951     // fold (fadd (fadd x, c1), c2) -> (fadd x, (fadd c1, c2))
8952     if (N1CFP && N0.getOpcode() == ISD::FADD && N0.getNode()->hasOneUse() &&
8953         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1)))
8954       return DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(0),
8955                          DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1), N1,
8956                                      Flags),
8957                          Flags);
8958 
8959     // If allowed, fold (fadd (fneg x), x) -> 0.0
8960     if (AllowNewConst && N0.getOpcode() == ISD::FNEG && N0.getOperand(0) == N1)
8961       return DAG.getConstantFP(0.0, DL, VT);
8962 
8963     // If allowed, fold (fadd x, (fneg x)) -> 0.0
8964     if (AllowNewConst && N1.getOpcode() == ISD::FNEG && N1.getOperand(0) == N0)
8965       return DAG.getConstantFP(0.0, DL, VT);
8966 
8967     // We can fold chains of FADD's of the same value into multiplications.
8968     // This transform is not safe in general because we are reducing the number
8969     // of rounding steps.
8970     if (TLI.isOperationLegalOrCustom(ISD::FMUL, VT) && !N0CFP && !N1CFP) {
8971       if (N0.getOpcode() == ISD::FMUL) {
8972         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
8973         bool CFP01 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(1));
8974 
8975         // (fadd (fmul x, c), x) -> (fmul x, c+1)
8976         if (CFP01 && !CFP00 && N0.getOperand(0) == N1) {
8977           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
8978                                        DAG.getConstantFP(1.0, DL, VT), Flags);
8979           return DAG.getNode(ISD::FMUL, DL, VT, N1, NewCFP, Flags);
8980         }
8981 
8982         // (fadd (fmul x, c), (fadd x, x)) -> (fmul x, c+2)
8983         if (CFP01 && !CFP00 && N1.getOpcode() == ISD::FADD &&
8984             N1.getOperand(0) == N1.getOperand(1) &&
8985             N0.getOperand(0) == N1.getOperand(0)) {
8986           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1),
8987                                        DAG.getConstantFP(2.0, DL, VT), Flags);
8988           return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), NewCFP, Flags);
8989         }
8990       }
8991 
8992       if (N1.getOpcode() == ISD::FMUL) {
8993         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
8994         bool CFP11 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(1));
8995 
8996         // (fadd x, (fmul x, c)) -> (fmul x, c+1)
8997         if (CFP11 && !CFP10 && N1.getOperand(0) == N0) {
8998           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
8999                                        DAG.getConstantFP(1.0, DL, VT), Flags);
9000           return DAG.getNode(ISD::FMUL, DL, VT, N0, NewCFP, Flags);
9001         }
9002 
9003         // (fadd (fadd x, x), (fmul x, c)) -> (fmul x, c+2)
9004         if (CFP11 && !CFP10 && N0.getOpcode() == ISD::FADD &&
9005             N0.getOperand(0) == N0.getOperand(1) &&
9006             N1.getOperand(0) == N0.getOperand(0)) {
9007           SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1),
9008                                        DAG.getConstantFP(2.0, DL, VT), Flags);
9009           return DAG.getNode(ISD::FMUL, DL, VT, N1.getOperand(0), NewCFP, Flags);
9010         }
9011       }
9012 
9013       if (N0.getOpcode() == ISD::FADD && AllowNewConst) {
9014         bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0));
9015         // (fadd (fadd x, x), x) -> (fmul x, 3.0)
9016         if (!CFP00 && N0.getOperand(0) == N0.getOperand(1) &&
9017             (N0.getOperand(0) == N1)) {
9018           return DAG.getNode(ISD::FMUL, DL, VT,
9019                              N1, DAG.getConstantFP(3.0, DL, VT), Flags);
9020         }
9021       }
9022 
9023       if (N1.getOpcode() == ISD::FADD && AllowNewConst) {
9024         bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0));
9025         // (fadd x, (fadd x, x)) -> (fmul x, 3.0)
9026         if (!CFP10 && N1.getOperand(0) == N1.getOperand(1) &&
9027             N1.getOperand(0) == N0) {
9028           return DAG.getNode(ISD::FMUL, DL, VT,
9029                              N0, DAG.getConstantFP(3.0, DL, VT), Flags);
9030         }
9031       }
9032 
9033       // (fadd (fadd x, x), (fadd x, x)) -> (fmul x, 4.0)
9034       if (AllowNewConst &&
9035           N0.getOpcode() == ISD::FADD && N1.getOpcode() == ISD::FADD &&
9036           N0.getOperand(0) == N0.getOperand(1) &&
9037           N1.getOperand(0) == N1.getOperand(1) &&
9038           N0.getOperand(0) == N1.getOperand(0)) {
9039         return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0),
9040                            DAG.getConstantFP(4.0, DL, VT), Flags);
9041       }
9042     }
9043   } // enable-unsafe-fp-math
9044 
9045   // FADD -> FMA combines:
9046   if (SDValue Fused = visitFADDForFMACombine(N)) {
9047     AddToWorklist(Fused.getNode());
9048     return Fused;
9049   }
9050   return SDValue();
9051 }
9052 
9053 SDValue DAGCombiner::visitFSUB(SDNode *N) {
9054   SDValue N0 = N->getOperand(0);
9055   SDValue N1 = N->getOperand(1);
9056   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9057   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9058   EVT VT = N->getValueType(0);
9059   SDLoc DL(N);
9060   const TargetOptions &Options = DAG.getTarget().Options;
9061   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
9062 
9063   // fold vector ops
9064   if (VT.isVector())
9065     if (SDValue FoldedVOp = SimplifyVBinOp(N))
9066       return FoldedVOp;
9067 
9068   // fold (fsub c1, c2) -> c1-c2
9069   if (N0CFP && N1CFP)
9070     return DAG.getNode(ISD::FSUB, DL, VT, N0, N1, Flags);
9071 
9072   // fold (fsub A, (fneg B)) -> (fadd A, B)
9073   if (isNegatibleForFree(N1, LegalOperations, TLI, &Options))
9074     return DAG.getNode(ISD::FADD, DL, VT, N0,
9075                        GetNegatedExpression(N1, DAG, LegalOperations), Flags);
9076 
9077   // FIXME: Auto-upgrade the target/function-level option.
9078   if (Options.UnsafeFPMath || N->getFlags()->hasNoSignedZeros()) {
9079     // (fsub 0, B) -> -B
9080     if (N0CFP && N0CFP->isZero()) {
9081       if (isNegatibleForFree(N1, LegalOperations, TLI, &Options))
9082         return GetNegatedExpression(N1, DAG, LegalOperations);
9083       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
9084         return DAG.getNode(ISD::FNEG, DL, VT, N1, Flags);
9085     }
9086   }
9087 
9088   // If 'unsafe math' is enabled, fold lots of things.
9089   if (Options.UnsafeFPMath) {
9090     // (fsub A, 0) -> A
9091     if (N1CFP && N1CFP->isZero())
9092       return N0;
9093 
9094     // (fsub x, x) -> 0.0
9095     if (N0 == N1)
9096       return DAG.getConstantFP(0.0f, DL, VT);
9097 
9098     // (fsub x, (fadd x, y)) -> (fneg y)
9099     // (fsub x, (fadd y, x)) -> (fneg y)
9100     if (N1.getOpcode() == ISD::FADD) {
9101       SDValue N10 = N1->getOperand(0);
9102       SDValue N11 = N1->getOperand(1);
9103 
9104       if (N10 == N0 && isNegatibleForFree(N11, LegalOperations, TLI, &Options))
9105         return GetNegatedExpression(N11, DAG, LegalOperations);
9106 
9107       if (N11 == N0 && isNegatibleForFree(N10, LegalOperations, TLI, &Options))
9108         return GetNegatedExpression(N10, DAG, LegalOperations);
9109     }
9110   }
9111 
9112   // FSUB -> FMA combines:
9113   if (SDValue Fused = visitFSUBForFMACombine(N)) {
9114     AddToWorklist(Fused.getNode());
9115     return Fused;
9116   }
9117 
9118   return SDValue();
9119 }
9120 
9121 SDValue DAGCombiner::visitFMUL(SDNode *N) {
9122   SDValue N0 = N->getOperand(0);
9123   SDValue N1 = N->getOperand(1);
9124   ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9125   ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9126   EVT VT = N->getValueType(0);
9127   SDLoc DL(N);
9128   const TargetOptions &Options = DAG.getTarget().Options;
9129   const SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
9130 
9131   // fold vector ops
9132   if (VT.isVector()) {
9133     // This just handles C1 * C2 for vectors. Other vector folds are below.
9134     if (SDValue FoldedVOp = SimplifyVBinOp(N))
9135       return FoldedVOp;
9136   }
9137 
9138   // fold (fmul c1, c2) -> c1*c2
9139   if (N0CFP && N1CFP)
9140     return DAG.getNode(ISD::FMUL, DL, VT, N0, N1, Flags);
9141 
9142   // canonicalize constant to RHS
9143   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9144      !isConstantFPBuildVectorOrConstantFP(N1))
9145     return DAG.getNode(ISD::FMUL, DL, VT, N1, N0, Flags);
9146 
9147   // fold (fmul A, 1.0) -> A
9148   if (N1CFP && N1CFP->isExactlyValue(1.0))
9149     return N0;
9150 
9151   if (Options.UnsafeFPMath) {
9152     // fold (fmul A, 0) -> 0
9153     if (N1CFP && N1CFP->isZero())
9154       return N1;
9155 
9156     // fold (fmul (fmul x, c1), c2) -> (fmul x, (fmul c1, c2))
9157     if (N0.getOpcode() == ISD::FMUL) {
9158       // Fold scalars or any vector constants (not just splats).
9159       // This fold is done in general by InstCombine, but extra fmul insts
9160       // may have been generated during lowering.
9161       SDValue N00 = N0.getOperand(0);
9162       SDValue N01 = N0.getOperand(1);
9163       auto *BV1 = dyn_cast<BuildVectorSDNode>(N1);
9164       auto *BV00 = dyn_cast<BuildVectorSDNode>(N00);
9165       auto *BV01 = dyn_cast<BuildVectorSDNode>(N01);
9166 
9167       // Check 1: Make sure that the first operand of the inner multiply is NOT
9168       // a constant. Otherwise, we may induce infinite looping.
9169       if (!(isConstOrConstSplatFP(N00) || (BV00 && BV00->isConstant()))) {
9170         // Check 2: Make sure that the second operand of the inner multiply and
9171         // the second operand of the outer multiply are constants.
9172         if ((N1CFP && isConstOrConstSplatFP(N01)) ||
9173             (BV1 && BV01 && BV1->isConstant() && BV01->isConstant())) {
9174           SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, N01, N1, Flags);
9175           return DAG.getNode(ISD::FMUL, DL, VT, N00, MulConsts, Flags);
9176         }
9177       }
9178     }
9179 
9180     // fold (fmul (fadd x, x), c) -> (fmul x, (fmul 2.0, c))
9181     // Undo the fmul 2.0, x -> fadd x, x transformation, since if it occurs
9182     // during an early run of DAGCombiner can prevent folding with fmuls
9183     // inserted during lowering.
9184     if (N0.getOpcode() == ISD::FADD &&
9185         (N0.getOperand(0) == N0.getOperand(1)) &&
9186         N0.hasOneUse()) {
9187       const SDValue Two = DAG.getConstantFP(2.0, DL, VT);
9188       SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, Two, N1, Flags);
9189       return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), MulConsts, Flags);
9190     }
9191   }
9192 
9193   // fold (fmul X, 2.0) -> (fadd X, X)
9194   if (N1CFP && N1CFP->isExactlyValue(+2.0))
9195     return DAG.getNode(ISD::FADD, DL, VT, N0, N0, Flags);
9196 
9197   // fold (fmul X, -1.0) -> (fneg X)
9198   if (N1CFP && N1CFP->isExactlyValue(-1.0))
9199     if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
9200       return DAG.getNode(ISD::FNEG, DL, VT, N0);
9201 
9202   // fold (fmul (fneg X), (fneg Y)) -> (fmul X, Y)
9203   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) {
9204     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) {
9205       // Both can be negated for free, check to see if at least one is cheaper
9206       // negated.
9207       if (LHSNeg == 2 || RHSNeg == 2)
9208         return DAG.getNode(ISD::FMUL, DL, VT,
9209                            GetNegatedExpression(N0, DAG, LegalOperations),
9210                            GetNegatedExpression(N1, DAG, LegalOperations),
9211                            Flags);
9212     }
9213   }
9214 
9215   // FMUL -> FMA combines:
9216   if (SDValue Fused = visitFMULForFMADistributiveCombine(N)) {
9217     AddToWorklist(Fused.getNode());
9218     return Fused;
9219   }
9220 
9221   return SDValue();
9222 }
9223 
9224 SDValue DAGCombiner::visitFMA(SDNode *N) {
9225   SDValue N0 = N->getOperand(0);
9226   SDValue N1 = N->getOperand(1);
9227   SDValue N2 = N->getOperand(2);
9228   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9229   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
9230   EVT VT = N->getValueType(0);
9231   SDLoc DL(N);
9232   const TargetOptions &Options = DAG.getTarget().Options;
9233 
9234   // Constant fold FMA.
9235   if (isa<ConstantFPSDNode>(N0) &&
9236       isa<ConstantFPSDNode>(N1) &&
9237       isa<ConstantFPSDNode>(N2)) {
9238     return DAG.getNode(ISD::FMA, DL, VT, N0, N1, N2);
9239   }
9240 
9241   if (Options.UnsafeFPMath) {
9242     if (N0CFP && N0CFP->isZero())
9243       return N2;
9244     if (N1CFP && N1CFP->isZero())
9245       return N2;
9246   }
9247   // TODO: The FMA node should have flags that propagate to these nodes.
9248   if (N0CFP && N0CFP->isExactlyValue(1.0))
9249     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N1, N2);
9250   if (N1CFP && N1CFP->isExactlyValue(1.0))
9251     return DAG.getNode(ISD::FADD, SDLoc(N), VT, N0, N2);
9252 
9253   // Canonicalize (fma c, x, y) -> (fma x, c, y)
9254   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9255      !isConstantFPBuildVectorOrConstantFP(N1))
9256     return DAG.getNode(ISD::FMA, SDLoc(N), VT, N1, N0, N2);
9257 
9258   // TODO: FMA nodes should have flags that propagate to the created nodes.
9259   // For now, create a Flags object for use with all unsafe math transforms.
9260   SDNodeFlags Flags;
9261   Flags.setUnsafeAlgebra(true);
9262 
9263   if (Options.UnsafeFPMath) {
9264     // (fma x, c1, (fmul x, c2)) -> (fmul x, c1+c2)
9265     if (N2.getOpcode() == ISD::FMUL && N0 == N2.getOperand(0) &&
9266         isConstantFPBuildVectorOrConstantFP(N1) &&
9267         isConstantFPBuildVectorOrConstantFP(N2.getOperand(1))) {
9268       return DAG.getNode(ISD::FMUL, DL, VT, N0,
9269                          DAG.getNode(ISD::FADD, DL, VT, N1, N2.getOperand(1),
9270                                      &Flags), &Flags);
9271     }
9272 
9273     // (fma (fmul x, c1), c2, y) -> (fma x, c1*c2, y)
9274     if (N0.getOpcode() == ISD::FMUL &&
9275         isConstantFPBuildVectorOrConstantFP(N1) &&
9276         isConstantFPBuildVectorOrConstantFP(N0.getOperand(1))) {
9277       return DAG.getNode(ISD::FMA, DL, VT,
9278                          N0.getOperand(0),
9279                          DAG.getNode(ISD::FMUL, DL, VT, N1, N0.getOperand(1),
9280                                      &Flags),
9281                          N2);
9282     }
9283   }
9284 
9285   // (fma x, 1, y) -> (fadd x, y)
9286   // (fma x, -1, y) -> (fadd (fneg x), y)
9287   if (N1CFP) {
9288     if (N1CFP->isExactlyValue(1.0))
9289       // TODO: The FMA node should have flags that propagate to this node.
9290       return DAG.getNode(ISD::FADD, DL, VT, N0, N2);
9291 
9292     if (N1CFP->isExactlyValue(-1.0) &&
9293         (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))) {
9294       SDValue RHSNeg = DAG.getNode(ISD::FNEG, DL, VT, N0);
9295       AddToWorklist(RHSNeg.getNode());
9296       // TODO: The FMA node should have flags that propagate to this node.
9297       return DAG.getNode(ISD::FADD, DL, VT, N2, RHSNeg);
9298     }
9299   }
9300 
9301   if (Options.UnsafeFPMath) {
9302     // (fma x, c, x) -> (fmul x, (c+1))
9303     if (N1CFP && N0 == N2) {
9304       return DAG.getNode(ISD::FMUL, DL, VT, N0,
9305                          DAG.getNode(ISD::FADD, DL, VT, N1,
9306                                      DAG.getConstantFP(1.0, DL, VT), &Flags),
9307                          &Flags);
9308     }
9309 
9310     // (fma x, c, (fneg x)) -> (fmul x, (c-1))
9311     if (N1CFP && N2.getOpcode() == ISD::FNEG && N2.getOperand(0) == N0) {
9312       return DAG.getNode(ISD::FMUL, DL, VT, N0,
9313                          DAG.getNode(ISD::FADD, DL, VT, N1,
9314                                      DAG.getConstantFP(-1.0, DL, VT), &Flags),
9315                          &Flags);
9316     }
9317   }
9318 
9319   return SDValue();
9320 }
9321 
9322 // Combine multiple FDIVs with the same divisor into multiple FMULs by the
9323 // reciprocal.
9324 // E.g., (a / D; b / D;) -> (recip = 1.0 / D; a * recip; b * recip)
9325 // Notice that this is not always beneficial. One reason is different targets
9326 // may have different costs for FDIV and FMUL, so sometimes the cost of two
9327 // FDIVs may be lower than the cost of one FDIV and two FMULs. Another reason
9328 // is the critical path is increased from "one FDIV" to "one FDIV + one FMUL".
9329 SDValue DAGCombiner::combineRepeatedFPDivisors(SDNode *N) {
9330   bool UnsafeMath = DAG.getTarget().Options.UnsafeFPMath;
9331   const SDNodeFlags *Flags = N->getFlags();
9332   if (!UnsafeMath && !Flags->hasAllowReciprocal())
9333     return SDValue();
9334 
9335   // Skip if current node is a reciprocal.
9336   SDValue N0 = N->getOperand(0);
9337   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9338   if (N0CFP && N0CFP->isExactlyValue(1.0))
9339     return SDValue();
9340 
9341   // Exit early if the target does not want this transform or if there can't
9342   // possibly be enough uses of the divisor to make the transform worthwhile.
9343   SDValue N1 = N->getOperand(1);
9344   unsigned MinUses = TLI.combineRepeatedFPDivisors();
9345   if (!MinUses || N1->use_size() < MinUses)
9346     return SDValue();
9347 
9348   // Find all FDIV users of the same divisor.
9349   // Use a set because duplicates may be present in the user list.
9350   SetVector<SDNode *> Users;
9351   for (auto *U : N1->uses()) {
9352     if (U->getOpcode() == ISD::FDIV && U->getOperand(1) == N1) {
9353       // This division is eligible for optimization only if global unsafe math
9354       // is enabled or if this division allows reciprocal formation.
9355       if (UnsafeMath || U->getFlags()->hasAllowReciprocal())
9356         Users.insert(U);
9357     }
9358   }
9359 
9360   // Now that we have the actual number of divisor uses, make sure it meets
9361   // the minimum threshold specified by the target.
9362   if (Users.size() < MinUses)
9363     return SDValue();
9364 
9365   EVT VT = N->getValueType(0);
9366   SDLoc DL(N);
9367   SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
9368   SDValue Reciprocal = DAG.getNode(ISD::FDIV, DL, VT, FPOne, N1, Flags);
9369 
9370   // Dividend / Divisor -> Dividend * Reciprocal
9371   for (auto *U : Users) {
9372     SDValue Dividend = U->getOperand(0);
9373     if (Dividend != FPOne) {
9374       SDValue NewNode = DAG.getNode(ISD::FMUL, SDLoc(U), VT, Dividend,
9375                                     Reciprocal, Flags);
9376       CombineTo(U, NewNode);
9377     } else if (U != Reciprocal.getNode()) {
9378       // In the absence of fast-math-flags, this user node is always the
9379       // same node as Reciprocal, but with FMF they may be different nodes.
9380       CombineTo(U, Reciprocal);
9381     }
9382   }
9383   return SDValue(N, 0);  // N was replaced.
9384 }
9385 
9386 SDValue DAGCombiner::visitFDIV(SDNode *N) {
9387   SDValue N0 = N->getOperand(0);
9388   SDValue N1 = N->getOperand(1);
9389   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9390   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
9391   EVT VT = N->getValueType(0);
9392   SDLoc DL(N);
9393   const TargetOptions &Options = DAG.getTarget().Options;
9394   SDNodeFlags *Flags = &cast<BinaryWithFlagsSDNode>(N)->Flags;
9395 
9396   // fold vector ops
9397   if (VT.isVector())
9398     if (SDValue FoldedVOp = SimplifyVBinOp(N))
9399       return FoldedVOp;
9400 
9401   // fold (fdiv c1, c2) -> c1/c2
9402   if (N0CFP && N1CFP)
9403     return DAG.getNode(ISD::FDIV, SDLoc(N), VT, N0, N1, Flags);
9404 
9405   if (Options.UnsafeFPMath) {
9406     // fold (fdiv X, c2) -> fmul X, 1/c2 if losing precision is acceptable.
9407     if (N1CFP) {
9408       // Compute the reciprocal 1.0 / c2.
9409       const APFloat &N1APF = N1CFP->getValueAPF();
9410       APFloat Recip(N1APF.getSemantics(), 1); // 1.0
9411       APFloat::opStatus st = Recip.divide(N1APF, APFloat::rmNearestTiesToEven);
9412       // Only do the transform if the reciprocal is a legal fp immediate that
9413       // isn't too nasty (eg NaN, denormal, ...).
9414       if ((st == APFloat::opOK || st == APFloat::opInexact) && // Not too nasty
9415           (!LegalOperations ||
9416            // FIXME: custom lowering of ConstantFP might fail (see e.g. ARM
9417            // backend)... we should handle this gracefully after Legalize.
9418            // TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT) ||
9419            TLI.isOperationLegal(llvm::ISD::ConstantFP, VT) ||
9420            TLI.isFPImmLegal(Recip, VT)))
9421         return DAG.getNode(ISD::FMUL, DL, VT, N0,
9422                            DAG.getConstantFP(Recip, DL, VT), Flags);
9423     }
9424 
9425     // If this FDIV is part of a reciprocal square root, it may be folded
9426     // into a target-specific square root estimate instruction.
9427     if (N1.getOpcode() == ISD::FSQRT) {
9428       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0), Flags)) {
9429         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
9430       }
9431     } else if (N1.getOpcode() == ISD::FP_EXTEND &&
9432                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
9433       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
9434                                           Flags)) {
9435         RV = DAG.getNode(ISD::FP_EXTEND, SDLoc(N1), VT, RV);
9436         AddToWorklist(RV.getNode());
9437         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
9438       }
9439     } else if (N1.getOpcode() == ISD::FP_ROUND &&
9440                N1.getOperand(0).getOpcode() == ISD::FSQRT) {
9441       if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0),
9442                                           Flags)) {
9443         RV = DAG.getNode(ISD::FP_ROUND, SDLoc(N1), VT, RV, N1.getOperand(1));
9444         AddToWorklist(RV.getNode());
9445         return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
9446       }
9447     } else if (N1.getOpcode() == ISD::FMUL) {
9448       // Look through an FMUL. Even though this won't remove the FDIV directly,
9449       // it's still worthwhile to get rid of the FSQRT if possible.
9450       SDValue SqrtOp;
9451       SDValue OtherOp;
9452       if (N1.getOperand(0).getOpcode() == ISD::FSQRT) {
9453         SqrtOp = N1.getOperand(0);
9454         OtherOp = N1.getOperand(1);
9455       } else if (N1.getOperand(1).getOpcode() == ISD::FSQRT) {
9456         SqrtOp = N1.getOperand(1);
9457         OtherOp = N1.getOperand(0);
9458       }
9459       if (SqrtOp.getNode()) {
9460         // We found a FSQRT, so try to make this fold:
9461         // x / (y * sqrt(z)) -> x * (rsqrt(z) / y)
9462         if (SDValue RV = buildRsqrtEstimate(SqrtOp.getOperand(0), Flags)) {
9463           RV = DAG.getNode(ISD::FDIV, SDLoc(N1), VT, RV, OtherOp, Flags);
9464           AddToWorklist(RV.getNode());
9465           return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
9466         }
9467       }
9468     }
9469 
9470     // Fold into a reciprocal estimate and multiply instead of a real divide.
9471     if (SDValue RV = BuildReciprocalEstimate(N1, Flags)) {
9472       AddToWorklist(RV.getNode());
9473       return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags);
9474     }
9475   }
9476 
9477   // (fdiv (fneg X), (fneg Y)) -> (fdiv X, Y)
9478   if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) {
9479     if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) {
9480       // Both can be negated for free, check to see if at least one is cheaper
9481       // negated.
9482       if (LHSNeg == 2 || RHSNeg == 2)
9483         return DAG.getNode(ISD::FDIV, SDLoc(N), VT,
9484                            GetNegatedExpression(N0, DAG, LegalOperations),
9485                            GetNegatedExpression(N1, DAG, LegalOperations),
9486                            Flags);
9487     }
9488   }
9489 
9490   if (SDValue CombineRepeatedDivisors = combineRepeatedFPDivisors(N))
9491     return CombineRepeatedDivisors;
9492 
9493   return SDValue();
9494 }
9495 
9496 SDValue DAGCombiner::visitFREM(SDNode *N) {
9497   SDValue N0 = N->getOperand(0);
9498   SDValue N1 = N->getOperand(1);
9499   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9500   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
9501   EVT VT = N->getValueType(0);
9502 
9503   // fold (frem c1, c2) -> fmod(c1,c2)
9504   if (N0CFP && N1CFP)
9505     return DAG.getNode(ISD::FREM, SDLoc(N), VT, N0, N1,
9506                        &cast<BinaryWithFlagsSDNode>(N)->Flags);
9507 
9508   return SDValue();
9509 }
9510 
9511 SDValue DAGCombiner::visitFSQRT(SDNode *N) {
9512   if (!DAG.getTarget().Options.UnsafeFPMath)
9513     return SDValue();
9514 
9515   SDValue N0 = N->getOperand(0);
9516   if (TLI.isFsqrtCheap(N0, DAG))
9517     return SDValue();
9518 
9519   // TODO: FSQRT nodes should have flags that propagate to the created nodes.
9520   // For now, create a Flags object for use with all unsafe math transforms.
9521   SDNodeFlags Flags;
9522   Flags.setUnsafeAlgebra(true);
9523   return buildSqrtEstimate(N0, &Flags);
9524 }
9525 
9526 /// copysign(x, fp_extend(y)) -> copysign(x, y)
9527 /// copysign(x, fp_round(y)) -> copysign(x, y)
9528 static inline bool CanCombineFCOPYSIGN_EXTEND_ROUND(SDNode *N) {
9529   SDValue N1 = N->getOperand(1);
9530   if ((N1.getOpcode() == ISD::FP_EXTEND ||
9531        N1.getOpcode() == ISD::FP_ROUND)) {
9532     // Do not optimize out type conversion of f128 type yet.
9533     // For some targets like x86_64, configuration is changed to keep one f128
9534     // value in one SSE register, but instruction selection cannot handle
9535     // FCOPYSIGN on SSE registers yet.
9536     EVT N1VT = N1->getValueType(0);
9537     EVT N1Op0VT = N1->getOperand(0)->getValueType(0);
9538     return (N1VT == N1Op0VT || N1Op0VT != MVT::f128);
9539   }
9540   return false;
9541 }
9542 
9543 SDValue DAGCombiner::visitFCOPYSIGN(SDNode *N) {
9544   SDValue N0 = N->getOperand(0);
9545   SDValue N1 = N->getOperand(1);
9546   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9547   ConstantFPSDNode *N1CFP = dyn_cast<ConstantFPSDNode>(N1);
9548   EVT VT = N->getValueType(0);
9549 
9550   if (N0CFP && N1CFP) // Constant fold
9551     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1);
9552 
9553   if (N1CFP) {
9554     const APFloat &V = N1CFP->getValueAPF();
9555     // copysign(x, c1) -> fabs(x)       iff ispos(c1)
9556     // copysign(x, c1) -> fneg(fabs(x)) iff isneg(c1)
9557     if (!V.isNegative()) {
9558       if (!LegalOperations || TLI.isOperationLegal(ISD::FABS, VT))
9559         return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9560     } else {
9561       if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))
9562         return DAG.getNode(ISD::FNEG, SDLoc(N), VT,
9563                            DAG.getNode(ISD::FABS, SDLoc(N0), VT, N0));
9564     }
9565   }
9566 
9567   // copysign(fabs(x), y) -> copysign(x, y)
9568   // copysign(fneg(x), y) -> copysign(x, y)
9569   // copysign(copysign(x,z), y) -> copysign(x, y)
9570   if (N0.getOpcode() == ISD::FABS || N0.getOpcode() == ISD::FNEG ||
9571       N0.getOpcode() == ISD::FCOPYSIGN)
9572     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0.getOperand(0), N1);
9573 
9574   // copysign(x, abs(y)) -> abs(x)
9575   if (N1.getOpcode() == ISD::FABS)
9576     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
9577 
9578   // copysign(x, copysign(y,z)) -> copysign(x, z)
9579   if (N1.getOpcode() == ISD::FCOPYSIGN)
9580     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(1));
9581 
9582   // copysign(x, fp_extend(y)) -> copysign(x, y)
9583   // copysign(x, fp_round(y)) -> copysign(x, y)
9584   if (CanCombineFCOPYSIGN_EXTEND_ROUND(N))
9585     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(0));
9586 
9587   return SDValue();
9588 }
9589 
9590 SDValue DAGCombiner::visitSINT_TO_FP(SDNode *N) {
9591   SDValue N0 = N->getOperand(0);
9592   EVT VT = N->getValueType(0);
9593   EVT OpVT = N0.getValueType();
9594 
9595   // fold (sint_to_fp c1) -> c1fp
9596   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
9597       // ...but only if the target supports immediate floating-point values
9598       (!LegalOperations ||
9599        TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT)))
9600     return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
9601 
9602   // If the input is a legal type, and SINT_TO_FP is not legal on this target,
9603   // but UINT_TO_FP is legal on this target, try to convert.
9604   if (!TLI.isOperationLegalOrCustom(ISD::SINT_TO_FP, OpVT) &&
9605       TLI.isOperationLegalOrCustom(ISD::UINT_TO_FP, OpVT)) {
9606     // If the sign bit is known to be zero, we can change this to UINT_TO_FP.
9607     if (DAG.SignBitIsZero(N0))
9608       return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
9609   }
9610 
9611   // The next optimizations are desirable only if SELECT_CC can be lowered.
9612   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
9613     // fold (sint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
9614     if (N0.getOpcode() == ISD::SETCC && N0.getValueType() == MVT::i1 &&
9615         !VT.isVector() &&
9616         (!LegalOperations ||
9617          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9618       SDLoc DL(N);
9619       SDValue Ops[] =
9620         { N0.getOperand(0), N0.getOperand(1),
9621           DAG.getConstantFP(-1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9622           N0.getOperand(2) };
9623       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9624     }
9625 
9626     // fold (sint_to_fp (zext (setcc x, y, cc))) ->
9627     //      (select_cc x, y, 1.0, 0.0,, cc)
9628     if (N0.getOpcode() == ISD::ZERO_EXTEND &&
9629         N0.getOperand(0).getOpcode() == ISD::SETCC &&!VT.isVector() &&
9630         (!LegalOperations ||
9631          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9632       SDLoc DL(N);
9633       SDValue Ops[] =
9634         { N0.getOperand(0).getOperand(0), N0.getOperand(0).getOperand(1),
9635           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9636           N0.getOperand(0).getOperand(2) };
9637       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9638     }
9639   }
9640 
9641   return SDValue();
9642 }
9643 
9644 SDValue DAGCombiner::visitUINT_TO_FP(SDNode *N) {
9645   SDValue N0 = N->getOperand(0);
9646   EVT VT = N->getValueType(0);
9647   EVT OpVT = N0.getValueType();
9648 
9649   // fold (uint_to_fp c1) -> c1fp
9650   if (DAG.isConstantIntBuildVectorOrConstantInt(N0) &&
9651       // ...but only if the target supports immediate floating-point values
9652       (!LegalOperations ||
9653        TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT)))
9654     return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0);
9655 
9656   // If the input is a legal type, and UINT_TO_FP is not legal on this target,
9657   // but SINT_TO_FP is legal on this target, try to convert.
9658   if (!TLI.isOperationLegalOrCustom(ISD::UINT_TO_FP, OpVT) &&
9659       TLI.isOperationLegalOrCustom(ISD::SINT_TO_FP, OpVT)) {
9660     // If the sign bit is known to be zero, we can change this to SINT_TO_FP.
9661     if (DAG.SignBitIsZero(N0))
9662       return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0);
9663   }
9664 
9665   // The next optimizations are desirable only if SELECT_CC can be lowered.
9666   if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) {
9667     // fold (uint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc)
9668 
9669     if (N0.getOpcode() == ISD::SETCC && !VT.isVector() &&
9670         (!LegalOperations ||
9671          TLI.isOperationLegalOrCustom(llvm::ISD::ConstantFP, VT))) {
9672       SDLoc DL(N);
9673       SDValue Ops[] =
9674         { N0.getOperand(0), N0.getOperand(1),
9675           DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT),
9676           N0.getOperand(2) };
9677       return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
9678     }
9679   }
9680 
9681   return SDValue();
9682 }
9683 
9684 // Fold (fp_to_{s/u}int ({s/u}int_to_fpx)) -> zext x, sext x, trunc x, or x
9685 static SDValue FoldIntToFPToInt(SDNode *N, SelectionDAG &DAG) {
9686   SDValue N0 = N->getOperand(0);
9687   EVT VT = N->getValueType(0);
9688 
9689   if (N0.getOpcode() != ISD::UINT_TO_FP && N0.getOpcode() != ISD::SINT_TO_FP)
9690     return SDValue();
9691 
9692   SDValue Src = N0.getOperand(0);
9693   EVT SrcVT = Src.getValueType();
9694   bool IsInputSigned = N0.getOpcode() == ISD::SINT_TO_FP;
9695   bool IsOutputSigned = N->getOpcode() == ISD::FP_TO_SINT;
9696 
9697   // We can safely assume the conversion won't overflow the output range,
9698   // because (for example) (uint8_t)18293.f is undefined behavior.
9699 
9700   // Since we can assume the conversion won't overflow, our decision as to
9701   // whether the input will fit in the float should depend on the minimum
9702   // of the input range and output range.
9703 
9704   // This means this is also safe for a signed input and unsigned output, since
9705   // a negative input would lead to undefined behavior.
9706   unsigned InputSize = (int)SrcVT.getScalarSizeInBits() - IsInputSigned;
9707   unsigned OutputSize = (int)VT.getScalarSizeInBits() - IsOutputSigned;
9708   unsigned ActualSize = std::min(InputSize, OutputSize);
9709   const fltSemantics &sem = DAG.EVTToAPFloatSemantics(N0.getValueType());
9710 
9711   // We can only fold away the float conversion if the input range can be
9712   // represented exactly in the float range.
9713   if (APFloat::semanticsPrecision(sem) >= ActualSize) {
9714     if (VT.getScalarSizeInBits() > SrcVT.getScalarSizeInBits()) {
9715       unsigned ExtOp = IsInputSigned && IsOutputSigned ? ISD::SIGN_EXTEND
9716                                                        : ISD::ZERO_EXTEND;
9717       return DAG.getNode(ExtOp, SDLoc(N), VT, Src);
9718     }
9719     if (VT.getScalarSizeInBits() < SrcVT.getScalarSizeInBits())
9720       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Src);
9721     return DAG.getBitcast(VT, Src);
9722   }
9723   return SDValue();
9724 }
9725 
9726 SDValue DAGCombiner::visitFP_TO_SINT(SDNode *N) {
9727   SDValue N0 = N->getOperand(0);
9728   EVT VT = N->getValueType(0);
9729 
9730   // fold (fp_to_sint c1fp) -> c1
9731   if (isConstantFPBuildVectorOrConstantFP(N0))
9732     return DAG.getNode(ISD::FP_TO_SINT, SDLoc(N), VT, N0);
9733 
9734   return FoldIntToFPToInt(N, DAG);
9735 }
9736 
9737 SDValue DAGCombiner::visitFP_TO_UINT(SDNode *N) {
9738   SDValue N0 = N->getOperand(0);
9739   EVT VT = N->getValueType(0);
9740 
9741   // fold (fp_to_uint c1fp) -> c1
9742   if (isConstantFPBuildVectorOrConstantFP(N0))
9743     return DAG.getNode(ISD::FP_TO_UINT, SDLoc(N), VT, N0);
9744 
9745   return FoldIntToFPToInt(N, DAG);
9746 }
9747 
9748 SDValue DAGCombiner::visitFP_ROUND(SDNode *N) {
9749   SDValue N0 = N->getOperand(0);
9750   SDValue N1 = N->getOperand(1);
9751   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9752   EVT VT = N->getValueType(0);
9753 
9754   // fold (fp_round c1fp) -> c1fp
9755   if (N0CFP)
9756     return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT, N0, N1);
9757 
9758   // fold (fp_round (fp_extend x)) -> x
9759   if (N0.getOpcode() == ISD::FP_EXTEND && VT == N0.getOperand(0).getValueType())
9760     return N0.getOperand(0);
9761 
9762   // fold (fp_round (fp_round x)) -> (fp_round x)
9763   if (N0.getOpcode() == ISD::FP_ROUND) {
9764     const bool NIsTrunc = N->getConstantOperandVal(1) == 1;
9765     const bool N0IsTrunc = N0.getConstantOperandVal(1) == 1;
9766 
9767     // Skip this folding if it results in an fp_round from f80 to f16.
9768     //
9769     // f80 to f16 always generates an expensive (and as yet, unimplemented)
9770     // libcall to __truncxfhf2 instead of selecting native f16 conversion
9771     // instructions from f32 or f64.  Moreover, the first (value-preserving)
9772     // fp_round from f80 to either f32 or f64 may become a NOP in platforms like
9773     // x86.
9774     if (N0.getOperand(0).getValueType() == MVT::f80 && VT == MVT::f16)
9775       return SDValue();
9776 
9777     // If the first fp_round isn't a value preserving truncation, it might
9778     // introduce a tie in the second fp_round, that wouldn't occur in the
9779     // single-step fp_round we want to fold to.
9780     // In other words, double rounding isn't the same as rounding.
9781     // Also, this is a value preserving truncation iff both fp_round's are.
9782     if (DAG.getTarget().Options.UnsafeFPMath || N0IsTrunc) {
9783       SDLoc DL(N);
9784       return DAG.getNode(ISD::FP_ROUND, DL, VT, N0.getOperand(0),
9785                          DAG.getIntPtrConstant(NIsTrunc && N0IsTrunc, DL));
9786     }
9787   }
9788 
9789   // fold (fp_round (copysign X, Y)) -> (copysign (fp_round X), Y)
9790   if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse()) {
9791     SDValue Tmp = DAG.getNode(ISD::FP_ROUND, SDLoc(N0), VT,
9792                               N0.getOperand(0), N1);
9793     AddToWorklist(Tmp.getNode());
9794     return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT,
9795                        Tmp, N0.getOperand(1));
9796   }
9797 
9798   return SDValue();
9799 }
9800 
9801 SDValue DAGCombiner::visitFP_ROUND_INREG(SDNode *N) {
9802   SDValue N0 = N->getOperand(0);
9803   EVT VT = N->getValueType(0);
9804   EVT EVT = cast<VTSDNode>(N->getOperand(1))->getVT();
9805   ConstantFPSDNode *N0CFP = dyn_cast<ConstantFPSDNode>(N0);
9806 
9807   // fold (fp_round_inreg c1fp) -> c1fp
9808   if (N0CFP && isTypeLegal(EVT)) {
9809     SDLoc DL(N);
9810     SDValue Round = DAG.getConstantFP(*N0CFP->getConstantFPValue(), DL, EVT);
9811     return DAG.getNode(ISD::FP_EXTEND, DL, VT, Round);
9812   }
9813 
9814   return SDValue();
9815 }
9816 
9817 SDValue DAGCombiner::visitFP_EXTEND(SDNode *N) {
9818   SDValue N0 = N->getOperand(0);
9819   EVT VT = N->getValueType(0);
9820 
9821   // If this is fp_round(fpextend), don't fold it, allow ourselves to be folded.
9822   if (N->hasOneUse() &&
9823       N->use_begin()->getOpcode() == ISD::FP_ROUND)
9824     return SDValue();
9825 
9826   // fold (fp_extend c1fp) -> c1fp
9827   if (isConstantFPBuildVectorOrConstantFP(N0))
9828     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, N0);
9829 
9830   // fold (fp_extend (fp16_to_fp op)) -> (fp16_to_fp op)
9831   if (N0.getOpcode() == ISD::FP16_TO_FP &&
9832       TLI.getOperationAction(ISD::FP16_TO_FP, VT) == TargetLowering::Legal)
9833     return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), VT, N0.getOperand(0));
9834 
9835   // Turn fp_extend(fp_round(X, 1)) -> x since the fp_round doesn't affect the
9836   // value of X.
9837   if (N0.getOpcode() == ISD::FP_ROUND
9838       && N0.getConstantOperandVal(1) == 1) {
9839     SDValue In = N0.getOperand(0);
9840     if (In.getValueType() == VT) return In;
9841     if (VT.bitsLT(In.getValueType()))
9842       return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT,
9843                          In, N0.getOperand(1));
9844     return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, In);
9845   }
9846 
9847   // fold (fpext (load x)) -> (fpext (fptrunc (extload x)))
9848   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
9849        TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) {
9850     LoadSDNode *LN0 = cast<LoadSDNode>(N0);
9851     SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT,
9852                                      LN0->getChain(),
9853                                      LN0->getBasePtr(), N0.getValueType(),
9854                                      LN0->getMemOperand());
9855     CombineTo(N, ExtLoad);
9856     CombineTo(N0.getNode(),
9857               DAG.getNode(ISD::FP_ROUND, SDLoc(N0),
9858                           N0.getValueType(), ExtLoad,
9859                           DAG.getIntPtrConstant(1, SDLoc(N0))),
9860               ExtLoad.getValue(1));
9861     return SDValue(N, 0);   // Return N so it doesn't get rechecked!
9862   }
9863 
9864   return SDValue();
9865 }
9866 
9867 SDValue DAGCombiner::visitFCEIL(SDNode *N) {
9868   SDValue N0 = N->getOperand(0);
9869   EVT VT = N->getValueType(0);
9870 
9871   // fold (fceil c1) -> fceil(c1)
9872   if (isConstantFPBuildVectorOrConstantFP(N0))
9873     return DAG.getNode(ISD::FCEIL, SDLoc(N), VT, N0);
9874 
9875   return SDValue();
9876 }
9877 
9878 SDValue DAGCombiner::visitFTRUNC(SDNode *N) {
9879   SDValue N0 = N->getOperand(0);
9880   EVT VT = N->getValueType(0);
9881 
9882   // fold (ftrunc c1) -> ftrunc(c1)
9883   if (isConstantFPBuildVectorOrConstantFP(N0))
9884     return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0);
9885 
9886   return SDValue();
9887 }
9888 
9889 SDValue DAGCombiner::visitFFLOOR(SDNode *N) {
9890   SDValue N0 = N->getOperand(0);
9891   EVT VT = N->getValueType(0);
9892 
9893   // fold (ffloor c1) -> ffloor(c1)
9894   if (isConstantFPBuildVectorOrConstantFP(N0))
9895     return DAG.getNode(ISD::FFLOOR, SDLoc(N), VT, N0);
9896 
9897   return SDValue();
9898 }
9899 
9900 // FIXME: FNEG and FABS have a lot in common; refactor.
9901 SDValue DAGCombiner::visitFNEG(SDNode *N) {
9902   SDValue N0 = N->getOperand(0);
9903   EVT VT = N->getValueType(0);
9904 
9905   // Constant fold FNEG.
9906   if (isConstantFPBuildVectorOrConstantFP(N0))
9907     return DAG.getNode(ISD::FNEG, SDLoc(N), VT, N0);
9908 
9909   if (isNegatibleForFree(N0, LegalOperations, DAG.getTargetLoweringInfo(),
9910                          &DAG.getTarget().Options))
9911     return GetNegatedExpression(N0, DAG, LegalOperations);
9912 
9913   // Transform fneg(bitconvert(x)) -> bitconvert(x ^ sign) to avoid loading
9914   // constant pool values.
9915   if (!TLI.isFNegFree(VT) &&
9916       N0.getOpcode() == ISD::BITCAST &&
9917       N0.getNode()->hasOneUse()) {
9918     SDValue Int = N0.getOperand(0);
9919     EVT IntVT = Int.getValueType();
9920     if (IntVT.isInteger() && !IntVT.isVector()) {
9921       APInt SignMask;
9922       if (N0.getValueType().isVector()) {
9923         // For a vector, get a mask such as 0x80... per scalar element
9924         // and splat it.
9925         SignMask = APInt::getSignBit(N0.getScalarValueSizeInBits());
9926         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
9927       } else {
9928         // For a scalar, just generate 0x80...
9929         SignMask = APInt::getSignBit(IntVT.getSizeInBits());
9930       }
9931       SDLoc DL0(N0);
9932       Int = DAG.getNode(ISD::XOR, DL0, IntVT, Int,
9933                         DAG.getConstant(SignMask, DL0, IntVT));
9934       AddToWorklist(Int.getNode());
9935       return DAG.getBitcast(VT, Int);
9936     }
9937   }
9938 
9939   // (fneg (fmul c, x)) -> (fmul -c, x)
9940   if (N0.getOpcode() == ISD::FMUL &&
9941       (N0.getNode()->hasOneUse() || !TLI.isFNegFree(VT))) {
9942     ConstantFPSDNode *CFP1 = dyn_cast<ConstantFPSDNode>(N0.getOperand(1));
9943     if (CFP1) {
9944       APFloat CVal = CFP1->getValueAPF();
9945       CVal.changeSign();
9946       if (Level >= AfterLegalizeDAG &&
9947           (TLI.isFPImmLegal(CVal, VT) ||
9948            TLI.isOperationLegal(ISD::ConstantFP, VT)))
9949         return DAG.getNode(ISD::FMUL, SDLoc(N), VT, N0.getOperand(0),
9950                            DAG.getNode(ISD::FNEG, SDLoc(N), VT,
9951                                        N0.getOperand(1)),
9952                            &cast<BinaryWithFlagsSDNode>(N0)->Flags);
9953     }
9954   }
9955 
9956   return SDValue();
9957 }
9958 
9959 SDValue DAGCombiner::visitFMINNUM(SDNode *N) {
9960   SDValue N0 = N->getOperand(0);
9961   SDValue N1 = N->getOperand(1);
9962   EVT VT = N->getValueType(0);
9963   const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9964   const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9965 
9966   if (N0CFP && N1CFP) {
9967     const APFloat &C0 = N0CFP->getValueAPF();
9968     const APFloat &C1 = N1CFP->getValueAPF();
9969     return DAG.getConstantFP(minnum(C0, C1), SDLoc(N), VT);
9970   }
9971 
9972   // Canonicalize to constant on RHS.
9973   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9974      !isConstantFPBuildVectorOrConstantFP(N1))
9975     return DAG.getNode(ISD::FMINNUM, SDLoc(N), VT, N1, N0);
9976 
9977   return SDValue();
9978 }
9979 
9980 SDValue DAGCombiner::visitFMAXNUM(SDNode *N) {
9981   SDValue N0 = N->getOperand(0);
9982   SDValue N1 = N->getOperand(1);
9983   EVT VT = N->getValueType(0);
9984   const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0);
9985   const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1);
9986 
9987   if (N0CFP && N1CFP) {
9988     const APFloat &C0 = N0CFP->getValueAPF();
9989     const APFloat &C1 = N1CFP->getValueAPF();
9990     return DAG.getConstantFP(maxnum(C0, C1), SDLoc(N), VT);
9991   }
9992 
9993   // Canonicalize to constant on RHS.
9994   if (isConstantFPBuildVectorOrConstantFP(N0) &&
9995      !isConstantFPBuildVectorOrConstantFP(N1))
9996     return DAG.getNode(ISD::FMAXNUM, SDLoc(N), VT, N1, N0);
9997 
9998   return SDValue();
9999 }
10000 
10001 SDValue DAGCombiner::visitFABS(SDNode *N) {
10002   SDValue N0 = N->getOperand(0);
10003   EVT VT = N->getValueType(0);
10004 
10005   // fold (fabs c1) -> fabs(c1)
10006   if (isConstantFPBuildVectorOrConstantFP(N0))
10007     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0);
10008 
10009   // fold (fabs (fabs x)) -> (fabs x)
10010   if (N0.getOpcode() == ISD::FABS)
10011     return N->getOperand(0);
10012 
10013   // fold (fabs (fneg x)) -> (fabs x)
10014   // fold (fabs (fcopysign x, y)) -> (fabs x)
10015   if (N0.getOpcode() == ISD::FNEG || N0.getOpcode() == ISD::FCOPYSIGN)
10016     return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0.getOperand(0));
10017 
10018   // Transform fabs(bitconvert(x)) -> bitconvert(x & ~sign) to avoid loading
10019   // constant pool values.
10020   if (!TLI.isFAbsFree(VT) &&
10021       N0.getOpcode() == ISD::BITCAST &&
10022       N0.getNode()->hasOneUse()) {
10023     SDValue Int = N0.getOperand(0);
10024     EVT IntVT = Int.getValueType();
10025     if (IntVT.isInteger() && !IntVT.isVector()) {
10026       APInt SignMask;
10027       if (N0.getValueType().isVector()) {
10028         // For a vector, get a mask such as 0x7f... per scalar element
10029         // and splat it.
10030         SignMask = ~APInt::getSignBit(N0.getScalarValueSizeInBits());
10031         SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask);
10032       } else {
10033         // For a scalar, just generate 0x7f...
10034         SignMask = ~APInt::getSignBit(IntVT.getSizeInBits());
10035       }
10036       SDLoc DL(N0);
10037       Int = DAG.getNode(ISD::AND, DL, IntVT, Int,
10038                         DAG.getConstant(SignMask, DL, IntVT));
10039       AddToWorklist(Int.getNode());
10040       return DAG.getBitcast(N->getValueType(0), Int);
10041     }
10042   }
10043 
10044   return SDValue();
10045 }
10046 
10047 SDValue DAGCombiner::visitBRCOND(SDNode *N) {
10048   SDValue Chain = N->getOperand(0);
10049   SDValue N1 = N->getOperand(1);
10050   SDValue N2 = N->getOperand(2);
10051 
10052   // If N is a constant we could fold this into a fallthrough or unconditional
10053   // branch. However that doesn't happen very often in normal code, because
10054   // Instcombine/SimplifyCFG should have handled the available opportunities.
10055   // If we did this folding here, it would be necessary to update the
10056   // MachineBasicBlock CFG, which is awkward.
10057 
10058   // fold a brcond with a setcc condition into a BR_CC node if BR_CC is legal
10059   // on the target.
10060   if (N1.getOpcode() == ISD::SETCC &&
10061       TLI.isOperationLegalOrCustom(ISD::BR_CC,
10062                                    N1.getOperand(0).getValueType())) {
10063     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
10064                        Chain, N1.getOperand(2),
10065                        N1.getOperand(0), N1.getOperand(1), N2);
10066   }
10067 
10068   if ((N1.hasOneUse() && N1.getOpcode() == ISD::SRL) ||
10069       ((N1.getOpcode() == ISD::TRUNCATE && N1.hasOneUse()) &&
10070        (N1.getOperand(0).hasOneUse() &&
10071         N1.getOperand(0).getOpcode() == ISD::SRL))) {
10072     SDNode *Trunc = nullptr;
10073     if (N1.getOpcode() == ISD::TRUNCATE) {
10074       // Look pass the truncate.
10075       Trunc = N1.getNode();
10076       N1 = N1.getOperand(0);
10077     }
10078 
10079     // Match this pattern so that we can generate simpler code:
10080     //
10081     //   %a = ...
10082     //   %b = and i32 %a, 2
10083     //   %c = srl i32 %b, 1
10084     //   brcond i32 %c ...
10085     //
10086     // into
10087     //
10088     //   %a = ...
10089     //   %b = and i32 %a, 2
10090     //   %c = setcc eq %b, 0
10091     //   brcond %c ...
10092     //
10093     // This applies only when the AND constant value has one bit set and the
10094     // SRL constant is equal to the log2 of the AND constant. The back-end is
10095     // smart enough to convert the result into a TEST/JMP sequence.
10096     SDValue Op0 = N1.getOperand(0);
10097     SDValue Op1 = N1.getOperand(1);
10098 
10099     if (Op0.getOpcode() == ISD::AND &&
10100         Op1.getOpcode() == ISD::Constant) {
10101       SDValue AndOp1 = Op0.getOperand(1);
10102 
10103       if (AndOp1.getOpcode() == ISD::Constant) {
10104         const APInt &AndConst = cast<ConstantSDNode>(AndOp1)->getAPIntValue();
10105 
10106         if (AndConst.isPowerOf2() &&
10107             cast<ConstantSDNode>(Op1)->getAPIntValue()==AndConst.logBase2()) {
10108           SDLoc DL(N);
10109           SDValue SetCC =
10110             DAG.getSetCC(DL,
10111                          getSetCCResultType(Op0.getValueType()),
10112                          Op0, DAG.getConstant(0, DL, Op0.getValueType()),
10113                          ISD::SETNE);
10114 
10115           SDValue NewBRCond = DAG.getNode(ISD::BRCOND, DL,
10116                                           MVT::Other, Chain, SetCC, N2);
10117           // Don't add the new BRCond into the worklist or else SimplifySelectCC
10118           // will convert it back to (X & C1) >> C2.
10119           CombineTo(N, NewBRCond, false);
10120           // Truncate is dead.
10121           if (Trunc)
10122             deleteAndRecombine(Trunc);
10123           // Replace the uses of SRL with SETCC
10124           WorklistRemover DeadNodes(*this);
10125           DAG.ReplaceAllUsesOfValueWith(N1, SetCC);
10126           deleteAndRecombine(N1.getNode());
10127           return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10128         }
10129       }
10130     }
10131 
10132     if (Trunc)
10133       // Restore N1 if the above transformation doesn't match.
10134       N1 = N->getOperand(1);
10135   }
10136 
10137   // Transform br(xor(x, y)) -> br(x != y)
10138   // Transform br(xor(xor(x,y), 1)) -> br (x == y)
10139   if (N1.hasOneUse() && N1.getOpcode() == ISD::XOR) {
10140     SDNode *TheXor = N1.getNode();
10141     SDValue Op0 = TheXor->getOperand(0);
10142     SDValue Op1 = TheXor->getOperand(1);
10143     if (Op0.getOpcode() == Op1.getOpcode()) {
10144       // Avoid missing important xor optimizations.
10145       if (SDValue Tmp = visitXOR(TheXor)) {
10146         if (Tmp.getNode() != TheXor) {
10147           DEBUG(dbgs() << "\nReplacing.8 ";
10148                 TheXor->dump(&DAG);
10149                 dbgs() << "\nWith: ";
10150                 Tmp.getNode()->dump(&DAG);
10151                 dbgs() << '\n');
10152           WorklistRemover DeadNodes(*this);
10153           DAG.ReplaceAllUsesOfValueWith(N1, Tmp);
10154           deleteAndRecombine(TheXor);
10155           return DAG.getNode(ISD::BRCOND, SDLoc(N),
10156                              MVT::Other, Chain, Tmp, N2);
10157         }
10158 
10159         // visitXOR has changed XOR's operands or replaced the XOR completely,
10160         // bail out.
10161         return SDValue(N, 0);
10162       }
10163     }
10164 
10165     if (Op0.getOpcode() != ISD::SETCC && Op1.getOpcode() != ISD::SETCC) {
10166       bool Equal = false;
10167       if (isOneConstant(Op0) && Op0.hasOneUse() &&
10168           Op0.getOpcode() == ISD::XOR) {
10169         TheXor = Op0.getNode();
10170         Equal = true;
10171       }
10172 
10173       EVT SetCCVT = N1.getValueType();
10174       if (LegalTypes)
10175         SetCCVT = getSetCCResultType(SetCCVT);
10176       SDValue SetCC = DAG.getSetCC(SDLoc(TheXor),
10177                                    SetCCVT,
10178                                    Op0, Op1,
10179                                    Equal ? ISD::SETEQ : ISD::SETNE);
10180       // Replace the uses of XOR with SETCC
10181       WorklistRemover DeadNodes(*this);
10182       DAG.ReplaceAllUsesOfValueWith(N1, SetCC);
10183       deleteAndRecombine(N1.getNode());
10184       return DAG.getNode(ISD::BRCOND, SDLoc(N),
10185                          MVT::Other, Chain, SetCC, N2);
10186     }
10187   }
10188 
10189   return SDValue();
10190 }
10191 
10192 // Operand List for BR_CC: Chain, CondCC, CondLHS, CondRHS, DestBB.
10193 //
10194 SDValue DAGCombiner::visitBR_CC(SDNode *N) {
10195   CondCodeSDNode *CC = cast<CondCodeSDNode>(N->getOperand(1));
10196   SDValue CondLHS = N->getOperand(2), CondRHS = N->getOperand(3);
10197 
10198   // If N is a constant we could fold this into a fallthrough or unconditional
10199   // branch. However that doesn't happen very often in normal code, because
10200   // Instcombine/SimplifyCFG should have handled the available opportunities.
10201   // If we did this folding here, it would be necessary to update the
10202   // MachineBasicBlock CFG, which is awkward.
10203 
10204   // Use SimplifySetCC to simplify SETCC's.
10205   SDValue Simp = SimplifySetCC(getSetCCResultType(CondLHS.getValueType()),
10206                                CondLHS, CondRHS, CC->get(), SDLoc(N),
10207                                false);
10208   if (Simp.getNode()) AddToWorklist(Simp.getNode());
10209 
10210   // fold to a simpler setcc
10211   if (Simp.getNode() && Simp.getOpcode() == ISD::SETCC)
10212     return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other,
10213                        N->getOperand(0), Simp.getOperand(2),
10214                        Simp.getOperand(0), Simp.getOperand(1),
10215                        N->getOperand(4));
10216 
10217   return SDValue();
10218 }
10219 
10220 /// Return true if 'Use' is a load or a store that uses N as its base pointer
10221 /// and that N may be folded in the load / store addressing mode.
10222 static bool canFoldInAddressingMode(SDNode *N, SDNode *Use,
10223                                     SelectionDAG &DAG,
10224                                     const TargetLowering &TLI) {
10225   EVT VT;
10226   unsigned AS;
10227 
10228   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(Use)) {
10229     if (LD->isIndexed() || LD->getBasePtr().getNode() != N)
10230       return false;
10231     VT = LD->getMemoryVT();
10232     AS = LD->getAddressSpace();
10233   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(Use)) {
10234     if (ST->isIndexed() || ST->getBasePtr().getNode() != N)
10235       return false;
10236     VT = ST->getMemoryVT();
10237     AS = ST->getAddressSpace();
10238   } else
10239     return false;
10240 
10241   TargetLowering::AddrMode AM;
10242   if (N->getOpcode() == ISD::ADD) {
10243     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
10244     if (Offset)
10245       // [reg +/- imm]
10246       AM.BaseOffs = Offset->getSExtValue();
10247     else
10248       // [reg +/- reg]
10249       AM.Scale = 1;
10250   } else if (N->getOpcode() == ISD::SUB) {
10251     ConstantSDNode *Offset = dyn_cast<ConstantSDNode>(N->getOperand(1));
10252     if (Offset)
10253       // [reg +/- imm]
10254       AM.BaseOffs = -Offset->getSExtValue();
10255     else
10256       // [reg +/- reg]
10257       AM.Scale = 1;
10258   } else
10259     return false;
10260 
10261   return TLI.isLegalAddressingMode(DAG.getDataLayout(), AM,
10262                                    VT.getTypeForEVT(*DAG.getContext()), AS);
10263 }
10264 
10265 /// Try turning a load/store into a pre-indexed load/store when the base
10266 /// pointer is an add or subtract and it has other uses besides the load/store.
10267 /// After the transformation, the new indexed load/store has effectively folded
10268 /// the add/subtract in and all of its other uses are redirected to the
10269 /// new load/store.
10270 bool DAGCombiner::CombineToPreIndexedLoadStore(SDNode *N) {
10271   if (Level < AfterLegalizeDAG)
10272     return false;
10273 
10274   bool isLoad = true;
10275   SDValue Ptr;
10276   EVT VT;
10277   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
10278     if (LD->isIndexed())
10279       return false;
10280     VT = LD->getMemoryVT();
10281     if (!TLI.isIndexedLoadLegal(ISD::PRE_INC, VT) &&
10282         !TLI.isIndexedLoadLegal(ISD::PRE_DEC, VT))
10283       return false;
10284     Ptr = LD->getBasePtr();
10285   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
10286     if (ST->isIndexed())
10287       return false;
10288     VT = ST->getMemoryVT();
10289     if (!TLI.isIndexedStoreLegal(ISD::PRE_INC, VT) &&
10290         !TLI.isIndexedStoreLegal(ISD::PRE_DEC, VT))
10291       return false;
10292     Ptr = ST->getBasePtr();
10293     isLoad = false;
10294   } else {
10295     return false;
10296   }
10297 
10298   // If the pointer is not an add/sub, or if it doesn't have multiple uses, bail
10299   // out.  There is no reason to make this a preinc/predec.
10300   if ((Ptr.getOpcode() != ISD::ADD && Ptr.getOpcode() != ISD::SUB) ||
10301       Ptr.getNode()->hasOneUse())
10302     return false;
10303 
10304   // Ask the target to do addressing mode selection.
10305   SDValue BasePtr;
10306   SDValue Offset;
10307   ISD::MemIndexedMode AM = ISD::UNINDEXED;
10308   if (!TLI.getPreIndexedAddressParts(N, BasePtr, Offset, AM, DAG))
10309     return false;
10310 
10311   // Backends without true r+i pre-indexed forms may need to pass a
10312   // constant base with a variable offset so that constant coercion
10313   // will work with the patterns in canonical form.
10314   bool Swapped = false;
10315   if (isa<ConstantSDNode>(BasePtr)) {
10316     std::swap(BasePtr, Offset);
10317     Swapped = true;
10318   }
10319 
10320   // Don't create a indexed load / store with zero offset.
10321   if (isNullConstant(Offset))
10322     return false;
10323 
10324   // Try turning it into a pre-indexed load / store except when:
10325   // 1) The new base ptr is a frame index.
10326   // 2) If N is a store and the new base ptr is either the same as or is a
10327   //    predecessor of the value being stored.
10328   // 3) Another use of old base ptr is a predecessor of N. If ptr is folded
10329   //    that would create a cycle.
10330   // 4) All uses are load / store ops that use it as old base ptr.
10331 
10332   // Check #1.  Preinc'ing a frame index would require copying the stack pointer
10333   // (plus the implicit offset) to a register to preinc anyway.
10334   if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
10335     return false;
10336 
10337   // Check #2.
10338   if (!isLoad) {
10339     SDValue Val = cast<StoreSDNode>(N)->getValue();
10340     if (Val == BasePtr || BasePtr.getNode()->isPredecessorOf(Val.getNode()))
10341       return false;
10342   }
10343 
10344   // Caches for hasPredecessorHelper.
10345   SmallPtrSet<const SDNode *, 32> Visited;
10346   SmallVector<const SDNode *, 16> Worklist;
10347   Worklist.push_back(N);
10348 
10349   // If the offset is a constant, there may be other adds of constants that
10350   // can be folded with this one. We should do this to avoid having to keep
10351   // a copy of the original base pointer.
10352   SmallVector<SDNode *, 16> OtherUses;
10353   if (isa<ConstantSDNode>(Offset))
10354     for (SDNode::use_iterator UI = BasePtr.getNode()->use_begin(),
10355                               UE = BasePtr.getNode()->use_end();
10356          UI != UE; ++UI) {
10357       SDUse &Use = UI.getUse();
10358       // Skip the use that is Ptr and uses of other results from BasePtr's
10359       // node (important for nodes that return multiple results).
10360       if (Use.getUser() == Ptr.getNode() || Use != BasePtr)
10361         continue;
10362 
10363       if (SDNode::hasPredecessorHelper(Use.getUser(), Visited, Worklist))
10364         continue;
10365 
10366       if (Use.getUser()->getOpcode() != ISD::ADD &&
10367           Use.getUser()->getOpcode() != ISD::SUB) {
10368         OtherUses.clear();
10369         break;
10370       }
10371 
10372       SDValue Op1 = Use.getUser()->getOperand((UI.getOperandNo() + 1) & 1);
10373       if (!isa<ConstantSDNode>(Op1)) {
10374         OtherUses.clear();
10375         break;
10376       }
10377 
10378       // FIXME: In some cases, we can be smarter about this.
10379       if (Op1.getValueType() != Offset.getValueType()) {
10380         OtherUses.clear();
10381         break;
10382       }
10383 
10384       OtherUses.push_back(Use.getUser());
10385     }
10386 
10387   if (Swapped)
10388     std::swap(BasePtr, Offset);
10389 
10390   // Now check for #3 and #4.
10391   bool RealUse = false;
10392 
10393   for (SDNode *Use : Ptr.getNode()->uses()) {
10394     if (Use == N)
10395       continue;
10396     if (SDNode::hasPredecessorHelper(Use, Visited, Worklist))
10397       return false;
10398 
10399     // If Ptr may be folded in addressing mode of other use, then it's
10400     // not profitable to do this transformation.
10401     if (!canFoldInAddressingMode(Ptr.getNode(), Use, DAG, TLI))
10402       RealUse = true;
10403   }
10404 
10405   if (!RealUse)
10406     return false;
10407 
10408   SDValue Result;
10409   if (isLoad)
10410     Result = DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
10411                                 BasePtr, Offset, AM);
10412   else
10413     Result = DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
10414                                  BasePtr, Offset, AM);
10415   ++PreIndexedNodes;
10416   ++NodesCombined;
10417   DEBUG(dbgs() << "\nReplacing.4 ";
10418         N->dump(&DAG);
10419         dbgs() << "\nWith: ";
10420         Result.getNode()->dump(&DAG);
10421         dbgs() << '\n');
10422   WorklistRemover DeadNodes(*this);
10423   if (isLoad) {
10424     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
10425     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
10426   } else {
10427     DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
10428   }
10429 
10430   // Finally, since the node is now dead, remove it from the graph.
10431   deleteAndRecombine(N);
10432 
10433   if (Swapped)
10434     std::swap(BasePtr, Offset);
10435 
10436   // Replace other uses of BasePtr that can be updated to use Ptr
10437   for (unsigned i = 0, e = OtherUses.size(); i != e; ++i) {
10438     unsigned OffsetIdx = 1;
10439     if (OtherUses[i]->getOperand(OffsetIdx).getNode() == BasePtr.getNode())
10440       OffsetIdx = 0;
10441     assert(OtherUses[i]->getOperand(!OffsetIdx).getNode() ==
10442            BasePtr.getNode() && "Expected BasePtr operand");
10443 
10444     // We need to replace ptr0 in the following expression:
10445     //   x0 * offset0 + y0 * ptr0 = t0
10446     // knowing that
10447     //   x1 * offset1 + y1 * ptr0 = t1 (the indexed load/store)
10448     //
10449     // where x0, x1, y0 and y1 in {-1, 1} are given by the types of the
10450     // indexed load/store and the expresion that needs to be re-written.
10451     //
10452     // Therefore, we have:
10453     //   t0 = (x0 * offset0 - x1 * y0 * y1 *offset1) + (y0 * y1) * t1
10454 
10455     ConstantSDNode *CN =
10456       cast<ConstantSDNode>(OtherUses[i]->getOperand(OffsetIdx));
10457     int X0, X1, Y0, Y1;
10458     const APInt &Offset0 = CN->getAPIntValue();
10459     APInt Offset1 = cast<ConstantSDNode>(Offset)->getAPIntValue();
10460 
10461     X0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 1) ? -1 : 1;
10462     Y0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 0) ? -1 : 1;
10463     X1 = (AM == ISD::PRE_DEC && !Swapped) ? -1 : 1;
10464     Y1 = (AM == ISD::PRE_DEC && Swapped) ? -1 : 1;
10465 
10466     unsigned Opcode = (Y0 * Y1 < 0) ? ISD::SUB : ISD::ADD;
10467 
10468     APInt CNV = Offset0;
10469     if (X0 < 0) CNV = -CNV;
10470     if (X1 * Y0 * Y1 < 0) CNV = CNV + Offset1;
10471     else CNV = CNV - Offset1;
10472 
10473     SDLoc DL(OtherUses[i]);
10474 
10475     // We can now generate the new expression.
10476     SDValue NewOp1 = DAG.getConstant(CNV, DL, CN->getValueType(0));
10477     SDValue NewOp2 = Result.getValue(isLoad ? 1 : 0);
10478 
10479     SDValue NewUse = DAG.getNode(Opcode,
10480                                  DL,
10481                                  OtherUses[i]->getValueType(0), NewOp1, NewOp2);
10482     DAG.ReplaceAllUsesOfValueWith(SDValue(OtherUses[i], 0), NewUse);
10483     deleteAndRecombine(OtherUses[i]);
10484   }
10485 
10486   // Replace the uses of Ptr with uses of the updated base value.
10487   DAG.ReplaceAllUsesOfValueWith(Ptr, Result.getValue(isLoad ? 1 : 0));
10488   deleteAndRecombine(Ptr.getNode());
10489 
10490   return true;
10491 }
10492 
10493 /// Try to combine a load/store with a add/sub of the base pointer node into a
10494 /// post-indexed load/store. The transformation folded the add/subtract into the
10495 /// new indexed load/store effectively and all of its uses are redirected to the
10496 /// new load/store.
10497 bool DAGCombiner::CombineToPostIndexedLoadStore(SDNode *N) {
10498   if (Level < AfterLegalizeDAG)
10499     return false;
10500 
10501   bool isLoad = true;
10502   SDValue Ptr;
10503   EVT VT;
10504   if (LoadSDNode *LD  = dyn_cast<LoadSDNode>(N)) {
10505     if (LD->isIndexed())
10506       return false;
10507     VT = LD->getMemoryVT();
10508     if (!TLI.isIndexedLoadLegal(ISD::POST_INC, VT) &&
10509         !TLI.isIndexedLoadLegal(ISD::POST_DEC, VT))
10510       return false;
10511     Ptr = LD->getBasePtr();
10512   } else if (StoreSDNode *ST  = dyn_cast<StoreSDNode>(N)) {
10513     if (ST->isIndexed())
10514       return false;
10515     VT = ST->getMemoryVT();
10516     if (!TLI.isIndexedStoreLegal(ISD::POST_INC, VT) &&
10517         !TLI.isIndexedStoreLegal(ISD::POST_DEC, VT))
10518       return false;
10519     Ptr = ST->getBasePtr();
10520     isLoad = false;
10521   } else {
10522     return false;
10523   }
10524 
10525   if (Ptr.getNode()->hasOneUse())
10526     return false;
10527 
10528   for (SDNode *Op : Ptr.getNode()->uses()) {
10529     if (Op == N ||
10530         (Op->getOpcode() != ISD::ADD && Op->getOpcode() != ISD::SUB))
10531       continue;
10532 
10533     SDValue BasePtr;
10534     SDValue Offset;
10535     ISD::MemIndexedMode AM = ISD::UNINDEXED;
10536     if (TLI.getPostIndexedAddressParts(N, Op, BasePtr, Offset, AM, DAG)) {
10537       // Don't create a indexed load / store with zero offset.
10538       if (isNullConstant(Offset))
10539         continue;
10540 
10541       // Try turning it into a post-indexed load / store except when
10542       // 1) All uses are load / store ops that use it as base ptr (and
10543       //    it may be folded as addressing mmode).
10544       // 2) Op must be independent of N, i.e. Op is neither a predecessor
10545       //    nor a successor of N. Otherwise, if Op is folded that would
10546       //    create a cycle.
10547 
10548       if (isa<FrameIndexSDNode>(BasePtr) || isa<RegisterSDNode>(BasePtr))
10549         continue;
10550 
10551       // Check for #1.
10552       bool TryNext = false;
10553       for (SDNode *Use : BasePtr.getNode()->uses()) {
10554         if (Use == Ptr.getNode())
10555           continue;
10556 
10557         // If all the uses are load / store addresses, then don't do the
10558         // transformation.
10559         if (Use->getOpcode() == ISD::ADD || Use->getOpcode() == ISD::SUB){
10560           bool RealUse = false;
10561           for (SDNode *UseUse : Use->uses()) {
10562             if (!canFoldInAddressingMode(Use, UseUse, DAG, TLI))
10563               RealUse = true;
10564           }
10565 
10566           if (!RealUse) {
10567             TryNext = true;
10568             break;
10569           }
10570         }
10571       }
10572 
10573       if (TryNext)
10574         continue;
10575 
10576       // Check for #2
10577       if (!Op->isPredecessorOf(N) && !N->isPredecessorOf(Op)) {
10578         SDValue Result = isLoad
10579           ? DAG.getIndexedLoad(SDValue(N,0), SDLoc(N),
10580                                BasePtr, Offset, AM)
10581           : DAG.getIndexedStore(SDValue(N,0), SDLoc(N),
10582                                 BasePtr, Offset, AM);
10583         ++PostIndexedNodes;
10584         ++NodesCombined;
10585         DEBUG(dbgs() << "\nReplacing.5 ";
10586               N->dump(&DAG);
10587               dbgs() << "\nWith: ";
10588               Result.getNode()->dump(&DAG);
10589               dbgs() << '\n');
10590         WorklistRemover DeadNodes(*this);
10591         if (isLoad) {
10592           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0));
10593           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2));
10594         } else {
10595           DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1));
10596         }
10597 
10598         // Finally, since the node is now dead, remove it from the graph.
10599         deleteAndRecombine(N);
10600 
10601         // Replace the uses of Use with uses of the updated base value.
10602         DAG.ReplaceAllUsesOfValueWith(SDValue(Op, 0),
10603                                       Result.getValue(isLoad ? 1 : 0));
10604         deleteAndRecombine(Op);
10605         return true;
10606       }
10607     }
10608   }
10609 
10610   return false;
10611 }
10612 
10613 /// \brief Return the base-pointer arithmetic from an indexed \p LD.
10614 SDValue DAGCombiner::SplitIndexingFromLoad(LoadSDNode *LD) {
10615   ISD::MemIndexedMode AM = LD->getAddressingMode();
10616   assert(AM != ISD::UNINDEXED);
10617   SDValue BP = LD->getOperand(1);
10618   SDValue Inc = LD->getOperand(2);
10619 
10620   // Some backends use TargetConstants for load offsets, but don't expect
10621   // TargetConstants in general ADD nodes. We can convert these constants into
10622   // regular Constants (if the constant is not opaque).
10623   assert((Inc.getOpcode() != ISD::TargetConstant ||
10624           !cast<ConstantSDNode>(Inc)->isOpaque()) &&
10625          "Cannot split out indexing using opaque target constants");
10626   if (Inc.getOpcode() == ISD::TargetConstant) {
10627     ConstantSDNode *ConstInc = cast<ConstantSDNode>(Inc);
10628     Inc = DAG.getConstant(*ConstInc->getConstantIntValue(), SDLoc(Inc),
10629                           ConstInc->getValueType(0));
10630   }
10631 
10632   unsigned Opc =
10633       (AM == ISD::PRE_INC || AM == ISD::POST_INC ? ISD::ADD : ISD::SUB);
10634   return DAG.getNode(Opc, SDLoc(LD), BP.getSimpleValueType(), BP, Inc);
10635 }
10636 
10637 SDValue DAGCombiner::visitLOAD(SDNode *N) {
10638   LoadSDNode *LD  = cast<LoadSDNode>(N);
10639   SDValue Chain = LD->getChain();
10640   SDValue Ptr   = LD->getBasePtr();
10641 
10642   // If load is not volatile and there are no uses of the loaded value (and
10643   // the updated indexed value in case of indexed loads), change uses of the
10644   // chain value into uses of the chain input (i.e. delete the dead load).
10645   if (!LD->isVolatile()) {
10646     if (N->getValueType(1) == MVT::Other) {
10647       // Unindexed loads.
10648       if (!N->hasAnyUseOfValue(0)) {
10649         // It's not safe to use the two value CombineTo variant here. e.g.
10650         // v1, chain2 = load chain1, loc
10651         // v2, chain3 = load chain2, loc
10652         // v3         = add v2, c
10653         // Now we replace use of chain2 with chain1.  This makes the second load
10654         // isomorphic to the one we are deleting, and thus makes this load live.
10655         DEBUG(dbgs() << "\nReplacing.6 ";
10656               N->dump(&DAG);
10657               dbgs() << "\nWith chain: ";
10658               Chain.getNode()->dump(&DAG);
10659               dbgs() << "\n");
10660         WorklistRemover DeadNodes(*this);
10661         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
10662 
10663         if (N->use_empty())
10664           deleteAndRecombine(N);
10665 
10666         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10667       }
10668     } else {
10669       // Indexed loads.
10670       assert(N->getValueType(2) == MVT::Other && "Malformed indexed loads?");
10671 
10672       // If this load has an opaque TargetConstant offset, then we cannot split
10673       // the indexing into an add/sub directly (that TargetConstant may not be
10674       // valid for a different type of node, and we cannot convert an opaque
10675       // target constant into a regular constant).
10676       bool HasOTCInc = LD->getOperand(2).getOpcode() == ISD::TargetConstant &&
10677                        cast<ConstantSDNode>(LD->getOperand(2))->isOpaque();
10678 
10679       if (!N->hasAnyUseOfValue(0) &&
10680           ((MaySplitLoadIndex && !HasOTCInc) || !N->hasAnyUseOfValue(1))) {
10681         SDValue Undef = DAG.getUNDEF(N->getValueType(0));
10682         SDValue Index;
10683         if (N->hasAnyUseOfValue(1) && MaySplitLoadIndex && !HasOTCInc) {
10684           Index = SplitIndexingFromLoad(LD);
10685           // Try to fold the base pointer arithmetic into subsequent loads and
10686           // stores.
10687           AddUsersToWorklist(N);
10688         } else
10689           Index = DAG.getUNDEF(N->getValueType(1));
10690         DEBUG(dbgs() << "\nReplacing.7 ";
10691               N->dump(&DAG);
10692               dbgs() << "\nWith: ";
10693               Undef.getNode()->dump(&DAG);
10694               dbgs() << " and 2 other values\n");
10695         WorklistRemover DeadNodes(*this);
10696         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Undef);
10697         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Index);
10698         DAG.ReplaceAllUsesOfValueWith(SDValue(N, 2), Chain);
10699         deleteAndRecombine(N);
10700         return SDValue(N, 0);   // Return N so it doesn't get rechecked!
10701       }
10702     }
10703   }
10704 
10705   // If this load is directly stored, replace the load value with the stored
10706   // value.
10707   // TODO: Handle store large -> read small portion.
10708   // TODO: Handle TRUNCSTORE/LOADEXT
10709   if (OptLevel != CodeGenOpt::None &&
10710       ISD::isNormalLoad(N) && !LD->isVolatile()) {
10711     if (ISD::isNON_TRUNCStore(Chain.getNode())) {
10712       StoreSDNode *PrevST = cast<StoreSDNode>(Chain);
10713       if (PrevST->getBasePtr() == Ptr &&
10714           PrevST->getValue().getValueType() == N->getValueType(0))
10715       return CombineTo(N, Chain.getOperand(1), Chain);
10716     }
10717   }
10718 
10719   // Try to infer better alignment information than the load already has.
10720   if (OptLevel != CodeGenOpt::None && LD->isUnindexed()) {
10721     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
10722       if (Align > LD->getMemOperand()->getBaseAlignment()) {
10723         SDValue NewLoad = DAG.getExtLoad(
10724             LD->getExtensionType(), SDLoc(N), LD->getValueType(0), Chain, Ptr,
10725             LD->getPointerInfo(), LD->getMemoryVT(), Align,
10726             LD->getMemOperand()->getFlags(), LD->getAAInfo());
10727         if (NewLoad.getNode() != N)
10728           return CombineTo(N, NewLoad, SDValue(NewLoad.getNode(), 1), true);
10729       }
10730     }
10731   }
10732 
10733   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
10734                                                   : DAG.getSubtarget().useAA();
10735 #ifndef NDEBUG
10736   if (CombinerAAOnlyFunc.getNumOccurrences() &&
10737       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
10738     UseAA = false;
10739 #endif
10740   if (UseAA && LD->isUnindexed()) {
10741     // Walk up chain skipping non-aliasing memory nodes.
10742     SDValue BetterChain = FindBetterChain(N, Chain);
10743 
10744     // If there is a better chain.
10745     if (Chain != BetterChain) {
10746       SDValue ReplLoad;
10747 
10748       // Replace the chain to void dependency.
10749       if (LD->getExtensionType() == ISD::NON_EXTLOAD) {
10750         ReplLoad = DAG.getLoad(N->getValueType(0), SDLoc(LD),
10751                                BetterChain, Ptr, LD->getMemOperand());
10752       } else {
10753         ReplLoad = DAG.getExtLoad(LD->getExtensionType(), SDLoc(LD),
10754                                   LD->getValueType(0),
10755                                   BetterChain, Ptr, LD->getMemoryVT(),
10756                                   LD->getMemOperand());
10757       }
10758 
10759       // Create token factor to keep old chain connected.
10760       SDValue Token = DAG.getNode(ISD::TokenFactor, SDLoc(N),
10761                                   MVT::Other, Chain, ReplLoad.getValue(1));
10762 
10763       // Make sure the new and old chains are cleaned up.
10764       AddToWorklist(Token.getNode());
10765 
10766       // Replace uses with load result and token factor. Don't add users
10767       // to work list.
10768       return CombineTo(N, ReplLoad.getValue(0), Token, false);
10769     }
10770   }
10771 
10772   // Try transforming N to an indexed load.
10773   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
10774     return SDValue(N, 0);
10775 
10776   // Try to slice up N to more direct loads if the slices are mapped to
10777   // different register banks or pairing can take place.
10778   if (SliceUpLoad(N))
10779     return SDValue(N, 0);
10780 
10781   return SDValue();
10782 }
10783 
10784 namespace {
10785 /// \brief Helper structure used to slice a load in smaller loads.
10786 /// Basically a slice is obtained from the following sequence:
10787 /// Origin = load Ty1, Base
10788 /// Shift = srl Ty1 Origin, CstTy Amount
10789 /// Inst = trunc Shift to Ty2
10790 ///
10791 /// Then, it will be rewriten into:
10792 /// Slice = load SliceTy, Base + SliceOffset
10793 /// [Inst = zext Slice to Ty2], only if SliceTy <> Ty2
10794 ///
10795 /// SliceTy is deduced from the number of bits that are actually used to
10796 /// build Inst.
10797 struct LoadedSlice {
10798   /// \brief Helper structure used to compute the cost of a slice.
10799   struct Cost {
10800     /// Are we optimizing for code size.
10801     bool ForCodeSize;
10802     /// Various cost.
10803     unsigned Loads;
10804     unsigned Truncates;
10805     unsigned CrossRegisterBanksCopies;
10806     unsigned ZExts;
10807     unsigned Shift;
10808 
10809     Cost(bool ForCodeSize = false)
10810         : ForCodeSize(ForCodeSize), Loads(0), Truncates(0),
10811           CrossRegisterBanksCopies(0), ZExts(0), Shift(0) {}
10812 
10813     /// \brief Get the cost of one isolated slice.
10814     Cost(const LoadedSlice &LS, bool ForCodeSize = false)
10815         : ForCodeSize(ForCodeSize), Loads(1), Truncates(0),
10816           CrossRegisterBanksCopies(0), ZExts(0), Shift(0) {
10817       EVT TruncType = LS.Inst->getValueType(0);
10818       EVT LoadedType = LS.getLoadedType();
10819       if (TruncType != LoadedType &&
10820           !LS.DAG->getTargetLoweringInfo().isZExtFree(LoadedType, TruncType))
10821         ZExts = 1;
10822     }
10823 
10824     /// \brief Account for slicing gain in the current cost.
10825     /// Slicing provide a few gains like removing a shift or a
10826     /// truncate. This method allows to grow the cost of the original
10827     /// load with the gain from this slice.
10828     void addSliceGain(const LoadedSlice &LS) {
10829       // Each slice saves a truncate.
10830       const TargetLowering &TLI = LS.DAG->getTargetLoweringInfo();
10831       if (!TLI.isTruncateFree(LS.Inst->getOperand(0).getValueType(),
10832                               LS.Inst->getValueType(0)))
10833         ++Truncates;
10834       // If there is a shift amount, this slice gets rid of it.
10835       if (LS.Shift)
10836         ++Shift;
10837       // If this slice can merge a cross register bank copy, account for it.
10838       if (LS.canMergeExpensiveCrossRegisterBankCopy())
10839         ++CrossRegisterBanksCopies;
10840     }
10841 
10842     Cost &operator+=(const Cost &RHS) {
10843       Loads += RHS.Loads;
10844       Truncates += RHS.Truncates;
10845       CrossRegisterBanksCopies += RHS.CrossRegisterBanksCopies;
10846       ZExts += RHS.ZExts;
10847       Shift += RHS.Shift;
10848       return *this;
10849     }
10850 
10851     bool operator==(const Cost &RHS) const {
10852       return Loads == RHS.Loads && Truncates == RHS.Truncates &&
10853              CrossRegisterBanksCopies == RHS.CrossRegisterBanksCopies &&
10854              ZExts == RHS.ZExts && Shift == RHS.Shift;
10855     }
10856 
10857     bool operator!=(const Cost &RHS) const { return !(*this == RHS); }
10858 
10859     bool operator<(const Cost &RHS) const {
10860       // Assume cross register banks copies are as expensive as loads.
10861       // FIXME: Do we want some more target hooks?
10862       unsigned ExpensiveOpsLHS = Loads + CrossRegisterBanksCopies;
10863       unsigned ExpensiveOpsRHS = RHS.Loads + RHS.CrossRegisterBanksCopies;
10864       // Unless we are optimizing for code size, consider the
10865       // expensive operation first.
10866       if (!ForCodeSize && ExpensiveOpsLHS != ExpensiveOpsRHS)
10867         return ExpensiveOpsLHS < ExpensiveOpsRHS;
10868       return (Truncates + ZExts + Shift + ExpensiveOpsLHS) <
10869              (RHS.Truncates + RHS.ZExts + RHS.Shift + ExpensiveOpsRHS);
10870     }
10871 
10872     bool operator>(const Cost &RHS) const { return RHS < *this; }
10873 
10874     bool operator<=(const Cost &RHS) const { return !(RHS < *this); }
10875 
10876     bool operator>=(const Cost &RHS) const { return !(*this < RHS); }
10877   };
10878   // The last instruction that represent the slice. This should be a
10879   // truncate instruction.
10880   SDNode *Inst;
10881   // The original load instruction.
10882   LoadSDNode *Origin;
10883   // The right shift amount in bits from the original load.
10884   unsigned Shift;
10885   // The DAG from which Origin came from.
10886   // This is used to get some contextual information about legal types, etc.
10887   SelectionDAG *DAG;
10888 
10889   LoadedSlice(SDNode *Inst = nullptr, LoadSDNode *Origin = nullptr,
10890               unsigned Shift = 0, SelectionDAG *DAG = nullptr)
10891       : Inst(Inst), Origin(Origin), Shift(Shift), DAG(DAG) {}
10892 
10893   /// \brief Get the bits used in a chunk of bits \p BitWidth large.
10894   /// \return Result is \p BitWidth and has used bits set to 1 and
10895   ///         not used bits set to 0.
10896   APInt getUsedBits() const {
10897     // Reproduce the trunc(lshr) sequence:
10898     // - Start from the truncated value.
10899     // - Zero extend to the desired bit width.
10900     // - Shift left.
10901     assert(Origin && "No original load to compare against.");
10902     unsigned BitWidth = Origin->getValueSizeInBits(0);
10903     assert(Inst && "This slice is not bound to an instruction");
10904     assert(Inst->getValueSizeInBits(0) <= BitWidth &&
10905            "Extracted slice is bigger than the whole type!");
10906     APInt UsedBits(Inst->getValueSizeInBits(0), 0);
10907     UsedBits.setAllBits();
10908     UsedBits = UsedBits.zext(BitWidth);
10909     UsedBits <<= Shift;
10910     return UsedBits;
10911   }
10912 
10913   /// \brief Get the size of the slice to be loaded in bytes.
10914   unsigned getLoadedSize() const {
10915     unsigned SliceSize = getUsedBits().countPopulation();
10916     assert(!(SliceSize & 0x7) && "Size is not a multiple of a byte.");
10917     return SliceSize / 8;
10918   }
10919 
10920   /// \brief Get the type that will be loaded for this slice.
10921   /// Note: This may not be the final type for the slice.
10922   EVT getLoadedType() const {
10923     assert(DAG && "Missing context");
10924     LLVMContext &Ctxt = *DAG->getContext();
10925     return EVT::getIntegerVT(Ctxt, getLoadedSize() * 8);
10926   }
10927 
10928   /// \brief Get the alignment of the load used for this slice.
10929   unsigned getAlignment() const {
10930     unsigned Alignment = Origin->getAlignment();
10931     unsigned Offset = getOffsetFromBase();
10932     if (Offset != 0)
10933       Alignment = MinAlign(Alignment, Alignment + Offset);
10934     return Alignment;
10935   }
10936 
10937   /// \brief Check if this slice can be rewritten with legal operations.
10938   bool isLegal() const {
10939     // An invalid slice is not legal.
10940     if (!Origin || !Inst || !DAG)
10941       return false;
10942 
10943     // Offsets are for indexed load only, we do not handle that.
10944     if (!Origin->getOffset().isUndef())
10945       return false;
10946 
10947     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
10948 
10949     // Check that the type is legal.
10950     EVT SliceType = getLoadedType();
10951     if (!TLI.isTypeLegal(SliceType))
10952       return false;
10953 
10954     // Check that the load is legal for this type.
10955     if (!TLI.isOperationLegal(ISD::LOAD, SliceType))
10956       return false;
10957 
10958     // Check that the offset can be computed.
10959     // 1. Check its type.
10960     EVT PtrType = Origin->getBasePtr().getValueType();
10961     if (PtrType == MVT::Untyped || PtrType.isExtended())
10962       return false;
10963 
10964     // 2. Check that it fits in the immediate.
10965     if (!TLI.isLegalAddImmediate(getOffsetFromBase()))
10966       return false;
10967 
10968     // 3. Check that the computation is legal.
10969     if (!TLI.isOperationLegal(ISD::ADD, PtrType))
10970       return false;
10971 
10972     // Check that the zext is legal if it needs one.
10973     EVT TruncateType = Inst->getValueType(0);
10974     if (TruncateType != SliceType &&
10975         !TLI.isOperationLegal(ISD::ZERO_EXTEND, TruncateType))
10976       return false;
10977 
10978     return true;
10979   }
10980 
10981   /// \brief Get the offset in bytes of this slice in the original chunk of
10982   /// bits.
10983   /// \pre DAG != nullptr.
10984   uint64_t getOffsetFromBase() const {
10985     assert(DAG && "Missing context.");
10986     bool IsBigEndian = DAG->getDataLayout().isBigEndian();
10987     assert(!(Shift & 0x7) && "Shifts not aligned on Bytes are not supported.");
10988     uint64_t Offset = Shift / 8;
10989     unsigned TySizeInBytes = Origin->getValueSizeInBits(0) / 8;
10990     assert(!(Origin->getValueSizeInBits(0) & 0x7) &&
10991            "The size of the original loaded type is not a multiple of a"
10992            " byte.");
10993     // If Offset is bigger than TySizeInBytes, it means we are loading all
10994     // zeros. This should have been optimized before in the process.
10995     assert(TySizeInBytes > Offset &&
10996            "Invalid shift amount for given loaded size");
10997     if (IsBigEndian)
10998       Offset = TySizeInBytes - Offset - getLoadedSize();
10999     return Offset;
11000   }
11001 
11002   /// \brief Generate the sequence of instructions to load the slice
11003   /// represented by this object and redirect the uses of this slice to
11004   /// this new sequence of instructions.
11005   /// \pre this->Inst && this->Origin are valid Instructions and this
11006   /// object passed the legal check: LoadedSlice::isLegal returned true.
11007   /// \return The last instruction of the sequence used to load the slice.
11008   SDValue loadSlice() const {
11009     assert(Inst && Origin && "Unable to replace a non-existing slice.");
11010     const SDValue &OldBaseAddr = Origin->getBasePtr();
11011     SDValue BaseAddr = OldBaseAddr;
11012     // Get the offset in that chunk of bytes w.r.t. the endianness.
11013     int64_t Offset = static_cast<int64_t>(getOffsetFromBase());
11014     assert(Offset >= 0 && "Offset too big to fit in int64_t!");
11015     if (Offset) {
11016       // BaseAddr = BaseAddr + Offset.
11017       EVT ArithType = BaseAddr.getValueType();
11018       SDLoc DL(Origin);
11019       BaseAddr = DAG->getNode(ISD::ADD, DL, ArithType, BaseAddr,
11020                               DAG->getConstant(Offset, DL, ArithType));
11021     }
11022 
11023     // Create the type of the loaded slice according to its size.
11024     EVT SliceType = getLoadedType();
11025 
11026     // Create the load for the slice.
11027     SDValue LastInst =
11028         DAG->getLoad(SliceType, SDLoc(Origin), Origin->getChain(), BaseAddr,
11029                      Origin->getPointerInfo().getWithOffset(Offset),
11030                      getAlignment(), Origin->getMemOperand()->getFlags());
11031     // If the final type is not the same as the loaded type, this means that
11032     // we have to pad with zero. Create a zero extend for that.
11033     EVT FinalType = Inst->getValueType(0);
11034     if (SliceType != FinalType)
11035       LastInst =
11036           DAG->getNode(ISD::ZERO_EXTEND, SDLoc(LastInst), FinalType, LastInst);
11037     return LastInst;
11038   }
11039 
11040   /// \brief Check if this slice can be merged with an expensive cross register
11041   /// bank copy. E.g.,
11042   /// i = load i32
11043   /// f = bitcast i32 i to float
11044   bool canMergeExpensiveCrossRegisterBankCopy() const {
11045     if (!Inst || !Inst->hasOneUse())
11046       return false;
11047     SDNode *Use = *Inst->use_begin();
11048     if (Use->getOpcode() != ISD::BITCAST)
11049       return false;
11050     assert(DAG && "Missing context");
11051     const TargetLowering &TLI = DAG->getTargetLoweringInfo();
11052     EVT ResVT = Use->getValueType(0);
11053     const TargetRegisterClass *ResRC = TLI.getRegClassFor(ResVT.getSimpleVT());
11054     const TargetRegisterClass *ArgRC =
11055         TLI.getRegClassFor(Use->getOperand(0).getValueType().getSimpleVT());
11056     if (ArgRC == ResRC || !TLI.isOperationLegal(ISD::LOAD, ResVT))
11057       return false;
11058 
11059     // At this point, we know that we perform a cross-register-bank copy.
11060     // Check if it is expensive.
11061     const TargetRegisterInfo *TRI = DAG->getSubtarget().getRegisterInfo();
11062     // Assume bitcasts are cheap, unless both register classes do not
11063     // explicitly share a common sub class.
11064     if (!TRI || TRI->getCommonSubClass(ArgRC, ResRC))
11065       return false;
11066 
11067     // Check if it will be merged with the load.
11068     // 1. Check the alignment constraint.
11069     unsigned RequiredAlignment = DAG->getDataLayout().getABITypeAlignment(
11070         ResVT.getTypeForEVT(*DAG->getContext()));
11071 
11072     if (RequiredAlignment > getAlignment())
11073       return false;
11074 
11075     // 2. Check that the load is a legal operation for that type.
11076     if (!TLI.isOperationLegal(ISD::LOAD, ResVT))
11077       return false;
11078 
11079     // 3. Check that we do not have a zext in the way.
11080     if (Inst->getValueType(0) != getLoadedType())
11081       return false;
11082 
11083     return true;
11084   }
11085 };
11086 }
11087 
11088 /// \brief Check that all bits set in \p UsedBits form a dense region, i.e.,
11089 /// \p UsedBits looks like 0..0 1..1 0..0.
11090 static bool areUsedBitsDense(const APInt &UsedBits) {
11091   // If all the bits are one, this is dense!
11092   if (UsedBits.isAllOnesValue())
11093     return true;
11094 
11095   // Get rid of the unused bits on the right.
11096   APInt NarrowedUsedBits = UsedBits.lshr(UsedBits.countTrailingZeros());
11097   // Get rid of the unused bits on the left.
11098   if (NarrowedUsedBits.countLeadingZeros())
11099     NarrowedUsedBits = NarrowedUsedBits.trunc(NarrowedUsedBits.getActiveBits());
11100   // Check that the chunk of bits is completely used.
11101   return NarrowedUsedBits.isAllOnesValue();
11102 }
11103 
11104 /// \brief Check whether or not \p First and \p Second are next to each other
11105 /// in memory. This means that there is no hole between the bits loaded
11106 /// by \p First and the bits loaded by \p Second.
11107 static bool areSlicesNextToEachOther(const LoadedSlice &First,
11108                                      const LoadedSlice &Second) {
11109   assert(First.Origin == Second.Origin && First.Origin &&
11110          "Unable to match different memory origins.");
11111   APInt UsedBits = First.getUsedBits();
11112   assert((UsedBits & Second.getUsedBits()) == 0 &&
11113          "Slices are not supposed to overlap.");
11114   UsedBits |= Second.getUsedBits();
11115   return areUsedBitsDense(UsedBits);
11116 }
11117 
11118 /// \brief Adjust the \p GlobalLSCost according to the target
11119 /// paring capabilities and the layout of the slices.
11120 /// \pre \p GlobalLSCost should account for at least as many loads as
11121 /// there is in the slices in \p LoadedSlices.
11122 static void adjustCostForPairing(SmallVectorImpl<LoadedSlice> &LoadedSlices,
11123                                  LoadedSlice::Cost &GlobalLSCost) {
11124   unsigned NumberOfSlices = LoadedSlices.size();
11125   // If there is less than 2 elements, no pairing is possible.
11126   if (NumberOfSlices < 2)
11127     return;
11128 
11129   // Sort the slices so that elements that are likely to be next to each
11130   // other in memory are next to each other in the list.
11131   std::sort(LoadedSlices.begin(), LoadedSlices.end(),
11132             [](const LoadedSlice &LHS, const LoadedSlice &RHS) {
11133     assert(LHS.Origin == RHS.Origin && "Different bases not implemented.");
11134     return LHS.getOffsetFromBase() < RHS.getOffsetFromBase();
11135   });
11136   const TargetLowering &TLI = LoadedSlices[0].DAG->getTargetLoweringInfo();
11137   // First (resp. Second) is the first (resp. Second) potentially candidate
11138   // to be placed in a paired load.
11139   const LoadedSlice *First = nullptr;
11140   const LoadedSlice *Second = nullptr;
11141   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice,
11142                 // Set the beginning of the pair.
11143                                                            First = Second) {
11144 
11145     Second = &LoadedSlices[CurrSlice];
11146 
11147     // If First is NULL, it means we start a new pair.
11148     // Get to the next slice.
11149     if (!First)
11150       continue;
11151 
11152     EVT LoadedType = First->getLoadedType();
11153 
11154     // If the types of the slices are different, we cannot pair them.
11155     if (LoadedType != Second->getLoadedType())
11156       continue;
11157 
11158     // Check if the target supplies paired loads for this type.
11159     unsigned RequiredAlignment = 0;
11160     if (!TLI.hasPairedLoad(LoadedType, RequiredAlignment)) {
11161       // move to the next pair, this type is hopeless.
11162       Second = nullptr;
11163       continue;
11164     }
11165     // Check if we meet the alignment requirement.
11166     if (RequiredAlignment > First->getAlignment())
11167       continue;
11168 
11169     // Check that both loads are next to each other in memory.
11170     if (!areSlicesNextToEachOther(*First, *Second))
11171       continue;
11172 
11173     assert(GlobalLSCost.Loads > 0 && "We save more loads than we created!");
11174     --GlobalLSCost.Loads;
11175     // Move to the next pair.
11176     Second = nullptr;
11177   }
11178 }
11179 
11180 /// \brief Check the profitability of all involved LoadedSlice.
11181 /// Currently, it is considered profitable if there is exactly two
11182 /// involved slices (1) which are (2) next to each other in memory, and
11183 /// whose cost (\see LoadedSlice::Cost) is smaller than the original load (3).
11184 ///
11185 /// Note: The order of the elements in \p LoadedSlices may be modified, but not
11186 /// the elements themselves.
11187 ///
11188 /// FIXME: When the cost model will be mature enough, we can relax
11189 /// constraints (1) and (2).
11190 static bool isSlicingProfitable(SmallVectorImpl<LoadedSlice> &LoadedSlices,
11191                                 const APInt &UsedBits, bool ForCodeSize) {
11192   unsigned NumberOfSlices = LoadedSlices.size();
11193   if (StressLoadSlicing)
11194     return NumberOfSlices > 1;
11195 
11196   // Check (1).
11197   if (NumberOfSlices != 2)
11198     return false;
11199 
11200   // Check (2).
11201   if (!areUsedBitsDense(UsedBits))
11202     return false;
11203 
11204   // Check (3).
11205   LoadedSlice::Cost OrigCost(ForCodeSize), GlobalSlicingCost(ForCodeSize);
11206   // The original code has one big load.
11207   OrigCost.Loads = 1;
11208   for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice) {
11209     const LoadedSlice &LS = LoadedSlices[CurrSlice];
11210     // Accumulate the cost of all the slices.
11211     LoadedSlice::Cost SliceCost(LS, ForCodeSize);
11212     GlobalSlicingCost += SliceCost;
11213 
11214     // Account as cost in the original configuration the gain obtained
11215     // with the current slices.
11216     OrigCost.addSliceGain(LS);
11217   }
11218 
11219   // If the target supports paired load, adjust the cost accordingly.
11220   adjustCostForPairing(LoadedSlices, GlobalSlicingCost);
11221   return OrigCost > GlobalSlicingCost;
11222 }
11223 
11224 /// \brief If the given load, \p LI, is used only by trunc or trunc(lshr)
11225 /// operations, split it in the various pieces being extracted.
11226 ///
11227 /// This sort of thing is introduced by SROA.
11228 /// This slicing takes care not to insert overlapping loads.
11229 /// \pre LI is a simple load (i.e., not an atomic or volatile load).
11230 bool DAGCombiner::SliceUpLoad(SDNode *N) {
11231   if (Level < AfterLegalizeDAG)
11232     return false;
11233 
11234   LoadSDNode *LD = cast<LoadSDNode>(N);
11235   if (LD->isVolatile() || !ISD::isNormalLoad(LD) ||
11236       !LD->getValueType(0).isInteger())
11237     return false;
11238 
11239   // Keep track of already used bits to detect overlapping values.
11240   // In that case, we will just abort the transformation.
11241   APInt UsedBits(LD->getValueSizeInBits(0), 0);
11242 
11243   SmallVector<LoadedSlice, 4> LoadedSlices;
11244 
11245   // Check if this load is used as several smaller chunks of bits.
11246   // Basically, look for uses in trunc or trunc(lshr) and record a new chain
11247   // of computation for each trunc.
11248   for (SDNode::use_iterator UI = LD->use_begin(), UIEnd = LD->use_end();
11249        UI != UIEnd; ++UI) {
11250     // Skip the uses of the chain.
11251     if (UI.getUse().getResNo() != 0)
11252       continue;
11253 
11254     SDNode *User = *UI;
11255     unsigned Shift = 0;
11256 
11257     // Check if this is a trunc(lshr).
11258     if (User->getOpcode() == ISD::SRL && User->hasOneUse() &&
11259         isa<ConstantSDNode>(User->getOperand(1))) {
11260       Shift = cast<ConstantSDNode>(User->getOperand(1))->getZExtValue();
11261       User = *User->use_begin();
11262     }
11263 
11264     // At this point, User is a Truncate, iff we encountered, trunc or
11265     // trunc(lshr).
11266     if (User->getOpcode() != ISD::TRUNCATE)
11267       return false;
11268 
11269     // The width of the type must be a power of 2 and greater than 8-bits.
11270     // Otherwise the load cannot be represented in LLVM IR.
11271     // Moreover, if we shifted with a non-8-bits multiple, the slice
11272     // will be across several bytes. We do not support that.
11273     unsigned Width = User->getValueSizeInBits(0);
11274     if (Width < 8 || !isPowerOf2_32(Width) || (Shift & 0x7))
11275       return 0;
11276 
11277     // Build the slice for this chain of computations.
11278     LoadedSlice LS(User, LD, Shift, &DAG);
11279     APInt CurrentUsedBits = LS.getUsedBits();
11280 
11281     // Check if this slice overlaps with another.
11282     if ((CurrentUsedBits & UsedBits) != 0)
11283       return false;
11284     // Update the bits used globally.
11285     UsedBits |= CurrentUsedBits;
11286 
11287     // Check if the new slice would be legal.
11288     if (!LS.isLegal())
11289       return false;
11290 
11291     // Record the slice.
11292     LoadedSlices.push_back(LS);
11293   }
11294 
11295   // Abort slicing if it does not seem to be profitable.
11296   if (!isSlicingProfitable(LoadedSlices, UsedBits, ForCodeSize))
11297     return false;
11298 
11299   ++SlicedLoads;
11300 
11301   // Rewrite each chain to use an independent load.
11302   // By construction, each chain can be represented by a unique load.
11303 
11304   // Prepare the argument for the new token factor for all the slices.
11305   SmallVector<SDValue, 8> ArgChains;
11306   for (SmallVectorImpl<LoadedSlice>::const_iterator
11307            LSIt = LoadedSlices.begin(),
11308            LSItEnd = LoadedSlices.end();
11309        LSIt != LSItEnd; ++LSIt) {
11310     SDValue SliceInst = LSIt->loadSlice();
11311     CombineTo(LSIt->Inst, SliceInst, true);
11312     if (SliceInst.getOpcode() != ISD::LOAD)
11313       SliceInst = SliceInst.getOperand(0);
11314     assert(SliceInst->getOpcode() == ISD::LOAD &&
11315            "It takes more than a zext to get to the loaded slice!!");
11316     ArgChains.push_back(SliceInst.getValue(1));
11317   }
11318 
11319   SDValue Chain = DAG.getNode(ISD::TokenFactor, SDLoc(LD), MVT::Other,
11320                               ArgChains);
11321   DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
11322   return true;
11323 }
11324 
11325 /// Check to see if V is (and load (ptr), imm), where the load is having
11326 /// specific bytes cleared out.  If so, return the byte size being masked out
11327 /// and the shift amount.
11328 static std::pair<unsigned, unsigned>
11329 CheckForMaskedLoad(SDValue V, SDValue Ptr, SDValue Chain) {
11330   std::pair<unsigned, unsigned> Result(0, 0);
11331 
11332   // Check for the structure we're looking for.
11333   if (V->getOpcode() != ISD::AND ||
11334       !isa<ConstantSDNode>(V->getOperand(1)) ||
11335       !ISD::isNormalLoad(V->getOperand(0).getNode()))
11336     return Result;
11337 
11338   // Check the chain and pointer.
11339   LoadSDNode *LD = cast<LoadSDNode>(V->getOperand(0));
11340   if (LD->getBasePtr() != Ptr) return Result;  // Not from same pointer.
11341 
11342   // The store should be chained directly to the load or be an operand of a
11343   // tokenfactor.
11344   if (LD == Chain.getNode())
11345     ; // ok.
11346   else if (Chain->getOpcode() != ISD::TokenFactor)
11347     return Result; // Fail.
11348   else {
11349     bool isOk = false;
11350     for (const SDValue &ChainOp : Chain->op_values())
11351       if (ChainOp.getNode() == LD) {
11352         isOk = true;
11353         break;
11354       }
11355     if (!isOk) return Result;
11356   }
11357 
11358   // This only handles simple types.
11359   if (V.getValueType() != MVT::i16 &&
11360       V.getValueType() != MVT::i32 &&
11361       V.getValueType() != MVT::i64)
11362     return Result;
11363 
11364   // Check the constant mask.  Invert it so that the bits being masked out are
11365   // 0 and the bits being kept are 1.  Use getSExtValue so that leading bits
11366   // follow the sign bit for uniformity.
11367   uint64_t NotMask = ~cast<ConstantSDNode>(V->getOperand(1))->getSExtValue();
11368   unsigned NotMaskLZ = countLeadingZeros(NotMask);
11369   if (NotMaskLZ & 7) return Result;  // Must be multiple of a byte.
11370   unsigned NotMaskTZ = countTrailingZeros(NotMask);
11371   if (NotMaskTZ & 7) return Result;  // Must be multiple of a byte.
11372   if (NotMaskLZ == 64) return Result;  // All zero mask.
11373 
11374   // See if we have a continuous run of bits.  If so, we have 0*1+0*
11375   if (countTrailingOnes(NotMask >> NotMaskTZ) + NotMaskTZ + NotMaskLZ != 64)
11376     return Result;
11377 
11378   // Adjust NotMaskLZ down to be from the actual size of the int instead of i64.
11379   if (V.getValueType() != MVT::i64 && NotMaskLZ)
11380     NotMaskLZ -= 64-V.getValueSizeInBits();
11381 
11382   unsigned MaskedBytes = (V.getValueSizeInBits()-NotMaskLZ-NotMaskTZ)/8;
11383   switch (MaskedBytes) {
11384   case 1:
11385   case 2:
11386   case 4: break;
11387   default: return Result; // All one mask, or 5-byte mask.
11388   }
11389 
11390   // Verify that the first bit starts at a multiple of mask so that the access
11391   // is aligned the same as the access width.
11392   if (NotMaskTZ && NotMaskTZ/8 % MaskedBytes) return Result;
11393 
11394   Result.first = MaskedBytes;
11395   Result.second = NotMaskTZ/8;
11396   return Result;
11397 }
11398 
11399 
11400 /// Check to see if IVal is something that provides a value as specified by
11401 /// MaskInfo. If so, replace the specified store with a narrower store of
11402 /// truncated IVal.
11403 static SDNode *
11404 ShrinkLoadReplaceStoreWithStore(const std::pair<unsigned, unsigned> &MaskInfo,
11405                                 SDValue IVal, StoreSDNode *St,
11406                                 DAGCombiner *DC) {
11407   unsigned NumBytes = MaskInfo.first;
11408   unsigned ByteShift = MaskInfo.second;
11409   SelectionDAG &DAG = DC->getDAG();
11410 
11411   // Check to see if IVal is all zeros in the part being masked in by the 'or'
11412   // that uses this.  If not, this is not a replacement.
11413   APInt Mask = ~APInt::getBitsSet(IVal.getValueSizeInBits(),
11414                                   ByteShift*8, (ByteShift+NumBytes)*8);
11415   if (!DAG.MaskedValueIsZero(IVal, Mask)) return nullptr;
11416 
11417   // Check that it is legal on the target to do this.  It is legal if the new
11418   // VT we're shrinking to (i8/i16/i32) is legal or we're still before type
11419   // legalization.
11420   MVT VT = MVT::getIntegerVT(NumBytes*8);
11421   if (!DC->isTypeLegal(VT))
11422     return nullptr;
11423 
11424   // Okay, we can do this!  Replace the 'St' store with a store of IVal that is
11425   // shifted by ByteShift and truncated down to NumBytes.
11426   if (ByteShift) {
11427     SDLoc DL(IVal);
11428     IVal = DAG.getNode(ISD::SRL, DL, IVal.getValueType(), IVal,
11429                        DAG.getConstant(ByteShift*8, DL,
11430                                     DC->getShiftAmountTy(IVal.getValueType())));
11431   }
11432 
11433   // Figure out the offset for the store and the alignment of the access.
11434   unsigned StOffset;
11435   unsigned NewAlign = St->getAlignment();
11436 
11437   if (DAG.getDataLayout().isLittleEndian())
11438     StOffset = ByteShift;
11439   else
11440     StOffset = IVal.getValueType().getStoreSize() - ByteShift - NumBytes;
11441 
11442   SDValue Ptr = St->getBasePtr();
11443   if (StOffset) {
11444     SDLoc DL(IVal);
11445     Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(),
11446                       Ptr, DAG.getConstant(StOffset, DL, Ptr.getValueType()));
11447     NewAlign = MinAlign(NewAlign, StOffset);
11448   }
11449 
11450   // Truncate down to the new size.
11451   IVal = DAG.getNode(ISD::TRUNCATE, SDLoc(IVal), VT, IVal);
11452 
11453   ++OpsNarrowed;
11454   return DAG
11455       .getStore(St->getChain(), SDLoc(St), IVal, Ptr,
11456                 St->getPointerInfo().getWithOffset(StOffset), NewAlign)
11457       .getNode();
11458 }
11459 
11460 
11461 /// Look for sequence of load / op / store where op is one of 'or', 'xor', and
11462 /// 'and' of immediates. If 'op' is only touching some of the loaded bits, try
11463 /// narrowing the load and store if it would end up being a win for performance
11464 /// or code size.
11465 SDValue DAGCombiner::ReduceLoadOpStoreWidth(SDNode *N) {
11466   StoreSDNode *ST  = cast<StoreSDNode>(N);
11467   if (ST->isVolatile())
11468     return SDValue();
11469 
11470   SDValue Chain = ST->getChain();
11471   SDValue Value = ST->getValue();
11472   SDValue Ptr   = ST->getBasePtr();
11473   EVT VT = Value.getValueType();
11474 
11475   if (ST->isTruncatingStore() || VT.isVector() || !Value.hasOneUse())
11476     return SDValue();
11477 
11478   unsigned Opc = Value.getOpcode();
11479 
11480   // If this is "store (or X, Y), P" and X is "(and (load P), cst)", where cst
11481   // is a byte mask indicating a consecutive number of bytes, check to see if
11482   // Y is known to provide just those bytes.  If so, we try to replace the
11483   // load + replace + store sequence with a single (narrower) store, which makes
11484   // the load dead.
11485   if (Opc == ISD::OR) {
11486     std::pair<unsigned, unsigned> MaskedLoad;
11487     MaskedLoad = CheckForMaskedLoad(Value.getOperand(0), Ptr, Chain);
11488     if (MaskedLoad.first)
11489       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
11490                                                   Value.getOperand(1), ST,this))
11491         return SDValue(NewST, 0);
11492 
11493     // Or is commutative, so try swapping X and Y.
11494     MaskedLoad = CheckForMaskedLoad(Value.getOperand(1), Ptr, Chain);
11495     if (MaskedLoad.first)
11496       if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad,
11497                                                   Value.getOperand(0), ST,this))
11498         return SDValue(NewST, 0);
11499   }
11500 
11501   if ((Opc != ISD::OR && Opc != ISD::XOR && Opc != ISD::AND) ||
11502       Value.getOperand(1).getOpcode() != ISD::Constant)
11503     return SDValue();
11504 
11505   SDValue N0 = Value.getOperand(0);
11506   if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() &&
11507       Chain == SDValue(N0.getNode(), 1)) {
11508     LoadSDNode *LD = cast<LoadSDNode>(N0);
11509     if (LD->getBasePtr() != Ptr ||
11510         LD->getPointerInfo().getAddrSpace() !=
11511         ST->getPointerInfo().getAddrSpace())
11512       return SDValue();
11513 
11514     // Find the type to narrow it the load / op / store to.
11515     SDValue N1 = Value.getOperand(1);
11516     unsigned BitWidth = N1.getValueSizeInBits();
11517     APInt Imm = cast<ConstantSDNode>(N1)->getAPIntValue();
11518     if (Opc == ISD::AND)
11519       Imm ^= APInt::getAllOnesValue(BitWidth);
11520     if (Imm == 0 || Imm.isAllOnesValue())
11521       return SDValue();
11522     unsigned ShAmt = Imm.countTrailingZeros();
11523     unsigned MSB = BitWidth - Imm.countLeadingZeros() - 1;
11524     unsigned NewBW = NextPowerOf2(MSB - ShAmt);
11525     EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
11526     // The narrowing should be profitable, the load/store operation should be
11527     // legal (or custom) and the store size should be equal to the NewVT width.
11528     while (NewBW < BitWidth &&
11529            (NewVT.getStoreSizeInBits() != NewBW ||
11530             !TLI.isOperationLegalOrCustom(Opc, NewVT) ||
11531             !TLI.isNarrowingProfitable(VT, NewVT))) {
11532       NewBW = NextPowerOf2(NewBW);
11533       NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW);
11534     }
11535     if (NewBW >= BitWidth)
11536       return SDValue();
11537 
11538     // If the lsb changed does not start at the type bitwidth boundary,
11539     // start at the previous one.
11540     if (ShAmt % NewBW)
11541       ShAmt = (((ShAmt + NewBW - 1) / NewBW) * NewBW) - NewBW;
11542     APInt Mask = APInt::getBitsSet(BitWidth, ShAmt,
11543                                    std::min(BitWidth, ShAmt + NewBW));
11544     if ((Imm & Mask) == Imm) {
11545       APInt NewImm = (Imm & Mask).lshr(ShAmt).trunc(NewBW);
11546       if (Opc == ISD::AND)
11547         NewImm ^= APInt::getAllOnesValue(NewBW);
11548       uint64_t PtrOff = ShAmt / 8;
11549       // For big endian targets, we need to adjust the offset to the pointer to
11550       // load the correct bytes.
11551       if (DAG.getDataLayout().isBigEndian())
11552         PtrOff = (BitWidth + 7 - NewBW) / 8 - PtrOff;
11553 
11554       unsigned NewAlign = MinAlign(LD->getAlignment(), PtrOff);
11555       Type *NewVTTy = NewVT.getTypeForEVT(*DAG.getContext());
11556       if (NewAlign < DAG.getDataLayout().getABITypeAlignment(NewVTTy))
11557         return SDValue();
11558 
11559       SDValue NewPtr = DAG.getNode(ISD::ADD, SDLoc(LD),
11560                                    Ptr.getValueType(), Ptr,
11561                                    DAG.getConstant(PtrOff, SDLoc(LD),
11562                                                    Ptr.getValueType()));
11563       SDValue NewLD =
11564           DAG.getLoad(NewVT, SDLoc(N0), LD->getChain(), NewPtr,
11565                       LD->getPointerInfo().getWithOffset(PtrOff), NewAlign,
11566                       LD->getMemOperand()->getFlags(), LD->getAAInfo());
11567       SDValue NewVal = DAG.getNode(Opc, SDLoc(Value), NewVT, NewLD,
11568                                    DAG.getConstant(NewImm, SDLoc(Value),
11569                                                    NewVT));
11570       SDValue NewST =
11571           DAG.getStore(Chain, SDLoc(N), NewVal, NewPtr,
11572                        ST->getPointerInfo().getWithOffset(PtrOff), NewAlign);
11573 
11574       AddToWorklist(NewPtr.getNode());
11575       AddToWorklist(NewLD.getNode());
11576       AddToWorklist(NewVal.getNode());
11577       WorklistRemover DeadNodes(*this);
11578       DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLD.getValue(1));
11579       ++OpsNarrowed;
11580       return NewST;
11581     }
11582   }
11583 
11584   return SDValue();
11585 }
11586 
11587 /// For a given floating point load / store pair, if the load value isn't used
11588 /// by any other operations, then consider transforming the pair to integer
11589 /// load / store operations if the target deems the transformation profitable.
11590 SDValue DAGCombiner::TransformFPLoadStorePair(SDNode *N) {
11591   StoreSDNode *ST  = cast<StoreSDNode>(N);
11592   SDValue Chain = ST->getChain();
11593   SDValue Value = ST->getValue();
11594   if (ISD::isNormalStore(ST) && ISD::isNormalLoad(Value.getNode()) &&
11595       Value.hasOneUse() &&
11596       Chain == SDValue(Value.getNode(), 1)) {
11597     LoadSDNode *LD = cast<LoadSDNode>(Value);
11598     EVT VT = LD->getMemoryVT();
11599     if (!VT.isFloatingPoint() ||
11600         VT != ST->getMemoryVT() ||
11601         LD->isNonTemporal() ||
11602         ST->isNonTemporal() ||
11603         LD->getPointerInfo().getAddrSpace() != 0 ||
11604         ST->getPointerInfo().getAddrSpace() != 0)
11605       return SDValue();
11606 
11607     EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits());
11608     if (!TLI.isOperationLegal(ISD::LOAD, IntVT) ||
11609         !TLI.isOperationLegal(ISD::STORE, IntVT) ||
11610         !TLI.isDesirableToTransformToIntegerOp(ISD::LOAD, VT) ||
11611         !TLI.isDesirableToTransformToIntegerOp(ISD::STORE, VT))
11612       return SDValue();
11613 
11614     unsigned LDAlign = LD->getAlignment();
11615     unsigned STAlign = ST->getAlignment();
11616     Type *IntVTTy = IntVT.getTypeForEVT(*DAG.getContext());
11617     unsigned ABIAlign = DAG.getDataLayout().getABITypeAlignment(IntVTTy);
11618     if (LDAlign < ABIAlign || STAlign < ABIAlign)
11619       return SDValue();
11620 
11621     SDValue NewLD =
11622         DAG.getLoad(IntVT, SDLoc(Value), LD->getChain(), LD->getBasePtr(),
11623                     LD->getPointerInfo(), LDAlign);
11624 
11625     SDValue NewST =
11626         DAG.getStore(NewLD.getValue(1), SDLoc(N), NewLD, ST->getBasePtr(),
11627                      ST->getPointerInfo(), STAlign);
11628 
11629     AddToWorklist(NewLD.getNode());
11630     AddToWorklist(NewST.getNode());
11631     WorklistRemover DeadNodes(*this);
11632     DAG.ReplaceAllUsesOfValueWith(Value.getValue(1), NewLD.getValue(1));
11633     ++LdStFP2Int;
11634     return NewST;
11635   }
11636 
11637   return SDValue();
11638 }
11639 
11640 // This is a helper function for visitMUL to check the profitability
11641 // of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2).
11642 // MulNode is the original multiply, AddNode is (add x, c1),
11643 // and ConstNode is c2.
11644 //
11645 // If the (add x, c1) has multiple uses, we could increase
11646 // the number of adds if we make this transformation.
11647 // It would only be worth doing this if we can remove a
11648 // multiply in the process. Check for that here.
11649 // To illustrate:
11650 //     (A + c1) * c3
11651 //     (A + c2) * c3
11652 // We're checking for cases where we have common "c3 * A" expressions.
11653 bool DAGCombiner::isMulAddWithConstProfitable(SDNode *MulNode,
11654                                               SDValue &AddNode,
11655                                               SDValue &ConstNode) {
11656   APInt Val;
11657 
11658   // If the add only has one use, this would be OK to do.
11659   if (AddNode.getNode()->hasOneUse())
11660     return true;
11661 
11662   // Walk all the users of the constant with which we're multiplying.
11663   for (SDNode *Use : ConstNode->uses()) {
11664 
11665     if (Use == MulNode) // This use is the one we're on right now. Skip it.
11666       continue;
11667 
11668     if (Use->getOpcode() == ISD::MUL) { // We have another multiply use.
11669       SDNode *OtherOp;
11670       SDNode *MulVar = AddNode.getOperand(0).getNode();
11671 
11672       // OtherOp is what we're multiplying against the constant.
11673       if (Use->getOperand(0) == ConstNode)
11674         OtherOp = Use->getOperand(1).getNode();
11675       else
11676         OtherOp = Use->getOperand(0).getNode();
11677 
11678       // Check to see if multiply is with the same operand of our "add".
11679       //
11680       //     ConstNode  = CONST
11681       //     Use = ConstNode * A  <-- visiting Use. OtherOp is A.
11682       //     ...
11683       //     AddNode  = (A + c1)  <-- MulVar is A.
11684       //         = AddNode * ConstNode   <-- current visiting instruction.
11685       //
11686       // If we make this transformation, we will have a common
11687       // multiply (ConstNode * A) that we can save.
11688       if (OtherOp == MulVar)
11689         return true;
11690 
11691       // Now check to see if a future expansion will give us a common
11692       // multiply.
11693       //
11694       //     ConstNode  = CONST
11695       //     AddNode    = (A + c1)
11696       //     ...   = AddNode * ConstNode <-- current visiting instruction.
11697       //     ...
11698       //     OtherOp = (A + c2)
11699       //     Use     = OtherOp * ConstNode <-- visiting Use.
11700       //
11701       // If we make this transformation, we will have a common
11702       // multiply (CONST * A) after we also do the same transformation
11703       // to the "t2" instruction.
11704       if (OtherOp->getOpcode() == ISD::ADD &&
11705           DAG.isConstantIntBuildVectorOrConstantInt(OtherOp->getOperand(1)) &&
11706           OtherOp->getOperand(0).getNode() == MulVar)
11707         return true;
11708     }
11709   }
11710 
11711   // Didn't find a case where this would be profitable.
11712   return false;
11713 }
11714 
11715 SDValue DAGCombiner::getMergedConstantVectorStore(
11716     SelectionDAG &DAG, const SDLoc &SL, ArrayRef<MemOpLink> Stores,
11717     SmallVectorImpl<SDValue> &Chains, EVT Ty) const {
11718   SmallVector<SDValue, 8> BuildVector;
11719 
11720   for (unsigned I = 0, E = Ty.getVectorNumElements(); I != E; ++I) {
11721     StoreSDNode *St = cast<StoreSDNode>(Stores[I].MemNode);
11722     Chains.push_back(St->getChain());
11723     BuildVector.push_back(St->getValue());
11724   }
11725 
11726   return DAG.getBuildVector(Ty, SL, BuildVector);
11727 }
11728 
11729 bool DAGCombiner::MergeStoresOfConstantsOrVecElts(
11730                   SmallVectorImpl<MemOpLink> &StoreNodes, EVT MemVT,
11731                   unsigned NumStores, bool IsConstantSrc, bool UseVector) {
11732   // Make sure we have something to merge.
11733   if (NumStores < 2)
11734     return false;
11735 
11736   int64_t ElementSizeBytes = MemVT.getSizeInBits() / 8;
11737   LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
11738   unsigned LatestNodeUsed = 0;
11739 
11740   for (unsigned i=0; i < NumStores; ++i) {
11741     // Find a chain for the new wide-store operand. Notice that some
11742     // of the store nodes that we found may not be selected for inclusion
11743     // in the wide store. The chain we use needs to be the chain of the
11744     // latest store node which is *used* and replaced by the wide store.
11745     if (StoreNodes[i].SequenceNum < StoreNodes[LatestNodeUsed].SequenceNum)
11746       LatestNodeUsed = i;
11747   }
11748 
11749   SmallVector<SDValue, 8> Chains;
11750 
11751   // The latest Node in the DAG.
11752   LSBaseSDNode *LatestOp = StoreNodes[LatestNodeUsed].MemNode;
11753   SDLoc DL(StoreNodes[0].MemNode);
11754 
11755   SDValue StoredVal;
11756   if (UseVector) {
11757     bool IsVec = MemVT.isVector();
11758     unsigned Elts = NumStores;
11759     if (IsVec) {
11760       // When merging vector stores, get the total number of elements.
11761       Elts *= MemVT.getVectorNumElements();
11762     }
11763     // Get the type for the merged vector store.
11764     EVT Ty = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
11765     assert(TLI.isTypeLegal(Ty) && "Illegal vector store");
11766 
11767     if (IsConstantSrc) {
11768       StoredVal = getMergedConstantVectorStore(DAG, DL, StoreNodes, Chains, Ty);
11769     } else {
11770       SmallVector<SDValue, 8> Ops;
11771       for (unsigned i = 0; i < NumStores; ++i) {
11772         StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
11773         SDValue Val = St->getValue();
11774         // All operands of BUILD_VECTOR / CONCAT_VECTOR must have the same type.
11775         if (Val.getValueType() != MemVT)
11776           return false;
11777         Ops.push_back(Val);
11778         Chains.push_back(St->getChain());
11779       }
11780 
11781       // Build the extracted vector elements back into a vector.
11782       StoredVal = DAG.getNode(IsVec ? ISD::CONCAT_VECTORS : ISD::BUILD_VECTOR,
11783                               DL, Ty, Ops);    }
11784   } else {
11785     // We should always use a vector store when merging extracted vector
11786     // elements, so this path implies a store of constants.
11787     assert(IsConstantSrc && "Merged vector elements should use vector store");
11788 
11789     unsigned SizeInBits = NumStores * ElementSizeBytes * 8;
11790     APInt StoreInt(SizeInBits, 0);
11791 
11792     // Construct a single integer constant which is made of the smaller
11793     // constant inputs.
11794     bool IsLE = DAG.getDataLayout().isLittleEndian();
11795     for (unsigned i = 0; i < NumStores; ++i) {
11796       unsigned Idx = IsLE ? (NumStores - 1 - i) : i;
11797       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[Idx].MemNode);
11798       Chains.push_back(St->getChain());
11799 
11800       SDValue Val = St->getValue();
11801       StoreInt <<= ElementSizeBytes * 8;
11802       if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Val)) {
11803         StoreInt |= C->getAPIntValue().zext(SizeInBits);
11804       } else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(Val)) {
11805         StoreInt |= C->getValueAPF().bitcastToAPInt().zext(SizeInBits);
11806       } else {
11807         llvm_unreachable("Invalid constant element type");
11808       }
11809     }
11810 
11811     // Create the new Load and Store operations.
11812     EVT StoreTy = EVT::getIntegerVT(*DAG.getContext(), SizeInBits);
11813     StoredVal = DAG.getConstant(StoreInt, DL, StoreTy);
11814   }
11815 
11816   assert(!Chains.empty());
11817 
11818   SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
11819   SDValue NewStore = DAG.getStore(NewChain, DL, StoredVal,
11820                                   FirstInChain->getBasePtr(),
11821                                   FirstInChain->getPointerInfo(),
11822                                   FirstInChain->getAlignment());
11823 
11824   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11825                                                   : DAG.getSubtarget().useAA();
11826   if (UseAA) {
11827     // Replace all merged stores with the new store.
11828     for (unsigned i = 0; i < NumStores; ++i)
11829       CombineTo(StoreNodes[i].MemNode, NewStore);
11830   } else {
11831     // Replace the last store with the new store.
11832     CombineTo(LatestOp, NewStore);
11833     // Erase all other stores.
11834     for (unsigned i = 0; i < NumStores; ++i) {
11835       if (StoreNodes[i].MemNode == LatestOp)
11836         continue;
11837       StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
11838       // ReplaceAllUsesWith will replace all uses that existed when it was
11839       // called, but graph optimizations may cause new ones to appear. For
11840       // example, the case in pr14333 looks like
11841       //
11842       //  St's chain -> St -> another store -> X
11843       //
11844       // And the only difference from St to the other store is the chain.
11845       // When we change it's chain to be St's chain they become identical,
11846       // get CSEed and the net result is that X is now a use of St.
11847       // Since we know that St is redundant, just iterate.
11848       while (!St->use_empty())
11849         DAG.ReplaceAllUsesWith(SDValue(St, 0), St->getChain());
11850       deleteAndRecombine(St);
11851     }
11852   }
11853 
11854   StoreNodes.erase(StoreNodes.begin() + NumStores, StoreNodes.end());
11855   return true;
11856 }
11857 
11858 void DAGCombiner::getStoreMergeAndAliasCandidates(
11859     StoreSDNode* St, SmallVectorImpl<MemOpLink> &StoreNodes,
11860     SmallVectorImpl<LSBaseSDNode*> &AliasLoadNodes) {
11861   // This holds the base pointer, index, and the offset in bytes from the base
11862   // pointer.
11863   BaseIndexOffset BasePtr = BaseIndexOffset::match(St->getBasePtr(), DAG);
11864 
11865   // We must have a base and an offset.
11866   if (!BasePtr.Base.getNode())
11867     return;
11868 
11869   // Do not handle stores to undef base pointers.
11870   if (BasePtr.Base.isUndef())
11871     return;
11872 
11873   // Walk up the chain and look for nodes with offsets from the same
11874   // base pointer. Stop when reaching an instruction with a different kind
11875   // or instruction which has a different base pointer.
11876   EVT MemVT = St->getMemoryVT();
11877   unsigned Seq = 0;
11878   StoreSDNode *Index = St;
11879 
11880 
11881   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
11882                                                   : DAG.getSubtarget().useAA();
11883 
11884   if (UseAA) {
11885     // Look at other users of the same chain. Stores on the same chain do not
11886     // alias. If combiner-aa is enabled, non-aliasing stores are canonicalized
11887     // to be on the same chain, so don't bother looking at adjacent chains.
11888 
11889     SDValue Chain = St->getChain();
11890     for (auto I = Chain->use_begin(), E = Chain->use_end(); I != E; ++I) {
11891       if (StoreSDNode *OtherST = dyn_cast<StoreSDNode>(*I)) {
11892         if (I.getOperandNo() != 0)
11893           continue;
11894 
11895         if (OtherST->isVolatile() || OtherST->isIndexed())
11896           continue;
11897 
11898         if (OtherST->getMemoryVT() != MemVT)
11899           continue;
11900 
11901         BaseIndexOffset Ptr = BaseIndexOffset::match(OtherST->getBasePtr(), DAG);
11902 
11903         if (Ptr.equalBaseIndex(BasePtr))
11904           StoreNodes.push_back(MemOpLink(OtherST, Ptr.Offset, Seq++));
11905       }
11906     }
11907 
11908     return;
11909   }
11910 
11911   while (Index) {
11912     // If the chain has more than one use, then we can't reorder the mem ops.
11913     if (Index != St && !SDValue(Index, 0)->hasOneUse())
11914       break;
11915 
11916     // Find the base pointer and offset for this memory node.
11917     BaseIndexOffset Ptr = BaseIndexOffset::match(Index->getBasePtr(), DAG);
11918 
11919     // Check that the base pointer is the same as the original one.
11920     if (!Ptr.equalBaseIndex(BasePtr))
11921       break;
11922 
11923     // The memory operands must not be volatile.
11924     if (Index->isVolatile() || Index->isIndexed())
11925       break;
11926 
11927     // No truncation.
11928     if (Index->isTruncatingStore())
11929       break;
11930 
11931     // The stored memory type must be the same.
11932     if (Index->getMemoryVT() != MemVT)
11933       break;
11934 
11935     // We do not allow under-aligned stores in order to prevent
11936     // overriding stores. NOTE: this is a bad hack. Alignment SHOULD
11937     // be irrelevant here; what MATTERS is that we not move memory
11938     // operations that potentially overlap past each-other.
11939     if (Index->getAlignment() < MemVT.getStoreSize())
11940       break;
11941 
11942     // We found a potential memory operand to merge.
11943     StoreNodes.push_back(MemOpLink(Index, Ptr.Offset, Seq++));
11944 
11945     // Find the next memory operand in the chain. If the next operand in the
11946     // chain is a store then move up and continue the scan with the next
11947     // memory operand. If the next operand is a load save it and use alias
11948     // information to check if it interferes with anything.
11949     SDNode *NextInChain = Index->getChain().getNode();
11950     while (1) {
11951       if (StoreSDNode *STn = dyn_cast<StoreSDNode>(NextInChain)) {
11952         // We found a store node. Use it for the next iteration.
11953         Index = STn;
11954         break;
11955       } else if (LoadSDNode *Ldn = dyn_cast<LoadSDNode>(NextInChain)) {
11956         if (Ldn->isVolatile()) {
11957           Index = nullptr;
11958           break;
11959         }
11960 
11961         // Save the load node for later. Continue the scan.
11962         AliasLoadNodes.push_back(Ldn);
11963         NextInChain = Ldn->getChain().getNode();
11964         continue;
11965       } else {
11966         Index = nullptr;
11967         break;
11968       }
11969     }
11970   }
11971 }
11972 
11973 // We need to check that merging these stores does not cause a loop
11974 // in the DAG. Any store candidate may depend on another candidate
11975 // indirectly through its operand (we already consider dependencies
11976 // through the chain). Check in parallel by searching up from
11977 // non-chain operands of candidates.
11978 bool DAGCombiner::checkMergeStoreCandidatesForDependencies(
11979     SmallVectorImpl<MemOpLink> &StoreNodes) {
11980   SmallPtrSet<const SDNode *, 16> Visited;
11981   SmallVector<const SDNode *, 8> Worklist;
11982   // search ops of store candidates
11983   for (unsigned i = 0; i < StoreNodes.size(); ++i) {
11984     SDNode *n = StoreNodes[i].MemNode;
11985     // Potential loops may happen only through non-chain operands
11986     for (unsigned j = 1; j < n->getNumOperands(); ++j)
11987       Worklist.push_back(n->getOperand(j).getNode());
11988   }
11989   // search through DAG. We can stop early if we find a storenode
11990   for (unsigned i = 0; i < StoreNodes.size(); ++i) {
11991     if (SDNode::hasPredecessorHelper(StoreNodes[i].MemNode, Visited, Worklist))
11992       return false;
11993   }
11994   return true;
11995 }
11996 
11997 bool DAGCombiner::MergeConsecutiveStores(
11998     StoreSDNode* St, SmallVectorImpl<MemOpLink> &StoreNodes) {
11999   if (OptLevel == CodeGenOpt::None)
12000     return false;
12001 
12002   EVT MemVT = St->getMemoryVT();
12003   int64_t ElementSizeBytes = MemVT.getSizeInBits() / 8;
12004   bool NoVectors = DAG.getMachineFunction().getFunction()->hasFnAttribute(
12005       Attribute::NoImplicitFloat);
12006 
12007   // This function cannot currently deal with non-byte-sized memory sizes.
12008   if (ElementSizeBytes * 8 != MemVT.getSizeInBits())
12009     return false;
12010 
12011   if (!MemVT.isSimple())
12012     return false;
12013 
12014   // Perform an early exit check. Do not bother looking at stored values that
12015   // are not constants, loads, or extracted vector elements.
12016   SDValue StoredVal = St->getValue();
12017   bool IsLoadSrc = isa<LoadSDNode>(StoredVal);
12018   bool IsConstantSrc = isa<ConstantSDNode>(StoredVal) ||
12019                        isa<ConstantFPSDNode>(StoredVal);
12020   bool IsExtractVecSrc = (StoredVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT ||
12021                           StoredVal.getOpcode() == ISD::EXTRACT_SUBVECTOR);
12022 
12023   if (!IsConstantSrc && !IsLoadSrc && !IsExtractVecSrc)
12024     return false;
12025 
12026   // Don't merge vectors into wider vectors if the source data comes from loads.
12027   // TODO: This restriction can be lifted by using logic similar to the
12028   // ExtractVecSrc case.
12029   if (MemVT.isVector() && IsLoadSrc)
12030     return false;
12031 
12032   // Only look at ends of store sequences.
12033   SDValue Chain = SDValue(St, 0);
12034   if (Chain->hasOneUse() && Chain->use_begin()->getOpcode() == ISD::STORE)
12035     return false;
12036 
12037   // Save the LoadSDNodes that we find in the chain.
12038   // We need to make sure that these nodes do not interfere with
12039   // any of the store nodes.
12040   SmallVector<LSBaseSDNode*, 8> AliasLoadNodes;
12041 
12042   getStoreMergeAndAliasCandidates(St, StoreNodes, AliasLoadNodes);
12043 
12044   // Check if there is anything to merge.
12045   if (StoreNodes.size() < 2)
12046     return false;
12047 
12048   // only do dependence check in AA case
12049   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
12050                                                   : DAG.getSubtarget().useAA();
12051   if (UseAA && !checkMergeStoreCandidatesForDependencies(StoreNodes))
12052     return false;
12053 
12054   // Sort the memory operands according to their distance from the
12055   // base pointer.  As a secondary criteria: make sure stores coming
12056   // later in the code come first in the list. This is important for
12057   // the non-UseAA case, because we're merging stores into the FINAL
12058   // store along a chain which potentially contains aliasing stores.
12059   // Thus, if there are multiple stores to the same address, the last
12060   // one can be considered for merging but not the others.
12061   std::sort(StoreNodes.begin(), StoreNodes.end(),
12062             [](MemOpLink LHS, MemOpLink RHS) {
12063     return LHS.OffsetFromBase < RHS.OffsetFromBase ||
12064            (LHS.OffsetFromBase == RHS.OffsetFromBase &&
12065             LHS.SequenceNum < RHS.SequenceNum);
12066   });
12067 
12068   // Scan the memory operations on the chain and find the first non-consecutive
12069   // store memory address.
12070   unsigned LastConsecutiveStore = 0;
12071   int64_t StartAddress = StoreNodes[0].OffsetFromBase;
12072   for (unsigned i = 0, e = StoreNodes.size(); i < e; ++i) {
12073 
12074     // Check that the addresses are consecutive starting from the second
12075     // element in the list of stores.
12076     if (i > 0) {
12077       int64_t CurrAddress = StoreNodes[i].OffsetFromBase;
12078       if (CurrAddress - StartAddress != (ElementSizeBytes * i))
12079         break;
12080     }
12081 
12082     // Check if this store interferes with any of the loads that we found.
12083     // If we find a load that alias with this store. Stop the sequence.
12084     if (any_of(AliasLoadNodes, [&](LSBaseSDNode *Ldn) {
12085           return isAlias(Ldn, StoreNodes[i].MemNode);
12086         }))
12087       break;
12088 
12089     // Mark this node as useful.
12090     LastConsecutiveStore = i;
12091   }
12092 
12093   // The node with the lowest store address.
12094   LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode;
12095   unsigned FirstStoreAS = FirstInChain->getAddressSpace();
12096   unsigned FirstStoreAlign = FirstInChain->getAlignment();
12097   LLVMContext &Context = *DAG.getContext();
12098   const DataLayout &DL = DAG.getDataLayout();
12099 
12100   // Store the constants into memory as one consecutive store.
12101   if (IsConstantSrc) {
12102     unsigned LastLegalType = 0;
12103     unsigned LastLegalVectorType = 0;
12104     bool NonZero = false;
12105     for (unsigned i=0; i<LastConsecutiveStore+1; ++i) {
12106       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
12107       SDValue StoredVal = St->getValue();
12108 
12109       if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(StoredVal)) {
12110         NonZero |= !C->isNullValue();
12111       } else if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(StoredVal)) {
12112         NonZero |= !C->getConstantFPValue()->isNullValue();
12113       } else {
12114         // Non-constant.
12115         break;
12116       }
12117 
12118       // Find a legal type for the constant store.
12119       unsigned SizeInBits = (i+1) * ElementSizeBytes * 8;
12120       EVT StoreTy = EVT::getIntegerVT(Context, SizeInBits);
12121       bool IsFast;
12122       if (TLI.isTypeLegal(StoreTy) &&
12123           TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
12124                                  FirstStoreAlign, &IsFast) && IsFast) {
12125         LastLegalType = i+1;
12126       // Or check whether a truncstore is legal.
12127       } else if (TLI.getTypeAction(Context, StoreTy) ==
12128                  TargetLowering::TypePromoteInteger) {
12129         EVT LegalizedStoredValueTy =
12130           TLI.getTypeToTransformTo(Context, StoredVal.getValueType());
12131         if (TLI.isTruncStoreLegal(LegalizedStoredValueTy, StoreTy) &&
12132             TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
12133                                    FirstStoreAS, FirstStoreAlign, &IsFast) &&
12134             IsFast) {
12135           LastLegalType = i + 1;
12136         }
12137       }
12138 
12139       // We only use vectors if the constant is known to be zero or the target
12140       // allows it and the function is not marked with the noimplicitfloat
12141       // attribute.
12142       if ((!NonZero || TLI.storeOfVectorConstantIsCheap(MemVT, i+1,
12143                                                         FirstStoreAS)) &&
12144           !NoVectors) {
12145         // Find a legal type for the vector store.
12146         EVT Ty = EVT::getVectorVT(Context, MemVT, i+1);
12147         if (TLI.isTypeLegal(Ty) &&
12148             TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS,
12149                                    FirstStoreAlign, &IsFast) && IsFast)
12150           LastLegalVectorType = i + 1;
12151       }
12152     }
12153 
12154     // Check if we found a legal integer type to store.
12155     if (LastLegalType == 0 && LastLegalVectorType == 0)
12156       return false;
12157 
12158     bool UseVector = (LastLegalVectorType > LastLegalType) && !NoVectors;
12159     unsigned NumElem = UseVector ? LastLegalVectorType : LastLegalType;
12160 
12161     return MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumElem,
12162                                            true, UseVector);
12163   }
12164 
12165   // When extracting multiple vector elements, try to store them
12166   // in one vector store rather than a sequence of scalar stores.
12167   if (IsExtractVecSrc) {
12168     unsigned NumStoresToMerge = 0;
12169     bool IsVec = MemVT.isVector();
12170     for (unsigned i = 0; i < LastConsecutiveStore + 1; ++i) {
12171       StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
12172       unsigned StoreValOpcode = St->getValue().getOpcode();
12173       // This restriction could be loosened.
12174       // Bail out if any stored values are not elements extracted from a vector.
12175       // It should be possible to handle mixed sources, but load sources need
12176       // more careful handling (see the block of code below that handles
12177       // consecutive loads).
12178       if (StoreValOpcode != ISD::EXTRACT_VECTOR_ELT &&
12179           StoreValOpcode != ISD::EXTRACT_SUBVECTOR)
12180         return false;
12181 
12182       // Find a legal type for the vector store.
12183       unsigned Elts = i + 1;
12184       if (IsVec) {
12185         // When merging vector stores, get the total number of elements.
12186         Elts *= MemVT.getVectorNumElements();
12187       }
12188       EVT Ty = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts);
12189       bool IsFast;
12190       if (TLI.isTypeLegal(Ty) &&
12191           TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS,
12192                                  FirstStoreAlign, &IsFast) && IsFast)
12193         NumStoresToMerge = i + 1;
12194     }
12195 
12196     return MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumStoresToMerge,
12197                                            false, true);
12198   }
12199 
12200   // Below we handle the case of multiple consecutive stores that
12201   // come from multiple consecutive loads. We merge them into a single
12202   // wide load and a single wide store.
12203 
12204   // Look for load nodes which are used by the stored values.
12205   SmallVector<MemOpLink, 8> LoadNodes;
12206 
12207   // Find acceptable loads. Loads need to have the same chain (token factor),
12208   // must not be zext, volatile, indexed, and they must be consecutive.
12209   BaseIndexOffset LdBasePtr;
12210   for (unsigned i=0; i<LastConsecutiveStore+1; ++i) {
12211     StoreSDNode *St  = cast<StoreSDNode>(StoreNodes[i].MemNode);
12212     LoadSDNode *Ld = dyn_cast<LoadSDNode>(St->getValue());
12213     if (!Ld) break;
12214 
12215     // Loads must only have one use.
12216     if (!Ld->hasNUsesOfValue(1, 0))
12217       break;
12218 
12219     // The memory operands must not be volatile.
12220     if (Ld->isVolatile() || Ld->isIndexed())
12221       break;
12222 
12223     // We do not accept ext loads.
12224     if (Ld->getExtensionType() != ISD::NON_EXTLOAD)
12225       break;
12226 
12227     // The stored memory type must be the same.
12228     if (Ld->getMemoryVT() != MemVT)
12229       break;
12230 
12231     BaseIndexOffset LdPtr = BaseIndexOffset::match(Ld->getBasePtr(), DAG);
12232     // If this is not the first ptr that we check.
12233     if (LdBasePtr.Base.getNode()) {
12234       // The base ptr must be the same.
12235       if (!LdPtr.equalBaseIndex(LdBasePtr))
12236         break;
12237     } else {
12238       // Check that all other base pointers are the same as this one.
12239       LdBasePtr = LdPtr;
12240     }
12241 
12242     // We found a potential memory operand to merge.
12243     LoadNodes.push_back(MemOpLink(Ld, LdPtr.Offset, 0));
12244   }
12245 
12246   if (LoadNodes.size() < 2)
12247     return false;
12248 
12249   // If we have load/store pair instructions and we only have two values,
12250   // don't bother.
12251   unsigned RequiredAlignment;
12252   if (LoadNodes.size() == 2 && TLI.hasPairedLoad(MemVT, RequiredAlignment) &&
12253       St->getAlignment() >= RequiredAlignment)
12254     return false;
12255 
12256   LoadSDNode *FirstLoad = cast<LoadSDNode>(LoadNodes[0].MemNode);
12257   unsigned FirstLoadAS = FirstLoad->getAddressSpace();
12258   unsigned FirstLoadAlign = FirstLoad->getAlignment();
12259 
12260   // Scan the memory operations on the chain and find the first non-consecutive
12261   // load memory address. These variables hold the index in the store node
12262   // array.
12263   unsigned LastConsecutiveLoad = 0;
12264   // This variable refers to the size and not index in the array.
12265   unsigned LastLegalVectorType = 0;
12266   unsigned LastLegalIntegerType = 0;
12267   StartAddress = LoadNodes[0].OffsetFromBase;
12268   SDValue FirstChain = FirstLoad->getChain();
12269   for (unsigned i = 1; i < LoadNodes.size(); ++i) {
12270     // All loads must share the same chain.
12271     if (LoadNodes[i].MemNode->getChain() != FirstChain)
12272       break;
12273 
12274     int64_t CurrAddress = LoadNodes[i].OffsetFromBase;
12275     if (CurrAddress - StartAddress != (ElementSizeBytes * i))
12276       break;
12277     LastConsecutiveLoad = i;
12278     // Find a legal type for the vector store.
12279     EVT StoreTy = EVT::getVectorVT(Context, MemVT, i+1);
12280     bool IsFastSt, IsFastLd;
12281     if (TLI.isTypeLegal(StoreTy) &&
12282         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
12283                                FirstStoreAlign, &IsFastSt) && IsFastSt &&
12284         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS,
12285                                FirstLoadAlign, &IsFastLd) && IsFastLd) {
12286       LastLegalVectorType = i + 1;
12287     }
12288 
12289     // Find a legal type for the integer store.
12290     unsigned SizeInBits = (i+1) * ElementSizeBytes * 8;
12291     StoreTy = EVT::getIntegerVT(Context, SizeInBits);
12292     if (TLI.isTypeLegal(StoreTy) &&
12293         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS,
12294                                FirstStoreAlign, &IsFastSt) && IsFastSt &&
12295         TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS,
12296                                FirstLoadAlign, &IsFastLd) && IsFastLd)
12297       LastLegalIntegerType = i + 1;
12298     // Or check whether a truncstore and extload is legal.
12299     else if (TLI.getTypeAction(Context, StoreTy) ==
12300              TargetLowering::TypePromoteInteger) {
12301       EVT LegalizedStoredValueTy =
12302         TLI.getTypeToTransformTo(Context, StoreTy);
12303       if (TLI.isTruncStoreLegal(LegalizedStoredValueTy, StoreTy) &&
12304           TLI.isLoadExtLegal(ISD::ZEXTLOAD, LegalizedStoredValueTy, StoreTy) &&
12305           TLI.isLoadExtLegal(ISD::SEXTLOAD, LegalizedStoredValueTy, StoreTy) &&
12306           TLI.isLoadExtLegal(ISD::EXTLOAD, LegalizedStoredValueTy, StoreTy) &&
12307           TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
12308                                  FirstStoreAS, FirstStoreAlign, &IsFastSt) &&
12309           IsFastSt &&
12310           TLI.allowsMemoryAccess(Context, DL, LegalizedStoredValueTy,
12311                                  FirstLoadAS, FirstLoadAlign, &IsFastLd) &&
12312           IsFastLd)
12313         LastLegalIntegerType = i+1;
12314     }
12315   }
12316 
12317   // Only use vector types if the vector type is larger than the integer type.
12318   // If they are the same, use integers.
12319   bool UseVectorTy = LastLegalVectorType > LastLegalIntegerType && !NoVectors;
12320   unsigned LastLegalType = std::max(LastLegalVectorType, LastLegalIntegerType);
12321 
12322   // We add +1 here because the LastXXX variables refer to location while
12323   // the NumElem refers to array/index size.
12324   unsigned NumElem = std::min(LastConsecutiveStore, LastConsecutiveLoad) + 1;
12325   NumElem = std::min(LastLegalType, NumElem);
12326 
12327   if (NumElem < 2)
12328     return false;
12329 
12330   // Collect the chains from all merged stores.
12331   SmallVector<SDValue, 8> MergeStoreChains;
12332   MergeStoreChains.push_back(StoreNodes[0].MemNode->getChain());
12333 
12334   // The latest Node in the DAG.
12335   unsigned LatestNodeUsed = 0;
12336   for (unsigned i=1; i<NumElem; ++i) {
12337     // Find a chain for the new wide-store operand. Notice that some
12338     // of the store nodes that we found may not be selected for inclusion
12339     // in the wide store. The chain we use needs to be the chain of the
12340     // latest store node which is *used* and replaced by the wide store.
12341     if (StoreNodes[i].SequenceNum < StoreNodes[LatestNodeUsed].SequenceNum)
12342       LatestNodeUsed = i;
12343 
12344     MergeStoreChains.push_back(StoreNodes[i].MemNode->getChain());
12345   }
12346 
12347   LSBaseSDNode *LatestOp = StoreNodes[LatestNodeUsed].MemNode;
12348 
12349   // Find if it is better to use vectors or integers to load and store
12350   // to memory.
12351   EVT JointMemOpVT;
12352   if (UseVectorTy) {
12353     JointMemOpVT = EVT::getVectorVT(Context, MemVT, NumElem);
12354   } else {
12355     unsigned SizeInBits = NumElem * ElementSizeBytes * 8;
12356     JointMemOpVT = EVT::getIntegerVT(Context, SizeInBits);
12357   }
12358 
12359   SDLoc LoadDL(LoadNodes[0].MemNode);
12360   SDLoc StoreDL(StoreNodes[0].MemNode);
12361 
12362   // The merged loads are required to have the same incoming chain, so
12363   // using the first's chain is acceptable.
12364   SDValue NewLoad = DAG.getLoad(JointMemOpVT, LoadDL, FirstLoad->getChain(),
12365                                 FirstLoad->getBasePtr(),
12366                                 FirstLoad->getPointerInfo(), FirstLoadAlign);
12367 
12368   SDValue NewStoreChain =
12369     DAG.getNode(ISD::TokenFactor, StoreDL, MVT::Other, MergeStoreChains);
12370 
12371   SDValue NewStore =
12372       DAG.getStore(NewStoreChain, StoreDL, NewLoad, FirstInChain->getBasePtr(),
12373                    FirstInChain->getPointerInfo(), FirstStoreAlign);
12374 
12375   // Transfer chain users from old loads to the new load.
12376   for (unsigned i = 0; i < NumElem; ++i) {
12377     LoadSDNode *Ld = cast<LoadSDNode>(LoadNodes[i].MemNode);
12378     DAG.ReplaceAllUsesOfValueWith(SDValue(Ld, 1),
12379                                   SDValue(NewLoad.getNode(), 1));
12380   }
12381 
12382   if (UseAA) {
12383     // Replace the all stores with the new store.
12384     for (unsigned i = 0; i < NumElem; ++i)
12385       CombineTo(StoreNodes[i].MemNode, NewStore);
12386   } else {
12387     // Replace the last store with the new store.
12388     CombineTo(LatestOp, NewStore);
12389     // Erase all other stores.
12390     for (unsigned i = 0; i < NumElem; ++i) {
12391       // Remove all Store nodes.
12392       if (StoreNodes[i].MemNode == LatestOp)
12393         continue;
12394       StoreSDNode *St = cast<StoreSDNode>(StoreNodes[i].MemNode);
12395       DAG.ReplaceAllUsesOfValueWith(SDValue(St, 0), St->getChain());
12396       deleteAndRecombine(St);
12397     }
12398   }
12399 
12400   StoreNodes.erase(StoreNodes.begin() + NumElem, StoreNodes.end());
12401   return true;
12402 }
12403 
12404 SDValue DAGCombiner::replaceStoreChain(StoreSDNode *ST, SDValue BetterChain) {
12405   SDLoc SL(ST);
12406   SDValue ReplStore;
12407 
12408   // Replace the chain to avoid dependency.
12409   if (ST->isTruncatingStore()) {
12410     ReplStore = DAG.getTruncStore(BetterChain, SL, ST->getValue(),
12411                                   ST->getBasePtr(), ST->getMemoryVT(),
12412                                   ST->getMemOperand());
12413   } else {
12414     ReplStore = DAG.getStore(BetterChain, SL, ST->getValue(), ST->getBasePtr(),
12415                              ST->getMemOperand());
12416   }
12417 
12418   // Create token to keep both nodes around.
12419   SDValue Token = DAG.getNode(ISD::TokenFactor, SL,
12420                               MVT::Other, ST->getChain(), ReplStore);
12421 
12422   // Make sure the new and old chains are cleaned up.
12423   AddToWorklist(Token.getNode());
12424 
12425   // Don't add users to work list.
12426   return CombineTo(ST, Token, false);
12427 }
12428 
12429 SDValue DAGCombiner::replaceStoreOfFPConstant(StoreSDNode *ST) {
12430   SDValue Value = ST->getValue();
12431   if (Value.getOpcode() == ISD::TargetConstantFP)
12432     return SDValue();
12433 
12434   SDLoc DL(ST);
12435 
12436   SDValue Chain = ST->getChain();
12437   SDValue Ptr = ST->getBasePtr();
12438 
12439   const ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Value);
12440 
12441   // NOTE: If the original store is volatile, this transform must not increase
12442   // the number of stores.  For example, on x86-32 an f64 can be stored in one
12443   // processor operation but an i64 (which is not legal) requires two.  So the
12444   // transform should not be done in this case.
12445 
12446   SDValue Tmp;
12447   switch (CFP->getSimpleValueType(0).SimpleTy) {
12448   default:
12449     llvm_unreachable("Unknown FP type");
12450   case MVT::f16:    // We don't do this for these yet.
12451   case MVT::f80:
12452   case MVT::f128:
12453   case MVT::ppcf128:
12454     return SDValue();
12455   case MVT::f32:
12456     if ((isTypeLegal(MVT::i32) && !LegalOperations && !ST->isVolatile()) ||
12457         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
12458       ;
12459       Tmp = DAG.getConstant((uint32_t)CFP->getValueAPF().
12460                             bitcastToAPInt().getZExtValue(), SDLoc(CFP),
12461                             MVT::i32);
12462       return DAG.getStore(Chain, DL, Tmp, Ptr, ST->getMemOperand());
12463     }
12464 
12465     return SDValue();
12466   case MVT::f64:
12467     if ((TLI.isTypeLegal(MVT::i64) && !LegalOperations &&
12468          !ST->isVolatile()) ||
12469         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i64)) {
12470       ;
12471       Tmp = DAG.getConstant(CFP->getValueAPF().bitcastToAPInt().
12472                             getZExtValue(), SDLoc(CFP), MVT::i64);
12473       return DAG.getStore(Chain, DL, Tmp,
12474                           Ptr, ST->getMemOperand());
12475     }
12476 
12477     if (!ST->isVolatile() &&
12478         TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) {
12479       // Many FP stores are not made apparent until after legalize, e.g. for
12480       // argument passing.  Since this is so common, custom legalize the
12481       // 64-bit integer store into two 32-bit stores.
12482       uint64_t Val = CFP->getValueAPF().bitcastToAPInt().getZExtValue();
12483       SDValue Lo = DAG.getConstant(Val & 0xFFFFFFFF, SDLoc(CFP), MVT::i32);
12484       SDValue Hi = DAG.getConstant(Val >> 32, SDLoc(CFP), MVT::i32);
12485       if (DAG.getDataLayout().isBigEndian())
12486         std::swap(Lo, Hi);
12487 
12488       unsigned Alignment = ST->getAlignment();
12489       MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
12490       AAMDNodes AAInfo = ST->getAAInfo();
12491 
12492       SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
12493                                  ST->getAlignment(), MMOFlags, AAInfo);
12494       Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
12495                         DAG.getConstant(4, DL, Ptr.getValueType()));
12496       Alignment = MinAlign(Alignment, 4U);
12497       SDValue St1 = DAG.getStore(Chain, DL, Hi, Ptr,
12498                                  ST->getPointerInfo().getWithOffset(4),
12499                                  Alignment, MMOFlags, AAInfo);
12500       return DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
12501                          St0, St1);
12502     }
12503 
12504     return SDValue();
12505   }
12506 }
12507 
12508 SDValue DAGCombiner::visitSTORE(SDNode *N) {
12509   StoreSDNode *ST  = cast<StoreSDNode>(N);
12510   SDValue Chain = ST->getChain();
12511   SDValue Value = ST->getValue();
12512   SDValue Ptr   = ST->getBasePtr();
12513 
12514   // If this is a store of a bit convert, store the input value if the
12515   // resultant store does not need a higher alignment than the original.
12516   if (Value.getOpcode() == ISD::BITCAST && !ST->isTruncatingStore() &&
12517       ST->isUnindexed()) {
12518     EVT SVT = Value.getOperand(0).getValueType();
12519     if (((!LegalOperations && !ST->isVolatile()) ||
12520          TLI.isOperationLegalOrCustom(ISD::STORE, SVT)) &&
12521         TLI.isStoreBitCastBeneficial(Value.getValueType(), SVT)) {
12522       unsigned OrigAlign = ST->getAlignment();
12523       bool Fast = false;
12524       if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), SVT,
12525                                  ST->getAddressSpace(), OrigAlign, &Fast) &&
12526           Fast) {
12527         return DAG.getStore(Chain, SDLoc(N), Value.getOperand(0), Ptr,
12528                             ST->getPointerInfo(), OrigAlign,
12529                             ST->getMemOperand()->getFlags(), ST->getAAInfo());
12530       }
12531     }
12532   }
12533 
12534   // Turn 'store undef, Ptr' -> nothing.
12535   if (Value.isUndef() && ST->isUnindexed())
12536     return Chain;
12537 
12538   // Try to infer better alignment information than the store already has.
12539   if (OptLevel != CodeGenOpt::None && ST->isUnindexed()) {
12540     if (unsigned Align = DAG.InferPtrAlignment(Ptr)) {
12541       if (Align > ST->getAlignment()) {
12542         SDValue NewStore =
12543             DAG.getTruncStore(Chain, SDLoc(N), Value, Ptr, ST->getPointerInfo(),
12544                               ST->getMemoryVT(), Align,
12545                               ST->getMemOperand()->getFlags(), ST->getAAInfo());
12546         if (NewStore.getNode() != N)
12547           return CombineTo(ST, NewStore, true);
12548       }
12549     }
12550   }
12551 
12552   // Try transforming a pair floating point load / store ops to integer
12553   // load / store ops.
12554   if (SDValue NewST = TransformFPLoadStorePair(N))
12555     return NewST;
12556 
12557   bool UseAA = CombinerAA.getNumOccurrences() > 0 ? CombinerAA
12558                                                   : DAG.getSubtarget().useAA();
12559 #ifndef NDEBUG
12560   if (CombinerAAOnlyFunc.getNumOccurrences() &&
12561       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
12562     UseAA = false;
12563 #endif
12564   if (UseAA && ST->isUnindexed()) {
12565     // FIXME: We should do this even without AA enabled. AA will just allow
12566     // FindBetterChain to work in more situations. The problem with this is that
12567     // any combine that expects memory operations to be on consecutive chains
12568     // first needs to be updated to look for users of the same chain.
12569 
12570     // Walk up chain skipping non-aliasing memory nodes, on this store and any
12571     // adjacent stores.
12572     if (findBetterNeighborChains(ST)) {
12573       // replaceStoreChain uses CombineTo, which handled all of the worklist
12574       // manipulation. Return the original node to not do anything else.
12575       return SDValue(ST, 0);
12576     }
12577     Chain = ST->getChain();
12578   }
12579 
12580   // Try transforming N to an indexed store.
12581   if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N))
12582     return SDValue(N, 0);
12583 
12584   // FIXME: is there such a thing as a truncating indexed store?
12585   if (ST->isTruncatingStore() && ST->isUnindexed() &&
12586       Value.getValueType().isInteger()) {
12587     // See if we can simplify the input to this truncstore with knowledge that
12588     // only the low bits are being used.  For example:
12589     // "truncstore (or (shl x, 8), y), i8"  -> "truncstore y, i8"
12590     SDValue Shorter = GetDemandedBits(
12591         Value, APInt::getLowBitsSet(Value.getScalarValueSizeInBits(),
12592                                     ST->getMemoryVT().getScalarSizeInBits()));
12593     AddToWorklist(Value.getNode());
12594     if (Shorter.getNode())
12595       return DAG.getTruncStore(Chain, SDLoc(N), Shorter,
12596                                Ptr, ST->getMemoryVT(), ST->getMemOperand());
12597 
12598     // Otherwise, see if we can simplify the operation with
12599     // SimplifyDemandedBits, which only works if the value has a single use.
12600     if (SimplifyDemandedBits(
12601             Value,
12602             APInt::getLowBitsSet(Value.getScalarValueSizeInBits(),
12603                                  ST->getMemoryVT().getScalarSizeInBits())))
12604       return SDValue(N, 0);
12605   }
12606 
12607   // If this is a load followed by a store to the same location, then the store
12608   // is dead/noop.
12609   if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Value)) {
12610     if (Ld->getBasePtr() == Ptr && ST->getMemoryVT() == Ld->getMemoryVT() &&
12611         ST->isUnindexed() && !ST->isVolatile() &&
12612         // There can't be any side effects between the load and store, such as
12613         // a call or store.
12614         Chain.reachesChainWithoutSideEffects(SDValue(Ld, 1))) {
12615       // The store is dead, remove it.
12616       return Chain;
12617     }
12618   }
12619 
12620   // If this is a store followed by a store with the same value to the same
12621   // location, then the store is dead/noop.
12622   if (StoreSDNode *ST1 = dyn_cast<StoreSDNode>(Chain)) {
12623     if (ST1->getBasePtr() == Ptr && ST->getMemoryVT() == ST1->getMemoryVT() &&
12624         ST1->getValue() == Value && ST->isUnindexed() && !ST->isVolatile() &&
12625         ST1->isUnindexed() && !ST1->isVolatile()) {
12626       // The store is dead, remove it.
12627       return Chain;
12628     }
12629   }
12630 
12631   // If this is an FP_ROUND or TRUNC followed by a store, fold this into a
12632   // truncating store.  We can do this even if this is already a truncstore.
12633   if ((Value.getOpcode() == ISD::FP_ROUND || Value.getOpcode() == ISD::TRUNCATE)
12634       && Value.getNode()->hasOneUse() && ST->isUnindexed() &&
12635       TLI.isTruncStoreLegal(Value.getOperand(0).getValueType(),
12636                             ST->getMemoryVT())) {
12637     return DAG.getTruncStore(Chain, SDLoc(N), Value.getOperand(0),
12638                              Ptr, ST->getMemoryVT(), ST->getMemOperand());
12639   }
12640 
12641   // Only perform this optimization before the types are legal, because we
12642   // don't want to perform this optimization on every DAGCombine invocation.
12643   if (!LegalTypes) {
12644     for (;;) {
12645       // There can be multiple store sequences on the same chain.
12646       // Keep trying to merge store sequences until we are unable to do so
12647       // or until we merge the last store on the chain.
12648       SmallVector<MemOpLink, 8> StoreNodes;
12649       bool Changed = MergeConsecutiveStores(ST, StoreNodes);
12650       if (!Changed) break;
12651 
12652       if (any_of(StoreNodes,
12653                  [ST](const MemOpLink &Link) { return Link.MemNode == ST; })) {
12654         // ST has been merged and no longer exists.
12655         return SDValue(N, 0);
12656       }
12657     }
12658   }
12659 
12660   // Turn 'store float 1.0, Ptr' -> 'store int 0x12345678, Ptr'
12661   //
12662   // Make sure to do this only after attempting to merge stores in order to
12663   //  avoid changing the types of some subset of stores due to visit order,
12664   //  preventing their merging.
12665   if (isa<ConstantFPSDNode>(Value)) {
12666     if (SDValue NewSt = replaceStoreOfFPConstant(ST))
12667       return NewSt;
12668   }
12669 
12670   if (SDValue NewSt = splitMergedValStore(ST))
12671     return NewSt;
12672 
12673   return ReduceLoadOpStoreWidth(N);
12674 }
12675 
12676 /// For the instruction sequence of store below, F and I values
12677 /// are bundled together as an i64 value before being stored into memory.
12678 /// Sometimes it is more efficent to generate separate stores for F and I,
12679 /// which can remove the bitwise instructions or sink them to colder places.
12680 ///
12681 ///   (store (or (zext (bitcast F to i32) to i64),
12682 ///              (shl (zext I to i64), 32)), addr)  -->
12683 ///   (store F, addr) and (store I, addr+4)
12684 ///
12685 /// Similarly, splitting for other merged store can also be beneficial, like:
12686 /// For pair of {i32, i32}, i64 store --> two i32 stores.
12687 /// For pair of {i32, i16}, i64 store --> two i32 stores.
12688 /// For pair of {i16, i16}, i32 store --> two i16 stores.
12689 /// For pair of {i16, i8},  i32 store --> two i16 stores.
12690 /// For pair of {i8, i8},   i16 store --> two i8 stores.
12691 ///
12692 /// We allow each target to determine specifically which kind of splitting is
12693 /// supported.
12694 ///
12695 /// The store patterns are commonly seen from the simple code snippet below
12696 /// if only std::make_pair(...) is sroa transformed before inlined into hoo.
12697 ///   void goo(const std::pair<int, float> &);
12698 ///   hoo() {
12699 ///     ...
12700 ///     goo(std::make_pair(tmp, ftmp));
12701 ///     ...
12702 ///   }
12703 ///
12704 SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) {
12705   if (OptLevel == CodeGenOpt::None)
12706     return SDValue();
12707 
12708   SDValue Val = ST->getValue();
12709   SDLoc DL(ST);
12710 
12711   // Match OR operand.
12712   if (!Val.getValueType().isScalarInteger() || Val.getOpcode() != ISD::OR)
12713     return SDValue();
12714 
12715   // Match SHL operand and get Lower and Higher parts of Val.
12716   SDValue Op1 = Val.getOperand(0);
12717   SDValue Op2 = Val.getOperand(1);
12718   SDValue Lo, Hi;
12719   if (Op1.getOpcode() != ISD::SHL) {
12720     std::swap(Op1, Op2);
12721     if (Op1.getOpcode() != ISD::SHL)
12722       return SDValue();
12723   }
12724   Lo = Op2;
12725   Hi = Op1.getOperand(0);
12726   if (!Op1.hasOneUse())
12727     return SDValue();
12728 
12729   // Match shift amount to HalfValBitSize.
12730   unsigned HalfValBitSize = Val.getValueSizeInBits() / 2;
12731   ConstantSDNode *ShAmt = dyn_cast<ConstantSDNode>(Op1.getOperand(1));
12732   if (!ShAmt || ShAmt->getAPIntValue() != HalfValBitSize)
12733     return SDValue();
12734 
12735   // Lo and Hi are zero-extended from int with size less equal than 32
12736   // to i64.
12737   if (Lo.getOpcode() != ISD::ZERO_EXTEND || !Lo.hasOneUse() ||
12738       !Lo.getOperand(0).getValueType().isScalarInteger() ||
12739       Lo.getOperand(0).getValueSizeInBits() > HalfValBitSize ||
12740       Hi.getOpcode() != ISD::ZERO_EXTEND || !Hi.hasOneUse() ||
12741       !Hi.getOperand(0).getValueType().isScalarInteger() ||
12742       Hi.getOperand(0).getValueSizeInBits() > HalfValBitSize)
12743     return SDValue();
12744 
12745   // Use the EVT of low and high parts before bitcast as the input
12746   // of target query.
12747   EVT LowTy = (Lo.getOperand(0).getOpcode() == ISD::BITCAST)
12748                   ? Lo.getOperand(0).getValueType()
12749                   : Lo.getValueType();
12750   EVT HighTy = (Hi.getOperand(0).getOpcode() == ISD::BITCAST)
12751                    ? Hi.getOperand(0).getValueType()
12752                    : Hi.getValueType();
12753   if (!TLI.isMultiStoresCheaperThanBitsMerge(LowTy, HighTy))
12754     return SDValue();
12755 
12756   // Start to split store.
12757   unsigned Alignment = ST->getAlignment();
12758   MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags();
12759   AAMDNodes AAInfo = ST->getAAInfo();
12760 
12761   // Change the sizes of Lo and Hi's value types to HalfValBitSize.
12762   EVT VT = EVT::getIntegerVT(*DAG.getContext(), HalfValBitSize);
12763   Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Lo.getOperand(0));
12764   Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Hi.getOperand(0));
12765 
12766   SDValue Chain = ST->getChain();
12767   SDValue Ptr = ST->getBasePtr();
12768   // Lower value store.
12769   SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(),
12770                              ST->getAlignment(), MMOFlags, AAInfo);
12771   Ptr =
12772       DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
12773                   DAG.getConstant(HalfValBitSize / 8, DL, Ptr.getValueType()));
12774   // Higher value store.
12775   SDValue St1 =
12776       DAG.getStore(St0, DL, Hi, Ptr,
12777                    ST->getPointerInfo().getWithOffset(HalfValBitSize / 8),
12778                    Alignment / 2, MMOFlags, AAInfo);
12779   return St1;
12780 }
12781 
12782 SDValue DAGCombiner::visitINSERT_VECTOR_ELT(SDNode *N) {
12783   SDValue InVec = N->getOperand(0);
12784   SDValue InVal = N->getOperand(1);
12785   SDValue EltNo = N->getOperand(2);
12786   SDLoc DL(N);
12787 
12788   // If the inserted element is an UNDEF, just use the input vector.
12789   if (InVal.isUndef())
12790     return InVec;
12791 
12792   EVT VT = InVec.getValueType();
12793 
12794   // Check that we know which element is being inserted
12795   if (!isa<ConstantSDNode>(EltNo))
12796     return SDValue();
12797   unsigned Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
12798 
12799   // Canonicalize insert_vector_elt dag nodes.
12800   // Example:
12801   // (insert_vector_elt (insert_vector_elt A, Idx0), Idx1)
12802   // -> (insert_vector_elt (insert_vector_elt A, Idx1), Idx0)
12803   //
12804   // Do this only if the child insert_vector node has one use; also
12805   // do this only if indices are both constants and Idx1 < Idx0.
12806   if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT && InVec.hasOneUse()
12807       && isa<ConstantSDNode>(InVec.getOperand(2))) {
12808     unsigned OtherElt =
12809       cast<ConstantSDNode>(InVec.getOperand(2))->getZExtValue();
12810     if (Elt < OtherElt) {
12811       // Swap nodes.
12812       SDValue NewOp = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT,
12813                                   InVec.getOperand(0), InVal, EltNo);
12814       AddToWorklist(NewOp.getNode());
12815       return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(InVec.getNode()),
12816                          VT, NewOp, InVec.getOperand(1), InVec.getOperand(2));
12817     }
12818   }
12819 
12820   // If we can't generate a legal BUILD_VECTOR, exit
12821   if (LegalOperations && !TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))
12822     return SDValue();
12823 
12824   // Check that the operand is a BUILD_VECTOR (or UNDEF, which can essentially
12825   // be converted to a BUILD_VECTOR).  Fill in the Ops vector with the
12826   // vector elements.
12827   SmallVector<SDValue, 8> Ops;
12828   // Do not combine these two vectors if the output vector will not replace
12829   // the input vector.
12830   if (InVec.getOpcode() == ISD::BUILD_VECTOR && InVec.hasOneUse()) {
12831     Ops.append(InVec.getNode()->op_begin(),
12832                InVec.getNode()->op_end());
12833   } else if (InVec.isUndef()) {
12834     unsigned NElts = VT.getVectorNumElements();
12835     Ops.append(NElts, DAG.getUNDEF(InVal.getValueType()));
12836   } else {
12837     return SDValue();
12838   }
12839 
12840   // Insert the element
12841   if (Elt < Ops.size()) {
12842     // All the operands of BUILD_VECTOR must have the same type;
12843     // we enforce that here.
12844     EVT OpVT = Ops[0].getValueType();
12845     Ops[Elt] = OpVT.isInteger() ? DAG.getAnyExtOrTrunc(InVal, DL, OpVT) : InVal;
12846   }
12847 
12848   // Return the new vector
12849   return DAG.getBuildVector(VT, DL, Ops);
12850 }
12851 
12852 SDValue DAGCombiner::ReplaceExtractVectorEltOfLoadWithNarrowedLoad(
12853     SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad) {
12854   assert(!OriginalLoad->isVolatile());
12855 
12856   EVT ResultVT = EVE->getValueType(0);
12857   EVT VecEltVT = InVecVT.getVectorElementType();
12858   unsigned Align = OriginalLoad->getAlignment();
12859   unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment(
12860       VecEltVT.getTypeForEVT(*DAG.getContext()));
12861 
12862   if (NewAlign > Align || !TLI.isOperationLegalOrCustom(ISD::LOAD, VecEltVT))
12863     return SDValue();
12864 
12865   ISD::LoadExtType ExtTy = ResultVT.bitsGT(VecEltVT) ?
12866     ISD::NON_EXTLOAD : ISD::EXTLOAD;
12867   if (!TLI.shouldReduceLoadWidth(OriginalLoad, ExtTy, VecEltVT))
12868     return SDValue();
12869 
12870   Align = NewAlign;
12871 
12872   SDValue NewPtr = OriginalLoad->getBasePtr();
12873   SDValue Offset;
12874   EVT PtrType = NewPtr.getValueType();
12875   MachinePointerInfo MPI;
12876   SDLoc DL(EVE);
12877   if (auto *ConstEltNo = dyn_cast<ConstantSDNode>(EltNo)) {
12878     int Elt = ConstEltNo->getZExtValue();
12879     unsigned PtrOff = VecEltVT.getSizeInBits() * Elt / 8;
12880     Offset = DAG.getConstant(PtrOff, DL, PtrType);
12881     MPI = OriginalLoad->getPointerInfo().getWithOffset(PtrOff);
12882   } else {
12883     Offset = DAG.getZExtOrTrunc(EltNo, DL, PtrType);
12884     Offset = DAG.getNode(
12885         ISD::MUL, DL, PtrType, Offset,
12886         DAG.getConstant(VecEltVT.getStoreSize(), DL, PtrType));
12887     MPI = OriginalLoad->getPointerInfo();
12888   }
12889   NewPtr = DAG.getNode(ISD::ADD, DL, PtrType, NewPtr, Offset);
12890 
12891   // The replacement we need to do here is a little tricky: we need to
12892   // replace an extractelement of a load with a load.
12893   // Use ReplaceAllUsesOfValuesWith to do the replacement.
12894   // Note that this replacement assumes that the extractvalue is the only
12895   // use of the load; that's okay because we don't want to perform this
12896   // transformation in other cases anyway.
12897   SDValue Load;
12898   SDValue Chain;
12899   if (ResultVT.bitsGT(VecEltVT)) {
12900     // If the result type of vextract is wider than the load, then issue an
12901     // extending load instead.
12902     ISD::LoadExtType ExtType = TLI.isLoadExtLegal(ISD::ZEXTLOAD, ResultVT,
12903                                                   VecEltVT)
12904                                    ? ISD::ZEXTLOAD
12905                                    : ISD::EXTLOAD;
12906     Load = DAG.getExtLoad(ExtType, SDLoc(EVE), ResultVT,
12907                           OriginalLoad->getChain(), NewPtr, MPI, VecEltVT,
12908                           Align, OriginalLoad->getMemOperand()->getFlags(),
12909                           OriginalLoad->getAAInfo());
12910     Chain = Load.getValue(1);
12911   } else {
12912     Load = DAG.getLoad(VecEltVT, SDLoc(EVE), OriginalLoad->getChain(), NewPtr,
12913                        MPI, Align, OriginalLoad->getMemOperand()->getFlags(),
12914                        OriginalLoad->getAAInfo());
12915     Chain = Load.getValue(1);
12916     if (ResultVT.bitsLT(VecEltVT))
12917       Load = DAG.getNode(ISD::TRUNCATE, SDLoc(EVE), ResultVT, Load);
12918     else
12919       Load = DAG.getBitcast(ResultVT, Load);
12920   }
12921   WorklistRemover DeadNodes(*this);
12922   SDValue From[] = { SDValue(EVE, 0), SDValue(OriginalLoad, 1) };
12923   SDValue To[] = { Load, Chain };
12924   DAG.ReplaceAllUsesOfValuesWith(From, To, 2);
12925   // Since we're explicitly calling ReplaceAllUses, add the new node to the
12926   // worklist explicitly as well.
12927   AddToWorklist(Load.getNode());
12928   AddUsersToWorklist(Load.getNode()); // Add users too
12929   // Make sure to revisit this node to clean it up; it will usually be dead.
12930   AddToWorklist(EVE);
12931   ++OpsNarrowed;
12932   return SDValue(EVE, 0);
12933 }
12934 
12935 SDValue DAGCombiner::visitEXTRACT_VECTOR_ELT(SDNode *N) {
12936   // (vextract (scalar_to_vector val, 0) -> val
12937   SDValue InVec = N->getOperand(0);
12938   EVT VT = InVec.getValueType();
12939   EVT NVT = N->getValueType(0);
12940 
12941   if (InVec.getOpcode() == ISD::SCALAR_TO_VECTOR) {
12942     // Check if the result type doesn't match the inserted element type. A
12943     // SCALAR_TO_VECTOR may truncate the inserted element and the
12944     // EXTRACT_VECTOR_ELT may widen the extracted vector.
12945     SDValue InOp = InVec.getOperand(0);
12946     if (InOp.getValueType() != NVT) {
12947       assert(InOp.getValueType().isInteger() && NVT.isInteger());
12948       return DAG.getSExtOrTrunc(InOp, SDLoc(InVec), NVT);
12949     }
12950     return InOp;
12951   }
12952 
12953   SDValue EltNo = N->getOperand(1);
12954   ConstantSDNode *ConstEltNo = dyn_cast<ConstantSDNode>(EltNo);
12955 
12956   // extract_vector_elt (build_vector x, y), 1 -> y
12957   if (ConstEltNo &&
12958       InVec.getOpcode() == ISD::BUILD_VECTOR &&
12959       TLI.isTypeLegal(VT) &&
12960       (InVec.hasOneUse() ||
12961        TLI.aggressivelyPreferBuildVectorSources(VT))) {
12962     SDValue Elt = InVec.getOperand(ConstEltNo->getZExtValue());
12963     EVT InEltVT = Elt.getValueType();
12964 
12965     // Sometimes build_vector's scalar input types do not match result type.
12966     if (NVT == InEltVT)
12967       return Elt;
12968 
12969     // TODO: It may be useful to truncate if free if the build_vector implicitly
12970     // converts.
12971   }
12972 
12973   // extract_vector_elt (v2i32 (bitcast i64:x)), 0 -> i32 (trunc i64:x)
12974   if (ConstEltNo && InVec.getOpcode() == ISD::BITCAST && InVec.hasOneUse() &&
12975       ConstEltNo->isNullValue() && VT.isInteger()) {
12976     SDValue BCSrc = InVec.getOperand(0);
12977     if (BCSrc.getValueType().isScalarInteger())
12978       return DAG.getNode(ISD::TRUNCATE, SDLoc(N), NVT, BCSrc);
12979   }
12980 
12981   // extract_vector_elt (insert_vector_elt vec, val, idx), idx) -> val
12982   //
12983   // This only really matters if the index is non-constant since other combines
12984   // on the constant elements already work.
12985   if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT &&
12986       EltNo == InVec.getOperand(2)) {
12987     SDValue Elt = InVec.getOperand(1);
12988     return VT.isInteger() ? DAG.getAnyExtOrTrunc(Elt, SDLoc(N), NVT) : Elt;
12989   }
12990 
12991   // Transform: (EXTRACT_VECTOR_ELT( VECTOR_SHUFFLE )) -> EXTRACT_VECTOR_ELT.
12992   // We only perform this optimization before the op legalization phase because
12993   // we may introduce new vector instructions which are not backed by TD
12994   // patterns. For example on AVX, extracting elements from a wide vector
12995   // without using extract_subvector. However, if we can find an underlying
12996   // scalar value, then we can always use that.
12997   if (ConstEltNo && InVec.getOpcode() == ISD::VECTOR_SHUFFLE) {
12998     int NumElem = VT.getVectorNumElements();
12999     ShuffleVectorSDNode *SVOp = cast<ShuffleVectorSDNode>(InVec);
13000     // Find the new index to extract from.
13001     int OrigElt = SVOp->getMaskElt(ConstEltNo->getZExtValue());
13002 
13003     // Extracting an undef index is undef.
13004     if (OrigElt == -1)
13005       return DAG.getUNDEF(NVT);
13006 
13007     // Select the right vector half to extract from.
13008     SDValue SVInVec;
13009     if (OrigElt < NumElem) {
13010       SVInVec = InVec->getOperand(0);
13011     } else {
13012       SVInVec = InVec->getOperand(1);
13013       OrigElt -= NumElem;
13014     }
13015 
13016     if (SVInVec.getOpcode() == ISD::BUILD_VECTOR) {
13017       SDValue InOp = SVInVec.getOperand(OrigElt);
13018       if (InOp.getValueType() != NVT) {
13019         assert(InOp.getValueType().isInteger() && NVT.isInteger());
13020         InOp = DAG.getSExtOrTrunc(InOp, SDLoc(SVInVec), NVT);
13021       }
13022 
13023       return InOp;
13024     }
13025 
13026     // FIXME: We should handle recursing on other vector shuffles and
13027     // scalar_to_vector here as well.
13028 
13029     if (!LegalOperations) {
13030       EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout());
13031       return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(N), NVT, SVInVec,
13032                          DAG.getConstant(OrigElt, SDLoc(SVOp), IndexTy));
13033     }
13034   }
13035 
13036   bool BCNumEltsChanged = false;
13037   EVT ExtVT = VT.getVectorElementType();
13038   EVT LVT = ExtVT;
13039 
13040   // If the result of load has to be truncated, then it's not necessarily
13041   // profitable.
13042   if (NVT.bitsLT(LVT) && !TLI.isTruncateFree(LVT, NVT))
13043     return SDValue();
13044 
13045   if (InVec.getOpcode() == ISD::BITCAST) {
13046     // Don't duplicate a load with other uses.
13047     if (!InVec.hasOneUse())
13048       return SDValue();
13049 
13050     EVT BCVT = InVec.getOperand(0).getValueType();
13051     if (!BCVT.isVector() || ExtVT.bitsGT(BCVT.getVectorElementType()))
13052       return SDValue();
13053     if (VT.getVectorNumElements() != BCVT.getVectorNumElements())
13054       BCNumEltsChanged = true;
13055     InVec = InVec.getOperand(0);
13056     ExtVT = BCVT.getVectorElementType();
13057   }
13058 
13059   // (vextract (vN[if]M load $addr), i) -> ([if]M load $addr + i * size)
13060   if (!LegalOperations && !ConstEltNo && InVec.hasOneUse() &&
13061       ISD::isNormalLoad(InVec.getNode()) &&
13062       !N->getOperand(1)->hasPredecessor(InVec.getNode())) {
13063     SDValue Index = N->getOperand(1);
13064     if (LoadSDNode *OrigLoad = dyn_cast<LoadSDNode>(InVec)) {
13065       if (!OrigLoad->isVolatile()) {
13066         return ReplaceExtractVectorEltOfLoadWithNarrowedLoad(N, VT, Index,
13067                                                              OrigLoad);
13068       }
13069     }
13070   }
13071 
13072   // Perform only after legalization to ensure build_vector / vector_shuffle
13073   // optimizations have already been done.
13074   if (!LegalOperations) return SDValue();
13075 
13076   // (vextract (v4f32 load $addr), c) -> (f32 load $addr+c*size)
13077   // (vextract (v4f32 s2v (f32 load $addr)), c) -> (f32 load $addr+c*size)
13078   // (vextract (v4f32 shuffle (load $addr), <1,u,u,u>), 0) -> (f32 load $addr)
13079 
13080   if (ConstEltNo) {
13081     int Elt = cast<ConstantSDNode>(EltNo)->getZExtValue();
13082 
13083     LoadSDNode *LN0 = nullptr;
13084     const ShuffleVectorSDNode *SVN = nullptr;
13085     if (ISD::isNormalLoad(InVec.getNode())) {
13086       LN0 = cast<LoadSDNode>(InVec);
13087     } else if (InVec.getOpcode() == ISD::SCALAR_TO_VECTOR &&
13088                InVec.getOperand(0).getValueType() == ExtVT &&
13089                ISD::isNormalLoad(InVec.getOperand(0).getNode())) {
13090       // Don't duplicate a load with other uses.
13091       if (!InVec.hasOneUse())
13092         return SDValue();
13093 
13094       LN0 = cast<LoadSDNode>(InVec.getOperand(0));
13095     } else if ((SVN = dyn_cast<ShuffleVectorSDNode>(InVec))) {
13096       // (vextract (vector_shuffle (load $addr), v2, <1, u, u, u>), 1)
13097       // =>
13098       // (load $addr+1*size)
13099 
13100       // Don't duplicate a load with other uses.
13101       if (!InVec.hasOneUse())
13102         return SDValue();
13103 
13104       // If the bit convert changed the number of elements, it is unsafe
13105       // to examine the mask.
13106       if (BCNumEltsChanged)
13107         return SDValue();
13108 
13109       // Select the input vector, guarding against out of range extract vector.
13110       unsigned NumElems = VT.getVectorNumElements();
13111       int Idx = (Elt > (int)NumElems) ? -1 : SVN->getMaskElt(Elt);
13112       InVec = (Idx < (int)NumElems) ? InVec.getOperand(0) : InVec.getOperand(1);
13113 
13114       if (InVec.getOpcode() == ISD::BITCAST) {
13115         // Don't duplicate a load with other uses.
13116         if (!InVec.hasOneUse())
13117           return SDValue();
13118 
13119         InVec = InVec.getOperand(0);
13120       }
13121       if (ISD::isNormalLoad(InVec.getNode())) {
13122         LN0 = cast<LoadSDNode>(InVec);
13123         Elt = (Idx < (int)NumElems) ? Idx : Idx - (int)NumElems;
13124         EltNo = DAG.getConstant(Elt, SDLoc(EltNo), EltNo.getValueType());
13125       }
13126     }
13127 
13128     // Make sure we found a non-volatile load and the extractelement is
13129     // the only use.
13130     if (!LN0 || !LN0->hasNUsesOfValue(1,0) || LN0->isVolatile())
13131       return SDValue();
13132 
13133     // If Idx was -1 above, Elt is going to be -1, so just return undef.
13134     if (Elt == -1)
13135       return DAG.getUNDEF(LVT);
13136 
13137     return ReplaceExtractVectorEltOfLoadWithNarrowedLoad(N, VT, EltNo, LN0);
13138   }
13139 
13140   return SDValue();
13141 }
13142 
13143 // Simplify (build_vec (ext )) to (bitcast (build_vec ))
13144 SDValue DAGCombiner::reduceBuildVecExtToExtBuildVec(SDNode *N) {
13145   // We perform this optimization post type-legalization because
13146   // the type-legalizer often scalarizes integer-promoted vectors.
13147   // Performing this optimization before may create bit-casts which
13148   // will be type-legalized to complex code sequences.
13149   // We perform this optimization only before the operation legalizer because we
13150   // may introduce illegal operations.
13151   if (Level != AfterLegalizeVectorOps && Level != AfterLegalizeTypes)
13152     return SDValue();
13153 
13154   unsigned NumInScalars = N->getNumOperands();
13155   SDLoc DL(N);
13156   EVT VT = N->getValueType(0);
13157 
13158   // Check to see if this is a BUILD_VECTOR of a bunch of values
13159   // which come from any_extend or zero_extend nodes. If so, we can create
13160   // a new BUILD_VECTOR using bit-casts which may enable other BUILD_VECTOR
13161   // optimizations. We do not handle sign-extend because we can't fill the sign
13162   // using shuffles.
13163   EVT SourceType = MVT::Other;
13164   bool AllAnyExt = true;
13165 
13166   for (unsigned i = 0; i != NumInScalars; ++i) {
13167     SDValue In = N->getOperand(i);
13168     // Ignore undef inputs.
13169     if (In.isUndef()) continue;
13170 
13171     bool AnyExt  = In.getOpcode() == ISD::ANY_EXTEND;
13172     bool ZeroExt = In.getOpcode() == ISD::ZERO_EXTEND;
13173 
13174     // Abort if the element is not an extension.
13175     if (!ZeroExt && !AnyExt) {
13176       SourceType = MVT::Other;
13177       break;
13178     }
13179 
13180     // The input is a ZeroExt or AnyExt. Check the original type.
13181     EVT InTy = In.getOperand(0).getValueType();
13182 
13183     // Check that all of the widened source types are the same.
13184     if (SourceType == MVT::Other)
13185       // First time.
13186       SourceType = InTy;
13187     else if (InTy != SourceType) {
13188       // Multiple income types. Abort.
13189       SourceType = MVT::Other;
13190       break;
13191     }
13192 
13193     // Check if all of the extends are ANY_EXTENDs.
13194     AllAnyExt &= AnyExt;
13195   }
13196 
13197   // In order to have valid types, all of the inputs must be extended from the
13198   // same source type and all of the inputs must be any or zero extend.
13199   // Scalar sizes must be a power of two.
13200   EVT OutScalarTy = VT.getScalarType();
13201   bool ValidTypes = SourceType != MVT::Other &&
13202                  isPowerOf2_32(OutScalarTy.getSizeInBits()) &&
13203                  isPowerOf2_32(SourceType.getSizeInBits());
13204 
13205   // Create a new simpler BUILD_VECTOR sequence which other optimizations can
13206   // turn into a single shuffle instruction.
13207   if (!ValidTypes)
13208     return SDValue();
13209 
13210   bool isLE = DAG.getDataLayout().isLittleEndian();
13211   unsigned ElemRatio = OutScalarTy.getSizeInBits()/SourceType.getSizeInBits();
13212   assert(ElemRatio > 1 && "Invalid element size ratio");
13213   SDValue Filler = AllAnyExt ? DAG.getUNDEF(SourceType):
13214                                DAG.getConstant(0, DL, SourceType);
13215 
13216   unsigned NewBVElems = ElemRatio * VT.getVectorNumElements();
13217   SmallVector<SDValue, 8> Ops(NewBVElems, Filler);
13218 
13219   // Populate the new build_vector
13220   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
13221     SDValue Cast = N->getOperand(i);
13222     assert((Cast.getOpcode() == ISD::ANY_EXTEND ||
13223             Cast.getOpcode() == ISD::ZERO_EXTEND ||
13224             Cast.isUndef()) && "Invalid cast opcode");
13225     SDValue In;
13226     if (Cast.isUndef())
13227       In = DAG.getUNDEF(SourceType);
13228     else
13229       In = Cast->getOperand(0);
13230     unsigned Index = isLE ? (i * ElemRatio) :
13231                             (i * ElemRatio + (ElemRatio - 1));
13232 
13233     assert(Index < Ops.size() && "Invalid index");
13234     Ops[Index] = In;
13235   }
13236 
13237   // The type of the new BUILD_VECTOR node.
13238   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SourceType, NewBVElems);
13239   assert(VecVT.getSizeInBits() == VT.getSizeInBits() &&
13240          "Invalid vector size");
13241   // Check if the new vector type is legal.
13242   if (!isTypeLegal(VecVT)) return SDValue();
13243 
13244   // Make the new BUILD_VECTOR.
13245   SDValue BV = DAG.getBuildVector(VecVT, DL, Ops);
13246 
13247   // The new BUILD_VECTOR node has the potential to be further optimized.
13248   AddToWorklist(BV.getNode());
13249   // Bitcast to the desired type.
13250   return DAG.getBitcast(VT, BV);
13251 }
13252 
13253 SDValue DAGCombiner::reduceBuildVecConvertToConvertBuildVec(SDNode *N) {
13254   EVT VT = N->getValueType(0);
13255 
13256   unsigned NumInScalars = N->getNumOperands();
13257   SDLoc DL(N);
13258 
13259   EVT SrcVT = MVT::Other;
13260   unsigned Opcode = ISD::DELETED_NODE;
13261   unsigned NumDefs = 0;
13262 
13263   for (unsigned i = 0; i != NumInScalars; ++i) {
13264     SDValue In = N->getOperand(i);
13265     unsigned Opc = In.getOpcode();
13266 
13267     if (Opc == ISD::UNDEF)
13268       continue;
13269 
13270     // If all scalar values are floats and converted from integers.
13271     if (Opcode == ISD::DELETED_NODE &&
13272         (Opc == ISD::UINT_TO_FP || Opc == ISD::SINT_TO_FP)) {
13273       Opcode = Opc;
13274     }
13275 
13276     if (Opc != Opcode)
13277       return SDValue();
13278 
13279     EVT InVT = In.getOperand(0).getValueType();
13280 
13281     // If all scalar values are typed differently, bail out. It's chosen to
13282     // simplify BUILD_VECTOR of integer types.
13283     if (SrcVT == MVT::Other)
13284       SrcVT = InVT;
13285     if (SrcVT != InVT)
13286       return SDValue();
13287     NumDefs++;
13288   }
13289 
13290   // If the vector has just one element defined, it's not worth to fold it into
13291   // a vectorized one.
13292   if (NumDefs < 2)
13293     return SDValue();
13294 
13295   assert((Opcode == ISD::UINT_TO_FP || Opcode == ISD::SINT_TO_FP)
13296          && "Should only handle conversion from integer to float.");
13297   assert(SrcVT != MVT::Other && "Cannot determine source type!");
13298 
13299   EVT NVT = EVT::getVectorVT(*DAG.getContext(), SrcVT, NumInScalars);
13300 
13301   if (!TLI.isOperationLegalOrCustom(Opcode, NVT))
13302     return SDValue();
13303 
13304   // Just because the floating-point vector type is legal does not necessarily
13305   // mean that the corresponding integer vector type is.
13306   if (!isTypeLegal(NVT))
13307     return SDValue();
13308 
13309   SmallVector<SDValue, 8> Opnds;
13310   for (unsigned i = 0; i != NumInScalars; ++i) {
13311     SDValue In = N->getOperand(i);
13312 
13313     if (In.isUndef())
13314       Opnds.push_back(DAG.getUNDEF(SrcVT));
13315     else
13316       Opnds.push_back(In.getOperand(0));
13317   }
13318   SDValue BV = DAG.getBuildVector(NVT, DL, Opnds);
13319   AddToWorklist(BV.getNode());
13320 
13321   return DAG.getNode(Opcode, DL, VT, BV);
13322 }
13323 
13324 SDValue DAGCombiner::createBuildVecShuffle(const SDLoc &DL, SDNode *N,
13325                                            ArrayRef<int> VectorMask,
13326                                            SDValue VecIn1, SDValue VecIn2,
13327                                            unsigned LeftIdx) {
13328   MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout());
13329   SDValue ZeroIdx = DAG.getConstant(0, DL, IdxTy);
13330 
13331   EVT VT = N->getValueType(0);
13332   EVT InVT1 = VecIn1.getValueType();
13333   EVT InVT2 = VecIn2.getNode() ? VecIn2.getValueType() : InVT1;
13334 
13335   unsigned Vec2Offset = InVT1.getVectorNumElements();
13336   unsigned NumElems = VT.getVectorNumElements();
13337   unsigned ShuffleNumElems = NumElems;
13338 
13339   // We can't generate a shuffle node with mismatched input and output types.
13340   // Try to make the types match the type of the output.
13341   if (InVT1 != VT || InVT2 != VT) {
13342     if ((VT.getSizeInBits() % InVT1.getSizeInBits() == 0) && InVT1 == InVT2) {
13343       // If the output vector length is a multiple of both input lengths,
13344       // we can concatenate them and pad the rest with undefs.
13345       unsigned NumConcats = VT.getSizeInBits() / InVT1.getSizeInBits();
13346       assert(NumConcats >= 2 && "Concat needs at least two inputs!");
13347       SmallVector<SDValue, 2> ConcatOps(NumConcats, DAG.getUNDEF(InVT1));
13348       ConcatOps[0] = VecIn1;
13349       ConcatOps[1] = VecIn2 ? VecIn2 : DAG.getUNDEF(InVT1);
13350       VecIn1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, ConcatOps);
13351       VecIn2 = SDValue();
13352     } else if (InVT1.getSizeInBits() == VT.getSizeInBits() * 2) {
13353       if (!TLI.isExtractSubvectorCheap(VT, NumElems))
13354         return SDValue();
13355 
13356       if (!VecIn2.getNode()) {
13357         // If we only have one input vector, and it's twice the size of the
13358         // output, split it in two.
13359         VecIn2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1,
13360                              DAG.getConstant(NumElems, DL, IdxTy));
13361         VecIn1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1, ZeroIdx);
13362         // Since we now have shorter input vectors, adjust the offset of the
13363         // second vector's start.
13364         Vec2Offset = NumElems;
13365       } else if (InVT2.getSizeInBits() <= InVT1.getSizeInBits()) {
13366         // VecIn1 is wider than the output, and we have another, possibly
13367         // smaller input. Pad the smaller input with undefs, shuffle at the
13368         // input vector width, and extract the output.
13369         // The shuffle type is different than VT, so check legality again.
13370         if (LegalOperations &&
13371             !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, InVT1))
13372           return SDValue();
13373 
13374         // Legalizing INSERT_SUBVECTOR is tricky - you basically have to
13375         // lower it back into a BUILD_VECTOR. So if the inserted type is
13376         // illegal, don't even try.
13377         if (InVT1 != InVT2) {
13378           if (!TLI.isTypeLegal(InVT2))
13379             return SDValue();
13380           VecIn2 = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, InVT1,
13381                                DAG.getUNDEF(InVT1), VecIn2, ZeroIdx);
13382         }
13383         ShuffleNumElems = NumElems * 2;
13384       } else {
13385         // Both VecIn1 and VecIn2 are wider than the output, and VecIn2 is wider
13386         // than VecIn1. We can't handle this for now - this case will disappear
13387         // when we start sorting the vectors by type.
13388         return SDValue();
13389       }
13390     } else {
13391       // TODO: Support cases where the length mismatch isn't exactly by a
13392       // factor of 2.
13393       // TODO: Move this check upwards, so that if we have bad type
13394       // mismatches, we don't create any DAG nodes.
13395       return SDValue();
13396     }
13397   }
13398 
13399   // Initialize mask to undef.
13400   SmallVector<int, 8> Mask(ShuffleNumElems, -1);
13401 
13402   // Only need to run up to the number of elements actually used, not the
13403   // total number of elements in the shuffle - if we are shuffling a wider
13404   // vector, the high lanes should be set to undef.
13405   for (unsigned i = 0; i != NumElems; ++i) {
13406     if (VectorMask[i] <= 0)
13407       continue;
13408 
13409     unsigned ExtIndex = N->getOperand(i).getConstantOperandVal(1);
13410     if (VectorMask[i] == (int)LeftIdx) {
13411       Mask[i] = ExtIndex;
13412     } else if (VectorMask[i] == (int)LeftIdx + 1) {
13413       Mask[i] = Vec2Offset + ExtIndex;
13414     }
13415   }
13416 
13417   // The type the input vectors may have changed above.
13418   InVT1 = VecIn1.getValueType();
13419 
13420   // If we already have a VecIn2, it should have the same type as VecIn1.
13421   // If we don't, get an undef/zero vector of the appropriate type.
13422   VecIn2 = VecIn2.getNode() ? VecIn2 : DAG.getUNDEF(InVT1);
13423   assert(InVT1 == VecIn2.getValueType() && "Unexpected second input type.");
13424 
13425   SDValue Shuffle = DAG.getVectorShuffle(InVT1, DL, VecIn1, VecIn2, Mask);
13426   if (ShuffleNumElems > NumElems)
13427     Shuffle = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Shuffle, ZeroIdx);
13428 
13429   return Shuffle;
13430 }
13431 
13432 // Check to see if this is a BUILD_VECTOR of a bunch of EXTRACT_VECTOR_ELT
13433 // operations. If the types of the vectors we're extracting from allow it,
13434 // turn this into a vector_shuffle node.
13435 SDValue DAGCombiner::reduceBuildVecToShuffle(SDNode *N) {
13436   SDLoc DL(N);
13437   EVT VT = N->getValueType(0);
13438 
13439   // Only type-legal BUILD_VECTOR nodes are converted to shuffle nodes.
13440   if (!isTypeLegal(VT))
13441     return SDValue();
13442 
13443   // May only combine to shuffle after legalize if shuffle is legal.
13444   if (LegalOperations && !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, VT))
13445     return SDValue();
13446 
13447   bool UsesZeroVector = false;
13448   unsigned NumElems = N->getNumOperands();
13449 
13450   // Record, for each element of the newly built vector, which input vector
13451   // that element comes from. -1 stands for undef, 0 for the zero vector,
13452   // and positive values for the input vectors.
13453   // VectorMask maps each element to its vector number, and VecIn maps vector
13454   // numbers to their initial SDValues.
13455 
13456   SmallVector<int, 8> VectorMask(NumElems, -1);
13457   SmallVector<SDValue, 8> VecIn;
13458   VecIn.push_back(SDValue());
13459 
13460   for (unsigned i = 0; i != NumElems; ++i) {
13461     SDValue Op = N->getOperand(i);
13462 
13463     if (Op.isUndef())
13464       continue;
13465 
13466     // See if we can use a blend with a zero vector.
13467     // TODO: Should we generalize this to a blend with an arbitrary constant
13468     // vector?
13469     if (isNullConstant(Op) || isNullFPConstant(Op)) {
13470       UsesZeroVector = true;
13471       VectorMask[i] = 0;
13472       continue;
13473     }
13474 
13475     // Not an undef or zero. If the input is something other than an
13476     // EXTRACT_VECTOR_ELT with a constant index, bail out.
13477     if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
13478         !isa<ConstantSDNode>(Op.getOperand(1)))
13479       return SDValue();
13480 
13481     SDValue ExtractedFromVec = Op.getOperand(0);
13482 
13483     // All inputs must have the same element type as the output.
13484     if (VT.getVectorElementType() !=
13485         ExtractedFromVec.getValueType().getVectorElementType())
13486       return SDValue();
13487 
13488     // Have we seen this input vector before?
13489     // The vectors are expected to be tiny (usually 1 or 2 elements), so using
13490     // a map back from SDValues to numbers isn't worth it.
13491     unsigned Idx = std::distance(
13492         VecIn.begin(), std::find(VecIn.begin(), VecIn.end(), ExtractedFromVec));
13493     if (Idx == VecIn.size())
13494       VecIn.push_back(ExtractedFromVec);
13495 
13496     VectorMask[i] = Idx;
13497   }
13498 
13499   // If we didn't find at least one input vector, bail out.
13500   if (VecIn.size() < 2)
13501     return SDValue();
13502 
13503   // TODO: We want to sort the vectors by descending length, so that adjacent
13504   // pairs have similar length, and the longer vector is always first in the
13505   // pair.
13506 
13507   // TODO: Should this fire if some of the input vectors has illegal type (like
13508   // it does now), or should we let legalization run its course first?
13509 
13510   // Shuffle phase:
13511   // Take pairs of vectors, and shuffle them so that the result has elements
13512   // from these vectors in the correct places.
13513   // For example, given:
13514   // t10: i32 = extract_vector_elt t1, Constant:i64<0>
13515   // t11: i32 = extract_vector_elt t2, Constant:i64<0>
13516   // t12: i32 = extract_vector_elt t3, Constant:i64<0>
13517   // t13: i32 = extract_vector_elt t1, Constant:i64<1>
13518   // t14: v4i32 = BUILD_VECTOR t10, t11, t12, t13
13519   // We will generate:
13520   // t20: v4i32 = vector_shuffle<0,4,u,1> t1, t2
13521   // t21: v4i32 = vector_shuffle<u,u,0,u> t3, undef
13522   SmallVector<SDValue, 4> Shuffles;
13523   for (unsigned In = 0, Len = (VecIn.size() / 2); In < Len; ++In) {
13524     unsigned LeftIdx = 2 * In + 1;
13525     SDValue VecLeft = VecIn[LeftIdx];
13526     SDValue VecRight =
13527         (LeftIdx + 1) < VecIn.size() ? VecIn[LeftIdx + 1] : SDValue();
13528 
13529     if (SDValue Shuffle = createBuildVecShuffle(DL, N, VectorMask, VecLeft,
13530                                                 VecRight, LeftIdx))
13531       Shuffles.push_back(Shuffle);
13532     else
13533       return SDValue();
13534   }
13535 
13536   // If we need the zero vector as an "ingredient" in the blend tree, add it
13537   // to the list of shuffles.
13538   if (UsesZeroVector)
13539     Shuffles.push_back(VT.isInteger() ? DAG.getConstant(0, DL, VT)
13540                                       : DAG.getConstantFP(0.0, DL, VT));
13541 
13542   // If we only have one shuffle, we're done.
13543   if (Shuffles.size() == 1)
13544     return Shuffles[0];
13545 
13546   // Update the vector mask to point to the post-shuffle vectors.
13547   for (int &Vec : VectorMask)
13548     if (Vec == 0)
13549       Vec = Shuffles.size() - 1;
13550     else
13551       Vec = (Vec - 1) / 2;
13552 
13553   // More than one shuffle. Generate a binary tree of blends, e.g. if from
13554   // the previous step we got the set of shuffles t10, t11, t12, t13, we will
13555   // generate:
13556   // t10: v8i32 = vector_shuffle<0,8,u,u,u,u,u,u> t1, t2
13557   // t11: v8i32 = vector_shuffle<u,u,0,8,u,u,u,u> t3, t4
13558   // t12: v8i32 = vector_shuffle<u,u,u,u,0,8,u,u> t5, t6
13559   // t13: v8i32 = vector_shuffle<u,u,u,u,u,u,0,8> t7, t8
13560   // t20: v8i32 = vector_shuffle<0,1,10,11,u,u,u,u> t10, t11
13561   // t21: v8i32 = vector_shuffle<u,u,u,u,4,5,14,15> t12, t13
13562   // t30: v8i32 = vector_shuffle<0,1,2,3,12,13,14,15> t20, t21
13563 
13564   // Make sure the initial size of the shuffle list is even.
13565   if (Shuffles.size() % 2)
13566     Shuffles.push_back(DAG.getUNDEF(VT));
13567 
13568   for (unsigned CurSize = Shuffles.size(); CurSize > 1; CurSize /= 2) {
13569     if (CurSize % 2) {
13570       Shuffles[CurSize] = DAG.getUNDEF(VT);
13571       CurSize++;
13572     }
13573     for (unsigned In = 0, Len = CurSize / 2; In < Len; ++In) {
13574       int Left = 2 * In;
13575       int Right = 2 * In + 1;
13576       SmallVector<int, 8> Mask(NumElems, -1);
13577       for (unsigned i = 0; i != NumElems; ++i) {
13578         if (VectorMask[i] == Left) {
13579           Mask[i] = i;
13580           VectorMask[i] = In;
13581         } else if (VectorMask[i] == Right) {
13582           Mask[i] = i + NumElems;
13583           VectorMask[i] = In;
13584         }
13585       }
13586 
13587       Shuffles[In] =
13588           DAG.getVectorShuffle(VT, DL, Shuffles[Left], Shuffles[Right], Mask);
13589     }
13590   }
13591 
13592   return Shuffles[0];
13593 }
13594 
13595 SDValue DAGCombiner::visitBUILD_VECTOR(SDNode *N) {
13596   EVT VT = N->getValueType(0);
13597 
13598   // A vector built entirely of undefs is undef.
13599   if (ISD::allOperandsUndef(N))
13600     return DAG.getUNDEF(VT);
13601 
13602   if (SDValue V = reduceBuildVecExtToExtBuildVec(N))
13603     return V;
13604 
13605   if (SDValue V = reduceBuildVecConvertToConvertBuildVec(N))
13606     return V;
13607 
13608   if (SDValue V = reduceBuildVecToShuffle(N))
13609     return V;
13610 
13611   return SDValue();
13612 }
13613 
13614 static SDValue combineConcatVectorOfScalars(SDNode *N, SelectionDAG &DAG) {
13615   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
13616   EVT OpVT = N->getOperand(0).getValueType();
13617 
13618   // If the operands are legal vectors, leave them alone.
13619   if (TLI.isTypeLegal(OpVT))
13620     return SDValue();
13621 
13622   SDLoc DL(N);
13623   EVT VT = N->getValueType(0);
13624   SmallVector<SDValue, 8> Ops;
13625 
13626   EVT SVT = EVT::getIntegerVT(*DAG.getContext(), OpVT.getSizeInBits());
13627   SDValue ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
13628 
13629   // Keep track of what we encounter.
13630   bool AnyInteger = false;
13631   bool AnyFP = false;
13632   for (const SDValue &Op : N->ops()) {
13633     if (ISD::BITCAST == Op.getOpcode() &&
13634         !Op.getOperand(0).getValueType().isVector())
13635       Ops.push_back(Op.getOperand(0));
13636     else if (ISD::UNDEF == Op.getOpcode())
13637       Ops.push_back(ScalarUndef);
13638     else
13639       return SDValue();
13640 
13641     // Note whether we encounter an integer or floating point scalar.
13642     // If it's neither, bail out, it could be something weird like x86mmx.
13643     EVT LastOpVT = Ops.back().getValueType();
13644     if (LastOpVT.isFloatingPoint())
13645       AnyFP = true;
13646     else if (LastOpVT.isInteger())
13647       AnyInteger = true;
13648     else
13649       return SDValue();
13650   }
13651 
13652   // If any of the operands is a floating point scalar bitcast to a vector,
13653   // use floating point types throughout, and bitcast everything.
13654   // Replace UNDEFs by another scalar UNDEF node, of the final desired type.
13655   if (AnyFP) {
13656     SVT = EVT::getFloatingPointVT(OpVT.getSizeInBits());
13657     ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT);
13658     if (AnyInteger) {
13659       for (SDValue &Op : Ops) {
13660         if (Op.getValueType() == SVT)
13661           continue;
13662         if (Op.isUndef())
13663           Op = ScalarUndef;
13664         else
13665           Op = DAG.getBitcast(SVT, Op);
13666       }
13667     }
13668   }
13669 
13670   EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SVT,
13671                                VT.getSizeInBits() / SVT.getSizeInBits());
13672   return DAG.getBitcast(VT, DAG.getBuildVector(VecVT, DL, Ops));
13673 }
13674 
13675 // Check to see if this is a CONCAT_VECTORS of a bunch of EXTRACT_SUBVECTOR
13676 // operations. If so, and if the EXTRACT_SUBVECTOR vector inputs come from at
13677 // most two distinct vectors the same size as the result, attempt to turn this
13678 // into a legal shuffle.
13679 static SDValue combineConcatVectorOfExtracts(SDNode *N, SelectionDAG &DAG) {
13680   EVT VT = N->getValueType(0);
13681   EVT OpVT = N->getOperand(0).getValueType();
13682   int NumElts = VT.getVectorNumElements();
13683   int NumOpElts = OpVT.getVectorNumElements();
13684 
13685   SDValue SV0 = DAG.getUNDEF(VT), SV1 = DAG.getUNDEF(VT);
13686   SmallVector<int, 8> Mask;
13687 
13688   for (SDValue Op : N->ops()) {
13689     // Peek through any bitcast.
13690     while (Op.getOpcode() == ISD::BITCAST)
13691       Op = Op.getOperand(0);
13692 
13693     // UNDEF nodes convert to UNDEF shuffle mask values.
13694     if (Op.isUndef()) {
13695       Mask.append((unsigned)NumOpElts, -1);
13696       continue;
13697     }
13698 
13699     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
13700       return SDValue();
13701 
13702     // What vector are we extracting the subvector from and at what index?
13703     SDValue ExtVec = Op.getOperand(0);
13704 
13705     // We want the EVT of the original extraction to correctly scale the
13706     // extraction index.
13707     EVT ExtVT = ExtVec.getValueType();
13708 
13709     // Peek through any bitcast.
13710     while (ExtVec.getOpcode() == ISD::BITCAST)
13711       ExtVec = ExtVec.getOperand(0);
13712 
13713     // UNDEF nodes convert to UNDEF shuffle mask values.
13714     if (ExtVec.isUndef()) {
13715       Mask.append((unsigned)NumOpElts, -1);
13716       continue;
13717     }
13718 
13719     if (!isa<ConstantSDNode>(Op.getOperand(1)))
13720       return SDValue();
13721     int ExtIdx = cast<ConstantSDNode>(Op.getOperand(1))->getZExtValue();
13722 
13723     // Ensure that we are extracting a subvector from a vector the same
13724     // size as the result.
13725     if (ExtVT.getSizeInBits() != VT.getSizeInBits())
13726       return SDValue();
13727 
13728     // Scale the subvector index to account for any bitcast.
13729     int NumExtElts = ExtVT.getVectorNumElements();
13730     if (0 == (NumExtElts % NumElts))
13731       ExtIdx /= (NumExtElts / NumElts);
13732     else if (0 == (NumElts % NumExtElts))
13733       ExtIdx *= (NumElts / NumExtElts);
13734     else
13735       return SDValue();
13736 
13737     // At most we can reference 2 inputs in the final shuffle.
13738     if (SV0.isUndef() || SV0 == ExtVec) {
13739       SV0 = ExtVec;
13740       for (int i = 0; i != NumOpElts; ++i)
13741         Mask.push_back(i + ExtIdx);
13742     } else if (SV1.isUndef() || SV1 == ExtVec) {
13743       SV1 = ExtVec;
13744       for (int i = 0; i != NumOpElts; ++i)
13745         Mask.push_back(i + ExtIdx + NumElts);
13746     } else {
13747       return SDValue();
13748     }
13749   }
13750 
13751   if (!DAG.getTargetLoweringInfo().isShuffleMaskLegal(Mask, VT))
13752     return SDValue();
13753 
13754   return DAG.getVectorShuffle(VT, SDLoc(N), DAG.getBitcast(VT, SV0),
13755                               DAG.getBitcast(VT, SV1), Mask);
13756 }
13757 
13758 SDValue DAGCombiner::visitCONCAT_VECTORS(SDNode *N) {
13759   // If we only have one input vector, we don't need to do any concatenation.
13760   if (N->getNumOperands() == 1)
13761     return N->getOperand(0);
13762 
13763   // Check if all of the operands are undefs.
13764   EVT VT = N->getValueType(0);
13765   if (ISD::allOperandsUndef(N))
13766     return DAG.getUNDEF(VT);
13767 
13768   // Optimize concat_vectors where all but the first of the vectors are undef.
13769   if (std::all_of(std::next(N->op_begin()), N->op_end(), [](const SDValue &Op) {
13770         return Op.isUndef();
13771       })) {
13772     SDValue In = N->getOperand(0);
13773     assert(In.getValueType().isVector() && "Must concat vectors");
13774 
13775     // Transform: concat_vectors(scalar, undef) -> scalar_to_vector(sclr).
13776     if (In->getOpcode() == ISD::BITCAST &&
13777         !In->getOperand(0)->getValueType(0).isVector()) {
13778       SDValue Scalar = In->getOperand(0);
13779 
13780       // If the bitcast type isn't legal, it might be a trunc of a legal type;
13781       // look through the trunc so we can still do the transform:
13782       //   concat_vectors(trunc(scalar), undef) -> scalar_to_vector(scalar)
13783       if (Scalar->getOpcode() == ISD::TRUNCATE &&
13784           !TLI.isTypeLegal(Scalar.getValueType()) &&
13785           TLI.isTypeLegal(Scalar->getOperand(0).getValueType()))
13786         Scalar = Scalar->getOperand(0);
13787 
13788       EVT SclTy = Scalar->getValueType(0);
13789 
13790       if (!SclTy.isFloatingPoint() && !SclTy.isInteger())
13791         return SDValue();
13792 
13793       EVT NVT = EVT::getVectorVT(*DAG.getContext(), SclTy,
13794                                  VT.getSizeInBits() / SclTy.getSizeInBits());
13795       if (!TLI.isTypeLegal(NVT) || !TLI.isTypeLegal(Scalar.getValueType()))
13796         return SDValue();
13797 
13798       SDValue Res = DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), NVT, Scalar);
13799       return DAG.getBitcast(VT, Res);
13800     }
13801   }
13802 
13803   // Fold any combination of BUILD_VECTOR or UNDEF nodes into one BUILD_VECTOR.
13804   // We have already tested above for an UNDEF only concatenation.
13805   // fold (concat_vectors (BUILD_VECTOR A, B, ...), (BUILD_VECTOR C, D, ...))
13806   // -> (BUILD_VECTOR A, B, ..., C, D, ...)
13807   auto IsBuildVectorOrUndef = [](const SDValue &Op) {
13808     return ISD::UNDEF == Op.getOpcode() || ISD::BUILD_VECTOR == Op.getOpcode();
13809   };
13810   if (llvm::all_of(N->ops(), IsBuildVectorOrUndef)) {
13811     SmallVector<SDValue, 8> Opnds;
13812     EVT SVT = VT.getScalarType();
13813 
13814     EVT MinVT = SVT;
13815     if (!SVT.isFloatingPoint()) {
13816       // If BUILD_VECTOR are from built from integer, they may have different
13817       // operand types. Get the smallest type and truncate all operands to it.
13818       bool FoundMinVT = false;
13819       for (const SDValue &Op : N->ops())
13820         if (ISD::BUILD_VECTOR == Op.getOpcode()) {
13821           EVT OpSVT = Op.getOperand(0)->getValueType(0);
13822           MinVT = (!FoundMinVT || OpSVT.bitsLE(MinVT)) ? OpSVT : MinVT;
13823           FoundMinVT = true;
13824         }
13825       assert(FoundMinVT && "Concat vector type mismatch");
13826     }
13827 
13828     for (const SDValue &Op : N->ops()) {
13829       EVT OpVT = Op.getValueType();
13830       unsigned NumElts = OpVT.getVectorNumElements();
13831 
13832       if (ISD::UNDEF == Op.getOpcode())
13833         Opnds.append(NumElts, DAG.getUNDEF(MinVT));
13834 
13835       if (ISD::BUILD_VECTOR == Op.getOpcode()) {
13836         if (SVT.isFloatingPoint()) {
13837           assert(SVT == OpVT.getScalarType() && "Concat vector type mismatch");
13838           Opnds.append(Op->op_begin(), Op->op_begin() + NumElts);
13839         } else {
13840           for (unsigned i = 0; i != NumElts; ++i)
13841             Opnds.push_back(
13842                 DAG.getNode(ISD::TRUNCATE, SDLoc(N), MinVT, Op.getOperand(i)));
13843         }
13844       }
13845     }
13846 
13847     assert(VT.getVectorNumElements() == Opnds.size() &&
13848            "Concat vector type mismatch");
13849     return DAG.getBuildVector(VT, SDLoc(N), Opnds);
13850   }
13851 
13852   // Fold CONCAT_VECTORS of only bitcast scalars (or undef) to BUILD_VECTOR.
13853   if (SDValue V = combineConcatVectorOfScalars(N, DAG))
13854     return V;
13855 
13856   // Fold CONCAT_VECTORS of EXTRACT_SUBVECTOR (or undef) to VECTOR_SHUFFLE.
13857   if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT))
13858     if (SDValue V = combineConcatVectorOfExtracts(N, DAG))
13859       return V;
13860 
13861   // Type legalization of vectors and DAG canonicalization of SHUFFLE_VECTOR
13862   // nodes often generate nop CONCAT_VECTOR nodes.
13863   // Scan the CONCAT_VECTOR operands and look for a CONCAT operations that
13864   // place the incoming vectors at the exact same location.
13865   SDValue SingleSource = SDValue();
13866   unsigned PartNumElem = N->getOperand(0).getValueType().getVectorNumElements();
13867 
13868   for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
13869     SDValue Op = N->getOperand(i);
13870 
13871     if (Op.isUndef())
13872       continue;
13873 
13874     // Check if this is the identity extract:
13875     if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR)
13876       return SDValue();
13877 
13878     // Find the single incoming vector for the extract_subvector.
13879     if (SingleSource.getNode()) {
13880       if (Op.getOperand(0) != SingleSource)
13881         return SDValue();
13882     } else {
13883       SingleSource = Op.getOperand(0);
13884 
13885       // Check the source type is the same as the type of the result.
13886       // If not, this concat may extend the vector, so we can not
13887       // optimize it away.
13888       if (SingleSource.getValueType() != N->getValueType(0))
13889         return SDValue();
13890     }
13891 
13892     unsigned IdentityIndex = i * PartNumElem;
13893     ConstantSDNode *CS = dyn_cast<ConstantSDNode>(Op.getOperand(1));
13894     // The extract index must be constant.
13895     if (!CS)
13896       return SDValue();
13897 
13898     // Check that we are reading from the identity index.
13899     if (CS->getZExtValue() != IdentityIndex)
13900       return SDValue();
13901   }
13902 
13903   if (SingleSource.getNode())
13904     return SingleSource;
13905 
13906   return SDValue();
13907 }
13908 
13909 SDValue DAGCombiner::visitEXTRACT_SUBVECTOR(SDNode* N) {
13910   EVT NVT = N->getValueType(0);
13911   SDValue V = N->getOperand(0);
13912 
13913   // Extract from UNDEF is UNDEF.
13914   if (V.isUndef())
13915     return DAG.getUNDEF(NVT);
13916 
13917   // Combine:
13918   //    (extract_subvec (concat V1, V2, ...), i)
13919   // Into:
13920   //    Vi if possible
13921   // Only operand 0 is checked as 'concat' assumes all inputs of the same
13922   // type.
13923   if (V->getOpcode() == ISD::CONCAT_VECTORS &&
13924       isa<ConstantSDNode>(N->getOperand(1)) &&
13925       V->getOperand(0).getValueType() == NVT) {
13926     unsigned Idx = N->getConstantOperandVal(1);
13927     unsigned NumElems = NVT.getVectorNumElements();
13928     assert((Idx % NumElems) == 0 &&
13929            "IDX in concat is not a multiple of the result vector length.");
13930     return V->getOperand(Idx / NumElems);
13931   }
13932 
13933   // Skip bitcasting
13934   if (V->getOpcode() == ISD::BITCAST)
13935     V = V.getOperand(0);
13936 
13937   if (V->getOpcode() == ISD::INSERT_SUBVECTOR) {
13938     // Handle only simple case where vector being inserted and vector
13939     // being extracted are of same size.
13940     EVT SmallVT = V->getOperand(1).getValueType();
13941     if (!NVT.bitsEq(SmallVT))
13942       return SDValue();
13943 
13944     // Only handle cases where both indexes are constants.
13945     ConstantSDNode *ExtIdx = dyn_cast<ConstantSDNode>(N->getOperand(1));
13946     ConstantSDNode *InsIdx = dyn_cast<ConstantSDNode>(V->getOperand(2));
13947 
13948     if (InsIdx && ExtIdx) {
13949       // Combine:
13950       //    (extract_subvec (insert_subvec V1, V2, InsIdx), ExtIdx)
13951       // Into:
13952       //    indices are equal or bit offsets are equal => V1
13953       //    otherwise => (extract_subvec V1, ExtIdx)
13954       if (InsIdx->getZExtValue() * SmallVT.getScalarSizeInBits() ==
13955           ExtIdx->getZExtValue() * NVT.getScalarSizeInBits())
13956         return DAG.getBitcast(NVT, V->getOperand(1));
13957       return DAG.getNode(
13958           ISD::EXTRACT_SUBVECTOR, SDLoc(N), NVT,
13959           DAG.getBitcast(N->getOperand(0).getValueType(), V->getOperand(0)),
13960           N->getOperand(1));
13961     }
13962   }
13963 
13964   return SDValue();
13965 }
13966 
13967 static SDValue simplifyShuffleOperandRecursively(SmallBitVector &UsedElements,
13968                                                  SDValue V, SelectionDAG &DAG) {
13969   SDLoc DL(V);
13970   EVT VT = V.getValueType();
13971 
13972   switch (V.getOpcode()) {
13973   default:
13974     return V;
13975 
13976   case ISD::CONCAT_VECTORS: {
13977     EVT OpVT = V->getOperand(0).getValueType();
13978     int OpSize = OpVT.getVectorNumElements();
13979     SmallBitVector OpUsedElements(OpSize, false);
13980     bool FoundSimplification = false;
13981     SmallVector<SDValue, 4> NewOps;
13982     NewOps.reserve(V->getNumOperands());
13983     for (int i = 0, NumOps = V->getNumOperands(); i < NumOps; ++i) {
13984       SDValue Op = V->getOperand(i);
13985       bool OpUsed = false;
13986       for (int j = 0; j < OpSize; ++j)
13987         if (UsedElements[i * OpSize + j]) {
13988           OpUsedElements[j] = true;
13989           OpUsed = true;
13990         }
13991       NewOps.push_back(
13992           OpUsed ? simplifyShuffleOperandRecursively(OpUsedElements, Op, DAG)
13993                  : DAG.getUNDEF(OpVT));
13994       FoundSimplification |= Op == NewOps.back();
13995       OpUsedElements.reset();
13996     }
13997     if (FoundSimplification)
13998       V = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, NewOps);
13999     return V;
14000   }
14001 
14002   case ISD::INSERT_SUBVECTOR: {
14003     SDValue BaseV = V->getOperand(0);
14004     SDValue SubV = V->getOperand(1);
14005     auto *IdxN = dyn_cast<ConstantSDNode>(V->getOperand(2));
14006     if (!IdxN)
14007       return V;
14008 
14009     int SubSize = SubV.getValueType().getVectorNumElements();
14010     int Idx = IdxN->getZExtValue();
14011     bool SubVectorUsed = false;
14012     SmallBitVector SubUsedElements(SubSize, false);
14013     for (int i = 0; i < SubSize; ++i)
14014       if (UsedElements[i + Idx]) {
14015         SubVectorUsed = true;
14016         SubUsedElements[i] = true;
14017         UsedElements[i + Idx] = false;
14018       }
14019 
14020     // Now recurse on both the base and sub vectors.
14021     SDValue SimplifiedSubV =
14022         SubVectorUsed
14023             ? simplifyShuffleOperandRecursively(SubUsedElements, SubV, DAG)
14024             : DAG.getUNDEF(SubV.getValueType());
14025     SDValue SimplifiedBaseV = simplifyShuffleOperandRecursively(UsedElements, BaseV, DAG);
14026     if (SimplifiedSubV != SubV || SimplifiedBaseV != BaseV)
14027       V = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VT,
14028                       SimplifiedBaseV, SimplifiedSubV, V->getOperand(2));
14029     return V;
14030   }
14031   }
14032 }
14033 
14034 static SDValue simplifyShuffleOperands(ShuffleVectorSDNode *SVN, SDValue N0,
14035                                        SDValue N1, SelectionDAG &DAG) {
14036   EVT VT = SVN->getValueType(0);
14037   int NumElts = VT.getVectorNumElements();
14038   SmallBitVector N0UsedElements(NumElts, false), N1UsedElements(NumElts, false);
14039   for (int M : SVN->getMask())
14040     if (M >= 0 && M < NumElts)
14041       N0UsedElements[M] = true;
14042     else if (M >= NumElts)
14043       N1UsedElements[M - NumElts] = true;
14044 
14045   SDValue S0 = simplifyShuffleOperandRecursively(N0UsedElements, N0, DAG);
14046   SDValue S1 = simplifyShuffleOperandRecursively(N1UsedElements, N1, DAG);
14047   if (S0 == N0 && S1 == N1)
14048     return SDValue();
14049 
14050   return DAG.getVectorShuffle(VT, SDLoc(SVN), S0, S1, SVN->getMask());
14051 }
14052 
14053 // Tries to turn a shuffle of two CONCAT_VECTORS into a single concat,
14054 // or turn a shuffle of a single concat into simpler shuffle then concat.
14055 static SDValue partitionShuffleOfConcats(SDNode *N, SelectionDAG &DAG) {
14056   EVT VT = N->getValueType(0);
14057   unsigned NumElts = VT.getVectorNumElements();
14058 
14059   SDValue N0 = N->getOperand(0);
14060   SDValue N1 = N->getOperand(1);
14061   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
14062 
14063   SmallVector<SDValue, 4> Ops;
14064   EVT ConcatVT = N0.getOperand(0).getValueType();
14065   unsigned NumElemsPerConcat = ConcatVT.getVectorNumElements();
14066   unsigned NumConcats = NumElts / NumElemsPerConcat;
14067 
14068   // Special case: shuffle(concat(A,B)) can be more efficiently represented
14069   // as concat(shuffle(A,B),UNDEF) if the shuffle doesn't set any of the high
14070   // half vector elements.
14071   if (NumElemsPerConcat * 2 == NumElts && N1.isUndef() &&
14072       std::all_of(SVN->getMask().begin() + NumElemsPerConcat,
14073                   SVN->getMask().end(), [](int i) { return i == -1; })) {
14074     N0 = DAG.getVectorShuffle(ConcatVT, SDLoc(N), N0.getOperand(0), N0.getOperand(1),
14075                               makeArrayRef(SVN->getMask().begin(), NumElemsPerConcat));
14076     N1 = DAG.getUNDEF(ConcatVT);
14077     return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N0, N1);
14078   }
14079 
14080   // Look at every vector that's inserted. We're looking for exact
14081   // subvector-sized copies from a concatenated vector
14082   for (unsigned I = 0; I != NumConcats; ++I) {
14083     // Make sure we're dealing with a copy.
14084     unsigned Begin = I * NumElemsPerConcat;
14085     bool AllUndef = true, NoUndef = true;
14086     for (unsigned J = Begin; J != Begin + NumElemsPerConcat; ++J) {
14087       if (SVN->getMaskElt(J) >= 0)
14088         AllUndef = false;
14089       else
14090         NoUndef = false;
14091     }
14092 
14093     if (NoUndef) {
14094       if (SVN->getMaskElt(Begin) % NumElemsPerConcat != 0)
14095         return SDValue();
14096 
14097       for (unsigned J = 1; J != NumElemsPerConcat; ++J)
14098         if (SVN->getMaskElt(Begin + J - 1) + 1 != SVN->getMaskElt(Begin + J))
14099           return SDValue();
14100 
14101       unsigned FirstElt = SVN->getMaskElt(Begin) / NumElemsPerConcat;
14102       if (FirstElt < N0.getNumOperands())
14103         Ops.push_back(N0.getOperand(FirstElt));
14104       else
14105         Ops.push_back(N1.getOperand(FirstElt - N0.getNumOperands()));
14106 
14107     } else if (AllUndef) {
14108       Ops.push_back(DAG.getUNDEF(N0.getOperand(0).getValueType()));
14109     } else { // Mixed with general masks and undefs, can't do optimization.
14110       return SDValue();
14111     }
14112   }
14113 
14114   return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops);
14115 }
14116 
14117 // Attempt to combine a shuffle of 2 inputs of 'scalar sources' -
14118 // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR.
14119 //
14120 // SHUFFLE(BUILD_VECTOR(), BUILD_VECTOR()) -> BUILD_VECTOR() is always
14121 // a simplification in some sense, but it isn't appropriate in general: some
14122 // BUILD_VECTORs are substantially cheaper than others. The general case
14123 // of a BUILD_VECTOR requires inserting each element individually (or
14124 // performing the equivalent in a temporary stack variable). A BUILD_VECTOR of
14125 // all constants is a single constant pool load.  A BUILD_VECTOR where each
14126 // element is identical is a splat.  A BUILD_VECTOR where most of the operands
14127 // are undef lowers to a small number of element insertions.
14128 //
14129 // To deal with this, we currently use a bunch of mostly arbitrary heuristics.
14130 // We don't fold shuffles where one side is a non-zero constant, and we don't
14131 // fold shuffles if the resulting BUILD_VECTOR would have duplicate
14132 // non-constant operands. This seems to work out reasonably well in practice.
14133 static SDValue combineShuffleOfScalars(ShuffleVectorSDNode *SVN,
14134                                        SelectionDAG &DAG,
14135                                        const TargetLowering &TLI) {
14136   EVT VT = SVN->getValueType(0);
14137   unsigned NumElts = VT.getVectorNumElements();
14138   SDValue N0 = SVN->getOperand(0);
14139   SDValue N1 = SVN->getOperand(1);
14140 
14141   if (!N0->hasOneUse() || !N1->hasOneUse())
14142     return SDValue();
14143   // If only one of N1,N2 is constant, bail out if it is not ALL_ZEROS as
14144   // discussed above.
14145   if (!N1.isUndef()) {
14146     bool N0AnyConst = isAnyConstantBuildVector(N0.getNode());
14147     bool N1AnyConst = isAnyConstantBuildVector(N1.getNode());
14148     if (N0AnyConst && !N1AnyConst && !ISD::isBuildVectorAllZeros(N0.getNode()))
14149       return SDValue();
14150     if (!N0AnyConst && N1AnyConst && !ISD::isBuildVectorAllZeros(N1.getNode()))
14151       return SDValue();
14152   }
14153 
14154   SmallVector<SDValue, 8> Ops;
14155   SmallSet<SDValue, 16> DuplicateOps;
14156   for (int M : SVN->getMask()) {
14157     SDValue Op = DAG.getUNDEF(VT.getScalarType());
14158     if (M >= 0) {
14159       int Idx = M < (int)NumElts ? M : M - NumElts;
14160       SDValue &S = (M < (int)NumElts ? N0 : N1);
14161       if (S.getOpcode() == ISD::BUILD_VECTOR) {
14162         Op = S.getOperand(Idx);
14163       } else if (S.getOpcode() == ISD::SCALAR_TO_VECTOR) {
14164         if (Idx == 0)
14165           Op = S.getOperand(0);
14166       } else {
14167         // Operand can't be combined - bail out.
14168         return SDValue();
14169       }
14170     }
14171 
14172     // Don't duplicate a non-constant BUILD_VECTOR operand; semantically, this is
14173     // fine, but it's likely to generate low-quality code if the target can't
14174     // reconstruct an appropriate shuffle.
14175     if (!Op.isUndef() && !isa<ConstantSDNode>(Op) && !isa<ConstantFPSDNode>(Op))
14176       if (!DuplicateOps.insert(Op).second)
14177         return SDValue();
14178 
14179     Ops.push_back(Op);
14180   }
14181   // BUILD_VECTOR requires all inputs to be of the same type, find the
14182   // maximum type and extend them all.
14183   EVT SVT = VT.getScalarType();
14184   if (SVT.isInteger())
14185     for (SDValue &Op : Ops)
14186       SVT = (SVT.bitsLT(Op.getValueType()) ? Op.getValueType() : SVT);
14187   if (SVT != VT.getScalarType())
14188     for (SDValue &Op : Ops)
14189       Op = TLI.isZExtFree(Op.getValueType(), SVT)
14190                ? DAG.getZExtOrTrunc(Op, SDLoc(SVN), SVT)
14191                : DAG.getSExtOrTrunc(Op, SDLoc(SVN), SVT);
14192   return DAG.getBuildVector(VT, SDLoc(SVN), Ops);
14193 }
14194 
14195 // Match shuffles that can be converted to any_vector_extend_in_reg.
14196 // This is often generated during legalization.
14197 // e.g. v4i32 <0,u,1,u> -> (v2i64 any_vector_extend_in_reg(v4i32 src))
14198 // TODO Add support for ZERO_EXTEND_VECTOR_INREG when we have a test case.
14199 SDValue combineShuffleToVectorExtend(ShuffleVectorSDNode *SVN,
14200                                      SelectionDAG &DAG,
14201                                      const TargetLowering &TLI,
14202                                      bool LegalOperations) {
14203   EVT VT = SVN->getValueType(0);
14204   bool IsBigEndian = DAG.getDataLayout().isBigEndian();
14205 
14206   // TODO Add support for big-endian when we have a test case.
14207   if (!VT.isInteger() || IsBigEndian)
14208     return SDValue();
14209 
14210   unsigned NumElts = VT.getVectorNumElements();
14211   unsigned EltSizeInBits = VT.getScalarSizeInBits();
14212   ArrayRef<int> Mask = SVN->getMask();
14213   SDValue N0 = SVN->getOperand(0);
14214 
14215   // shuffle<0,-1,1,-1> == (v2i64 anyextend_vector_inreg(v4i32))
14216   auto isAnyExtend = [&Mask, &NumElts](unsigned Scale) {
14217     for (unsigned i = 0; i != NumElts; ++i) {
14218       if (Mask[i] < 0)
14219         continue;
14220       if ((i % Scale) == 0 && Mask[i] == (int)(i / Scale))
14221         continue;
14222       return false;
14223     }
14224     return true;
14225   };
14226 
14227   // Attempt to match a '*_extend_vector_inreg' shuffle, we just search for
14228   // power-of-2 extensions as they are the most likely.
14229   for (unsigned Scale = 2; Scale < NumElts; Scale *= 2) {
14230     if (!isAnyExtend(Scale))
14231       continue;
14232 
14233     EVT OutSVT = EVT::getIntegerVT(*DAG.getContext(), EltSizeInBits * Scale);
14234     EVT OutVT = EVT::getVectorVT(*DAG.getContext(), OutSVT, NumElts / Scale);
14235     if (!LegalOperations ||
14236         TLI.isOperationLegalOrCustom(ISD::ANY_EXTEND_VECTOR_INREG, OutVT))
14237       return DAG.getBitcast(VT,
14238                             DAG.getAnyExtendVectorInReg(N0, SDLoc(SVN), OutVT));
14239   }
14240 
14241   return SDValue();
14242 }
14243 
14244 // Detect 'truncate_vector_inreg' style shuffles that pack the lower parts of
14245 // each source element of a large type into the lowest elements of a smaller
14246 // destination type. This is often generated during legalization.
14247 // If the source node itself was a '*_extend_vector_inreg' node then we should
14248 // then be able to remove it.
14249 SDValue combineTruncationShuffle(ShuffleVectorSDNode *SVN, SelectionDAG &DAG) {
14250   EVT VT = SVN->getValueType(0);
14251   bool IsBigEndian = DAG.getDataLayout().isBigEndian();
14252 
14253   // TODO Add support for big-endian when we have a test case.
14254   if (!VT.isInteger() || IsBigEndian)
14255     return SDValue();
14256 
14257   SDValue N0 = SVN->getOperand(0);
14258   while (N0.getOpcode() == ISD::BITCAST)
14259     N0 = N0.getOperand(0);
14260 
14261   unsigned Opcode = N0.getOpcode();
14262   if (Opcode != ISD::ANY_EXTEND_VECTOR_INREG &&
14263       Opcode != ISD::SIGN_EXTEND_VECTOR_INREG &&
14264       Opcode != ISD::ZERO_EXTEND_VECTOR_INREG)
14265     return SDValue();
14266 
14267   SDValue N00 = N0.getOperand(0);
14268   ArrayRef<int> Mask = SVN->getMask();
14269   unsigned NumElts = VT.getVectorNumElements();
14270   unsigned EltSizeInBits = VT.getScalarSizeInBits();
14271   unsigned ExtSrcSizeInBits = N00.getScalarValueSizeInBits();
14272 
14273   // (v4i32 truncate_vector_inreg(v2i64)) == shuffle<0,2-1,-1>
14274   // (v8i16 truncate_vector_inreg(v4i32)) == shuffle<0,2,4,6,-1,-1,-1,-1>
14275   // (v8i16 truncate_vector_inreg(v2i64)) == shuffle<0,4,-1,-1,-1,-1,-1,-1>
14276   auto isTruncate = [&Mask, &NumElts](unsigned Scale) {
14277     for (unsigned i = 0; i != NumElts; ++i) {
14278       if (Mask[i] < 0)
14279         continue;
14280       if ((i * Scale) < NumElts && Mask[i] == (int)(i * Scale))
14281         continue;
14282       return false;
14283     }
14284     return true;
14285   };
14286 
14287   // At the moment we just handle the case where we've truncated back to the
14288   // same size as before the extension.
14289   // TODO: handle more extension/truncation cases as cases arise.
14290   if (EltSizeInBits != ExtSrcSizeInBits)
14291     return SDValue();
14292 
14293   // Attempt to match a 'truncate_vector_inreg' shuffle, we just search for
14294   // power-of-2 truncations as they are the most likely.
14295   for (unsigned Scale = 2; Scale < NumElts; Scale *= 2)
14296     if (isTruncate(Scale))
14297       return DAG.getBitcast(VT, N00);
14298 
14299   return SDValue();
14300 }
14301 
14302 SDValue DAGCombiner::visitVECTOR_SHUFFLE(SDNode *N) {
14303   EVT VT = N->getValueType(0);
14304   unsigned NumElts = VT.getVectorNumElements();
14305 
14306   SDValue N0 = N->getOperand(0);
14307   SDValue N1 = N->getOperand(1);
14308 
14309   assert(N0.getValueType() == VT && "Vector shuffle must be normalized in DAG");
14310 
14311   // Canonicalize shuffle undef, undef -> undef
14312   if (N0.isUndef() && N1.isUndef())
14313     return DAG.getUNDEF(VT);
14314 
14315   ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
14316 
14317   // Canonicalize shuffle v, v -> v, undef
14318   if (N0 == N1) {
14319     SmallVector<int, 8> NewMask;
14320     for (unsigned i = 0; i != NumElts; ++i) {
14321       int Idx = SVN->getMaskElt(i);
14322       if (Idx >= (int)NumElts) Idx -= NumElts;
14323       NewMask.push_back(Idx);
14324     }
14325     return DAG.getVectorShuffle(VT, SDLoc(N), N0, DAG.getUNDEF(VT), NewMask);
14326   }
14327 
14328   // Canonicalize shuffle undef, v -> v, undef.  Commute the shuffle mask.
14329   if (N0.isUndef())
14330     return DAG.getCommutedVectorShuffle(*SVN);
14331 
14332   // Remove references to rhs if it is undef
14333   if (N1.isUndef()) {
14334     bool Changed = false;
14335     SmallVector<int, 8> NewMask;
14336     for (unsigned i = 0; i != NumElts; ++i) {
14337       int Idx = SVN->getMaskElt(i);
14338       if (Idx >= (int)NumElts) {
14339         Idx = -1;
14340         Changed = true;
14341       }
14342       NewMask.push_back(Idx);
14343     }
14344     if (Changed)
14345       return DAG.getVectorShuffle(VT, SDLoc(N), N0, N1, NewMask);
14346   }
14347 
14348   // If it is a splat, check if the argument vector is another splat or a
14349   // build_vector.
14350   if (SVN->isSplat() && SVN->getSplatIndex() < (int)NumElts) {
14351     SDNode *V = N0.getNode();
14352 
14353     // If this is a bit convert that changes the element type of the vector but
14354     // not the number of vector elements, look through it.  Be careful not to
14355     // look though conversions that change things like v4f32 to v2f64.
14356     if (V->getOpcode() == ISD::BITCAST) {
14357       SDValue ConvInput = V->getOperand(0);
14358       if (ConvInput.getValueType().isVector() &&
14359           ConvInput.getValueType().getVectorNumElements() == NumElts)
14360         V = ConvInput.getNode();
14361     }
14362 
14363     if (V->getOpcode() == ISD::BUILD_VECTOR) {
14364       assert(V->getNumOperands() == NumElts &&
14365              "BUILD_VECTOR has wrong number of operands");
14366       SDValue Base;
14367       bool AllSame = true;
14368       for (unsigned i = 0; i != NumElts; ++i) {
14369         if (!V->getOperand(i).isUndef()) {
14370           Base = V->getOperand(i);
14371           break;
14372         }
14373       }
14374       // Splat of <u, u, u, u>, return <u, u, u, u>
14375       if (!Base.getNode())
14376         return N0;
14377       for (unsigned i = 0; i != NumElts; ++i) {
14378         if (V->getOperand(i) != Base) {
14379           AllSame = false;
14380           break;
14381         }
14382       }
14383       // Splat of <x, x, x, x>, return <x, x, x, x>
14384       if (AllSame)
14385         return N0;
14386 
14387       // Canonicalize any other splat as a build_vector.
14388       const SDValue &Splatted = V->getOperand(SVN->getSplatIndex());
14389       SmallVector<SDValue, 8> Ops(NumElts, Splatted);
14390       SDValue NewBV = DAG.getBuildVector(V->getValueType(0), SDLoc(N), Ops);
14391 
14392       // We may have jumped through bitcasts, so the type of the
14393       // BUILD_VECTOR may not match the type of the shuffle.
14394       if (V->getValueType(0) != VT)
14395         NewBV = DAG.getBitcast(VT, NewBV);
14396       return NewBV;
14397     }
14398   }
14399 
14400   // There are various patterns used to build up a vector from smaller vectors,
14401   // subvectors, or elements. Scan chains of these and replace unused insertions
14402   // or components with undef.
14403   if (SDValue S = simplifyShuffleOperands(SVN, N0, N1, DAG))
14404     return S;
14405 
14406   // Match shuffles that can be converted to any_vector_extend_in_reg.
14407   if (SDValue V = combineShuffleToVectorExtend(SVN, DAG, TLI, LegalOperations))
14408     return V;
14409 
14410   // Combine "truncate_vector_in_reg" style shuffles.
14411   if (SDValue V = combineTruncationShuffle(SVN, DAG))
14412     return V;
14413 
14414   if (N0.getOpcode() == ISD::CONCAT_VECTORS &&
14415       Level < AfterLegalizeVectorOps &&
14416       (N1.isUndef() ||
14417       (N1.getOpcode() == ISD::CONCAT_VECTORS &&
14418        N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType()))) {
14419     if (SDValue V = partitionShuffleOfConcats(N, DAG))
14420       return V;
14421   }
14422 
14423   // Attempt to combine a shuffle of 2 inputs of 'scalar sources' -
14424   // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR.
14425   if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT))
14426     if (SDValue Res = combineShuffleOfScalars(SVN, DAG, TLI))
14427       return Res;
14428 
14429   // If this shuffle only has a single input that is a bitcasted shuffle,
14430   // attempt to merge the 2 shuffles and suitably bitcast the inputs/output
14431   // back to their original types.
14432   if (N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() &&
14433       N1.isUndef() && Level < AfterLegalizeVectorOps &&
14434       TLI.isTypeLegal(VT)) {
14435 
14436     // Peek through the bitcast only if there is one user.
14437     SDValue BC0 = N0;
14438     while (BC0.getOpcode() == ISD::BITCAST) {
14439       if (!BC0.hasOneUse())
14440         break;
14441       BC0 = BC0.getOperand(0);
14442     }
14443 
14444     auto ScaleShuffleMask = [](ArrayRef<int> Mask, int Scale) {
14445       if (Scale == 1)
14446         return SmallVector<int, 8>(Mask.begin(), Mask.end());
14447 
14448       SmallVector<int, 8> NewMask;
14449       for (int M : Mask)
14450         for (int s = 0; s != Scale; ++s)
14451           NewMask.push_back(M < 0 ? -1 : Scale * M + s);
14452       return NewMask;
14453     };
14454 
14455     if (BC0.getOpcode() == ISD::VECTOR_SHUFFLE && BC0.hasOneUse()) {
14456       EVT SVT = VT.getScalarType();
14457       EVT InnerVT = BC0->getValueType(0);
14458       EVT InnerSVT = InnerVT.getScalarType();
14459 
14460       // Determine which shuffle works with the smaller scalar type.
14461       EVT ScaleVT = SVT.bitsLT(InnerSVT) ? VT : InnerVT;
14462       EVT ScaleSVT = ScaleVT.getScalarType();
14463 
14464       if (TLI.isTypeLegal(ScaleVT) &&
14465           0 == (InnerSVT.getSizeInBits() % ScaleSVT.getSizeInBits()) &&
14466           0 == (SVT.getSizeInBits() % ScaleSVT.getSizeInBits())) {
14467 
14468         int InnerScale = InnerSVT.getSizeInBits() / ScaleSVT.getSizeInBits();
14469         int OuterScale = SVT.getSizeInBits() / ScaleSVT.getSizeInBits();
14470 
14471         // Scale the shuffle masks to the smaller scalar type.
14472         ShuffleVectorSDNode *InnerSVN = cast<ShuffleVectorSDNode>(BC0);
14473         SmallVector<int, 8> InnerMask =
14474             ScaleShuffleMask(InnerSVN->getMask(), InnerScale);
14475         SmallVector<int, 8> OuterMask =
14476             ScaleShuffleMask(SVN->getMask(), OuterScale);
14477 
14478         // Merge the shuffle masks.
14479         SmallVector<int, 8> NewMask;
14480         for (int M : OuterMask)
14481           NewMask.push_back(M < 0 ? -1 : InnerMask[M]);
14482 
14483         // Test for shuffle mask legality over both commutations.
14484         SDValue SV0 = BC0->getOperand(0);
14485         SDValue SV1 = BC0->getOperand(1);
14486         bool LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
14487         if (!LegalMask) {
14488           std::swap(SV0, SV1);
14489           ShuffleVectorSDNode::commuteMask(NewMask);
14490           LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT);
14491         }
14492 
14493         if (LegalMask) {
14494           SV0 = DAG.getBitcast(ScaleVT, SV0);
14495           SV1 = DAG.getBitcast(ScaleVT, SV1);
14496           return DAG.getBitcast(
14497               VT, DAG.getVectorShuffle(ScaleVT, SDLoc(N), SV0, SV1, NewMask));
14498         }
14499       }
14500     }
14501   }
14502 
14503   // Canonicalize shuffles according to rules:
14504   //  shuffle(A, shuffle(A, B)) -> shuffle(shuffle(A,B), A)
14505   //  shuffle(B, shuffle(A, B)) -> shuffle(shuffle(A,B), B)
14506   //  shuffle(B, shuffle(A, Undef)) -> shuffle(shuffle(A, Undef), B)
14507   if (N1.getOpcode() == ISD::VECTOR_SHUFFLE &&
14508       N0.getOpcode() != ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG &&
14509       TLI.isTypeLegal(VT)) {
14510     // The incoming shuffle must be of the same type as the result of the
14511     // current shuffle.
14512     assert(N1->getOperand(0).getValueType() == VT &&
14513            "Shuffle types don't match");
14514 
14515     SDValue SV0 = N1->getOperand(0);
14516     SDValue SV1 = N1->getOperand(1);
14517     bool HasSameOp0 = N0 == SV0;
14518     bool IsSV1Undef = SV1.isUndef();
14519     if (HasSameOp0 || IsSV1Undef || N0 == SV1)
14520       // Commute the operands of this shuffle so that next rule
14521       // will trigger.
14522       return DAG.getCommutedVectorShuffle(*SVN);
14523   }
14524 
14525   // Try to fold according to rules:
14526   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
14527   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
14528   //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
14529   // Don't try to fold shuffles with illegal type.
14530   // Only fold if this shuffle is the only user of the other shuffle.
14531   if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && N->isOnlyUserOf(N0.getNode()) &&
14532       Level < AfterLegalizeDAG && TLI.isTypeLegal(VT)) {
14533     ShuffleVectorSDNode *OtherSV = cast<ShuffleVectorSDNode>(N0);
14534 
14535     // Don't try to fold splats; they're likely to simplify somehow, or they
14536     // might be free.
14537     if (OtherSV->isSplat())
14538       return SDValue();
14539 
14540     // The incoming shuffle must be of the same type as the result of the
14541     // current shuffle.
14542     assert(OtherSV->getOperand(0).getValueType() == VT &&
14543            "Shuffle types don't match");
14544 
14545     SDValue SV0, SV1;
14546     SmallVector<int, 4> Mask;
14547     // Compute the combined shuffle mask for a shuffle with SV0 as the first
14548     // operand, and SV1 as the second operand.
14549     for (unsigned i = 0; i != NumElts; ++i) {
14550       int Idx = SVN->getMaskElt(i);
14551       if (Idx < 0) {
14552         // Propagate Undef.
14553         Mask.push_back(Idx);
14554         continue;
14555       }
14556 
14557       SDValue CurrentVec;
14558       if (Idx < (int)NumElts) {
14559         // This shuffle index refers to the inner shuffle N0. Lookup the inner
14560         // shuffle mask to identify which vector is actually referenced.
14561         Idx = OtherSV->getMaskElt(Idx);
14562         if (Idx < 0) {
14563           // Propagate Undef.
14564           Mask.push_back(Idx);
14565           continue;
14566         }
14567 
14568         CurrentVec = (Idx < (int) NumElts) ? OtherSV->getOperand(0)
14569                                            : OtherSV->getOperand(1);
14570       } else {
14571         // This shuffle index references an element within N1.
14572         CurrentVec = N1;
14573       }
14574 
14575       // Simple case where 'CurrentVec' is UNDEF.
14576       if (CurrentVec.isUndef()) {
14577         Mask.push_back(-1);
14578         continue;
14579       }
14580 
14581       // Canonicalize the shuffle index. We don't know yet if CurrentVec
14582       // will be the first or second operand of the combined shuffle.
14583       Idx = Idx % NumElts;
14584       if (!SV0.getNode() || SV0 == CurrentVec) {
14585         // Ok. CurrentVec is the left hand side.
14586         // Update the mask accordingly.
14587         SV0 = CurrentVec;
14588         Mask.push_back(Idx);
14589         continue;
14590       }
14591 
14592       // Bail out if we cannot convert the shuffle pair into a single shuffle.
14593       if (SV1.getNode() && SV1 != CurrentVec)
14594         return SDValue();
14595 
14596       // Ok. CurrentVec is the right hand side.
14597       // Update the mask accordingly.
14598       SV1 = CurrentVec;
14599       Mask.push_back(Idx + NumElts);
14600     }
14601 
14602     // Check if all indices in Mask are Undef. In case, propagate Undef.
14603     bool isUndefMask = true;
14604     for (unsigned i = 0; i != NumElts && isUndefMask; ++i)
14605       isUndefMask &= Mask[i] < 0;
14606 
14607     if (isUndefMask)
14608       return DAG.getUNDEF(VT);
14609 
14610     if (!SV0.getNode())
14611       SV0 = DAG.getUNDEF(VT);
14612     if (!SV1.getNode())
14613       SV1 = DAG.getUNDEF(VT);
14614 
14615     // Avoid introducing shuffles with illegal mask.
14616     if (!TLI.isShuffleMaskLegal(Mask, VT)) {
14617       ShuffleVectorSDNode::commuteMask(Mask);
14618 
14619       if (!TLI.isShuffleMaskLegal(Mask, VT))
14620         return SDValue();
14621 
14622       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, A, M2)
14623       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, A, M2)
14624       //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, B, M2)
14625       std::swap(SV0, SV1);
14626     }
14627 
14628     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2)
14629     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2)
14630     //   shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2)
14631     return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, Mask);
14632   }
14633 
14634   return SDValue();
14635 }
14636 
14637 SDValue DAGCombiner::visitSCALAR_TO_VECTOR(SDNode *N) {
14638   SDValue InVal = N->getOperand(0);
14639   EVT VT = N->getValueType(0);
14640 
14641   // Replace a SCALAR_TO_VECTOR(EXTRACT_VECTOR_ELT(V,C0)) pattern
14642   // with a VECTOR_SHUFFLE.
14643   if (InVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
14644     SDValue InVec = InVal->getOperand(0);
14645     SDValue EltNo = InVal->getOperand(1);
14646 
14647     // FIXME: We could support implicit truncation if the shuffle can be
14648     // scaled to a smaller vector scalar type.
14649     ConstantSDNode *C0 = dyn_cast<ConstantSDNode>(EltNo);
14650     if (C0 && VT == InVec.getValueType() &&
14651         VT.getScalarType() == InVal.getValueType()) {
14652       SmallVector<int, 8> NewMask(VT.getVectorNumElements(), -1);
14653       int Elt = C0->getZExtValue();
14654       NewMask[0] = Elt;
14655 
14656       if (TLI.isShuffleMaskLegal(NewMask, VT))
14657         return DAG.getVectorShuffle(VT, SDLoc(N), InVec, DAG.getUNDEF(VT),
14658                                     NewMask);
14659     }
14660   }
14661 
14662   return SDValue();
14663 }
14664 
14665 SDValue DAGCombiner::visitINSERT_SUBVECTOR(SDNode *N) {
14666   EVT VT = N->getValueType(0);
14667   SDValue N0 = N->getOperand(0);
14668   SDValue N1 = N->getOperand(1);
14669   SDValue N2 = N->getOperand(2);
14670 
14671   // If inserting an UNDEF, just return the original vector.
14672   if (N1.isUndef())
14673     return N0;
14674 
14675   // If this is an insert of an extracted vector into an undef vector, we can
14676   // just use the input to the extract.
14677   if (N0.isUndef() && N1.getOpcode() == ISD::EXTRACT_SUBVECTOR &&
14678       N1.getOperand(1) == N2 && N1.getOperand(0).getValueType() == VT)
14679     return N1.getOperand(0);
14680 
14681   // Combine INSERT_SUBVECTORs where we are inserting to the same index.
14682   // INSERT_SUBVECTOR( INSERT_SUBVECTOR( Vec, SubOld, Idx ), SubNew, Idx )
14683   // --> INSERT_SUBVECTOR( Vec, SubNew, Idx )
14684   if (N0.getOpcode() == ISD::INSERT_SUBVECTOR &&
14685       N0.getOperand(1).getValueType() == N1.getValueType() &&
14686       N0.getOperand(2) == N2)
14687     return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0.getOperand(0),
14688                        N1, N2);
14689 
14690   if (!isa<ConstantSDNode>(N2))
14691     return SDValue();
14692 
14693   unsigned InsIdx = cast<ConstantSDNode>(N2)->getZExtValue();
14694 
14695   // Canonicalize insert_subvector dag nodes.
14696   // Example:
14697   // (insert_subvector (insert_subvector A, Idx0), Idx1)
14698   // -> (insert_subvector (insert_subvector A, Idx1), Idx0)
14699   if (N0.getOpcode() == ISD::INSERT_SUBVECTOR && N0.hasOneUse() &&
14700       N1.getValueType() == N0.getOperand(1).getValueType() &&
14701       isa<ConstantSDNode>(N0.getOperand(2))) {
14702     unsigned OtherIdx = cast<ConstantSDNode>(N0.getOperand(2))->getZExtValue();
14703     if (InsIdx < OtherIdx) {
14704       // Swap nodes.
14705       SDValue NewOp = DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT,
14706                                   N0.getOperand(0), N1, N2);
14707       AddToWorklist(NewOp.getNode());
14708       return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N0.getNode()),
14709                          VT, NewOp, N0.getOperand(1), N0.getOperand(2));
14710     }
14711   }
14712 
14713   // If the input vector is a concatenation, and the insert replaces
14714   // one of the pieces, we can optimize into a single concat_vectors.
14715   if (N0.getOpcode() == ISD::CONCAT_VECTORS && N0.hasOneUse() &&
14716       N0.getOperand(0).getValueType() == N1.getValueType()) {
14717     unsigned Factor = N1.getValueType().getVectorNumElements();
14718 
14719     SmallVector<SDValue, 8> Ops(N0->op_begin(), N0->op_end());
14720     Ops[cast<ConstantSDNode>(N2)->getZExtValue() / Factor] = N1;
14721 
14722     return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops);
14723   }
14724 
14725   return SDValue();
14726 }
14727 
14728 SDValue DAGCombiner::visitFP_TO_FP16(SDNode *N) {
14729   SDValue N0 = N->getOperand(0);
14730 
14731   // fold (fp_to_fp16 (fp16_to_fp op)) -> op
14732   if (N0->getOpcode() == ISD::FP16_TO_FP)
14733     return N0->getOperand(0);
14734 
14735   return SDValue();
14736 }
14737 
14738 SDValue DAGCombiner::visitFP16_TO_FP(SDNode *N) {
14739   SDValue N0 = N->getOperand(0);
14740 
14741   // fold fp16_to_fp(op & 0xffff) -> fp16_to_fp(op)
14742   if (N0->getOpcode() == ISD::AND) {
14743     ConstantSDNode *AndConst = getAsNonOpaqueConstant(N0.getOperand(1));
14744     if (AndConst && AndConst->getAPIntValue() == 0xffff) {
14745       return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), N->getValueType(0),
14746                          N0.getOperand(0));
14747     }
14748   }
14749 
14750   return SDValue();
14751 }
14752 
14753 /// Returns a vector_shuffle if it able to transform an AND to a vector_shuffle
14754 /// with the destination vector and a zero vector.
14755 /// e.g. AND V, <0xffffffff, 0, 0xffffffff, 0>. ==>
14756 ///      vector_shuffle V, Zero, <0, 4, 2, 4>
14757 SDValue DAGCombiner::XformToShuffleWithZero(SDNode *N) {
14758   EVT VT = N->getValueType(0);
14759   SDValue LHS = N->getOperand(0);
14760   SDValue RHS = N->getOperand(1);
14761   SDLoc DL(N);
14762 
14763   // Make sure we're not running after operation legalization where it
14764   // may have custom lowered the vector shuffles.
14765   if (LegalOperations)
14766     return SDValue();
14767 
14768   if (N->getOpcode() != ISD::AND)
14769     return SDValue();
14770 
14771   if (RHS.getOpcode() == ISD::BITCAST)
14772     RHS = RHS.getOperand(0);
14773 
14774   if (RHS.getOpcode() != ISD::BUILD_VECTOR)
14775     return SDValue();
14776 
14777   EVT RVT = RHS.getValueType();
14778   unsigned NumElts = RHS.getNumOperands();
14779 
14780   // Attempt to create a valid clear mask, splitting the mask into
14781   // sub elements and checking to see if each is
14782   // all zeros or all ones - suitable for shuffle masking.
14783   auto BuildClearMask = [&](int Split) {
14784     int NumSubElts = NumElts * Split;
14785     int NumSubBits = RVT.getScalarSizeInBits() / Split;
14786 
14787     SmallVector<int, 8> Indices;
14788     for (int i = 0; i != NumSubElts; ++i) {
14789       int EltIdx = i / Split;
14790       int SubIdx = i % Split;
14791       SDValue Elt = RHS.getOperand(EltIdx);
14792       if (Elt.isUndef()) {
14793         Indices.push_back(-1);
14794         continue;
14795       }
14796 
14797       APInt Bits;
14798       if (isa<ConstantSDNode>(Elt))
14799         Bits = cast<ConstantSDNode>(Elt)->getAPIntValue();
14800       else if (isa<ConstantFPSDNode>(Elt))
14801         Bits = cast<ConstantFPSDNode>(Elt)->getValueAPF().bitcastToAPInt();
14802       else
14803         return SDValue();
14804 
14805       // Extract the sub element from the constant bit mask.
14806       if (DAG.getDataLayout().isBigEndian()) {
14807         Bits = Bits.lshr((Split - SubIdx - 1) * NumSubBits);
14808       } else {
14809         Bits = Bits.lshr(SubIdx * NumSubBits);
14810       }
14811 
14812       if (Split > 1)
14813         Bits = Bits.trunc(NumSubBits);
14814 
14815       if (Bits.isAllOnesValue())
14816         Indices.push_back(i);
14817       else if (Bits == 0)
14818         Indices.push_back(i + NumSubElts);
14819       else
14820         return SDValue();
14821     }
14822 
14823     // Let's see if the target supports this vector_shuffle.
14824     EVT ClearSVT = EVT::getIntegerVT(*DAG.getContext(), NumSubBits);
14825     EVT ClearVT = EVT::getVectorVT(*DAG.getContext(), ClearSVT, NumSubElts);
14826     if (!TLI.isVectorClearMaskLegal(Indices, ClearVT))
14827       return SDValue();
14828 
14829     SDValue Zero = DAG.getConstant(0, DL, ClearVT);
14830     return DAG.getBitcast(VT, DAG.getVectorShuffle(ClearVT, DL,
14831                                                    DAG.getBitcast(ClearVT, LHS),
14832                                                    Zero, Indices));
14833   };
14834 
14835   // Determine maximum split level (byte level masking).
14836   int MaxSplit = 1;
14837   if (RVT.getScalarSizeInBits() % 8 == 0)
14838     MaxSplit = RVT.getScalarSizeInBits() / 8;
14839 
14840   for (int Split = 1; Split <= MaxSplit; ++Split)
14841     if (RVT.getScalarSizeInBits() % Split == 0)
14842       if (SDValue S = BuildClearMask(Split))
14843         return S;
14844 
14845   return SDValue();
14846 }
14847 
14848 /// Visit a binary vector operation, like ADD.
14849 SDValue DAGCombiner::SimplifyVBinOp(SDNode *N) {
14850   assert(N->getValueType(0).isVector() &&
14851          "SimplifyVBinOp only works on vectors!");
14852 
14853   SDValue LHS = N->getOperand(0);
14854   SDValue RHS = N->getOperand(1);
14855   SDValue Ops[] = {LHS, RHS};
14856 
14857   // See if we can constant fold the vector operation.
14858   if (SDValue Fold = DAG.FoldConstantVectorArithmetic(
14859           N->getOpcode(), SDLoc(LHS), LHS.getValueType(), Ops, N->getFlags()))
14860     return Fold;
14861 
14862   // Try to convert a constant mask AND into a shuffle clear mask.
14863   if (SDValue Shuffle = XformToShuffleWithZero(N))
14864     return Shuffle;
14865 
14866   // Type legalization might introduce new shuffles in the DAG.
14867   // Fold (VBinOp (shuffle (A, Undef, Mask)), (shuffle (B, Undef, Mask)))
14868   //   -> (shuffle (VBinOp (A, B)), Undef, Mask).
14869   if (LegalTypes && isa<ShuffleVectorSDNode>(LHS) &&
14870       isa<ShuffleVectorSDNode>(RHS) && LHS.hasOneUse() && RHS.hasOneUse() &&
14871       LHS.getOperand(1).isUndef() &&
14872       RHS.getOperand(1).isUndef()) {
14873     ShuffleVectorSDNode *SVN0 = cast<ShuffleVectorSDNode>(LHS);
14874     ShuffleVectorSDNode *SVN1 = cast<ShuffleVectorSDNode>(RHS);
14875 
14876     if (SVN0->getMask().equals(SVN1->getMask())) {
14877       EVT VT = N->getValueType(0);
14878       SDValue UndefVector = LHS.getOperand(1);
14879       SDValue NewBinOp = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
14880                                      LHS.getOperand(0), RHS.getOperand(0),
14881                                      N->getFlags());
14882       AddUsersToWorklist(N);
14883       return DAG.getVectorShuffle(VT, SDLoc(N), NewBinOp, UndefVector,
14884                                   SVN0->getMask());
14885     }
14886   }
14887 
14888   return SDValue();
14889 }
14890 
14891 SDValue DAGCombiner::SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1,
14892                                     SDValue N2) {
14893   assert(N0.getOpcode() ==ISD::SETCC && "First argument must be a SetCC node!");
14894 
14895   SDValue SCC = SimplifySelectCC(DL, N0.getOperand(0), N0.getOperand(1), N1, N2,
14896                                  cast<CondCodeSDNode>(N0.getOperand(2))->get());
14897 
14898   // If we got a simplified select_cc node back from SimplifySelectCC, then
14899   // break it down into a new SETCC node, and a new SELECT node, and then return
14900   // the SELECT node, since we were called with a SELECT node.
14901   if (SCC.getNode()) {
14902     // Check to see if we got a select_cc back (to turn into setcc/select).
14903     // Otherwise, just return whatever node we got back, like fabs.
14904     if (SCC.getOpcode() == ISD::SELECT_CC) {
14905       SDValue SETCC = DAG.getNode(ISD::SETCC, SDLoc(N0),
14906                                   N0.getValueType(),
14907                                   SCC.getOperand(0), SCC.getOperand(1),
14908                                   SCC.getOperand(4));
14909       AddToWorklist(SETCC.getNode());
14910       return DAG.getSelect(SDLoc(SCC), SCC.getValueType(), SETCC,
14911                            SCC.getOperand(2), SCC.getOperand(3));
14912     }
14913 
14914     return SCC;
14915   }
14916   return SDValue();
14917 }
14918 
14919 /// Given a SELECT or a SELECT_CC node, where LHS and RHS are the two values
14920 /// being selected between, see if we can simplify the select.  Callers of this
14921 /// should assume that TheSelect is deleted if this returns true.  As such, they
14922 /// should return the appropriate thing (e.g. the node) back to the top-level of
14923 /// the DAG combiner loop to avoid it being looked at.
14924 bool DAGCombiner::SimplifySelectOps(SDNode *TheSelect, SDValue LHS,
14925                                     SDValue RHS) {
14926 
14927   // fold (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
14928   // The select + setcc is redundant, because fsqrt returns NaN for X < 0.
14929   if (const ConstantFPSDNode *NaN = isConstOrConstSplatFP(LHS)) {
14930     if (NaN->isNaN() && RHS.getOpcode() == ISD::FSQRT) {
14931       // We have: (select (setcc ?, ?, ?), NaN, (fsqrt ?))
14932       SDValue Sqrt = RHS;
14933       ISD::CondCode CC;
14934       SDValue CmpLHS;
14935       const ConstantFPSDNode *Zero = nullptr;
14936 
14937       if (TheSelect->getOpcode() == ISD::SELECT_CC) {
14938         CC = dyn_cast<CondCodeSDNode>(TheSelect->getOperand(4))->get();
14939         CmpLHS = TheSelect->getOperand(0);
14940         Zero = isConstOrConstSplatFP(TheSelect->getOperand(1));
14941       } else {
14942         // SELECT or VSELECT
14943         SDValue Cmp = TheSelect->getOperand(0);
14944         if (Cmp.getOpcode() == ISD::SETCC) {
14945           CC = dyn_cast<CondCodeSDNode>(Cmp.getOperand(2))->get();
14946           CmpLHS = Cmp.getOperand(0);
14947           Zero = isConstOrConstSplatFP(Cmp.getOperand(1));
14948         }
14949       }
14950       if (Zero && Zero->isZero() &&
14951           Sqrt.getOperand(0) == CmpLHS && (CC == ISD::SETOLT ||
14952           CC == ISD::SETULT || CC == ISD::SETLT)) {
14953         // We have: (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x))
14954         CombineTo(TheSelect, Sqrt);
14955         return true;
14956       }
14957     }
14958   }
14959   // Cannot simplify select with vector condition
14960   if (TheSelect->getOperand(0).getValueType().isVector()) return false;
14961 
14962   // If this is a select from two identical things, try to pull the operation
14963   // through the select.
14964   if (LHS.getOpcode() != RHS.getOpcode() ||
14965       !LHS.hasOneUse() || !RHS.hasOneUse())
14966     return false;
14967 
14968   // If this is a load and the token chain is identical, replace the select
14969   // of two loads with a load through a select of the address to load from.
14970   // This triggers in things like "select bool X, 10.0, 123.0" after the FP
14971   // constants have been dropped into the constant pool.
14972   if (LHS.getOpcode() == ISD::LOAD) {
14973     LoadSDNode *LLD = cast<LoadSDNode>(LHS);
14974     LoadSDNode *RLD = cast<LoadSDNode>(RHS);
14975 
14976     // Token chains must be identical.
14977     if (LHS.getOperand(0) != RHS.getOperand(0) ||
14978         // Do not let this transformation reduce the number of volatile loads.
14979         LLD->isVolatile() || RLD->isVolatile() ||
14980         // FIXME: If either is a pre/post inc/dec load,
14981         // we'd need to split out the address adjustment.
14982         LLD->isIndexed() || RLD->isIndexed() ||
14983         // If this is an EXTLOAD, the VT's must match.
14984         LLD->getMemoryVT() != RLD->getMemoryVT() ||
14985         // If this is an EXTLOAD, the kind of extension must match.
14986         (LLD->getExtensionType() != RLD->getExtensionType() &&
14987          // The only exception is if one of the extensions is anyext.
14988          LLD->getExtensionType() != ISD::EXTLOAD &&
14989          RLD->getExtensionType() != ISD::EXTLOAD) ||
14990         // FIXME: this discards src value information.  This is
14991         // over-conservative. It would be beneficial to be able to remember
14992         // both potential memory locations.  Since we are discarding
14993         // src value info, don't do the transformation if the memory
14994         // locations are not in the default address space.
14995         LLD->getPointerInfo().getAddrSpace() != 0 ||
14996         RLD->getPointerInfo().getAddrSpace() != 0 ||
14997         !TLI.isOperationLegalOrCustom(TheSelect->getOpcode(),
14998                                       LLD->getBasePtr().getValueType()))
14999       return false;
15000 
15001     // Check that the select condition doesn't reach either load.  If so,
15002     // folding this will induce a cycle into the DAG.  If not, this is safe to
15003     // xform, so create a select of the addresses.
15004     SDValue Addr;
15005     if (TheSelect->getOpcode() == ISD::SELECT) {
15006       SDNode *CondNode = TheSelect->getOperand(0).getNode();
15007       if ((LLD->hasAnyUseOfValue(1) && LLD->isPredecessorOf(CondNode)) ||
15008           (RLD->hasAnyUseOfValue(1) && RLD->isPredecessorOf(CondNode)))
15009         return false;
15010       // The loads must not depend on one another.
15011       if (LLD->isPredecessorOf(RLD) ||
15012           RLD->isPredecessorOf(LLD))
15013         return false;
15014       Addr = DAG.getSelect(SDLoc(TheSelect),
15015                            LLD->getBasePtr().getValueType(),
15016                            TheSelect->getOperand(0), LLD->getBasePtr(),
15017                            RLD->getBasePtr());
15018     } else {  // Otherwise SELECT_CC
15019       SDNode *CondLHS = TheSelect->getOperand(0).getNode();
15020       SDNode *CondRHS = TheSelect->getOperand(1).getNode();
15021 
15022       if ((LLD->hasAnyUseOfValue(1) &&
15023            (LLD->isPredecessorOf(CondLHS) || LLD->isPredecessorOf(CondRHS))) ||
15024           (RLD->hasAnyUseOfValue(1) &&
15025            (RLD->isPredecessorOf(CondLHS) || RLD->isPredecessorOf(CondRHS))))
15026         return false;
15027 
15028       Addr = DAG.getNode(ISD::SELECT_CC, SDLoc(TheSelect),
15029                          LLD->getBasePtr().getValueType(),
15030                          TheSelect->getOperand(0),
15031                          TheSelect->getOperand(1),
15032                          LLD->getBasePtr(), RLD->getBasePtr(),
15033                          TheSelect->getOperand(4));
15034     }
15035 
15036     SDValue Load;
15037     // It is safe to replace the two loads if they have different alignments,
15038     // but the new load must be the minimum (most restrictive) alignment of the
15039     // inputs.
15040     unsigned Alignment = std::min(LLD->getAlignment(), RLD->getAlignment());
15041     MachineMemOperand::Flags MMOFlags = LLD->getMemOperand()->getFlags();
15042     if (!RLD->isInvariant())
15043       MMOFlags &= ~MachineMemOperand::MOInvariant;
15044     if (!RLD->isDereferenceable())
15045       MMOFlags &= ~MachineMemOperand::MODereferenceable;
15046     if (LLD->getExtensionType() == ISD::NON_EXTLOAD) {
15047       // FIXME: Discards pointer and AA info.
15048       Load = DAG.getLoad(TheSelect->getValueType(0), SDLoc(TheSelect),
15049                          LLD->getChain(), Addr, MachinePointerInfo(), Alignment,
15050                          MMOFlags);
15051     } else {
15052       // FIXME: Discards pointer and AA info.
15053       Load = DAG.getExtLoad(
15054           LLD->getExtensionType() == ISD::EXTLOAD ? RLD->getExtensionType()
15055                                                   : LLD->getExtensionType(),
15056           SDLoc(TheSelect), TheSelect->getValueType(0), LLD->getChain(), Addr,
15057           MachinePointerInfo(), LLD->getMemoryVT(), Alignment, MMOFlags);
15058     }
15059 
15060     // Users of the select now use the result of the load.
15061     CombineTo(TheSelect, Load);
15062 
15063     // Users of the old loads now use the new load's chain.  We know the
15064     // old-load value is dead now.
15065     CombineTo(LHS.getNode(), Load.getValue(0), Load.getValue(1));
15066     CombineTo(RHS.getNode(), Load.getValue(0), Load.getValue(1));
15067     return true;
15068   }
15069 
15070   return false;
15071 }
15072 
15073 /// Try to fold an expression of the form (N0 cond N1) ? N2 : N3 to a shift and
15074 /// bitwise 'and'.
15075 SDValue DAGCombiner::foldSelectCCToShiftAnd(const SDLoc &DL, SDValue N0,
15076                                             SDValue N1, SDValue N2, SDValue N3,
15077                                             ISD::CondCode CC) {
15078   // If this is a select where the false operand is zero and the compare is a
15079   // check of the sign bit, see if we can perform the "gzip trick":
15080   // select_cc setlt X, 0, A, 0 -> and (sra X, size(X)-1), A
15081   // select_cc setgt X, 0, A, 0 -> and (not (sra X, size(X)-1)), A
15082   EVT XType = N0.getValueType();
15083   EVT AType = N2.getValueType();
15084   if (!isNullConstant(N3) || !XType.bitsGE(AType))
15085     return SDValue();
15086 
15087   // If the comparison is testing for a positive value, we have to invert
15088   // the sign bit mask, so only do that transform if the target has a bitwise
15089   // 'and not' instruction (the invert is free).
15090   if (CC == ISD::SETGT && TLI.hasAndNot(N2)) {
15091     // (X > -1) ? A : 0
15092     // (X >  0) ? X : 0 <-- This is canonical signed max.
15093     if (!(isAllOnesConstant(N1) || (isNullConstant(N1) && N0 == N2)))
15094       return SDValue();
15095   } else if (CC == ISD::SETLT) {
15096     // (X <  0) ? A : 0
15097     // (X <  1) ? X : 0 <-- This is un-canonicalized signed min.
15098     if (!(isNullConstant(N1) || (isOneConstant(N1) && N0 == N2)))
15099       return SDValue();
15100   } else {
15101     return SDValue();
15102   }
15103 
15104   // and (sra X, size(X)-1), A -> "and (srl X, C2), A" iff A is a single-bit
15105   // constant.
15106   EVT ShiftAmtTy = getShiftAmountTy(N0.getValueType());
15107   auto *N2C = dyn_cast<ConstantSDNode>(N2.getNode());
15108   if (N2C && ((N2C->getAPIntValue() & (N2C->getAPIntValue() - 1)) == 0)) {
15109     unsigned ShCt = XType.getSizeInBits() - N2C->getAPIntValue().logBase2() - 1;
15110     SDValue ShiftAmt = DAG.getConstant(ShCt, DL, ShiftAmtTy);
15111     SDValue Shift = DAG.getNode(ISD::SRL, DL, XType, N0, ShiftAmt);
15112     AddToWorklist(Shift.getNode());
15113 
15114     if (XType.bitsGT(AType)) {
15115       Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
15116       AddToWorklist(Shift.getNode());
15117     }
15118 
15119     if (CC == ISD::SETGT)
15120       Shift = DAG.getNOT(DL, Shift, AType);
15121 
15122     return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
15123   }
15124 
15125   SDValue ShiftAmt = DAG.getConstant(XType.getSizeInBits() - 1, DL, ShiftAmtTy);
15126   SDValue Shift = DAG.getNode(ISD::SRA, DL, XType, N0, ShiftAmt);
15127   AddToWorklist(Shift.getNode());
15128 
15129   if (XType.bitsGT(AType)) {
15130     Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift);
15131     AddToWorklist(Shift.getNode());
15132   }
15133 
15134   if (CC == ISD::SETGT)
15135     Shift = DAG.getNOT(DL, Shift, AType);
15136 
15137   return DAG.getNode(ISD::AND, DL, AType, Shift, N2);
15138 }
15139 
15140 /// Simplify an expression of the form (N0 cond N1) ? N2 : N3
15141 /// where 'cond' is the comparison specified by CC.
15142 SDValue DAGCombiner::SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1,
15143                                       SDValue N2, SDValue N3, ISD::CondCode CC,
15144                                       bool NotExtCompare) {
15145   // (x ? y : y) -> y.
15146   if (N2 == N3) return N2;
15147 
15148   EVT VT = N2.getValueType();
15149   ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N1.getNode());
15150   ConstantSDNode *N2C = dyn_cast<ConstantSDNode>(N2.getNode());
15151 
15152   // Determine if the condition we're dealing with is constant
15153   SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()),
15154                               N0, N1, CC, DL, false);
15155   if (SCC.getNode()) AddToWorklist(SCC.getNode());
15156 
15157   if (ConstantSDNode *SCCC = dyn_cast_or_null<ConstantSDNode>(SCC.getNode())) {
15158     // fold select_cc true, x, y -> x
15159     // fold select_cc false, x, y -> y
15160     return !SCCC->isNullValue() ? N2 : N3;
15161   }
15162 
15163   // Check to see if we can simplify the select into an fabs node
15164   if (ConstantFPSDNode *CFP = dyn_cast<ConstantFPSDNode>(N1)) {
15165     // Allow either -0.0 or 0.0
15166     if (CFP->isZero()) {
15167       // select (setg[te] X, +/-0.0), X, fneg(X) -> fabs
15168       if ((CC == ISD::SETGE || CC == ISD::SETGT) &&
15169           N0 == N2 && N3.getOpcode() == ISD::FNEG &&
15170           N2 == N3.getOperand(0))
15171         return DAG.getNode(ISD::FABS, DL, VT, N0);
15172 
15173       // select (setl[te] X, +/-0.0), fneg(X), X -> fabs
15174       if ((CC == ISD::SETLT || CC == ISD::SETLE) &&
15175           N0 == N3 && N2.getOpcode() == ISD::FNEG &&
15176           N2.getOperand(0) == N3)
15177         return DAG.getNode(ISD::FABS, DL, VT, N3);
15178     }
15179   }
15180 
15181   // Turn "(a cond b) ? 1.0f : 2.0f" into "load (tmp + ((a cond b) ? 0 : 4)"
15182   // where "tmp" is a constant pool entry containing an array with 1.0 and 2.0
15183   // in it.  This is a win when the constant is not otherwise available because
15184   // it replaces two constant pool loads with one.  We only do this if the FP
15185   // type is known to be legal, because if it isn't, then we are before legalize
15186   // types an we want the other legalization to happen first (e.g. to avoid
15187   // messing with soft float) and if the ConstantFP is not legal, because if
15188   // it is legal, we may not need to store the FP constant in a constant pool.
15189   if (ConstantFPSDNode *TV = dyn_cast<ConstantFPSDNode>(N2))
15190     if (ConstantFPSDNode *FV = dyn_cast<ConstantFPSDNode>(N3)) {
15191       if (TLI.isTypeLegal(N2.getValueType()) &&
15192           (TLI.getOperationAction(ISD::ConstantFP, N2.getValueType()) !=
15193                TargetLowering::Legal &&
15194            !TLI.isFPImmLegal(TV->getValueAPF(), TV->getValueType(0)) &&
15195            !TLI.isFPImmLegal(FV->getValueAPF(), FV->getValueType(0))) &&
15196           // If both constants have multiple uses, then we won't need to do an
15197           // extra load, they are likely around in registers for other users.
15198           (TV->hasOneUse() || FV->hasOneUse())) {
15199         Constant *Elts[] = {
15200           const_cast<ConstantFP*>(FV->getConstantFPValue()),
15201           const_cast<ConstantFP*>(TV->getConstantFPValue())
15202         };
15203         Type *FPTy = Elts[0]->getType();
15204         const DataLayout &TD = DAG.getDataLayout();
15205 
15206         // Create a ConstantArray of the two constants.
15207         Constant *CA = ConstantArray::get(ArrayType::get(FPTy, 2), Elts);
15208         SDValue CPIdx =
15209             DAG.getConstantPool(CA, TLI.getPointerTy(DAG.getDataLayout()),
15210                                 TD.getPrefTypeAlignment(FPTy));
15211         unsigned Alignment = cast<ConstantPoolSDNode>(CPIdx)->getAlignment();
15212 
15213         // Get the offsets to the 0 and 1 element of the array so that we can
15214         // select between them.
15215         SDValue Zero = DAG.getIntPtrConstant(0, DL);
15216         unsigned EltSize = (unsigned)TD.getTypeAllocSize(Elts[0]->getType());
15217         SDValue One = DAG.getIntPtrConstant(EltSize, SDLoc(FV));
15218 
15219         SDValue Cond = DAG.getSetCC(DL,
15220                                     getSetCCResultType(N0.getValueType()),
15221                                     N0, N1, CC);
15222         AddToWorklist(Cond.getNode());
15223         SDValue CstOffset = DAG.getSelect(DL, Zero.getValueType(),
15224                                           Cond, One, Zero);
15225         AddToWorklist(CstOffset.getNode());
15226         CPIdx = DAG.getNode(ISD::ADD, DL, CPIdx.getValueType(), CPIdx,
15227                             CstOffset);
15228         AddToWorklist(CPIdx.getNode());
15229         return DAG.getLoad(
15230             TV->getValueType(0), DL, DAG.getEntryNode(), CPIdx,
15231             MachinePointerInfo::getConstantPool(DAG.getMachineFunction()),
15232             Alignment);
15233       }
15234     }
15235 
15236   if (SDValue V = foldSelectCCToShiftAnd(DL, N0, N1, N2, N3, CC))
15237     return V;
15238 
15239   // fold (select_cc seteq (and x, y), 0, 0, A) -> (and (shr (shl x)) A)
15240   // where y is has a single bit set.
15241   // A plaintext description would be, we can turn the SELECT_CC into an AND
15242   // when the condition can be materialized as an all-ones register.  Any
15243   // single bit-test can be materialized as an all-ones register with
15244   // shift-left and shift-right-arith.
15245   if (CC == ISD::SETEQ && N0->getOpcode() == ISD::AND &&
15246       N0->getValueType(0) == VT && isNullConstant(N1) && isNullConstant(N2)) {
15247     SDValue AndLHS = N0->getOperand(0);
15248     ConstantSDNode *ConstAndRHS = dyn_cast<ConstantSDNode>(N0->getOperand(1));
15249     if (ConstAndRHS && ConstAndRHS->getAPIntValue().countPopulation() == 1) {
15250       // Shift the tested bit over the sign bit.
15251       const APInt &AndMask = ConstAndRHS->getAPIntValue();
15252       SDValue ShlAmt =
15253         DAG.getConstant(AndMask.countLeadingZeros(), SDLoc(AndLHS),
15254                         getShiftAmountTy(AndLHS.getValueType()));
15255       SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N0), VT, AndLHS, ShlAmt);
15256 
15257       // Now arithmetic right shift it all the way over, so the result is either
15258       // all-ones, or zero.
15259       SDValue ShrAmt =
15260         DAG.getConstant(AndMask.getBitWidth() - 1, SDLoc(Shl),
15261                         getShiftAmountTy(Shl.getValueType()));
15262       SDValue Shr = DAG.getNode(ISD::SRA, SDLoc(N0), VT, Shl, ShrAmt);
15263 
15264       return DAG.getNode(ISD::AND, DL, VT, Shr, N3);
15265     }
15266   }
15267 
15268   // fold select C, 16, 0 -> shl C, 4
15269   if (N2C && isNullConstant(N3) && N2C->getAPIntValue().isPowerOf2() &&
15270       TLI.getBooleanContents(N0.getValueType()) ==
15271           TargetLowering::ZeroOrOneBooleanContent) {
15272 
15273     // If the caller doesn't want us to simplify this into a zext of a compare,
15274     // don't do it.
15275     if (NotExtCompare && N2C->isOne())
15276       return SDValue();
15277 
15278     // Get a SetCC of the condition
15279     // NOTE: Don't create a SETCC if it's not legal on this target.
15280     if (!LegalOperations ||
15281         TLI.isOperationLegal(ISD::SETCC, N0.getValueType())) {
15282       SDValue Temp, SCC;
15283       // cast from setcc result type to select result type
15284       if (LegalTypes) {
15285         SCC  = DAG.getSetCC(DL, getSetCCResultType(N0.getValueType()),
15286                             N0, N1, CC);
15287         if (N2.getValueType().bitsLT(SCC.getValueType()))
15288           Temp = DAG.getZeroExtendInReg(SCC, SDLoc(N2),
15289                                         N2.getValueType());
15290         else
15291           Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2),
15292                              N2.getValueType(), SCC);
15293       } else {
15294         SCC  = DAG.getSetCC(SDLoc(N0), MVT::i1, N0, N1, CC);
15295         Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2),
15296                            N2.getValueType(), SCC);
15297       }
15298 
15299       AddToWorklist(SCC.getNode());
15300       AddToWorklist(Temp.getNode());
15301 
15302       if (N2C->isOne())
15303         return Temp;
15304 
15305       // shl setcc result by log2 n2c
15306       return DAG.getNode(
15307           ISD::SHL, DL, N2.getValueType(), Temp,
15308           DAG.getConstant(N2C->getAPIntValue().logBase2(), SDLoc(Temp),
15309                           getShiftAmountTy(Temp.getValueType())));
15310     }
15311   }
15312 
15313   // Check to see if this is an integer abs.
15314   // select_cc setg[te] X,  0,  X, -X ->
15315   // select_cc setgt    X, -1,  X, -X ->
15316   // select_cc setl[te] X,  0, -X,  X ->
15317   // select_cc setlt    X,  1, -X,  X ->
15318   // Y = sra (X, size(X)-1); xor (add (X, Y), Y)
15319   if (N1C) {
15320     ConstantSDNode *SubC = nullptr;
15321     if (((N1C->isNullValue() && (CC == ISD::SETGT || CC == ISD::SETGE)) ||
15322          (N1C->isAllOnesValue() && CC == ISD::SETGT)) &&
15323         N0 == N2 && N3.getOpcode() == ISD::SUB && N0 == N3.getOperand(1))
15324       SubC = dyn_cast<ConstantSDNode>(N3.getOperand(0));
15325     else if (((N1C->isNullValue() && (CC == ISD::SETLT || CC == ISD::SETLE)) ||
15326               (N1C->isOne() && CC == ISD::SETLT)) &&
15327              N0 == N3 && N2.getOpcode() == ISD::SUB && N0 == N2.getOperand(1))
15328       SubC = dyn_cast<ConstantSDNode>(N2.getOperand(0));
15329 
15330     EVT XType = N0.getValueType();
15331     if (SubC && SubC->isNullValue() && XType.isInteger()) {
15332       SDLoc DL(N0);
15333       SDValue Shift = DAG.getNode(ISD::SRA, DL, XType,
15334                                   N0,
15335                                   DAG.getConstant(XType.getSizeInBits() - 1, DL,
15336                                          getShiftAmountTy(N0.getValueType())));
15337       SDValue Add = DAG.getNode(ISD::ADD, DL,
15338                                 XType, N0, Shift);
15339       AddToWorklist(Shift.getNode());
15340       AddToWorklist(Add.getNode());
15341       return DAG.getNode(ISD::XOR, DL, XType, Add, Shift);
15342     }
15343   }
15344 
15345   // select_cc seteq X, 0, sizeof(X), ctlz(X) -> ctlz(X)
15346   // select_cc seteq X, 0, sizeof(X), ctlz_zero_undef(X) -> ctlz(X)
15347   // select_cc seteq X, 0, sizeof(X), cttz(X) -> cttz(X)
15348   // select_cc seteq X, 0, sizeof(X), cttz_zero_undef(X) -> cttz(X)
15349   // select_cc setne X, 0, ctlz(X), sizeof(X) -> ctlz(X)
15350   // select_cc setne X, 0, ctlz_zero_undef(X), sizeof(X) -> ctlz(X)
15351   // select_cc setne X, 0, cttz(X), sizeof(X) -> cttz(X)
15352   // select_cc setne X, 0, cttz_zero_undef(X), sizeof(X) -> cttz(X)
15353   if (N1C && N1C->isNullValue() && (CC == ISD::SETEQ || CC == ISD::SETNE)) {
15354     SDValue ValueOnZero = N2;
15355     SDValue Count = N3;
15356     // If the condition is NE instead of E, swap the operands.
15357     if (CC == ISD::SETNE)
15358       std::swap(ValueOnZero, Count);
15359     // Check if the value on zero is a constant equal to the bits in the type.
15360     if (auto *ValueOnZeroC = dyn_cast<ConstantSDNode>(ValueOnZero)) {
15361       if (ValueOnZeroC->getAPIntValue() == VT.getSizeInBits()) {
15362         // If the other operand is cttz/cttz_zero_undef of N0, and cttz is
15363         // legal, combine to just cttz.
15364         if ((Count.getOpcode() == ISD::CTTZ ||
15365              Count.getOpcode() == ISD::CTTZ_ZERO_UNDEF) &&
15366             N0 == Count.getOperand(0) &&
15367             (!LegalOperations || TLI.isOperationLegal(ISD::CTTZ, VT)))
15368           return DAG.getNode(ISD::CTTZ, DL, VT, N0);
15369         // If the other operand is ctlz/ctlz_zero_undef of N0, and ctlz is
15370         // legal, combine to just ctlz.
15371         if ((Count.getOpcode() == ISD::CTLZ ||
15372              Count.getOpcode() == ISD::CTLZ_ZERO_UNDEF) &&
15373             N0 == Count.getOperand(0) &&
15374             (!LegalOperations || TLI.isOperationLegal(ISD::CTLZ, VT)))
15375           return DAG.getNode(ISD::CTLZ, DL, VT, N0);
15376       }
15377     }
15378   }
15379 
15380   return SDValue();
15381 }
15382 
15383 /// This is a stub for TargetLowering::SimplifySetCC.
15384 SDValue DAGCombiner::SimplifySetCC(EVT VT, SDValue N0, SDValue N1,
15385                                    ISD::CondCode Cond, const SDLoc &DL,
15386                                    bool foldBooleans) {
15387   TargetLowering::DAGCombinerInfo
15388     DagCombineInfo(DAG, Level, false, this);
15389   return TLI.SimplifySetCC(VT, N0, N1, Cond, foldBooleans, DagCombineInfo, DL);
15390 }
15391 
15392 /// Given an ISD::SDIV node expressing a divide by constant, return
15393 /// a DAG expression to select that will generate the same value by multiplying
15394 /// by a magic number.
15395 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
15396 SDValue DAGCombiner::BuildSDIV(SDNode *N) {
15397   // when optimising for minimum size, we don't want to expand a div to a mul
15398   // and a shift.
15399   if (DAG.getMachineFunction().getFunction()->optForMinSize())
15400     return SDValue();
15401 
15402   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
15403   if (!C)
15404     return SDValue();
15405 
15406   // Avoid division by zero.
15407   if (C->isNullValue())
15408     return SDValue();
15409 
15410   std::vector<SDNode*> Built;
15411   SDValue S =
15412       TLI.BuildSDIV(N, C->getAPIntValue(), DAG, LegalOperations, &Built);
15413 
15414   for (SDNode *N : Built)
15415     AddToWorklist(N);
15416   return S;
15417 }
15418 
15419 /// Given an ISD::SDIV node expressing a divide by constant power of 2, return a
15420 /// DAG expression that will generate the same value by right shifting.
15421 SDValue DAGCombiner::BuildSDIVPow2(SDNode *N) {
15422   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
15423   if (!C)
15424     return SDValue();
15425 
15426   // Avoid division by zero.
15427   if (C->isNullValue())
15428     return SDValue();
15429 
15430   std::vector<SDNode *> Built;
15431   SDValue S = TLI.BuildSDIVPow2(N, C->getAPIntValue(), DAG, &Built);
15432 
15433   for (SDNode *N : Built)
15434     AddToWorklist(N);
15435   return S;
15436 }
15437 
15438 /// Given an ISD::UDIV node expressing a divide by constant, return a DAG
15439 /// expression that will generate the same value by multiplying by a magic
15440 /// number.
15441 /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide".
15442 SDValue DAGCombiner::BuildUDIV(SDNode *N) {
15443   // when optimising for minimum size, we don't want to expand a div to a mul
15444   // and a shift.
15445   if (DAG.getMachineFunction().getFunction()->optForMinSize())
15446     return SDValue();
15447 
15448   ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1));
15449   if (!C)
15450     return SDValue();
15451 
15452   // Avoid division by zero.
15453   if (C->isNullValue())
15454     return SDValue();
15455 
15456   std::vector<SDNode*> Built;
15457   SDValue S =
15458       TLI.BuildUDIV(N, C->getAPIntValue(), DAG, LegalOperations, &Built);
15459 
15460   for (SDNode *N : Built)
15461     AddToWorklist(N);
15462   return S;
15463 }
15464 
15465 /// Determines the LogBase2 value for a non-null input value using the
15466 /// transform: LogBase2(V) = (EltBits - 1) - ctlz(V).
15467 SDValue DAGCombiner::BuildLogBase2(SDValue V, const SDLoc &DL) {
15468   EVT VT = V.getValueType();
15469   unsigned EltBits = VT.getScalarSizeInBits();
15470   SDValue Ctlz = DAG.getNode(ISD::CTLZ, DL, VT, V);
15471   SDValue Base = DAG.getConstant(EltBits - 1, DL, VT);
15472   SDValue LogBase2 = DAG.getNode(ISD::SUB, DL, VT, Base, Ctlz);
15473   return LogBase2;
15474 }
15475 
15476 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
15477 /// For the reciprocal, we need to find the zero of the function:
15478 ///   F(X) = A X - 1 [which has a zero at X = 1/A]
15479 ///     =>
15480 ///   X_{i+1} = X_i (2 - A X_i) = X_i + X_i (1 - A X_i) [this second form
15481 ///     does not require additional intermediate precision]
15482 SDValue DAGCombiner::BuildReciprocalEstimate(SDValue Op, SDNodeFlags *Flags) {
15483   if (Level >= AfterLegalizeDAG)
15484     return SDValue();
15485 
15486   // TODO: Handle half and/or extended types?
15487   EVT VT = Op.getValueType();
15488   if (VT.getScalarType() != MVT::f32 && VT.getScalarType() != MVT::f64)
15489     return SDValue();
15490 
15491   // If estimates are explicitly disabled for this function, we're done.
15492   MachineFunction &MF = DAG.getMachineFunction();
15493   int Enabled = TLI.getRecipEstimateDivEnabled(VT, MF);
15494   if (Enabled == TLI.ReciprocalEstimate::Disabled)
15495     return SDValue();
15496 
15497   // Estimates may be explicitly enabled for this type with a custom number of
15498   // refinement steps.
15499   int Iterations = TLI.getDivRefinementSteps(VT, MF);
15500   if (SDValue Est = TLI.getRecipEstimate(Op, DAG, Enabled, Iterations)) {
15501     AddToWorklist(Est.getNode());
15502 
15503     if (Iterations) {
15504       EVT VT = Op.getValueType();
15505       SDLoc DL(Op);
15506       SDValue FPOne = DAG.getConstantFP(1.0, DL, VT);
15507 
15508       // Newton iterations: Est = Est + Est (1 - Arg * Est)
15509       for (int i = 0; i < Iterations; ++i) {
15510         SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Op, Est, Flags);
15511         AddToWorklist(NewEst.getNode());
15512 
15513         NewEst = DAG.getNode(ISD::FSUB, DL, VT, FPOne, NewEst, Flags);
15514         AddToWorklist(NewEst.getNode());
15515 
15516         NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
15517         AddToWorklist(NewEst.getNode());
15518 
15519         Est = DAG.getNode(ISD::FADD, DL, VT, Est, NewEst, Flags);
15520         AddToWorklist(Est.getNode());
15521       }
15522     }
15523     return Est;
15524   }
15525 
15526   return SDValue();
15527 }
15528 
15529 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
15530 /// For the reciprocal sqrt, we need to find the zero of the function:
15531 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
15532 ///     =>
15533 ///   X_{i+1} = X_i (1.5 - A X_i^2 / 2)
15534 /// As a result, we precompute A/2 prior to the iteration loop.
15535 SDValue DAGCombiner::buildSqrtNROneConst(SDValue Arg, SDValue Est,
15536                                          unsigned Iterations,
15537                                          SDNodeFlags *Flags, bool Reciprocal) {
15538   EVT VT = Arg.getValueType();
15539   SDLoc DL(Arg);
15540   SDValue ThreeHalves = DAG.getConstantFP(1.5, DL, VT);
15541 
15542   // We now need 0.5 * Arg which we can write as (1.5 * Arg - Arg) so that
15543   // this entire sequence requires only one FP constant.
15544   SDValue HalfArg = DAG.getNode(ISD::FMUL, DL, VT, ThreeHalves, Arg, Flags);
15545   AddToWorklist(HalfArg.getNode());
15546 
15547   HalfArg = DAG.getNode(ISD::FSUB, DL, VT, HalfArg, Arg, Flags);
15548   AddToWorklist(HalfArg.getNode());
15549 
15550   // Newton iterations: Est = Est * (1.5 - HalfArg * Est * Est)
15551   for (unsigned i = 0; i < Iterations; ++i) {
15552     SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, Est, Flags);
15553     AddToWorklist(NewEst.getNode());
15554 
15555     NewEst = DAG.getNode(ISD::FMUL, DL, VT, HalfArg, NewEst, Flags);
15556     AddToWorklist(NewEst.getNode());
15557 
15558     NewEst = DAG.getNode(ISD::FSUB, DL, VT, ThreeHalves, NewEst, Flags);
15559     AddToWorklist(NewEst.getNode());
15560 
15561     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags);
15562     AddToWorklist(Est.getNode());
15563   }
15564 
15565   // If non-reciprocal square root is requested, multiply the result by Arg.
15566   if (!Reciprocal) {
15567     Est = DAG.getNode(ISD::FMUL, DL, VT, Est, Arg, Flags);
15568     AddToWorklist(Est.getNode());
15569   }
15570 
15571   return Est;
15572 }
15573 
15574 /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i)
15575 /// For the reciprocal sqrt, we need to find the zero of the function:
15576 ///   F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)]
15577 ///     =>
15578 ///   X_{i+1} = (-0.5 * X_i) * (A * X_i * X_i + (-3.0))
15579 SDValue DAGCombiner::buildSqrtNRTwoConst(SDValue Arg, SDValue Est,
15580                                          unsigned Iterations,
15581                                          SDNodeFlags *Flags, bool Reciprocal) {
15582   EVT VT = Arg.getValueType();
15583   SDLoc DL(Arg);
15584   SDValue MinusThree = DAG.getConstantFP(-3.0, DL, VT);
15585   SDValue MinusHalf = DAG.getConstantFP(-0.5, DL, VT);
15586 
15587   // This routine must enter the loop below to work correctly
15588   // when (Reciprocal == false).
15589   assert(Iterations > 0);
15590 
15591   // Newton iterations for reciprocal square root:
15592   // E = (E * -0.5) * ((A * E) * E + -3.0)
15593   for (unsigned i = 0; i < Iterations; ++i) {
15594     SDValue AE = DAG.getNode(ISD::FMUL, DL, VT, Arg, Est, Flags);
15595     AddToWorklist(AE.getNode());
15596 
15597     SDValue AEE = DAG.getNode(ISD::FMUL, DL, VT, AE, Est, Flags);
15598     AddToWorklist(AEE.getNode());
15599 
15600     SDValue RHS = DAG.getNode(ISD::FADD, DL, VT, AEE, MinusThree, Flags);
15601     AddToWorklist(RHS.getNode());
15602 
15603     // When calculating a square root at the last iteration build:
15604     // S = ((A * E) * -0.5) * ((A * E) * E + -3.0)
15605     // (notice a common subexpression)
15606     SDValue LHS;
15607     if (Reciprocal || (i + 1) < Iterations) {
15608       // RSQRT: LHS = (E * -0.5)
15609       LHS = DAG.getNode(ISD::FMUL, DL, VT, Est, MinusHalf, Flags);
15610     } else {
15611       // SQRT: LHS = (A * E) * -0.5
15612       LHS = DAG.getNode(ISD::FMUL, DL, VT, AE, MinusHalf, Flags);
15613     }
15614     AddToWorklist(LHS.getNode());
15615 
15616     Est = DAG.getNode(ISD::FMUL, DL, VT, LHS, RHS, Flags);
15617     AddToWorklist(Est.getNode());
15618   }
15619 
15620   return Est;
15621 }
15622 
15623 /// Build code to calculate either rsqrt(Op) or sqrt(Op). In the latter case
15624 /// Op*rsqrt(Op) is actually computed, so additional postprocessing is needed if
15625 /// Op can be zero.
15626 SDValue DAGCombiner::buildSqrtEstimateImpl(SDValue Op, SDNodeFlags *Flags,
15627                                            bool Reciprocal) {
15628   if (Level >= AfterLegalizeDAG)
15629     return SDValue();
15630 
15631   // TODO: Handle half and/or extended types?
15632   EVT VT = Op.getValueType();
15633   if (VT.getScalarType() != MVT::f32 && VT.getScalarType() != MVT::f64)
15634     return SDValue();
15635 
15636   // If estimates are explicitly disabled for this function, we're done.
15637   MachineFunction &MF = DAG.getMachineFunction();
15638   int Enabled = TLI.getRecipEstimateSqrtEnabled(VT, MF);
15639   if (Enabled == TLI.ReciprocalEstimate::Disabled)
15640     return SDValue();
15641 
15642   // Estimates may be explicitly enabled for this type with a custom number of
15643   // refinement steps.
15644   int Iterations = TLI.getSqrtRefinementSteps(VT, MF);
15645 
15646   bool UseOneConstNR = false;
15647   if (SDValue Est =
15648       TLI.getSqrtEstimate(Op, DAG, Enabled, Iterations, UseOneConstNR,
15649                           Reciprocal)) {
15650     AddToWorklist(Est.getNode());
15651 
15652     if (Iterations) {
15653       Est = UseOneConstNR
15654             ? buildSqrtNROneConst(Op, Est, Iterations, Flags, Reciprocal)
15655             : buildSqrtNRTwoConst(Op, Est, Iterations, Flags, Reciprocal);
15656 
15657       if (!Reciprocal) {
15658         // Unfortunately, Est is now NaN if the input was exactly 0.0.
15659         // Select out this case and force the answer to 0.0.
15660         EVT VT = Op.getValueType();
15661         SDLoc DL(Op);
15662 
15663         SDValue FPZero = DAG.getConstantFP(0.0, DL, VT);
15664         EVT CCVT = getSetCCResultType(VT);
15665         SDValue ZeroCmp = DAG.getSetCC(DL, CCVT, Op, FPZero, ISD::SETEQ);
15666         AddToWorklist(ZeroCmp.getNode());
15667 
15668         Est = DAG.getNode(VT.isVector() ? ISD::VSELECT : ISD::SELECT, DL, VT,
15669                           ZeroCmp, FPZero, Est);
15670         AddToWorklist(Est.getNode());
15671       }
15672     }
15673     return Est;
15674   }
15675 
15676   return SDValue();
15677 }
15678 
15679 SDValue DAGCombiner::buildRsqrtEstimate(SDValue Op, SDNodeFlags *Flags) {
15680   return buildSqrtEstimateImpl(Op, Flags, true);
15681 }
15682 
15683 SDValue DAGCombiner::buildSqrtEstimate(SDValue Op, SDNodeFlags *Flags) {
15684   return buildSqrtEstimateImpl(Op, Flags, false);
15685 }
15686 
15687 /// Return true if base is a frame index, which is known not to alias with
15688 /// anything but itself.  Provides base object and offset as results.
15689 static bool FindBaseOffset(SDValue Ptr, SDValue &Base, int64_t &Offset,
15690                            const GlobalValue *&GV, const void *&CV) {
15691   // Assume it is a primitive operation.
15692   Base = Ptr; Offset = 0; GV = nullptr; CV = nullptr;
15693 
15694   // If it's an adding a simple constant then integrate the offset.
15695   if (Base.getOpcode() == ISD::ADD) {
15696     if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Base.getOperand(1))) {
15697       Base = Base.getOperand(0);
15698       Offset += C->getZExtValue();
15699     }
15700   }
15701 
15702   // Return the underlying GlobalValue, and update the Offset.  Return false
15703   // for GlobalAddressSDNode since the same GlobalAddress may be represented
15704   // by multiple nodes with different offsets.
15705   if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Base)) {
15706     GV = G->getGlobal();
15707     Offset += G->getOffset();
15708     return false;
15709   }
15710 
15711   // Return the underlying Constant value, and update the Offset.  Return false
15712   // for ConstantSDNodes since the same constant pool entry may be represented
15713   // by multiple nodes with different offsets.
15714   if (ConstantPoolSDNode *C = dyn_cast<ConstantPoolSDNode>(Base)) {
15715     CV = C->isMachineConstantPoolEntry() ? (const void *)C->getMachineCPVal()
15716                                          : (const void *)C->getConstVal();
15717     Offset += C->getOffset();
15718     return false;
15719   }
15720   // If it's any of the following then it can't alias with anything but itself.
15721   return isa<FrameIndexSDNode>(Base);
15722 }
15723 
15724 /// Return true if there is any possibility that the two addresses overlap.
15725 bool DAGCombiner::isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const {
15726   // If they are the same then they must be aliases.
15727   if (Op0->getBasePtr() == Op1->getBasePtr()) return true;
15728 
15729   // If they are both volatile then they cannot be reordered.
15730   if (Op0->isVolatile() && Op1->isVolatile()) return true;
15731 
15732   // If one operation reads from invariant memory, and the other may store, they
15733   // cannot alias. These should really be checking the equivalent of mayWrite,
15734   // but it only matters for memory nodes other than load /store.
15735   if (Op0->isInvariant() && Op1->writeMem())
15736     return false;
15737 
15738   if (Op1->isInvariant() && Op0->writeMem())
15739     return false;
15740 
15741   // Gather base node and offset information.
15742   SDValue Base1, Base2;
15743   int64_t Offset1, Offset2;
15744   const GlobalValue *GV1, *GV2;
15745   const void *CV1, *CV2;
15746   bool isFrameIndex1 = FindBaseOffset(Op0->getBasePtr(),
15747                                       Base1, Offset1, GV1, CV1);
15748   bool isFrameIndex2 = FindBaseOffset(Op1->getBasePtr(),
15749                                       Base2, Offset2, GV2, CV2);
15750 
15751   // If they have a same base address then check to see if they overlap.
15752   if (Base1 == Base2 || (GV1 && (GV1 == GV2)) || (CV1 && (CV1 == CV2)))
15753     return !((Offset1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= Offset2 ||
15754              (Offset2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= Offset1);
15755 
15756   // It is possible for different frame indices to alias each other, mostly
15757   // when tail call optimization reuses return address slots for arguments.
15758   // To catch this case, look up the actual index of frame indices to compute
15759   // the real alias relationship.
15760   if (isFrameIndex1 && isFrameIndex2) {
15761     MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
15762     Offset1 += MFI.getObjectOffset(cast<FrameIndexSDNode>(Base1)->getIndex());
15763     Offset2 += MFI.getObjectOffset(cast<FrameIndexSDNode>(Base2)->getIndex());
15764     return !((Offset1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= Offset2 ||
15765              (Offset2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= Offset1);
15766   }
15767 
15768   // Otherwise, if we know what the bases are, and they aren't identical, then
15769   // we know they cannot alias.
15770   if ((isFrameIndex1 || CV1 || GV1) && (isFrameIndex2 || CV2 || GV2))
15771     return false;
15772 
15773   // If we know required SrcValue1 and SrcValue2 have relatively large alignment
15774   // compared to the size and offset of the access, we may be able to prove they
15775   // do not alias.  This check is conservative for now to catch cases created by
15776   // splitting vector types.
15777   if ((Op0->getOriginalAlignment() == Op1->getOriginalAlignment()) &&
15778       (Op0->getSrcValueOffset() != Op1->getSrcValueOffset()) &&
15779       (Op0->getMemoryVT().getSizeInBits() >> 3 ==
15780        Op1->getMemoryVT().getSizeInBits() >> 3) &&
15781       (Op0->getOriginalAlignment() > (Op0->getMemoryVT().getSizeInBits() >> 3))) {
15782     int64_t OffAlign1 = Op0->getSrcValueOffset() % Op0->getOriginalAlignment();
15783     int64_t OffAlign2 = Op1->getSrcValueOffset() % Op1->getOriginalAlignment();
15784 
15785     // There is no overlap between these relatively aligned accesses of similar
15786     // size, return no alias.
15787     if ((OffAlign1 + (Op0->getMemoryVT().getSizeInBits() >> 3)) <= OffAlign2 ||
15788         (OffAlign2 + (Op1->getMemoryVT().getSizeInBits() >> 3)) <= OffAlign1)
15789       return false;
15790   }
15791 
15792   bool UseAA = CombinerGlobalAA.getNumOccurrences() > 0
15793                    ? CombinerGlobalAA
15794                    : DAG.getSubtarget().useAA();
15795 #ifndef NDEBUG
15796   if (CombinerAAOnlyFunc.getNumOccurrences() &&
15797       CombinerAAOnlyFunc != DAG.getMachineFunction().getName())
15798     UseAA = false;
15799 #endif
15800   if (UseAA &&
15801       Op0->getMemOperand()->getValue() && Op1->getMemOperand()->getValue()) {
15802     // Use alias analysis information.
15803     int64_t MinOffset = std::min(Op0->getSrcValueOffset(),
15804                                  Op1->getSrcValueOffset());
15805     int64_t Overlap1 = (Op0->getMemoryVT().getSizeInBits() >> 3) +
15806         Op0->getSrcValueOffset() - MinOffset;
15807     int64_t Overlap2 = (Op1->getMemoryVT().getSizeInBits() >> 3) +
15808         Op1->getSrcValueOffset() - MinOffset;
15809     AliasResult AAResult =
15810         AA.alias(MemoryLocation(Op0->getMemOperand()->getValue(), Overlap1,
15811                                 UseTBAA ? Op0->getAAInfo() : AAMDNodes()),
15812                  MemoryLocation(Op1->getMemOperand()->getValue(), Overlap2,
15813                                 UseTBAA ? Op1->getAAInfo() : AAMDNodes()));
15814     if (AAResult == NoAlias)
15815       return false;
15816   }
15817 
15818   // Otherwise we have to assume they alias.
15819   return true;
15820 }
15821 
15822 /// Walk up chain skipping non-aliasing memory nodes,
15823 /// looking for aliasing nodes and adding them to the Aliases vector.
15824 void DAGCombiner::GatherAllAliases(SDNode *N, SDValue OriginalChain,
15825                                    SmallVectorImpl<SDValue> &Aliases) {
15826   SmallVector<SDValue, 8> Chains;     // List of chains to visit.
15827   SmallPtrSet<SDNode *, 16> Visited;  // Visited node set.
15828 
15829   // Get alias information for node.
15830   bool IsLoad = isa<LoadSDNode>(N) && !cast<LSBaseSDNode>(N)->isVolatile();
15831 
15832   // Starting off.
15833   Chains.push_back(OriginalChain);
15834   unsigned Depth = 0;
15835 
15836   // Look at each chain and determine if it is an alias.  If so, add it to the
15837   // aliases list.  If not, then continue up the chain looking for the next
15838   // candidate.
15839   while (!Chains.empty()) {
15840     SDValue Chain = Chains.pop_back_val();
15841 
15842     // For TokenFactor nodes, look at each operand and only continue up the
15843     // chain until we reach the depth limit.
15844     //
15845     // FIXME: The depth check could be made to return the last non-aliasing
15846     // chain we found before we hit a tokenfactor rather than the original
15847     // chain.
15848     if (Depth > TLI.getGatherAllAliasesMaxDepth()) {
15849       Aliases.clear();
15850       Aliases.push_back(OriginalChain);
15851       return;
15852     }
15853 
15854     // Don't bother if we've been before.
15855     if (!Visited.insert(Chain.getNode()).second)
15856       continue;
15857 
15858     switch (Chain.getOpcode()) {
15859     case ISD::EntryToken:
15860       // Entry token is ideal chain operand, but handled in FindBetterChain.
15861       break;
15862 
15863     case ISD::LOAD:
15864     case ISD::STORE: {
15865       // Get alias information for Chain.
15866       bool IsOpLoad = isa<LoadSDNode>(Chain.getNode()) &&
15867           !cast<LSBaseSDNode>(Chain.getNode())->isVolatile();
15868 
15869       // If chain is alias then stop here.
15870       if (!(IsLoad && IsOpLoad) &&
15871           isAlias(cast<LSBaseSDNode>(N), cast<LSBaseSDNode>(Chain.getNode()))) {
15872         Aliases.push_back(Chain);
15873       } else {
15874         // Look further up the chain.
15875         Chains.push_back(Chain.getOperand(0));
15876         ++Depth;
15877       }
15878       break;
15879     }
15880 
15881     case ISD::TokenFactor:
15882       // We have to check each of the operands of the token factor for "small"
15883       // token factors, so we queue them up.  Adding the operands to the queue
15884       // (stack) in reverse order maintains the original order and increases the
15885       // likelihood that getNode will find a matching token factor (CSE.)
15886       if (Chain.getNumOperands() > 16) {
15887         Aliases.push_back(Chain);
15888         break;
15889       }
15890       for (unsigned n = Chain.getNumOperands(); n;)
15891         Chains.push_back(Chain.getOperand(--n));
15892       ++Depth;
15893       break;
15894 
15895     default:
15896       // For all other instructions we will just have to take what we can get.
15897       Aliases.push_back(Chain);
15898       break;
15899     }
15900   }
15901 }
15902 
15903 /// Walk up chain skipping non-aliasing memory nodes, looking for a better chain
15904 /// (aliasing node.)
15905 SDValue DAGCombiner::FindBetterChain(SDNode *N, SDValue OldChain) {
15906   SmallVector<SDValue, 8> Aliases;  // Ops for replacing token factor.
15907 
15908   // Accumulate all the aliases to this node.
15909   GatherAllAliases(N, OldChain, Aliases);
15910 
15911   // If no operands then chain to entry token.
15912   if (Aliases.size() == 0)
15913     return DAG.getEntryNode();
15914 
15915   // If a single operand then chain to it.  We don't need to revisit it.
15916   if (Aliases.size() == 1)
15917     return Aliases[0];
15918 
15919   // Construct a custom tailored token factor.
15920   return DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Aliases);
15921 }
15922 
15923 bool DAGCombiner::findBetterNeighborChains(StoreSDNode *St) {
15924   // This holds the base pointer, index, and the offset in bytes from the base
15925   // pointer.
15926   BaseIndexOffset BasePtr = BaseIndexOffset::match(St->getBasePtr(), DAG);
15927 
15928   // We must have a base and an offset.
15929   if (!BasePtr.Base.getNode())
15930     return false;
15931 
15932   // Do not handle stores to undef base pointers.
15933   if (BasePtr.Base.isUndef())
15934     return false;
15935 
15936   SmallVector<StoreSDNode *, 8> ChainedStores;
15937   ChainedStores.push_back(St);
15938 
15939   // Walk up the chain and look for nodes with offsets from the same
15940   // base pointer. Stop when reaching an instruction with a different kind
15941   // or instruction which has a different base pointer.
15942   StoreSDNode *Index = St;
15943   while (Index) {
15944     // If the chain has more than one use, then we can't reorder the mem ops.
15945     if (Index != St && !SDValue(Index, 0)->hasOneUse())
15946       break;
15947 
15948     if (Index->isVolatile() || Index->isIndexed())
15949       break;
15950 
15951     // Find the base pointer and offset for this memory node.
15952     BaseIndexOffset Ptr = BaseIndexOffset::match(Index->getBasePtr(), DAG);
15953 
15954     // Check that the base pointer is the same as the original one.
15955     if (!Ptr.equalBaseIndex(BasePtr))
15956       break;
15957 
15958     // Find the next memory operand in the chain. If the next operand in the
15959     // chain is a store then move up and continue the scan with the next
15960     // memory operand. If the next operand is a load save it and use alias
15961     // information to check if it interferes with anything.
15962     SDNode *NextInChain = Index->getChain().getNode();
15963     while (true) {
15964       if (StoreSDNode *STn = dyn_cast<StoreSDNode>(NextInChain)) {
15965         // We found a store node. Use it for the next iteration.
15966         if (STn->isVolatile() || STn->isIndexed()) {
15967           Index = nullptr;
15968           break;
15969         }
15970         ChainedStores.push_back(STn);
15971         Index = STn;
15972         break;
15973       } else if (LoadSDNode *Ldn = dyn_cast<LoadSDNode>(NextInChain)) {
15974         NextInChain = Ldn->getChain().getNode();
15975         continue;
15976       } else {
15977         Index = nullptr;
15978         break;
15979       }
15980     }
15981   }
15982 
15983   bool MadeChangeToSt = false;
15984   SmallVector<std::pair<StoreSDNode *, SDValue>, 8> BetterChains;
15985 
15986   for (StoreSDNode *ChainedStore : ChainedStores) {
15987     SDValue Chain = ChainedStore->getChain();
15988     SDValue BetterChain = FindBetterChain(ChainedStore, Chain);
15989 
15990     if (Chain != BetterChain) {
15991       if (ChainedStore == St)
15992         MadeChangeToSt = true;
15993       BetterChains.push_back(std::make_pair(ChainedStore, BetterChain));
15994     }
15995   }
15996 
15997   // Do all replacements after finding the replacements to make to avoid making
15998   // the chains more complicated by introducing new TokenFactors.
15999   for (auto Replacement : BetterChains)
16000     replaceStoreChain(Replacement.first, Replacement.second);
16001 
16002   return MadeChangeToSt;
16003 }
16004 
16005 /// This is the entry point for the file.
16006 void SelectionDAG::Combine(CombineLevel Level, AliasAnalysis &AA,
16007                            CodeGenOpt::Level OptLevel) {
16008   /// This is the main entry point to this class.
16009   DAGCombiner(*this, AA, OptLevel).Run(Level);
16010 }
16011