1 //===-- VEISelLowering.cpp - VE DAG Lowering Implementation ---------------===//
2 //
3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4 // See https://llvm.org/LICENSE.txt for license information.
5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6 //
7 //===----------------------------------------------------------------------===//
8 //
9 // This file implements the interfaces that VE uses to lower LLVM code into a
10 // selection DAG.
11 //
12 //===----------------------------------------------------------------------===//
13 
14 #include "VEISelLowering.h"
15 #include "MCTargetDesc/VEMCExpr.h"
16 #include "VECustomDAG.h"
17 #include "VEInstrBuilder.h"
18 #include "VEMachineFunctionInfo.h"
19 #include "VERegisterInfo.h"
20 #include "VETargetMachine.h"
21 #include "llvm/ADT/StringSwitch.h"
22 #include "llvm/CodeGen/CallingConvLower.h"
23 #include "llvm/CodeGen/MachineFrameInfo.h"
24 #include "llvm/CodeGen/MachineFunction.h"
25 #include "llvm/CodeGen/MachineInstrBuilder.h"
26 #include "llvm/CodeGen/MachineJumpTableInfo.h"
27 #include "llvm/CodeGen/MachineModuleInfo.h"
28 #include "llvm/CodeGen/MachineRegisterInfo.h"
29 #include "llvm/CodeGen/SelectionDAG.h"
30 #include "llvm/CodeGen/TargetLoweringObjectFileImpl.h"
31 #include "llvm/IR/DerivedTypes.h"
32 #include "llvm/IR/Function.h"
33 #include "llvm/IR/IRBuilder.h"
34 #include "llvm/IR/Module.h"
35 #include "llvm/Support/ErrorHandling.h"
36 #include "llvm/Support/KnownBits.h"
37 using namespace llvm;
38 
39 #define DEBUG_TYPE "ve-lower"
40 
41 //===----------------------------------------------------------------------===//
42 // Calling Convention Implementation
43 //===----------------------------------------------------------------------===//
44 
45 #include "VEGenCallingConv.inc"
46 
47 CCAssignFn *getReturnCC(CallingConv::ID CallConv) {
48   switch (CallConv) {
49   default:
50     return RetCC_VE_C;
51   case CallingConv::Fast:
52     return RetCC_VE_Fast;
53   }
54 }
55 
56 CCAssignFn *getParamCC(CallingConv::ID CallConv, bool IsVarArg) {
57   if (IsVarArg)
58     return CC_VE2;
59   switch (CallConv) {
60   default:
61     return CC_VE_C;
62   case CallingConv::Fast:
63     return CC_VE_Fast;
64   }
65 }
66 
67 bool VETargetLowering::CanLowerReturn(
68     CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg,
69     const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context) const {
70   CCAssignFn *RetCC = getReturnCC(CallConv);
71   SmallVector<CCValAssign, 16> RVLocs;
72   CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context);
73   return CCInfo.CheckReturn(Outs, RetCC);
74 }
75 
76 static const MVT AllVectorVTs[] = {MVT::v256i32, MVT::v512i32, MVT::v256i64,
77                                    MVT::v256f32, MVT::v512f32, MVT::v256f64};
78 
79 static const MVT AllMaskVTs[] = {MVT::v256i1, MVT::v512i1};
80 
81 static const MVT AllPackedVTs[] = {MVT::v512i32, MVT::v512f32};
82 
83 void VETargetLowering::initRegisterClasses() {
84   // Set up the register classes.
85   addRegisterClass(MVT::i32, &VE::I32RegClass);
86   addRegisterClass(MVT::i64, &VE::I64RegClass);
87   addRegisterClass(MVT::f32, &VE::F32RegClass);
88   addRegisterClass(MVT::f64, &VE::I64RegClass);
89   addRegisterClass(MVT::f128, &VE::F128RegClass);
90 
91   if (Subtarget->enableVPU()) {
92     for (MVT VecVT : AllVectorVTs)
93       addRegisterClass(VecVT, &VE::V64RegClass);
94     addRegisterClass(MVT::v256i1, &VE::VMRegClass);
95     addRegisterClass(MVT::v512i1, &VE::VM512RegClass);
96   }
97 }
98 
99 void VETargetLowering::initSPUActions() {
100   const auto &TM = getTargetMachine();
101   /// Load & Store {
102 
103   // VE doesn't have i1 sign extending load.
104   for (MVT VT : MVT::integer_valuetypes()) {
105     setLoadExtAction(ISD::SEXTLOAD, VT, MVT::i1, Promote);
106     setLoadExtAction(ISD::ZEXTLOAD, VT, MVT::i1, Promote);
107     setLoadExtAction(ISD::EXTLOAD, VT, MVT::i1, Promote);
108     setTruncStoreAction(VT, MVT::i1, Expand);
109   }
110 
111   // VE doesn't have floating point extload/truncstore, so expand them.
112   for (MVT FPVT : MVT::fp_valuetypes()) {
113     for (MVT OtherFPVT : MVT::fp_valuetypes()) {
114       setLoadExtAction(ISD::EXTLOAD, FPVT, OtherFPVT, Expand);
115       setTruncStoreAction(FPVT, OtherFPVT, Expand);
116     }
117   }
118 
119   // VE doesn't have fp128 load/store, so expand them in custom lower.
120   setOperationAction(ISD::LOAD, MVT::f128, Custom);
121   setOperationAction(ISD::STORE, MVT::f128, Custom);
122 
123   /// } Load & Store
124 
125   // Custom legalize address nodes into LO/HI parts.
126   MVT PtrVT = MVT::getIntegerVT(TM.getPointerSizeInBits(0));
127   setOperationAction(ISD::BlockAddress, PtrVT, Custom);
128   setOperationAction(ISD::GlobalAddress, PtrVT, Custom);
129   setOperationAction(ISD::GlobalTLSAddress, PtrVT, Custom);
130   setOperationAction(ISD::ConstantPool, PtrVT, Custom);
131   setOperationAction(ISD::JumpTable, PtrVT, Custom);
132 
133   /// VAARG handling {
134   setOperationAction(ISD::VASTART, MVT::Other, Custom);
135   // VAARG needs to be lowered to access with 8 bytes alignment.
136   setOperationAction(ISD::VAARG, MVT::Other, Custom);
137   // Use the default implementation.
138   setOperationAction(ISD::VACOPY, MVT::Other, Expand);
139   setOperationAction(ISD::VAEND, MVT::Other, Expand);
140   /// } VAARG handling
141 
142   /// Stack {
143   setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i32, Custom);
144   setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i64, Custom);
145 
146   // Use the default implementation.
147   setOperationAction(ISD::STACKSAVE, MVT::Other, Expand);
148   setOperationAction(ISD::STACKRESTORE, MVT::Other, Expand);
149   /// } Stack
150 
151   /// Branch {
152 
153   // VE doesn't have BRCOND
154   setOperationAction(ISD::BRCOND, MVT::Other, Expand);
155 
156   // BR_JT is not implemented yet.
157   setOperationAction(ISD::BR_JT, MVT::Other, Expand);
158 
159   /// } Branch
160 
161   /// Int Ops {
162   for (MVT IntVT : {MVT::i32, MVT::i64}) {
163     // VE has no REM or DIVREM operations.
164     setOperationAction(ISD::UREM, IntVT, Expand);
165     setOperationAction(ISD::SREM, IntVT, Expand);
166     setOperationAction(ISD::SDIVREM, IntVT, Expand);
167     setOperationAction(ISD::UDIVREM, IntVT, Expand);
168 
169     // VE has no SHL_PARTS/SRA_PARTS/SRL_PARTS operations.
170     setOperationAction(ISD::SHL_PARTS, IntVT, Expand);
171     setOperationAction(ISD::SRA_PARTS, IntVT, Expand);
172     setOperationAction(ISD::SRL_PARTS, IntVT, Expand);
173 
174     // VE has no MULHU/S or U/SMUL_LOHI operations.
175     // TODO: Use MPD instruction to implement SMUL_LOHI for i32 type.
176     setOperationAction(ISD::MULHU, IntVT, Expand);
177     setOperationAction(ISD::MULHS, IntVT, Expand);
178     setOperationAction(ISD::UMUL_LOHI, IntVT, Expand);
179     setOperationAction(ISD::SMUL_LOHI, IntVT, Expand);
180 
181     // VE has no CTTZ, ROTL, ROTR operations.
182     setOperationAction(ISD::CTTZ, IntVT, Expand);
183     setOperationAction(ISD::ROTL, IntVT, Expand);
184     setOperationAction(ISD::ROTR, IntVT, Expand);
185 
186     // VE has 64 bits instruction which works as i64 BSWAP operation.  This
187     // instruction works fine as i32 BSWAP operation with an additional
188     // parameter.  Use isel patterns to lower BSWAP.
189     setOperationAction(ISD::BSWAP, IntVT, Legal);
190 
191     // VE has only 64 bits instructions which work as i64 BITREVERSE/CTLZ/CTPOP
192     // operations.  Use isel patterns for i64, promote for i32.
193     LegalizeAction Act = (IntVT == MVT::i32) ? Promote : Legal;
194     setOperationAction(ISD::BITREVERSE, IntVT, Act);
195     setOperationAction(ISD::CTLZ, IntVT, Act);
196     setOperationAction(ISD::CTLZ_ZERO_UNDEF, IntVT, Act);
197     setOperationAction(ISD::CTPOP, IntVT, Act);
198 
199     // VE has only 64 bits instructions which work as i64 AND/OR/XOR operations.
200     // Use isel patterns for i64, promote for i32.
201     setOperationAction(ISD::AND, IntVT, Act);
202     setOperationAction(ISD::OR, IntVT, Act);
203     setOperationAction(ISD::XOR, IntVT, Act);
204   }
205   /// } Int Ops
206 
207   /// Conversion {
208   // VE doesn't have instructions for fp<->uint, so expand them by llvm
209   setOperationAction(ISD::FP_TO_UINT, MVT::i32, Promote); // use i64
210   setOperationAction(ISD::UINT_TO_FP, MVT::i32, Promote); // use i64
211   setOperationAction(ISD::FP_TO_UINT, MVT::i64, Expand);
212   setOperationAction(ISD::UINT_TO_FP, MVT::i64, Expand);
213 
214   // fp16 not supported
215   for (MVT FPVT : MVT::fp_valuetypes()) {
216     setOperationAction(ISD::FP16_TO_FP, FPVT, Expand);
217     setOperationAction(ISD::FP_TO_FP16, FPVT, Expand);
218   }
219   /// } Conversion
220 
221   /// Floating-point Ops {
222   /// Note: Floating-point operations are fneg, fadd, fsub, fmul, fdiv, frem,
223   ///       and fcmp.
224 
225   // VE doesn't have following floating point operations.
226   for (MVT VT : MVT::fp_valuetypes()) {
227     setOperationAction(ISD::FNEG, VT, Expand);
228     setOperationAction(ISD::FREM, VT, Expand);
229   }
230 
231   // VE doesn't have fdiv of f128.
232   setOperationAction(ISD::FDIV, MVT::f128, Expand);
233 
234   for (MVT FPVT : {MVT::f32, MVT::f64}) {
235     // f32 and f64 uses ConstantFP.  f128 uses ConstantPool.
236     setOperationAction(ISD::ConstantFP, FPVT, Legal);
237   }
238   /// } Floating-point Ops
239 
240   /// Floating-point math functions {
241 
242   // VE doesn't have following floating point math functions.
243   for (MVT VT : MVT::fp_valuetypes()) {
244     setOperationAction(ISD::FABS, VT, Expand);
245     setOperationAction(ISD::FCOPYSIGN, VT, Expand);
246     setOperationAction(ISD::FCOS, VT, Expand);
247     setOperationAction(ISD::FSIN, VT, Expand);
248     setOperationAction(ISD::FSQRT, VT, Expand);
249   }
250 
251   /// } Floating-point math functions
252 
253   /// Atomic instructions {
254 
255   setMaxAtomicSizeInBitsSupported(64);
256   setMinCmpXchgSizeInBits(32);
257   setSupportsUnalignedAtomics(false);
258 
259   // Use custom inserter for ATOMIC_FENCE.
260   setOperationAction(ISD::ATOMIC_FENCE, MVT::Other, Custom);
261 
262   // Other atomic instructions.
263   for (MVT VT : MVT::integer_valuetypes()) {
264     // Support i8/i16 atomic swap.
265     setOperationAction(ISD::ATOMIC_SWAP, VT, Custom);
266 
267     // FIXME: Support "atmam" instructions.
268     setOperationAction(ISD::ATOMIC_LOAD_ADD, VT, Expand);
269     setOperationAction(ISD::ATOMIC_LOAD_SUB, VT, Expand);
270     setOperationAction(ISD::ATOMIC_LOAD_AND, VT, Expand);
271     setOperationAction(ISD::ATOMIC_LOAD_OR, VT, Expand);
272 
273     // VE doesn't have follwing instructions.
274     setOperationAction(ISD::ATOMIC_CMP_SWAP_WITH_SUCCESS, VT, Expand);
275     setOperationAction(ISD::ATOMIC_LOAD_CLR, VT, Expand);
276     setOperationAction(ISD::ATOMIC_LOAD_XOR, VT, Expand);
277     setOperationAction(ISD::ATOMIC_LOAD_NAND, VT, Expand);
278     setOperationAction(ISD::ATOMIC_LOAD_MIN, VT, Expand);
279     setOperationAction(ISD::ATOMIC_LOAD_MAX, VT, Expand);
280     setOperationAction(ISD::ATOMIC_LOAD_UMIN, VT, Expand);
281     setOperationAction(ISD::ATOMIC_LOAD_UMAX, VT, Expand);
282   }
283 
284   /// } Atomic instructions
285 
286   /// SJLJ instructions {
287   setOperationAction(ISD::EH_SJLJ_LONGJMP, MVT::Other, Custom);
288   setOperationAction(ISD::EH_SJLJ_SETJMP, MVT::i32, Custom);
289   setOperationAction(ISD::EH_SJLJ_SETUP_DISPATCH, MVT::Other, Custom);
290   if (TM.Options.ExceptionModel == ExceptionHandling::SjLj)
291     setLibcallName(RTLIB::UNWIND_RESUME, "_Unwind_SjLj_Resume");
292   /// } SJLJ instructions
293 
294   // Intrinsic instructions
295   setOperationAction(ISD::INTRINSIC_WO_CHAIN, MVT::Other, Custom);
296 }
297 
298 void VETargetLowering::initVPUActions() {
299   for (MVT LegalMaskVT : AllMaskVTs)
300     setOperationAction(ISD::BUILD_VECTOR, LegalMaskVT, Custom);
301 
302   for (unsigned Opc : {ISD::AND, ISD::OR, ISD::XOR})
303     setOperationAction(Opc, MVT::v512i1, Custom);
304 
305   for (MVT LegalVecVT : AllVectorVTs) {
306     setOperationAction(ISD::BUILD_VECTOR, LegalVecVT, Custom);
307     setOperationAction(ISD::INSERT_VECTOR_ELT, LegalVecVT, Legal);
308     setOperationAction(ISD::EXTRACT_VECTOR_ELT, LegalVecVT, Legal);
309     // Translate all vector instructions with legal element types to VVP_*
310     // nodes.
311     // TODO We will custom-widen into VVP_* nodes in the future. While we are
312     // buildling the infrastructure for this, we only do this for legal vector
313     // VTs.
314 #define HANDLE_VP_TO_VVP(VP_OPC, VVP_NAME)                                     \
315   setOperationAction(ISD::VP_OPC, LegalVecVT, Custom);
316 #define ADD_VVP_OP(VVP_NAME, ISD_NAME)                                         \
317   setOperationAction(ISD::ISD_NAME, LegalVecVT, Custom);
318 #include "VVPNodes.def"
319   }
320 
321   for (MVT LegalPackedVT : AllPackedVTs) {
322     setOperationAction(ISD::INSERT_VECTOR_ELT, LegalPackedVT, Custom);
323     setOperationAction(ISD::EXTRACT_VECTOR_ELT, LegalPackedVT, Custom);
324   }
325 }
326 
327 SDValue
328 VETargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv,
329                               bool IsVarArg,
330                               const SmallVectorImpl<ISD::OutputArg> &Outs,
331                               const SmallVectorImpl<SDValue> &OutVals,
332                               const SDLoc &DL, SelectionDAG &DAG) const {
333   // CCValAssign - represent the assignment of the return value to locations.
334   SmallVector<CCValAssign, 16> RVLocs;
335 
336   // CCState - Info about the registers and stack slot.
337   CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), RVLocs,
338                  *DAG.getContext());
339 
340   // Analyze return values.
341   CCInfo.AnalyzeReturn(Outs, getReturnCC(CallConv));
342 
343   SDValue Flag;
344   SmallVector<SDValue, 4> RetOps(1, Chain);
345 
346   // Copy the result values into the output registers.
347   for (unsigned i = 0; i != RVLocs.size(); ++i) {
348     CCValAssign &VA = RVLocs[i];
349     assert(VA.isRegLoc() && "Can only return in registers!");
350     assert(!VA.needsCustom() && "Unexpected custom lowering");
351     SDValue OutVal = OutVals[i];
352 
353     // Integer return values must be sign or zero extended by the callee.
354     switch (VA.getLocInfo()) {
355     case CCValAssign::Full:
356       break;
357     case CCValAssign::SExt:
358       OutVal = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), OutVal);
359       break;
360     case CCValAssign::ZExt:
361       OutVal = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), OutVal);
362       break;
363     case CCValAssign::AExt:
364       OutVal = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), OutVal);
365       break;
366     case CCValAssign::BCvt: {
367       // Convert a float return value to i64 with padding.
368       //     63     31   0
369       //    +------+------+
370       //    | float|   0  |
371       //    +------+------+
372       assert(VA.getLocVT() == MVT::i64);
373       assert(VA.getValVT() == MVT::f32);
374       SDValue Undef = SDValue(
375           DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0);
376       SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
377       OutVal = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL,
378                                           MVT::i64, Undef, OutVal, Sub_f32),
379                        0);
380       break;
381     }
382     default:
383       llvm_unreachable("Unknown loc info!");
384     }
385 
386     Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), OutVal, Flag);
387 
388     // Guarantee that all emitted copies are stuck together with flags.
389     Flag = Chain.getValue(1);
390     RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
391   }
392 
393   RetOps[0] = Chain; // Update chain.
394 
395   // Add the flag if we have it.
396   if (Flag.getNode())
397     RetOps.push_back(Flag);
398 
399   return DAG.getNode(VEISD::RET_FLAG, DL, MVT::Other, RetOps);
400 }
401 
402 SDValue VETargetLowering::LowerFormalArguments(
403     SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
404     const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
405     SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
406   MachineFunction &MF = DAG.getMachineFunction();
407 
408   // Get the base offset of the incoming arguments stack space.
409   unsigned ArgsBaseOffset = Subtarget->getRsaSize();
410   // Get the size of the preserved arguments area
411   unsigned ArgsPreserved = 64;
412 
413   // Analyze arguments according to CC_VE.
414   SmallVector<CCValAssign, 16> ArgLocs;
415   CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), ArgLocs,
416                  *DAG.getContext());
417   // Allocate the preserved area first.
418   CCInfo.AllocateStack(ArgsPreserved, Align(8));
419   // We already allocated the preserved area, so the stack offset computed
420   // by CC_VE would be correct now.
421   CCInfo.AnalyzeFormalArguments(Ins, getParamCC(CallConv, false));
422 
423   for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) {
424     CCValAssign &VA = ArgLocs[i];
425     assert(!VA.needsCustom() && "Unexpected custom lowering");
426     if (VA.isRegLoc()) {
427       // This argument is passed in a register.
428       // All integer register arguments are promoted by the caller to i64.
429 
430       // Create a virtual register for the promoted live-in value.
431       Register VReg =
432           MF.addLiveIn(VA.getLocReg(), getRegClassFor(VA.getLocVT()));
433       SDValue Arg = DAG.getCopyFromReg(Chain, DL, VReg, VA.getLocVT());
434 
435       // The caller promoted the argument, so insert an Assert?ext SDNode so we
436       // won't promote the value again in this function.
437       switch (VA.getLocInfo()) {
438       case CCValAssign::SExt:
439         Arg = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), Arg,
440                           DAG.getValueType(VA.getValVT()));
441         break;
442       case CCValAssign::ZExt:
443         Arg = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), Arg,
444                           DAG.getValueType(VA.getValVT()));
445         break;
446       case CCValAssign::BCvt: {
447         // Extract a float argument from i64 with padding.
448         //     63     31   0
449         //    +------+------+
450         //    | float|   0  |
451         //    +------+------+
452         assert(VA.getLocVT() == MVT::i64);
453         assert(VA.getValVT() == MVT::f32);
454         SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
455         Arg = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
456                                          MVT::f32, Arg, Sub_f32),
457                       0);
458         break;
459       }
460       default:
461         break;
462       }
463 
464       // Truncate the register down to the argument type.
465       if (VA.isExtInLoc())
466         Arg = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Arg);
467 
468       InVals.push_back(Arg);
469       continue;
470     }
471 
472     // The registers are exhausted. This argument was passed on the stack.
473     assert(VA.isMemLoc());
474     // The CC_VE_Full/Half functions compute stack offsets relative to the
475     // beginning of the arguments area at %fp + the size of reserved area.
476     unsigned Offset = VA.getLocMemOffset() + ArgsBaseOffset;
477     unsigned ValSize = VA.getValVT().getSizeInBits() / 8;
478 
479     // Adjust offset for a float argument by adding 4 since the argument is
480     // stored in 8 bytes buffer with offset like below.  LLVM generates
481     // 4 bytes load instruction, so need to adjust offset here.  This
482     // adjustment is required in only LowerFormalArguments.  In LowerCall,
483     // a float argument is converted to i64 first, and stored as 8 bytes
484     // data, which is required by ABI, so no need for adjustment.
485     //    0      4
486     //    +------+------+
487     //    | empty| float|
488     //    +------+------+
489     if (VA.getValVT() == MVT::f32)
490       Offset += 4;
491 
492     int FI = MF.getFrameInfo().CreateFixedObject(ValSize, Offset, true);
493     InVals.push_back(
494         DAG.getLoad(VA.getValVT(), DL, Chain,
495                     DAG.getFrameIndex(FI, getPointerTy(MF.getDataLayout())),
496                     MachinePointerInfo::getFixedStack(MF, FI)));
497   }
498 
499   if (!IsVarArg)
500     return Chain;
501 
502   // This function takes variable arguments, some of which may have been passed
503   // in registers %s0-%s8.
504   //
505   // The va_start intrinsic needs to know the offset to the first variable
506   // argument.
507   // TODO: need to calculate offset correctly once we support f128.
508   unsigned ArgOffset = ArgLocs.size() * 8;
509   VEMachineFunctionInfo *FuncInfo = MF.getInfo<VEMachineFunctionInfo>();
510   // Skip the reserved area at the top of stack.
511   FuncInfo->setVarArgsFrameOffset(ArgOffset + ArgsBaseOffset);
512 
513   return Chain;
514 }
515 
516 // FIXME? Maybe this could be a TableGen attribute on some registers and
517 // this table could be generated automatically from RegInfo.
518 Register VETargetLowering::getRegisterByName(const char *RegName, LLT VT,
519                                              const MachineFunction &MF) const {
520   Register Reg = StringSwitch<Register>(RegName)
521                      .Case("sp", VE::SX11)    // Stack pointer
522                      .Case("fp", VE::SX9)     // Frame pointer
523                      .Case("sl", VE::SX8)     // Stack limit
524                      .Case("lr", VE::SX10)    // Link register
525                      .Case("tp", VE::SX14)    // Thread pointer
526                      .Case("outer", VE::SX12) // Outer regiser
527                      .Case("info", VE::SX17)  // Info area register
528                      .Case("got", VE::SX15)   // Global offset table register
529                      .Case("plt", VE::SX16) // Procedure linkage table register
530                      .Default(0);
531 
532   if (Reg)
533     return Reg;
534 
535   report_fatal_error("Invalid register name global variable");
536 }
537 
538 //===----------------------------------------------------------------------===//
539 // TargetLowering Implementation
540 //===----------------------------------------------------------------------===//
541 
542 SDValue VETargetLowering::LowerCall(TargetLowering::CallLoweringInfo &CLI,
543                                     SmallVectorImpl<SDValue> &InVals) const {
544   SelectionDAG &DAG = CLI.DAG;
545   SDLoc DL = CLI.DL;
546   SDValue Chain = CLI.Chain;
547   auto PtrVT = getPointerTy(DAG.getDataLayout());
548 
549   // VE target does not yet support tail call optimization.
550   CLI.IsTailCall = false;
551 
552   // Get the base offset of the outgoing arguments stack space.
553   unsigned ArgsBaseOffset = Subtarget->getRsaSize();
554   // Get the size of the preserved arguments area
555   unsigned ArgsPreserved = 8 * 8u;
556 
557   // Analyze operands of the call, assigning locations to each operand.
558   SmallVector<CCValAssign, 16> ArgLocs;
559   CCState CCInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), ArgLocs,
560                  *DAG.getContext());
561   // Allocate the preserved area first.
562   CCInfo.AllocateStack(ArgsPreserved, Align(8));
563   // We already allocated the preserved area, so the stack offset computed
564   // by CC_VE would be correct now.
565   CCInfo.AnalyzeCallOperands(CLI.Outs, getParamCC(CLI.CallConv, false));
566 
567   // VE requires to use both register and stack for varargs or no-prototyped
568   // functions.
569   bool UseBoth = CLI.IsVarArg;
570 
571   // Analyze operands again if it is required to store BOTH.
572   SmallVector<CCValAssign, 16> ArgLocs2;
573   CCState CCInfo2(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(),
574                   ArgLocs2, *DAG.getContext());
575   if (UseBoth)
576     CCInfo2.AnalyzeCallOperands(CLI.Outs, getParamCC(CLI.CallConv, true));
577 
578   // Get the size of the outgoing arguments stack space requirement.
579   unsigned ArgsSize = CCInfo.getNextStackOffset();
580 
581   // Keep stack frames 16-byte aligned.
582   ArgsSize = alignTo(ArgsSize, 16);
583 
584   // Adjust the stack pointer to make room for the arguments.
585   // FIXME: Use hasReservedCallFrame to avoid %sp adjustments around all calls
586   // with more than 6 arguments.
587   Chain = DAG.getCALLSEQ_START(Chain, ArgsSize, 0, DL);
588 
589   // Collect the set of registers to pass to the function and their values.
590   // This will be emitted as a sequence of CopyToReg nodes glued to the call
591   // instruction.
592   SmallVector<std::pair<unsigned, SDValue>, 8> RegsToPass;
593 
594   // Collect chains from all the memory opeations that copy arguments to the
595   // stack. They must follow the stack pointer adjustment above and precede the
596   // call instruction itself.
597   SmallVector<SDValue, 8> MemOpChains;
598 
599   // VE needs to get address of callee function in a register
600   // So, prepare to copy it to SX12 here.
601 
602   // If the callee is a GlobalAddress node (quite common, every direct call is)
603   // turn it into a TargetGlobalAddress node so that legalize doesn't hack it.
604   // Likewise ExternalSymbol -> TargetExternalSymbol.
605   SDValue Callee = CLI.Callee;
606 
607   bool IsPICCall = isPositionIndependent();
608 
609   // PC-relative references to external symbols should go through $stub.
610   // If so, we need to prepare GlobalBaseReg first.
611   const TargetMachine &TM = DAG.getTarget();
612   const Module *Mod = DAG.getMachineFunction().getFunction().getParent();
613   const GlobalValue *GV = nullptr;
614   auto *CalleeG = dyn_cast<GlobalAddressSDNode>(Callee);
615   if (CalleeG)
616     GV = CalleeG->getGlobal();
617   bool Local = TM.shouldAssumeDSOLocal(*Mod, GV);
618   bool UsePlt = !Local;
619   MachineFunction &MF = DAG.getMachineFunction();
620 
621   // Turn GlobalAddress/ExternalSymbol node into a value node
622   // containing the address of them here.
623   if (CalleeG) {
624     if (IsPICCall) {
625       if (UsePlt)
626         Subtarget->getInstrInfo()->getGlobalBaseReg(&MF);
627       Callee = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, 0);
628       Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee);
629     } else {
630       Callee =
631           makeHiLoPair(Callee, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG);
632     }
633   } else if (ExternalSymbolSDNode *E = dyn_cast<ExternalSymbolSDNode>(Callee)) {
634     if (IsPICCall) {
635       if (UsePlt)
636         Subtarget->getInstrInfo()->getGlobalBaseReg(&MF);
637       Callee = DAG.getTargetExternalSymbol(E->getSymbol(), PtrVT, 0);
638       Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee);
639     } else {
640       Callee =
641           makeHiLoPair(Callee, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG);
642     }
643   }
644 
645   RegsToPass.push_back(std::make_pair(VE::SX12, Callee));
646 
647   for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) {
648     CCValAssign &VA = ArgLocs[i];
649     SDValue Arg = CLI.OutVals[i];
650 
651     // Promote the value if needed.
652     switch (VA.getLocInfo()) {
653     default:
654       llvm_unreachable("Unknown location info!");
655     case CCValAssign::Full:
656       break;
657     case CCValAssign::SExt:
658       Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Arg);
659       break;
660     case CCValAssign::ZExt:
661       Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Arg);
662       break;
663     case CCValAssign::AExt:
664       Arg = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Arg);
665       break;
666     case CCValAssign::BCvt: {
667       // Convert a float argument to i64 with padding.
668       //     63     31   0
669       //    +------+------+
670       //    | float|   0  |
671       //    +------+------+
672       assert(VA.getLocVT() == MVT::i64);
673       assert(VA.getValVT() == MVT::f32);
674       SDValue Undef = SDValue(
675           DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0);
676       SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
677       Arg = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL,
678                                        MVT::i64, Undef, Arg, Sub_f32),
679                     0);
680       break;
681     }
682     }
683 
684     if (VA.isRegLoc()) {
685       RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg));
686       if (!UseBoth)
687         continue;
688       VA = ArgLocs2[i];
689     }
690 
691     assert(VA.isMemLoc());
692 
693     // Create a store off the stack pointer for this argument.
694     SDValue StackPtr = DAG.getRegister(VE::SX11, PtrVT);
695     // The argument area starts at %fp/%sp + the size of reserved area.
696     SDValue PtrOff =
697         DAG.getIntPtrConstant(VA.getLocMemOffset() + ArgsBaseOffset, DL);
698     PtrOff = DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr, PtrOff);
699     MemOpChains.push_back(
700         DAG.getStore(Chain, DL, Arg, PtrOff, MachinePointerInfo()));
701   }
702 
703   // Emit all stores, make sure they occur before the call.
704   if (!MemOpChains.empty())
705     Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains);
706 
707   // Build a sequence of CopyToReg nodes glued together with token chain and
708   // glue operands which copy the outgoing args into registers. The InGlue is
709   // necessary since all emitted instructions must be stuck together in order
710   // to pass the live physical registers.
711   SDValue InGlue;
712   for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) {
713     Chain = DAG.getCopyToReg(Chain, DL, RegsToPass[i].first,
714                              RegsToPass[i].second, InGlue);
715     InGlue = Chain.getValue(1);
716   }
717 
718   // Build the operands for the call instruction itself.
719   SmallVector<SDValue, 8> Ops;
720   Ops.push_back(Chain);
721   for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i)
722     Ops.push_back(DAG.getRegister(RegsToPass[i].first,
723                                   RegsToPass[i].second.getValueType()));
724 
725   // Add a register mask operand representing the call-preserved registers.
726   const VERegisterInfo *TRI = Subtarget->getRegisterInfo();
727   const uint32_t *Mask =
728       TRI->getCallPreservedMask(DAG.getMachineFunction(), CLI.CallConv);
729   assert(Mask && "Missing call preserved mask for calling convention");
730   Ops.push_back(DAG.getRegisterMask(Mask));
731 
732   // Make sure the CopyToReg nodes are glued to the call instruction which
733   // consumes the registers.
734   if (InGlue.getNode())
735     Ops.push_back(InGlue);
736 
737   // Now the call itself.
738   SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
739   Chain = DAG.getNode(VEISD::CALL, DL, NodeTys, Ops);
740   InGlue = Chain.getValue(1);
741 
742   // Revert the stack pointer immediately after the call.
743   Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(ArgsSize, DL, true),
744                              DAG.getIntPtrConstant(0, DL, true), InGlue, DL);
745   InGlue = Chain.getValue(1);
746 
747   // Now extract the return values. This is more or less the same as
748   // LowerFormalArguments.
749 
750   // Assign locations to each value returned by this call.
751   SmallVector<CCValAssign, 16> RVLocs;
752   CCState RVInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), RVLocs,
753                  *DAG.getContext());
754 
755   // Set inreg flag manually for codegen generated library calls that
756   // return float.
757   if (CLI.Ins.size() == 1 && CLI.Ins[0].VT == MVT::f32 && !CLI.CB)
758     CLI.Ins[0].Flags.setInReg();
759 
760   RVInfo.AnalyzeCallResult(CLI.Ins, getReturnCC(CLI.CallConv));
761 
762   // Copy all of the result registers out of their specified physreg.
763   for (unsigned i = 0; i != RVLocs.size(); ++i) {
764     CCValAssign &VA = RVLocs[i];
765     assert(!VA.needsCustom() && "Unexpected custom lowering");
766     Register Reg = VA.getLocReg();
767 
768     // When returning 'inreg {i32, i32 }', two consecutive i32 arguments can
769     // reside in the same register in the high and low bits. Reuse the
770     // CopyFromReg previous node to avoid duplicate copies.
771     SDValue RV;
772     if (RegisterSDNode *SrcReg = dyn_cast<RegisterSDNode>(Chain.getOperand(1)))
773       if (SrcReg->getReg() == Reg && Chain->getOpcode() == ISD::CopyFromReg)
774         RV = Chain.getValue(0);
775 
776     // But usually we'll create a new CopyFromReg for a different register.
777     if (!RV.getNode()) {
778       RV = DAG.getCopyFromReg(Chain, DL, Reg, RVLocs[i].getLocVT(), InGlue);
779       Chain = RV.getValue(1);
780       InGlue = Chain.getValue(2);
781     }
782 
783     // The callee promoted the return value, so insert an Assert?ext SDNode so
784     // we won't promote the value again in this function.
785     switch (VA.getLocInfo()) {
786     case CCValAssign::SExt:
787       RV = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), RV,
788                        DAG.getValueType(VA.getValVT()));
789       break;
790     case CCValAssign::ZExt:
791       RV = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), RV,
792                        DAG.getValueType(VA.getValVT()));
793       break;
794     case CCValAssign::BCvt: {
795       // Extract a float return value from i64 with padding.
796       //     63     31   0
797       //    +------+------+
798       //    | float|   0  |
799       //    +------+------+
800       assert(VA.getLocVT() == MVT::i64);
801       assert(VA.getValVT() == MVT::f32);
802       SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
803       RV = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
804                                       MVT::f32, RV, Sub_f32),
805                    0);
806       break;
807     }
808     default:
809       break;
810     }
811 
812     // Truncate the register down to the return value type.
813     if (VA.isExtInLoc())
814       RV = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), RV);
815 
816     InVals.push_back(RV);
817   }
818 
819   return Chain;
820 }
821 
822 bool VETargetLowering::isOffsetFoldingLegal(
823     const GlobalAddressSDNode *GA) const {
824   // VE uses 64 bit addressing, so we need multiple instructions to generate
825   // an address.  Folding address with offset increases the number of
826   // instructions, so that we disable it here.  Offsets will be folded in
827   // the DAG combine later if it worth to do so.
828   return false;
829 }
830 
831 /// isFPImmLegal - Returns true if the target can instruction select the
832 /// specified FP immediate natively. If false, the legalizer will
833 /// materialize the FP immediate as a load from a constant pool.
834 bool VETargetLowering::isFPImmLegal(const APFloat &Imm, EVT VT,
835                                     bool ForCodeSize) const {
836   return VT == MVT::f32 || VT == MVT::f64;
837 }
838 
839 /// Determine if the target supports unaligned memory accesses.
840 ///
841 /// This function returns true if the target allows unaligned memory accesses
842 /// of the specified type in the given address space. If true, it also returns
843 /// whether the unaligned memory access is "fast" in the last argument by
844 /// reference. This is used, for example, in situations where an array
845 /// copy/move/set is converted to a sequence of store operations. Its use
846 /// helps to ensure that such replacements don't generate code that causes an
847 /// alignment error (trap) on the target machine.
848 bool VETargetLowering::allowsMisalignedMemoryAccesses(EVT VT,
849                                                       unsigned AddrSpace,
850                                                       Align A,
851                                                       MachineMemOperand::Flags,
852                                                       bool *Fast) const {
853   if (Fast) {
854     // It's fast anytime on VE
855     *Fast = true;
856   }
857   return true;
858 }
859 
860 VETargetLowering::VETargetLowering(const TargetMachine &TM,
861                                    const VESubtarget &STI)
862     : TargetLowering(TM), Subtarget(&STI) {
863   // Instructions which use registers as conditionals examine all the
864   // bits (as does the pseudo SELECT_CC expansion). I don't think it
865   // matters much whether it's ZeroOrOneBooleanContent, or
866   // ZeroOrNegativeOneBooleanContent, so, arbitrarily choose the
867   // former.
868   setBooleanContents(ZeroOrOneBooleanContent);
869   setBooleanVectorContents(ZeroOrOneBooleanContent);
870 
871   initRegisterClasses();
872   initSPUActions();
873   initVPUActions();
874 
875   setStackPointerRegisterToSaveRestore(VE::SX11);
876 
877   // We have target-specific dag combine patterns for the following nodes:
878   setTargetDAGCombine(ISD::TRUNCATE);
879 
880   // Set function alignment to 16 bytes
881   setMinFunctionAlignment(Align(16));
882 
883   // VE stores all argument by 8 bytes alignment
884   setMinStackArgumentAlignment(Align(8));
885 
886   computeRegisterProperties(Subtarget->getRegisterInfo());
887 }
888 
889 const char *VETargetLowering::getTargetNodeName(unsigned Opcode) const {
890 #define TARGET_NODE_CASE(NAME)                                                 \
891   case VEISD::NAME:                                                            \
892     return "VEISD::" #NAME;
893   switch ((VEISD::NodeType)Opcode) {
894   case VEISD::FIRST_NUMBER:
895     break;
896     TARGET_NODE_CASE(CALL)
897     TARGET_NODE_CASE(EH_SJLJ_LONGJMP)
898     TARGET_NODE_CASE(EH_SJLJ_SETJMP)
899     TARGET_NODE_CASE(EH_SJLJ_SETUP_DISPATCH)
900     TARGET_NODE_CASE(GETFUNPLT)
901     TARGET_NODE_CASE(GETSTACKTOP)
902     TARGET_NODE_CASE(GETTLSADDR)
903     TARGET_NODE_CASE(GLOBAL_BASE_REG)
904     TARGET_NODE_CASE(Hi)
905     TARGET_NODE_CASE(Lo)
906     TARGET_NODE_CASE(MEMBARRIER)
907     TARGET_NODE_CASE(RET_FLAG)
908     TARGET_NODE_CASE(TS1AM)
909     TARGET_NODE_CASE(VEC_UNPACK_LO)
910     TARGET_NODE_CASE(VEC_UNPACK_HI)
911     TARGET_NODE_CASE(VEC_PACK)
912     TARGET_NODE_CASE(VEC_BROADCAST)
913     TARGET_NODE_CASE(REPL_I32)
914     TARGET_NODE_CASE(REPL_F32)
915 
916     TARGET_NODE_CASE(LEGALAVL)
917 
918     // Register the VVP_* SDNodes.
919 #define ADD_VVP_OP(VVP_NAME, ...) TARGET_NODE_CASE(VVP_NAME)
920 #include "VVPNodes.def"
921   }
922 #undef TARGET_NODE_CASE
923   return nullptr;
924 }
925 
926 EVT VETargetLowering::getSetCCResultType(const DataLayout &, LLVMContext &,
927                                          EVT VT) const {
928   return MVT::i32;
929 }
930 
931 // Convert to a target node and set target flags.
932 SDValue VETargetLowering::withTargetFlags(SDValue Op, unsigned TF,
933                                           SelectionDAG &DAG) const {
934   if (const GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(Op))
935     return DAG.getTargetGlobalAddress(GA->getGlobal(), SDLoc(GA),
936                                       GA->getValueType(0), GA->getOffset(), TF);
937 
938   if (const BlockAddressSDNode *BA = dyn_cast<BlockAddressSDNode>(Op))
939     return DAG.getTargetBlockAddress(BA->getBlockAddress(), Op.getValueType(),
940                                      0, TF);
941 
942   if (const ConstantPoolSDNode *CP = dyn_cast<ConstantPoolSDNode>(Op))
943     return DAG.getTargetConstantPool(CP->getConstVal(), CP->getValueType(0),
944                                      CP->getAlign(), CP->getOffset(), TF);
945 
946   if (const ExternalSymbolSDNode *ES = dyn_cast<ExternalSymbolSDNode>(Op))
947     return DAG.getTargetExternalSymbol(ES->getSymbol(), ES->getValueType(0),
948                                        TF);
949 
950   if (const JumpTableSDNode *JT = dyn_cast<JumpTableSDNode>(Op))
951     return DAG.getTargetJumpTable(JT->getIndex(), JT->getValueType(0), TF);
952 
953   llvm_unreachable("Unhandled address SDNode");
954 }
955 
956 // Split Op into high and low parts according to HiTF and LoTF.
957 // Return an ADD node combining the parts.
958 SDValue VETargetLowering::makeHiLoPair(SDValue Op, unsigned HiTF, unsigned LoTF,
959                                        SelectionDAG &DAG) const {
960   SDLoc DL(Op);
961   EVT VT = Op.getValueType();
962   SDValue Hi = DAG.getNode(VEISD::Hi, DL, VT, withTargetFlags(Op, HiTF, DAG));
963   SDValue Lo = DAG.getNode(VEISD::Lo, DL, VT, withTargetFlags(Op, LoTF, DAG));
964   return DAG.getNode(ISD::ADD, DL, VT, Hi, Lo);
965 }
966 
967 // Build SDNodes for producing an address from a GlobalAddress, ConstantPool,
968 // or ExternalSymbol SDNode.
969 SDValue VETargetLowering::makeAddress(SDValue Op, SelectionDAG &DAG) const {
970   SDLoc DL(Op);
971   EVT PtrVT = Op.getValueType();
972 
973   // Handle PIC mode first. VE needs a got load for every variable!
974   if (isPositionIndependent()) {
975     auto GlobalN = dyn_cast<GlobalAddressSDNode>(Op);
976 
977     if (isa<ConstantPoolSDNode>(Op) || isa<JumpTableSDNode>(Op) ||
978         (GlobalN && GlobalN->getGlobal()->hasLocalLinkage())) {
979       // Create following instructions for local linkage PIC code.
980       //     lea %reg, label@gotoff_lo
981       //     and %reg, %reg, (32)0
982       //     lea.sl %reg, label@gotoff_hi(%reg, %got)
983       SDValue HiLo = makeHiLoPair(Op, VEMCExpr::VK_VE_GOTOFF_HI32,
984                                   VEMCExpr::VK_VE_GOTOFF_LO32, DAG);
985       SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT);
986       return DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo);
987     }
988     // Create following instructions for not local linkage PIC code.
989     //     lea %reg, label@got_lo
990     //     and %reg, %reg, (32)0
991     //     lea.sl %reg, label@got_hi(%reg)
992     //     ld %reg, (%reg, %got)
993     SDValue HiLo = makeHiLoPair(Op, VEMCExpr::VK_VE_GOT_HI32,
994                                 VEMCExpr::VK_VE_GOT_LO32, DAG);
995     SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT);
996     SDValue AbsAddr = DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo);
997     return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), AbsAddr,
998                        MachinePointerInfo::getGOT(DAG.getMachineFunction()));
999   }
1000 
1001   // This is one of the absolute code models.
1002   switch (getTargetMachine().getCodeModel()) {
1003   default:
1004     llvm_unreachable("Unsupported absolute code model");
1005   case CodeModel::Small:
1006   case CodeModel::Medium:
1007   case CodeModel::Large:
1008     // abs64.
1009     return makeHiLoPair(Op, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG);
1010   }
1011 }
1012 
1013 /// Custom Lower {
1014 
1015 // The mappings for emitLeading/TrailingFence for VE is designed by following
1016 // http://www.cl.cam.ac.uk/~pes20/cpp/cpp0xmappings.html
1017 Instruction *VETargetLowering::emitLeadingFence(IRBuilderBase &Builder,
1018                                                 Instruction *Inst,
1019                                                 AtomicOrdering Ord) const {
1020   switch (Ord) {
1021   case AtomicOrdering::NotAtomic:
1022   case AtomicOrdering::Unordered:
1023     llvm_unreachable("Invalid fence: unordered/non-atomic");
1024   case AtomicOrdering::Monotonic:
1025   case AtomicOrdering::Acquire:
1026     return nullptr; // Nothing to do
1027   case AtomicOrdering::Release:
1028   case AtomicOrdering::AcquireRelease:
1029     return Builder.CreateFence(AtomicOrdering::Release);
1030   case AtomicOrdering::SequentiallyConsistent:
1031     if (!Inst->hasAtomicStore())
1032       return nullptr; // Nothing to do
1033     return Builder.CreateFence(AtomicOrdering::SequentiallyConsistent);
1034   }
1035   llvm_unreachable("Unknown fence ordering in emitLeadingFence");
1036 }
1037 
1038 Instruction *VETargetLowering::emitTrailingFence(IRBuilderBase &Builder,
1039                                                  Instruction *Inst,
1040                                                  AtomicOrdering Ord) const {
1041   switch (Ord) {
1042   case AtomicOrdering::NotAtomic:
1043   case AtomicOrdering::Unordered:
1044     llvm_unreachable("Invalid fence: unordered/not-atomic");
1045   case AtomicOrdering::Monotonic:
1046   case AtomicOrdering::Release:
1047     return nullptr; // Nothing to do
1048   case AtomicOrdering::Acquire:
1049   case AtomicOrdering::AcquireRelease:
1050     return Builder.CreateFence(AtomicOrdering::Acquire);
1051   case AtomicOrdering::SequentiallyConsistent:
1052     return Builder.CreateFence(AtomicOrdering::SequentiallyConsistent);
1053   }
1054   llvm_unreachable("Unknown fence ordering in emitTrailingFence");
1055 }
1056 
1057 SDValue VETargetLowering::lowerATOMIC_FENCE(SDValue Op,
1058                                             SelectionDAG &DAG) const {
1059   SDLoc DL(Op);
1060   AtomicOrdering FenceOrdering = static_cast<AtomicOrdering>(
1061       cast<ConstantSDNode>(Op.getOperand(1))->getZExtValue());
1062   SyncScope::ID FenceSSID = static_cast<SyncScope::ID>(
1063       cast<ConstantSDNode>(Op.getOperand(2))->getZExtValue());
1064 
1065   // VE uses Release consistency, so need a fence instruction if it is a
1066   // cross-thread fence.
1067   if (FenceSSID == SyncScope::System) {
1068     switch (FenceOrdering) {
1069     case AtomicOrdering::NotAtomic:
1070     case AtomicOrdering::Unordered:
1071     case AtomicOrdering::Monotonic:
1072       // No need to generate fencem instruction here.
1073       break;
1074     case AtomicOrdering::Acquire:
1075       // Generate "fencem 2" as acquire fence.
1076       return SDValue(DAG.getMachineNode(VE::FENCEM, DL, MVT::Other,
1077                                         DAG.getTargetConstant(2, DL, MVT::i32),
1078                                         Op.getOperand(0)),
1079                      0);
1080     case AtomicOrdering::Release:
1081       // Generate "fencem 1" as release fence.
1082       return SDValue(DAG.getMachineNode(VE::FENCEM, DL, MVT::Other,
1083                                         DAG.getTargetConstant(1, DL, MVT::i32),
1084                                         Op.getOperand(0)),
1085                      0);
1086     case AtomicOrdering::AcquireRelease:
1087     case AtomicOrdering::SequentiallyConsistent:
1088       // Generate "fencem 3" as acq_rel and seq_cst fence.
1089       // FIXME: "fencem 3" doesn't wait for for PCIe deveices accesses,
1090       //        so  seq_cst may require more instruction for them.
1091       return SDValue(DAG.getMachineNode(VE::FENCEM, DL, MVT::Other,
1092                                         DAG.getTargetConstant(3, DL, MVT::i32),
1093                                         Op.getOperand(0)),
1094                      0);
1095     }
1096   }
1097 
1098   // MEMBARRIER is a compiler barrier; it codegens to a no-op.
1099   return DAG.getNode(VEISD::MEMBARRIER, DL, MVT::Other, Op.getOperand(0));
1100 }
1101 
1102 TargetLowering::AtomicExpansionKind
1103 VETargetLowering::shouldExpandAtomicRMWInIR(AtomicRMWInst *AI) const {
1104   // We have TS1AM implementation for i8/i16/i32/i64, so use it.
1105   if (AI->getOperation() == AtomicRMWInst::Xchg) {
1106     return AtomicExpansionKind::None;
1107   }
1108   // FIXME: Support "ATMAM" instruction for LOAD_ADD/SUB/AND/OR.
1109 
1110   // Otherwise, expand it using compare and exchange instruction to not call
1111   // __sync_fetch_and_* functions.
1112   return AtomicExpansionKind::CmpXChg;
1113 }
1114 
1115 static SDValue prepareTS1AM(SDValue Op, SelectionDAG &DAG, SDValue &Flag,
1116                             SDValue &Bits) {
1117   SDLoc DL(Op);
1118   AtomicSDNode *N = cast<AtomicSDNode>(Op);
1119   SDValue Ptr = N->getOperand(1);
1120   SDValue Val = N->getOperand(2);
1121   EVT PtrVT = Ptr.getValueType();
1122   bool Byte = N->getMemoryVT() == MVT::i8;
1123   //   Remainder = AND Ptr, 3
1124   //   Flag = 1 << Remainder  ; If Byte is true (1 byte swap flag)
1125   //   Flag = 3 << Remainder  ; If Byte is false (2 bytes swap flag)
1126   //   Bits = Remainder << 3
1127   //   NewVal = Val << Bits
1128   SDValue Const3 = DAG.getConstant(3, DL, PtrVT);
1129   SDValue Remainder = DAG.getNode(ISD::AND, DL, PtrVT, {Ptr, Const3});
1130   SDValue Mask = Byte ? DAG.getConstant(1, DL, MVT::i32)
1131                       : DAG.getConstant(3, DL, MVT::i32);
1132   Flag = DAG.getNode(ISD::SHL, DL, MVT::i32, {Mask, Remainder});
1133   Bits = DAG.getNode(ISD::SHL, DL, PtrVT, {Remainder, Const3});
1134   return DAG.getNode(ISD::SHL, DL, Val.getValueType(), {Val, Bits});
1135 }
1136 
1137 static SDValue finalizeTS1AM(SDValue Op, SelectionDAG &DAG, SDValue Data,
1138                              SDValue Bits) {
1139   SDLoc DL(Op);
1140   EVT VT = Data.getValueType();
1141   bool Byte = cast<AtomicSDNode>(Op)->getMemoryVT() == MVT::i8;
1142   //   NewData = Data >> Bits
1143   //   Result = NewData & 0xff   ; If Byte is true (1 byte)
1144   //   Result = NewData & 0xffff ; If Byte is false (2 bytes)
1145 
1146   SDValue NewData = DAG.getNode(ISD::SRL, DL, VT, Data, Bits);
1147   return DAG.getNode(ISD::AND, DL, VT,
1148                      {NewData, DAG.getConstant(Byte ? 0xff : 0xffff, DL, VT)});
1149 }
1150 
1151 SDValue VETargetLowering::lowerATOMIC_SWAP(SDValue Op,
1152                                            SelectionDAG &DAG) const {
1153   SDLoc DL(Op);
1154   AtomicSDNode *N = cast<AtomicSDNode>(Op);
1155 
1156   if (N->getMemoryVT() == MVT::i8) {
1157     // For i8, use "ts1am"
1158     //   Input:
1159     //     ATOMIC_SWAP Ptr, Val, Order
1160     //
1161     //   Output:
1162     //     Remainder = AND Ptr, 3
1163     //     Flag = 1 << Remainder   ; 1 byte swap flag for TS1AM inst.
1164     //     Bits = Remainder << 3
1165     //     NewVal = Val << Bits
1166     //
1167     //     Aligned = AND Ptr, -4
1168     //     Data = TS1AM Aligned, Flag, NewVal
1169     //
1170     //     NewData = Data >> Bits
1171     //     Result = NewData & 0xff ; 1 byte result
1172     SDValue Flag;
1173     SDValue Bits;
1174     SDValue NewVal = prepareTS1AM(Op, DAG, Flag, Bits);
1175 
1176     SDValue Ptr = N->getOperand(1);
1177     SDValue Aligned = DAG.getNode(ISD::AND, DL, Ptr.getValueType(),
1178                                   {Ptr, DAG.getConstant(-4, DL, MVT::i64)});
1179     SDValue TS1AM = DAG.getAtomic(VEISD::TS1AM, DL, N->getMemoryVT(),
1180                                   DAG.getVTList(Op.getNode()->getValueType(0),
1181                                                 Op.getNode()->getValueType(1)),
1182                                   {N->getChain(), Aligned, Flag, NewVal},
1183                                   N->getMemOperand());
1184 
1185     SDValue Result = finalizeTS1AM(Op, DAG, TS1AM, Bits);
1186     SDValue Chain = TS1AM.getValue(1);
1187     return DAG.getMergeValues({Result, Chain}, DL);
1188   }
1189   if (N->getMemoryVT() == MVT::i16) {
1190     // For i16, use "ts1am"
1191     SDValue Flag;
1192     SDValue Bits;
1193     SDValue NewVal = prepareTS1AM(Op, DAG, Flag, Bits);
1194 
1195     SDValue Ptr = N->getOperand(1);
1196     SDValue Aligned = DAG.getNode(ISD::AND, DL, Ptr.getValueType(),
1197                                   {Ptr, DAG.getConstant(-4, DL, MVT::i64)});
1198     SDValue TS1AM = DAG.getAtomic(VEISD::TS1AM, DL, N->getMemoryVT(),
1199                                   DAG.getVTList(Op.getNode()->getValueType(0),
1200                                                 Op.getNode()->getValueType(1)),
1201                                   {N->getChain(), Aligned, Flag, NewVal},
1202                                   N->getMemOperand());
1203 
1204     SDValue Result = finalizeTS1AM(Op, DAG, TS1AM, Bits);
1205     SDValue Chain = TS1AM.getValue(1);
1206     return DAG.getMergeValues({Result, Chain}, DL);
1207   }
1208   // Otherwise, let llvm legalize it.
1209   return Op;
1210 }
1211 
1212 SDValue VETargetLowering::lowerGlobalAddress(SDValue Op,
1213                                              SelectionDAG &DAG) const {
1214   return makeAddress(Op, DAG);
1215 }
1216 
1217 SDValue VETargetLowering::lowerBlockAddress(SDValue Op,
1218                                             SelectionDAG &DAG) const {
1219   return makeAddress(Op, DAG);
1220 }
1221 
1222 SDValue VETargetLowering::lowerConstantPool(SDValue Op,
1223                                             SelectionDAG &DAG) const {
1224   return makeAddress(Op, DAG);
1225 }
1226 
1227 SDValue
1228 VETargetLowering::lowerToTLSGeneralDynamicModel(SDValue Op,
1229                                                 SelectionDAG &DAG) const {
1230   SDLoc DL(Op);
1231 
1232   // Generate the following code:
1233   //   t1: ch,glue = callseq_start t0, 0, 0
1234   //   t2: i64,ch,glue = VEISD::GETTLSADDR t1, label, t1:1
1235   //   t3: ch,glue = callseq_end t2, 0, 0, t2:2
1236   //   t4: i64,ch,glue = CopyFromReg t3, Register:i64 $sx0, t3:1
1237   SDValue Label = withTargetFlags(Op, 0, DAG);
1238   EVT PtrVT = Op.getValueType();
1239 
1240   // Lowering the machine isd will make sure everything is in the right
1241   // location.
1242   SDValue Chain = DAG.getEntryNode();
1243   SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
1244   const uint32_t *Mask = Subtarget->getRegisterInfo()->getCallPreservedMask(
1245       DAG.getMachineFunction(), CallingConv::C);
1246   Chain = DAG.getCALLSEQ_START(Chain, 64, 0, DL);
1247   SDValue Args[] = {Chain, Label, DAG.getRegisterMask(Mask), Chain.getValue(1)};
1248   Chain = DAG.getNode(VEISD::GETTLSADDR, DL, NodeTys, Args);
1249   Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(64, DL, true),
1250                              DAG.getIntPtrConstant(0, DL, true),
1251                              Chain.getValue(1), DL);
1252   Chain = DAG.getCopyFromReg(Chain, DL, VE::SX0, PtrVT, Chain.getValue(1));
1253 
1254   // GETTLSADDR will be codegen'ed as call. Inform MFI that function has calls.
1255   MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
1256   MFI.setHasCalls(true);
1257 
1258   // Also generate code to prepare a GOT register if it is PIC.
1259   if (isPositionIndependent()) {
1260     MachineFunction &MF = DAG.getMachineFunction();
1261     Subtarget->getInstrInfo()->getGlobalBaseReg(&MF);
1262   }
1263 
1264   return Chain;
1265 }
1266 
1267 SDValue VETargetLowering::lowerGlobalTLSAddress(SDValue Op,
1268                                                 SelectionDAG &DAG) const {
1269   // The current implementation of nld (2.26) doesn't allow local exec model
1270   // code described in VE-tls_v1.1.pdf (*1) as its input. Instead, we always
1271   // generate the general dynamic model code sequence.
1272   //
1273   // *1: https://www.nec.com/en/global/prod/hpc/aurora/document/VE-tls_v1.1.pdf
1274   return lowerToTLSGeneralDynamicModel(Op, DAG);
1275 }
1276 
1277 SDValue VETargetLowering::lowerJumpTable(SDValue Op, SelectionDAG &DAG) const {
1278   return makeAddress(Op, DAG);
1279 }
1280 
1281 // Lower a f128 load into two f64 loads.
1282 static SDValue lowerLoadF128(SDValue Op, SelectionDAG &DAG) {
1283   SDLoc DL(Op);
1284   LoadSDNode *LdNode = dyn_cast<LoadSDNode>(Op.getNode());
1285   assert(LdNode && LdNode->getOffset().isUndef() && "Unexpected node type");
1286   unsigned Alignment = LdNode->getAlign().value();
1287   if (Alignment > 8)
1288     Alignment = 8;
1289 
1290   SDValue Lo64 =
1291       DAG.getLoad(MVT::f64, DL, LdNode->getChain(), LdNode->getBasePtr(),
1292                   LdNode->getPointerInfo(), Alignment,
1293                   LdNode->isVolatile() ? MachineMemOperand::MOVolatile
1294                                        : MachineMemOperand::MONone);
1295   EVT AddrVT = LdNode->getBasePtr().getValueType();
1296   SDValue HiPtr = DAG.getNode(ISD::ADD, DL, AddrVT, LdNode->getBasePtr(),
1297                               DAG.getConstant(8, DL, AddrVT));
1298   SDValue Hi64 =
1299       DAG.getLoad(MVT::f64, DL, LdNode->getChain(), HiPtr,
1300                   LdNode->getPointerInfo(), Alignment,
1301                   LdNode->isVolatile() ? MachineMemOperand::MOVolatile
1302                                        : MachineMemOperand::MONone);
1303 
1304   SDValue SubRegEven = DAG.getTargetConstant(VE::sub_even, DL, MVT::i32);
1305   SDValue SubRegOdd = DAG.getTargetConstant(VE::sub_odd, DL, MVT::i32);
1306 
1307   // VE stores Hi64 to 8(addr) and Lo64 to 0(addr)
1308   SDNode *InFP128 =
1309       DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::f128);
1310   InFP128 = DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, MVT::f128,
1311                                SDValue(InFP128, 0), Hi64, SubRegEven);
1312   InFP128 = DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, MVT::f128,
1313                                SDValue(InFP128, 0), Lo64, SubRegOdd);
1314   SDValue OutChains[2] = {SDValue(Lo64.getNode(), 1),
1315                           SDValue(Hi64.getNode(), 1)};
1316   SDValue OutChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1317   SDValue Ops[2] = {SDValue(InFP128, 0), OutChain};
1318   return DAG.getMergeValues(Ops, DL);
1319 }
1320 
1321 SDValue VETargetLowering::lowerLOAD(SDValue Op, SelectionDAG &DAG) const {
1322   LoadSDNode *LdNode = cast<LoadSDNode>(Op.getNode());
1323 
1324   SDValue BasePtr = LdNode->getBasePtr();
1325   if (isa<FrameIndexSDNode>(BasePtr.getNode())) {
1326     // Do not expand store instruction with frame index here because of
1327     // dependency problems.  We expand it later in eliminateFrameIndex().
1328     return Op;
1329   }
1330 
1331   EVT MemVT = LdNode->getMemoryVT();
1332   if (MemVT == MVT::f128)
1333     return lowerLoadF128(Op, DAG);
1334 
1335   return Op;
1336 }
1337 
1338 // Lower a f128 store into two f64 stores.
1339 static SDValue lowerStoreF128(SDValue Op, SelectionDAG &DAG) {
1340   SDLoc DL(Op);
1341   StoreSDNode *StNode = dyn_cast<StoreSDNode>(Op.getNode());
1342   assert(StNode && StNode->getOffset().isUndef() && "Unexpected node type");
1343 
1344   SDValue SubRegEven = DAG.getTargetConstant(VE::sub_even, DL, MVT::i32);
1345   SDValue SubRegOdd = DAG.getTargetConstant(VE::sub_odd, DL, MVT::i32);
1346 
1347   SDNode *Hi64 = DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, MVT::i64,
1348                                     StNode->getValue(), SubRegEven);
1349   SDNode *Lo64 = DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, MVT::i64,
1350                                     StNode->getValue(), SubRegOdd);
1351 
1352   unsigned Alignment = StNode->getAlign().value();
1353   if (Alignment > 8)
1354     Alignment = 8;
1355 
1356   // VE stores Hi64 to 8(addr) and Lo64 to 0(addr)
1357   SDValue OutChains[2];
1358   OutChains[0] =
1359       DAG.getStore(StNode->getChain(), DL, SDValue(Lo64, 0),
1360                    StNode->getBasePtr(), MachinePointerInfo(), Alignment,
1361                    StNode->isVolatile() ? MachineMemOperand::MOVolatile
1362                                         : MachineMemOperand::MONone);
1363   EVT AddrVT = StNode->getBasePtr().getValueType();
1364   SDValue HiPtr = DAG.getNode(ISD::ADD, DL, AddrVT, StNode->getBasePtr(),
1365                               DAG.getConstant(8, DL, AddrVT));
1366   OutChains[1] =
1367       DAG.getStore(StNode->getChain(), DL, SDValue(Hi64, 0), HiPtr,
1368                    MachinePointerInfo(), Alignment,
1369                    StNode->isVolatile() ? MachineMemOperand::MOVolatile
1370                                         : MachineMemOperand::MONone);
1371   return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1372 }
1373 
1374 SDValue VETargetLowering::lowerSTORE(SDValue Op, SelectionDAG &DAG) const {
1375   StoreSDNode *StNode = cast<StoreSDNode>(Op.getNode());
1376   assert(StNode && StNode->getOffset().isUndef() && "Unexpected node type");
1377 
1378   SDValue BasePtr = StNode->getBasePtr();
1379   if (isa<FrameIndexSDNode>(BasePtr.getNode())) {
1380     // Do not expand store instruction with frame index here because of
1381     // dependency problems.  We expand it later in eliminateFrameIndex().
1382     return Op;
1383   }
1384 
1385   EVT MemVT = StNode->getMemoryVT();
1386   if (MemVT == MVT::f128)
1387     return lowerStoreF128(Op, DAG);
1388 
1389   // Otherwise, ask llvm to expand it.
1390   return SDValue();
1391 }
1392 
1393 SDValue VETargetLowering::lowerVASTART(SDValue Op, SelectionDAG &DAG) const {
1394   MachineFunction &MF = DAG.getMachineFunction();
1395   VEMachineFunctionInfo *FuncInfo = MF.getInfo<VEMachineFunctionInfo>();
1396   auto PtrVT = getPointerTy(DAG.getDataLayout());
1397 
1398   // Need frame address to find the address of VarArgsFrameIndex.
1399   MF.getFrameInfo().setFrameAddressIsTaken(true);
1400 
1401   // vastart just stores the address of the VarArgsFrameIndex slot into the
1402   // memory location argument.
1403   SDLoc DL(Op);
1404   SDValue Offset =
1405       DAG.getNode(ISD::ADD, DL, PtrVT, DAG.getRegister(VE::SX9, PtrVT),
1406                   DAG.getIntPtrConstant(FuncInfo->getVarArgsFrameOffset(), DL));
1407   const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
1408   return DAG.getStore(Op.getOperand(0), DL, Offset, Op.getOperand(1),
1409                       MachinePointerInfo(SV));
1410 }
1411 
1412 SDValue VETargetLowering::lowerVAARG(SDValue Op, SelectionDAG &DAG) const {
1413   SDNode *Node = Op.getNode();
1414   EVT VT = Node->getValueType(0);
1415   SDValue InChain = Node->getOperand(0);
1416   SDValue VAListPtr = Node->getOperand(1);
1417   EVT PtrVT = VAListPtr.getValueType();
1418   const Value *SV = cast<SrcValueSDNode>(Node->getOperand(2))->getValue();
1419   SDLoc DL(Node);
1420   SDValue VAList =
1421       DAG.getLoad(PtrVT, DL, InChain, VAListPtr, MachinePointerInfo(SV));
1422   SDValue Chain = VAList.getValue(1);
1423   SDValue NextPtr;
1424 
1425   if (VT == MVT::f128) {
1426     // VE f128 values must be stored with 16 bytes alignment.  We doesn't
1427     // know the actual alignment of VAList, so we take alignment of it
1428     // dyanmically.
1429     int Align = 16;
1430     VAList = DAG.getNode(ISD::ADD, DL, PtrVT, VAList,
1431                          DAG.getConstant(Align - 1, DL, PtrVT));
1432     VAList = DAG.getNode(ISD::AND, DL, PtrVT, VAList,
1433                          DAG.getConstant(-Align, DL, PtrVT));
1434     // Increment the pointer, VAList, by 16 to the next vaarg.
1435     NextPtr =
1436         DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(16, DL));
1437   } else if (VT == MVT::f32) {
1438     // float --> need special handling like below.
1439     //    0      4
1440     //    +------+------+
1441     //    | empty| float|
1442     //    +------+------+
1443     // Increment the pointer, VAList, by 8 to the next vaarg.
1444     NextPtr =
1445         DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL));
1446     // Then, adjust VAList.
1447     unsigned InternalOffset = 4;
1448     VAList = DAG.getNode(ISD::ADD, DL, PtrVT, VAList,
1449                          DAG.getConstant(InternalOffset, DL, PtrVT));
1450   } else {
1451     // Increment the pointer, VAList, by 8 to the next vaarg.
1452     NextPtr =
1453         DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL));
1454   }
1455 
1456   // Store the incremented VAList to the legalized pointer.
1457   InChain = DAG.getStore(Chain, DL, NextPtr, VAListPtr, MachinePointerInfo(SV));
1458 
1459   // Load the actual argument out of the pointer VAList.
1460   // We can't count on greater alignment than the word size.
1461   return DAG.getLoad(VT, DL, InChain, VAList, MachinePointerInfo(),
1462                      std::min(PtrVT.getSizeInBits(), VT.getSizeInBits()) / 8);
1463 }
1464 
1465 SDValue VETargetLowering::lowerDYNAMIC_STACKALLOC(SDValue Op,
1466                                                   SelectionDAG &DAG) const {
1467   // Generate following code.
1468   //   (void)__llvm_grow_stack(size);
1469   //   ret = GETSTACKTOP;        // pseudo instruction
1470   SDLoc DL(Op);
1471 
1472   // Get the inputs.
1473   SDNode *Node = Op.getNode();
1474   SDValue Chain = Op.getOperand(0);
1475   SDValue Size = Op.getOperand(1);
1476   MaybeAlign Alignment(Op.getConstantOperandVal(2));
1477   EVT VT = Node->getValueType(0);
1478 
1479   // Chain the dynamic stack allocation so that it doesn't modify the stack
1480   // pointer when other instructions are using the stack.
1481   Chain = DAG.getCALLSEQ_START(Chain, 0, 0, DL);
1482 
1483   const TargetFrameLowering &TFI = *Subtarget->getFrameLowering();
1484   Align StackAlign = TFI.getStackAlign();
1485   bool NeedsAlign = Alignment.valueOrOne() > StackAlign;
1486 
1487   // Prepare arguments
1488   TargetLowering::ArgListTy Args;
1489   TargetLowering::ArgListEntry Entry;
1490   Entry.Node = Size;
1491   Entry.Ty = Entry.Node.getValueType().getTypeForEVT(*DAG.getContext());
1492   Args.push_back(Entry);
1493   if (NeedsAlign) {
1494     Entry.Node = DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT);
1495     Entry.Ty = Entry.Node.getValueType().getTypeForEVT(*DAG.getContext());
1496     Args.push_back(Entry);
1497   }
1498   Type *RetTy = Type::getVoidTy(*DAG.getContext());
1499 
1500   EVT PtrVT = Op.getValueType();
1501   SDValue Callee;
1502   if (NeedsAlign) {
1503     Callee = DAG.getTargetExternalSymbol("__ve_grow_stack_align", PtrVT, 0);
1504   } else {
1505     Callee = DAG.getTargetExternalSymbol("__ve_grow_stack", PtrVT, 0);
1506   }
1507 
1508   TargetLowering::CallLoweringInfo CLI(DAG);
1509   CLI.setDebugLoc(DL)
1510       .setChain(Chain)
1511       .setCallee(CallingConv::PreserveAll, RetTy, Callee, std::move(Args))
1512       .setDiscardResult(true);
1513   std::pair<SDValue, SDValue> pair = LowerCallTo(CLI);
1514   Chain = pair.second;
1515   SDValue Result = DAG.getNode(VEISD::GETSTACKTOP, DL, VT, Chain);
1516   if (NeedsAlign) {
1517     Result = DAG.getNode(ISD::ADD, DL, VT, Result,
1518                          DAG.getConstant((Alignment->value() - 1ULL), DL, VT));
1519     Result = DAG.getNode(ISD::AND, DL, VT, Result,
1520                          DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT));
1521   }
1522   //  Chain = Result.getValue(1);
1523   Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(0, DL, true),
1524                              DAG.getIntPtrConstant(0, DL, true), SDValue(), DL);
1525 
1526   SDValue Ops[2] = {Result, Chain};
1527   return DAG.getMergeValues(Ops, DL);
1528 }
1529 
1530 SDValue VETargetLowering::lowerEH_SJLJ_LONGJMP(SDValue Op,
1531                                                SelectionDAG &DAG) const {
1532   SDLoc DL(Op);
1533   return DAG.getNode(VEISD::EH_SJLJ_LONGJMP, DL, MVT::Other, Op.getOperand(0),
1534                      Op.getOperand(1));
1535 }
1536 
1537 SDValue VETargetLowering::lowerEH_SJLJ_SETJMP(SDValue Op,
1538                                               SelectionDAG &DAG) const {
1539   SDLoc DL(Op);
1540   return DAG.getNode(VEISD::EH_SJLJ_SETJMP, DL,
1541                      DAG.getVTList(MVT::i32, MVT::Other), Op.getOperand(0),
1542                      Op.getOperand(1));
1543 }
1544 
1545 SDValue VETargetLowering::lowerEH_SJLJ_SETUP_DISPATCH(SDValue Op,
1546                                                       SelectionDAG &DAG) const {
1547   SDLoc DL(Op);
1548   return DAG.getNode(VEISD::EH_SJLJ_SETUP_DISPATCH, DL, MVT::Other,
1549                      Op.getOperand(0));
1550 }
1551 
1552 static SDValue lowerFRAMEADDR(SDValue Op, SelectionDAG &DAG,
1553                               const VETargetLowering &TLI,
1554                               const VESubtarget *Subtarget) {
1555   SDLoc DL(Op);
1556   MachineFunction &MF = DAG.getMachineFunction();
1557   EVT PtrVT = TLI.getPointerTy(MF.getDataLayout());
1558 
1559   MachineFrameInfo &MFI = MF.getFrameInfo();
1560   MFI.setFrameAddressIsTaken(true);
1561 
1562   unsigned Depth = Op.getConstantOperandVal(0);
1563   const VERegisterInfo *RegInfo = Subtarget->getRegisterInfo();
1564   Register FrameReg = RegInfo->getFrameRegister(MF);
1565   SDValue FrameAddr =
1566       DAG.getCopyFromReg(DAG.getEntryNode(), DL, FrameReg, PtrVT);
1567   while (Depth--)
1568     FrameAddr = DAG.getLoad(Op.getValueType(), DL, DAG.getEntryNode(),
1569                             FrameAddr, MachinePointerInfo());
1570   return FrameAddr;
1571 }
1572 
1573 static SDValue lowerRETURNADDR(SDValue Op, SelectionDAG &DAG,
1574                                const VETargetLowering &TLI,
1575                                const VESubtarget *Subtarget) {
1576   MachineFunction &MF = DAG.getMachineFunction();
1577   MachineFrameInfo &MFI = MF.getFrameInfo();
1578   MFI.setReturnAddressIsTaken(true);
1579 
1580   if (TLI.verifyReturnAddressArgumentIsConstant(Op, DAG))
1581     return SDValue();
1582 
1583   SDValue FrameAddr = lowerFRAMEADDR(Op, DAG, TLI, Subtarget);
1584 
1585   SDLoc DL(Op);
1586   EVT VT = Op.getValueType();
1587   SDValue Offset = DAG.getConstant(8, DL, VT);
1588   return DAG.getLoad(VT, DL, DAG.getEntryNode(),
1589                      DAG.getNode(ISD::ADD, DL, VT, FrameAddr, Offset),
1590                      MachinePointerInfo());
1591 }
1592 
1593 SDValue VETargetLowering::lowerINTRINSIC_WO_CHAIN(SDValue Op,
1594                                                   SelectionDAG &DAG) const {
1595   SDLoc DL(Op);
1596   unsigned IntNo = cast<ConstantSDNode>(Op.getOperand(0))->getZExtValue();
1597   switch (IntNo) {
1598   default: // Don't custom lower most intrinsics.
1599     return SDValue();
1600   case Intrinsic::eh_sjlj_lsda: {
1601     MachineFunction &MF = DAG.getMachineFunction();
1602     MVT VT = Op.getSimpleValueType();
1603     const VETargetMachine *TM =
1604         static_cast<const VETargetMachine *>(&DAG.getTarget());
1605 
1606     // Create GCC_except_tableXX string.  The real symbol for that will be
1607     // generated in EHStreamer::emitExceptionTable() later.  So, we just
1608     // borrow it's name here.
1609     TM->getStrList()->push_back(std::string(
1610         (Twine("GCC_except_table") + Twine(MF.getFunctionNumber())).str()));
1611     SDValue Addr =
1612         DAG.getTargetExternalSymbol(TM->getStrList()->back().c_str(), VT, 0);
1613     if (isPositionIndependent()) {
1614       Addr = makeHiLoPair(Addr, VEMCExpr::VK_VE_GOTOFF_HI32,
1615                           VEMCExpr::VK_VE_GOTOFF_LO32, DAG);
1616       SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, VT);
1617       return DAG.getNode(ISD::ADD, DL, VT, GlobalBase, Addr);
1618     }
1619     return makeHiLoPair(Addr, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG);
1620   }
1621   }
1622 }
1623 
1624 static bool getUniqueInsertion(SDNode *N, unsigned &UniqueIdx) {
1625   if (!isa<BuildVectorSDNode>(N))
1626     return false;
1627   const auto *BVN = cast<BuildVectorSDNode>(N);
1628 
1629   // Find first non-undef insertion.
1630   unsigned Idx;
1631   for (Idx = 0; Idx < BVN->getNumOperands(); ++Idx) {
1632     auto ElemV = BVN->getOperand(Idx);
1633     if (!ElemV->isUndef())
1634       break;
1635   }
1636   // Catch the (hypothetical) all-undef case.
1637   if (Idx == BVN->getNumOperands())
1638     return false;
1639   // Remember insertion.
1640   UniqueIdx = Idx++;
1641   // Verify that all other insertions are undef.
1642   for (; Idx < BVN->getNumOperands(); ++Idx) {
1643     auto ElemV = BVN->getOperand(Idx);
1644     if (!ElemV->isUndef())
1645       return false;
1646   }
1647   return true;
1648 }
1649 
1650 static SDValue getSplatValue(SDNode *N) {
1651   if (auto *BuildVec = dyn_cast<BuildVectorSDNode>(N)) {
1652     return BuildVec->getSplatValue();
1653   }
1654   return SDValue();
1655 }
1656 
1657 SDValue VETargetLowering::lowerBUILD_VECTOR(SDValue Op,
1658                                             SelectionDAG &DAG) const {
1659   VECustomDAG CDAG(DAG, Op);
1660   MVT ResultVT = Op.getSimpleValueType();
1661 
1662   // If there is just one element, expand to INSERT_VECTOR_ELT.
1663   unsigned UniqueIdx;
1664   if (getUniqueInsertion(Op.getNode(), UniqueIdx)) {
1665     SDValue AccuV = CDAG.getUNDEF(Op.getValueType());
1666     auto ElemV = Op->getOperand(UniqueIdx);
1667     SDValue IdxV = CDAG.getConstant(UniqueIdx, MVT::i64);
1668     return CDAG.getNode(ISD::INSERT_VECTOR_ELT, ResultVT, {AccuV, ElemV, IdxV});
1669   }
1670 
1671   // Else emit a broadcast.
1672   if (SDValue ScalarV = getSplatValue(Op.getNode())) {
1673     unsigned NumEls = ResultVT.getVectorNumElements();
1674     auto AVL = CDAG.getConstant(NumEls, MVT::i32);
1675     return CDAG.getBroadcast(ResultVT, ScalarV, AVL);
1676   }
1677 
1678   // Expand
1679   return SDValue();
1680 }
1681 
1682 TargetLowering::LegalizeAction
1683 VETargetLowering::getCustomOperationAction(SDNode &Op) const {
1684   // Custom legalization on VVP_* and VEC_* opcodes is required to pack-legalize
1685   // these operations (transform nodes such that their AVL parameter refers to
1686   // packs of 64bit, instead of number of elements.
1687 
1688   // Packing opcodes are created with a pack-legal AVL (LEGALAVL). No need to
1689   // re-visit them.
1690   if (isPackingSupportOpcode(Op.getOpcode()))
1691     return Legal;
1692 
1693   // Custom lower to legalize AVL for packed mode.
1694   if (isVVPOrVEC(Op.getOpcode()))
1695     return Custom;
1696   return Legal;
1697 }
1698 
1699 SDValue VETargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
1700   LLVM_DEBUG(dbgs() << "::LowerOperation"; Op->print(dbgs()););
1701   unsigned Opcode = Op.getOpcode();
1702   if (ISD::isVPOpcode(Opcode))
1703     return lowerToVVP(Op, DAG);
1704 
1705   switch (Opcode) {
1706   default:
1707     llvm_unreachable("Should not custom lower this!");
1708   case ISD::ATOMIC_FENCE:
1709     return lowerATOMIC_FENCE(Op, DAG);
1710   case ISD::ATOMIC_SWAP:
1711     return lowerATOMIC_SWAP(Op, DAG);
1712   case ISD::BlockAddress:
1713     return lowerBlockAddress(Op, DAG);
1714   case ISD::ConstantPool:
1715     return lowerConstantPool(Op, DAG);
1716   case ISD::DYNAMIC_STACKALLOC:
1717     return lowerDYNAMIC_STACKALLOC(Op, DAG);
1718   case ISD::EH_SJLJ_LONGJMP:
1719     return lowerEH_SJLJ_LONGJMP(Op, DAG);
1720   case ISD::EH_SJLJ_SETJMP:
1721     return lowerEH_SJLJ_SETJMP(Op, DAG);
1722   case ISD::EH_SJLJ_SETUP_DISPATCH:
1723     return lowerEH_SJLJ_SETUP_DISPATCH(Op, DAG);
1724   case ISD::FRAMEADDR:
1725     return lowerFRAMEADDR(Op, DAG, *this, Subtarget);
1726   case ISD::GlobalAddress:
1727     return lowerGlobalAddress(Op, DAG);
1728   case ISD::GlobalTLSAddress:
1729     return lowerGlobalTLSAddress(Op, DAG);
1730   case ISD::INTRINSIC_WO_CHAIN:
1731     return lowerINTRINSIC_WO_CHAIN(Op, DAG);
1732   case ISD::JumpTable:
1733     return lowerJumpTable(Op, DAG);
1734   case ISD::LOAD:
1735     return lowerLOAD(Op, DAG);
1736   case ISD::RETURNADDR:
1737     return lowerRETURNADDR(Op, DAG, *this, Subtarget);
1738   case ISD::BUILD_VECTOR:
1739     return lowerBUILD_VECTOR(Op, DAG);
1740   case ISD::STORE:
1741     return lowerSTORE(Op, DAG);
1742   case ISD::VASTART:
1743     return lowerVASTART(Op, DAG);
1744   case ISD::VAARG:
1745     return lowerVAARG(Op, DAG);
1746 
1747   case ISD::INSERT_VECTOR_ELT:
1748     return lowerINSERT_VECTOR_ELT(Op, DAG);
1749   case ISD::EXTRACT_VECTOR_ELT:
1750     return lowerEXTRACT_VECTOR_ELT(Op, DAG);
1751 
1752   // Legalize the AVL of this internal node.
1753   case VEISD::VEC_BROADCAST:
1754 #define ADD_VVP_OP(VVP_NAME, ...) case VEISD::VVP_NAME:
1755 #include "VVPNodes.def"
1756     // AVL already legalized.
1757     if (getAnnotatedNodeAVL(Op).second)
1758       return Op;
1759     return legalizeInternalVectorOp(Op, DAG);
1760 
1761     // Translate into a VEC_*/VVP_* layer operation.
1762 #define ADD_VVP_OP(VVP_NAME, ISD_NAME) case ISD::ISD_NAME:
1763 #include "VVPNodes.def"
1764     if (isMaskArithmetic(Op) && isPackedVectorType(Op.getValueType()))
1765       return splitMaskArithmetic(Op, DAG);
1766     return lowerToVVP(Op, DAG);
1767   }
1768 }
1769 /// } Custom Lower
1770 
1771 void VETargetLowering::ReplaceNodeResults(SDNode *N,
1772                                           SmallVectorImpl<SDValue> &Results,
1773                                           SelectionDAG &DAG) const {
1774   switch (N->getOpcode()) {
1775   case ISD::ATOMIC_SWAP:
1776     // Let LLVM expand atomic swap instruction through LowerOperation.
1777     return;
1778   default:
1779     LLVM_DEBUG(N->dumpr(&DAG));
1780     llvm_unreachable("Do not know how to custom type legalize this operation!");
1781   }
1782 }
1783 
1784 /// JumpTable for VE.
1785 ///
1786 ///   VE cannot generate relocatable symbol in jump table.  VE cannot
1787 ///   generate expressions using symbols in both text segment and data
1788 ///   segment like below.
1789 ///             .4byte  .LBB0_2-.LJTI0_0
1790 ///   So, we generate offset from the top of function like below as
1791 ///   a custom label.
1792 ///             .4byte  .LBB0_2-<function name>
1793 
1794 unsigned VETargetLowering::getJumpTableEncoding() const {
1795   // Use custom label for PIC.
1796   if (isPositionIndependent())
1797     return MachineJumpTableInfo::EK_Custom32;
1798 
1799   // Otherwise, use the normal jump table encoding heuristics.
1800   return TargetLowering::getJumpTableEncoding();
1801 }
1802 
1803 const MCExpr *VETargetLowering::LowerCustomJumpTableEntry(
1804     const MachineJumpTableInfo *MJTI, const MachineBasicBlock *MBB,
1805     unsigned Uid, MCContext &Ctx) const {
1806   assert(isPositionIndependent());
1807 
1808   // Generate custom label for PIC like below.
1809   //    .4bytes  .LBB0_2-<function name>
1810   const auto *Value = MCSymbolRefExpr::create(MBB->getSymbol(), Ctx);
1811   MCSymbol *Sym = Ctx.getOrCreateSymbol(MBB->getParent()->getName().data());
1812   const auto *Base = MCSymbolRefExpr::create(Sym, Ctx);
1813   return MCBinaryExpr::createSub(Value, Base, Ctx);
1814 }
1815 
1816 SDValue VETargetLowering::getPICJumpTableRelocBase(SDValue Table,
1817                                                    SelectionDAG &DAG) const {
1818   assert(isPositionIndependent());
1819   SDLoc DL(Table);
1820   Function *Function = &DAG.getMachineFunction().getFunction();
1821   assert(Function != nullptr);
1822   auto PtrTy = getPointerTy(DAG.getDataLayout(), Function->getAddressSpace());
1823 
1824   // In the jump table, we have following values in PIC mode.
1825   //    .4bytes  .LBB0_2-<function name>
1826   // We need to add this value and the address of this function to generate
1827   // .LBB0_2 label correctly under PIC mode.  So, we want to generate following
1828   // instructions:
1829   //     lea %reg, fun@gotoff_lo
1830   //     and %reg, %reg, (32)0
1831   //     lea.sl %reg, fun@gotoff_hi(%reg, %got)
1832   // In order to do so, we need to genarate correctly marked DAG node using
1833   // makeHiLoPair.
1834   SDValue Op = DAG.getGlobalAddress(Function, DL, PtrTy);
1835   SDValue HiLo = makeHiLoPair(Op, VEMCExpr::VK_VE_GOTOFF_HI32,
1836                               VEMCExpr::VK_VE_GOTOFF_LO32, DAG);
1837   SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrTy);
1838   return DAG.getNode(ISD::ADD, DL, PtrTy, GlobalBase, HiLo);
1839 }
1840 
1841 Register VETargetLowering::prepareMBB(MachineBasicBlock &MBB,
1842                                       MachineBasicBlock::iterator I,
1843                                       MachineBasicBlock *TargetBB,
1844                                       const DebugLoc &DL) const {
1845   MachineFunction *MF = MBB.getParent();
1846   MachineRegisterInfo &MRI = MF->getRegInfo();
1847   const VEInstrInfo *TII = Subtarget->getInstrInfo();
1848 
1849   const TargetRegisterClass *RC = &VE::I64RegClass;
1850   Register Tmp1 = MRI.createVirtualRegister(RC);
1851   Register Tmp2 = MRI.createVirtualRegister(RC);
1852   Register Result = MRI.createVirtualRegister(RC);
1853 
1854   if (isPositionIndependent()) {
1855     // Create following instructions for local linkage PIC code.
1856     //     lea %Tmp1, TargetBB@gotoff_lo
1857     //     and %Tmp2, %Tmp1, (32)0
1858     //     lea.sl %Result, TargetBB@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
1859     BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
1860         .addImm(0)
1861         .addImm(0)
1862         .addMBB(TargetBB, VEMCExpr::VK_VE_GOTOFF_LO32);
1863     BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
1864         .addReg(Tmp1, getKillRegState(true))
1865         .addImm(M0(32));
1866     BuildMI(MBB, I, DL, TII->get(VE::LEASLrri), Result)
1867         .addReg(VE::SX15)
1868         .addReg(Tmp2, getKillRegState(true))
1869         .addMBB(TargetBB, VEMCExpr::VK_VE_GOTOFF_HI32);
1870   } else {
1871     // Create following instructions for non-PIC code.
1872     //     lea     %Tmp1, TargetBB@lo
1873     //     and     %Tmp2, %Tmp1, (32)0
1874     //     lea.sl  %Result, TargetBB@hi(%Tmp2)
1875     BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
1876         .addImm(0)
1877         .addImm(0)
1878         .addMBB(TargetBB, VEMCExpr::VK_VE_LO32);
1879     BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
1880         .addReg(Tmp1, getKillRegState(true))
1881         .addImm(M0(32));
1882     BuildMI(MBB, I, DL, TII->get(VE::LEASLrii), Result)
1883         .addReg(Tmp2, getKillRegState(true))
1884         .addImm(0)
1885         .addMBB(TargetBB, VEMCExpr::VK_VE_HI32);
1886   }
1887   return Result;
1888 }
1889 
1890 Register VETargetLowering::prepareSymbol(MachineBasicBlock &MBB,
1891                                          MachineBasicBlock::iterator I,
1892                                          StringRef Symbol, const DebugLoc &DL,
1893                                          bool IsLocal = false,
1894                                          bool IsCall = false) const {
1895   MachineFunction *MF = MBB.getParent();
1896   MachineRegisterInfo &MRI = MF->getRegInfo();
1897   const VEInstrInfo *TII = Subtarget->getInstrInfo();
1898 
1899   const TargetRegisterClass *RC = &VE::I64RegClass;
1900   Register Result = MRI.createVirtualRegister(RC);
1901 
1902   if (isPositionIndependent()) {
1903     if (IsCall && !IsLocal) {
1904       // Create following instructions for non-local linkage PIC code function
1905       // calls.  These instructions uses IC and magic number -24, so we expand
1906       // them in VEAsmPrinter.cpp from GETFUNPLT pseudo instruction.
1907       //     lea %Reg, Symbol@plt_lo(-24)
1908       //     and %Reg, %Reg, (32)0
1909       //     sic %s16
1910       //     lea.sl %Result, Symbol@plt_hi(%Reg, %s16) ; %s16 is PLT
1911       BuildMI(MBB, I, DL, TII->get(VE::GETFUNPLT), Result)
1912           .addExternalSymbol("abort");
1913     } else if (IsLocal) {
1914       Register Tmp1 = MRI.createVirtualRegister(RC);
1915       Register Tmp2 = MRI.createVirtualRegister(RC);
1916       // Create following instructions for local linkage PIC code.
1917       //     lea %Tmp1, Symbol@gotoff_lo
1918       //     and %Tmp2, %Tmp1, (32)0
1919       //     lea.sl %Result, Symbol@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
1920       BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
1921           .addImm(0)
1922           .addImm(0)
1923           .addExternalSymbol(Symbol.data(), VEMCExpr::VK_VE_GOTOFF_LO32);
1924       BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
1925           .addReg(Tmp1, getKillRegState(true))
1926           .addImm(M0(32));
1927       BuildMI(MBB, I, DL, TII->get(VE::LEASLrri), Result)
1928           .addReg(VE::SX15)
1929           .addReg(Tmp2, getKillRegState(true))
1930           .addExternalSymbol(Symbol.data(), VEMCExpr::VK_VE_GOTOFF_HI32);
1931     } else {
1932       Register Tmp1 = MRI.createVirtualRegister(RC);
1933       Register Tmp2 = MRI.createVirtualRegister(RC);
1934       // Create following instructions for not local linkage PIC code.
1935       //     lea %Tmp1, Symbol@got_lo
1936       //     and %Tmp2, %Tmp1, (32)0
1937       //     lea.sl %Tmp3, Symbol@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
1938       //     ld %Result, 0(%Tmp3)
1939       Register Tmp3 = MRI.createVirtualRegister(RC);
1940       BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
1941           .addImm(0)
1942           .addImm(0)
1943           .addExternalSymbol(Symbol.data(), VEMCExpr::VK_VE_GOT_LO32);
1944       BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
1945           .addReg(Tmp1, getKillRegState(true))
1946           .addImm(M0(32));
1947       BuildMI(MBB, I, DL, TII->get(VE::LEASLrri), Tmp3)
1948           .addReg(VE::SX15)
1949           .addReg(Tmp2, getKillRegState(true))
1950           .addExternalSymbol(Symbol.data(), VEMCExpr::VK_VE_GOT_HI32);
1951       BuildMI(MBB, I, DL, TII->get(VE::LDrii), Result)
1952           .addReg(Tmp3, getKillRegState(true))
1953           .addImm(0)
1954           .addImm(0);
1955     }
1956   } else {
1957     Register Tmp1 = MRI.createVirtualRegister(RC);
1958     Register Tmp2 = MRI.createVirtualRegister(RC);
1959     // Create following instructions for non-PIC code.
1960     //     lea     %Tmp1, Symbol@lo
1961     //     and     %Tmp2, %Tmp1, (32)0
1962     //     lea.sl  %Result, Symbol@hi(%Tmp2)
1963     BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
1964         .addImm(0)
1965         .addImm(0)
1966         .addExternalSymbol(Symbol.data(), VEMCExpr::VK_VE_LO32);
1967     BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
1968         .addReg(Tmp1, getKillRegState(true))
1969         .addImm(M0(32));
1970     BuildMI(MBB, I, DL, TII->get(VE::LEASLrii), Result)
1971         .addReg(Tmp2, getKillRegState(true))
1972         .addImm(0)
1973         .addExternalSymbol(Symbol.data(), VEMCExpr::VK_VE_HI32);
1974   }
1975   return Result;
1976 }
1977 
1978 void VETargetLowering::setupEntryBlockForSjLj(MachineInstr &MI,
1979                                               MachineBasicBlock *MBB,
1980                                               MachineBasicBlock *DispatchBB,
1981                                               int FI, int Offset) const {
1982   DebugLoc DL = MI.getDebugLoc();
1983   const VEInstrInfo *TII = Subtarget->getInstrInfo();
1984 
1985   Register LabelReg =
1986       prepareMBB(*MBB, MachineBasicBlock::iterator(MI), DispatchBB, DL);
1987 
1988   // Store an address of DispatchBB to a given jmpbuf[1] where has next IC
1989   // referenced by longjmp (throw) later.
1990   MachineInstrBuilder MIB = BuildMI(*MBB, MI, DL, TII->get(VE::STrii));
1991   addFrameReference(MIB, FI, Offset); // jmpbuf[1]
1992   MIB.addReg(LabelReg, getKillRegState(true));
1993 }
1994 
1995 MachineBasicBlock *
1996 VETargetLowering::emitEHSjLjSetJmp(MachineInstr &MI,
1997                                    MachineBasicBlock *MBB) const {
1998   DebugLoc DL = MI.getDebugLoc();
1999   MachineFunction *MF = MBB->getParent();
2000   const TargetInstrInfo *TII = Subtarget->getInstrInfo();
2001   const TargetRegisterInfo *TRI = Subtarget->getRegisterInfo();
2002   MachineRegisterInfo &MRI = MF->getRegInfo();
2003 
2004   const BasicBlock *BB = MBB->getBasicBlock();
2005   MachineFunction::iterator I = ++MBB->getIterator();
2006 
2007   // Memory Reference.
2008   SmallVector<MachineMemOperand *, 2> MMOs(MI.memoperands_begin(),
2009                                            MI.memoperands_end());
2010   Register BufReg = MI.getOperand(1).getReg();
2011 
2012   Register DstReg;
2013 
2014   DstReg = MI.getOperand(0).getReg();
2015   const TargetRegisterClass *RC = MRI.getRegClass(DstReg);
2016   assert(TRI->isTypeLegalForClass(*RC, MVT::i32) && "Invalid destination!");
2017   (void)TRI;
2018   Register MainDestReg = MRI.createVirtualRegister(RC);
2019   Register RestoreDestReg = MRI.createVirtualRegister(RC);
2020 
2021   // For `v = call @llvm.eh.sjlj.setjmp(buf)`, we generate following
2022   // instructions.  SP/FP must be saved in jmpbuf before `llvm.eh.sjlj.setjmp`.
2023   //
2024   // ThisMBB:
2025   //   buf[3] = %s17 iff %s17 is used as BP
2026   //   buf[1] = RestoreMBB as IC after longjmp
2027   //   # SjLjSetup RestoreMBB
2028   //
2029   // MainMBB:
2030   //   v_main = 0
2031   //
2032   // SinkMBB:
2033   //   v = phi(v_main, MainMBB, v_restore, RestoreMBB)
2034   //   ...
2035   //
2036   // RestoreMBB:
2037   //   %s17 = buf[3] = iff %s17 is used as BP
2038   //   v_restore = 1
2039   //   goto SinkMBB
2040 
2041   MachineBasicBlock *ThisMBB = MBB;
2042   MachineBasicBlock *MainMBB = MF->CreateMachineBasicBlock(BB);
2043   MachineBasicBlock *SinkMBB = MF->CreateMachineBasicBlock(BB);
2044   MachineBasicBlock *RestoreMBB = MF->CreateMachineBasicBlock(BB);
2045   MF->insert(I, MainMBB);
2046   MF->insert(I, SinkMBB);
2047   MF->push_back(RestoreMBB);
2048   RestoreMBB->setHasAddressTaken();
2049 
2050   // Transfer the remainder of BB and its successor edges to SinkMBB.
2051   SinkMBB->splice(SinkMBB->begin(), MBB,
2052                   std::next(MachineBasicBlock::iterator(MI)), MBB->end());
2053   SinkMBB->transferSuccessorsAndUpdatePHIs(MBB);
2054 
2055   // ThisMBB:
2056   Register LabelReg =
2057       prepareMBB(*MBB, MachineBasicBlock::iterator(MI), RestoreMBB, DL);
2058 
2059   // Store BP in buf[3] iff this function is using BP.
2060   const VEFrameLowering *TFI = Subtarget->getFrameLowering();
2061   if (TFI->hasBP(*MF)) {
2062     MachineInstrBuilder MIB = BuildMI(*MBB, MI, DL, TII->get(VE::STrii));
2063     MIB.addReg(BufReg);
2064     MIB.addImm(0);
2065     MIB.addImm(24);
2066     MIB.addReg(VE::SX17);
2067     MIB.setMemRefs(MMOs);
2068   }
2069 
2070   // Store IP in buf[1].
2071   MachineInstrBuilder MIB = BuildMI(*MBB, MI, DL, TII->get(VE::STrii));
2072   MIB.add(MI.getOperand(1)); // we can preserve the kill flags here.
2073   MIB.addImm(0);
2074   MIB.addImm(8);
2075   MIB.addReg(LabelReg, getKillRegState(true));
2076   MIB.setMemRefs(MMOs);
2077 
2078   // SP/FP are already stored in jmpbuf before `llvm.eh.sjlj.setjmp`.
2079 
2080   // Insert setup.
2081   MIB =
2082       BuildMI(*ThisMBB, MI, DL, TII->get(VE::EH_SjLj_Setup)).addMBB(RestoreMBB);
2083 
2084   const VERegisterInfo *RegInfo = Subtarget->getRegisterInfo();
2085   MIB.addRegMask(RegInfo->getNoPreservedMask());
2086   ThisMBB->addSuccessor(MainMBB);
2087   ThisMBB->addSuccessor(RestoreMBB);
2088 
2089   // MainMBB:
2090   BuildMI(MainMBB, DL, TII->get(VE::LEAzii), MainDestReg)
2091       .addImm(0)
2092       .addImm(0)
2093       .addImm(0);
2094   MainMBB->addSuccessor(SinkMBB);
2095 
2096   // SinkMBB:
2097   BuildMI(*SinkMBB, SinkMBB->begin(), DL, TII->get(VE::PHI), DstReg)
2098       .addReg(MainDestReg)
2099       .addMBB(MainMBB)
2100       .addReg(RestoreDestReg)
2101       .addMBB(RestoreMBB);
2102 
2103   // RestoreMBB:
2104   // Restore BP from buf[3] iff this function is using BP.  The address of
2105   // buf is in SX10.
2106   // FIXME: Better to not use SX10 here
2107   if (TFI->hasBP(*MF)) {
2108     MachineInstrBuilder MIB =
2109         BuildMI(RestoreMBB, DL, TII->get(VE::LDrii), VE::SX17);
2110     MIB.addReg(VE::SX10);
2111     MIB.addImm(0);
2112     MIB.addImm(24);
2113     MIB.setMemRefs(MMOs);
2114   }
2115   BuildMI(RestoreMBB, DL, TII->get(VE::LEAzii), RestoreDestReg)
2116       .addImm(0)
2117       .addImm(0)
2118       .addImm(1);
2119   BuildMI(RestoreMBB, DL, TII->get(VE::BRCFLa_t)).addMBB(SinkMBB);
2120   RestoreMBB->addSuccessor(SinkMBB);
2121 
2122   MI.eraseFromParent();
2123   return SinkMBB;
2124 }
2125 
2126 MachineBasicBlock *
2127 VETargetLowering::emitEHSjLjLongJmp(MachineInstr &MI,
2128                                     MachineBasicBlock *MBB) const {
2129   DebugLoc DL = MI.getDebugLoc();
2130   MachineFunction *MF = MBB->getParent();
2131   const TargetInstrInfo *TII = Subtarget->getInstrInfo();
2132   MachineRegisterInfo &MRI = MF->getRegInfo();
2133 
2134   // Memory Reference.
2135   SmallVector<MachineMemOperand *, 2> MMOs(MI.memoperands_begin(),
2136                                            MI.memoperands_end());
2137   Register BufReg = MI.getOperand(0).getReg();
2138 
2139   Register Tmp = MRI.createVirtualRegister(&VE::I64RegClass);
2140   // Since FP is only updated here but NOT referenced, it's treated as GPR.
2141   Register FP = VE::SX9;
2142   Register SP = VE::SX11;
2143 
2144   MachineInstrBuilder MIB;
2145 
2146   MachineBasicBlock *ThisMBB = MBB;
2147 
2148   // For `call @llvm.eh.sjlj.longjmp(buf)`, we generate following instructions.
2149   //
2150   // ThisMBB:
2151   //   %fp = load buf[0]
2152   //   %jmp = load buf[1]
2153   //   %s10 = buf        ; Store an address of buf to SX10 for RestoreMBB
2154   //   %sp = load buf[2] ; generated by llvm.eh.sjlj.setjmp.
2155   //   jmp %jmp
2156 
2157   // Reload FP.
2158   MIB = BuildMI(*ThisMBB, MI, DL, TII->get(VE::LDrii), FP);
2159   MIB.addReg(BufReg);
2160   MIB.addImm(0);
2161   MIB.addImm(0);
2162   MIB.setMemRefs(MMOs);
2163 
2164   // Reload IP.
2165   MIB = BuildMI(*ThisMBB, MI, DL, TII->get(VE::LDrii), Tmp);
2166   MIB.addReg(BufReg);
2167   MIB.addImm(0);
2168   MIB.addImm(8);
2169   MIB.setMemRefs(MMOs);
2170 
2171   // Copy BufReg to SX10 for later use in setjmp.
2172   // FIXME: Better to not use SX10 here
2173   BuildMI(*ThisMBB, MI, DL, TII->get(VE::ORri), VE::SX10)
2174       .addReg(BufReg)
2175       .addImm(0);
2176 
2177   // Reload SP.
2178   MIB = BuildMI(*ThisMBB, MI, DL, TII->get(VE::LDrii), SP);
2179   MIB.add(MI.getOperand(0)); // we can preserve the kill flags here.
2180   MIB.addImm(0);
2181   MIB.addImm(16);
2182   MIB.setMemRefs(MMOs);
2183 
2184   // Jump.
2185   BuildMI(*ThisMBB, MI, DL, TII->get(VE::BCFLari_t))
2186       .addReg(Tmp, getKillRegState(true))
2187       .addImm(0);
2188 
2189   MI.eraseFromParent();
2190   return ThisMBB;
2191 }
2192 
2193 MachineBasicBlock *
2194 VETargetLowering::emitSjLjDispatchBlock(MachineInstr &MI,
2195                                         MachineBasicBlock *BB) const {
2196   DebugLoc DL = MI.getDebugLoc();
2197   MachineFunction *MF = BB->getParent();
2198   MachineFrameInfo &MFI = MF->getFrameInfo();
2199   MachineRegisterInfo &MRI = MF->getRegInfo();
2200   const VEInstrInfo *TII = Subtarget->getInstrInfo();
2201   int FI = MFI.getFunctionContextIndex();
2202 
2203   // Get a mapping of the call site numbers to all of the landing pads they're
2204   // associated with.
2205   DenseMap<unsigned, SmallVector<MachineBasicBlock *, 2>> CallSiteNumToLPad;
2206   unsigned MaxCSNum = 0;
2207   for (auto &MBB : *MF) {
2208     if (!MBB.isEHPad())
2209       continue;
2210 
2211     MCSymbol *Sym = nullptr;
2212     for (const auto &MI : MBB) {
2213       if (MI.isDebugInstr())
2214         continue;
2215 
2216       assert(MI.isEHLabel() && "expected EH_LABEL");
2217       Sym = MI.getOperand(0).getMCSymbol();
2218       break;
2219     }
2220 
2221     if (!MF->hasCallSiteLandingPad(Sym))
2222       continue;
2223 
2224     for (unsigned CSI : MF->getCallSiteLandingPad(Sym)) {
2225       CallSiteNumToLPad[CSI].push_back(&MBB);
2226       MaxCSNum = std::max(MaxCSNum, CSI);
2227     }
2228   }
2229 
2230   // Get an ordered list of the machine basic blocks for the jump table.
2231   std::vector<MachineBasicBlock *> LPadList;
2232   SmallPtrSet<MachineBasicBlock *, 32> InvokeBBs;
2233   LPadList.reserve(CallSiteNumToLPad.size());
2234 
2235   for (unsigned CSI = 1; CSI <= MaxCSNum; ++CSI) {
2236     for (auto &LP : CallSiteNumToLPad[CSI]) {
2237       LPadList.push_back(LP);
2238       InvokeBBs.insert(LP->pred_begin(), LP->pred_end());
2239     }
2240   }
2241 
2242   assert(!LPadList.empty() &&
2243          "No landing pad destinations for the dispatch jump table!");
2244 
2245   // The %fn_context is allocated like below (from --print-after=sjljehprepare):
2246   //   %fn_context = alloca { i8*, i64, [4 x i64], i8*, i8*, [5 x i8*] }
2247   //
2248   // This `[5 x i8*]` is jmpbuf, so jmpbuf[1] is FI+72.
2249   // First `i64` is callsite, so callsite is FI+8.
2250   static const int OffsetIC = 72;
2251   static const int OffsetCS = 8;
2252 
2253   // Create the MBBs for the dispatch code like following:
2254   //
2255   // ThisMBB:
2256   //   Prepare DispatchBB address and store it to buf[1].
2257   //   ...
2258   //
2259   // DispatchBB:
2260   //   %s15 = GETGOT iff isPositionIndependent
2261   //   %callsite = load callsite
2262   //   brgt.l.t #size of callsites, %callsite, DispContBB
2263   //
2264   // TrapBB:
2265   //   Call abort.
2266   //
2267   // DispContBB:
2268   //   %breg = address of jump table
2269   //   %pc = load and calculate next pc from %breg and %callsite
2270   //   jmp %pc
2271 
2272   // Shove the dispatch's address into the return slot in the function context.
2273   MachineBasicBlock *DispatchBB = MF->CreateMachineBasicBlock();
2274   DispatchBB->setIsEHPad(true);
2275 
2276   // Trap BB will causes trap like `assert(0)`.
2277   MachineBasicBlock *TrapBB = MF->CreateMachineBasicBlock();
2278   DispatchBB->addSuccessor(TrapBB);
2279 
2280   MachineBasicBlock *DispContBB = MF->CreateMachineBasicBlock();
2281   DispatchBB->addSuccessor(DispContBB);
2282 
2283   // Insert MBBs.
2284   MF->push_back(DispatchBB);
2285   MF->push_back(DispContBB);
2286   MF->push_back(TrapBB);
2287 
2288   // Insert code to call abort in the TrapBB.
2289   Register Abort = prepareSymbol(*TrapBB, TrapBB->end(), "abort", DL,
2290                                  /* Local */ false, /* Call */ true);
2291   BuildMI(TrapBB, DL, TII->get(VE::BSICrii), VE::SX10)
2292       .addReg(Abort, getKillRegState(true))
2293       .addImm(0)
2294       .addImm(0);
2295 
2296   // Insert code into the entry block that creates and registers the function
2297   // context.
2298   setupEntryBlockForSjLj(MI, BB, DispatchBB, FI, OffsetIC);
2299 
2300   // Create the jump table and associated information
2301   unsigned JTE = getJumpTableEncoding();
2302   MachineJumpTableInfo *JTI = MF->getOrCreateJumpTableInfo(JTE);
2303   unsigned MJTI = JTI->createJumpTableIndex(LPadList);
2304 
2305   const VERegisterInfo &RI = TII->getRegisterInfo();
2306   // Add a register mask with no preserved registers.  This results in all
2307   // registers being marked as clobbered.
2308   BuildMI(DispatchBB, DL, TII->get(VE::NOP))
2309       .addRegMask(RI.getNoPreservedMask());
2310 
2311   if (isPositionIndependent()) {
2312     // Force to generate GETGOT, since current implementation doesn't store GOT
2313     // register.
2314     BuildMI(DispatchBB, DL, TII->get(VE::GETGOT), VE::SX15);
2315   }
2316 
2317   // IReg is used as an index in a memory operand and therefore can't be SP
2318   const TargetRegisterClass *RC = &VE::I64RegClass;
2319   Register IReg = MRI.createVirtualRegister(RC);
2320   addFrameReference(BuildMI(DispatchBB, DL, TII->get(VE::LDLZXrii), IReg), FI,
2321                     OffsetCS);
2322   if (LPadList.size() < 64) {
2323     BuildMI(DispatchBB, DL, TII->get(VE::BRCFLir_t))
2324         .addImm(VECC::CC_ILE)
2325         .addImm(LPadList.size())
2326         .addReg(IReg)
2327         .addMBB(TrapBB);
2328   } else {
2329     assert(LPadList.size() <= 0x7FFFFFFF && "Too large Landing Pad!");
2330     Register TmpReg = MRI.createVirtualRegister(RC);
2331     BuildMI(DispatchBB, DL, TII->get(VE::LEAzii), TmpReg)
2332         .addImm(0)
2333         .addImm(0)
2334         .addImm(LPadList.size());
2335     BuildMI(DispatchBB, DL, TII->get(VE::BRCFLrr_t))
2336         .addImm(VECC::CC_ILE)
2337         .addReg(TmpReg, getKillRegState(true))
2338         .addReg(IReg)
2339         .addMBB(TrapBB);
2340   }
2341 
2342   Register BReg = MRI.createVirtualRegister(RC);
2343   Register Tmp1 = MRI.createVirtualRegister(RC);
2344   Register Tmp2 = MRI.createVirtualRegister(RC);
2345 
2346   if (isPositionIndependent()) {
2347     // Create following instructions for local linkage PIC code.
2348     //     lea    %Tmp1, .LJTI0_0@gotoff_lo
2349     //     and    %Tmp2, %Tmp1, (32)0
2350     //     lea.sl %BReg, .LJTI0_0@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
2351     BuildMI(DispContBB, DL, TII->get(VE::LEAzii), Tmp1)
2352         .addImm(0)
2353         .addImm(0)
2354         .addJumpTableIndex(MJTI, VEMCExpr::VK_VE_GOTOFF_LO32);
2355     BuildMI(DispContBB, DL, TII->get(VE::ANDrm), Tmp2)
2356         .addReg(Tmp1, getKillRegState(true))
2357         .addImm(M0(32));
2358     BuildMI(DispContBB, DL, TII->get(VE::LEASLrri), BReg)
2359         .addReg(VE::SX15)
2360         .addReg(Tmp2, getKillRegState(true))
2361         .addJumpTableIndex(MJTI, VEMCExpr::VK_VE_GOTOFF_HI32);
2362   } else {
2363     // Create following instructions for non-PIC code.
2364     //     lea     %Tmp1, .LJTI0_0@lo
2365     //     and     %Tmp2, %Tmp1, (32)0
2366     //     lea.sl  %BReg, .LJTI0_0@hi(%Tmp2)
2367     BuildMI(DispContBB, DL, TII->get(VE::LEAzii), Tmp1)
2368         .addImm(0)
2369         .addImm(0)
2370         .addJumpTableIndex(MJTI, VEMCExpr::VK_VE_LO32);
2371     BuildMI(DispContBB, DL, TII->get(VE::ANDrm), Tmp2)
2372         .addReg(Tmp1, getKillRegState(true))
2373         .addImm(M0(32));
2374     BuildMI(DispContBB, DL, TII->get(VE::LEASLrii), BReg)
2375         .addReg(Tmp2, getKillRegState(true))
2376         .addImm(0)
2377         .addJumpTableIndex(MJTI, VEMCExpr::VK_VE_HI32);
2378   }
2379 
2380   switch (JTE) {
2381   case MachineJumpTableInfo::EK_BlockAddress: {
2382     // Generate simple block address code for no-PIC model.
2383     //     sll %Tmp1, %IReg, 3
2384     //     lds %TReg, 0(%Tmp1, %BReg)
2385     //     bcfla %TReg
2386 
2387     Register TReg = MRI.createVirtualRegister(RC);
2388     Register Tmp1 = MRI.createVirtualRegister(RC);
2389 
2390     BuildMI(DispContBB, DL, TII->get(VE::SLLri), Tmp1)
2391         .addReg(IReg, getKillRegState(true))
2392         .addImm(3);
2393     BuildMI(DispContBB, DL, TII->get(VE::LDrri), TReg)
2394         .addReg(BReg, getKillRegState(true))
2395         .addReg(Tmp1, getKillRegState(true))
2396         .addImm(0);
2397     BuildMI(DispContBB, DL, TII->get(VE::BCFLari_t))
2398         .addReg(TReg, getKillRegState(true))
2399         .addImm(0);
2400     break;
2401   }
2402   case MachineJumpTableInfo::EK_Custom32: {
2403     // Generate block address code using differences from the function pointer
2404     // for PIC model.
2405     //     sll %Tmp1, %IReg, 2
2406     //     ldl.zx %OReg, 0(%Tmp1, %BReg)
2407     //     Prepare function address in BReg2.
2408     //     adds.l %TReg, %BReg2, %OReg
2409     //     bcfla %TReg
2410 
2411     assert(isPositionIndependent());
2412     Register OReg = MRI.createVirtualRegister(RC);
2413     Register TReg = MRI.createVirtualRegister(RC);
2414     Register Tmp1 = MRI.createVirtualRegister(RC);
2415 
2416     BuildMI(DispContBB, DL, TII->get(VE::SLLri), Tmp1)
2417         .addReg(IReg, getKillRegState(true))
2418         .addImm(2);
2419     BuildMI(DispContBB, DL, TII->get(VE::LDLZXrri), OReg)
2420         .addReg(BReg, getKillRegState(true))
2421         .addReg(Tmp1, getKillRegState(true))
2422         .addImm(0);
2423     Register BReg2 =
2424         prepareSymbol(*DispContBB, DispContBB->end(),
2425                       DispContBB->getParent()->getName(), DL, /* Local */ true);
2426     BuildMI(DispContBB, DL, TII->get(VE::ADDSLrr), TReg)
2427         .addReg(OReg, getKillRegState(true))
2428         .addReg(BReg2, getKillRegState(true));
2429     BuildMI(DispContBB, DL, TII->get(VE::BCFLari_t))
2430         .addReg(TReg, getKillRegState(true))
2431         .addImm(0);
2432     break;
2433   }
2434   default:
2435     llvm_unreachable("Unexpected jump table encoding");
2436   }
2437 
2438   // Add the jump table entries as successors to the MBB.
2439   SmallPtrSet<MachineBasicBlock *, 8> SeenMBBs;
2440   for (auto &LP : LPadList)
2441     if (SeenMBBs.insert(LP).second)
2442       DispContBB->addSuccessor(LP);
2443 
2444   // N.B. the order the invoke BBs are processed in doesn't matter here.
2445   SmallVector<MachineBasicBlock *, 64> MBBLPads;
2446   const MCPhysReg *SavedRegs = MF->getRegInfo().getCalleeSavedRegs();
2447   for (MachineBasicBlock *MBB : InvokeBBs) {
2448     // Remove the landing pad successor from the invoke block and replace it
2449     // with the new dispatch block.
2450     // Keep a copy of Successors since it's modified inside the loop.
2451     SmallVector<MachineBasicBlock *, 8> Successors(MBB->succ_rbegin(),
2452                                                    MBB->succ_rend());
2453     // FIXME: Avoid quadratic complexity.
2454     for (auto MBBS : Successors) {
2455       if (MBBS->isEHPad()) {
2456         MBB->removeSuccessor(MBBS);
2457         MBBLPads.push_back(MBBS);
2458       }
2459     }
2460 
2461     MBB->addSuccessor(DispatchBB);
2462 
2463     // Find the invoke call and mark all of the callee-saved registers as
2464     // 'implicit defined' so that they're spilled.  This prevents code from
2465     // moving instructions to before the EH block, where they will never be
2466     // executed.
2467     for (auto &II : reverse(*MBB)) {
2468       if (!II.isCall())
2469         continue;
2470 
2471       DenseMap<Register, bool> DefRegs;
2472       for (auto &MOp : II.operands())
2473         if (MOp.isReg())
2474           DefRegs[MOp.getReg()] = true;
2475 
2476       MachineInstrBuilder MIB(*MF, &II);
2477       for (unsigned RI = 0; SavedRegs[RI]; ++RI) {
2478         Register Reg = SavedRegs[RI];
2479         if (!DefRegs[Reg])
2480           MIB.addReg(Reg, RegState::ImplicitDefine | RegState::Dead);
2481       }
2482 
2483       break;
2484     }
2485   }
2486 
2487   // Mark all former landing pads as non-landing pads.  The dispatch is the only
2488   // landing pad now.
2489   for (auto &LP : MBBLPads)
2490     LP->setIsEHPad(false);
2491 
2492   // The instruction is gone now.
2493   MI.eraseFromParent();
2494   return BB;
2495 }
2496 
2497 MachineBasicBlock *
2498 VETargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
2499                                               MachineBasicBlock *BB) const {
2500   switch (MI.getOpcode()) {
2501   default:
2502     llvm_unreachable("Unknown Custom Instruction!");
2503   case VE::EH_SjLj_LongJmp:
2504     return emitEHSjLjLongJmp(MI, BB);
2505   case VE::EH_SjLj_SetJmp:
2506     return emitEHSjLjSetJmp(MI, BB);
2507   case VE::EH_SjLj_Setup_Dispatch:
2508     return emitSjLjDispatchBlock(MI, BB);
2509   }
2510 }
2511 
2512 static bool isI32Insn(const SDNode *User, const SDNode *N) {
2513   switch (User->getOpcode()) {
2514   default:
2515     return false;
2516   case ISD::ADD:
2517   case ISD::SUB:
2518   case ISD::MUL:
2519   case ISD::SDIV:
2520   case ISD::UDIV:
2521   case ISD::SETCC:
2522   case ISD::SMIN:
2523   case ISD::SMAX:
2524   case ISD::SHL:
2525   case ISD::SRA:
2526   case ISD::BSWAP:
2527   case ISD::SINT_TO_FP:
2528   case ISD::UINT_TO_FP:
2529   case ISD::BR_CC:
2530   case ISD::BITCAST:
2531   case ISD::ATOMIC_CMP_SWAP:
2532   case ISD::ATOMIC_SWAP:
2533     return true;
2534   case ISD::SRL:
2535     if (N->getOperand(0).getOpcode() != ISD::SRL)
2536       return true;
2537     // (srl (trunc (srl ...))) may be optimized by combining srl, so
2538     // doesn't optimize trunc now.
2539     return false;
2540   case ISD::SELECT_CC:
2541     if (User->getOperand(2).getNode() != N &&
2542         User->getOperand(3).getNode() != N)
2543       return true;
2544     LLVM_FALLTHROUGH;
2545   case ISD::AND:
2546   case ISD::OR:
2547   case ISD::XOR:
2548   case ISD::SELECT:
2549   case ISD::CopyToReg:
2550     // Check all use of selections, bit operations, and copies.  If all of them
2551     // are safe, optimize truncate to extract_subreg.
2552     for (const SDNode *U : User->uses()) {
2553       switch (U->getOpcode()) {
2554       default:
2555         // If the use is an instruction which treats the source operand as i32,
2556         // it is safe to avoid truncate here.
2557         if (isI32Insn(U, N))
2558           continue;
2559         break;
2560       case ISD::ANY_EXTEND:
2561       case ISD::SIGN_EXTEND:
2562       case ISD::ZERO_EXTEND: {
2563         // Special optimizations to the combination of ext and trunc.
2564         // (ext ... (select ... (trunc ...))) is safe to avoid truncate here
2565         // since this truncate instruction clears higher 32 bits which is filled
2566         // by one of ext instructions later.
2567         assert(N->getValueType(0) == MVT::i32 &&
2568                "find truncate to not i32 integer");
2569         if (User->getOpcode() == ISD::SELECT_CC ||
2570             User->getOpcode() == ISD::SELECT)
2571           continue;
2572         break;
2573       }
2574       }
2575       return false;
2576     }
2577     return true;
2578   }
2579 }
2580 
2581 // Optimize TRUNCATE in DAG combining.  Optimizing it in CUSTOM lower is
2582 // sometime too early.  Optimizing it in DAG pattern matching in VEInstrInfo.td
2583 // is sometime too late.  So, doing it at here.
2584 SDValue VETargetLowering::combineTRUNCATE(SDNode *N,
2585                                           DAGCombinerInfo &DCI) const {
2586   assert(N->getOpcode() == ISD::TRUNCATE &&
2587          "Should be called with a TRUNCATE node");
2588 
2589   SelectionDAG &DAG = DCI.DAG;
2590   SDLoc DL(N);
2591   EVT VT = N->getValueType(0);
2592 
2593   // We prefer to do this when all types are legal.
2594   if (!DCI.isAfterLegalizeDAG())
2595     return SDValue();
2596 
2597   // Skip combine TRUNCATE atm if the operand of TRUNCATE might be a constant.
2598   if (N->getOperand(0)->getOpcode() == ISD::SELECT_CC &&
2599       isa<ConstantSDNode>(N->getOperand(0)->getOperand(0)) &&
2600       isa<ConstantSDNode>(N->getOperand(0)->getOperand(1)))
2601     return SDValue();
2602 
2603   // Check all use of this TRUNCATE.
2604   for (const SDNode *User : N->uses()) {
2605     // Make sure that we're not going to replace TRUNCATE for non i32
2606     // instructions.
2607     //
2608     // FIXME: Although we could sometimes handle this, and it does occur in
2609     // practice that one of the condition inputs to the select is also one of
2610     // the outputs, we currently can't deal with this.
2611     if (isI32Insn(User, N))
2612       continue;
2613 
2614     return SDValue();
2615   }
2616 
2617   SDValue SubI32 = DAG.getTargetConstant(VE::sub_i32, DL, MVT::i32);
2618   return SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, VT,
2619                                     N->getOperand(0), SubI32),
2620                  0);
2621 }
2622 
2623 SDValue VETargetLowering::PerformDAGCombine(SDNode *N,
2624                                             DAGCombinerInfo &DCI) const {
2625   switch (N->getOpcode()) {
2626   default:
2627     break;
2628   case ISD::TRUNCATE:
2629     return combineTRUNCATE(N, DCI);
2630   }
2631 
2632   return SDValue();
2633 }
2634 
2635 //===----------------------------------------------------------------------===//
2636 // VE Inline Assembly Support
2637 //===----------------------------------------------------------------------===//
2638 
2639 VETargetLowering::ConstraintType
2640 VETargetLowering::getConstraintType(StringRef Constraint) const {
2641   if (Constraint.size() == 1) {
2642     switch (Constraint[0]) {
2643     default:
2644       break;
2645     case 'v': // vector registers
2646       return C_RegisterClass;
2647     }
2648   }
2649   return TargetLowering::getConstraintType(Constraint);
2650 }
2651 
2652 std::pair<unsigned, const TargetRegisterClass *>
2653 VETargetLowering::getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI,
2654                                                StringRef Constraint,
2655                                                MVT VT) const {
2656   const TargetRegisterClass *RC = nullptr;
2657   if (Constraint.size() == 1) {
2658     switch (Constraint[0]) {
2659     default:
2660       return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
2661     case 'r':
2662       RC = &VE::I64RegClass;
2663       break;
2664     case 'v':
2665       RC = &VE::V64RegClass;
2666       break;
2667     }
2668     return std::make_pair(0U, RC);
2669   }
2670 
2671   return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
2672 }
2673 
2674 //===----------------------------------------------------------------------===//
2675 // VE Target Optimization Support
2676 //===----------------------------------------------------------------------===//
2677 
2678 unsigned VETargetLowering::getMinimumJumpTableEntries() const {
2679   // Specify 8 for PIC model to relieve the impact of PIC load instructions.
2680   if (isJumpTableRelative())
2681     return 8;
2682 
2683   return TargetLowering::getMinimumJumpTableEntries();
2684 }
2685 
2686 bool VETargetLowering::hasAndNot(SDValue Y) const {
2687   EVT VT = Y.getValueType();
2688 
2689   // VE doesn't have vector and not instruction.
2690   if (VT.isVector())
2691     return false;
2692 
2693   // VE allows different immediate values for X and Y where ~X & Y.
2694   // Only simm7 works for X, and only mimm works for Y on VE.  However, this
2695   // function is used to check whether an immediate value is OK for and-not
2696   // instruction as both X and Y.  Generating additional instruction to
2697   // retrieve an immediate value is no good since the purpose of this
2698   // function is to convert a series of 3 instructions to another series of
2699   // 3 instructions with better parallelism.  Therefore, we return false
2700   // for all immediate values now.
2701   // FIXME: Change hasAndNot function to have two operands to make it work
2702   //        correctly with Aurora VE.
2703   if (isa<ConstantSDNode>(Y))
2704     return false;
2705 
2706   // It's ok for generic registers.
2707   return true;
2708 }
2709 
2710 SDValue VETargetLowering::lowerEXTRACT_VECTOR_ELT(SDValue Op,
2711                                                   SelectionDAG &DAG) const {
2712   assert(Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT && "Unknown opcode!");
2713   MVT VT = Op.getOperand(0).getSimpleValueType();
2714 
2715   // Special treatment for packed V64 types.
2716   assert(VT == MVT::v512i32 || VT == MVT::v512f32);
2717   (void)VT;
2718   // Example of codes:
2719   //   %packed_v = extractelt %vr, %idx / 2
2720   //   %v = %packed_v >> (%idx % 2 * 32)
2721   //   %res = %v & 0xffffffff
2722 
2723   SDValue Vec = Op.getOperand(0);
2724   SDValue Idx = Op.getOperand(1);
2725   SDLoc DL(Op);
2726   SDValue Result = Op;
2727   if (false /* Idx->isConstant() */) {
2728     // TODO: optimized implementation using constant values
2729   } else {
2730     SDValue Const1 = DAG.getConstant(1, DL, MVT::i64);
2731     SDValue HalfIdx = DAG.getNode(ISD::SRL, DL, MVT::i64, {Idx, Const1});
2732     SDValue PackedElt =
2733         SDValue(DAG.getMachineNode(VE::LVSvr, DL, MVT::i64, {Vec, HalfIdx}), 0);
2734     SDValue AndIdx = DAG.getNode(ISD::AND, DL, MVT::i64, {Idx, Const1});
2735     SDValue Shift = DAG.getNode(ISD::XOR, DL, MVT::i64, {AndIdx, Const1});
2736     SDValue Const5 = DAG.getConstant(5, DL, MVT::i64);
2737     Shift = DAG.getNode(ISD::SHL, DL, MVT::i64, {Shift, Const5});
2738     PackedElt = DAG.getNode(ISD::SRL, DL, MVT::i64, {PackedElt, Shift});
2739     SDValue Mask = DAG.getConstant(0xFFFFFFFFL, DL, MVT::i64);
2740     PackedElt = DAG.getNode(ISD::AND, DL, MVT::i64, {PackedElt, Mask});
2741     SDValue SubI32 = DAG.getTargetConstant(VE::sub_i32, DL, MVT::i32);
2742     Result = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
2743                                         MVT::i32, PackedElt, SubI32),
2744                      0);
2745 
2746     if (Op.getSimpleValueType() == MVT::f32) {
2747       Result = DAG.getBitcast(MVT::f32, Result);
2748     } else {
2749       assert(Op.getSimpleValueType() == MVT::i32);
2750     }
2751   }
2752   return Result;
2753 }
2754 
2755 SDValue VETargetLowering::lowerINSERT_VECTOR_ELT(SDValue Op,
2756                                                  SelectionDAG &DAG) const {
2757   assert(Op.getOpcode() == ISD::INSERT_VECTOR_ELT && "Unknown opcode!");
2758   MVT VT = Op.getOperand(0).getSimpleValueType();
2759 
2760   // Special treatment for packed V64 types.
2761   assert(VT == MVT::v512i32 || VT == MVT::v512f32);
2762   (void)VT;
2763   // The v512i32 and v512f32 starts from upper bits (0..31).  This "upper
2764   // bits" required `val << 32` from C implementation's point of view.
2765   //
2766   // Example of codes:
2767   //   %packed_elt = extractelt %vr, (%idx >> 1)
2768   //   %shift = ((%idx & 1) ^ 1) << 5
2769   //   %packed_elt &= 0xffffffff00000000 >> shift
2770   //   %packed_elt |= (zext %val) << shift
2771   //   %vr = insertelt %vr, %packed_elt, (%idx >> 1)
2772 
2773   SDLoc DL(Op);
2774   SDValue Vec = Op.getOperand(0);
2775   SDValue Val = Op.getOperand(1);
2776   SDValue Idx = Op.getOperand(2);
2777   if (Idx.getSimpleValueType() == MVT::i32)
2778     Idx = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i64, Idx);
2779   if (Val.getSimpleValueType() == MVT::f32)
2780     Val = DAG.getBitcast(MVT::i32, Val);
2781   assert(Val.getSimpleValueType() == MVT::i32);
2782   Val = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i64, Val);
2783 
2784   SDValue Result = Op;
2785   if (false /* Idx->isConstant()*/) {
2786     // TODO: optimized implementation using constant values
2787   } else {
2788     SDValue Const1 = DAG.getConstant(1, DL, MVT::i64);
2789     SDValue HalfIdx = DAG.getNode(ISD::SRL, DL, MVT::i64, {Idx, Const1});
2790     SDValue PackedElt =
2791         SDValue(DAG.getMachineNode(VE::LVSvr, DL, MVT::i64, {Vec, HalfIdx}), 0);
2792     SDValue AndIdx = DAG.getNode(ISD::AND, DL, MVT::i64, {Idx, Const1});
2793     SDValue Shift = DAG.getNode(ISD::XOR, DL, MVT::i64, {AndIdx, Const1});
2794     SDValue Const5 = DAG.getConstant(5, DL, MVT::i64);
2795     Shift = DAG.getNode(ISD::SHL, DL, MVT::i64, {Shift, Const5});
2796     SDValue Mask = DAG.getConstant(0xFFFFFFFF00000000L, DL, MVT::i64);
2797     Mask = DAG.getNode(ISD::SRL, DL, MVT::i64, {Mask, Shift});
2798     PackedElt = DAG.getNode(ISD::AND, DL, MVT::i64, {PackedElt, Mask});
2799     Val = DAG.getNode(ISD::SHL, DL, MVT::i64, {Val, Shift});
2800     PackedElt = DAG.getNode(ISD::OR, DL, MVT::i64, {PackedElt, Val});
2801     Result =
2802         SDValue(DAG.getMachineNode(VE::LSVrr_v, DL, Vec.getSimpleValueType(),
2803                                    {HalfIdx, PackedElt, Vec}),
2804                 0);
2805   }
2806   return Result;
2807 }
2808