1 //===-- VEISelLowering.cpp - VE DAG Lowering Implementation ---------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This file implements the interfaces that VE uses to lower LLVM code into a 10 // selection DAG. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "VEISelLowering.h" 15 #include "MCTargetDesc/VEMCExpr.h" 16 #include "VEMachineFunctionInfo.h" 17 #include "VERegisterInfo.h" 18 #include "VETargetMachine.h" 19 #include "llvm/ADT/StringSwitch.h" 20 #include "llvm/CodeGen/CallingConvLower.h" 21 #include "llvm/CodeGen/MachineFrameInfo.h" 22 #include "llvm/CodeGen/MachineFunction.h" 23 #include "llvm/CodeGen/MachineInstrBuilder.h" 24 #include "llvm/CodeGen/MachineModuleInfo.h" 25 #include "llvm/CodeGen/MachineRegisterInfo.h" 26 #include "llvm/CodeGen/SelectionDAG.h" 27 #include "llvm/CodeGen/TargetLoweringObjectFileImpl.h" 28 #include "llvm/IR/DerivedTypes.h" 29 #include "llvm/IR/Function.h" 30 #include "llvm/IR/Module.h" 31 #include "llvm/Support/ErrorHandling.h" 32 #include "llvm/Support/KnownBits.h" 33 using namespace llvm; 34 35 #define DEBUG_TYPE "ve-lower" 36 37 //===----------------------------------------------------------------------===// 38 // Calling Convention Implementation 39 //===----------------------------------------------------------------------===// 40 41 #include "VEGenCallingConv.inc" 42 43 bool VETargetLowering::CanLowerReturn( 44 CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg, 45 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context) const { 46 CCAssignFn *RetCC = RetCC_VE; 47 SmallVector<CCValAssign, 16> RVLocs; 48 CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context); 49 return CCInfo.CheckReturn(Outs, RetCC); 50 } 51 52 SDValue 53 VETargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv, 54 bool IsVarArg, 55 const SmallVectorImpl<ISD::OutputArg> &Outs, 56 const SmallVectorImpl<SDValue> &OutVals, 57 const SDLoc &DL, SelectionDAG &DAG) const { 58 // CCValAssign - represent the assignment of the return value to locations. 59 SmallVector<CCValAssign, 16> RVLocs; 60 61 // CCState - Info about the registers and stack slot. 62 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), RVLocs, 63 *DAG.getContext()); 64 65 // Analyze return values. 66 CCInfo.AnalyzeReturn(Outs, RetCC_VE); 67 68 SDValue Flag; 69 SmallVector<SDValue, 4> RetOps(1, Chain); 70 71 // Copy the result values into the output registers. 72 for (unsigned i = 0; i != RVLocs.size(); ++i) { 73 CCValAssign &VA = RVLocs[i]; 74 assert(VA.isRegLoc() && "Can only return in registers!"); 75 SDValue OutVal = OutVals[i]; 76 77 // Integer return values must be sign or zero extended by the callee. 78 switch (VA.getLocInfo()) { 79 case CCValAssign::Full: 80 break; 81 case CCValAssign::SExt: 82 OutVal = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), OutVal); 83 break; 84 case CCValAssign::ZExt: 85 OutVal = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), OutVal); 86 break; 87 case CCValAssign::AExt: 88 OutVal = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), OutVal); 89 break; 90 case CCValAssign::BCvt: { 91 // Convert a float return value to i64 with padding. 92 // 63 31 0 93 // +------+------+ 94 // | float| 0 | 95 // +------+------+ 96 assert(VA.getLocVT() == MVT::i64); 97 assert(VA.getValVT() == MVT::f32); 98 SDValue Undef = SDValue( 99 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0); 100 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 101 OutVal = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, 102 MVT::i64, Undef, OutVal, Sub_f32), 103 0); 104 break; 105 } 106 default: 107 llvm_unreachable("Unknown loc info!"); 108 } 109 110 assert(!VA.needsCustom() && "Unexpected custom lowering"); 111 112 Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), OutVal, Flag); 113 114 // Guarantee that all emitted copies are stuck together with flags. 115 Flag = Chain.getValue(1); 116 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT())); 117 } 118 119 RetOps[0] = Chain; // Update chain. 120 121 // Add the flag if we have it. 122 if (Flag.getNode()) 123 RetOps.push_back(Flag); 124 125 return DAG.getNode(VEISD::RET_FLAG, DL, MVT::Other, RetOps); 126 } 127 128 SDValue VETargetLowering::LowerFormalArguments( 129 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, 130 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL, 131 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const { 132 MachineFunction &MF = DAG.getMachineFunction(); 133 134 // Get the base offset of the incoming arguments stack space. 135 unsigned ArgsBaseOffset = 176; 136 // Get the size of the preserved arguments area 137 unsigned ArgsPreserved = 64; 138 139 // Analyze arguments according to CC_VE. 140 SmallVector<CCValAssign, 16> ArgLocs; 141 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), ArgLocs, 142 *DAG.getContext()); 143 // Allocate the preserved area first. 144 CCInfo.AllocateStack(ArgsPreserved, Align(8)); 145 // We already allocated the preserved area, so the stack offset computed 146 // by CC_VE would be correct now. 147 CCInfo.AnalyzeFormalArguments(Ins, CC_VE); 148 149 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { 150 CCValAssign &VA = ArgLocs[i]; 151 if (VA.isRegLoc()) { 152 // This argument is passed in a register. 153 // All integer register arguments are promoted by the caller to i64. 154 155 // Create a virtual register for the promoted live-in value. 156 unsigned VReg = 157 MF.addLiveIn(VA.getLocReg(), getRegClassFor(VA.getLocVT())); 158 SDValue Arg = DAG.getCopyFromReg(Chain, DL, VReg, VA.getLocVT()); 159 160 // Get the high bits for i32 struct elements. 161 if (VA.getValVT() == MVT::i32 && VA.needsCustom()) 162 Arg = DAG.getNode(ISD::SRL, DL, VA.getLocVT(), Arg, 163 DAG.getConstant(32, DL, MVT::i32)); 164 165 // The caller promoted the argument, so insert an Assert?ext SDNode so we 166 // won't promote the value again in this function. 167 switch (VA.getLocInfo()) { 168 case CCValAssign::SExt: 169 Arg = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), Arg, 170 DAG.getValueType(VA.getValVT())); 171 break; 172 case CCValAssign::ZExt: 173 Arg = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), Arg, 174 DAG.getValueType(VA.getValVT())); 175 break; 176 case CCValAssign::BCvt: { 177 // Extract a float argument from i64 with padding. 178 // 63 31 0 179 // +------+------+ 180 // | float| 0 | 181 // +------+------+ 182 assert(VA.getLocVT() == MVT::i64); 183 assert(VA.getValVT() == MVT::f32); 184 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 185 Arg = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, 186 MVT::f32, Arg, Sub_f32), 187 0); 188 break; 189 } 190 default: 191 break; 192 } 193 194 // Truncate the register down to the argument type. 195 if (VA.isExtInLoc()) 196 Arg = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Arg); 197 198 InVals.push_back(Arg); 199 continue; 200 } 201 202 // The registers are exhausted. This argument was passed on the stack. 203 assert(VA.isMemLoc()); 204 // The CC_VE_Full/Half functions compute stack offsets relative to the 205 // beginning of the arguments area at %fp+176. 206 unsigned Offset = VA.getLocMemOffset() + ArgsBaseOffset; 207 unsigned ValSize = VA.getValVT().getSizeInBits() / 8; 208 209 // Adjust offset for a float argument by adding 4 since the argument is 210 // stored in 8 bytes buffer with offset like below. LLVM generates 211 // 4 bytes load instruction, so need to adjust offset here. This 212 // adjustment is required in only LowerFormalArguments. In LowerCall, 213 // a float argument is converted to i64 first, and stored as 8 bytes 214 // data, which is required by ABI, so no need for adjustment. 215 // 0 4 216 // +------+------+ 217 // | empty| float| 218 // +------+------+ 219 if (VA.getValVT() == MVT::f32) 220 Offset += 4; 221 222 int FI = MF.getFrameInfo().CreateFixedObject(ValSize, Offset, true); 223 InVals.push_back( 224 DAG.getLoad(VA.getValVT(), DL, Chain, 225 DAG.getFrameIndex(FI, getPointerTy(MF.getDataLayout())), 226 MachinePointerInfo::getFixedStack(MF, FI))); 227 } 228 229 if (!IsVarArg) 230 return Chain; 231 232 // This function takes variable arguments, some of which may have been passed 233 // in registers %s0-%s8. 234 // 235 // The va_start intrinsic needs to know the offset to the first variable 236 // argument. 237 // TODO: need to calculate offset correctly once we support f128. 238 unsigned ArgOffset = ArgLocs.size() * 8; 239 VEMachineFunctionInfo *FuncInfo = MF.getInfo<VEMachineFunctionInfo>(); 240 // Skip the 176 bytes of register save area. 241 FuncInfo->setVarArgsFrameOffset(ArgOffset + ArgsBaseOffset); 242 243 return Chain; 244 } 245 246 // FIXME? Maybe this could be a TableGen attribute on some registers and 247 // this table could be generated automatically from RegInfo. 248 Register VETargetLowering::getRegisterByName(const char *RegName, LLT VT, 249 const MachineFunction &MF) const { 250 Register Reg = StringSwitch<Register>(RegName) 251 .Case("sp", VE::SX11) // Stack pointer 252 .Case("fp", VE::SX9) // Frame pointer 253 .Case("sl", VE::SX8) // Stack limit 254 .Case("lr", VE::SX10) // Link register 255 .Case("tp", VE::SX14) // Thread pointer 256 .Case("outer", VE::SX12) // Outer regiser 257 .Case("info", VE::SX17) // Info area register 258 .Case("got", VE::SX15) // Global offset table register 259 .Case("plt", VE::SX16) // Procedure linkage table register 260 .Default(0); 261 262 if (Reg) 263 return Reg; 264 265 report_fatal_error("Invalid register name global variable"); 266 } 267 268 //===----------------------------------------------------------------------===// 269 // TargetLowering Implementation 270 //===----------------------------------------------------------------------===// 271 272 SDValue VETargetLowering::LowerCall(TargetLowering::CallLoweringInfo &CLI, 273 SmallVectorImpl<SDValue> &InVals) const { 274 SelectionDAG &DAG = CLI.DAG; 275 SDLoc DL = CLI.DL; 276 SDValue Chain = CLI.Chain; 277 auto PtrVT = getPointerTy(DAG.getDataLayout()); 278 279 // VE target does not yet support tail call optimization. 280 CLI.IsTailCall = false; 281 282 // Get the base offset of the outgoing arguments stack space. 283 unsigned ArgsBaseOffset = 176; 284 // Get the size of the preserved arguments area 285 unsigned ArgsPreserved = 8 * 8u; 286 287 // Analyze operands of the call, assigning locations to each operand. 288 SmallVector<CCValAssign, 16> ArgLocs; 289 CCState CCInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), ArgLocs, 290 *DAG.getContext()); 291 // Allocate the preserved area first. 292 CCInfo.AllocateStack(ArgsPreserved, Align(8)); 293 // We already allocated the preserved area, so the stack offset computed 294 // by CC_VE would be correct now. 295 CCInfo.AnalyzeCallOperands(CLI.Outs, CC_VE); 296 297 // VE requires to use both register and stack for varargs or no-prototyped 298 // functions. 299 bool UseBoth = CLI.IsVarArg; 300 301 // Analyze operands again if it is required to store BOTH. 302 SmallVector<CCValAssign, 16> ArgLocs2; 303 CCState CCInfo2(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), 304 ArgLocs2, *DAG.getContext()); 305 if (UseBoth) 306 CCInfo2.AnalyzeCallOperands(CLI.Outs, CC_VE2); 307 308 // Get the size of the outgoing arguments stack space requirement. 309 unsigned ArgsSize = CCInfo.getNextStackOffset(); 310 311 // Keep stack frames 16-byte aligned. 312 ArgsSize = alignTo(ArgsSize, 16); 313 314 // Adjust the stack pointer to make room for the arguments. 315 // FIXME: Use hasReservedCallFrame to avoid %sp adjustments around all calls 316 // with more than 6 arguments. 317 Chain = DAG.getCALLSEQ_START(Chain, ArgsSize, 0, DL); 318 319 // Collect the set of registers to pass to the function and their values. 320 // This will be emitted as a sequence of CopyToReg nodes glued to the call 321 // instruction. 322 SmallVector<std::pair<unsigned, SDValue>, 8> RegsToPass; 323 324 // Collect chains from all the memory opeations that copy arguments to the 325 // stack. They must follow the stack pointer adjustment above and precede the 326 // call instruction itself. 327 SmallVector<SDValue, 8> MemOpChains; 328 329 // VE needs to get address of callee function in a register 330 // So, prepare to copy it to SX12 here. 331 332 // If the callee is a GlobalAddress node (quite common, every direct call is) 333 // turn it into a TargetGlobalAddress node so that legalize doesn't hack it. 334 // Likewise ExternalSymbol -> TargetExternalSymbol. 335 SDValue Callee = CLI.Callee; 336 337 bool IsPICCall = isPositionIndependent(); 338 339 // PC-relative references to external symbols should go through $stub. 340 // If so, we need to prepare GlobalBaseReg first. 341 const TargetMachine &TM = DAG.getTarget(); 342 const Module *Mod = DAG.getMachineFunction().getFunction().getParent(); 343 const GlobalValue *GV = nullptr; 344 auto *CalleeG = dyn_cast<GlobalAddressSDNode>(Callee); 345 if (CalleeG) 346 GV = CalleeG->getGlobal(); 347 bool Local = TM.shouldAssumeDSOLocal(*Mod, GV); 348 bool UsePlt = !Local; 349 MachineFunction &MF = DAG.getMachineFunction(); 350 351 // Turn GlobalAddress/ExternalSymbol node into a value node 352 // containing the address of them here. 353 if (CalleeG) { 354 if (IsPICCall) { 355 if (UsePlt) 356 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF); 357 Callee = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, 0); 358 Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee); 359 } else { 360 Callee = 361 makeHiLoPair(Callee, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG); 362 } 363 } else if (ExternalSymbolSDNode *E = dyn_cast<ExternalSymbolSDNode>(Callee)) { 364 if (IsPICCall) { 365 if (UsePlt) 366 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF); 367 Callee = DAG.getTargetExternalSymbol(E->getSymbol(), PtrVT, 0); 368 Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee); 369 } else { 370 Callee = 371 makeHiLoPair(Callee, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG); 372 } 373 } 374 375 RegsToPass.push_back(std::make_pair(VE::SX12, Callee)); 376 377 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { 378 CCValAssign &VA = ArgLocs[i]; 379 SDValue Arg = CLI.OutVals[i]; 380 381 // Promote the value if needed. 382 switch (VA.getLocInfo()) { 383 default: 384 llvm_unreachable("Unknown location info!"); 385 case CCValAssign::Full: 386 break; 387 case CCValAssign::SExt: 388 Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Arg); 389 break; 390 case CCValAssign::ZExt: 391 Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Arg); 392 break; 393 case CCValAssign::AExt: 394 Arg = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Arg); 395 break; 396 case CCValAssign::BCvt: { 397 // Convert a float argument to i64 with padding. 398 // 63 31 0 399 // +------+------+ 400 // | float| 0 | 401 // +------+------+ 402 assert(VA.getLocVT() == MVT::i64); 403 assert(VA.getValVT() == MVT::f32); 404 SDValue Undef = SDValue( 405 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0); 406 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 407 Arg = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, 408 MVT::i64, Undef, Arg, Sub_f32), 409 0); 410 break; 411 } 412 } 413 414 if (VA.isRegLoc()) { 415 RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg)); 416 if (!UseBoth) 417 continue; 418 VA = ArgLocs2[i]; 419 } 420 421 assert(VA.isMemLoc()); 422 423 // Create a store off the stack pointer for this argument. 424 SDValue StackPtr = DAG.getRegister(VE::SX11, PtrVT); 425 // The argument area starts at %fp+176 in the callee frame, 426 // %sp+176 in ours. 427 SDValue PtrOff = 428 DAG.getIntPtrConstant(VA.getLocMemOffset() + ArgsBaseOffset, DL); 429 PtrOff = DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr, PtrOff); 430 MemOpChains.push_back( 431 DAG.getStore(Chain, DL, Arg, PtrOff, MachinePointerInfo())); 432 } 433 434 // Emit all stores, make sure they occur before the call. 435 if (!MemOpChains.empty()) 436 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains); 437 438 // Build a sequence of CopyToReg nodes glued together with token chain and 439 // glue operands which copy the outgoing args into registers. The InGlue is 440 // necessary since all emitted instructions must be stuck together in order 441 // to pass the live physical registers. 442 SDValue InGlue; 443 for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) { 444 Chain = DAG.getCopyToReg(Chain, DL, RegsToPass[i].first, 445 RegsToPass[i].second, InGlue); 446 InGlue = Chain.getValue(1); 447 } 448 449 // Build the operands for the call instruction itself. 450 SmallVector<SDValue, 8> Ops; 451 Ops.push_back(Chain); 452 for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) 453 Ops.push_back(DAG.getRegister(RegsToPass[i].first, 454 RegsToPass[i].second.getValueType())); 455 456 // Add a register mask operand representing the call-preserved registers. 457 const VERegisterInfo *TRI = Subtarget->getRegisterInfo(); 458 const uint32_t *Mask = 459 TRI->getCallPreservedMask(DAG.getMachineFunction(), CLI.CallConv); 460 assert(Mask && "Missing call preserved mask for calling convention"); 461 Ops.push_back(DAG.getRegisterMask(Mask)); 462 463 // Make sure the CopyToReg nodes are glued to the call instruction which 464 // consumes the registers. 465 if (InGlue.getNode()) 466 Ops.push_back(InGlue); 467 468 // Now the call itself. 469 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue); 470 Chain = DAG.getNode(VEISD::CALL, DL, NodeTys, Ops); 471 InGlue = Chain.getValue(1); 472 473 // Revert the stack pointer immediately after the call. 474 Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(ArgsSize, DL, true), 475 DAG.getIntPtrConstant(0, DL, true), InGlue, DL); 476 InGlue = Chain.getValue(1); 477 478 // Now extract the return values. This is more or less the same as 479 // LowerFormalArguments. 480 481 // Assign locations to each value returned by this call. 482 SmallVector<CCValAssign, 16> RVLocs; 483 CCState RVInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), RVLocs, 484 *DAG.getContext()); 485 486 // Set inreg flag manually for codegen generated library calls that 487 // return float. 488 if (CLI.Ins.size() == 1 && CLI.Ins[0].VT == MVT::f32 && !CLI.CB) 489 CLI.Ins[0].Flags.setInReg(); 490 491 RVInfo.AnalyzeCallResult(CLI.Ins, RetCC_VE); 492 493 // Copy all of the result registers out of their specified physreg. 494 for (unsigned i = 0; i != RVLocs.size(); ++i) { 495 CCValAssign &VA = RVLocs[i]; 496 unsigned Reg = VA.getLocReg(); 497 498 // When returning 'inreg {i32, i32 }', two consecutive i32 arguments can 499 // reside in the same register in the high and low bits. Reuse the 500 // CopyFromReg previous node to avoid duplicate copies. 501 SDValue RV; 502 if (RegisterSDNode *SrcReg = dyn_cast<RegisterSDNode>(Chain.getOperand(1))) 503 if (SrcReg->getReg() == Reg && Chain->getOpcode() == ISD::CopyFromReg) 504 RV = Chain.getValue(0); 505 506 // But usually we'll create a new CopyFromReg for a different register. 507 if (!RV.getNode()) { 508 RV = DAG.getCopyFromReg(Chain, DL, Reg, RVLocs[i].getLocVT(), InGlue); 509 Chain = RV.getValue(1); 510 InGlue = Chain.getValue(2); 511 } 512 513 // Get the high bits for i32 struct elements. 514 if (VA.getValVT() == MVT::i32 && VA.needsCustom()) 515 RV = DAG.getNode(ISD::SRL, DL, VA.getLocVT(), RV, 516 DAG.getConstant(32, DL, MVT::i32)); 517 518 // The callee promoted the return value, so insert an Assert?ext SDNode so 519 // we won't promote the value again in this function. 520 switch (VA.getLocInfo()) { 521 case CCValAssign::SExt: 522 RV = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), RV, 523 DAG.getValueType(VA.getValVT())); 524 break; 525 case CCValAssign::ZExt: 526 RV = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), RV, 527 DAG.getValueType(VA.getValVT())); 528 break; 529 case CCValAssign::BCvt: { 530 // Extract a float return value from i64 with padding. 531 // 63 31 0 532 // +------+------+ 533 // | float| 0 | 534 // +------+------+ 535 assert(VA.getLocVT() == MVT::i64); 536 assert(VA.getValVT() == MVT::f32); 537 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 538 RV = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, 539 MVT::f32, RV, Sub_f32), 540 0); 541 break; 542 } 543 default: 544 break; 545 } 546 547 // Truncate the register down to the return value type. 548 if (VA.isExtInLoc()) 549 RV = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), RV); 550 551 InVals.push_back(RV); 552 } 553 554 return Chain; 555 } 556 557 bool VETargetLowering::isOffsetFoldingLegal( 558 const GlobalAddressSDNode *GA) const { 559 // VE uses 64 bit addressing, so we need multiple instructions to generate 560 // an address. Folding address with offset increases the number of 561 // instructions, so that we disable it here. Offsets will be folded in 562 // the DAG combine later if it worth to do so. 563 return false; 564 } 565 566 /// isFPImmLegal - Returns true if the target can instruction select the 567 /// specified FP immediate natively. If false, the legalizer will 568 /// materialize the FP immediate as a load from a constant pool. 569 bool VETargetLowering::isFPImmLegal(const APFloat &Imm, EVT VT, 570 bool ForCodeSize) const { 571 return VT == MVT::f32 || VT == MVT::f64; 572 } 573 574 /// Determine if the target supports unaligned memory accesses. 575 /// 576 /// This function returns true if the target allows unaligned memory accesses 577 /// of the specified type in the given address space. If true, it also returns 578 /// whether the unaligned memory access is "fast" in the last argument by 579 /// reference. This is used, for example, in situations where an array 580 /// copy/move/set is converted to a sequence of store operations. Its use 581 /// helps to ensure that such replacements don't generate code that causes an 582 /// alignment error (trap) on the target machine. 583 bool VETargetLowering::allowsMisalignedMemoryAccesses(EVT VT, 584 unsigned AddrSpace, 585 unsigned Align, 586 MachineMemOperand::Flags, 587 bool *Fast) const { 588 if (Fast) { 589 // It's fast anytime on VE 590 *Fast = true; 591 } 592 return true; 593 } 594 595 bool VETargetLowering::hasAndNot(SDValue Y) const { 596 EVT VT = Y.getValueType(); 597 598 // VE doesn't have vector and not instruction. 599 if (VT.isVector()) 600 return false; 601 602 // VE allows different immediate values for X and Y where ~X & Y. 603 // Only simm7 works for X, and only mimm works for Y on VE. However, this 604 // function is used to check whether an immediate value is OK for and-not 605 // instruction as both X and Y. Generating additional instruction to 606 // retrieve an immediate value is no good since the purpose of this 607 // function is to convert a series of 3 instructions to another series of 608 // 3 instructions with better parallelism. Therefore, we return false 609 // for all immediate values now. 610 // FIXME: Change hasAndNot function to have two operands to make it work 611 // correctly with Aurora VE. 612 if (isa<ConstantSDNode>(Y)) 613 return false; 614 615 // It's ok for generic registers. 616 return true; 617 } 618 619 VETargetLowering::VETargetLowering(const TargetMachine &TM, 620 const VESubtarget &STI) 621 : TargetLowering(TM), Subtarget(&STI) { 622 // Instructions which use registers as conditionals examine all the 623 // bits (as does the pseudo SELECT_CC expansion). I don't think it 624 // matters much whether it's ZeroOrOneBooleanContent, or 625 // ZeroOrNegativeOneBooleanContent, so, arbitrarily choose the 626 // former. 627 setBooleanContents(ZeroOrOneBooleanContent); 628 setBooleanVectorContents(ZeroOrOneBooleanContent); 629 630 // Set up the register classes. 631 addRegisterClass(MVT::i32, &VE::I32RegClass); 632 addRegisterClass(MVT::i64, &VE::I64RegClass); 633 addRegisterClass(MVT::f32, &VE::F32RegClass); 634 addRegisterClass(MVT::f64, &VE::I64RegClass); 635 addRegisterClass(MVT::f128, &VE::F128RegClass); 636 637 /// Load & Store { 638 639 // VE doesn't have i1 sign extending load. 640 for (MVT VT : MVT::integer_valuetypes()) { 641 setLoadExtAction(ISD::SEXTLOAD, VT, MVT::i1, Promote); 642 setLoadExtAction(ISD::ZEXTLOAD, VT, MVT::i1, Promote); 643 setLoadExtAction(ISD::EXTLOAD, VT, MVT::i1, Promote); 644 setTruncStoreAction(VT, MVT::i1, Expand); 645 } 646 647 // VE doesn't have floating point extload/truncstore, so expand them. 648 for (MVT FPVT : MVT::fp_valuetypes()) { 649 for (MVT OtherFPVT : MVT::fp_valuetypes()) { 650 setLoadExtAction(ISD::EXTLOAD, FPVT, OtherFPVT, Expand); 651 setTruncStoreAction(FPVT, OtherFPVT, Expand); 652 } 653 } 654 655 // VE doesn't have fp128 load/store, so expand them in custom lower. 656 setOperationAction(ISD::LOAD, MVT::f128, Custom); 657 setOperationAction(ISD::STORE, MVT::f128, Custom); 658 659 /// } Load & Store 660 661 // Custom legalize address nodes into LO/HI parts. 662 MVT PtrVT = MVT::getIntegerVT(TM.getPointerSizeInBits(0)); 663 setOperationAction(ISD::BlockAddress, PtrVT, Custom); 664 setOperationAction(ISD::GlobalAddress, PtrVT, Custom); 665 setOperationAction(ISD::GlobalTLSAddress, PtrVT, Custom); 666 setOperationAction(ISD::ConstantPool, PtrVT, Custom); 667 668 /// VAARG handling { 669 setOperationAction(ISD::VASTART, MVT::Other, Custom); 670 // VAARG needs to be lowered to access with 8 bytes alignment. 671 setOperationAction(ISD::VAARG, MVT::Other, Custom); 672 // Use the default implementation. 673 setOperationAction(ISD::VACOPY, MVT::Other, Expand); 674 setOperationAction(ISD::VAEND, MVT::Other, Expand); 675 /// } VAARG handling 676 677 /// Stack { 678 setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i32, Custom); 679 setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i64, Custom); 680 /// } Stack 681 682 /// Branch { 683 // VE doesn't have BRCOND 684 setOperationAction(ISD::BRCOND, MVT::Other, Expand); 685 /// } Branch 686 687 /// Int Ops { 688 for (MVT IntVT : {MVT::i32, MVT::i64}) { 689 // VE has no REM or DIVREM operations. 690 setOperationAction(ISD::UREM, IntVT, Expand); 691 setOperationAction(ISD::SREM, IntVT, Expand); 692 setOperationAction(ISD::SDIVREM, IntVT, Expand); 693 setOperationAction(ISD::UDIVREM, IntVT, Expand); 694 695 // VE has no SHL_PARTS/SRA_PARTS/SRL_PARTS operations. 696 setOperationAction(ISD::SHL_PARTS, IntVT, Expand); 697 setOperationAction(ISD::SRA_PARTS, IntVT, Expand); 698 setOperationAction(ISD::SRL_PARTS, IntVT, Expand); 699 700 // VE has no MULHU/S or U/SMUL_LOHI operations. 701 // TODO: Use MPD instruction to implement SMUL_LOHI for i32 type. 702 setOperationAction(ISD::MULHU, IntVT, Expand); 703 setOperationAction(ISD::MULHS, IntVT, Expand); 704 setOperationAction(ISD::UMUL_LOHI, IntVT, Expand); 705 setOperationAction(ISD::SMUL_LOHI, IntVT, Expand); 706 707 // VE has no CTTZ, ROTL, ROTR operations. 708 setOperationAction(ISD::CTTZ, IntVT, Expand); 709 setOperationAction(ISD::ROTL, IntVT, Expand); 710 setOperationAction(ISD::ROTR, IntVT, Expand); 711 712 // VE has 64 bits instruction which works as i64 BSWAP operation. This 713 // instruction works fine as i32 BSWAP operation with an additional 714 // parameter. Use isel patterns to lower BSWAP. 715 setOperationAction(ISD::BSWAP, IntVT, Legal); 716 717 // VE has only 64 bits instructions which work as i64 BITREVERSE/CTLZ/CTPOP 718 // operations. Use isel patterns for i64, promote for i32. 719 LegalizeAction Act = (IntVT == MVT::i32) ? Promote : Legal; 720 setOperationAction(ISD::BITREVERSE, IntVT, Act); 721 setOperationAction(ISD::CTLZ, IntVT, Act); 722 setOperationAction(ISD::CTLZ_ZERO_UNDEF, IntVT, Act); 723 setOperationAction(ISD::CTPOP, IntVT, Act); 724 725 // VE has only 64 bits instructions which work as i64 AND/OR/XOR operations. 726 // Use isel patterns for i64, promote for i32. 727 setOperationAction(ISD::AND, IntVT, Act); 728 setOperationAction(ISD::OR, IntVT, Act); 729 setOperationAction(ISD::XOR, IntVT, Act); 730 } 731 /// } Int Ops 732 733 /// Conversion { 734 // VE doesn't have instructions for fp<->uint, so expand them by llvm 735 setOperationAction(ISD::FP_TO_UINT, MVT::i32, Promote); // use i64 736 setOperationAction(ISD::UINT_TO_FP, MVT::i32, Promote); // use i64 737 setOperationAction(ISD::FP_TO_UINT, MVT::i64, Expand); 738 setOperationAction(ISD::UINT_TO_FP, MVT::i64, Expand); 739 740 // fp16 not supported 741 for (MVT FPVT : MVT::fp_valuetypes()) { 742 setOperationAction(ISD::FP16_TO_FP, FPVT, Expand); 743 setOperationAction(ISD::FP_TO_FP16, FPVT, Expand); 744 } 745 /// } Conversion 746 747 /// Floating-point Ops { 748 /// Note: Floating-point operations are fneg, fadd, fsub, fmul, fdiv, frem, 749 /// and fcmp. 750 751 // VE doesn't have following floating point operations. 752 for (MVT VT : MVT::fp_valuetypes()) { 753 setOperationAction(ISD::FNEG, VT, Expand); 754 setOperationAction(ISD::FREM, VT, Expand); 755 } 756 757 // VE doesn't have fdiv of f128. 758 setOperationAction(ISD::FDIV, MVT::f128, Expand); 759 760 for (MVT FPVT : {MVT::f32, MVT::f64}) { 761 // f32 and f64 uses ConstantFP. f128 uses ConstantPool. 762 setOperationAction(ISD::ConstantFP, FPVT, Legal); 763 } 764 /// } Floating-point Ops 765 766 /// Floating-point math functions { 767 768 // VE doesn't have following floating point math functions. 769 for (MVT VT : MVT::fp_valuetypes()) { 770 setOperationAction(ISD::FCOPYSIGN, VT, Expand); 771 } 772 773 /// } Floating-point math functions 774 775 setStackPointerRegisterToSaveRestore(VE::SX11); 776 777 // We have target-specific dag combine patterns for the following nodes: 778 setTargetDAGCombine(ISD::TRUNCATE); 779 780 // Set function alignment to 16 bytes 781 setMinFunctionAlignment(Align(16)); 782 783 // VE stores all argument by 8 bytes alignment 784 setMinStackArgumentAlignment(Align(8)); 785 786 computeRegisterProperties(Subtarget->getRegisterInfo()); 787 } 788 789 const char *VETargetLowering::getTargetNodeName(unsigned Opcode) const { 790 #define TARGET_NODE_CASE(NAME) \ 791 case VEISD::NAME: \ 792 return "VEISD::" #NAME; 793 switch ((VEISD::NodeType)Opcode) { 794 case VEISD::FIRST_NUMBER: 795 break; 796 TARGET_NODE_CASE(Lo) 797 TARGET_NODE_CASE(Hi) 798 TARGET_NODE_CASE(GETFUNPLT) 799 TARGET_NODE_CASE(GETSTACKTOP) 800 TARGET_NODE_CASE(GETTLSADDR) 801 TARGET_NODE_CASE(CALL) 802 TARGET_NODE_CASE(RET_FLAG) 803 TARGET_NODE_CASE(GLOBAL_BASE_REG) 804 } 805 #undef TARGET_NODE_CASE 806 return nullptr; 807 } 808 809 EVT VETargetLowering::getSetCCResultType(const DataLayout &, LLVMContext &, 810 EVT VT) const { 811 return MVT::i32; 812 } 813 814 // Convert to a target node and set target flags. 815 SDValue VETargetLowering::withTargetFlags(SDValue Op, unsigned TF, 816 SelectionDAG &DAG) const { 817 if (const GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(Op)) 818 return DAG.getTargetGlobalAddress(GA->getGlobal(), SDLoc(GA), 819 GA->getValueType(0), GA->getOffset(), TF); 820 821 if (const BlockAddressSDNode *BA = dyn_cast<BlockAddressSDNode>(Op)) 822 return DAG.getTargetBlockAddress(BA->getBlockAddress(), Op.getValueType(), 823 0, TF); 824 825 if (const ConstantPoolSDNode *CP = dyn_cast<ConstantPoolSDNode>(Op)) 826 return DAG.getTargetConstantPool(CP->getConstVal(), CP->getValueType(0), 827 CP->getAlign(), CP->getOffset(), TF); 828 829 if (const ExternalSymbolSDNode *ES = dyn_cast<ExternalSymbolSDNode>(Op)) 830 return DAG.getTargetExternalSymbol(ES->getSymbol(), ES->getValueType(0), 831 TF); 832 833 llvm_unreachable("Unhandled address SDNode"); 834 } 835 836 // Split Op into high and low parts according to HiTF and LoTF. 837 // Return an ADD node combining the parts. 838 SDValue VETargetLowering::makeHiLoPair(SDValue Op, unsigned HiTF, unsigned LoTF, 839 SelectionDAG &DAG) const { 840 SDLoc DL(Op); 841 EVT VT = Op.getValueType(); 842 SDValue Hi = DAG.getNode(VEISD::Hi, DL, VT, withTargetFlags(Op, HiTF, DAG)); 843 SDValue Lo = DAG.getNode(VEISD::Lo, DL, VT, withTargetFlags(Op, LoTF, DAG)); 844 return DAG.getNode(ISD::ADD, DL, VT, Hi, Lo); 845 } 846 847 // Build SDNodes for producing an address from a GlobalAddress, ConstantPool, 848 // or ExternalSymbol SDNode. 849 SDValue VETargetLowering::makeAddress(SDValue Op, SelectionDAG &DAG) const { 850 SDLoc DL(Op); 851 EVT PtrVT = Op.getValueType(); 852 853 // Handle PIC mode first. VE needs a got load for every variable! 854 if (isPositionIndependent()) { 855 // GLOBAL_BASE_REG codegen'ed with call. Inform MFI that this 856 // function has calls. 857 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); 858 MFI.setHasCalls(true); 859 auto GlobalN = dyn_cast<GlobalAddressSDNode>(Op); 860 861 if (isa<ConstantPoolSDNode>(Op) || 862 (GlobalN && GlobalN->getGlobal()->hasLocalLinkage())) { 863 // Create following instructions for local linkage PIC code. 864 // lea %s35, %gotoff_lo(.LCPI0_0) 865 // and %s35, %s35, (32)0 866 // lea.sl %s35, %gotoff_hi(.LCPI0_0)(%s35) 867 // adds.l %s35, %s15, %s35 ; %s15 is GOT 868 // FIXME: use lea.sl %s35, %gotoff_hi(.LCPI0_0)(%s35, %s15) 869 SDValue HiLo = makeHiLoPair(Op, VEMCExpr::VK_VE_GOTOFF_HI32, 870 VEMCExpr::VK_VE_GOTOFF_LO32, DAG); 871 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT); 872 return DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo); 873 } 874 // Create following instructions for not local linkage PIC code. 875 // lea %s35, %got_lo(.LCPI0_0) 876 // and %s35, %s35, (32)0 877 // lea.sl %s35, %got_hi(.LCPI0_0)(%s35) 878 // adds.l %s35, %s15, %s35 ; %s15 is GOT 879 // ld %s35, (,%s35) 880 // FIXME: use lea.sl %s35, %gotoff_hi(.LCPI0_0)(%s35, %s15) 881 SDValue HiLo = makeHiLoPair(Op, VEMCExpr::VK_VE_GOT_HI32, 882 VEMCExpr::VK_VE_GOT_LO32, DAG); 883 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT); 884 SDValue AbsAddr = DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo); 885 return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), AbsAddr, 886 MachinePointerInfo::getGOT(DAG.getMachineFunction())); 887 } 888 889 // This is one of the absolute code models. 890 switch (getTargetMachine().getCodeModel()) { 891 default: 892 llvm_unreachable("Unsupported absolute code model"); 893 case CodeModel::Small: 894 case CodeModel::Medium: 895 case CodeModel::Large: 896 // abs64. 897 return makeHiLoPair(Op, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG); 898 } 899 } 900 901 /// Custom Lower { 902 903 SDValue VETargetLowering::lowerGlobalAddress(SDValue Op, 904 SelectionDAG &DAG) const { 905 return makeAddress(Op, DAG); 906 } 907 908 SDValue VETargetLowering::lowerBlockAddress(SDValue Op, 909 SelectionDAG &DAG) const { 910 return makeAddress(Op, DAG); 911 } 912 913 SDValue VETargetLowering::lowerConstantPool(SDValue Op, 914 SelectionDAG &DAG) const { 915 return makeAddress(Op, DAG); 916 } 917 918 SDValue 919 VETargetLowering::lowerToTLSGeneralDynamicModel(SDValue Op, 920 SelectionDAG &DAG) const { 921 SDLoc DL(Op); 922 923 // Generate the following code: 924 // t1: ch,glue = callseq_start t0, 0, 0 925 // t2: i64,ch,glue = VEISD::GETTLSADDR t1, label, t1:1 926 // t3: ch,glue = callseq_end t2, 0, 0, t2:2 927 // t4: i64,ch,glue = CopyFromReg t3, Register:i64 $sx0, t3:1 928 SDValue Label = withTargetFlags(Op, 0, DAG); 929 EVT PtrVT = Op.getValueType(); 930 931 // Lowering the machine isd will make sure everything is in the right 932 // location. 933 SDValue Chain = DAG.getEntryNode(); 934 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue); 935 const uint32_t *Mask = Subtarget->getRegisterInfo()->getCallPreservedMask( 936 DAG.getMachineFunction(), CallingConv::C); 937 Chain = DAG.getCALLSEQ_START(Chain, 64, 0, DL); 938 SDValue Args[] = {Chain, Label, DAG.getRegisterMask(Mask), Chain.getValue(1)}; 939 Chain = DAG.getNode(VEISD::GETTLSADDR, DL, NodeTys, Args); 940 Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(64, DL, true), 941 DAG.getIntPtrConstant(0, DL, true), 942 Chain.getValue(1), DL); 943 Chain = DAG.getCopyFromReg(Chain, DL, VE::SX0, PtrVT, Chain.getValue(1)); 944 945 // GETTLSADDR will be codegen'ed as call. Inform MFI that function has calls. 946 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); 947 MFI.setHasCalls(true); 948 949 // Also generate code to prepare a GOT register if it is PIC. 950 if (isPositionIndependent()) { 951 MachineFunction &MF = DAG.getMachineFunction(); 952 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF); 953 } 954 955 return Chain; 956 } 957 958 SDValue VETargetLowering::lowerGlobalTLSAddress(SDValue Op, 959 SelectionDAG &DAG) const { 960 // The current implementation of nld (2.26) doesn't allow local exec model 961 // code described in VE-tls_v1.1.pdf (*1) as its input. Instead, we always 962 // generate the general dynamic model code sequence. 963 // 964 // *1: https://www.nec.com/en/global/prod/hpc/aurora/document/VE-tls_v1.1.pdf 965 return lowerToTLSGeneralDynamicModel(Op, DAG); 966 } 967 968 // Lower a f128 load into two f64 loads. 969 static SDValue lowerLoadF128(SDValue Op, SelectionDAG &DAG) { 970 SDLoc DL(Op); 971 LoadSDNode *LdNode = dyn_cast<LoadSDNode>(Op.getNode()); 972 assert(LdNode && LdNode->getOffset().isUndef() && "Unexpected node type"); 973 unsigned Alignment = LdNode->getAlign().value(); 974 if (Alignment > 8) 975 Alignment = 8; 976 977 SDValue Lo64 = 978 DAG.getLoad(MVT::f64, DL, LdNode->getChain(), LdNode->getBasePtr(), 979 LdNode->getPointerInfo(), Alignment, 980 LdNode->isVolatile() ? MachineMemOperand::MOVolatile 981 : MachineMemOperand::MONone); 982 EVT AddrVT = LdNode->getBasePtr().getValueType(); 983 SDValue HiPtr = DAG.getNode(ISD::ADD, DL, AddrVT, LdNode->getBasePtr(), 984 DAG.getConstant(8, DL, AddrVT)); 985 SDValue Hi64 = 986 DAG.getLoad(MVT::f64, DL, LdNode->getChain(), HiPtr, 987 LdNode->getPointerInfo(), Alignment, 988 LdNode->isVolatile() ? MachineMemOperand::MOVolatile 989 : MachineMemOperand::MONone); 990 991 SDValue SubRegEven = DAG.getTargetConstant(VE::sub_even, DL, MVT::i32); 992 SDValue SubRegOdd = DAG.getTargetConstant(VE::sub_odd, DL, MVT::i32); 993 994 // VE stores Hi64 to 8(addr) and Lo64 to 0(addr) 995 SDNode *InFP128 = 996 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::f128); 997 InFP128 = DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, MVT::f128, 998 SDValue(InFP128, 0), Hi64, SubRegEven); 999 InFP128 = DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, MVT::f128, 1000 SDValue(InFP128, 0), Lo64, SubRegOdd); 1001 SDValue OutChains[2] = {SDValue(Lo64.getNode(), 1), 1002 SDValue(Hi64.getNode(), 1)}; 1003 SDValue OutChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains); 1004 SDValue Ops[2] = {SDValue(InFP128, 0), OutChain}; 1005 return DAG.getMergeValues(Ops, DL); 1006 } 1007 1008 SDValue VETargetLowering::lowerLOAD(SDValue Op, SelectionDAG &DAG) const { 1009 LoadSDNode *LdNode = cast<LoadSDNode>(Op.getNode()); 1010 1011 SDValue BasePtr = LdNode->getBasePtr(); 1012 if (isa<FrameIndexSDNode>(BasePtr.getNode())) { 1013 // Do not expand store instruction with frame index here because of 1014 // dependency problems. We expand it later in eliminateFrameIndex(). 1015 return Op; 1016 } 1017 1018 EVT MemVT = LdNode->getMemoryVT(); 1019 if (MemVT == MVT::f128) 1020 return lowerLoadF128(Op, DAG); 1021 1022 return Op; 1023 } 1024 1025 // Lower a f128 store into two f64 stores. 1026 static SDValue lowerStoreF128(SDValue Op, SelectionDAG &DAG) { 1027 SDLoc DL(Op); 1028 StoreSDNode *StNode = dyn_cast<StoreSDNode>(Op.getNode()); 1029 assert(StNode && StNode->getOffset().isUndef() && "Unexpected node type"); 1030 1031 SDValue SubRegEven = DAG.getTargetConstant(VE::sub_even, DL, MVT::i32); 1032 SDValue SubRegOdd = DAG.getTargetConstant(VE::sub_odd, DL, MVT::i32); 1033 1034 SDNode *Hi64 = DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, MVT::i64, 1035 StNode->getValue(), SubRegEven); 1036 SDNode *Lo64 = DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, MVT::i64, 1037 StNode->getValue(), SubRegOdd); 1038 1039 unsigned Alignment = StNode->getAlign().value(); 1040 if (Alignment > 8) 1041 Alignment = 8; 1042 1043 // VE stores Hi64 to 8(addr) and Lo64 to 0(addr) 1044 SDValue OutChains[2]; 1045 OutChains[0] = 1046 DAG.getStore(StNode->getChain(), DL, SDValue(Lo64, 0), 1047 StNode->getBasePtr(), MachinePointerInfo(), Alignment, 1048 StNode->isVolatile() ? MachineMemOperand::MOVolatile 1049 : MachineMemOperand::MONone); 1050 EVT AddrVT = StNode->getBasePtr().getValueType(); 1051 SDValue HiPtr = DAG.getNode(ISD::ADD, DL, AddrVT, StNode->getBasePtr(), 1052 DAG.getConstant(8, DL, AddrVT)); 1053 OutChains[1] = 1054 DAG.getStore(StNode->getChain(), DL, SDValue(Hi64, 0), HiPtr, 1055 MachinePointerInfo(), Alignment, 1056 StNode->isVolatile() ? MachineMemOperand::MOVolatile 1057 : MachineMemOperand::MONone); 1058 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains); 1059 } 1060 1061 SDValue VETargetLowering::lowerSTORE(SDValue Op, SelectionDAG &DAG) const { 1062 StoreSDNode *StNode = cast<StoreSDNode>(Op.getNode()); 1063 assert(StNode && StNode->getOffset().isUndef() && "Unexpected node type"); 1064 1065 SDValue BasePtr = StNode->getBasePtr(); 1066 if (isa<FrameIndexSDNode>(BasePtr.getNode())) { 1067 // Do not expand store instruction with frame index here because of 1068 // dependency problems. We expand it later in eliminateFrameIndex(). 1069 return Op; 1070 } 1071 1072 EVT MemVT = StNode->getMemoryVT(); 1073 if (MemVT == MVT::f128) 1074 return lowerStoreF128(Op, DAG); 1075 1076 // Otherwise, ask llvm to expand it. 1077 return SDValue(); 1078 } 1079 1080 SDValue VETargetLowering::lowerVASTART(SDValue Op, SelectionDAG &DAG) const { 1081 MachineFunction &MF = DAG.getMachineFunction(); 1082 VEMachineFunctionInfo *FuncInfo = MF.getInfo<VEMachineFunctionInfo>(); 1083 auto PtrVT = getPointerTy(DAG.getDataLayout()); 1084 1085 // Need frame address to find the address of VarArgsFrameIndex. 1086 MF.getFrameInfo().setFrameAddressIsTaken(true); 1087 1088 // vastart just stores the address of the VarArgsFrameIndex slot into the 1089 // memory location argument. 1090 SDLoc DL(Op); 1091 SDValue Offset = 1092 DAG.getNode(ISD::ADD, DL, PtrVT, DAG.getRegister(VE::SX9, PtrVT), 1093 DAG.getIntPtrConstant(FuncInfo->getVarArgsFrameOffset(), DL)); 1094 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue(); 1095 return DAG.getStore(Op.getOperand(0), DL, Offset, Op.getOperand(1), 1096 MachinePointerInfo(SV)); 1097 } 1098 1099 SDValue VETargetLowering::lowerVAARG(SDValue Op, SelectionDAG &DAG) const { 1100 SDNode *Node = Op.getNode(); 1101 EVT VT = Node->getValueType(0); 1102 SDValue InChain = Node->getOperand(0); 1103 SDValue VAListPtr = Node->getOperand(1); 1104 EVT PtrVT = VAListPtr.getValueType(); 1105 const Value *SV = cast<SrcValueSDNode>(Node->getOperand(2))->getValue(); 1106 SDLoc DL(Node); 1107 SDValue VAList = 1108 DAG.getLoad(PtrVT, DL, InChain, VAListPtr, MachinePointerInfo(SV)); 1109 SDValue Chain = VAList.getValue(1); 1110 SDValue NextPtr; 1111 1112 if (VT == MVT::f128) { 1113 // VE f128 values must be stored with 16 bytes alignment. We doesn't 1114 // know the actual alignment of VAList, so we take alignment of it 1115 // dyanmically. 1116 int Align = 16; 1117 VAList = DAG.getNode(ISD::ADD, DL, PtrVT, VAList, 1118 DAG.getConstant(Align - 1, DL, PtrVT)); 1119 VAList = DAG.getNode(ISD::AND, DL, PtrVT, VAList, 1120 DAG.getConstant(-Align, DL, PtrVT)); 1121 // Increment the pointer, VAList, by 16 to the next vaarg. 1122 NextPtr = 1123 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(16, DL)); 1124 } else if (VT == MVT::f32) { 1125 // float --> need special handling like below. 1126 // 0 4 1127 // +------+------+ 1128 // | empty| float| 1129 // +------+------+ 1130 // Increment the pointer, VAList, by 8 to the next vaarg. 1131 NextPtr = 1132 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL)); 1133 // Then, adjust VAList. 1134 unsigned InternalOffset = 4; 1135 VAList = DAG.getNode(ISD::ADD, DL, PtrVT, VAList, 1136 DAG.getConstant(InternalOffset, DL, PtrVT)); 1137 } else { 1138 // Increment the pointer, VAList, by 8 to the next vaarg. 1139 NextPtr = 1140 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL)); 1141 } 1142 1143 // Store the incremented VAList to the legalized pointer. 1144 InChain = DAG.getStore(Chain, DL, NextPtr, VAListPtr, MachinePointerInfo(SV)); 1145 1146 // Load the actual argument out of the pointer VAList. 1147 // We can't count on greater alignment than the word size. 1148 return DAG.getLoad(VT, DL, InChain, VAList, MachinePointerInfo(), 1149 std::min(PtrVT.getSizeInBits(), VT.getSizeInBits()) / 8); 1150 } 1151 1152 SDValue VETargetLowering::lowerDYNAMIC_STACKALLOC(SDValue Op, 1153 SelectionDAG &DAG) const { 1154 // Generate following code. 1155 // (void)__llvm_grow_stack(size); 1156 // ret = GETSTACKTOP; // pseudo instruction 1157 SDLoc DL(Op); 1158 1159 // Get the inputs. 1160 SDNode *Node = Op.getNode(); 1161 SDValue Chain = Op.getOperand(0); 1162 SDValue Size = Op.getOperand(1); 1163 MaybeAlign Alignment(Op.getConstantOperandVal(2)); 1164 EVT VT = Node->getValueType(0); 1165 1166 // Chain the dynamic stack allocation so that it doesn't modify the stack 1167 // pointer when other instructions are using the stack. 1168 Chain = DAG.getCALLSEQ_START(Chain, 0, 0, DL); 1169 1170 const TargetFrameLowering &TFI = *Subtarget->getFrameLowering(); 1171 Align StackAlign = TFI.getStackAlign(); 1172 bool NeedsAlign = Alignment.valueOrOne() > StackAlign; 1173 1174 // Prepare arguments 1175 TargetLowering::ArgListTy Args; 1176 TargetLowering::ArgListEntry Entry; 1177 Entry.Node = Size; 1178 Entry.Ty = Entry.Node.getValueType().getTypeForEVT(*DAG.getContext()); 1179 Args.push_back(Entry); 1180 if (NeedsAlign) { 1181 Entry.Node = DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT); 1182 Entry.Ty = Entry.Node.getValueType().getTypeForEVT(*DAG.getContext()); 1183 Args.push_back(Entry); 1184 } 1185 Type *RetTy = Type::getVoidTy(*DAG.getContext()); 1186 1187 EVT PtrVT = Op.getValueType(); 1188 SDValue Callee; 1189 if (NeedsAlign) { 1190 Callee = DAG.getTargetExternalSymbol("__ve_grow_stack_align", PtrVT, 0); 1191 } else { 1192 Callee = DAG.getTargetExternalSymbol("__ve_grow_stack", PtrVT, 0); 1193 } 1194 1195 TargetLowering::CallLoweringInfo CLI(DAG); 1196 CLI.setDebugLoc(DL) 1197 .setChain(Chain) 1198 .setCallee(CallingConv::PreserveAll, RetTy, Callee, std::move(Args)) 1199 .setDiscardResult(true); 1200 std::pair<SDValue, SDValue> pair = LowerCallTo(CLI); 1201 Chain = pair.second; 1202 SDValue Result = DAG.getNode(VEISD::GETSTACKTOP, DL, VT, Chain); 1203 if (NeedsAlign) { 1204 Result = DAG.getNode(ISD::ADD, DL, VT, Result, 1205 DAG.getConstant((Alignment->value() - 1ULL), DL, VT)); 1206 Result = DAG.getNode(ISD::AND, DL, VT, Result, 1207 DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT)); 1208 } 1209 // Chain = Result.getValue(1); 1210 Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(0, DL, true), 1211 DAG.getIntPtrConstant(0, DL, true), SDValue(), DL); 1212 1213 SDValue Ops[2] = {Result, Chain}; 1214 return DAG.getMergeValues(Ops, DL); 1215 } 1216 1217 SDValue VETargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const { 1218 switch (Op.getOpcode()) { 1219 default: 1220 llvm_unreachable("Should not custom lower this!"); 1221 case ISD::BlockAddress: 1222 return lowerBlockAddress(Op, DAG); 1223 case ISD::ConstantPool: 1224 return lowerConstantPool(Op, DAG); 1225 case ISD::DYNAMIC_STACKALLOC: 1226 return lowerDYNAMIC_STACKALLOC(Op, DAG); 1227 case ISD::GlobalAddress: 1228 return lowerGlobalAddress(Op, DAG); 1229 case ISD::GlobalTLSAddress: 1230 return lowerGlobalTLSAddress(Op, DAG); 1231 case ISD::LOAD: 1232 return lowerLOAD(Op, DAG); 1233 case ISD::STORE: 1234 return lowerSTORE(Op, DAG); 1235 case ISD::VASTART: 1236 return lowerVASTART(Op, DAG); 1237 case ISD::VAARG: 1238 return lowerVAARG(Op, DAG); 1239 } 1240 } 1241 /// } Custom Lower 1242 1243 static bool isI32Insn(const SDNode *User, const SDNode *N) { 1244 switch (User->getOpcode()) { 1245 default: 1246 return false; 1247 case ISD::ADD: 1248 case ISD::SUB: 1249 case ISD::MUL: 1250 case ISD::SDIV: 1251 case ISD::UDIV: 1252 case ISD::SETCC: 1253 case ISD::SMIN: 1254 case ISD::SMAX: 1255 case ISD::SHL: 1256 case ISD::SRA: 1257 case ISD::BSWAP: 1258 case ISD::SINT_TO_FP: 1259 case ISD::UINT_TO_FP: 1260 case ISD::BR_CC: 1261 case ISD::BITCAST: 1262 case ISD::ATOMIC_CMP_SWAP: 1263 case ISD::ATOMIC_SWAP: 1264 return true; 1265 case ISD::SRL: 1266 if (N->getOperand(0).getOpcode() != ISD::SRL) 1267 return true; 1268 // (srl (trunc (srl ...))) may be optimized by combining srl, so 1269 // doesn't optimize trunc now. 1270 return false; 1271 case ISD::SELECT_CC: 1272 if (User->getOperand(2).getNode() != N && 1273 User->getOperand(3).getNode() != N) 1274 return true; 1275 LLVM_FALLTHROUGH; 1276 case ISD::AND: 1277 case ISD::OR: 1278 case ISD::XOR: 1279 case ISD::SELECT: 1280 case ISD::CopyToReg: 1281 // Check all use of selections, bit operations, and copies. If all of them 1282 // are safe, optimize truncate to extract_subreg. 1283 for (SDNode::use_iterator UI = User->use_begin(), UE = User->use_end(); 1284 UI != UE; ++UI) { 1285 switch ((*UI)->getOpcode()) { 1286 default: 1287 // If the use is an instruction which treats the source operand as i32, 1288 // it is safe to avoid truncate here. 1289 if (isI32Insn(*UI, N)) 1290 continue; 1291 break; 1292 case ISD::ANY_EXTEND: 1293 case ISD::SIGN_EXTEND: 1294 case ISD::ZERO_EXTEND: { 1295 // Special optimizations to the combination of ext and trunc. 1296 // (ext ... (select ... (trunc ...))) is safe to avoid truncate here 1297 // since this truncate instruction clears higher 32 bits which is filled 1298 // by one of ext instructions later. 1299 assert(N->getValueType(0) == MVT::i32 && 1300 "find truncate to not i32 integer"); 1301 if (User->getOpcode() == ISD::SELECT_CC || 1302 User->getOpcode() == ISD::SELECT) 1303 continue; 1304 break; 1305 } 1306 } 1307 return false; 1308 } 1309 return true; 1310 } 1311 } 1312 1313 // Optimize TRUNCATE in DAG combining. Optimizing it in CUSTOM lower is 1314 // sometime too early. Optimizing it in DAG pattern matching in VEInstrInfo.td 1315 // is sometime too late. So, doing it at here. 1316 SDValue VETargetLowering::combineTRUNCATE(SDNode *N, 1317 DAGCombinerInfo &DCI) const { 1318 assert(N->getOpcode() == ISD::TRUNCATE && 1319 "Should be called with a TRUNCATE node"); 1320 1321 SelectionDAG &DAG = DCI.DAG; 1322 SDLoc DL(N); 1323 EVT VT = N->getValueType(0); 1324 1325 // We prefer to do this when all types are legal. 1326 if (!DCI.isAfterLegalizeDAG()) 1327 return SDValue(); 1328 1329 // Skip combine TRUNCATE atm if the operand of TRUNCATE might be a constant. 1330 if (N->getOperand(0)->getOpcode() == ISD::SELECT_CC && 1331 isa<ConstantSDNode>(N->getOperand(0)->getOperand(0)) && 1332 isa<ConstantSDNode>(N->getOperand(0)->getOperand(1))) 1333 return SDValue(); 1334 1335 // Check all use of this TRUNCATE. 1336 for (SDNode::use_iterator UI = N->use_begin(), UE = N->use_end(); UI != UE; 1337 ++UI) { 1338 SDNode *User = *UI; 1339 1340 // Make sure that we're not going to replace TRUNCATE for non i32 1341 // instructions. 1342 // 1343 // FIXME: Although we could sometimes handle this, and it does occur in 1344 // practice that one of the condition inputs to the select is also one of 1345 // the outputs, we currently can't deal with this. 1346 if (isI32Insn(User, N)) 1347 continue; 1348 1349 return SDValue(); 1350 } 1351 1352 SDValue SubI32 = DAG.getTargetConstant(VE::sub_i32, DL, MVT::i32); 1353 return SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, VT, 1354 N->getOperand(0), SubI32), 1355 0); 1356 } 1357 1358 SDValue VETargetLowering::PerformDAGCombine(SDNode *N, 1359 DAGCombinerInfo &DCI) const { 1360 switch (N->getOpcode()) { 1361 default: 1362 break; 1363 case ISD::TRUNCATE: 1364 return combineTRUNCATE(N, DCI); 1365 } 1366 1367 return SDValue(); 1368 } 1369