1 //===-- VEISelLowering.cpp - VE DAG Lowering Implementation ---------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This file implements the interfaces that VE uses to lower LLVM code into a 10 // selection DAG. 11 // 12 //===----------------------------------------------------------------------===// 13 14 #include "VEISelLowering.h" 15 #include "MCTargetDesc/VEMCExpr.h" 16 #include "VEMachineFunctionInfo.h" 17 #include "VERegisterInfo.h" 18 #include "VETargetMachine.h" 19 #include "llvm/ADT/StringSwitch.h" 20 #include "llvm/CodeGen/CallingConvLower.h" 21 #include "llvm/CodeGen/MachineFrameInfo.h" 22 #include "llvm/CodeGen/MachineFunction.h" 23 #include "llvm/CodeGen/MachineInstrBuilder.h" 24 #include "llvm/CodeGen/MachineModuleInfo.h" 25 #include "llvm/CodeGen/MachineRegisterInfo.h" 26 #include "llvm/CodeGen/SelectionDAG.h" 27 #include "llvm/CodeGen/TargetLoweringObjectFileImpl.h" 28 #include "llvm/IR/DerivedTypes.h" 29 #include "llvm/IR/Function.h" 30 #include "llvm/IR/Module.h" 31 #include "llvm/Support/ErrorHandling.h" 32 #include "llvm/Support/KnownBits.h" 33 using namespace llvm; 34 35 #define DEBUG_TYPE "ve-lower" 36 37 //===----------------------------------------------------------------------===// 38 // Calling Convention Implementation 39 //===----------------------------------------------------------------------===// 40 41 #include "VEGenCallingConv.inc" 42 43 bool VETargetLowering::CanLowerReturn( 44 CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg, 45 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context) const { 46 CCAssignFn *RetCC = RetCC_VE; 47 SmallVector<CCValAssign, 16> RVLocs; 48 CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context); 49 return CCInfo.CheckReturn(Outs, RetCC); 50 } 51 52 SDValue 53 VETargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv, 54 bool IsVarArg, 55 const SmallVectorImpl<ISD::OutputArg> &Outs, 56 const SmallVectorImpl<SDValue> &OutVals, 57 const SDLoc &DL, SelectionDAG &DAG) const { 58 // CCValAssign - represent the assignment of the return value to locations. 59 SmallVector<CCValAssign, 16> RVLocs; 60 61 // CCState - Info about the registers and stack slot. 62 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), RVLocs, 63 *DAG.getContext()); 64 65 // Analyze return values. 66 CCInfo.AnalyzeReturn(Outs, RetCC_VE); 67 68 SDValue Flag; 69 SmallVector<SDValue, 4> RetOps(1, Chain); 70 71 // Copy the result values into the output registers. 72 for (unsigned i = 0; i != RVLocs.size(); ++i) { 73 CCValAssign &VA = RVLocs[i]; 74 assert(VA.isRegLoc() && "Can only return in registers!"); 75 SDValue OutVal = OutVals[i]; 76 77 // Integer return values must be sign or zero extended by the callee. 78 switch (VA.getLocInfo()) { 79 case CCValAssign::Full: 80 break; 81 case CCValAssign::SExt: 82 OutVal = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), OutVal); 83 break; 84 case CCValAssign::ZExt: 85 OutVal = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), OutVal); 86 break; 87 case CCValAssign::AExt: 88 OutVal = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), OutVal); 89 break; 90 case CCValAssign::BCvt: { 91 // Convert a float return value to i64 with padding. 92 // 63 31 0 93 // +------+------+ 94 // | float| 0 | 95 // +------+------+ 96 assert(VA.getLocVT() == MVT::i64); 97 assert(VA.getValVT() == MVT::f32); 98 SDValue Undef = SDValue( 99 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0); 100 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 101 OutVal = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, 102 MVT::i64, Undef, OutVal, Sub_f32), 103 0); 104 break; 105 } 106 default: 107 llvm_unreachable("Unknown loc info!"); 108 } 109 110 assert(!VA.needsCustom() && "Unexpected custom lowering"); 111 112 Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), OutVal, Flag); 113 114 // Guarantee that all emitted copies are stuck together with flags. 115 Flag = Chain.getValue(1); 116 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT())); 117 } 118 119 RetOps[0] = Chain; // Update chain. 120 121 // Add the flag if we have it. 122 if (Flag.getNode()) 123 RetOps.push_back(Flag); 124 125 return DAG.getNode(VEISD::RET_FLAG, DL, MVT::Other, RetOps); 126 } 127 128 SDValue VETargetLowering::LowerFormalArguments( 129 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, 130 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL, 131 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const { 132 MachineFunction &MF = DAG.getMachineFunction(); 133 134 // Get the base offset of the incoming arguments stack space. 135 unsigned ArgsBaseOffset = 176; 136 // Get the size of the preserved arguments area 137 unsigned ArgsPreserved = 64; 138 139 // Analyze arguments according to CC_VE. 140 SmallVector<CCValAssign, 16> ArgLocs; 141 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), ArgLocs, 142 *DAG.getContext()); 143 // Allocate the preserved area first. 144 CCInfo.AllocateStack(ArgsPreserved, Align(8)); 145 // We already allocated the preserved area, so the stack offset computed 146 // by CC_VE would be correct now. 147 CCInfo.AnalyzeFormalArguments(Ins, CC_VE); 148 149 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { 150 CCValAssign &VA = ArgLocs[i]; 151 if (VA.isRegLoc()) { 152 // This argument is passed in a register. 153 // All integer register arguments are promoted by the caller to i64. 154 155 // Create a virtual register for the promoted live-in value. 156 unsigned VReg = 157 MF.addLiveIn(VA.getLocReg(), getRegClassFor(VA.getLocVT())); 158 SDValue Arg = DAG.getCopyFromReg(Chain, DL, VReg, VA.getLocVT()); 159 160 // Get the high bits for i32 struct elements. 161 if (VA.getValVT() == MVT::i32 && VA.needsCustom()) 162 Arg = DAG.getNode(ISD::SRL, DL, VA.getLocVT(), Arg, 163 DAG.getConstant(32, DL, MVT::i32)); 164 165 // The caller promoted the argument, so insert an Assert?ext SDNode so we 166 // won't promote the value again in this function. 167 switch (VA.getLocInfo()) { 168 case CCValAssign::SExt: 169 Arg = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), Arg, 170 DAG.getValueType(VA.getValVT())); 171 break; 172 case CCValAssign::ZExt: 173 Arg = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), Arg, 174 DAG.getValueType(VA.getValVT())); 175 break; 176 case CCValAssign::BCvt: { 177 // Extract a float argument from i64 with padding. 178 // 63 31 0 179 // +------+------+ 180 // | float| 0 | 181 // +------+------+ 182 assert(VA.getLocVT() == MVT::i64); 183 assert(VA.getValVT() == MVT::f32); 184 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 185 Arg = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, 186 MVT::f32, Arg, Sub_f32), 187 0); 188 break; 189 } 190 default: 191 break; 192 } 193 194 // Truncate the register down to the argument type. 195 if (VA.isExtInLoc()) 196 Arg = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Arg); 197 198 InVals.push_back(Arg); 199 continue; 200 } 201 202 // The registers are exhausted. This argument was passed on the stack. 203 assert(VA.isMemLoc()); 204 // The CC_VE_Full/Half functions compute stack offsets relative to the 205 // beginning of the arguments area at %fp+176. 206 unsigned Offset = VA.getLocMemOffset() + ArgsBaseOffset; 207 unsigned ValSize = VA.getValVT().getSizeInBits() / 8; 208 209 // Adjust offset for a float argument by adding 4 since the argument is 210 // stored in 8 bytes buffer with offset like below. LLVM generates 211 // 4 bytes load instruction, so need to adjust offset here. This 212 // adjustment is required in only LowerFormalArguments. In LowerCall, 213 // a float argument is converted to i64 first, and stored as 8 bytes 214 // data, which is required by ABI, so no need for adjustment. 215 // 0 4 216 // +------+------+ 217 // | empty| float| 218 // +------+------+ 219 if (VA.getValVT() == MVT::f32) 220 Offset += 4; 221 222 int FI = MF.getFrameInfo().CreateFixedObject(ValSize, Offset, true); 223 InVals.push_back( 224 DAG.getLoad(VA.getValVT(), DL, Chain, 225 DAG.getFrameIndex(FI, getPointerTy(MF.getDataLayout())), 226 MachinePointerInfo::getFixedStack(MF, FI))); 227 } 228 229 if (!IsVarArg) 230 return Chain; 231 232 // This function takes variable arguments, some of which may have been passed 233 // in registers %s0-%s8. 234 // 235 // The va_start intrinsic needs to know the offset to the first variable 236 // argument. 237 // TODO: need to calculate offset correctly once we support f128. 238 unsigned ArgOffset = ArgLocs.size() * 8; 239 VEMachineFunctionInfo *FuncInfo = MF.getInfo<VEMachineFunctionInfo>(); 240 // Skip the 176 bytes of register save area. 241 FuncInfo->setVarArgsFrameOffset(ArgOffset + ArgsBaseOffset); 242 243 return Chain; 244 } 245 246 // FIXME? Maybe this could be a TableGen attribute on some registers and 247 // this table could be generated automatically from RegInfo. 248 Register VETargetLowering::getRegisterByName(const char *RegName, LLT VT, 249 const MachineFunction &MF) const { 250 Register Reg = StringSwitch<Register>(RegName) 251 .Case("sp", VE::SX11) // Stack pointer 252 .Case("fp", VE::SX9) // Frame pointer 253 .Case("sl", VE::SX8) // Stack limit 254 .Case("lr", VE::SX10) // Link register 255 .Case("tp", VE::SX14) // Thread pointer 256 .Case("outer", VE::SX12) // Outer regiser 257 .Case("info", VE::SX17) // Info area register 258 .Case("got", VE::SX15) // Global offset table register 259 .Case("plt", VE::SX16) // Procedure linkage table register 260 .Default(0); 261 262 if (Reg) 263 return Reg; 264 265 report_fatal_error("Invalid register name global variable"); 266 } 267 268 //===----------------------------------------------------------------------===// 269 // TargetLowering Implementation 270 //===----------------------------------------------------------------------===// 271 272 SDValue VETargetLowering::LowerCall(TargetLowering::CallLoweringInfo &CLI, 273 SmallVectorImpl<SDValue> &InVals) const { 274 SelectionDAG &DAG = CLI.DAG; 275 SDLoc DL = CLI.DL; 276 SDValue Chain = CLI.Chain; 277 auto PtrVT = getPointerTy(DAG.getDataLayout()); 278 279 // VE target does not yet support tail call optimization. 280 CLI.IsTailCall = false; 281 282 // Get the base offset of the outgoing arguments stack space. 283 unsigned ArgsBaseOffset = 176; 284 // Get the size of the preserved arguments area 285 unsigned ArgsPreserved = 8 * 8u; 286 287 // Analyze operands of the call, assigning locations to each operand. 288 SmallVector<CCValAssign, 16> ArgLocs; 289 CCState CCInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), ArgLocs, 290 *DAG.getContext()); 291 // Allocate the preserved area first. 292 CCInfo.AllocateStack(ArgsPreserved, Align(8)); 293 // We already allocated the preserved area, so the stack offset computed 294 // by CC_VE would be correct now. 295 CCInfo.AnalyzeCallOperands(CLI.Outs, CC_VE); 296 297 // VE requires to use both register and stack for varargs or no-prototyped 298 // functions. 299 bool UseBoth = CLI.IsVarArg; 300 301 // Analyze operands again if it is required to store BOTH. 302 SmallVector<CCValAssign, 16> ArgLocs2; 303 CCState CCInfo2(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), 304 ArgLocs2, *DAG.getContext()); 305 if (UseBoth) 306 CCInfo2.AnalyzeCallOperands(CLI.Outs, CC_VE2); 307 308 // Get the size of the outgoing arguments stack space requirement. 309 unsigned ArgsSize = CCInfo.getNextStackOffset(); 310 311 // Keep stack frames 16-byte aligned. 312 ArgsSize = alignTo(ArgsSize, 16); 313 314 // Adjust the stack pointer to make room for the arguments. 315 // FIXME: Use hasReservedCallFrame to avoid %sp adjustments around all calls 316 // with more than 6 arguments. 317 Chain = DAG.getCALLSEQ_START(Chain, ArgsSize, 0, DL); 318 319 // Collect the set of registers to pass to the function and their values. 320 // This will be emitted as a sequence of CopyToReg nodes glued to the call 321 // instruction. 322 SmallVector<std::pair<unsigned, SDValue>, 8> RegsToPass; 323 324 // Collect chains from all the memory opeations that copy arguments to the 325 // stack. They must follow the stack pointer adjustment above and precede the 326 // call instruction itself. 327 SmallVector<SDValue, 8> MemOpChains; 328 329 // VE needs to get address of callee function in a register 330 // So, prepare to copy it to SX12 here. 331 332 // If the callee is a GlobalAddress node (quite common, every direct call is) 333 // turn it into a TargetGlobalAddress node so that legalize doesn't hack it. 334 // Likewise ExternalSymbol -> TargetExternalSymbol. 335 SDValue Callee = CLI.Callee; 336 337 bool IsPICCall = isPositionIndependent(); 338 339 // PC-relative references to external symbols should go through $stub. 340 // If so, we need to prepare GlobalBaseReg first. 341 const TargetMachine &TM = DAG.getTarget(); 342 const Module *Mod = DAG.getMachineFunction().getFunction().getParent(); 343 const GlobalValue *GV = nullptr; 344 auto *CalleeG = dyn_cast<GlobalAddressSDNode>(Callee); 345 if (CalleeG) 346 GV = CalleeG->getGlobal(); 347 bool Local = TM.shouldAssumeDSOLocal(*Mod, GV); 348 bool UsePlt = !Local; 349 MachineFunction &MF = DAG.getMachineFunction(); 350 351 // Turn GlobalAddress/ExternalSymbol node into a value node 352 // containing the address of them here. 353 if (CalleeG) { 354 if (IsPICCall) { 355 if (UsePlt) 356 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF); 357 Callee = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, 0); 358 Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee); 359 } else { 360 Callee = 361 makeHiLoPair(Callee, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG); 362 } 363 } else if (ExternalSymbolSDNode *E = dyn_cast<ExternalSymbolSDNode>(Callee)) { 364 if (IsPICCall) { 365 if (UsePlt) 366 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF); 367 Callee = DAG.getTargetExternalSymbol(E->getSymbol(), PtrVT, 0); 368 Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee); 369 } else { 370 Callee = 371 makeHiLoPair(Callee, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG); 372 } 373 } 374 375 RegsToPass.push_back(std::make_pair(VE::SX12, Callee)); 376 377 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { 378 CCValAssign &VA = ArgLocs[i]; 379 SDValue Arg = CLI.OutVals[i]; 380 381 // Promote the value if needed. 382 switch (VA.getLocInfo()) { 383 default: 384 llvm_unreachable("Unknown location info!"); 385 case CCValAssign::Full: 386 break; 387 case CCValAssign::SExt: 388 Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Arg); 389 break; 390 case CCValAssign::ZExt: 391 Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Arg); 392 break; 393 case CCValAssign::AExt: 394 Arg = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Arg); 395 break; 396 case CCValAssign::BCvt: { 397 // Convert a float argument to i64 with padding. 398 // 63 31 0 399 // +------+------+ 400 // | float| 0 | 401 // +------+------+ 402 assert(VA.getLocVT() == MVT::i64); 403 assert(VA.getValVT() == MVT::f32); 404 SDValue Undef = SDValue( 405 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0); 406 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 407 Arg = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, 408 MVT::i64, Undef, Arg, Sub_f32), 409 0); 410 break; 411 } 412 } 413 414 if (VA.isRegLoc()) { 415 RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg)); 416 if (!UseBoth) 417 continue; 418 VA = ArgLocs2[i]; 419 } 420 421 assert(VA.isMemLoc()); 422 423 // Create a store off the stack pointer for this argument. 424 SDValue StackPtr = DAG.getRegister(VE::SX11, PtrVT); 425 // The argument area starts at %fp+176 in the callee frame, 426 // %sp+176 in ours. 427 SDValue PtrOff = 428 DAG.getIntPtrConstant(VA.getLocMemOffset() + ArgsBaseOffset, DL); 429 PtrOff = DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr, PtrOff); 430 MemOpChains.push_back( 431 DAG.getStore(Chain, DL, Arg, PtrOff, MachinePointerInfo())); 432 } 433 434 // Emit all stores, make sure they occur before the call. 435 if (!MemOpChains.empty()) 436 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains); 437 438 // Build a sequence of CopyToReg nodes glued together with token chain and 439 // glue operands which copy the outgoing args into registers. The InGlue is 440 // necessary since all emitted instructions must be stuck together in order 441 // to pass the live physical registers. 442 SDValue InGlue; 443 for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) { 444 Chain = DAG.getCopyToReg(Chain, DL, RegsToPass[i].first, 445 RegsToPass[i].second, InGlue); 446 InGlue = Chain.getValue(1); 447 } 448 449 // Build the operands for the call instruction itself. 450 SmallVector<SDValue, 8> Ops; 451 Ops.push_back(Chain); 452 for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) 453 Ops.push_back(DAG.getRegister(RegsToPass[i].first, 454 RegsToPass[i].second.getValueType())); 455 456 // Add a register mask operand representing the call-preserved registers. 457 const VERegisterInfo *TRI = Subtarget->getRegisterInfo(); 458 const uint32_t *Mask = 459 TRI->getCallPreservedMask(DAG.getMachineFunction(), CLI.CallConv); 460 assert(Mask && "Missing call preserved mask for calling convention"); 461 Ops.push_back(DAG.getRegisterMask(Mask)); 462 463 // Make sure the CopyToReg nodes are glued to the call instruction which 464 // consumes the registers. 465 if (InGlue.getNode()) 466 Ops.push_back(InGlue); 467 468 // Now the call itself. 469 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue); 470 Chain = DAG.getNode(VEISD::CALL, DL, NodeTys, Ops); 471 InGlue = Chain.getValue(1); 472 473 // Revert the stack pointer immediately after the call. 474 Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(ArgsSize, DL, true), 475 DAG.getIntPtrConstant(0, DL, true), InGlue, DL); 476 InGlue = Chain.getValue(1); 477 478 // Now extract the return values. This is more or less the same as 479 // LowerFormalArguments. 480 481 // Assign locations to each value returned by this call. 482 SmallVector<CCValAssign, 16> RVLocs; 483 CCState RVInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), RVLocs, 484 *DAG.getContext()); 485 486 // Set inreg flag manually for codegen generated library calls that 487 // return float. 488 if (CLI.Ins.size() == 1 && CLI.Ins[0].VT == MVT::f32 && !CLI.CB) 489 CLI.Ins[0].Flags.setInReg(); 490 491 RVInfo.AnalyzeCallResult(CLI.Ins, RetCC_VE); 492 493 // Copy all of the result registers out of their specified physreg. 494 for (unsigned i = 0; i != RVLocs.size(); ++i) { 495 CCValAssign &VA = RVLocs[i]; 496 unsigned Reg = VA.getLocReg(); 497 498 // When returning 'inreg {i32, i32 }', two consecutive i32 arguments can 499 // reside in the same register in the high and low bits. Reuse the 500 // CopyFromReg previous node to avoid duplicate copies. 501 SDValue RV; 502 if (RegisterSDNode *SrcReg = dyn_cast<RegisterSDNode>(Chain.getOperand(1))) 503 if (SrcReg->getReg() == Reg && Chain->getOpcode() == ISD::CopyFromReg) 504 RV = Chain.getValue(0); 505 506 // But usually we'll create a new CopyFromReg for a different register. 507 if (!RV.getNode()) { 508 RV = DAG.getCopyFromReg(Chain, DL, Reg, RVLocs[i].getLocVT(), InGlue); 509 Chain = RV.getValue(1); 510 InGlue = Chain.getValue(2); 511 } 512 513 // Get the high bits for i32 struct elements. 514 if (VA.getValVT() == MVT::i32 && VA.needsCustom()) 515 RV = DAG.getNode(ISD::SRL, DL, VA.getLocVT(), RV, 516 DAG.getConstant(32, DL, MVT::i32)); 517 518 // The callee promoted the return value, so insert an Assert?ext SDNode so 519 // we won't promote the value again in this function. 520 switch (VA.getLocInfo()) { 521 case CCValAssign::SExt: 522 RV = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), RV, 523 DAG.getValueType(VA.getValVT())); 524 break; 525 case CCValAssign::ZExt: 526 RV = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), RV, 527 DAG.getValueType(VA.getValVT())); 528 break; 529 case CCValAssign::BCvt: { 530 // Extract a float return value from i64 with padding. 531 // 63 31 0 532 // +------+------+ 533 // | float| 0 | 534 // +------+------+ 535 assert(VA.getLocVT() == MVT::i64); 536 assert(VA.getValVT() == MVT::f32); 537 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32); 538 RV = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, 539 MVT::f32, RV, Sub_f32), 540 0); 541 break; 542 } 543 default: 544 break; 545 } 546 547 // Truncate the register down to the return value type. 548 if (VA.isExtInLoc()) 549 RV = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), RV); 550 551 InVals.push_back(RV); 552 } 553 554 return Chain; 555 } 556 557 /// isFPImmLegal - Returns true if the target can instruction select the 558 /// specified FP immediate natively. If false, the legalizer will 559 /// materialize the FP immediate as a load from a constant pool. 560 bool VETargetLowering::isFPImmLegal(const APFloat &Imm, EVT VT, 561 bool ForCodeSize) const { 562 return VT == MVT::f32 || VT == MVT::f64; 563 } 564 565 /// Determine if the target supports unaligned memory accesses. 566 /// 567 /// This function returns true if the target allows unaligned memory accesses 568 /// of the specified type in the given address space. If true, it also returns 569 /// whether the unaligned memory access is "fast" in the last argument by 570 /// reference. This is used, for example, in situations where an array 571 /// copy/move/set is converted to a sequence of store operations. Its use 572 /// helps to ensure that such replacements don't generate code that causes an 573 /// alignment error (trap) on the target machine. 574 bool VETargetLowering::allowsMisalignedMemoryAccesses(EVT VT, 575 unsigned AddrSpace, 576 unsigned Align, 577 MachineMemOperand::Flags, 578 bool *Fast) const { 579 if (Fast) { 580 // It's fast anytime on VE 581 *Fast = true; 582 } 583 return true; 584 } 585 586 bool VETargetLowering::hasAndNot(SDValue Y) const { 587 EVT VT = Y.getValueType(); 588 589 // VE doesn't have vector and not instruction. 590 if (VT.isVector()) 591 return false; 592 593 // VE allows different immediate values for X and Y where ~X & Y. 594 // Only simm7 works for X, and only mimm works for Y on VE. However, this 595 // function is used to check whether an immediate value is OK for and-not 596 // instruction as both X and Y. Generating additional instruction to 597 // retrieve an immediate value is no good since the purpose of this 598 // function is to convert a series of 3 instructions to another series of 599 // 3 instructions with better parallelism. Therefore, we return false 600 // for all immediate values now. 601 // FIXME: Change hasAndNot function to have two operands to make it work 602 // correctly with Aurora VE. 603 if (isa<ConstantSDNode>(Y)) 604 return false; 605 606 // It's ok for generic registers. 607 return true; 608 } 609 610 VETargetLowering::VETargetLowering(const TargetMachine &TM, 611 const VESubtarget &STI) 612 : TargetLowering(TM), Subtarget(&STI) { 613 // Instructions which use registers as conditionals examine all the 614 // bits (as does the pseudo SELECT_CC expansion). I don't think it 615 // matters much whether it's ZeroOrOneBooleanContent, or 616 // ZeroOrNegativeOneBooleanContent, so, arbitrarily choose the 617 // former. 618 setBooleanContents(ZeroOrOneBooleanContent); 619 setBooleanVectorContents(ZeroOrOneBooleanContent); 620 621 // Set up the register classes. 622 addRegisterClass(MVT::i32, &VE::I32RegClass); 623 addRegisterClass(MVT::i64, &VE::I64RegClass); 624 addRegisterClass(MVT::f32, &VE::F32RegClass); 625 addRegisterClass(MVT::f64, &VE::I64RegClass); 626 627 /// Load & Store { 628 for (MVT FPVT : MVT::fp_valuetypes()) { 629 for (MVT OtherFPVT : MVT::fp_valuetypes()) { 630 // Turn FP extload into load/fpextend 631 setLoadExtAction(ISD::EXTLOAD, FPVT, OtherFPVT, Expand); 632 633 // Turn FP truncstore into trunc + store. 634 setTruncStoreAction(FPVT, OtherFPVT, Expand); 635 } 636 } 637 638 // VE doesn't have i1 sign extending load 639 for (MVT VT : MVT::integer_valuetypes()) { 640 setLoadExtAction(ISD::SEXTLOAD, VT, MVT::i1, Promote); 641 setLoadExtAction(ISD::ZEXTLOAD, VT, MVT::i1, Promote); 642 setLoadExtAction(ISD::EXTLOAD, VT, MVT::i1, Promote); 643 setTruncStoreAction(VT, MVT::i1, Expand); 644 } 645 /// } Load & Store 646 647 // Custom legalize address nodes into LO/HI parts. 648 MVT PtrVT = MVT::getIntegerVT(TM.getPointerSizeInBits(0)); 649 setOperationAction(ISD::BlockAddress, PtrVT, Custom); 650 setOperationAction(ISD::GlobalAddress, PtrVT, Custom); 651 setOperationAction(ISD::GlobalTLSAddress, PtrVT, Custom); 652 653 /// VAARG handling { 654 setOperationAction(ISD::VASTART, MVT::Other, Custom); 655 // VAARG needs to be lowered to access with 8 bytes alignment. 656 setOperationAction(ISD::VAARG, MVT::Other, Custom); 657 // Use the default implementation. 658 setOperationAction(ISD::VACOPY, MVT::Other, Expand); 659 setOperationAction(ISD::VAEND, MVT::Other, Expand); 660 /// } VAARG handling 661 662 /// Stack { 663 setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i32, Custom); 664 setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i64, Custom); 665 /// } Stack 666 667 /// Int Ops { 668 for (MVT IntVT : {MVT::i32, MVT::i64}) { 669 // VE has no REM or DIVREM operations. 670 setOperationAction(ISD::UREM, IntVT, Expand); 671 setOperationAction(ISD::SREM, IntVT, Expand); 672 setOperationAction(ISD::SDIVREM, IntVT, Expand); 673 setOperationAction(ISD::UDIVREM, IntVT, Expand); 674 675 setOperationAction(ISD::CTTZ, IntVT, Expand); 676 setOperationAction(ISD::ROTL, IntVT, Expand); 677 setOperationAction(ISD::ROTR, IntVT, Expand); 678 679 // Use isel patterns for i32 and i64 680 setOperationAction(ISD::BSWAP, IntVT, Legal); 681 setOperationAction(ISD::CTLZ, IntVT, Legal); 682 setOperationAction(ISD::CTPOP, IntVT, Legal); 683 684 // Use isel patterns for i64, Promote i32 685 LegalizeAction Act = (IntVT == MVT::i32) ? Promote : Legal; 686 setOperationAction(ISD::BITREVERSE, IntVT, Act); 687 } 688 /// } Int Ops 689 690 /// Conversion { 691 // VE doesn't have instructions for fp<->uint, so expand them by llvm 692 setOperationAction(ISD::FP_TO_UINT, MVT::i32, Promote); // use i64 693 setOperationAction(ISD::UINT_TO_FP, MVT::i32, Promote); // use i64 694 setOperationAction(ISD::FP_TO_UINT, MVT::i64, Expand); 695 setOperationAction(ISD::UINT_TO_FP, MVT::i64, Expand); 696 697 // fp16 not supported 698 for (MVT FPVT : MVT::fp_valuetypes()) { 699 setOperationAction(ISD::FP16_TO_FP, FPVT, Expand); 700 setOperationAction(ISD::FP_TO_FP16, FPVT, Expand); 701 } 702 /// } Conversion 703 704 setStackPointerRegisterToSaveRestore(VE::SX11); 705 706 // Set function alignment to 16 bytes 707 setMinFunctionAlignment(Align(16)); 708 709 // VE stores all argument by 8 bytes alignment 710 setMinStackArgumentAlignment(Align(8)); 711 712 computeRegisterProperties(Subtarget->getRegisterInfo()); 713 } 714 715 const char *VETargetLowering::getTargetNodeName(unsigned Opcode) const { 716 #define TARGET_NODE_CASE(NAME) \ 717 case VEISD::NAME: \ 718 return "VEISD::" #NAME; 719 switch ((VEISD::NodeType)Opcode) { 720 case VEISD::FIRST_NUMBER: 721 break; 722 TARGET_NODE_CASE(Lo) 723 TARGET_NODE_CASE(Hi) 724 TARGET_NODE_CASE(GETFUNPLT) 725 TARGET_NODE_CASE(GETSTACKTOP) 726 TARGET_NODE_CASE(GETTLSADDR) 727 TARGET_NODE_CASE(CALL) 728 TARGET_NODE_CASE(RET_FLAG) 729 TARGET_NODE_CASE(GLOBAL_BASE_REG) 730 } 731 #undef TARGET_NODE_CASE 732 return nullptr; 733 } 734 735 EVT VETargetLowering::getSetCCResultType(const DataLayout &, LLVMContext &, 736 EVT VT) const { 737 return MVT::i32; 738 } 739 740 // Convert to a target node and set target flags. 741 SDValue VETargetLowering::withTargetFlags(SDValue Op, unsigned TF, 742 SelectionDAG &DAG) const { 743 if (const GlobalAddressSDNode *GA = dyn_cast<GlobalAddressSDNode>(Op)) 744 return DAG.getTargetGlobalAddress(GA->getGlobal(), SDLoc(GA), 745 GA->getValueType(0), GA->getOffset(), TF); 746 747 if (const BlockAddressSDNode *BA = dyn_cast<BlockAddressSDNode>(Op)) 748 return DAG.getTargetBlockAddress(BA->getBlockAddress(), Op.getValueType(), 749 0, TF); 750 751 if (const ExternalSymbolSDNode *ES = dyn_cast<ExternalSymbolSDNode>(Op)) 752 return DAG.getTargetExternalSymbol(ES->getSymbol(), ES->getValueType(0), 753 TF); 754 755 llvm_unreachable("Unhandled address SDNode"); 756 } 757 758 // Split Op into high and low parts according to HiTF and LoTF. 759 // Return an ADD node combining the parts. 760 SDValue VETargetLowering::makeHiLoPair(SDValue Op, unsigned HiTF, unsigned LoTF, 761 SelectionDAG &DAG) const { 762 SDLoc DL(Op); 763 EVT VT = Op.getValueType(); 764 SDValue Hi = DAG.getNode(VEISD::Hi, DL, VT, withTargetFlags(Op, HiTF, DAG)); 765 SDValue Lo = DAG.getNode(VEISD::Lo, DL, VT, withTargetFlags(Op, LoTF, DAG)); 766 return DAG.getNode(ISD::ADD, DL, VT, Hi, Lo); 767 } 768 769 // Build SDNodes for producing an address from a GlobalAddress, ConstantPool, 770 // or ExternalSymbol SDNode. 771 SDValue VETargetLowering::makeAddress(SDValue Op, SelectionDAG &DAG) const { 772 SDLoc DL(Op); 773 EVT PtrVT = Op.getValueType(); 774 775 // Handle PIC mode first. VE needs a got load for every variable! 776 if (isPositionIndependent()) { 777 // GLOBAL_BASE_REG codegen'ed with call. Inform MFI that this 778 // function has calls. 779 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); 780 MFI.setHasCalls(true); 781 auto GlobalN = dyn_cast<GlobalAddressSDNode>(Op); 782 783 if (isa<ConstantPoolSDNode>(Op) || 784 (GlobalN && GlobalN->getGlobal()->hasLocalLinkage())) { 785 // Create following instructions for local linkage PIC code. 786 // lea %s35, %gotoff_lo(.LCPI0_0) 787 // and %s35, %s35, (32)0 788 // lea.sl %s35, %gotoff_hi(.LCPI0_0)(%s35) 789 // adds.l %s35, %s15, %s35 ; %s15 is GOT 790 // FIXME: use lea.sl %s35, %gotoff_hi(.LCPI0_0)(%s35, %s15) 791 SDValue HiLo = makeHiLoPair(Op, VEMCExpr::VK_VE_GOTOFF_HI32, 792 VEMCExpr::VK_VE_GOTOFF_LO32, DAG); 793 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT); 794 return DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo); 795 } 796 // Create following instructions for not local linkage PIC code. 797 // lea %s35, %got_lo(.LCPI0_0) 798 // and %s35, %s35, (32)0 799 // lea.sl %s35, %got_hi(.LCPI0_0)(%s35) 800 // adds.l %s35, %s15, %s35 ; %s15 is GOT 801 // ld %s35, (,%s35) 802 // FIXME: use lea.sl %s35, %gotoff_hi(.LCPI0_0)(%s35, %s15) 803 SDValue HiLo = makeHiLoPair(Op, VEMCExpr::VK_VE_GOT_HI32, 804 VEMCExpr::VK_VE_GOT_LO32, DAG); 805 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT); 806 SDValue AbsAddr = DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo); 807 return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), AbsAddr, 808 MachinePointerInfo::getGOT(DAG.getMachineFunction())); 809 } 810 811 // This is one of the absolute code models. 812 switch (getTargetMachine().getCodeModel()) { 813 default: 814 llvm_unreachable("Unsupported absolute code model"); 815 case CodeModel::Small: 816 case CodeModel::Medium: 817 case CodeModel::Large: 818 // abs64. 819 return makeHiLoPair(Op, VEMCExpr::VK_VE_HI32, VEMCExpr::VK_VE_LO32, DAG); 820 } 821 } 822 823 /// Custom Lower { 824 825 SDValue VETargetLowering::LowerGlobalAddress(SDValue Op, 826 SelectionDAG &DAG) const { 827 return makeAddress(Op, DAG); 828 } 829 830 SDValue VETargetLowering::LowerBlockAddress(SDValue Op, 831 SelectionDAG &DAG) const { 832 return makeAddress(Op, DAG); 833 } 834 835 SDValue 836 VETargetLowering::LowerToTLSGeneralDynamicModel(SDValue Op, 837 SelectionDAG &DAG) const { 838 SDLoc dl(Op); 839 840 // Generate the following code: 841 // t1: ch,glue = callseq_start t0, 0, 0 842 // t2: i64,ch,glue = VEISD::GETTLSADDR t1, label, t1:1 843 // t3: ch,glue = callseq_end t2, 0, 0, t2:2 844 // t4: i64,ch,glue = CopyFromReg t3, Register:i64 $sx0, t3:1 845 SDValue Label = withTargetFlags(Op, 0, DAG); 846 EVT PtrVT = Op.getValueType(); 847 848 // Lowering the machine isd will make sure everything is in the right 849 // location. 850 SDValue Chain = DAG.getEntryNode(); 851 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue); 852 const uint32_t *Mask = Subtarget->getRegisterInfo()->getCallPreservedMask( 853 DAG.getMachineFunction(), CallingConv::C); 854 Chain = DAG.getCALLSEQ_START(Chain, 64, 0, dl); 855 SDValue Args[] = {Chain, Label, DAG.getRegisterMask(Mask), Chain.getValue(1)}; 856 Chain = DAG.getNode(VEISD::GETTLSADDR, dl, NodeTys, Args); 857 Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(64, dl, true), 858 DAG.getIntPtrConstant(0, dl, true), 859 Chain.getValue(1), dl); 860 Chain = DAG.getCopyFromReg(Chain, dl, VE::SX0, PtrVT, Chain.getValue(1)); 861 862 // GETTLSADDR will be codegen'ed as call. Inform MFI that function has calls. 863 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); 864 MFI.setHasCalls(true); 865 866 // Also generate code to prepare a GOT register if it is PIC. 867 if (isPositionIndependent()) { 868 MachineFunction &MF = DAG.getMachineFunction(); 869 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF); 870 } 871 872 return Chain; 873 } 874 875 SDValue VETargetLowering::LowerGlobalTLSAddress(SDValue Op, 876 SelectionDAG &DAG) const { 877 // The current implementation of nld (2.26) doesn't allow local exec model 878 // code described in VE-tls_v1.1.pdf (*1) as its input. Instead, we always 879 // generate the general dynamic model code sequence. 880 // 881 // *1: https://www.nec.com/en/global/prod/hpc/aurora/document/VE-tls_v1.1.pdf 882 return LowerToTLSGeneralDynamicModel(Op, DAG); 883 } 884 885 SDValue VETargetLowering::LowerVASTART(SDValue Op, SelectionDAG &DAG) const { 886 MachineFunction &MF = DAG.getMachineFunction(); 887 VEMachineFunctionInfo *FuncInfo = MF.getInfo<VEMachineFunctionInfo>(); 888 auto PtrVT = getPointerTy(DAG.getDataLayout()); 889 890 // Need frame address to find the address of VarArgsFrameIndex. 891 MF.getFrameInfo().setFrameAddressIsTaken(true); 892 893 // vastart just stores the address of the VarArgsFrameIndex slot into the 894 // memory location argument. 895 SDLoc DL(Op); 896 SDValue Offset = 897 DAG.getNode(ISD::ADD, DL, PtrVT, DAG.getRegister(VE::SX9, PtrVT), 898 DAG.getIntPtrConstant(FuncInfo->getVarArgsFrameOffset(), DL)); 899 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue(); 900 return DAG.getStore(Op.getOperand(0), DL, Offset, Op.getOperand(1), 901 MachinePointerInfo(SV)); 902 } 903 904 SDValue VETargetLowering::LowerVAARG(SDValue Op, SelectionDAG &DAG) const { 905 SDNode *Node = Op.getNode(); 906 EVT VT = Node->getValueType(0); 907 SDValue InChain = Node->getOperand(0); 908 SDValue VAListPtr = Node->getOperand(1); 909 EVT PtrVT = VAListPtr.getValueType(); 910 const Value *SV = cast<SrcValueSDNode>(Node->getOperand(2))->getValue(); 911 SDLoc DL(Node); 912 SDValue VAList = 913 DAG.getLoad(PtrVT, DL, InChain, VAListPtr, MachinePointerInfo(SV)); 914 SDValue Chain = VAList.getValue(1); 915 SDValue NextPtr; 916 917 if (VT == MVT::f32) { 918 // float --> need special handling like below. 919 // 0 4 920 // +------+------+ 921 // | empty| float| 922 // +------+------+ 923 // Increment the pointer, VAList, by 8 to the next vaarg. 924 NextPtr = 925 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL)); 926 // Then, adjust VAList. 927 unsigned InternalOffset = 4; 928 VAList = DAG.getNode(ISD::ADD, DL, PtrVT, VAList, 929 DAG.getConstant(InternalOffset, DL, PtrVT)); 930 } else { 931 // Increment the pointer, VAList, by 8 to the next vaarg. 932 NextPtr = 933 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL)); 934 } 935 936 // Store the incremented VAList to the legalized pointer. 937 InChain = DAG.getStore(Chain, DL, NextPtr, VAListPtr, MachinePointerInfo(SV)); 938 939 // Load the actual argument out of the pointer VAList. 940 // We can't count on greater alignment than the word size. 941 return DAG.getLoad(VT, DL, InChain, VAList, MachinePointerInfo(), 942 std::min(PtrVT.getSizeInBits(), VT.getSizeInBits()) / 8); 943 } 944 945 SDValue VETargetLowering::lowerDYNAMIC_STACKALLOC(SDValue Op, 946 SelectionDAG &DAG) const { 947 // Generate following code. 948 // (void)__llvm_grow_stack(size); 949 // ret = GETSTACKTOP; // pseudo instruction 950 SDLoc DL(Op); 951 952 // Get the inputs. 953 SDNode *Node = Op.getNode(); 954 SDValue Chain = Op.getOperand(0); 955 SDValue Size = Op.getOperand(1); 956 MaybeAlign Alignment(Op.getConstantOperandVal(2)); 957 EVT VT = Node->getValueType(0); 958 959 // Chain the dynamic stack allocation so that it doesn't modify the stack 960 // pointer when other instructions are using the stack. 961 Chain = DAG.getCALLSEQ_START(Chain, 0, 0, DL); 962 963 const TargetFrameLowering &TFI = *Subtarget->getFrameLowering(); 964 Align StackAlign = TFI.getStackAlign(); 965 bool NeedsAlign = Alignment.valueOrOne() > StackAlign; 966 967 // Prepare arguments 968 TargetLowering::ArgListTy Args; 969 TargetLowering::ArgListEntry Entry; 970 Entry.Node = Size; 971 Entry.Ty = Entry.Node.getValueType().getTypeForEVT(*DAG.getContext()); 972 Args.push_back(Entry); 973 if (NeedsAlign) { 974 Entry.Node = DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT); 975 Entry.Ty = Entry.Node.getValueType().getTypeForEVT(*DAG.getContext()); 976 Args.push_back(Entry); 977 } 978 Type *RetTy = Type::getVoidTy(*DAG.getContext()); 979 980 EVT PtrVT = Op.getValueType(); 981 SDValue Callee; 982 if (NeedsAlign) { 983 Callee = DAG.getTargetExternalSymbol("__ve_grow_stack_align", PtrVT, 0); 984 } else { 985 Callee = DAG.getTargetExternalSymbol("__ve_grow_stack", PtrVT, 0); 986 } 987 988 TargetLowering::CallLoweringInfo CLI(DAG); 989 CLI.setDebugLoc(DL) 990 .setChain(Chain) 991 .setCallee(CallingConv::PreserveAll, RetTy, Callee, std::move(Args)) 992 .setDiscardResult(true); 993 std::pair<SDValue, SDValue> pair = LowerCallTo(CLI); 994 Chain = pair.second; 995 SDValue Result = DAG.getNode(VEISD::GETSTACKTOP, DL, VT, Chain); 996 if (NeedsAlign) { 997 Result = DAG.getNode(ISD::ADD, DL, VT, Result, 998 DAG.getConstant((Alignment->value() - 1ULL), DL, VT)); 999 Result = DAG.getNode(ISD::AND, DL, VT, Result, 1000 DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT)); 1001 } 1002 // Chain = Result.getValue(1); 1003 Chain = DAG.getCALLSEQ_END(Chain, DAG.getIntPtrConstant(0, DL, true), 1004 DAG.getIntPtrConstant(0, DL, true), SDValue(), DL); 1005 1006 SDValue Ops[2] = {Result, Chain}; 1007 return DAG.getMergeValues(Ops, DL); 1008 } 1009 1010 SDValue VETargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const { 1011 switch (Op.getOpcode()) { 1012 default: 1013 llvm_unreachable("Should not custom lower this!"); 1014 case ISD::BlockAddress: 1015 return LowerBlockAddress(Op, DAG); 1016 case ISD::DYNAMIC_STACKALLOC: 1017 return lowerDYNAMIC_STACKALLOC(Op, DAG); 1018 case ISD::GlobalAddress: 1019 return LowerGlobalAddress(Op, DAG); 1020 case ISD::GlobalTLSAddress: 1021 return LowerGlobalTLSAddress(Op, DAG); 1022 case ISD::VASTART: 1023 return LowerVASTART(Op, DAG); 1024 case ISD::VAARG: 1025 return LowerVAARG(Op, DAG); 1026 } 1027 } 1028 /// } Custom Lower 1029