1 //===- AArch64FrameLowering.cpp - AArch64 Frame Lowering -------*- C++ -*-====// 2 // 3 // The LLVM Compiler Infrastructure 4 // 5 // This file is distributed under the University of Illinois Open Source 6 // License. See LICENSE.TXT for details. 7 // 8 //===----------------------------------------------------------------------===// 9 // 10 // This file contains the AArch64 implementation of TargetFrameLowering class. 11 // 12 // On AArch64, stack frames are structured as follows: 13 // 14 // The stack grows downward. 15 // 16 // All of the individual frame areas on the frame below are optional, i.e. it's 17 // possible to create a function so that the particular area isn't present 18 // in the frame. 19 // 20 // At function entry, the "frame" looks as follows: 21 // 22 // | | Higher address 23 // |-----------------------------------| 24 // | | 25 // | arguments passed on the stack | 26 // | | 27 // |-----------------------------------| <- sp 28 // | | Lower address 29 // 30 // 31 // After the prologue has run, the frame has the following general structure. 32 // Note that this doesn't depict the case where a red-zone is used. Also, 33 // technically the last frame area (VLAs) doesn't get created until in the 34 // main function body, after the prologue is run. However, it's depicted here 35 // for completeness. 36 // 37 // | | Higher address 38 // |-----------------------------------| 39 // | | 40 // | arguments passed on the stack | 41 // | | 42 // |-----------------------------------| 43 // | | 44 // | prev_fp, prev_lr | 45 // | (a.k.a. "frame record") | 46 // |-----------------------------------| <- fp(=x29) 47 // | | 48 // | other callee-saved registers | 49 // | | 50 // |-----------------------------------| 51 // |.empty.space.to.make.part.below....| 52 // |.aligned.in.case.it.needs.more.than| (size of this area is unknown at 53 // |.the.standard.16-byte.alignment....| compile time; if present) 54 // |-----------------------------------| 55 // | | 56 // | local variables of fixed size | 57 // | including spill slots | 58 // |-----------------------------------| <- bp(not defined by ABI, 59 // |.variable-sized.local.variables....| LLVM chooses X19) 60 // |.(VLAs)............................| (size of this area is unknown at 61 // |...................................| compile time) 62 // |-----------------------------------| <- sp 63 // | | Lower address 64 // 65 // 66 // To access the data in a frame, at-compile time, a constant offset must be 67 // computable from one of the pointers (fp, bp, sp) to access it. The size 68 // of the areas with a dotted background cannot be computed at compile-time 69 // if they are present, making it required to have all three of fp, bp and 70 // sp to be set up to be able to access all contents in the frame areas, 71 // assuming all of the frame areas are non-empty. 72 // 73 // For most functions, some of the frame areas are empty. For those functions, 74 // it may not be necessary to set up fp or bp: 75 // * A base pointer is definitely needed when there are both VLAs and local 76 // variables with more-than-default alignment requirements. 77 // * A frame pointer is definitely needed when there are local variables with 78 // more-than-default alignment requirements. 79 // 80 // In some cases when a base pointer is not strictly needed, it is generated 81 // anyway when offsets from the frame pointer to access local variables become 82 // so large that the offset can't be encoded in the immediate fields of loads 83 // or stores. 84 // 85 // FIXME: also explain the redzone concept. 86 // FIXME: also explain the concept of reserved call frames. 87 // 88 //===----------------------------------------------------------------------===// 89 90 #include "AArch64FrameLowering.h" 91 #include "AArch64InstrInfo.h" 92 #include "AArch64MachineFunctionInfo.h" 93 #include "AArch64RegisterInfo.h" 94 #include "AArch64Subtarget.h" 95 #include "AArch64TargetMachine.h" 96 #include "llvm/ADT/SmallVector.h" 97 #include "llvm/ADT/Statistic.h" 98 #include "llvm/CodeGen/LivePhysRegs.h" 99 #include "llvm/CodeGen/MachineBasicBlock.h" 100 #include "llvm/CodeGen/MachineFrameInfo.h" 101 #include "llvm/CodeGen/MachineFunction.h" 102 #include "llvm/CodeGen/MachineInstr.h" 103 #include "llvm/CodeGen/MachineInstrBuilder.h" 104 #include "llvm/CodeGen/MachineMemOperand.h" 105 #include "llvm/CodeGen/MachineModuleInfo.h" 106 #include "llvm/CodeGen/MachineOperand.h" 107 #include "llvm/CodeGen/MachineRegisterInfo.h" 108 #include "llvm/CodeGen/RegisterScavenging.h" 109 #include "llvm/IR/Attributes.h" 110 #include "llvm/IR/CallingConv.h" 111 #include "llvm/IR/DataLayout.h" 112 #include "llvm/IR/DebugLoc.h" 113 #include "llvm/IR/Function.h" 114 #include "llvm/MC/MCDwarf.h" 115 #include "llvm/Support/CommandLine.h" 116 #include "llvm/Support/Debug.h" 117 #include "llvm/Support/ErrorHandling.h" 118 #include "llvm/Support/MathExtras.h" 119 #include "llvm/Support/raw_ostream.h" 120 #include "llvm/Target/TargetInstrInfo.h" 121 #include "llvm/Target/TargetMachine.h" 122 #include "llvm/Target/TargetOptions.h" 123 #include "llvm/Target/TargetRegisterInfo.h" 124 #include "llvm/Target/TargetSubtargetInfo.h" 125 #include <cassert> 126 #include <cstdint> 127 #include <iterator> 128 #include <vector> 129 130 using namespace llvm; 131 132 #define DEBUG_TYPE "frame-info" 133 134 static cl::opt<bool> EnableRedZone("aarch64-redzone", 135 cl::desc("enable use of redzone on AArch64"), 136 cl::init(false), cl::Hidden); 137 138 STATISTIC(NumRedZoneFunctions, "Number of functions using red zone"); 139 140 bool AArch64FrameLowering::canUseRedZone(const MachineFunction &MF) const { 141 if (!EnableRedZone) 142 return false; 143 // Don't use the red zone if the function explicitly asks us not to. 144 // This is typically used for kernel code. 145 if (MF.getFunction()->hasFnAttribute(Attribute::NoRedZone)) 146 return false; 147 148 const MachineFrameInfo &MFI = MF.getFrameInfo(); 149 const AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 150 unsigned NumBytes = AFI->getLocalStackSize(); 151 152 return !(MFI.hasCalls() || hasFP(MF) || NumBytes > 128); 153 } 154 155 /// hasFP - Return true if the specified function should have a dedicated frame 156 /// pointer register. 157 bool AArch64FrameLowering::hasFP(const MachineFunction &MF) const { 158 const MachineFrameInfo &MFI = MF.getFrameInfo(); 159 const TargetRegisterInfo *RegInfo = MF.getSubtarget().getRegisterInfo(); 160 // Retain behavior of always omitting the FP for leaf functions when possible. 161 return (MFI.hasCalls() && 162 MF.getTarget().Options.DisableFramePointerElim(MF)) || 163 MFI.hasVarSizedObjects() || MFI.isFrameAddressTaken() || 164 MFI.hasStackMap() || MFI.hasPatchPoint() || 165 RegInfo->needsStackRealignment(MF); 166 } 167 168 /// hasReservedCallFrame - Under normal circumstances, when a frame pointer is 169 /// not required, we reserve argument space for call sites in the function 170 /// immediately on entry to the current function. This eliminates the need for 171 /// add/sub sp brackets around call sites. Returns true if the call frame is 172 /// included as part of the stack frame. 173 bool 174 AArch64FrameLowering::hasReservedCallFrame(const MachineFunction &MF) const { 175 return !MF.getFrameInfo().hasVarSizedObjects(); 176 } 177 178 MachineBasicBlock::iterator AArch64FrameLowering::eliminateCallFramePseudoInstr( 179 MachineFunction &MF, MachineBasicBlock &MBB, 180 MachineBasicBlock::iterator I) const { 181 const AArch64InstrInfo *TII = 182 static_cast<const AArch64InstrInfo *>(MF.getSubtarget().getInstrInfo()); 183 DebugLoc DL = I->getDebugLoc(); 184 unsigned Opc = I->getOpcode(); 185 bool IsDestroy = Opc == TII->getCallFrameDestroyOpcode(); 186 uint64_t CalleePopAmount = IsDestroy ? I->getOperand(1).getImm() : 0; 187 188 const TargetFrameLowering *TFI = MF.getSubtarget().getFrameLowering(); 189 if (!TFI->hasReservedCallFrame(MF)) { 190 unsigned Align = getStackAlignment(); 191 192 int64_t Amount = I->getOperand(0).getImm(); 193 Amount = alignTo(Amount, Align); 194 if (!IsDestroy) 195 Amount = -Amount; 196 197 // N.b. if CalleePopAmount is valid but zero (i.e. callee would pop, but it 198 // doesn't have to pop anything), then the first operand will be zero too so 199 // this adjustment is a no-op. 200 if (CalleePopAmount == 0) { 201 // FIXME: in-function stack adjustment for calls is limited to 24-bits 202 // because there's no guaranteed temporary register available. 203 // 204 // ADD/SUB (immediate) has only LSL #0 and LSL #12 available. 205 // 1) For offset <= 12-bit, we use LSL #0 206 // 2) For 12-bit <= offset <= 24-bit, we use two instructions. One uses 207 // LSL #0, and the other uses LSL #12. 208 // 209 // Most call frames will be allocated at the start of a function so 210 // this is OK, but it is a limitation that needs dealing with. 211 assert(Amount > -0xffffff && Amount < 0xffffff && "call frame too large"); 212 emitFrameOffset(MBB, I, DL, AArch64::SP, AArch64::SP, Amount, TII); 213 } 214 } else if (CalleePopAmount != 0) { 215 // If the calling convention demands that the callee pops arguments from the 216 // stack, we want to add it back if we have a reserved call frame. 217 assert(CalleePopAmount < 0xffffff && "call frame too large"); 218 emitFrameOffset(MBB, I, DL, AArch64::SP, AArch64::SP, -CalleePopAmount, 219 TII); 220 } 221 return MBB.erase(I); 222 } 223 224 void AArch64FrameLowering::emitCalleeSavedFrameMoves( 225 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI) const { 226 MachineFunction &MF = *MBB.getParent(); 227 MachineFrameInfo &MFI = MF.getFrameInfo(); 228 const TargetSubtargetInfo &STI = MF.getSubtarget(); 229 const MCRegisterInfo *MRI = STI.getRegisterInfo(); 230 const TargetInstrInfo *TII = STI.getInstrInfo(); 231 DebugLoc DL = MBB.findDebugLoc(MBBI); 232 233 // Add callee saved registers to move list. 234 const std::vector<CalleeSavedInfo> &CSI = MFI.getCalleeSavedInfo(); 235 if (CSI.empty()) 236 return; 237 238 for (const auto &Info : CSI) { 239 unsigned Reg = Info.getReg(); 240 int64_t Offset = 241 MFI.getObjectOffset(Info.getFrameIdx()) - getOffsetOfLocalArea(); 242 unsigned DwarfReg = MRI->getDwarfRegNum(Reg, true); 243 unsigned CFIIndex = MF.addFrameInst( 244 MCCFIInstruction::createOffset(nullptr, DwarfReg, Offset)); 245 BuildMI(MBB, MBBI, DL, TII->get(TargetOpcode::CFI_INSTRUCTION)) 246 .addCFIIndex(CFIIndex) 247 .setMIFlags(MachineInstr::FrameSetup); 248 } 249 } 250 251 // Find a scratch register that we can use at the start of the prologue to 252 // re-align the stack pointer. We avoid using callee-save registers since they 253 // may appear to be free when this is called from canUseAsPrologue (during 254 // shrink wrapping), but then no longer be free when this is called from 255 // emitPrologue. 256 // 257 // FIXME: This is a bit conservative, since in the above case we could use one 258 // of the callee-save registers as a scratch temp to re-align the stack pointer, 259 // but we would then have to make sure that we were in fact saving at least one 260 // callee-save register in the prologue, which is additional complexity that 261 // doesn't seem worth the benefit. 262 static unsigned findScratchNonCalleeSaveRegister(MachineBasicBlock *MBB) { 263 MachineFunction *MF = MBB->getParent(); 264 265 // If MBB is an entry block, use X9 as the scratch register 266 if (&MF->front() == MBB) 267 return AArch64::X9; 268 269 const TargetRegisterInfo &TRI = *MF->getSubtarget().getRegisterInfo(); 270 LivePhysRegs LiveRegs(&TRI); 271 LiveRegs.addLiveIns(*MBB); 272 273 // Mark callee saved registers as used so we will not choose them. 274 const AArch64Subtarget &Subtarget = MF->getSubtarget<AArch64Subtarget>(); 275 const AArch64RegisterInfo *RegInfo = Subtarget.getRegisterInfo(); 276 const MCPhysReg *CSRegs = RegInfo->getCalleeSavedRegs(MF); 277 for (unsigned i = 0; CSRegs[i]; ++i) 278 LiveRegs.addReg(CSRegs[i]); 279 280 // Prefer X9 since it was historically used for the prologue scratch reg. 281 const MachineRegisterInfo &MRI = MF->getRegInfo(); 282 if (LiveRegs.available(MRI, AArch64::X9)) 283 return AArch64::X9; 284 285 for (unsigned Reg : AArch64::GPR64RegClass) { 286 if (LiveRegs.available(MRI, Reg)) 287 return Reg; 288 } 289 return AArch64::NoRegister; 290 } 291 292 bool AArch64FrameLowering::canUseAsPrologue( 293 const MachineBasicBlock &MBB) const { 294 const MachineFunction *MF = MBB.getParent(); 295 MachineBasicBlock *TmpMBB = const_cast<MachineBasicBlock *>(&MBB); 296 const AArch64Subtarget &Subtarget = MF->getSubtarget<AArch64Subtarget>(); 297 const AArch64RegisterInfo *RegInfo = Subtarget.getRegisterInfo(); 298 299 // Don't need a scratch register if we're not going to re-align the stack. 300 if (!RegInfo->needsStackRealignment(*MF)) 301 return true; 302 // Otherwise, we can use any block as long as it has a scratch register 303 // available. 304 return findScratchNonCalleeSaveRegister(TmpMBB) != AArch64::NoRegister; 305 } 306 307 bool AArch64FrameLowering::shouldCombineCSRLocalStackBump( 308 MachineFunction &MF, unsigned StackBumpBytes) const { 309 AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 310 const MachineFrameInfo &MFI = MF.getFrameInfo(); 311 const AArch64Subtarget &Subtarget = MF.getSubtarget<AArch64Subtarget>(); 312 const AArch64RegisterInfo *RegInfo = Subtarget.getRegisterInfo(); 313 314 if (AFI->getLocalStackSize() == 0) 315 return false; 316 317 // 512 is the maximum immediate for stp/ldp that will be used for 318 // callee-save save/restores 319 if (StackBumpBytes >= 512) 320 return false; 321 322 if (MFI.hasVarSizedObjects()) 323 return false; 324 325 if (RegInfo->needsStackRealignment(MF)) 326 return false; 327 328 // This isn't strictly necessary, but it simplifies things a bit since the 329 // current RedZone handling code assumes the SP is adjusted by the 330 // callee-save save/restore code. 331 if (canUseRedZone(MF)) 332 return false; 333 334 return true; 335 } 336 337 // Convert callee-save register save/restore instruction to do stack pointer 338 // decrement/increment to allocate/deallocate the callee-save stack area by 339 // converting store/load to use pre/post increment version. 340 static MachineBasicBlock::iterator convertCalleeSaveRestoreToSPPrePostIncDec( 341 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, 342 const DebugLoc &DL, const TargetInstrInfo *TII, int CSStackSizeInc) { 343 unsigned NewOpc; 344 bool NewIsUnscaled = false; 345 switch (MBBI->getOpcode()) { 346 default: 347 llvm_unreachable("Unexpected callee-save save/restore opcode!"); 348 case AArch64::STPXi: 349 NewOpc = AArch64::STPXpre; 350 break; 351 case AArch64::STPDi: 352 NewOpc = AArch64::STPDpre; 353 break; 354 case AArch64::STRXui: 355 NewOpc = AArch64::STRXpre; 356 NewIsUnscaled = true; 357 break; 358 case AArch64::STRDui: 359 NewOpc = AArch64::STRDpre; 360 NewIsUnscaled = true; 361 break; 362 case AArch64::LDPXi: 363 NewOpc = AArch64::LDPXpost; 364 break; 365 case AArch64::LDPDi: 366 NewOpc = AArch64::LDPDpost; 367 break; 368 case AArch64::LDRXui: 369 NewOpc = AArch64::LDRXpost; 370 NewIsUnscaled = true; 371 break; 372 case AArch64::LDRDui: 373 NewOpc = AArch64::LDRDpost; 374 NewIsUnscaled = true; 375 break; 376 } 377 378 MachineInstrBuilder MIB = BuildMI(MBB, MBBI, DL, TII->get(NewOpc)); 379 MIB.addReg(AArch64::SP, RegState::Define); 380 381 // Copy all operands other than the immediate offset. 382 unsigned OpndIdx = 0; 383 for (unsigned OpndEnd = MBBI->getNumOperands() - 1; OpndIdx < OpndEnd; 384 ++OpndIdx) 385 MIB.add(MBBI->getOperand(OpndIdx)); 386 387 assert(MBBI->getOperand(OpndIdx).getImm() == 0 && 388 "Unexpected immediate offset in first/last callee-save save/restore " 389 "instruction!"); 390 assert(MBBI->getOperand(OpndIdx - 1).getReg() == AArch64::SP && 391 "Unexpected base register in callee-save save/restore instruction!"); 392 // Last operand is immediate offset that needs fixing. 393 assert(CSStackSizeInc % 8 == 0); 394 int64_t CSStackSizeIncImm = CSStackSizeInc; 395 if (!NewIsUnscaled) 396 CSStackSizeIncImm /= 8; 397 MIB.addImm(CSStackSizeIncImm); 398 399 MIB.setMIFlags(MBBI->getFlags()); 400 MIB.setMemRefs(MBBI->memoperands_begin(), MBBI->memoperands_end()); 401 402 return std::prev(MBB.erase(MBBI)); 403 } 404 405 // Fixup callee-save register save/restore instructions to take into account 406 // combined SP bump by adding the local stack size to the stack offsets. 407 static void fixupCalleeSaveRestoreStackOffset(MachineInstr &MI, 408 unsigned LocalStackSize) { 409 unsigned Opc = MI.getOpcode(); 410 (void)Opc; 411 assert((Opc == AArch64::STPXi || Opc == AArch64::STPDi || 412 Opc == AArch64::STRXui || Opc == AArch64::STRDui || 413 Opc == AArch64::LDPXi || Opc == AArch64::LDPDi || 414 Opc == AArch64::LDRXui || Opc == AArch64::LDRDui) && 415 "Unexpected callee-save save/restore opcode!"); 416 417 unsigned OffsetIdx = MI.getNumExplicitOperands() - 1; 418 assert(MI.getOperand(OffsetIdx - 1).getReg() == AArch64::SP && 419 "Unexpected base register in callee-save save/restore instruction!"); 420 // Last operand is immediate offset that needs fixing. 421 MachineOperand &OffsetOpnd = MI.getOperand(OffsetIdx); 422 // All generated opcodes have scaled offsets. 423 assert(LocalStackSize % 8 == 0); 424 OffsetOpnd.setImm(OffsetOpnd.getImm() + LocalStackSize / 8); 425 } 426 427 void AArch64FrameLowering::emitPrologue(MachineFunction &MF, 428 MachineBasicBlock &MBB) const { 429 MachineBasicBlock::iterator MBBI = MBB.begin(); 430 const MachineFrameInfo &MFI = MF.getFrameInfo(); 431 const Function *Fn = MF.getFunction(); 432 const AArch64Subtarget &Subtarget = MF.getSubtarget<AArch64Subtarget>(); 433 const AArch64RegisterInfo *RegInfo = Subtarget.getRegisterInfo(); 434 const TargetInstrInfo *TII = Subtarget.getInstrInfo(); 435 MachineModuleInfo &MMI = MF.getMMI(); 436 AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 437 bool needsFrameMoves = MMI.hasDebugInfo() || Fn->needsUnwindTableEntry(); 438 bool HasFP = hasFP(MF); 439 440 // Debug location must be unknown since the first debug location is used 441 // to determine the end of the prologue. 442 DebugLoc DL; 443 444 // All calls are tail calls in GHC calling conv, and functions have no 445 // prologue/epilogue. 446 if (MF.getFunction()->getCallingConv() == CallingConv::GHC) 447 return; 448 449 int NumBytes = (int)MFI.getStackSize(); 450 if (!AFI->hasStackFrame()) { 451 assert(!HasFP && "unexpected function without stack frame but with FP"); 452 453 // All of the stack allocation is for locals. 454 AFI->setLocalStackSize(NumBytes); 455 456 if (!NumBytes) 457 return; 458 // REDZONE: If the stack size is less than 128 bytes, we don't need 459 // to actually allocate. 460 if (canUseRedZone(MF)) 461 ++NumRedZoneFunctions; 462 else { 463 emitFrameOffset(MBB, MBBI, DL, AArch64::SP, AArch64::SP, -NumBytes, TII, 464 MachineInstr::FrameSetup); 465 466 // Label used to tie together the PROLOG_LABEL and the MachineMoves. 467 MCSymbol *FrameLabel = MMI.getContext().createTempSymbol(); 468 // Encode the stack size of the leaf function. 469 unsigned CFIIndex = MF.addFrameInst( 470 MCCFIInstruction::createDefCfaOffset(FrameLabel, -NumBytes)); 471 BuildMI(MBB, MBBI, DL, TII->get(TargetOpcode::CFI_INSTRUCTION)) 472 .addCFIIndex(CFIIndex) 473 .setMIFlags(MachineInstr::FrameSetup); 474 } 475 return; 476 } 477 478 auto CSStackSize = AFI->getCalleeSavedStackSize(); 479 // All of the remaining stack allocations are for locals. 480 AFI->setLocalStackSize(NumBytes - CSStackSize); 481 482 bool CombineSPBump = shouldCombineCSRLocalStackBump(MF, NumBytes); 483 if (CombineSPBump) { 484 emitFrameOffset(MBB, MBBI, DL, AArch64::SP, AArch64::SP, -NumBytes, TII, 485 MachineInstr::FrameSetup); 486 NumBytes = 0; 487 } else if (CSStackSize != 0) { 488 MBBI = convertCalleeSaveRestoreToSPPrePostIncDec(MBB, MBBI, DL, TII, 489 -CSStackSize); 490 NumBytes -= CSStackSize; 491 } 492 assert(NumBytes >= 0 && "Negative stack allocation size!?"); 493 494 // Move past the saves of the callee-saved registers, fixing up the offsets 495 // and pre-inc if we decided to combine the callee-save and local stack 496 // pointer bump above. 497 MachineBasicBlock::iterator End = MBB.end(); 498 while (MBBI != End && MBBI->getFlag(MachineInstr::FrameSetup)) { 499 if (CombineSPBump) 500 fixupCalleeSaveRestoreStackOffset(*MBBI, AFI->getLocalStackSize()); 501 ++MBBI; 502 } 503 if (HasFP) { 504 // Only set up FP if we actually need to. Frame pointer is fp = sp - 16. 505 int FPOffset = CSStackSize - 16; 506 if (CombineSPBump) 507 FPOffset += AFI->getLocalStackSize(); 508 509 // Issue sub fp, sp, FPOffset or 510 // mov fp,sp when FPOffset is zero. 511 // Note: All stores of callee-saved registers are marked as "FrameSetup". 512 // This code marks the instruction(s) that set the FP also. 513 emitFrameOffset(MBB, MBBI, DL, AArch64::FP, AArch64::SP, FPOffset, TII, 514 MachineInstr::FrameSetup); 515 } 516 517 // Allocate space for the rest of the frame. 518 if (NumBytes) { 519 const bool NeedsRealignment = RegInfo->needsStackRealignment(MF); 520 unsigned scratchSPReg = AArch64::SP; 521 522 if (NeedsRealignment) { 523 scratchSPReg = findScratchNonCalleeSaveRegister(&MBB); 524 assert(scratchSPReg != AArch64::NoRegister); 525 } 526 527 // If we're a leaf function, try using the red zone. 528 if (!canUseRedZone(MF)) 529 // FIXME: in the case of dynamic re-alignment, NumBytes doesn't have 530 // the correct value here, as NumBytes also includes padding bytes, 531 // which shouldn't be counted here. 532 emitFrameOffset(MBB, MBBI, DL, scratchSPReg, AArch64::SP, -NumBytes, TII, 533 MachineInstr::FrameSetup); 534 535 if (NeedsRealignment) { 536 const unsigned Alignment = MFI.getMaxAlignment(); 537 const unsigned NrBitsToZero = countTrailingZeros(Alignment); 538 assert(NrBitsToZero > 1); 539 assert(scratchSPReg != AArch64::SP); 540 541 // SUB X9, SP, NumBytes 542 // -- X9 is temporary register, so shouldn't contain any live data here, 543 // -- free to use. This is already produced by emitFrameOffset above. 544 // AND SP, X9, 0b11111...0000 545 // The logical immediates have a non-trivial encoding. The following 546 // formula computes the encoded immediate with all ones but 547 // NrBitsToZero zero bits as least significant bits. 548 uint32_t andMaskEncoded = (1 << 12) // = N 549 | ((64 - NrBitsToZero) << 6) // immr 550 | ((64 - NrBitsToZero - 1) << 0); // imms 551 552 BuildMI(MBB, MBBI, DL, TII->get(AArch64::ANDXri), AArch64::SP) 553 .addReg(scratchSPReg, RegState::Kill) 554 .addImm(andMaskEncoded); 555 AFI->setStackRealigned(true); 556 } 557 } 558 559 // If we need a base pointer, set it up here. It's whatever the value of the 560 // stack pointer is at this point. Any variable size objects will be allocated 561 // after this, so we can still use the base pointer to reference locals. 562 // 563 // FIXME: Clarify FrameSetup flags here. 564 // Note: Use emitFrameOffset() like above for FP if the FrameSetup flag is 565 // needed. 566 if (RegInfo->hasBasePointer(MF)) { 567 TII->copyPhysReg(MBB, MBBI, DL, RegInfo->getBaseRegister(), AArch64::SP, 568 false); 569 } 570 571 if (needsFrameMoves) { 572 const DataLayout &TD = MF.getDataLayout(); 573 const int StackGrowth = -TD.getPointerSize(0); 574 unsigned FramePtr = RegInfo->getFrameRegister(MF); 575 // An example of the prologue: 576 // 577 // .globl __foo 578 // .align 2 579 // __foo: 580 // Ltmp0: 581 // .cfi_startproc 582 // .cfi_personality 155, ___gxx_personality_v0 583 // Leh_func_begin: 584 // .cfi_lsda 16, Lexception33 585 // 586 // stp xa,bx, [sp, -#offset]! 587 // ... 588 // stp x28, x27, [sp, #offset-32] 589 // stp fp, lr, [sp, #offset-16] 590 // add fp, sp, #offset - 16 591 // sub sp, sp, #1360 592 // 593 // The Stack: 594 // +-------------------------------------------+ 595 // 10000 | ........ | ........ | ........ | ........ | 596 // 10004 | ........ | ........ | ........ | ........ | 597 // +-------------------------------------------+ 598 // 10008 | ........ | ........ | ........ | ........ | 599 // 1000c | ........ | ........ | ........ | ........ | 600 // +===========================================+ 601 // 10010 | X28 Register | 602 // 10014 | X28 Register | 603 // +-------------------------------------------+ 604 // 10018 | X27 Register | 605 // 1001c | X27 Register | 606 // +===========================================+ 607 // 10020 | Frame Pointer | 608 // 10024 | Frame Pointer | 609 // +-------------------------------------------+ 610 // 10028 | Link Register | 611 // 1002c | Link Register | 612 // +===========================================+ 613 // 10030 | ........ | ........ | ........ | ........ | 614 // 10034 | ........ | ........ | ........ | ........ | 615 // +-------------------------------------------+ 616 // 10038 | ........ | ........ | ........ | ........ | 617 // 1003c | ........ | ........ | ........ | ........ | 618 // +-------------------------------------------+ 619 // 620 // [sp] = 10030 :: >>initial value<< 621 // sp = 10020 :: stp fp, lr, [sp, #-16]! 622 // fp = sp == 10020 :: mov fp, sp 623 // [sp] == 10020 :: stp x28, x27, [sp, #-16]! 624 // sp == 10010 :: >>final value<< 625 // 626 // The frame pointer (w29) points to address 10020. If we use an offset of 627 // '16' from 'w29', we get the CFI offsets of -8 for w30, -16 for w29, -24 628 // for w27, and -32 for w28: 629 // 630 // Ltmp1: 631 // .cfi_def_cfa w29, 16 632 // Ltmp2: 633 // .cfi_offset w30, -8 634 // Ltmp3: 635 // .cfi_offset w29, -16 636 // Ltmp4: 637 // .cfi_offset w27, -24 638 // Ltmp5: 639 // .cfi_offset w28, -32 640 641 if (HasFP) { 642 // Define the current CFA rule to use the provided FP. 643 unsigned Reg = RegInfo->getDwarfRegNum(FramePtr, true); 644 unsigned CFIIndex = MF.addFrameInst( 645 MCCFIInstruction::createDefCfa(nullptr, Reg, 2 * StackGrowth)); 646 BuildMI(MBB, MBBI, DL, TII->get(TargetOpcode::CFI_INSTRUCTION)) 647 .addCFIIndex(CFIIndex) 648 .setMIFlags(MachineInstr::FrameSetup); 649 } else { 650 // Encode the stack size of the leaf function. 651 unsigned CFIIndex = MF.addFrameInst( 652 MCCFIInstruction::createDefCfaOffset(nullptr, -MFI.getStackSize())); 653 BuildMI(MBB, MBBI, DL, TII->get(TargetOpcode::CFI_INSTRUCTION)) 654 .addCFIIndex(CFIIndex) 655 .setMIFlags(MachineInstr::FrameSetup); 656 } 657 658 // Now emit the moves for whatever callee saved regs we have (including FP, 659 // LR if those are saved). 660 emitCalleeSavedFrameMoves(MBB, MBBI); 661 } 662 } 663 664 void AArch64FrameLowering::emitEpilogue(MachineFunction &MF, 665 MachineBasicBlock &MBB) const { 666 MachineBasicBlock::iterator MBBI = MBB.getLastNonDebugInstr(); 667 MachineFrameInfo &MFI = MF.getFrameInfo(); 668 const AArch64Subtarget &Subtarget = MF.getSubtarget<AArch64Subtarget>(); 669 const TargetInstrInfo *TII = Subtarget.getInstrInfo(); 670 DebugLoc DL; 671 bool IsTailCallReturn = false; 672 if (MBB.end() != MBBI) { 673 DL = MBBI->getDebugLoc(); 674 unsigned RetOpcode = MBBI->getOpcode(); 675 IsTailCallReturn = RetOpcode == AArch64::TCRETURNdi || 676 RetOpcode == AArch64::TCRETURNri; 677 } 678 int NumBytes = MFI.getStackSize(); 679 const AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 680 681 // All calls are tail calls in GHC calling conv, and functions have no 682 // prologue/epilogue. 683 if (MF.getFunction()->getCallingConv() == CallingConv::GHC) 684 return; 685 686 // Initial and residual are named for consistency with the prologue. Note that 687 // in the epilogue, the residual adjustment is executed first. 688 uint64_t ArgumentPopSize = 0; 689 if (IsTailCallReturn) { 690 MachineOperand &StackAdjust = MBBI->getOperand(1); 691 692 // For a tail-call in a callee-pops-arguments environment, some or all of 693 // the stack may actually be in use for the call's arguments, this is 694 // calculated during LowerCall and consumed here... 695 ArgumentPopSize = StackAdjust.getImm(); 696 } else { 697 // ... otherwise the amount to pop is *all* of the argument space, 698 // conveniently stored in the MachineFunctionInfo by 699 // LowerFormalArguments. This will, of course, be zero for the C calling 700 // convention. 701 ArgumentPopSize = AFI->getArgumentStackToRestore(); 702 } 703 704 // The stack frame should be like below, 705 // 706 // ---------------------- --- 707 // | | | 708 // | BytesInStackArgArea| CalleeArgStackSize 709 // | (NumReusableBytes) | (of tail call) 710 // | | --- 711 // | | | 712 // ---------------------| --- | 713 // | | | | 714 // | CalleeSavedReg | | | 715 // | (CalleeSavedStackSize)| | | 716 // | | | | 717 // ---------------------| | NumBytes 718 // | | StackSize (StackAdjustUp) 719 // | LocalStackSize | | | 720 // | (covering callee | | | 721 // | args) | | | 722 // | | | | 723 // ---------------------- --- --- 724 // 725 // So NumBytes = StackSize + BytesInStackArgArea - CalleeArgStackSize 726 // = StackSize + ArgumentPopSize 727 // 728 // AArch64TargetLowering::LowerCall figures out ArgumentPopSize and keeps 729 // it as the 2nd argument of AArch64ISD::TC_RETURN. 730 731 auto CSStackSize = AFI->getCalleeSavedStackSize(); 732 bool CombineSPBump = shouldCombineCSRLocalStackBump(MF, NumBytes); 733 734 if (!CombineSPBump && CSStackSize != 0) 735 convertCalleeSaveRestoreToSPPrePostIncDec( 736 MBB, std::prev(MBB.getFirstTerminator()), DL, TII, CSStackSize); 737 738 // Move past the restores of the callee-saved registers. 739 MachineBasicBlock::iterator LastPopI = MBB.getFirstTerminator(); 740 MachineBasicBlock::iterator Begin = MBB.begin(); 741 while (LastPopI != Begin) { 742 --LastPopI; 743 if (!LastPopI->getFlag(MachineInstr::FrameDestroy)) { 744 ++LastPopI; 745 break; 746 } else if (CombineSPBump) 747 fixupCalleeSaveRestoreStackOffset(*LastPopI, AFI->getLocalStackSize()); 748 } 749 750 // If there is a single SP update, insert it before the ret and we're done. 751 if (CombineSPBump) { 752 emitFrameOffset(MBB, MBB.getFirstTerminator(), DL, AArch64::SP, AArch64::SP, 753 NumBytes + ArgumentPopSize, TII, 754 MachineInstr::FrameDestroy); 755 return; 756 } 757 758 NumBytes -= CSStackSize; 759 assert(NumBytes >= 0 && "Negative stack allocation size!?"); 760 761 if (!hasFP(MF)) { 762 bool RedZone = canUseRedZone(MF); 763 // If this was a redzone leaf function, we don't need to restore the 764 // stack pointer (but we may need to pop stack args for fastcc). 765 if (RedZone && ArgumentPopSize == 0) 766 return; 767 768 bool NoCalleeSaveRestore = CSStackSize == 0; 769 int StackRestoreBytes = RedZone ? 0 : NumBytes; 770 if (NoCalleeSaveRestore) 771 StackRestoreBytes += ArgumentPopSize; 772 emitFrameOffset(MBB, LastPopI, DL, AArch64::SP, AArch64::SP, 773 StackRestoreBytes, TII, MachineInstr::FrameDestroy); 774 // If we were able to combine the local stack pop with the argument pop, 775 // then we're done. 776 if (NoCalleeSaveRestore || ArgumentPopSize == 0) 777 return; 778 NumBytes = 0; 779 } 780 781 // Restore the original stack pointer. 782 // FIXME: Rather than doing the math here, we should instead just use 783 // non-post-indexed loads for the restores if we aren't actually going to 784 // be able to save any instructions. 785 if (MFI.hasVarSizedObjects() || AFI->isStackRealigned()) 786 emitFrameOffset(MBB, LastPopI, DL, AArch64::SP, AArch64::FP, 787 -CSStackSize + 16, TII, MachineInstr::FrameDestroy); 788 else if (NumBytes) 789 emitFrameOffset(MBB, LastPopI, DL, AArch64::SP, AArch64::SP, NumBytes, TII, 790 MachineInstr::FrameDestroy); 791 792 // This must be placed after the callee-save restore code because that code 793 // assumes the SP is at the same location as it was after the callee-save save 794 // code in the prologue. 795 if (ArgumentPopSize) 796 emitFrameOffset(MBB, MBB.getFirstTerminator(), DL, AArch64::SP, AArch64::SP, 797 ArgumentPopSize, TII, MachineInstr::FrameDestroy); 798 } 799 800 /// getFrameIndexReference - Provide a base+offset reference to an FI slot for 801 /// debug info. It's the same as what we use for resolving the code-gen 802 /// references for now. FIXME: This can go wrong when references are 803 /// SP-relative and simple call frames aren't used. 804 int AArch64FrameLowering::getFrameIndexReference(const MachineFunction &MF, 805 int FI, 806 unsigned &FrameReg) const { 807 return resolveFrameIndexReference(MF, FI, FrameReg); 808 } 809 810 int AArch64FrameLowering::resolveFrameIndexReference(const MachineFunction &MF, 811 int FI, unsigned &FrameReg, 812 bool PreferFP) const { 813 const MachineFrameInfo &MFI = MF.getFrameInfo(); 814 const AArch64RegisterInfo *RegInfo = static_cast<const AArch64RegisterInfo *>( 815 MF.getSubtarget().getRegisterInfo()); 816 const AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 817 int FPOffset = MFI.getObjectOffset(FI) + 16; 818 int Offset = MFI.getObjectOffset(FI) + MFI.getStackSize(); 819 bool isFixed = MFI.isFixedObjectIndex(FI); 820 821 // Use frame pointer to reference fixed objects. Use it for locals if 822 // there are VLAs or a dynamically realigned SP (and thus the SP isn't 823 // reliable as a base). Make sure useFPForScavengingIndex() does the 824 // right thing for the emergency spill slot. 825 bool UseFP = false; 826 if (AFI->hasStackFrame()) { 827 // Note: Keeping the following as multiple 'if' statements rather than 828 // merging to a single expression for readability. 829 // 830 // Argument access should always use the FP. 831 if (isFixed) { 832 UseFP = hasFP(MF); 833 } else if (hasFP(MF) && !RegInfo->hasBasePointer(MF) && 834 !RegInfo->needsStackRealignment(MF)) { 835 // Use SP or FP, whichever gives us the best chance of the offset 836 // being in range for direct access. If the FPOffset is positive, 837 // that'll always be best, as the SP will be even further away. 838 // If the FPOffset is negative, we have to keep in mind that the 839 // available offset range for negative offsets is smaller than for 840 // positive ones. If we have variable sized objects, we're stuck with 841 // using the FP regardless, though, as the SP offset is unknown 842 // and we don't have a base pointer available. If an offset is 843 // available via the FP and the SP, use whichever is closest. 844 if (PreferFP || MFI.hasVarSizedObjects() || FPOffset >= 0 || 845 (FPOffset >= -256 && Offset > -FPOffset)) 846 UseFP = true; 847 } 848 } 849 850 assert((isFixed || !RegInfo->needsStackRealignment(MF) || !UseFP) && 851 "In the presence of dynamic stack pointer realignment, " 852 "non-argument objects cannot be accessed through the frame pointer"); 853 854 if (UseFP) { 855 FrameReg = RegInfo->getFrameRegister(MF); 856 return FPOffset; 857 } 858 859 // Use the base pointer if we have one. 860 if (RegInfo->hasBasePointer(MF)) 861 FrameReg = RegInfo->getBaseRegister(); 862 else { 863 FrameReg = AArch64::SP; 864 // If we're using the red zone for this function, the SP won't actually 865 // be adjusted, so the offsets will be negative. They're also all 866 // within range of the signed 9-bit immediate instructions. 867 if (canUseRedZone(MF)) 868 Offset -= AFI->getLocalStackSize(); 869 } 870 871 return Offset; 872 } 873 874 static unsigned getPrologueDeath(MachineFunction &MF, unsigned Reg) { 875 // Do not set a kill flag on values that are also marked as live-in. This 876 // happens with the @llvm-returnaddress intrinsic and with arguments passed in 877 // callee saved registers. 878 // Omitting the kill flags is conservatively correct even if the live-in 879 // is not used after all. 880 bool IsLiveIn = MF.getRegInfo().isLiveIn(Reg); 881 return getKillRegState(!IsLiveIn); 882 } 883 884 static bool produceCompactUnwindFrame(MachineFunction &MF) { 885 const AArch64Subtarget &Subtarget = MF.getSubtarget<AArch64Subtarget>(); 886 AttributeSet Attrs = MF.getFunction()->getAttributes(); 887 return Subtarget.isTargetMachO() && 888 !(Subtarget.getTargetLowering()->supportSwiftError() && 889 Attrs.hasAttrSomewhere(Attribute::SwiftError)); 890 } 891 892 namespace { 893 894 struct RegPairInfo { 895 unsigned Reg1 = AArch64::NoRegister; 896 unsigned Reg2 = AArch64::NoRegister; 897 int FrameIdx; 898 int Offset; 899 bool IsGPR; 900 901 RegPairInfo() = default; 902 903 bool isPaired() const { return Reg2 != AArch64::NoRegister; } 904 }; 905 906 } // end anonymous namespace 907 908 static void computeCalleeSaveRegisterPairs( 909 MachineFunction &MF, const std::vector<CalleeSavedInfo> &CSI, 910 const TargetRegisterInfo *TRI, SmallVectorImpl<RegPairInfo> &RegPairs) { 911 912 if (CSI.empty()) 913 return; 914 915 AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 916 MachineFrameInfo &MFI = MF.getFrameInfo(); 917 CallingConv::ID CC = MF.getFunction()->getCallingConv(); 918 unsigned Count = CSI.size(); 919 (void)CC; 920 // MachO's compact unwind format relies on all registers being stored in 921 // pairs. 922 assert((!produceCompactUnwindFrame(MF) || 923 CC == CallingConv::PreserveMost || 924 (Count & 1) == 0) && 925 "Odd number of callee-saved regs to spill!"); 926 unsigned Offset = AFI->getCalleeSavedStackSize(); 927 928 for (unsigned i = 0; i < Count; ++i) { 929 RegPairInfo RPI; 930 RPI.Reg1 = CSI[i].getReg(); 931 932 assert(AArch64::GPR64RegClass.contains(RPI.Reg1) || 933 AArch64::FPR64RegClass.contains(RPI.Reg1)); 934 RPI.IsGPR = AArch64::GPR64RegClass.contains(RPI.Reg1); 935 936 // Add the next reg to the pair if it is in the same register class. 937 if (i + 1 < Count) { 938 unsigned NextReg = CSI[i + 1].getReg(); 939 if ((RPI.IsGPR && AArch64::GPR64RegClass.contains(NextReg)) || 940 (!RPI.IsGPR && AArch64::FPR64RegClass.contains(NextReg))) 941 RPI.Reg2 = NextReg; 942 } 943 944 // GPRs and FPRs are saved in pairs of 64-bit regs. We expect the CSI 945 // list to come in sorted by frame index so that we can issue the store 946 // pair instructions directly. Assert if we see anything otherwise. 947 // 948 // The order of the registers in the list is controlled by 949 // getCalleeSavedRegs(), so they will always be in-order, as well. 950 assert((!RPI.isPaired() || 951 (CSI[i].getFrameIdx() + 1 == CSI[i + 1].getFrameIdx())) && 952 "Out of order callee saved regs!"); 953 954 // MachO's compact unwind format relies on all registers being stored in 955 // adjacent register pairs. 956 assert((!produceCompactUnwindFrame(MF) || 957 CC == CallingConv::PreserveMost || 958 (RPI.isPaired() && 959 ((RPI.Reg1 == AArch64::LR && RPI.Reg2 == AArch64::FP) || 960 RPI.Reg1 + 1 == RPI.Reg2))) && 961 "Callee-save registers not saved as adjacent register pair!"); 962 963 RPI.FrameIdx = CSI[i].getFrameIdx(); 964 965 if (Count * 8 != AFI->getCalleeSavedStackSize() && !RPI.isPaired()) { 966 // Round up size of non-pair to pair size if we need to pad the 967 // callee-save area to ensure 16-byte alignment. 968 Offset -= 16; 969 assert(MFI.getObjectAlignment(RPI.FrameIdx) <= 16); 970 MFI.setObjectAlignment(RPI.FrameIdx, 16); 971 AFI->setCalleeSaveStackHasFreeSpace(true); 972 } else 973 Offset -= RPI.isPaired() ? 16 : 8; 974 assert(Offset % 8 == 0); 975 RPI.Offset = Offset / 8; 976 assert((RPI.Offset >= -64 && RPI.Offset <= 63) && 977 "Offset out of bounds for LDP/STP immediate"); 978 979 RegPairs.push_back(RPI); 980 if (RPI.isPaired()) 981 ++i; 982 } 983 } 984 985 bool AArch64FrameLowering::spillCalleeSavedRegisters( 986 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, 987 const std::vector<CalleeSavedInfo> &CSI, 988 const TargetRegisterInfo *TRI) const { 989 MachineFunction &MF = *MBB.getParent(); 990 const TargetInstrInfo &TII = *MF.getSubtarget().getInstrInfo(); 991 DebugLoc DL; 992 SmallVector<RegPairInfo, 8> RegPairs; 993 994 computeCalleeSaveRegisterPairs(MF, CSI, TRI, RegPairs); 995 996 for (auto RPII = RegPairs.rbegin(), RPIE = RegPairs.rend(); RPII != RPIE; 997 ++RPII) { 998 RegPairInfo RPI = *RPII; 999 unsigned Reg1 = RPI.Reg1; 1000 unsigned Reg2 = RPI.Reg2; 1001 unsigned StrOpc; 1002 1003 // Issue sequence of spills for cs regs. The first spill may be converted 1004 // to a pre-decrement store later by emitPrologue if the callee-save stack 1005 // area allocation can't be combined with the local stack area allocation. 1006 // For example: 1007 // stp x22, x21, [sp, #0] // addImm(+0) 1008 // stp x20, x19, [sp, #16] // addImm(+2) 1009 // stp fp, lr, [sp, #32] // addImm(+4) 1010 // Rationale: This sequence saves uop updates compared to a sequence of 1011 // pre-increment spills like stp xi,xj,[sp,#-16]! 1012 // Note: Similar rationale and sequence for restores in epilog. 1013 if (RPI.IsGPR) 1014 StrOpc = RPI.isPaired() ? AArch64::STPXi : AArch64::STRXui; 1015 else 1016 StrOpc = RPI.isPaired() ? AArch64::STPDi : AArch64::STRDui; 1017 DEBUG(dbgs() << "CSR spill: (" << TRI->getName(Reg1); 1018 if (RPI.isPaired()) 1019 dbgs() << ", " << TRI->getName(Reg2); 1020 dbgs() << ") -> fi#(" << RPI.FrameIdx; 1021 if (RPI.isPaired()) 1022 dbgs() << ", " << RPI.FrameIdx+1; 1023 dbgs() << ")\n"); 1024 1025 MachineInstrBuilder MIB = BuildMI(MBB, MI, DL, TII.get(StrOpc)); 1026 MBB.addLiveIn(Reg1); 1027 if (RPI.isPaired()) { 1028 MBB.addLiveIn(Reg2); 1029 MIB.addReg(Reg2, getPrologueDeath(MF, Reg2)); 1030 MIB.addMemOperand(MF.getMachineMemOperand( 1031 MachinePointerInfo::getFixedStack(MF, RPI.FrameIdx + 1), 1032 MachineMemOperand::MOStore, 8, 8)); 1033 } 1034 MIB.addReg(Reg1, getPrologueDeath(MF, Reg1)) 1035 .addReg(AArch64::SP) 1036 .addImm(RPI.Offset) // [sp, #offset*8], where factor*8 is implicit 1037 .setMIFlag(MachineInstr::FrameSetup); 1038 MIB.addMemOperand(MF.getMachineMemOperand( 1039 MachinePointerInfo::getFixedStack(MF, RPI.FrameIdx), 1040 MachineMemOperand::MOStore, 8, 8)); 1041 } 1042 return true; 1043 } 1044 1045 bool AArch64FrameLowering::restoreCalleeSavedRegisters( 1046 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, 1047 const std::vector<CalleeSavedInfo> &CSI, 1048 const TargetRegisterInfo *TRI) const { 1049 MachineFunction &MF = *MBB.getParent(); 1050 const TargetInstrInfo &TII = *MF.getSubtarget().getInstrInfo(); 1051 DebugLoc DL; 1052 SmallVector<RegPairInfo, 8> RegPairs; 1053 1054 if (MI != MBB.end()) 1055 DL = MI->getDebugLoc(); 1056 1057 computeCalleeSaveRegisterPairs(MF, CSI, TRI, RegPairs); 1058 1059 for (auto RPII = RegPairs.begin(), RPIE = RegPairs.end(); RPII != RPIE; 1060 ++RPII) { 1061 RegPairInfo RPI = *RPII; 1062 unsigned Reg1 = RPI.Reg1; 1063 unsigned Reg2 = RPI.Reg2; 1064 1065 // Issue sequence of restores for cs regs. The last restore may be converted 1066 // to a post-increment load later by emitEpilogue if the callee-save stack 1067 // area allocation can't be combined with the local stack area allocation. 1068 // For example: 1069 // ldp fp, lr, [sp, #32] // addImm(+4) 1070 // ldp x20, x19, [sp, #16] // addImm(+2) 1071 // ldp x22, x21, [sp, #0] // addImm(+0) 1072 // Note: see comment in spillCalleeSavedRegisters() 1073 unsigned LdrOpc; 1074 if (RPI.IsGPR) 1075 LdrOpc = RPI.isPaired() ? AArch64::LDPXi : AArch64::LDRXui; 1076 else 1077 LdrOpc = RPI.isPaired() ? AArch64::LDPDi : AArch64::LDRDui; 1078 DEBUG(dbgs() << "CSR restore: (" << TRI->getName(Reg1); 1079 if (RPI.isPaired()) 1080 dbgs() << ", " << TRI->getName(Reg2); 1081 dbgs() << ") -> fi#(" << RPI.FrameIdx; 1082 if (RPI.isPaired()) 1083 dbgs() << ", " << RPI.FrameIdx+1; 1084 dbgs() << ")\n"); 1085 1086 MachineInstrBuilder MIB = BuildMI(MBB, MI, DL, TII.get(LdrOpc)); 1087 if (RPI.isPaired()) { 1088 MIB.addReg(Reg2, getDefRegState(true)); 1089 MIB.addMemOperand(MF.getMachineMemOperand( 1090 MachinePointerInfo::getFixedStack(MF, RPI.FrameIdx + 1), 1091 MachineMemOperand::MOLoad, 8, 8)); 1092 } 1093 MIB.addReg(Reg1, getDefRegState(true)) 1094 .addReg(AArch64::SP) 1095 .addImm(RPI.Offset) // [sp, #offset*8] where the factor*8 is implicit 1096 .setMIFlag(MachineInstr::FrameDestroy); 1097 MIB.addMemOperand(MF.getMachineMemOperand( 1098 MachinePointerInfo::getFixedStack(MF, RPI.FrameIdx), 1099 MachineMemOperand::MOLoad, 8, 8)); 1100 } 1101 return true; 1102 } 1103 1104 void AArch64FrameLowering::determineCalleeSaves(MachineFunction &MF, 1105 BitVector &SavedRegs, 1106 RegScavenger *RS) const { 1107 // All calls are tail calls in GHC calling conv, and functions have no 1108 // prologue/epilogue. 1109 if (MF.getFunction()->getCallingConv() == CallingConv::GHC) 1110 return; 1111 1112 TargetFrameLowering::determineCalleeSaves(MF, SavedRegs, RS); 1113 const AArch64RegisterInfo *RegInfo = static_cast<const AArch64RegisterInfo *>( 1114 MF.getSubtarget().getRegisterInfo()); 1115 AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 1116 unsigned UnspilledCSGPR = AArch64::NoRegister; 1117 unsigned UnspilledCSGPRPaired = AArch64::NoRegister; 1118 1119 // The frame record needs to be created by saving the appropriate registers 1120 if (hasFP(MF)) { 1121 SavedRegs.set(AArch64::FP); 1122 SavedRegs.set(AArch64::LR); 1123 } 1124 1125 unsigned BasePointerReg = AArch64::NoRegister; 1126 if (RegInfo->hasBasePointer(MF)) 1127 BasePointerReg = RegInfo->getBaseRegister(); 1128 1129 bool ExtraCSSpill = false; 1130 const MCPhysReg *CSRegs = RegInfo->getCalleeSavedRegs(&MF); 1131 // Figure out which callee-saved registers to save/restore. 1132 for (unsigned i = 0; CSRegs[i]; ++i) { 1133 const unsigned Reg = CSRegs[i]; 1134 1135 // Add the base pointer register to SavedRegs if it is callee-save. 1136 if (Reg == BasePointerReg) 1137 SavedRegs.set(Reg); 1138 1139 bool RegUsed = SavedRegs.test(Reg); 1140 unsigned PairedReg = CSRegs[i ^ 1]; 1141 if (!RegUsed) { 1142 if (AArch64::GPR64RegClass.contains(Reg) && 1143 !RegInfo->isReservedReg(MF, Reg)) { 1144 UnspilledCSGPR = Reg; 1145 UnspilledCSGPRPaired = PairedReg; 1146 } 1147 continue; 1148 } 1149 1150 // MachO's compact unwind format relies on all registers being stored in 1151 // pairs. 1152 // FIXME: the usual format is actually better if unwinding isn't needed. 1153 if (produceCompactUnwindFrame(MF) && !SavedRegs.test(PairedReg)) { 1154 SavedRegs.set(PairedReg); 1155 if (AArch64::GPR64RegClass.contains(PairedReg) && 1156 !RegInfo->isReservedReg(MF, PairedReg)) 1157 ExtraCSSpill = true; 1158 } 1159 } 1160 1161 DEBUG(dbgs() << "*** determineCalleeSaves\nUsed CSRs:"; 1162 for (int Reg = SavedRegs.find_first(); Reg != -1; 1163 Reg = SavedRegs.find_next(Reg)) 1164 dbgs() << ' ' << PrintReg(Reg, RegInfo); 1165 dbgs() << "\n";); 1166 1167 // If any callee-saved registers are used, the frame cannot be eliminated. 1168 unsigned NumRegsSpilled = SavedRegs.count(); 1169 bool CanEliminateFrame = NumRegsSpilled == 0; 1170 1171 // FIXME: Set BigStack if any stack slot references may be out of range. 1172 // For now, just conservatively guestimate based on unscaled indexing 1173 // range. We'll end up allocating an unnecessary spill slot a lot, but 1174 // realistically that's not a big deal at this stage of the game. 1175 // The CSR spill slots have not been allocated yet, so estimateStackSize 1176 // won't include them. 1177 MachineFrameInfo &MFI = MF.getFrameInfo(); 1178 unsigned CFSize = MFI.estimateStackSize(MF) + 8 * NumRegsSpilled; 1179 DEBUG(dbgs() << "Estimated stack frame size: " << CFSize << " bytes.\n"); 1180 bool BigStack = (CFSize >= 256); 1181 if (BigStack || !CanEliminateFrame || RegInfo->cannotEliminateFrame(MF)) 1182 AFI->setHasStackFrame(true); 1183 1184 // Estimate if we might need to scavenge a register at some point in order 1185 // to materialize a stack offset. If so, either spill one additional 1186 // callee-saved register or reserve a special spill slot to facilitate 1187 // register scavenging. If we already spilled an extra callee-saved register 1188 // above to keep the number of spills even, we don't need to do anything else 1189 // here. 1190 if (BigStack && !ExtraCSSpill) { 1191 if (UnspilledCSGPR != AArch64::NoRegister) { 1192 DEBUG(dbgs() << "Spilling " << PrintReg(UnspilledCSGPR, RegInfo) 1193 << " to get a scratch register.\n"); 1194 SavedRegs.set(UnspilledCSGPR); 1195 // MachO's compact unwind format relies on all registers being stored in 1196 // pairs, so if we need to spill one extra for BigStack, then we need to 1197 // store the pair. 1198 if (produceCompactUnwindFrame(MF)) 1199 SavedRegs.set(UnspilledCSGPRPaired); 1200 ExtraCSSpill = true; 1201 NumRegsSpilled = SavedRegs.count(); 1202 } 1203 1204 // If we didn't find an extra callee-saved register to spill, create 1205 // an emergency spill slot. 1206 if (!ExtraCSSpill) { 1207 const TargetRegisterClass *RC = &AArch64::GPR64RegClass; 1208 int FI = MFI.CreateStackObject(RC->getSize(), RC->getAlignment(), false); 1209 RS->addScavengingFrameIndex(FI); 1210 DEBUG(dbgs() << "No available CS registers, allocated fi#" << FI 1211 << " as the emergency spill slot.\n"); 1212 } 1213 } 1214 1215 // Round up to register pair alignment to avoid additional SP adjustment 1216 // instructions. 1217 AFI->setCalleeSavedStackSize(alignTo(8 * NumRegsSpilled, 16)); 1218 } 1219 1220 bool AArch64FrameLowering::enableStackSlotScavenging( 1221 const MachineFunction &MF) const { 1222 const AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 1223 return AFI->hasCalleeSaveStackFreeSpace(); 1224 } 1225