1 //===- AArch64InstrInfo.cpp - AArch64 Instruction Information -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This file contains the AArch64 implementation of the TargetInstrInfo class. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "AArch64InstrInfo.h" 14 #include "AArch64MachineFunctionInfo.h" 15 #include "AArch64Subtarget.h" 16 #include "MCTargetDesc/AArch64AddressingModes.h" 17 #include "Utils/AArch64BaseInfo.h" 18 #include "llvm/ADT/ArrayRef.h" 19 #include "llvm/ADT/STLExtras.h" 20 #include "llvm/ADT/SmallVector.h" 21 #include "llvm/CodeGen/MachineBasicBlock.h" 22 #include "llvm/CodeGen/MachineFrameInfo.h" 23 #include "llvm/CodeGen/MachineFunction.h" 24 #include "llvm/CodeGen/MachineInstr.h" 25 #include "llvm/CodeGen/MachineInstrBuilder.h" 26 #include "llvm/CodeGen/MachineMemOperand.h" 27 #include "llvm/CodeGen/MachineOperand.h" 28 #include "llvm/CodeGen/MachineRegisterInfo.h" 29 #include "llvm/CodeGen/MachineModuleInfo.h" 30 #include "llvm/CodeGen/StackMaps.h" 31 #include "llvm/CodeGen/TargetRegisterInfo.h" 32 #include "llvm/CodeGen/TargetSubtargetInfo.h" 33 #include "llvm/IR/DebugLoc.h" 34 #include "llvm/IR/GlobalValue.h" 35 #include "llvm/MC/MCInst.h" 36 #include "llvm/MC/MCInstrDesc.h" 37 #include "llvm/Support/Casting.h" 38 #include "llvm/Support/CodeGen.h" 39 #include "llvm/Support/CommandLine.h" 40 #include "llvm/Support/Compiler.h" 41 #include "llvm/Support/ErrorHandling.h" 42 #include "llvm/Support/MathExtras.h" 43 #include "llvm/Target/TargetMachine.h" 44 #include "llvm/Target/TargetOptions.h" 45 #include <cassert> 46 #include <cstdint> 47 #include <iterator> 48 #include <utility> 49 50 using namespace llvm; 51 52 #define GET_INSTRINFO_CTOR_DTOR 53 #include "AArch64GenInstrInfo.inc" 54 55 static cl::opt<unsigned> TBZDisplacementBits( 56 "aarch64-tbz-offset-bits", cl::Hidden, cl::init(14), 57 cl::desc("Restrict range of TB[N]Z instructions (DEBUG)")); 58 59 static cl::opt<unsigned> CBZDisplacementBits( 60 "aarch64-cbz-offset-bits", cl::Hidden, cl::init(19), 61 cl::desc("Restrict range of CB[N]Z instructions (DEBUG)")); 62 63 static cl::opt<unsigned> 64 BCCDisplacementBits("aarch64-bcc-offset-bits", cl::Hidden, cl::init(19), 65 cl::desc("Restrict range of Bcc instructions (DEBUG)")); 66 67 AArch64InstrInfo::AArch64InstrInfo(const AArch64Subtarget &STI) 68 : AArch64GenInstrInfo(AArch64::ADJCALLSTACKDOWN, AArch64::ADJCALLSTACKUP, 69 AArch64::CATCHRET), 70 RI(STI.getTargetTriple()), Subtarget(STI) {} 71 72 /// GetInstSize - Return the number of bytes of code the specified 73 /// instruction may be. This returns the maximum number of bytes. 74 unsigned AArch64InstrInfo::getInstSizeInBytes(const MachineInstr &MI) const { 75 const MachineBasicBlock &MBB = *MI.getParent(); 76 const MachineFunction *MF = MBB.getParent(); 77 const MCAsmInfo *MAI = MF->getTarget().getMCAsmInfo(); 78 79 if (MI.getOpcode() == AArch64::INLINEASM) 80 return getInlineAsmLength(MI.getOperand(0).getSymbolName(), *MAI); 81 82 // FIXME: We currently only handle pseudoinstructions that don't get expanded 83 // before the assembly printer. 84 unsigned NumBytes = 0; 85 const MCInstrDesc &Desc = MI.getDesc(); 86 switch (Desc.getOpcode()) { 87 default: 88 // Anything not explicitly designated otherwise is a normal 4-byte insn. 89 NumBytes = 4; 90 break; 91 case TargetOpcode::DBG_VALUE: 92 case TargetOpcode::EH_LABEL: 93 case TargetOpcode::IMPLICIT_DEF: 94 case TargetOpcode::KILL: 95 NumBytes = 0; 96 break; 97 case TargetOpcode::STACKMAP: 98 // The upper bound for a stackmap intrinsic is the full length of its shadow 99 NumBytes = StackMapOpers(&MI).getNumPatchBytes(); 100 assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!"); 101 break; 102 case TargetOpcode::PATCHPOINT: 103 // The size of the patchpoint intrinsic is the number of bytes requested 104 NumBytes = PatchPointOpers(&MI).getNumPatchBytes(); 105 assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!"); 106 break; 107 case AArch64::TLSDESC_CALLSEQ: 108 // This gets lowered to an instruction sequence which takes 16 bytes 109 NumBytes = 16; 110 break; 111 case AArch64::JumpTableDest32: 112 case AArch64::JumpTableDest16: 113 case AArch64::JumpTableDest8: 114 NumBytes = 12; 115 break; 116 case AArch64::SPACE: 117 NumBytes = MI.getOperand(1).getImm(); 118 break; 119 } 120 121 return NumBytes; 122 } 123 124 static void parseCondBranch(MachineInstr *LastInst, MachineBasicBlock *&Target, 125 SmallVectorImpl<MachineOperand> &Cond) { 126 // Block ends with fall-through condbranch. 127 switch (LastInst->getOpcode()) { 128 default: 129 llvm_unreachable("Unknown branch instruction?"); 130 case AArch64::Bcc: 131 Target = LastInst->getOperand(1).getMBB(); 132 Cond.push_back(LastInst->getOperand(0)); 133 break; 134 case AArch64::CBZW: 135 case AArch64::CBZX: 136 case AArch64::CBNZW: 137 case AArch64::CBNZX: 138 Target = LastInst->getOperand(1).getMBB(); 139 Cond.push_back(MachineOperand::CreateImm(-1)); 140 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode())); 141 Cond.push_back(LastInst->getOperand(0)); 142 break; 143 case AArch64::TBZW: 144 case AArch64::TBZX: 145 case AArch64::TBNZW: 146 case AArch64::TBNZX: 147 Target = LastInst->getOperand(2).getMBB(); 148 Cond.push_back(MachineOperand::CreateImm(-1)); 149 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode())); 150 Cond.push_back(LastInst->getOperand(0)); 151 Cond.push_back(LastInst->getOperand(1)); 152 } 153 } 154 155 static unsigned getBranchDisplacementBits(unsigned Opc) { 156 switch (Opc) { 157 default: 158 llvm_unreachable("unexpected opcode!"); 159 case AArch64::B: 160 return 64; 161 case AArch64::TBNZW: 162 case AArch64::TBZW: 163 case AArch64::TBNZX: 164 case AArch64::TBZX: 165 return TBZDisplacementBits; 166 case AArch64::CBNZW: 167 case AArch64::CBZW: 168 case AArch64::CBNZX: 169 case AArch64::CBZX: 170 return CBZDisplacementBits; 171 case AArch64::Bcc: 172 return BCCDisplacementBits; 173 } 174 } 175 176 bool AArch64InstrInfo::isBranchOffsetInRange(unsigned BranchOp, 177 int64_t BrOffset) const { 178 unsigned Bits = getBranchDisplacementBits(BranchOp); 179 assert(Bits >= 3 && "max branch displacement must be enough to jump" 180 "over conditional branch expansion"); 181 return isIntN(Bits, BrOffset / 4); 182 } 183 184 MachineBasicBlock * 185 AArch64InstrInfo::getBranchDestBlock(const MachineInstr &MI) const { 186 switch (MI.getOpcode()) { 187 default: 188 llvm_unreachable("unexpected opcode!"); 189 case AArch64::B: 190 return MI.getOperand(0).getMBB(); 191 case AArch64::TBZW: 192 case AArch64::TBNZW: 193 case AArch64::TBZX: 194 case AArch64::TBNZX: 195 return MI.getOperand(2).getMBB(); 196 case AArch64::CBZW: 197 case AArch64::CBNZW: 198 case AArch64::CBZX: 199 case AArch64::CBNZX: 200 case AArch64::Bcc: 201 return MI.getOperand(1).getMBB(); 202 } 203 } 204 205 // Branch analysis. 206 bool AArch64InstrInfo::analyzeBranch(MachineBasicBlock &MBB, 207 MachineBasicBlock *&TBB, 208 MachineBasicBlock *&FBB, 209 SmallVectorImpl<MachineOperand> &Cond, 210 bool AllowModify) const { 211 // If the block has no terminators, it just falls into the block after it. 212 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr(); 213 if (I == MBB.end()) 214 return false; 215 216 if (!isUnpredicatedTerminator(*I)) 217 return false; 218 219 // Get the last instruction in the block. 220 MachineInstr *LastInst = &*I; 221 222 // If there is only one terminator instruction, process it. 223 unsigned LastOpc = LastInst->getOpcode(); 224 if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) { 225 if (isUncondBranchOpcode(LastOpc)) { 226 TBB = LastInst->getOperand(0).getMBB(); 227 return false; 228 } 229 if (isCondBranchOpcode(LastOpc)) { 230 // Block ends with fall-through condbranch. 231 parseCondBranch(LastInst, TBB, Cond); 232 return false; 233 } 234 return true; // Can't handle indirect branch. 235 } 236 237 // Get the instruction before it if it is a terminator. 238 MachineInstr *SecondLastInst = &*I; 239 unsigned SecondLastOpc = SecondLastInst->getOpcode(); 240 241 // If AllowModify is true and the block ends with two or more unconditional 242 // branches, delete all but the first unconditional branch. 243 if (AllowModify && isUncondBranchOpcode(LastOpc)) { 244 while (isUncondBranchOpcode(SecondLastOpc)) { 245 LastInst->eraseFromParent(); 246 LastInst = SecondLastInst; 247 LastOpc = LastInst->getOpcode(); 248 if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) { 249 // Return now the only terminator is an unconditional branch. 250 TBB = LastInst->getOperand(0).getMBB(); 251 return false; 252 } else { 253 SecondLastInst = &*I; 254 SecondLastOpc = SecondLastInst->getOpcode(); 255 } 256 } 257 } 258 259 // If there are three terminators, we don't know what sort of block this is. 260 if (SecondLastInst && I != MBB.begin() && isUnpredicatedTerminator(*--I)) 261 return true; 262 263 // If the block ends with a B and a Bcc, handle it. 264 if (isCondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) { 265 parseCondBranch(SecondLastInst, TBB, Cond); 266 FBB = LastInst->getOperand(0).getMBB(); 267 return false; 268 } 269 270 // If the block ends with two unconditional branches, handle it. The second 271 // one is not executed, so remove it. 272 if (isUncondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) { 273 TBB = SecondLastInst->getOperand(0).getMBB(); 274 I = LastInst; 275 if (AllowModify) 276 I->eraseFromParent(); 277 return false; 278 } 279 280 // ...likewise if it ends with an indirect branch followed by an unconditional 281 // branch. 282 if (isIndirectBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) { 283 I = LastInst; 284 if (AllowModify) 285 I->eraseFromParent(); 286 return true; 287 } 288 289 // Otherwise, can't handle this. 290 return true; 291 } 292 293 bool AArch64InstrInfo::reverseBranchCondition( 294 SmallVectorImpl<MachineOperand> &Cond) const { 295 if (Cond[0].getImm() != -1) { 296 // Regular Bcc 297 AArch64CC::CondCode CC = (AArch64CC::CondCode)(int)Cond[0].getImm(); 298 Cond[0].setImm(AArch64CC::getInvertedCondCode(CC)); 299 } else { 300 // Folded compare-and-branch 301 switch (Cond[1].getImm()) { 302 default: 303 llvm_unreachable("Unknown conditional branch!"); 304 case AArch64::CBZW: 305 Cond[1].setImm(AArch64::CBNZW); 306 break; 307 case AArch64::CBNZW: 308 Cond[1].setImm(AArch64::CBZW); 309 break; 310 case AArch64::CBZX: 311 Cond[1].setImm(AArch64::CBNZX); 312 break; 313 case AArch64::CBNZX: 314 Cond[1].setImm(AArch64::CBZX); 315 break; 316 case AArch64::TBZW: 317 Cond[1].setImm(AArch64::TBNZW); 318 break; 319 case AArch64::TBNZW: 320 Cond[1].setImm(AArch64::TBZW); 321 break; 322 case AArch64::TBZX: 323 Cond[1].setImm(AArch64::TBNZX); 324 break; 325 case AArch64::TBNZX: 326 Cond[1].setImm(AArch64::TBZX); 327 break; 328 } 329 } 330 331 return false; 332 } 333 334 unsigned AArch64InstrInfo::removeBranch(MachineBasicBlock &MBB, 335 int *BytesRemoved) const { 336 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr(); 337 if (I == MBB.end()) 338 return 0; 339 340 if (!isUncondBranchOpcode(I->getOpcode()) && 341 !isCondBranchOpcode(I->getOpcode())) 342 return 0; 343 344 // Remove the branch. 345 I->eraseFromParent(); 346 347 I = MBB.end(); 348 349 if (I == MBB.begin()) { 350 if (BytesRemoved) 351 *BytesRemoved = 4; 352 return 1; 353 } 354 --I; 355 if (!isCondBranchOpcode(I->getOpcode())) { 356 if (BytesRemoved) 357 *BytesRemoved = 4; 358 return 1; 359 } 360 361 // Remove the branch. 362 I->eraseFromParent(); 363 if (BytesRemoved) 364 *BytesRemoved = 8; 365 366 return 2; 367 } 368 369 void AArch64InstrInfo::instantiateCondBranch( 370 MachineBasicBlock &MBB, const DebugLoc &DL, MachineBasicBlock *TBB, 371 ArrayRef<MachineOperand> Cond) const { 372 if (Cond[0].getImm() != -1) { 373 // Regular Bcc 374 BuildMI(&MBB, DL, get(AArch64::Bcc)).addImm(Cond[0].getImm()).addMBB(TBB); 375 } else { 376 // Folded compare-and-branch 377 // Note that we use addOperand instead of addReg to keep the flags. 378 const MachineInstrBuilder MIB = 379 BuildMI(&MBB, DL, get(Cond[1].getImm())).add(Cond[2]); 380 if (Cond.size() > 3) 381 MIB.addImm(Cond[3].getImm()); 382 MIB.addMBB(TBB); 383 } 384 } 385 386 unsigned AArch64InstrInfo::insertBranch( 387 MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, 388 ArrayRef<MachineOperand> Cond, const DebugLoc &DL, int *BytesAdded) const { 389 // Shouldn't be a fall through. 390 assert(TBB && "insertBranch must not be told to insert a fallthrough"); 391 392 if (!FBB) { 393 if (Cond.empty()) // Unconditional branch? 394 BuildMI(&MBB, DL, get(AArch64::B)).addMBB(TBB); 395 else 396 instantiateCondBranch(MBB, DL, TBB, Cond); 397 398 if (BytesAdded) 399 *BytesAdded = 4; 400 401 return 1; 402 } 403 404 // Two-way conditional branch. 405 instantiateCondBranch(MBB, DL, TBB, Cond); 406 BuildMI(&MBB, DL, get(AArch64::B)).addMBB(FBB); 407 408 if (BytesAdded) 409 *BytesAdded = 8; 410 411 return 2; 412 } 413 414 // Find the original register that VReg is copied from. 415 static unsigned removeCopies(const MachineRegisterInfo &MRI, unsigned VReg) { 416 while (TargetRegisterInfo::isVirtualRegister(VReg)) { 417 const MachineInstr *DefMI = MRI.getVRegDef(VReg); 418 if (!DefMI->isFullCopy()) 419 return VReg; 420 VReg = DefMI->getOperand(1).getReg(); 421 } 422 return VReg; 423 } 424 425 // Determine if VReg is defined by an instruction that can be folded into a 426 // csel instruction. If so, return the folded opcode, and the replacement 427 // register. 428 static unsigned canFoldIntoCSel(const MachineRegisterInfo &MRI, unsigned VReg, 429 unsigned *NewVReg = nullptr) { 430 VReg = removeCopies(MRI, VReg); 431 if (!TargetRegisterInfo::isVirtualRegister(VReg)) 432 return 0; 433 434 bool Is64Bit = AArch64::GPR64allRegClass.hasSubClassEq(MRI.getRegClass(VReg)); 435 const MachineInstr *DefMI = MRI.getVRegDef(VReg); 436 unsigned Opc = 0; 437 unsigned SrcOpNum = 0; 438 switch (DefMI->getOpcode()) { 439 case AArch64::ADDSXri: 440 case AArch64::ADDSWri: 441 // if NZCV is used, do not fold. 442 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1) 443 return 0; 444 // fall-through to ADDXri and ADDWri. 445 LLVM_FALLTHROUGH; 446 case AArch64::ADDXri: 447 case AArch64::ADDWri: 448 // add x, 1 -> csinc. 449 if (!DefMI->getOperand(2).isImm() || DefMI->getOperand(2).getImm() != 1 || 450 DefMI->getOperand(3).getImm() != 0) 451 return 0; 452 SrcOpNum = 1; 453 Opc = Is64Bit ? AArch64::CSINCXr : AArch64::CSINCWr; 454 break; 455 456 case AArch64::ORNXrr: 457 case AArch64::ORNWrr: { 458 // not x -> csinv, represented as orn dst, xzr, src. 459 unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg()); 460 if (ZReg != AArch64::XZR && ZReg != AArch64::WZR) 461 return 0; 462 SrcOpNum = 2; 463 Opc = Is64Bit ? AArch64::CSINVXr : AArch64::CSINVWr; 464 break; 465 } 466 467 case AArch64::SUBSXrr: 468 case AArch64::SUBSWrr: 469 // if NZCV is used, do not fold. 470 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1) 471 return 0; 472 // fall-through to SUBXrr and SUBWrr. 473 LLVM_FALLTHROUGH; 474 case AArch64::SUBXrr: 475 case AArch64::SUBWrr: { 476 // neg x -> csneg, represented as sub dst, xzr, src. 477 unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg()); 478 if (ZReg != AArch64::XZR && ZReg != AArch64::WZR) 479 return 0; 480 SrcOpNum = 2; 481 Opc = Is64Bit ? AArch64::CSNEGXr : AArch64::CSNEGWr; 482 break; 483 } 484 default: 485 return 0; 486 } 487 assert(Opc && SrcOpNum && "Missing parameters"); 488 489 if (NewVReg) 490 *NewVReg = DefMI->getOperand(SrcOpNum).getReg(); 491 return Opc; 492 } 493 494 bool AArch64InstrInfo::canInsertSelect(const MachineBasicBlock &MBB, 495 ArrayRef<MachineOperand> Cond, 496 unsigned TrueReg, unsigned FalseReg, 497 int &CondCycles, int &TrueCycles, 498 int &FalseCycles) const { 499 // Check register classes. 500 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 501 const TargetRegisterClass *RC = 502 RI.getCommonSubClass(MRI.getRegClass(TrueReg), MRI.getRegClass(FalseReg)); 503 if (!RC) 504 return false; 505 506 // Expanding cbz/tbz requires an extra cycle of latency on the condition. 507 unsigned ExtraCondLat = Cond.size() != 1; 508 509 // GPRs are handled by csel. 510 // FIXME: Fold in x+1, -x, and ~x when applicable. 511 if (AArch64::GPR64allRegClass.hasSubClassEq(RC) || 512 AArch64::GPR32allRegClass.hasSubClassEq(RC)) { 513 // Single-cycle csel, csinc, csinv, and csneg. 514 CondCycles = 1 + ExtraCondLat; 515 TrueCycles = FalseCycles = 1; 516 if (canFoldIntoCSel(MRI, TrueReg)) 517 TrueCycles = 0; 518 else if (canFoldIntoCSel(MRI, FalseReg)) 519 FalseCycles = 0; 520 return true; 521 } 522 523 // Scalar floating point is handled by fcsel. 524 // FIXME: Form fabs, fmin, and fmax when applicable. 525 if (AArch64::FPR64RegClass.hasSubClassEq(RC) || 526 AArch64::FPR32RegClass.hasSubClassEq(RC)) { 527 CondCycles = 5 + ExtraCondLat; 528 TrueCycles = FalseCycles = 2; 529 return true; 530 } 531 532 // Can't do vectors. 533 return false; 534 } 535 536 void AArch64InstrInfo::insertSelect(MachineBasicBlock &MBB, 537 MachineBasicBlock::iterator I, 538 const DebugLoc &DL, unsigned DstReg, 539 ArrayRef<MachineOperand> Cond, 540 unsigned TrueReg, unsigned FalseReg) const { 541 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 542 543 // Parse the condition code, see parseCondBranch() above. 544 AArch64CC::CondCode CC; 545 switch (Cond.size()) { 546 default: 547 llvm_unreachable("Unknown condition opcode in Cond"); 548 case 1: // b.cc 549 CC = AArch64CC::CondCode(Cond[0].getImm()); 550 break; 551 case 3: { // cbz/cbnz 552 // We must insert a compare against 0. 553 bool Is64Bit; 554 switch (Cond[1].getImm()) { 555 default: 556 llvm_unreachable("Unknown branch opcode in Cond"); 557 case AArch64::CBZW: 558 Is64Bit = false; 559 CC = AArch64CC::EQ; 560 break; 561 case AArch64::CBZX: 562 Is64Bit = true; 563 CC = AArch64CC::EQ; 564 break; 565 case AArch64::CBNZW: 566 Is64Bit = false; 567 CC = AArch64CC::NE; 568 break; 569 case AArch64::CBNZX: 570 Is64Bit = true; 571 CC = AArch64CC::NE; 572 break; 573 } 574 unsigned SrcReg = Cond[2].getReg(); 575 if (Is64Bit) { 576 // cmp reg, #0 is actually subs xzr, reg, #0. 577 MRI.constrainRegClass(SrcReg, &AArch64::GPR64spRegClass); 578 BuildMI(MBB, I, DL, get(AArch64::SUBSXri), AArch64::XZR) 579 .addReg(SrcReg) 580 .addImm(0) 581 .addImm(0); 582 } else { 583 MRI.constrainRegClass(SrcReg, &AArch64::GPR32spRegClass); 584 BuildMI(MBB, I, DL, get(AArch64::SUBSWri), AArch64::WZR) 585 .addReg(SrcReg) 586 .addImm(0) 587 .addImm(0); 588 } 589 break; 590 } 591 case 4: { // tbz/tbnz 592 // We must insert a tst instruction. 593 switch (Cond[1].getImm()) { 594 default: 595 llvm_unreachable("Unknown branch opcode in Cond"); 596 case AArch64::TBZW: 597 case AArch64::TBZX: 598 CC = AArch64CC::EQ; 599 break; 600 case AArch64::TBNZW: 601 case AArch64::TBNZX: 602 CC = AArch64CC::NE; 603 break; 604 } 605 // cmp reg, #foo is actually ands xzr, reg, #1<<foo. 606 if (Cond[1].getImm() == AArch64::TBZW || Cond[1].getImm() == AArch64::TBNZW) 607 BuildMI(MBB, I, DL, get(AArch64::ANDSWri), AArch64::WZR) 608 .addReg(Cond[2].getReg()) 609 .addImm( 610 AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 32)); 611 else 612 BuildMI(MBB, I, DL, get(AArch64::ANDSXri), AArch64::XZR) 613 .addReg(Cond[2].getReg()) 614 .addImm( 615 AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 64)); 616 break; 617 } 618 } 619 620 unsigned Opc = 0; 621 const TargetRegisterClass *RC = nullptr; 622 bool TryFold = false; 623 if (MRI.constrainRegClass(DstReg, &AArch64::GPR64RegClass)) { 624 RC = &AArch64::GPR64RegClass; 625 Opc = AArch64::CSELXr; 626 TryFold = true; 627 } else if (MRI.constrainRegClass(DstReg, &AArch64::GPR32RegClass)) { 628 RC = &AArch64::GPR32RegClass; 629 Opc = AArch64::CSELWr; 630 TryFold = true; 631 } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR64RegClass)) { 632 RC = &AArch64::FPR64RegClass; 633 Opc = AArch64::FCSELDrrr; 634 } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR32RegClass)) { 635 RC = &AArch64::FPR32RegClass; 636 Opc = AArch64::FCSELSrrr; 637 } 638 assert(RC && "Unsupported regclass"); 639 640 // Try folding simple instructions into the csel. 641 if (TryFold) { 642 unsigned NewVReg = 0; 643 unsigned FoldedOpc = canFoldIntoCSel(MRI, TrueReg, &NewVReg); 644 if (FoldedOpc) { 645 // The folded opcodes csinc, csinc and csneg apply the operation to 646 // FalseReg, so we need to invert the condition. 647 CC = AArch64CC::getInvertedCondCode(CC); 648 TrueReg = FalseReg; 649 } else 650 FoldedOpc = canFoldIntoCSel(MRI, FalseReg, &NewVReg); 651 652 // Fold the operation. Leave any dead instructions for DCE to clean up. 653 if (FoldedOpc) { 654 FalseReg = NewVReg; 655 Opc = FoldedOpc; 656 // The extends the live range of NewVReg. 657 MRI.clearKillFlags(NewVReg); 658 } 659 } 660 661 // Pull all virtual register into the appropriate class. 662 MRI.constrainRegClass(TrueReg, RC); 663 MRI.constrainRegClass(FalseReg, RC); 664 665 // Insert the csel. 666 BuildMI(MBB, I, DL, get(Opc), DstReg) 667 .addReg(TrueReg) 668 .addReg(FalseReg) 669 .addImm(CC); 670 } 671 672 /// Returns true if a MOVi32imm or MOVi64imm can be expanded to an ORRxx. 673 static bool canBeExpandedToORR(const MachineInstr &MI, unsigned BitSize) { 674 uint64_t Imm = MI.getOperand(1).getImm(); 675 uint64_t UImm = Imm << (64 - BitSize) >> (64 - BitSize); 676 uint64_t Encoding; 677 return AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding); 678 } 679 680 // FIXME: this implementation should be micro-architecture dependent, so a 681 // micro-architecture target hook should be introduced here in future. 682 bool AArch64InstrInfo::isAsCheapAsAMove(const MachineInstr &MI) const { 683 if (!Subtarget.hasCustomCheapAsMoveHandling()) 684 return MI.isAsCheapAsAMove(); 685 686 const unsigned Opcode = MI.getOpcode(); 687 688 // Firstly, check cases gated by features. 689 690 if (Subtarget.hasZeroCycleZeroingFP()) { 691 if (Opcode == AArch64::FMOVH0 || 692 Opcode == AArch64::FMOVS0 || 693 Opcode == AArch64::FMOVD0) 694 return true; 695 } 696 697 if (Subtarget.hasZeroCycleZeroingGP()) { 698 if (Opcode == TargetOpcode::COPY && 699 (MI.getOperand(1).getReg() == AArch64::WZR || 700 MI.getOperand(1).getReg() == AArch64::XZR)) 701 return true; 702 } 703 704 // Secondly, check cases specific to sub-targets. 705 706 if (Subtarget.hasExynosCheapAsMoveHandling()) { 707 if (isExynosCheapAsMove(MI)) 708 return true; 709 710 return MI.isAsCheapAsAMove(); 711 } 712 713 // Finally, check generic cases. 714 715 switch (Opcode) { 716 default: 717 return false; 718 719 // add/sub on register without shift 720 case AArch64::ADDWri: 721 case AArch64::ADDXri: 722 case AArch64::SUBWri: 723 case AArch64::SUBXri: 724 return (MI.getOperand(3).getImm() == 0); 725 726 // logical ops on immediate 727 case AArch64::ANDWri: 728 case AArch64::ANDXri: 729 case AArch64::EORWri: 730 case AArch64::EORXri: 731 case AArch64::ORRWri: 732 case AArch64::ORRXri: 733 return true; 734 735 // logical ops on register without shift 736 case AArch64::ANDWrr: 737 case AArch64::ANDXrr: 738 case AArch64::BICWrr: 739 case AArch64::BICXrr: 740 case AArch64::EONWrr: 741 case AArch64::EONXrr: 742 case AArch64::EORWrr: 743 case AArch64::EORXrr: 744 case AArch64::ORNWrr: 745 case AArch64::ORNXrr: 746 case AArch64::ORRWrr: 747 case AArch64::ORRXrr: 748 return true; 749 750 // If MOVi32imm or MOVi64imm can be expanded into ORRWri or 751 // ORRXri, it is as cheap as MOV 752 case AArch64::MOVi32imm: 753 return canBeExpandedToORR(MI, 32); 754 case AArch64::MOVi64imm: 755 return canBeExpandedToORR(MI, 64); 756 } 757 758 llvm_unreachable("Unknown opcode to check as cheap as a move!"); 759 } 760 761 bool AArch64InstrInfo::isFalkorShiftExtFast(const MachineInstr &MI) { 762 switch (MI.getOpcode()) { 763 default: 764 return false; 765 766 case AArch64::ADDWrs: 767 case AArch64::ADDXrs: 768 case AArch64::ADDSWrs: 769 case AArch64::ADDSXrs: { 770 unsigned Imm = MI.getOperand(3).getImm(); 771 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm); 772 if (ShiftVal == 0) 773 return true; 774 return AArch64_AM::getShiftType(Imm) == AArch64_AM::LSL && ShiftVal <= 5; 775 } 776 777 case AArch64::ADDWrx: 778 case AArch64::ADDXrx: 779 case AArch64::ADDXrx64: 780 case AArch64::ADDSWrx: 781 case AArch64::ADDSXrx: 782 case AArch64::ADDSXrx64: { 783 unsigned Imm = MI.getOperand(3).getImm(); 784 switch (AArch64_AM::getArithExtendType(Imm)) { 785 default: 786 return false; 787 case AArch64_AM::UXTB: 788 case AArch64_AM::UXTH: 789 case AArch64_AM::UXTW: 790 case AArch64_AM::UXTX: 791 return AArch64_AM::getArithShiftValue(Imm) <= 4; 792 } 793 } 794 795 case AArch64::SUBWrs: 796 case AArch64::SUBSWrs: { 797 unsigned Imm = MI.getOperand(3).getImm(); 798 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm); 799 return ShiftVal == 0 || 800 (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 31); 801 } 802 803 case AArch64::SUBXrs: 804 case AArch64::SUBSXrs: { 805 unsigned Imm = MI.getOperand(3).getImm(); 806 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm); 807 return ShiftVal == 0 || 808 (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 63); 809 } 810 811 case AArch64::SUBWrx: 812 case AArch64::SUBXrx: 813 case AArch64::SUBXrx64: 814 case AArch64::SUBSWrx: 815 case AArch64::SUBSXrx: 816 case AArch64::SUBSXrx64: { 817 unsigned Imm = MI.getOperand(3).getImm(); 818 switch (AArch64_AM::getArithExtendType(Imm)) { 819 default: 820 return false; 821 case AArch64_AM::UXTB: 822 case AArch64_AM::UXTH: 823 case AArch64_AM::UXTW: 824 case AArch64_AM::UXTX: 825 return AArch64_AM::getArithShiftValue(Imm) == 0; 826 } 827 } 828 829 case AArch64::LDRBBroW: 830 case AArch64::LDRBBroX: 831 case AArch64::LDRBroW: 832 case AArch64::LDRBroX: 833 case AArch64::LDRDroW: 834 case AArch64::LDRDroX: 835 case AArch64::LDRHHroW: 836 case AArch64::LDRHHroX: 837 case AArch64::LDRHroW: 838 case AArch64::LDRHroX: 839 case AArch64::LDRQroW: 840 case AArch64::LDRQroX: 841 case AArch64::LDRSBWroW: 842 case AArch64::LDRSBWroX: 843 case AArch64::LDRSBXroW: 844 case AArch64::LDRSBXroX: 845 case AArch64::LDRSHWroW: 846 case AArch64::LDRSHWroX: 847 case AArch64::LDRSHXroW: 848 case AArch64::LDRSHXroX: 849 case AArch64::LDRSWroW: 850 case AArch64::LDRSWroX: 851 case AArch64::LDRSroW: 852 case AArch64::LDRSroX: 853 case AArch64::LDRWroW: 854 case AArch64::LDRWroX: 855 case AArch64::LDRXroW: 856 case AArch64::LDRXroX: 857 case AArch64::PRFMroW: 858 case AArch64::PRFMroX: 859 case AArch64::STRBBroW: 860 case AArch64::STRBBroX: 861 case AArch64::STRBroW: 862 case AArch64::STRBroX: 863 case AArch64::STRDroW: 864 case AArch64::STRDroX: 865 case AArch64::STRHHroW: 866 case AArch64::STRHHroX: 867 case AArch64::STRHroW: 868 case AArch64::STRHroX: 869 case AArch64::STRQroW: 870 case AArch64::STRQroX: 871 case AArch64::STRSroW: 872 case AArch64::STRSroX: 873 case AArch64::STRWroW: 874 case AArch64::STRWroX: 875 case AArch64::STRXroW: 876 case AArch64::STRXroX: { 877 unsigned IsSigned = MI.getOperand(3).getImm(); 878 return !IsSigned; 879 } 880 } 881 } 882 883 bool AArch64InstrInfo::isSEHInstruction(const MachineInstr &MI) { 884 unsigned Opc = MI.getOpcode(); 885 switch (Opc) { 886 default: 887 return false; 888 case AArch64::SEH_StackAlloc: 889 case AArch64::SEH_SaveFPLR: 890 case AArch64::SEH_SaveFPLR_X: 891 case AArch64::SEH_SaveReg: 892 case AArch64::SEH_SaveReg_X: 893 case AArch64::SEH_SaveRegP: 894 case AArch64::SEH_SaveRegP_X: 895 case AArch64::SEH_SaveFReg: 896 case AArch64::SEH_SaveFReg_X: 897 case AArch64::SEH_SaveFRegP: 898 case AArch64::SEH_SaveFRegP_X: 899 case AArch64::SEH_SetFP: 900 case AArch64::SEH_AddFP: 901 case AArch64::SEH_Nop: 902 case AArch64::SEH_PrologEnd: 903 case AArch64::SEH_EpilogStart: 904 case AArch64::SEH_EpilogEnd: 905 return true; 906 } 907 } 908 909 bool AArch64InstrInfo::isCoalescableExtInstr(const MachineInstr &MI, 910 unsigned &SrcReg, unsigned &DstReg, 911 unsigned &SubIdx) const { 912 switch (MI.getOpcode()) { 913 default: 914 return false; 915 case AArch64::SBFMXri: // aka sxtw 916 case AArch64::UBFMXri: // aka uxtw 917 // Check for the 32 -> 64 bit extension case, these instructions can do 918 // much more. 919 if (MI.getOperand(2).getImm() != 0 || MI.getOperand(3).getImm() != 31) 920 return false; 921 // This is a signed or unsigned 32 -> 64 bit extension. 922 SrcReg = MI.getOperand(1).getReg(); 923 DstReg = MI.getOperand(0).getReg(); 924 SubIdx = AArch64::sub_32; 925 return true; 926 } 927 } 928 929 bool AArch64InstrInfo::areMemAccessesTriviallyDisjoint( 930 const MachineInstr &MIa, const MachineInstr &MIb, AliasAnalysis *AA) const { 931 const TargetRegisterInfo *TRI = &getRegisterInfo(); 932 const MachineOperand *BaseOpA = nullptr, *BaseOpB = nullptr; 933 int64_t OffsetA = 0, OffsetB = 0; 934 unsigned WidthA = 0, WidthB = 0; 935 936 assert(MIa.mayLoadOrStore() && "MIa must be a load or store."); 937 assert(MIb.mayLoadOrStore() && "MIb must be a load or store."); 938 939 if (MIa.hasUnmodeledSideEffects() || MIb.hasUnmodeledSideEffects() || 940 MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef()) 941 return false; 942 943 // Retrieve the base, offset from the base and width. Width 944 // is the size of memory that is being loaded/stored (e.g. 1, 2, 4, 8). If 945 // base are identical, and the offset of a lower memory access + 946 // the width doesn't overlap the offset of a higher memory access, 947 // then the memory accesses are different. 948 if (getMemOperandWithOffsetWidth(MIa, BaseOpA, OffsetA, WidthA, TRI) && 949 getMemOperandWithOffsetWidth(MIb, BaseOpB, OffsetB, WidthB, TRI)) { 950 if (BaseOpA->isIdenticalTo(*BaseOpB)) { 951 int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB; 952 int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA; 953 int LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB; 954 if (LowOffset + LowWidth <= HighOffset) 955 return true; 956 } 957 } 958 return false; 959 } 960 961 bool AArch64InstrInfo::isSchedulingBoundary(const MachineInstr &MI, 962 const MachineBasicBlock *MBB, 963 const MachineFunction &MF) const { 964 if (TargetInstrInfo::isSchedulingBoundary(MI, MBB, MF)) 965 return true; 966 switch (MI.getOpcode()) { 967 case AArch64::HINT: 968 // CSDB hints are scheduling barriers. 969 if (MI.getOperand(0).getImm() == 0x14) 970 return true; 971 break; 972 case AArch64::DSB: 973 case AArch64::ISB: 974 // DSB and ISB also are scheduling barriers. 975 return true; 976 default:; 977 } 978 return isSEHInstruction(MI); 979 } 980 981 /// analyzeCompare - For a comparison instruction, return the source registers 982 /// in SrcReg and SrcReg2, and the value it compares against in CmpValue. 983 /// Return true if the comparison instruction can be analyzed. 984 bool AArch64InstrInfo::analyzeCompare(const MachineInstr &MI, unsigned &SrcReg, 985 unsigned &SrcReg2, int &CmpMask, 986 int &CmpValue) const { 987 // The first operand can be a frame index where we'd normally expect a 988 // register. 989 assert(MI.getNumOperands() >= 2 && "All AArch64 cmps should have 2 operands"); 990 if (!MI.getOperand(1).isReg()) 991 return false; 992 993 switch (MI.getOpcode()) { 994 default: 995 break; 996 case AArch64::SUBSWrr: 997 case AArch64::SUBSWrs: 998 case AArch64::SUBSWrx: 999 case AArch64::SUBSXrr: 1000 case AArch64::SUBSXrs: 1001 case AArch64::SUBSXrx: 1002 case AArch64::ADDSWrr: 1003 case AArch64::ADDSWrs: 1004 case AArch64::ADDSWrx: 1005 case AArch64::ADDSXrr: 1006 case AArch64::ADDSXrs: 1007 case AArch64::ADDSXrx: 1008 // Replace SUBSWrr with SUBWrr if NZCV is not used. 1009 SrcReg = MI.getOperand(1).getReg(); 1010 SrcReg2 = MI.getOperand(2).getReg(); 1011 CmpMask = ~0; 1012 CmpValue = 0; 1013 return true; 1014 case AArch64::SUBSWri: 1015 case AArch64::ADDSWri: 1016 case AArch64::SUBSXri: 1017 case AArch64::ADDSXri: 1018 SrcReg = MI.getOperand(1).getReg(); 1019 SrcReg2 = 0; 1020 CmpMask = ~0; 1021 // FIXME: In order to convert CmpValue to 0 or 1 1022 CmpValue = MI.getOperand(2).getImm() != 0; 1023 return true; 1024 case AArch64::ANDSWri: 1025 case AArch64::ANDSXri: 1026 // ANDS does not use the same encoding scheme as the others xxxS 1027 // instructions. 1028 SrcReg = MI.getOperand(1).getReg(); 1029 SrcReg2 = 0; 1030 CmpMask = ~0; 1031 // FIXME:The return val type of decodeLogicalImmediate is uint64_t, 1032 // while the type of CmpValue is int. When converting uint64_t to int, 1033 // the high 32 bits of uint64_t will be lost. 1034 // In fact it causes a bug in spec2006-483.xalancbmk 1035 // CmpValue is only used to compare with zero in OptimizeCompareInstr 1036 CmpValue = AArch64_AM::decodeLogicalImmediate( 1037 MI.getOperand(2).getImm(), 1038 MI.getOpcode() == AArch64::ANDSWri ? 32 : 64) != 0; 1039 return true; 1040 } 1041 1042 return false; 1043 } 1044 1045 static bool UpdateOperandRegClass(MachineInstr &Instr) { 1046 MachineBasicBlock *MBB = Instr.getParent(); 1047 assert(MBB && "Can't get MachineBasicBlock here"); 1048 MachineFunction *MF = MBB->getParent(); 1049 assert(MF && "Can't get MachineFunction here"); 1050 const TargetInstrInfo *TII = MF->getSubtarget().getInstrInfo(); 1051 const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo(); 1052 MachineRegisterInfo *MRI = &MF->getRegInfo(); 1053 1054 for (unsigned OpIdx = 0, EndIdx = Instr.getNumOperands(); OpIdx < EndIdx; 1055 ++OpIdx) { 1056 MachineOperand &MO = Instr.getOperand(OpIdx); 1057 const TargetRegisterClass *OpRegCstraints = 1058 Instr.getRegClassConstraint(OpIdx, TII, TRI); 1059 1060 // If there's no constraint, there's nothing to do. 1061 if (!OpRegCstraints) 1062 continue; 1063 // If the operand is a frame index, there's nothing to do here. 1064 // A frame index operand will resolve correctly during PEI. 1065 if (MO.isFI()) 1066 continue; 1067 1068 assert(MO.isReg() && 1069 "Operand has register constraints without being a register!"); 1070 1071 unsigned Reg = MO.getReg(); 1072 if (TargetRegisterInfo::isPhysicalRegister(Reg)) { 1073 if (!OpRegCstraints->contains(Reg)) 1074 return false; 1075 } else if (!OpRegCstraints->hasSubClassEq(MRI->getRegClass(Reg)) && 1076 !MRI->constrainRegClass(Reg, OpRegCstraints)) 1077 return false; 1078 } 1079 1080 return true; 1081 } 1082 1083 /// Return the opcode that does not set flags when possible - otherwise 1084 /// return the original opcode. The caller is responsible to do the actual 1085 /// substitution and legality checking. 1086 static unsigned convertToNonFlagSettingOpc(const MachineInstr &MI) { 1087 // Don't convert all compare instructions, because for some the zero register 1088 // encoding becomes the sp register. 1089 bool MIDefinesZeroReg = false; 1090 if (MI.definesRegister(AArch64::WZR) || MI.definesRegister(AArch64::XZR)) 1091 MIDefinesZeroReg = true; 1092 1093 switch (MI.getOpcode()) { 1094 default: 1095 return MI.getOpcode(); 1096 case AArch64::ADDSWrr: 1097 return AArch64::ADDWrr; 1098 case AArch64::ADDSWri: 1099 return MIDefinesZeroReg ? AArch64::ADDSWri : AArch64::ADDWri; 1100 case AArch64::ADDSWrs: 1101 return MIDefinesZeroReg ? AArch64::ADDSWrs : AArch64::ADDWrs; 1102 case AArch64::ADDSWrx: 1103 return AArch64::ADDWrx; 1104 case AArch64::ADDSXrr: 1105 return AArch64::ADDXrr; 1106 case AArch64::ADDSXri: 1107 return MIDefinesZeroReg ? AArch64::ADDSXri : AArch64::ADDXri; 1108 case AArch64::ADDSXrs: 1109 return MIDefinesZeroReg ? AArch64::ADDSXrs : AArch64::ADDXrs; 1110 case AArch64::ADDSXrx: 1111 return AArch64::ADDXrx; 1112 case AArch64::SUBSWrr: 1113 return AArch64::SUBWrr; 1114 case AArch64::SUBSWri: 1115 return MIDefinesZeroReg ? AArch64::SUBSWri : AArch64::SUBWri; 1116 case AArch64::SUBSWrs: 1117 return MIDefinesZeroReg ? AArch64::SUBSWrs : AArch64::SUBWrs; 1118 case AArch64::SUBSWrx: 1119 return AArch64::SUBWrx; 1120 case AArch64::SUBSXrr: 1121 return AArch64::SUBXrr; 1122 case AArch64::SUBSXri: 1123 return MIDefinesZeroReg ? AArch64::SUBSXri : AArch64::SUBXri; 1124 case AArch64::SUBSXrs: 1125 return MIDefinesZeroReg ? AArch64::SUBSXrs : AArch64::SUBXrs; 1126 case AArch64::SUBSXrx: 1127 return AArch64::SUBXrx; 1128 } 1129 } 1130 1131 enum AccessKind { AK_Write = 0x01, AK_Read = 0x10, AK_All = 0x11 }; 1132 1133 /// True when condition flags are accessed (either by writing or reading) 1134 /// on the instruction trace starting at From and ending at To. 1135 /// 1136 /// Note: If From and To are from different blocks it's assumed CC are accessed 1137 /// on the path. 1138 static bool areCFlagsAccessedBetweenInstrs( 1139 MachineBasicBlock::iterator From, MachineBasicBlock::iterator To, 1140 const TargetRegisterInfo *TRI, const AccessKind AccessToCheck = AK_All) { 1141 // Early exit if To is at the beginning of the BB. 1142 if (To == To->getParent()->begin()) 1143 return true; 1144 1145 // Check whether the instructions are in the same basic block 1146 // If not, assume the condition flags might get modified somewhere. 1147 if (To->getParent() != From->getParent()) 1148 return true; 1149 1150 // From must be above To. 1151 assert(std::find_if(++To.getReverse(), To->getParent()->rend(), 1152 [From](MachineInstr &MI) { 1153 return MI.getIterator() == From; 1154 }) != To->getParent()->rend()); 1155 1156 // We iterate backward starting \p To until we hit \p From. 1157 for (--To; To != From; --To) { 1158 const MachineInstr &Instr = *To; 1159 1160 if (((AccessToCheck & AK_Write) && 1161 Instr.modifiesRegister(AArch64::NZCV, TRI)) || 1162 ((AccessToCheck & AK_Read) && Instr.readsRegister(AArch64::NZCV, TRI))) 1163 return true; 1164 } 1165 return false; 1166 } 1167 1168 /// Try to optimize a compare instruction. A compare instruction is an 1169 /// instruction which produces AArch64::NZCV. It can be truly compare 1170 /// instruction 1171 /// when there are no uses of its destination register. 1172 /// 1173 /// The following steps are tried in order: 1174 /// 1. Convert CmpInstr into an unconditional version. 1175 /// 2. Remove CmpInstr if above there is an instruction producing a needed 1176 /// condition code or an instruction which can be converted into such an 1177 /// instruction. 1178 /// Only comparison with zero is supported. 1179 bool AArch64InstrInfo::optimizeCompareInstr( 1180 MachineInstr &CmpInstr, unsigned SrcReg, unsigned SrcReg2, int CmpMask, 1181 int CmpValue, const MachineRegisterInfo *MRI) const { 1182 assert(CmpInstr.getParent()); 1183 assert(MRI); 1184 1185 // Replace SUBSWrr with SUBWrr if NZCV is not used. 1186 int DeadNZCVIdx = CmpInstr.findRegisterDefOperandIdx(AArch64::NZCV, true); 1187 if (DeadNZCVIdx != -1) { 1188 if (CmpInstr.definesRegister(AArch64::WZR) || 1189 CmpInstr.definesRegister(AArch64::XZR)) { 1190 CmpInstr.eraseFromParent(); 1191 return true; 1192 } 1193 unsigned Opc = CmpInstr.getOpcode(); 1194 unsigned NewOpc = convertToNonFlagSettingOpc(CmpInstr); 1195 if (NewOpc == Opc) 1196 return false; 1197 const MCInstrDesc &MCID = get(NewOpc); 1198 CmpInstr.setDesc(MCID); 1199 CmpInstr.RemoveOperand(DeadNZCVIdx); 1200 bool succeeded = UpdateOperandRegClass(CmpInstr); 1201 (void)succeeded; 1202 assert(succeeded && "Some operands reg class are incompatible!"); 1203 return true; 1204 } 1205 1206 // Continue only if we have a "ri" where immediate is zero. 1207 // FIXME:CmpValue has already been converted to 0 or 1 in analyzeCompare 1208 // function. 1209 assert((CmpValue == 0 || CmpValue == 1) && "CmpValue must be 0 or 1!"); 1210 if (CmpValue != 0 || SrcReg2 != 0) 1211 return false; 1212 1213 // CmpInstr is a Compare instruction if destination register is not used. 1214 if (!MRI->use_nodbg_empty(CmpInstr.getOperand(0).getReg())) 1215 return false; 1216 1217 return substituteCmpToZero(CmpInstr, SrcReg, MRI); 1218 } 1219 1220 /// Get opcode of S version of Instr. 1221 /// If Instr is S version its opcode is returned. 1222 /// AArch64::INSTRUCTION_LIST_END is returned if Instr does not have S version 1223 /// or we are not interested in it. 1224 static unsigned sForm(MachineInstr &Instr) { 1225 switch (Instr.getOpcode()) { 1226 default: 1227 return AArch64::INSTRUCTION_LIST_END; 1228 1229 case AArch64::ADDSWrr: 1230 case AArch64::ADDSWri: 1231 case AArch64::ADDSXrr: 1232 case AArch64::ADDSXri: 1233 case AArch64::SUBSWrr: 1234 case AArch64::SUBSWri: 1235 case AArch64::SUBSXrr: 1236 case AArch64::SUBSXri: 1237 return Instr.getOpcode(); 1238 1239 case AArch64::ADDWrr: 1240 return AArch64::ADDSWrr; 1241 case AArch64::ADDWri: 1242 return AArch64::ADDSWri; 1243 case AArch64::ADDXrr: 1244 return AArch64::ADDSXrr; 1245 case AArch64::ADDXri: 1246 return AArch64::ADDSXri; 1247 case AArch64::ADCWr: 1248 return AArch64::ADCSWr; 1249 case AArch64::ADCXr: 1250 return AArch64::ADCSXr; 1251 case AArch64::SUBWrr: 1252 return AArch64::SUBSWrr; 1253 case AArch64::SUBWri: 1254 return AArch64::SUBSWri; 1255 case AArch64::SUBXrr: 1256 return AArch64::SUBSXrr; 1257 case AArch64::SUBXri: 1258 return AArch64::SUBSXri; 1259 case AArch64::SBCWr: 1260 return AArch64::SBCSWr; 1261 case AArch64::SBCXr: 1262 return AArch64::SBCSXr; 1263 case AArch64::ANDWri: 1264 return AArch64::ANDSWri; 1265 case AArch64::ANDXri: 1266 return AArch64::ANDSXri; 1267 } 1268 } 1269 1270 /// Check if AArch64::NZCV should be alive in successors of MBB. 1271 static bool areCFlagsAliveInSuccessors(MachineBasicBlock *MBB) { 1272 for (auto *BB : MBB->successors()) 1273 if (BB->isLiveIn(AArch64::NZCV)) 1274 return true; 1275 return false; 1276 } 1277 1278 namespace { 1279 1280 struct UsedNZCV { 1281 bool N = false; 1282 bool Z = false; 1283 bool C = false; 1284 bool V = false; 1285 1286 UsedNZCV() = default; 1287 1288 UsedNZCV &operator|=(const UsedNZCV &UsedFlags) { 1289 this->N |= UsedFlags.N; 1290 this->Z |= UsedFlags.Z; 1291 this->C |= UsedFlags.C; 1292 this->V |= UsedFlags.V; 1293 return *this; 1294 } 1295 }; 1296 1297 } // end anonymous namespace 1298 1299 /// Find a condition code used by the instruction. 1300 /// Returns AArch64CC::Invalid if either the instruction does not use condition 1301 /// codes or we don't optimize CmpInstr in the presence of such instructions. 1302 static AArch64CC::CondCode findCondCodeUsedByInstr(const MachineInstr &Instr) { 1303 switch (Instr.getOpcode()) { 1304 default: 1305 return AArch64CC::Invalid; 1306 1307 case AArch64::Bcc: { 1308 int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV); 1309 assert(Idx >= 2); 1310 return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 2).getImm()); 1311 } 1312 1313 case AArch64::CSINVWr: 1314 case AArch64::CSINVXr: 1315 case AArch64::CSINCWr: 1316 case AArch64::CSINCXr: 1317 case AArch64::CSELWr: 1318 case AArch64::CSELXr: 1319 case AArch64::CSNEGWr: 1320 case AArch64::CSNEGXr: 1321 case AArch64::FCSELSrrr: 1322 case AArch64::FCSELDrrr: { 1323 int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV); 1324 assert(Idx >= 1); 1325 return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 1).getImm()); 1326 } 1327 } 1328 } 1329 1330 static UsedNZCV getUsedNZCV(AArch64CC::CondCode CC) { 1331 assert(CC != AArch64CC::Invalid); 1332 UsedNZCV UsedFlags; 1333 switch (CC) { 1334 default: 1335 break; 1336 1337 case AArch64CC::EQ: // Z set 1338 case AArch64CC::NE: // Z clear 1339 UsedFlags.Z = true; 1340 break; 1341 1342 case AArch64CC::HI: // Z clear and C set 1343 case AArch64CC::LS: // Z set or C clear 1344 UsedFlags.Z = true; 1345 LLVM_FALLTHROUGH; 1346 case AArch64CC::HS: // C set 1347 case AArch64CC::LO: // C clear 1348 UsedFlags.C = true; 1349 break; 1350 1351 case AArch64CC::MI: // N set 1352 case AArch64CC::PL: // N clear 1353 UsedFlags.N = true; 1354 break; 1355 1356 case AArch64CC::VS: // V set 1357 case AArch64CC::VC: // V clear 1358 UsedFlags.V = true; 1359 break; 1360 1361 case AArch64CC::GT: // Z clear, N and V the same 1362 case AArch64CC::LE: // Z set, N and V differ 1363 UsedFlags.Z = true; 1364 LLVM_FALLTHROUGH; 1365 case AArch64CC::GE: // N and V the same 1366 case AArch64CC::LT: // N and V differ 1367 UsedFlags.N = true; 1368 UsedFlags.V = true; 1369 break; 1370 } 1371 return UsedFlags; 1372 } 1373 1374 static bool isADDSRegImm(unsigned Opcode) { 1375 return Opcode == AArch64::ADDSWri || Opcode == AArch64::ADDSXri; 1376 } 1377 1378 static bool isSUBSRegImm(unsigned Opcode) { 1379 return Opcode == AArch64::SUBSWri || Opcode == AArch64::SUBSXri; 1380 } 1381 1382 /// Check if CmpInstr can be substituted by MI. 1383 /// 1384 /// CmpInstr can be substituted: 1385 /// - CmpInstr is either 'ADDS %vreg, 0' or 'SUBS %vreg, 0' 1386 /// - and, MI and CmpInstr are from the same MachineBB 1387 /// - and, condition flags are not alive in successors of the CmpInstr parent 1388 /// - and, if MI opcode is the S form there must be no defs of flags between 1389 /// MI and CmpInstr 1390 /// or if MI opcode is not the S form there must be neither defs of flags 1391 /// nor uses of flags between MI and CmpInstr. 1392 /// - and C/V flags are not used after CmpInstr 1393 static bool canInstrSubstituteCmpInstr(MachineInstr *MI, MachineInstr *CmpInstr, 1394 const TargetRegisterInfo *TRI) { 1395 assert(MI); 1396 assert(sForm(*MI) != AArch64::INSTRUCTION_LIST_END); 1397 assert(CmpInstr); 1398 1399 const unsigned CmpOpcode = CmpInstr->getOpcode(); 1400 if (!isADDSRegImm(CmpOpcode) && !isSUBSRegImm(CmpOpcode)) 1401 return false; 1402 1403 if (MI->getParent() != CmpInstr->getParent()) 1404 return false; 1405 1406 if (areCFlagsAliveInSuccessors(CmpInstr->getParent())) 1407 return false; 1408 1409 AccessKind AccessToCheck = AK_Write; 1410 if (sForm(*MI) != MI->getOpcode()) 1411 AccessToCheck = AK_All; 1412 if (areCFlagsAccessedBetweenInstrs(MI, CmpInstr, TRI, AccessToCheck)) 1413 return false; 1414 1415 UsedNZCV NZCVUsedAfterCmp; 1416 for (auto I = std::next(CmpInstr->getIterator()), 1417 E = CmpInstr->getParent()->instr_end(); 1418 I != E; ++I) { 1419 const MachineInstr &Instr = *I; 1420 if (Instr.readsRegister(AArch64::NZCV, TRI)) { 1421 AArch64CC::CondCode CC = findCondCodeUsedByInstr(Instr); 1422 if (CC == AArch64CC::Invalid) // Unsupported conditional instruction 1423 return false; 1424 NZCVUsedAfterCmp |= getUsedNZCV(CC); 1425 } 1426 1427 if (Instr.modifiesRegister(AArch64::NZCV, TRI)) 1428 break; 1429 } 1430 1431 return !NZCVUsedAfterCmp.C && !NZCVUsedAfterCmp.V; 1432 } 1433 1434 /// Substitute an instruction comparing to zero with another instruction 1435 /// which produces needed condition flags. 1436 /// 1437 /// Return true on success. 1438 bool AArch64InstrInfo::substituteCmpToZero( 1439 MachineInstr &CmpInstr, unsigned SrcReg, 1440 const MachineRegisterInfo *MRI) const { 1441 assert(MRI); 1442 // Get the unique definition of SrcReg. 1443 MachineInstr *MI = MRI->getUniqueVRegDef(SrcReg); 1444 if (!MI) 1445 return false; 1446 1447 const TargetRegisterInfo *TRI = &getRegisterInfo(); 1448 1449 unsigned NewOpc = sForm(*MI); 1450 if (NewOpc == AArch64::INSTRUCTION_LIST_END) 1451 return false; 1452 1453 if (!canInstrSubstituteCmpInstr(MI, &CmpInstr, TRI)) 1454 return false; 1455 1456 // Update the instruction to set NZCV. 1457 MI->setDesc(get(NewOpc)); 1458 CmpInstr.eraseFromParent(); 1459 bool succeeded = UpdateOperandRegClass(*MI); 1460 (void)succeeded; 1461 assert(succeeded && "Some operands reg class are incompatible!"); 1462 MI->addRegisterDefined(AArch64::NZCV, TRI); 1463 return true; 1464 } 1465 1466 bool AArch64InstrInfo::expandPostRAPseudo(MachineInstr &MI) const { 1467 if (MI.getOpcode() != TargetOpcode::LOAD_STACK_GUARD && 1468 MI.getOpcode() != AArch64::CATCHRET) 1469 return false; 1470 1471 MachineBasicBlock &MBB = *MI.getParent(); 1472 DebugLoc DL = MI.getDebugLoc(); 1473 1474 if (MI.getOpcode() == AArch64::CATCHRET) { 1475 // Skip to the first instruction before the epilog. 1476 const TargetInstrInfo *TII = 1477 MBB.getParent()->getSubtarget().getInstrInfo(); 1478 MachineBasicBlock *TargetMBB = MI.getOperand(0).getMBB(); 1479 auto MBBI = MachineBasicBlock::iterator(MI); 1480 MachineBasicBlock::iterator FirstEpilogSEH = std::prev(MBBI); 1481 while (FirstEpilogSEH->getFlag(MachineInstr::FrameDestroy) && 1482 FirstEpilogSEH != MBB.begin()) 1483 FirstEpilogSEH = std::prev(FirstEpilogSEH); 1484 if (FirstEpilogSEH != MBB.begin()) 1485 FirstEpilogSEH = std::next(FirstEpilogSEH); 1486 BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADRP)) 1487 .addReg(AArch64::X0, RegState::Define) 1488 .addMBB(TargetMBB); 1489 BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADDXri)) 1490 .addReg(AArch64::X0, RegState::Define) 1491 .addReg(AArch64::X0) 1492 .addMBB(TargetMBB) 1493 .addImm(0); 1494 return true; 1495 } 1496 1497 unsigned Reg = MI.getOperand(0).getReg(); 1498 const GlobalValue *GV = 1499 cast<GlobalValue>((*MI.memoperands_begin())->getValue()); 1500 const TargetMachine &TM = MBB.getParent()->getTarget(); 1501 unsigned char OpFlags = Subtarget.ClassifyGlobalReference(GV, TM); 1502 const unsigned char MO_NC = AArch64II::MO_NC; 1503 1504 if ((OpFlags & AArch64II::MO_GOT) != 0) { 1505 BuildMI(MBB, MI, DL, get(AArch64::LOADgot), Reg) 1506 .addGlobalAddress(GV, 0, OpFlags); 1507 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg) 1508 .addReg(Reg, RegState::Kill) 1509 .addImm(0) 1510 .addMemOperand(*MI.memoperands_begin()); 1511 } else if (TM.getCodeModel() == CodeModel::Large) { 1512 BuildMI(MBB, MI, DL, get(AArch64::MOVZXi), Reg) 1513 .addGlobalAddress(GV, 0, AArch64II::MO_G0 | MO_NC) 1514 .addImm(0); 1515 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg) 1516 .addReg(Reg, RegState::Kill) 1517 .addGlobalAddress(GV, 0, AArch64II::MO_G1 | MO_NC) 1518 .addImm(16); 1519 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg) 1520 .addReg(Reg, RegState::Kill) 1521 .addGlobalAddress(GV, 0, AArch64II::MO_G2 | MO_NC) 1522 .addImm(32); 1523 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg) 1524 .addReg(Reg, RegState::Kill) 1525 .addGlobalAddress(GV, 0, AArch64II::MO_G3) 1526 .addImm(48); 1527 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg) 1528 .addReg(Reg, RegState::Kill) 1529 .addImm(0) 1530 .addMemOperand(*MI.memoperands_begin()); 1531 } else if (TM.getCodeModel() == CodeModel::Tiny) { 1532 BuildMI(MBB, MI, DL, get(AArch64::ADR), Reg) 1533 .addGlobalAddress(GV, 0, OpFlags); 1534 } else { 1535 BuildMI(MBB, MI, DL, get(AArch64::ADRP), Reg) 1536 .addGlobalAddress(GV, 0, OpFlags | AArch64II::MO_PAGE); 1537 unsigned char LoFlags = OpFlags | AArch64II::MO_PAGEOFF | MO_NC; 1538 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg) 1539 .addReg(Reg, RegState::Kill) 1540 .addGlobalAddress(GV, 0, LoFlags) 1541 .addMemOperand(*MI.memoperands_begin()); 1542 } 1543 1544 MBB.erase(MI); 1545 1546 return true; 1547 } 1548 1549 // Return true if this instruction simply sets its single destination register 1550 // to zero. This is equivalent to a register rename of the zero-register. 1551 bool AArch64InstrInfo::isGPRZero(const MachineInstr &MI) { 1552 switch (MI.getOpcode()) { 1553 default: 1554 break; 1555 case AArch64::MOVZWi: 1556 case AArch64::MOVZXi: // movz Rd, #0 (LSL #0) 1557 if (MI.getOperand(1).isImm() && MI.getOperand(1).getImm() == 0) { 1558 assert(MI.getDesc().getNumOperands() == 3 && 1559 MI.getOperand(2).getImm() == 0 && "invalid MOVZi operands"); 1560 return true; 1561 } 1562 break; 1563 case AArch64::ANDWri: // and Rd, Rzr, #imm 1564 return MI.getOperand(1).getReg() == AArch64::WZR; 1565 case AArch64::ANDXri: 1566 return MI.getOperand(1).getReg() == AArch64::XZR; 1567 case TargetOpcode::COPY: 1568 return MI.getOperand(1).getReg() == AArch64::WZR; 1569 } 1570 return false; 1571 } 1572 1573 // Return true if this instruction simply renames a general register without 1574 // modifying bits. 1575 bool AArch64InstrInfo::isGPRCopy(const MachineInstr &MI) { 1576 switch (MI.getOpcode()) { 1577 default: 1578 break; 1579 case TargetOpcode::COPY: { 1580 // GPR32 copies will by lowered to ORRXrs 1581 unsigned DstReg = MI.getOperand(0).getReg(); 1582 return (AArch64::GPR32RegClass.contains(DstReg) || 1583 AArch64::GPR64RegClass.contains(DstReg)); 1584 } 1585 case AArch64::ORRXrs: // orr Xd, Xzr, Xm (LSL #0) 1586 if (MI.getOperand(1).getReg() == AArch64::XZR) { 1587 assert(MI.getDesc().getNumOperands() == 4 && 1588 MI.getOperand(3).getImm() == 0 && "invalid ORRrs operands"); 1589 return true; 1590 } 1591 break; 1592 case AArch64::ADDXri: // add Xd, Xn, #0 (LSL #0) 1593 if (MI.getOperand(2).getImm() == 0) { 1594 assert(MI.getDesc().getNumOperands() == 4 && 1595 MI.getOperand(3).getImm() == 0 && "invalid ADDXri operands"); 1596 return true; 1597 } 1598 break; 1599 } 1600 return false; 1601 } 1602 1603 // Return true if this instruction simply renames a general register without 1604 // modifying bits. 1605 bool AArch64InstrInfo::isFPRCopy(const MachineInstr &MI) { 1606 switch (MI.getOpcode()) { 1607 default: 1608 break; 1609 case TargetOpcode::COPY: { 1610 // FPR64 copies will by lowered to ORR.16b 1611 unsigned DstReg = MI.getOperand(0).getReg(); 1612 return (AArch64::FPR64RegClass.contains(DstReg) || 1613 AArch64::FPR128RegClass.contains(DstReg)); 1614 } 1615 case AArch64::ORRv16i8: 1616 if (MI.getOperand(1).getReg() == MI.getOperand(2).getReg()) { 1617 assert(MI.getDesc().getNumOperands() == 3 && MI.getOperand(0).isReg() && 1618 "invalid ORRv16i8 operands"); 1619 return true; 1620 } 1621 break; 1622 } 1623 return false; 1624 } 1625 1626 unsigned AArch64InstrInfo::isLoadFromStackSlot(const MachineInstr &MI, 1627 int &FrameIndex) const { 1628 switch (MI.getOpcode()) { 1629 default: 1630 break; 1631 case AArch64::LDRWui: 1632 case AArch64::LDRXui: 1633 case AArch64::LDRBui: 1634 case AArch64::LDRHui: 1635 case AArch64::LDRSui: 1636 case AArch64::LDRDui: 1637 case AArch64::LDRQui: 1638 if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() && 1639 MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) { 1640 FrameIndex = MI.getOperand(1).getIndex(); 1641 return MI.getOperand(0).getReg(); 1642 } 1643 break; 1644 } 1645 1646 return 0; 1647 } 1648 1649 unsigned AArch64InstrInfo::isStoreToStackSlot(const MachineInstr &MI, 1650 int &FrameIndex) const { 1651 switch (MI.getOpcode()) { 1652 default: 1653 break; 1654 case AArch64::STRWui: 1655 case AArch64::STRXui: 1656 case AArch64::STRBui: 1657 case AArch64::STRHui: 1658 case AArch64::STRSui: 1659 case AArch64::STRDui: 1660 case AArch64::STRQui: 1661 if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() && 1662 MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) { 1663 FrameIndex = MI.getOperand(1).getIndex(); 1664 return MI.getOperand(0).getReg(); 1665 } 1666 break; 1667 } 1668 return 0; 1669 } 1670 1671 /// Check all MachineMemOperands for a hint to suppress pairing. 1672 bool AArch64InstrInfo::isLdStPairSuppressed(const MachineInstr &MI) { 1673 return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) { 1674 return MMO->getFlags() & MOSuppressPair; 1675 }); 1676 } 1677 1678 /// Set a flag on the first MachineMemOperand to suppress pairing. 1679 void AArch64InstrInfo::suppressLdStPair(MachineInstr &MI) { 1680 if (MI.memoperands_empty()) 1681 return; 1682 (*MI.memoperands_begin())->setFlags(MOSuppressPair); 1683 } 1684 1685 /// Check all MachineMemOperands for a hint that the load/store is strided. 1686 bool AArch64InstrInfo::isStridedAccess(const MachineInstr &MI) { 1687 return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) { 1688 return MMO->getFlags() & MOStridedAccess; 1689 }); 1690 } 1691 1692 bool AArch64InstrInfo::isUnscaledLdSt(unsigned Opc) { 1693 switch (Opc) { 1694 default: 1695 return false; 1696 case AArch64::STURSi: 1697 case AArch64::STURDi: 1698 case AArch64::STURQi: 1699 case AArch64::STURBBi: 1700 case AArch64::STURHHi: 1701 case AArch64::STURWi: 1702 case AArch64::STURXi: 1703 case AArch64::LDURSi: 1704 case AArch64::LDURDi: 1705 case AArch64::LDURQi: 1706 case AArch64::LDURWi: 1707 case AArch64::LDURXi: 1708 case AArch64::LDURSWi: 1709 case AArch64::LDURHHi: 1710 case AArch64::LDURBBi: 1711 case AArch64::LDURSBWi: 1712 case AArch64::LDURSHWi: 1713 return true; 1714 } 1715 } 1716 1717 Optional<unsigned> AArch64InstrInfo::getUnscaledLdSt(unsigned Opc) { 1718 switch (Opc) { 1719 default: return {}; 1720 case AArch64::PRFMui: return AArch64::PRFUMi; 1721 case AArch64::LDRXui: return AArch64::LDURXi; 1722 case AArch64::LDRWui: return AArch64::LDURWi; 1723 case AArch64::LDRBui: return AArch64::LDURBi; 1724 case AArch64::LDRHui: return AArch64::LDURHi; 1725 case AArch64::LDRSui: return AArch64::LDURSi; 1726 case AArch64::LDRDui: return AArch64::LDURDi; 1727 case AArch64::LDRQui: return AArch64::LDURQi; 1728 case AArch64::LDRBBui: return AArch64::LDURBBi; 1729 case AArch64::LDRHHui: return AArch64::LDURHHi; 1730 case AArch64::LDRSBXui: return AArch64::LDURSBXi; 1731 case AArch64::LDRSBWui: return AArch64::LDURSBWi; 1732 case AArch64::LDRSHXui: return AArch64::LDURSHXi; 1733 case AArch64::LDRSHWui: return AArch64::LDURSHWi; 1734 case AArch64::LDRSWui: return AArch64::LDURSWi; 1735 case AArch64::STRXui: return AArch64::STURXi; 1736 case AArch64::STRWui: return AArch64::STURWi; 1737 case AArch64::STRBui: return AArch64::STURBi; 1738 case AArch64::STRHui: return AArch64::STURHi; 1739 case AArch64::STRSui: return AArch64::STURSi; 1740 case AArch64::STRDui: return AArch64::STURDi; 1741 case AArch64::STRQui: return AArch64::STURQi; 1742 case AArch64::STRBBui: return AArch64::STURBBi; 1743 case AArch64::STRHHui: return AArch64::STURHHi; 1744 } 1745 } 1746 1747 unsigned AArch64InstrInfo::getLoadStoreImmIdx(unsigned Opc) { 1748 switch (Opc) { 1749 default: 1750 return 2; 1751 case AArch64::LDPXi: 1752 case AArch64::LDPDi: 1753 case AArch64::STPXi: 1754 case AArch64::STPDi: 1755 case AArch64::LDNPXi: 1756 case AArch64::LDNPDi: 1757 case AArch64::STNPXi: 1758 case AArch64::STNPDi: 1759 case AArch64::LDPQi: 1760 case AArch64::STPQi: 1761 case AArch64::LDNPQi: 1762 case AArch64::STNPQi: 1763 case AArch64::LDPWi: 1764 case AArch64::LDPSi: 1765 case AArch64::STPWi: 1766 case AArch64::STPSi: 1767 case AArch64::LDNPWi: 1768 case AArch64::LDNPSi: 1769 case AArch64::STNPWi: 1770 case AArch64::STNPSi: 1771 case AArch64::LDG: 1772 return 3; 1773 case AArch64::ADDG: 1774 case AArch64::STGOffset: 1775 return 2; 1776 } 1777 } 1778 1779 bool AArch64InstrInfo::isPairableLdStInst(const MachineInstr &MI) { 1780 switch (MI.getOpcode()) { 1781 default: 1782 return false; 1783 // Scaled instructions. 1784 case AArch64::STRSui: 1785 case AArch64::STRDui: 1786 case AArch64::STRQui: 1787 case AArch64::STRXui: 1788 case AArch64::STRWui: 1789 case AArch64::LDRSui: 1790 case AArch64::LDRDui: 1791 case AArch64::LDRQui: 1792 case AArch64::LDRXui: 1793 case AArch64::LDRWui: 1794 case AArch64::LDRSWui: 1795 // Unscaled instructions. 1796 case AArch64::STURSi: 1797 case AArch64::STURDi: 1798 case AArch64::STURQi: 1799 case AArch64::STURWi: 1800 case AArch64::STURXi: 1801 case AArch64::LDURSi: 1802 case AArch64::LDURDi: 1803 case AArch64::LDURQi: 1804 case AArch64::LDURWi: 1805 case AArch64::LDURXi: 1806 case AArch64::LDURSWi: 1807 return true; 1808 } 1809 } 1810 1811 unsigned AArch64InstrInfo::convertToFlagSettingOpc(unsigned Opc, 1812 bool &Is64Bit) { 1813 switch (Opc) { 1814 default: 1815 llvm_unreachable("Opcode has no flag setting equivalent!"); 1816 // 32-bit cases: 1817 case AArch64::ADDWri: 1818 Is64Bit = false; 1819 return AArch64::ADDSWri; 1820 case AArch64::ADDWrr: 1821 Is64Bit = false; 1822 return AArch64::ADDSWrr; 1823 case AArch64::ADDWrs: 1824 Is64Bit = false; 1825 return AArch64::ADDSWrs; 1826 case AArch64::ADDWrx: 1827 Is64Bit = false; 1828 return AArch64::ADDSWrx; 1829 case AArch64::ANDWri: 1830 Is64Bit = false; 1831 return AArch64::ANDSWri; 1832 case AArch64::ANDWrr: 1833 Is64Bit = false; 1834 return AArch64::ANDSWrr; 1835 case AArch64::ANDWrs: 1836 Is64Bit = false; 1837 return AArch64::ANDSWrs; 1838 case AArch64::BICWrr: 1839 Is64Bit = false; 1840 return AArch64::BICSWrr; 1841 case AArch64::BICWrs: 1842 Is64Bit = false; 1843 return AArch64::BICSWrs; 1844 case AArch64::SUBWri: 1845 Is64Bit = false; 1846 return AArch64::SUBSWri; 1847 case AArch64::SUBWrr: 1848 Is64Bit = false; 1849 return AArch64::SUBSWrr; 1850 case AArch64::SUBWrs: 1851 Is64Bit = false; 1852 return AArch64::SUBSWrs; 1853 case AArch64::SUBWrx: 1854 Is64Bit = false; 1855 return AArch64::SUBSWrx; 1856 // 64-bit cases: 1857 case AArch64::ADDXri: 1858 Is64Bit = true; 1859 return AArch64::ADDSXri; 1860 case AArch64::ADDXrr: 1861 Is64Bit = true; 1862 return AArch64::ADDSXrr; 1863 case AArch64::ADDXrs: 1864 Is64Bit = true; 1865 return AArch64::ADDSXrs; 1866 case AArch64::ADDXrx: 1867 Is64Bit = true; 1868 return AArch64::ADDSXrx; 1869 case AArch64::ANDXri: 1870 Is64Bit = true; 1871 return AArch64::ANDSXri; 1872 case AArch64::ANDXrr: 1873 Is64Bit = true; 1874 return AArch64::ANDSXrr; 1875 case AArch64::ANDXrs: 1876 Is64Bit = true; 1877 return AArch64::ANDSXrs; 1878 case AArch64::BICXrr: 1879 Is64Bit = true; 1880 return AArch64::BICSXrr; 1881 case AArch64::BICXrs: 1882 Is64Bit = true; 1883 return AArch64::BICSXrs; 1884 case AArch64::SUBXri: 1885 Is64Bit = true; 1886 return AArch64::SUBSXri; 1887 case AArch64::SUBXrr: 1888 Is64Bit = true; 1889 return AArch64::SUBSXrr; 1890 case AArch64::SUBXrs: 1891 Is64Bit = true; 1892 return AArch64::SUBSXrs; 1893 case AArch64::SUBXrx: 1894 Is64Bit = true; 1895 return AArch64::SUBSXrx; 1896 } 1897 } 1898 1899 // Is this a candidate for ld/st merging or pairing? For example, we don't 1900 // touch volatiles or load/stores that have a hint to avoid pair formation. 1901 bool AArch64InstrInfo::isCandidateToMergeOrPair(const MachineInstr &MI) const { 1902 // If this is a volatile load/store, don't mess with it. 1903 if (MI.hasOrderedMemoryRef()) 1904 return false; 1905 1906 // Make sure this is a reg/fi+imm (as opposed to an address reloc). 1907 assert((MI.getOperand(1).isReg() || MI.getOperand(1).isFI()) && 1908 "Expected a reg or frame index operand."); 1909 if (!MI.getOperand(2).isImm()) 1910 return false; 1911 1912 // Can't merge/pair if the instruction modifies the base register. 1913 // e.g., ldr x0, [x0] 1914 // This case will never occur with an FI base. 1915 if (MI.getOperand(1).isReg()) { 1916 unsigned BaseReg = MI.getOperand(1).getReg(); 1917 const TargetRegisterInfo *TRI = &getRegisterInfo(); 1918 if (MI.modifiesRegister(BaseReg, TRI)) 1919 return false; 1920 } 1921 1922 // Check if this load/store has a hint to avoid pair formation. 1923 // MachineMemOperands hints are set by the AArch64StorePairSuppress pass. 1924 if (isLdStPairSuppressed(MI)) 1925 return false; 1926 1927 // On some CPUs quad load/store pairs are slower than two single load/stores. 1928 if (Subtarget.isPaired128Slow()) { 1929 switch (MI.getOpcode()) { 1930 default: 1931 break; 1932 case AArch64::LDURQi: 1933 case AArch64::STURQi: 1934 case AArch64::LDRQui: 1935 case AArch64::STRQui: 1936 return false; 1937 } 1938 } 1939 1940 return true; 1941 } 1942 1943 bool AArch64InstrInfo::getMemOperandWithOffset(const MachineInstr &LdSt, 1944 const MachineOperand *&BaseOp, 1945 int64_t &Offset, 1946 const TargetRegisterInfo *TRI) const { 1947 unsigned Width; 1948 return getMemOperandWithOffsetWidth(LdSt, BaseOp, Offset, Width, TRI); 1949 } 1950 1951 bool AArch64InstrInfo::getMemOperandWithOffsetWidth( 1952 const MachineInstr &LdSt, const MachineOperand *&BaseOp, int64_t &Offset, 1953 unsigned &Width, const TargetRegisterInfo *TRI) const { 1954 assert(LdSt.mayLoadOrStore() && "Expected a memory operation."); 1955 // Handle only loads/stores with base register followed by immediate offset. 1956 if (LdSt.getNumExplicitOperands() == 3) { 1957 // Non-paired instruction (e.g., ldr x1, [x0, #8]). 1958 if ((!LdSt.getOperand(1).isReg() && !LdSt.getOperand(1).isFI()) || 1959 !LdSt.getOperand(2).isImm()) 1960 return false; 1961 } else if (LdSt.getNumExplicitOperands() == 4) { 1962 // Paired instruction (e.g., ldp x1, x2, [x0, #8]). 1963 if (!LdSt.getOperand(1).isReg() || 1964 (!LdSt.getOperand(2).isReg() && !LdSt.getOperand(2).isFI()) || 1965 !LdSt.getOperand(3).isImm()) 1966 return false; 1967 } else 1968 return false; 1969 1970 // Get the scaling factor for the instruction and set the width for the 1971 // instruction. 1972 unsigned Scale = 0; 1973 int64_t Dummy1, Dummy2; 1974 1975 // If this returns false, then it's an instruction we don't want to handle. 1976 if (!getMemOpInfo(LdSt.getOpcode(), Scale, Width, Dummy1, Dummy2)) 1977 return false; 1978 1979 // Compute the offset. Offset is calculated as the immediate operand 1980 // multiplied by the scaling factor. Unscaled instructions have scaling factor 1981 // set to 1. 1982 if (LdSt.getNumExplicitOperands() == 3) { 1983 BaseOp = &LdSt.getOperand(1); 1984 Offset = LdSt.getOperand(2).getImm() * Scale; 1985 } else { 1986 assert(LdSt.getNumExplicitOperands() == 4 && "invalid number of operands"); 1987 BaseOp = &LdSt.getOperand(2); 1988 Offset = LdSt.getOperand(3).getImm() * Scale; 1989 } 1990 1991 assert((BaseOp->isReg() || BaseOp->isFI()) && 1992 "getMemOperandWithOffset only supports base " 1993 "operands of type register or frame index."); 1994 1995 return true; 1996 } 1997 1998 MachineOperand & 1999 AArch64InstrInfo::getMemOpBaseRegImmOfsOffsetOperand(MachineInstr &LdSt) const { 2000 assert(LdSt.mayLoadOrStore() && "Expected a memory operation."); 2001 MachineOperand &OfsOp = LdSt.getOperand(LdSt.getNumExplicitOperands() - 1); 2002 assert(OfsOp.isImm() && "Offset operand wasn't immediate."); 2003 return OfsOp; 2004 } 2005 2006 bool AArch64InstrInfo::getMemOpInfo(unsigned Opcode, unsigned &Scale, 2007 unsigned &Width, int64_t &MinOffset, 2008 int64_t &MaxOffset) { 2009 switch (Opcode) { 2010 // Not a memory operation or something we want to handle. 2011 default: 2012 Scale = Width = 0; 2013 MinOffset = MaxOffset = 0; 2014 return false; 2015 case AArch64::STRWpost: 2016 case AArch64::LDRWpost: 2017 Width = 32; 2018 Scale = 4; 2019 MinOffset = -256; 2020 MaxOffset = 255; 2021 break; 2022 case AArch64::LDURQi: 2023 case AArch64::STURQi: 2024 Width = 16; 2025 Scale = 1; 2026 MinOffset = -256; 2027 MaxOffset = 255; 2028 break; 2029 case AArch64::PRFUMi: 2030 case AArch64::LDURXi: 2031 case AArch64::LDURDi: 2032 case AArch64::STURXi: 2033 case AArch64::STURDi: 2034 Width = 8; 2035 Scale = 1; 2036 MinOffset = -256; 2037 MaxOffset = 255; 2038 break; 2039 case AArch64::LDURWi: 2040 case AArch64::LDURSi: 2041 case AArch64::LDURSWi: 2042 case AArch64::STURWi: 2043 case AArch64::STURSi: 2044 Width = 4; 2045 Scale = 1; 2046 MinOffset = -256; 2047 MaxOffset = 255; 2048 break; 2049 case AArch64::LDURHi: 2050 case AArch64::LDURHHi: 2051 case AArch64::LDURSHXi: 2052 case AArch64::LDURSHWi: 2053 case AArch64::STURHi: 2054 case AArch64::STURHHi: 2055 Width = 2; 2056 Scale = 1; 2057 MinOffset = -256; 2058 MaxOffset = 255; 2059 break; 2060 case AArch64::LDURBi: 2061 case AArch64::LDURBBi: 2062 case AArch64::LDURSBXi: 2063 case AArch64::LDURSBWi: 2064 case AArch64::STURBi: 2065 case AArch64::STURBBi: 2066 Width = 1; 2067 Scale = 1; 2068 MinOffset = -256; 2069 MaxOffset = 255; 2070 break; 2071 case AArch64::LDPQi: 2072 case AArch64::LDNPQi: 2073 case AArch64::STPQi: 2074 case AArch64::STNPQi: 2075 Scale = 16; 2076 Width = 32; 2077 MinOffset = -64; 2078 MaxOffset = 63; 2079 break; 2080 case AArch64::LDRQui: 2081 case AArch64::STRQui: 2082 Scale = Width = 16; 2083 MinOffset = 0; 2084 MaxOffset = 4095; 2085 break; 2086 case AArch64::LDPXi: 2087 case AArch64::LDPDi: 2088 case AArch64::LDNPXi: 2089 case AArch64::LDNPDi: 2090 case AArch64::STPXi: 2091 case AArch64::STPDi: 2092 case AArch64::STNPXi: 2093 case AArch64::STNPDi: 2094 Scale = 8; 2095 Width = 16; 2096 MinOffset = -64; 2097 MaxOffset = 63; 2098 break; 2099 case AArch64::PRFMui: 2100 case AArch64::LDRXui: 2101 case AArch64::LDRDui: 2102 case AArch64::STRXui: 2103 case AArch64::STRDui: 2104 Scale = Width = 8; 2105 MinOffset = 0; 2106 MaxOffset = 4095; 2107 break; 2108 case AArch64::LDPWi: 2109 case AArch64::LDPSi: 2110 case AArch64::LDNPWi: 2111 case AArch64::LDNPSi: 2112 case AArch64::STPWi: 2113 case AArch64::STPSi: 2114 case AArch64::STNPWi: 2115 case AArch64::STNPSi: 2116 Scale = 4; 2117 Width = 8; 2118 MinOffset = -64; 2119 MaxOffset = 63; 2120 break; 2121 case AArch64::LDRWui: 2122 case AArch64::LDRSui: 2123 case AArch64::LDRSWui: 2124 case AArch64::STRWui: 2125 case AArch64::STRSui: 2126 Scale = Width = 4; 2127 MinOffset = 0; 2128 MaxOffset = 4095; 2129 break; 2130 case AArch64::LDRHui: 2131 case AArch64::LDRHHui: 2132 case AArch64::LDRSHWui: 2133 case AArch64::LDRSHXui: 2134 case AArch64::STRHui: 2135 case AArch64::STRHHui: 2136 Scale = Width = 2; 2137 MinOffset = 0; 2138 MaxOffset = 4095; 2139 break; 2140 case AArch64::LDRBui: 2141 case AArch64::LDRBBui: 2142 case AArch64::LDRSBWui: 2143 case AArch64::LDRSBXui: 2144 case AArch64::STRBui: 2145 case AArch64::STRBBui: 2146 Scale = Width = 1; 2147 MinOffset = 0; 2148 MaxOffset = 4095; 2149 break; 2150 case AArch64::ADDG: 2151 Scale = 16; 2152 Width = 0; 2153 MinOffset = 0; 2154 MaxOffset = 63; 2155 break; 2156 case AArch64::LDG: 2157 case AArch64::STGOffset: 2158 Scale = Width = 16; 2159 MinOffset = -256; 2160 MaxOffset = 255; 2161 break; 2162 } 2163 2164 return true; 2165 } 2166 2167 static unsigned getOffsetStride(unsigned Opc) { 2168 switch (Opc) { 2169 default: 2170 return 0; 2171 case AArch64::LDURQi: 2172 case AArch64::STURQi: 2173 return 16; 2174 case AArch64::LDURXi: 2175 case AArch64::LDURDi: 2176 case AArch64::STURXi: 2177 case AArch64::STURDi: 2178 return 8; 2179 case AArch64::LDURWi: 2180 case AArch64::LDURSi: 2181 case AArch64::LDURSWi: 2182 case AArch64::STURWi: 2183 case AArch64::STURSi: 2184 return 4; 2185 } 2186 } 2187 2188 // Scale the unscaled offsets. Returns false if the unscaled offset can't be 2189 // scaled. 2190 static bool scaleOffset(unsigned Opc, int64_t &Offset) { 2191 unsigned OffsetStride = getOffsetStride(Opc); 2192 if (OffsetStride == 0) 2193 return false; 2194 // If the byte-offset isn't a multiple of the stride, we can't scale this 2195 // offset. 2196 if (Offset % OffsetStride != 0) 2197 return false; 2198 2199 // Convert the byte-offset used by unscaled into an "element" offset used 2200 // by the scaled pair load/store instructions. 2201 Offset /= OffsetStride; 2202 return true; 2203 } 2204 2205 // Unscale the scaled offsets. Returns false if the scaled offset can't be 2206 // unscaled. 2207 static bool unscaleOffset(unsigned Opc, int64_t &Offset) { 2208 unsigned OffsetStride = getOffsetStride(Opc); 2209 if (OffsetStride == 0) 2210 return false; 2211 2212 // Convert the "element" offset used by scaled pair load/store instructions 2213 // into the byte-offset used by unscaled. 2214 Offset *= OffsetStride; 2215 return true; 2216 } 2217 2218 static bool canPairLdStOpc(unsigned FirstOpc, unsigned SecondOpc) { 2219 if (FirstOpc == SecondOpc) 2220 return true; 2221 // We can also pair sign-ext and zero-ext instructions. 2222 switch (FirstOpc) { 2223 default: 2224 return false; 2225 case AArch64::LDRWui: 2226 case AArch64::LDURWi: 2227 return SecondOpc == AArch64::LDRSWui || SecondOpc == AArch64::LDURSWi; 2228 case AArch64::LDRSWui: 2229 case AArch64::LDURSWi: 2230 return SecondOpc == AArch64::LDRWui || SecondOpc == AArch64::LDURWi; 2231 } 2232 // These instructions can't be paired based on their opcodes. 2233 return false; 2234 } 2235 2236 static bool shouldClusterFI(const MachineFrameInfo &MFI, int FI1, 2237 int64_t Offset1, unsigned Opcode1, int FI2, 2238 int64_t Offset2, unsigned Opcode2) { 2239 // Accesses through fixed stack object frame indices may access a different 2240 // fixed stack slot. Check that the object offsets + offsets match. 2241 if (MFI.isFixedObjectIndex(FI1) && MFI.isFixedObjectIndex(FI2)) { 2242 int64_t ObjectOffset1 = MFI.getObjectOffset(FI1); 2243 int64_t ObjectOffset2 = MFI.getObjectOffset(FI2); 2244 assert(ObjectOffset1 <= ObjectOffset2 && "Object offsets are not ordered."); 2245 // Get the byte-offset from the object offset. 2246 if (!unscaleOffset(Opcode1, Offset1) || !unscaleOffset(Opcode2, Offset2)) 2247 return false; 2248 ObjectOffset1 += Offset1; 2249 ObjectOffset2 += Offset2; 2250 // Get the "element" index in the object. 2251 if (!scaleOffset(Opcode1, ObjectOffset1) || 2252 !scaleOffset(Opcode2, ObjectOffset2)) 2253 return false; 2254 return ObjectOffset1 + 1 == ObjectOffset2; 2255 } 2256 2257 return FI1 == FI2; 2258 } 2259 2260 /// Detect opportunities for ldp/stp formation. 2261 /// 2262 /// Only called for LdSt for which getMemOperandWithOffset returns true. 2263 bool AArch64InstrInfo::shouldClusterMemOps(const MachineOperand &BaseOp1, 2264 const MachineOperand &BaseOp2, 2265 unsigned NumLoads) const { 2266 const MachineInstr &FirstLdSt = *BaseOp1.getParent(); 2267 const MachineInstr &SecondLdSt = *BaseOp2.getParent(); 2268 if (BaseOp1.getType() != BaseOp2.getType()) 2269 return false; 2270 2271 assert((BaseOp1.isReg() || BaseOp1.isFI()) && 2272 "Only base registers and frame indices are supported."); 2273 2274 // Check for both base regs and base FI. 2275 if (BaseOp1.isReg() && BaseOp1.getReg() != BaseOp2.getReg()) 2276 return false; 2277 2278 // Only cluster up to a single pair. 2279 if (NumLoads > 1) 2280 return false; 2281 2282 if (!isPairableLdStInst(FirstLdSt) || !isPairableLdStInst(SecondLdSt)) 2283 return false; 2284 2285 // Can we pair these instructions based on their opcodes? 2286 unsigned FirstOpc = FirstLdSt.getOpcode(); 2287 unsigned SecondOpc = SecondLdSt.getOpcode(); 2288 if (!canPairLdStOpc(FirstOpc, SecondOpc)) 2289 return false; 2290 2291 // Can't merge volatiles or load/stores that have a hint to avoid pair 2292 // formation, for example. 2293 if (!isCandidateToMergeOrPair(FirstLdSt) || 2294 !isCandidateToMergeOrPair(SecondLdSt)) 2295 return false; 2296 2297 // isCandidateToMergeOrPair guarantees that operand 2 is an immediate. 2298 int64_t Offset1 = FirstLdSt.getOperand(2).getImm(); 2299 if (isUnscaledLdSt(FirstOpc) && !scaleOffset(FirstOpc, Offset1)) 2300 return false; 2301 2302 int64_t Offset2 = SecondLdSt.getOperand(2).getImm(); 2303 if (isUnscaledLdSt(SecondOpc) && !scaleOffset(SecondOpc, Offset2)) 2304 return false; 2305 2306 // Pairwise instructions have a 7-bit signed offset field. 2307 if (Offset1 > 63 || Offset1 < -64) 2308 return false; 2309 2310 // The caller should already have ordered First/SecondLdSt by offset. 2311 // Note: except for non-equal frame index bases 2312 if (BaseOp1.isFI()) { 2313 assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 >= Offset2) && 2314 "Caller should have ordered offsets."); 2315 2316 const MachineFrameInfo &MFI = 2317 FirstLdSt.getParent()->getParent()->getFrameInfo(); 2318 return shouldClusterFI(MFI, BaseOp1.getIndex(), Offset1, FirstOpc, 2319 BaseOp2.getIndex(), Offset2, SecondOpc); 2320 } 2321 2322 assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 <= Offset2) && 2323 "Caller should have ordered offsets."); 2324 2325 return Offset1 + 1 == Offset2; 2326 } 2327 2328 static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB, 2329 unsigned Reg, unsigned SubIdx, 2330 unsigned State, 2331 const TargetRegisterInfo *TRI) { 2332 if (!SubIdx) 2333 return MIB.addReg(Reg, State); 2334 2335 if (TargetRegisterInfo::isPhysicalRegister(Reg)) 2336 return MIB.addReg(TRI->getSubReg(Reg, SubIdx), State); 2337 return MIB.addReg(Reg, State, SubIdx); 2338 } 2339 2340 static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg, 2341 unsigned NumRegs) { 2342 // We really want the positive remainder mod 32 here, that happens to be 2343 // easily obtainable with a mask. 2344 return ((DestReg - SrcReg) & 0x1f) < NumRegs; 2345 } 2346 2347 void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB, 2348 MachineBasicBlock::iterator I, 2349 const DebugLoc &DL, unsigned DestReg, 2350 unsigned SrcReg, bool KillSrc, 2351 unsigned Opcode, 2352 ArrayRef<unsigned> Indices) const { 2353 assert(Subtarget.hasNEON() && "Unexpected register copy without NEON"); 2354 const TargetRegisterInfo *TRI = &getRegisterInfo(); 2355 uint16_t DestEncoding = TRI->getEncodingValue(DestReg); 2356 uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg); 2357 unsigned NumRegs = Indices.size(); 2358 2359 int SubReg = 0, End = NumRegs, Incr = 1; 2360 if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs)) { 2361 SubReg = NumRegs - 1; 2362 End = -1; 2363 Incr = -1; 2364 } 2365 2366 for (; SubReg != End; SubReg += Incr) { 2367 const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode)); 2368 AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI); 2369 AddSubReg(MIB, SrcReg, Indices[SubReg], 0, TRI); 2370 AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI); 2371 } 2372 } 2373 2374 void AArch64InstrInfo::copyGPRRegTuple(MachineBasicBlock &MBB, 2375 MachineBasicBlock::iterator I, 2376 DebugLoc DL, unsigned DestReg, 2377 unsigned SrcReg, bool KillSrc, 2378 unsigned Opcode, unsigned ZeroReg, 2379 llvm::ArrayRef<unsigned> Indices) const { 2380 const TargetRegisterInfo *TRI = &getRegisterInfo(); 2381 unsigned NumRegs = Indices.size(); 2382 2383 #ifndef NDEBUG 2384 uint16_t DestEncoding = TRI->getEncodingValue(DestReg); 2385 uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg); 2386 assert(DestEncoding % NumRegs == 0 && SrcEncoding % NumRegs == 0 && 2387 "GPR reg sequences should not be able to overlap"); 2388 #endif 2389 2390 for (unsigned SubReg = 0; SubReg != NumRegs; ++SubReg) { 2391 const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode)); 2392 AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI); 2393 MIB.addReg(ZeroReg); 2394 AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI); 2395 MIB.addImm(0); 2396 } 2397 } 2398 2399 void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB, 2400 MachineBasicBlock::iterator I, 2401 const DebugLoc &DL, unsigned DestReg, 2402 unsigned SrcReg, bool KillSrc) const { 2403 if (AArch64::GPR32spRegClass.contains(DestReg) && 2404 (AArch64::GPR32spRegClass.contains(SrcReg) || SrcReg == AArch64::WZR)) { 2405 const TargetRegisterInfo *TRI = &getRegisterInfo(); 2406 2407 if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) { 2408 // If either operand is WSP, expand to ADD #0. 2409 if (Subtarget.hasZeroCycleRegMove()) { 2410 // Cyclone recognizes "ADD Xd, Xn, #0" as a zero-cycle register move. 2411 unsigned DestRegX = TRI->getMatchingSuperReg(DestReg, AArch64::sub_32, 2412 &AArch64::GPR64spRegClass); 2413 unsigned SrcRegX = TRI->getMatchingSuperReg(SrcReg, AArch64::sub_32, 2414 &AArch64::GPR64spRegClass); 2415 // This instruction is reading and writing X registers. This may upset 2416 // the register scavenger and machine verifier, so we need to indicate 2417 // that we are reading an undefined value from SrcRegX, but a proper 2418 // value from SrcReg. 2419 BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestRegX) 2420 .addReg(SrcRegX, RegState::Undef) 2421 .addImm(0) 2422 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)) 2423 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc)); 2424 } else { 2425 BuildMI(MBB, I, DL, get(AArch64::ADDWri), DestReg) 2426 .addReg(SrcReg, getKillRegState(KillSrc)) 2427 .addImm(0) 2428 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2429 } 2430 } else if (SrcReg == AArch64::WZR && Subtarget.hasZeroCycleZeroingGP()) { 2431 BuildMI(MBB, I, DL, get(AArch64::MOVZWi), DestReg) 2432 .addImm(0) 2433 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2434 } else { 2435 if (Subtarget.hasZeroCycleRegMove()) { 2436 // Cyclone recognizes "ORR Xd, XZR, Xm" as a zero-cycle register move. 2437 unsigned DestRegX = TRI->getMatchingSuperReg(DestReg, AArch64::sub_32, 2438 &AArch64::GPR64spRegClass); 2439 unsigned SrcRegX = TRI->getMatchingSuperReg(SrcReg, AArch64::sub_32, 2440 &AArch64::GPR64spRegClass); 2441 // This instruction is reading and writing X registers. This may upset 2442 // the register scavenger and machine verifier, so we need to indicate 2443 // that we are reading an undefined value from SrcRegX, but a proper 2444 // value from SrcReg. 2445 BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestRegX) 2446 .addReg(AArch64::XZR) 2447 .addReg(SrcRegX, RegState::Undef) 2448 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc)); 2449 } else { 2450 // Otherwise, expand to ORR WZR. 2451 BuildMI(MBB, I, DL, get(AArch64::ORRWrr), DestReg) 2452 .addReg(AArch64::WZR) 2453 .addReg(SrcReg, getKillRegState(KillSrc)); 2454 } 2455 } 2456 return; 2457 } 2458 2459 if (AArch64::GPR64spRegClass.contains(DestReg) && 2460 (AArch64::GPR64spRegClass.contains(SrcReg) || SrcReg == AArch64::XZR)) { 2461 if (DestReg == AArch64::SP || SrcReg == AArch64::SP) { 2462 // If either operand is SP, expand to ADD #0. 2463 BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestReg) 2464 .addReg(SrcReg, getKillRegState(KillSrc)) 2465 .addImm(0) 2466 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2467 } else if (SrcReg == AArch64::XZR && Subtarget.hasZeroCycleZeroingGP()) { 2468 BuildMI(MBB, I, DL, get(AArch64::MOVZXi), DestReg) 2469 .addImm(0) 2470 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2471 } else { 2472 // Otherwise, expand to ORR XZR. 2473 BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestReg) 2474 .addReg(AArch64::XZR) 2475 .addReg(SrcReg, getKillRegState(KillSrc)); 2476 } 2477 return; 2478 } 2479 2480 // Copy a DDDD register quad by copying the individual sub-registers. 2481 if (AArch64::DDDDRegClass.contains(DestReg) && 2482 AArch64::DDDDRegClass.contains(SrcReg)) { 2483 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1, 2484 AArch64::dsub2, AArch64::dsub3}; 2485 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8, 2486 Indices); 2487 return; 2488 } 2489 2490 // Copy a DDD register triple by copying the individual sub-registers. 2491 if (AArch64::DDDRegClass.contains(DestReg) && 2492 AArch64::DDDRegClass.contains(SrcReg)) { 2493 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1, 2494 AArch64::dsub2}; 2495 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8, 2496 Indices); 2497 return; 2498 } 2499 2500 // Copy a DD register pair by copying the individual sub-registers. 2501 if (AArch64::DDRegClass.contains(DestReg) && 2502 AArch64::DDRegClass.contains(SrcReg)) { 2503 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1}; 2504 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8, 2505 Indices); 2506 return; 2507 } 2508 2509 // Copy a QQQQ register quad by copying the individual sub-registers. 2510 if (AArch64::QQQQRegClass.contains(DestReg) && 2511 AArch64::QQQQRegClass.contains(SrcReg)) { 2512 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1, 2513 AArch64::qsub2, AArch64::qsub3}; 2514 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8, 2515 Indices); 2516 return; 2517 } 2518 2519 // Copy a QQQ register triple by copying the individual sub-registers. 2520 if (AArch64::QQQRegClass.contains(DestReg) && 2521 AArch64::QQQRegClass.contains(SrcReg)) { 2522 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1, 2523 AArch64::qsub2}; 2524 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8, 2525 Indices); 2526 return; 2527 } 2528 2529 // Copy a QQ register pair by copying the individual sub-registers. 2530 if (AArch64::QQRegClass.contains(DestReg) && 2531 AArch64::QQRegClass.contains(SrcReg)) { 2532 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1}; 2533 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8, 2534 Indices); 2535 return; 2536 } 2537 2538 if (AArch64::XSeqPairsClassRegClass.contains(DestReg) && 2539 AArch64::XSeqPairsClassRegClass.contains(SrcReg)) { 2540 static const unsigned Indices[] = {AArch64::sube64, AArch64::subo64}; 2541 copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRXrs, 2542 AArch64::XZR, Indices); 2543 return; 2544 } 2545 2546 if (AArch64::WSeqPairsClassRegClass.contains(DestReg) && 2547 AArch64::WSeqPairsClassRegClass.contains(SrcReg)) { 2548 static const unsigned Indices[] = {AArch64::sube32, AArch64::subo32}; 2549 copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRWrs, 2550 AArch64::WZR, Indices); 2551 return; 2552 } 2553 2554 if (AArch64::FPR128RegClass.contains(DestReg) && 2555 AArch64::FPR128RegClass.contains(SrcReg)) { 2556 if (Subtarget.hasNEON()) { 2557 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2558 .addReg(SrcReg) 2559 .addReg(SrcReg, getKillRegState(KillSrc)); 2560 } else { 2561 BuildMI(MBB, I, DL, get(AArch64::STRQpre)) 2562 .addReg(AArch64::SP, RegState::Define) 2563 .addReg(SrcReg, getKillRegState(KillSrc)) 2564 .addReg(AArch64::SP) 2565 .addImm(-16); 2566 BuildMI(MBB, I, DL, get(AArch64::LDRQpre)) 2567 .addReg(AArch64::SP, RegState::Define) 2568 .addReg(DestReg, RegState::Define) 2569 .addReg(AArch64::SP) 2570 .addImm(16); 2571 } 2572 return; 2573 } 2574 2575 if (AArch64::FPR64RegClass.contains(DestReg) && 2576 AArch64::FPR64RegClass.contains(SrcReg)) { 2577 if (Subtarget.hasNEON()) { 2578 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::dsub, 2579 &AArch64::FPR128RegClass); 2580 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::dsub, 2581 &AArch64::FPR128RegClass); 2582 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2583 .addReg(SrcReg) 2584 .addReg(SrcReg, getKillRegState(KillSrc)); 2585 } else { 2586 BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestReg) 2587 .addReg(SrcReg, getKillRegState(KillSrc)); 2588 } 2589 return; 2590 } 2591 2592 if (AArch64::FPR32RegClass.contains(DestReg) && 2593 AArch64::FPR32RegClass.contains(SrcReg)) { 2594 if (Subtarget.hasNEON()) { 2595 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::ssub, 2596 &AArch64::FPR128RegClass); 2597 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::ssub, 2598 &AArch64::FPR128RegClass); 2599 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2600 .addReg(SrcReg) 2601 .addReg(SrcReg, getKillRegState(KillSrc)); 2602 } else { 2603 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg) 2604 .addReg(SrcReg, getKillRegState(KillSrc)); 2605 } 2606 return; 2607 } 2608 2609 if (AArch64::FPR16RegClass.contains(DestReg) && 2610 AArch64::FPR16RegClass.contains(SrcReg)) { 2611 if (Subtarget.hasNEON()) { 2612 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub, 2613 &AArch64::FPR128RegClass); 2614 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub, 2615 &AArch64::FPR128RegClass); 2616 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2617 .addReg(SrcReg) 2618 .addReg(SrcReg, getKillRegState(KillSrc)); 2619 } else { 2620 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub, 2621 &AArch64::FPR32RegClass); 2622 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub, 2623 &AArch64::FPR32RegClass); 2624 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg) 2625 .addReg(SrcReg, getKillRegState(KillSrc)); 2626 } 2627 return; 2628 } 2629 2630 if (AArch64::FPR8RegClass.contains(DestReg) && 2631 AArch64::FPR8RegClass.contains(SrcReg)) { 2632 if (Subtarget.hasNEON()) { 2633 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub, 2634 &AArch64::FPR128RegClass); 2635 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub, 2636 &AArch64::FPR128RegClass); 2637 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2638 .addReg(SrcReg) 2639 .addReg(SrcReg, getKillRegState(KillSrc)); 2640 } else { 2641 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub, 2642 &AArch64::FPR32RegClass); 2643 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub, 2644 &AArch64::FPR32RegClass); 2645 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg) 2646 .addReg(SrcReg, getKillRegState(KillSrc)); 2647 } 2648 return; 2649 } 2650 2651 // Copies between GPR64 and FPR64. 2652 if (AArch64::FPR64RegClass.contains(DestReg) && 2653 AArch64::GPR64RegClass.contains(SrcReg)) { 2654 BuildMI(MBB, I, DL, get(AArch64::FMOVXDr), DestReg) 2655 .addReg(SrcReg, getKillRegState(KillSrc)); 2656 return; 2657 } 2658 if (AArch64::GPR64RegClass.contains(DestReg) && 2659 AArch64::FPR64RegClass.contains(SrcReg)) { 2660 BuildMI(MBB, I, DL, get(AArch64::FMOVDXr), DestReg) 2661 .addReg(SrcReg, getKillRegState(KillSrc)); 2662 return; 2663 } 2664 // Copies between GPR32 and FPR32. 2665 if (AArch64::FPR32RegClass.contains(DestReg) && 2666 AArch64::GPR32RegClass.contains(SrcReg)) { 2667 BuildMI(MBB, I, DL, get(AArch64::FMOVWSr), DestReg) 2668 .addReg(SrcReg, getKillRegState(KillSrc)); 2669 return; 2670 } 2671 if (AArch64::GPR32RegClass.contains(DestReg) && 2672 AArch64::FPR32RegClass.contains(SrcReg)) { 2673 BuildMI(MBB, I, DL, get(AArch64::FMOVSWr), DestReg) 2674 .addReg(SrcReg, getKillRegState(KillSrc)); 2675 return; 2676 } 2677 2678 if (DestReg == AArch64::NZCV) { 2679 assert(AArch64::GPR64RegClass.contains(SrcReg) && "Invalid NZCV copy"); 2680 BuildMI(MBB, I, DL, get(AArch64::MSR)) 2681 .addImm(AArch64SysReg::NZCV) 2682 .addReg(SrcReg, getKillRegState(KillSrc)) 2683 .addReg(AArch64::NZCV, RegState::Implicit | RegState::Define); 2684 return; 2685 } 2686 2687 if (SrcReg == AArch64::NZCV) { 2688 assert(AArch64::GPR64RegClass.contains(DestReg) && "Invalid NZCV copy"); 2689 BuildMI(MBB, I, DL, get(AArch64::MRS), DestReg) 2690 .addImm(AArch64SysReg::NZCV) 2691 .addReg(AArch64::NZCV, RegState::Implicit | getKillRegState(KillSrc)); 2692 return; 2693 } 2694 2695 llvm_unreachable("unimplemented reg-to-reg copy"); 2696 } 2697 2698 static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI, 2699 MachineBasicBlock &MBB, 2700 MachineBasicBlock::iterator InsertBefore, 2701 const MCInstrDesc &MCID, 2702 unsigned SrcReg, bool IsKill, 2703 unsigned SubIdx0, unsigned SubIdx1, int FI, 2704 MachineMemOperand *MMO) { 2705 unsigned SrcReg0 = SrcReg; 2706 unsigned SrcReg1 = SrcReg; 2707 if (TargetRegisterInfo::isPhysicalRegister(SrcReg)) { 2708 SrcReg0 = TRI.getSubReg(SrcReg, SubIdx0); 2709 SubIdx0 = 0; 2710 SrcReg1 = TRI.getSubReg(SrcReg, SubIdx1); 2711 SubIdx1 = 0; 2712 } 2713 BuildMI(MBB, InsertBefore, DebugLoc(), MCID) 2714 .addReg(SrcReg0, getKillRegState(IsKill), SubIdx0) 2715 .addReg(SrcReg1, getKillRegState(IsKill), SubIdx1) 2716 .addFrameIndex(FI) 2717 .addImm(0) 2718 .addMemOperand(MMO); 2719 } 2720 2721 void AArch64InstrInfo::storeRegToStackSlot( 2722 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned SrcReg, 2723 bool isKill, int FI, const TargetRegisterClass *RC, 2724 const TargetRegisterInfo *TRI) const { 2725 MachineFunction &MF = *MBB.getParent(); 2726 MachineFrameInfo &MFI = MF.getFrameInfo(); 2727 unsigned Align = MFI.getObjectAlignment(FI); 2728 2729 MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI); 2730 MachineMemOperand *MMO = MF.getMachineMemOperand( 2731 PtrInfo, MachineMemOperand::MOStore, MFI.getObjectSize(FI), Align); 2732 unsigned Opc = 0; 2733 bool Offset = true; 2734 switch (TRI->getSpillSize(*RC)) { 2735 case 1: 2736 if (AArch64::FPR8RegClass.hasSubClassEq(RC)) 2737 Opc = AArch64::STRBui; 2738 break; 2739 case 2: 2740 if (AArch64::FPR16RegClass.hasSubClassEq(RC)) 2741 Opc = AArch64::STRHui; 2742 break; 2743 case 4: 2744 if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) { 2745 Opc = AArch64::STRWui; 2746 if (TargetRegisterInfo::isVirtualRegister(SrcReg)) 2747 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR32RegClass); 2748 else 2749 assert(SrcReg != AArch64::WSP); 2750 } else if (AArch64::FPR32RegClass.hasSubClassEq(RC)) 2751 Opc = AArch64::STRSui; 2752 break; 2753 case 8: 2754 if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) { 2755 Opc = AArch64::STRXui; 2756 if (TargetRegisterInfo::isVirtualRegister(SrcReg)) 2757 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass); 2758 else 2759 assert(SrcReg != AArch64::SP); 2760 } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) { 2761 Opc = AArch64::STRDui; 2762 } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) { 2763 storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI, 2764 get(AArch64::STPWi), SrcReg, isKill, 2765 AArch64::sube32, AArch64::subo32, FI, MMO); 2766 return; 2767 } 2768 break; 2769 case 16: 2770 if (AArch64::FPR128RegClass.hasSubClassEq(RC)) 2771 Opc = AArch64::STRQui; 2772 else if (AArch64::DDRegClass.hasSubClassEq(RC)) { 2773 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2774 Opc = AArch64::ST1Twov1d; 2775 Offset = false; 2776 } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) { 2777 storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI, 2778 get(AArch64::STPXi), SrcReg, isKill, 2779 AArch64::sube64, AArch64::subo64, FI, MMO); 2780 return; 2781 } 2782 break; 2783 case 24: 2784 if (AArch64::DDDRegClass.hasSubClassEq(RC)) { 2785 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2786 Opc = AArch64::ST1Threev1d; 2787 Offset = false; 2788 } 2789 break; 2790 case 32: 2791 if (AArch64::DDDDRegClass.hasSubClassEq(RC)) { 2792 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2793 Opc = AArch64::ST1Fourv1d; 2794 Offset = false; 2795 } else if (AArch64::QQRegClass.hasSubClassEq(RC)) { 2796 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2797 Opc = AArch64::ST1Twov2d; 2798 Offset = false; 2799 } 2800 break; 2801 case 48: 2802 if (AArch64::QQQRegClass.hasSubClassEq(RC)) { 2803 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2804 Opc = AArch64::ST1Threev2d; 2805 Offset = false; 2806 } 2807 break; 2808 case 64: 2809 if (AArch64::QQQQRegClass.hasSubClassEq(RC)) { 2810 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2811 Opc = AArch64::ST1Fourv2d; 2812 Offset = false; 2813 } 2814 break; 2815 } 2816 assert(Opc && "Unknown register class"); 2817 2818 const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc)) 2819 .addReg(SrcReg, getKillRegState(isKill)) 2820 .addFrameIndex(FI); 2821 2822 if (Offset) 2823 MI.addImm(0); 2824 MI.addMemOperand(MMO); 2825 } 2826 2827 static void loadRegPairFromStackSlot(const TargetRegisterInfo &TRI, 2828 MachineBasicBlock &MBB, 2829 MachineBasicBlock::iterator InsertBefore, 2830 const MCInstrDesc &MCID, 2831 unsigned DestReg, unsigned SubIdx0, 2832 unsigned SubIdx1, int FI, 2833 MachineMemOperand *MMO) { 2834 unsigned DestReg0 = DestReg; 2835 unsigned DestReg1 = DestReg; 2836 bool IsUndef = true; 2837 if (TargetRegisterInfo::isPhysicalRegister(DestReg)) { 2838 DestReg0 = TRI.getSubReg(DestReg, SubIdx0); 2839 SubIdx0 = 0; 2840 DestReg1 = TRI.getSubReg(DestReg, SubIdx1); 2841 SubIdx1 = 0; 2842 IsUndef = false; 2843 } 2844 BuildMI(MBB, InsertBefore, DebugLoc(), MCID) 2845 .addReg(DestReg0, RegState::Define | getUndefRegState(IsUndef), SubIdx0) 2846 .addReg(DestReg1, RegState::Define | getUndefRegState(IsUndef), SubIdx1) 2847 .addFrameIndex(FI) 2848 .addImm(0) 2849 .addMemOperand(MMO); 2850 } 2851 2852 void AArch64InstrInfo::loadRegFromStackSlot( 2853 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned DestReg, 2854 int FI, const TargetRegisterClass *RC, 2855 const TargetRegisterInfo *TRI) const { 2856 MachineFunction &MF = *MBB.getParent(); 2857 MachineFrameInfo &MFI = MF.getFrameInfo(); 2858 unsigned Align = MFI.getObjectAlignment(FI); 2859 MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI); 2860 MachineMemOperand *MMO = MF.getMachineMemOperand( 2861 PtrInfo, MachineMemOperand::MOLoad, MFI.getObjectSize(FI), Align); 2862 2863 unsigned Opc = 0; 2864 bool Offset = true; 2865 switch (TRI->getSpillSize(*RC)) { 2866 case 1: 2867 if (AArch64::FPR8RegClass.hasSubClassEq(RC)) 2868 Opc = AArch64::LDRBui; 2869 break; 2870 case 2: 2871 if (AArch64::FPR16RegClass.hasSubClassEq(RC)) 2872 Opc = AArch64::LDRHui; 2873 break; 2874 case 4: 2875 if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) { 2876 Opc = AArch64::LDRWui; 2877 if (TargetRegisterInfo::isVirtualRegister(DestReg)) 2878 MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR32RegClass); 2879 else 2880 assert(DestReg != AArch64::WSP); 2881 } else if (AArch64::FPR32RegClass.hasSubClassEq(RC)) 2882 Opc = AArch64::LDRSui; 2883 break; 2884 case 8: 2885 if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) { 2886 Opc = AArch64::LDRXui; 2887 if (TargetRegisterInfo::isVirtualRegister(DestReg)) 2888 MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR64RegClass); 2889 else 2890 assert(DestReg != AArch64::SP); 2891 } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) { 2892 Opc = AArch64::LDRDui; 2893 } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) { 2894 loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI, 2895 get(AArch64::LDPWi), DestReg, AArch64::sube32, 2896 AArch64::subo32, FI, MMO); 2897 return; 2898 } 2899 break; 2900 case 16: 2901 if (AArch64::FPR128RegClass.hasSubClassEq(RC)) 2902 Opc = AArch64::LDRQui; 2903 else if (AArch64::DDRegClass.hasSubClassEq(RC)) { 2904 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2905 Opc = AArch64::LD1Twov1d; 2906 Offset = false; 2907 } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) { 2908 loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI, 2909 get(AArch64::LDPXi), DestReg, AArch64::sube64, 2910 AArch64::subo64, FI, MMO); 2911 return; 2912 } 2913 break; 2914 case 24: 2915 if (AArch64::DDDRegClass.hasSubClassEq(RC)) { 2916 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2917 Opc = AArch64::LD1Threev1d; 2918 Offset = false; 2919 } 2920 break; 2921 case 32: 2922 if (AArch64::DDDDRegClass.hasSubClassEq(RC)) { 2923 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2924 Opc = AArch64::LD1Fourv1d; 2925 Offset = false; 2926 } else if (AArch64::QQRegClass.hasSubClassEq(RC)) { 2927 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2928 Opc = AArch64::LD1Twov2d; 2929 Offset = false; 2930 } 2931 break; 2932 case 48: 2933 if (AArch64::QQQRegClass.hasSubClassEq(RC)) { 2934 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2935 Opc = AArch64::LD1Threev2d; 2936 Offset = false; 2937 } 2938 break; 2939 case 64: 2940 if (AArch64::QQQQRegClass.hasSubClassEq(RC)) { 2941 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2942 Opc = AArch64::LD1Fourv2d; 2943 Offset = false; 2944 } 2945 break; 2946 } 2947 assert(Opc && "Unknown register class"); 2948 2949 const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc)) 2950 .addReg(DestReg, getDefRegState(true)) 2951 .addFrameIndex(FI); 2952 if (Offset) 2953 MI.addImm(0); 2954 MI.addMemOperand(MMO); 2955 } 2956 2957 void llvm::emitFrameOffset(MachineBasicBlock &MBB, 2958 MachineBasicBlock::iterator MBBI, const DebugLoc &DL, 2959 unsigned DestReg, unsigned SrcReg, int Offset, 2960 const TargetInstrInfo *TII, 2961 MachineInstr::MIFlag Flag, bool SetNZCV, 2962 bool NeedsWinCFI) { 2963 if (DestReg == SrcReg && Offset == 0) 2964 return; 2965 2966 assert((DestReg != AArch64::SP || Offset % 16 == 0) && 2967 "SP increment/decrement not 16-byte aligned"); 2968 2969 bool isSub = Offset < 0; 2970 if (isSub) 2971 Offset = -Offset; 2972 2973 // FIXME: If the offset won't fit in 24-bits, compute the offset into a 2974 // scratch register. If DestReg is a virtual register, use it as the 2975 // scratch register; otherwise, create a new virtual register (to be 2976 // replaced by the scavenger at the end of PEI). That case can be optimized 2977 // slightly if DestReg is SP which is always 16-byte aligned, so the scratch 2978 // register can be loaded with offset%8 and the add/sub can use an extending 2979 // instruction with LSL#3. 2980 // Currently the function handles any offsets but generates a poor sequence 2981 // of code. 2982 // assert(Offset < (1 << 24) && "unimplemented reg plus immediate"); 2983 2984 unsigned Opc; 2985 if (SetNZCV) 2986 Opc = isSub ? AArch64::SUBSXri : AArch64::ADDSXri; 2987 else 2988 Opc = isSub ? AArch64::SUBXri : AArch64::ADDXri; 2989 const unsigned MaxEncoding = 0xfff; 2990 const unsigned ShiftSize = 12; 2991 const unsigned MaxEncodableValue = MaxEncoding << ShiftSize; 2992 while (((unsigned)Offset) >= (1 << ShiftSize)) { 2993 unsigned ThisVal; 2994 if (((unsigned)Offset) > MaxEncodableValue) { 2995 ThisVal = MaxEncodableValue; 2996 } else { 2997 ThisVal = Offset & MaxEncodableValue; 2998 } 2999 assert((ThisVal >> ShiftSize) <= MaxEncoding && 3000 "Encoding cannot handle value that big"); 3001 BuildMI(MBB, MBBI, DL, TII->get(Opc), DestReg) 3002 .addReg(SrcReg) 3003 .addImm(ThisVal >> ShiftSize) 3004 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, ShiftSize)) 3005 .setMIFlag(Flag); 3006 3007 if (NeedsWinCFI && SrcReg == AArch64::SP && DestReg == AArch64::SP) 3008 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_StackAlloc)) 3009 .addImm(ThisVal) 3010 .setMIFlag(Flag); 3011 3012 SrcReg = DestReg; 3013 Offset -= ThisVal; 3014 if (Offset == 0) 3015 return; 3016 } 3017 BuildMI(MBB, MBBI, DL, TII->get(Opc), DestReg) 3018 .addReg(SrcReg) 3019 .addImm(Offset) 3020 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)) 3021 .setMIFlag(Flag); 3022 3023 if (NeedsWinCFI) { 3024 if ((DestReg == AArch64::FP && SrcReg == AArch64::SP) || 3025 (SrcReg == AArch64::FP && DestReg == AArch64::SP)) { 3026 if (Offset == 0) 3027 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_SetFP)). 3028 setMIFlag(Flag); 3029 else 3030 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_AddFP)). 3031 addImm(Offset).setMIFlag(Flag); 3032 } else if (DestReg == AArch64::SP) { 3033 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_StackAlloc)). 3034 addImm(Offset).setMIFlag(Flag); 3035 } 3036 } 3037 } 3038 3039 MachineInstr *AArch64InstrInfo::foldMemoryOperandImpl( 3040 MachineFunction &MF, MachineInstr &MI, ArrayRef<unsigned> Ops, 3041 MachineBasicBlock::iterator InsertPt, int FrameIndex, 3042 LiveIntervals *LIS) const { 3043 // This is a bit of a hack. Consider this instruction: 3044 // 3045 // %0 = COPY %sp; GPR64all:%0 3046 // 3047 // We explicitly chose GPR64all for the virtual register so such a copy might 3048 // be eliminated by RegisterCoalescer. However, that may not be possible, and 3049 // %0 may even spill. We can't spill %sp, and since it is in the GPR64all 3050 // register class, TargetInstrInfo::foldMemoryOperand() is going to try. 3051 // 3052 // To prevent that, we are going to constrain the %0 register class here. 3053 // 3054 // <rdar://problem/11522048> 3055 // 3056 if (MI.isFullCopy()) { 3057 unsigned DstReg = MI.getOperand(0).getReg(); 3058 unsigned SrcReg = MI.getOperand(1).getReg(); 3059 if (SrcReg == AArch64::SP && 3060 TargetRegisterInfo::isVirtualRegister(DstReg)) { 3061 MF.getRegInfo().constrainRegClass(DstReg, &AArch64::GPR64RegClass); 3062 return nullptr; 3063 } 3064 if (DstReg == AArch64::SP && 3065 TargetRegisterInfo::isVirtualRegister(SrcReg)) { 3066 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass); 3067 return nullptr; 3068 } 3069 } 3070 3071 // Handle the case where a copy is being spilled or filled but the source 3072 // and destination register class don't match. For example: 3073 // 3074 // %0 = COPY %xzr; GPR64common:%0 3075 // 3076 // In this case we can still safely fold away the COPY and generate the 3077 // following spill code: 3078 // 3079 // STRXui %xzr, %stack.0 3080 // 3081 // This also eliminates spilled cross register class COPYs (e.g. between x and 3082 // d regs) of the same size. For example: 3083 // 3084 // %0 = COPY %1; GPR64:%0, FPR64:%1 3085 // 3086 // will be filled as 3087 // 3088 // LDRDui %0, fi<#0> 3089 // 3090 // instead of 3091 // 3092 // LDRXui %Temp, fi<#0> 3093 // %0 = FMOV %Temp 3094 // 3095 if (MI.isCopy() && Ops.size() == 1 && 3096 // Make sure we're only folding the explicit COPY defs/uses. 3097 (Ops[0] == 0 || Ops[0] == 1)) { 3098 bool IsSpill = Ops[0] == 0; 3099 bool IsFill = !IsSpill; 3100 const TargetRegisterInfo &TRI = *MF.getSubtarget().getRegisterInfo(); 3101 const MachineRegisterInfo &MRI = MF.getRegInfo(); 3102 MachineBasicBlock &MBB = *MI.getParent(); 3103 const MachineOperand &DstMO = MI.getOperand(0); 3104 const MachineOperand &SrcMO = MI.getOperand(1); 3105 unsigned DstReg = DstMO.getReg(); 3106 unsigned SrcReg = SrcMO.getReg(); 3107 // This is slightly expensive to compute for physical regs since 3108 // getMinimalPhysRegClass is slow. 3109 auto getRegClass = [&](unsigned Reg) { 3110 return TargetRegisterInfo::isVirtualRegister(Reg) 3111 ? MRI.getRegClass(Reg) 3112 : TRI.getMinimalPhysRegClass(Reg); 3113 }; 3114 3115 if (DstMO.getSubReg() == 0 && SrcMO.getSubReg() == 0) { 3116 assert(TRI.getRegSizeInBits(*getRegClass(DstReg)) == 3117 TRI.getRegSizeInBits(*getRegClass(SrcReg)) && 3118 "Mismatched register size in non subreg COPY"); 3119 if (IsSpill) 3120 storeRegToStackSlot(MBB, InsertPt, SrcReg, SrcMO.isKill(), FrameIndex, 3121 getRegClass(SrcReg), &TRI); 3122 else 3123 loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex, 3124 getRegClass(DstReg), &TRI); 3125 return &*--InsertPt; 3126 } 3127 3128 // Handle cases like spilling def of: 3129 // 3130 // %0:sub_32<def,read-undef> = COPY %wzr; GPR64common:%0 3131 // 3132 // where the physical register source can be widened and stored to the full 3133 // virtual reg destination stack slot, in this case producing: 3134 // 3135 // STRXui %xzr, %stack.0 3136 // 3137 if (IsSpill && DstMO.isUndef() && 3138 TargetRegisterInfo::isPhysicalRegister(SrcReg)) { 3139 assert(SrcMO.getSubReg() == 0 && 3140 "Unexpected subreg on physical register"); 3141 const TargetRegisterClass *SpillRC; 3142 unsigned SpillSubreg; 3143 switch (DstMO.getSubReg()) { 3144 default: 3145 SpillRC = nullptr; 3146 break; 3147 case AArch64::sub_32: 3148 case AArch64::ssub: 3149 if (AArch64::GPR32RegClass.contains(SrcReg)) { 3150 SpillRC = &AArch64::GPR64RegClass; 3151 SpillSubreg = AArch64::sub_32; 3152 } else if (AArch64::FPR32RegClass.contains(SrcReg)) { 3153 SpillRC = &AArch64::FPR64RegClass; 3154 SpillSubreg = AArch64::ssub; 3155 } else 3156 SpillRC = nullptr; 3157 break; 3158 case AArch64::dsub: 3159 if (AArch64::FPR64RegClass.contains(SrcReg)) { 3160 SpillRC = &AArch64::FPR128RegClass; 3161 SpillSubreg = AArch64::dsub; 3162 } else 3163 SpillRC = nullptr; 3164 break; 3165 } 3166 3167 if (SpillRC) 3168 if (unsigned WidenedSrcReg = 3169 TRI.getMatchingSuperReg(SrcReg, SpillSubreg, SpillRC)) { 3170 storeRegToStackSlot(MBB, InsertPt, WidenedSrcReg, SrcMO.isKill(), 3171 FrameIndex, SpillRC, &TRI); 3172 return &*--InsertPt; 3173 } 3174 } 3175 3176 // Handle cases like filling use of: 3177 // 3178 // %0:sub_32<def,read-undef> = COPY %1; GPR64:%0, GPR32:%1 3179 // 3180 // where we can load the full virtual reg source stack slot, into the subreg 3181 // destination, in this case producing: 3182 // 3183 // LDRWui %0:sub_32<def,read-undef>, %stack.0 3184 // 3185 if (IsFill && SrcMO.getSubReg() == 0 && DstMO.isUndef()) { 3186 const TargetRegisterClass *FillRC; 3187 switch (DstMO.getSubReg()) { 3188 default: 3189 FillRC = nullptr; 3190 break; 3191 case AArch64::sub_32: 3192 FillRC = &AArch64::GPR32RegClass; 3193 break; 3194 case AArch64::ssub: 3195 FillRC = &AArch64::FPR32RegClass; 3196 break; 3197 case AArch64::dsub: 3198 FillRC = &AArch64::FPR64RegClass; 3199 break; 3200 } 3201 3202 if (FillRC) { 3203 assert(TRI.getRegSizeInBits(*getRegClass(SrcReg)) == 3204 TRI.getRegSizeInBits(*FillRC) && 3205 "Mismatched regclass size on folded subreg COPY"); 3206 loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex, FillRC, &TRI); 3207 MachineInstr &LoadMI = *--InsertPt; 3208 MachineOperand &LoadDst = LoadMI.getOperand(0); 3209 assert(LoadDst.getSubReg() == 0 && "unexpected subreg on fill load"); 3210 LoadDst.setSubReg(DstMO.getSubReg()); 3211 LoadDst.setIsUndef(); 3212 return &LoadMI; 3213 } 3214 } 3215 } 3216 3217 // Cannot fold. 3218 return nullptr; 3219 } 3220 3221 int llvm::isAArch64FrameOffsetLegal(const MachineInstr &MI, int &Offset, 3222 bool *OutUseUnscaledOp, 3223 unsigned *OutUnscaledOp, 3224 int *EmittableOffset) { 3225 // Set output values in case of early exit. 3226 if (EmittableOffset) 3227 *EmittableOffset = 0; 3228 if (OutUseUnscaledOp) 3229 *OutUseUnscaledOp = false; 3230 if (OutUnscaledOp) 3231 *OutUnscaledOp = 0; 3232 3233 // Exit early for structured vector spills/fills as they can't take an 3234 // immediate offset. 3235 switch (MI.getOpcode()) { 3236 default: 3237 break; 3238 case AArch64::LD1Twov2d: 3239 case AArch64::LD1Threev2d: 3240 case AArch64::LD1Fourv2d: 3241 case AArch64::LD1Twov1d: 3242 case AArch64::LD1Threev1d: 3243 case AArch64::LD1Fourv1d: 3244 case AArch64::ST1Twov2d: 3245 case AArch64::ST1Threev2d: 3246 case AArch64::ST1Fourv2d: 3247 case AArch64::ST1Twov1d: 3248 case AArch64::ST1Threev1d: 3249 case AArch64::ST1Fourv1d: 3250 return AArch64FrameOffsetCannotUpdate; 3251 } 3252 3253 // Get the min/max offset and the scale. 3254 unsigned Scale, Width; 3255 int64_t MinOff, MaxOff; 3256 if (!AArch64InstrInfo::getMemOpInfo(MI.getOpcode(), Scale, Width, MinOff, 3257 MaxOff)) 3258 llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal"); 3259 3260 // Construct the complete offset. 3261 const MachineOperand &ImmOpnd = 3262 MI.getOperand(AArch64InstrInfo::getLoadStoreImmIdx(MI.getOpcode())); 3263 Offset += ImmOpnd.getImm() * Scale; 3264 3265 // If the offset doesn't match the scale, we rewrite the instruction to 3266 // use the unscaled instruction instead. Likewise, if we have a negative 3267 // offset and there is an unscaled op to use. 3268 Optional<unsigned> UnscaledOp = 3269 AArch64InstrInfo::getUnscaledLdSt(MI.getOpcode()); 3270 bool useUnscaledOp = UnscaledOp && (Offset % Scale || Offset < 0); 3271 if (useUnscaledOp && 3272 !AArch64InstrInfo::getMemOpInfo(*UnscaledOp, Scale, Width, MinOff, MaxOff)) 3273 llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal"); 3274 3275 int64_t Remainder = Offset % Scale; 3276 assert(!(Remainder && useUnscaledOp) && 3277 "Cannot have remainder when using unscaled op"); 3278 3279 assert(MinOff < MaxOff && "Unexpected Min/Max offsets"); 3280 int NewOffset = Offset / Scale; 3281 if (MinOff <= NewOffset && NewOffset <= MaxOff) 3282 Offset = Remainder; 3283 else { 3284 NewOffset = NewOffset < 0 ? MinOff : MaxOff; 3285 Offset = Offset - NewOffset * Scale + Remainder; 3286 } 3287 3288 if (EmittableOffset) 3289 *EmittableOffset = NewOffset; 3290 if (OutUseUnscaledOp) 3291 *OutUseUnscaledOp = useUnscaledOp; 3292 if (OutUnscaledOp && UnscaledOp) 3293 *OutUnscaledOp = *UnscaledOp; 3294 3295 return AArch64FrameOffsetCanUpdate | 3296 (Offset == 0 ? AArch64FrameOffsetIsLegal : 0); 3297 } 3298 3299 bool llvm::rewriteAArch64FrameIndex(MachineInstr &MI, unsigned FrameRegIdx, 3300 unsigned FrameReg, int &Offset, 3301 const AArch64InstrInfo *TII) { 3302 unsigned Opcode = MI.getOpcode(); 3303 unsigned ImmIdx = FrameRegIdx + 1; 3304 3305 if (Opcode == AArch64::ADDSXri || Opcode == AArch64::ADDXri) { 3306 Offset += MI.getOperand(ImmIdx).getImm(); 3307 emitFrameOffset(*MI.getParent(), MI, MI.getDebugLoc(), 3308 MI.getOperand(0).getReg(), FrameReg, Offset, TII, 3309 MachineInstr::NoFlags, (Opcode == AArch64::ADDSXri)); 3310 MI.eraseFromParent(); 3311 Offset = 0; 3312 return true; 3313 } 3314 3315 int NewOffset; 3316 unsigned UnscaledOp; 3317 bool UseUnscaledOp; 3318 int Status = isAArch64FrameOffsetLegal(MI, Offset, &UseUnscaledOp, 3319 &UnscaledOp, &NewOffset); 3320 if (Status & AArch64FrameOffsetCanUpdate) { 3321 if (Status & AArch64FrameOffsetIsLegal) 3322 // Replace the FrameIndex with FrameReg. 3323 MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false); 3324 if (UseUnscaledOp) 3325 MI.setDesc(TII->get(UnscaledOp)); 3326 3327 MI.getOperand(ImmIdx).ChangeToImmediate(NewOffset); 3328 return Offset == 0; 3329 } 3330 3331 return false; 3332 } 3333 3334 void AArch64InstrInfo::getNoop(MCInst &NopInst) const { 3335 NopInst.setOpcode(AArch64::HINT); 3336 NopInst.addOperand(MCOperand::createImm(0)); 3337 } 3338 3339 // AArch64 supports MachineCombiner. 3340 bool AArch64InstrInfo::useMachineCombiner() const { return true; } 3341 3342 // True when Opc sets flag 3343 static bool isCombineInstrSettingFlag(unsigned Opc) { 3344 switch (Opc) { 3345 case AArch64::ADDSWrr: 3346 case AArch64::ADDSWri: 3347 case AArch64::ADDSXrr: 3348 case AArch64::ADDSXri: 3349 case AArch64::SUBSWrr: 3350 case AArch64::SUBSXrr: 3351 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi. 3352 case AArch64::SUBSWri: 3353 case AArch64::SUBSXri: 3354 return true; 3355 default: 3356 break; 3357 } 3358 return false; 3359 } 3360 3361 // 32b Opcodes that can be combined with a MUL 3362 static bool isCombineInstrCandidate32(unsigned Opc) { 3363 switch (Opc) { 3364 case AArch64::ADDWrr: 3365 case AArch64::ADDWri: 3366 case AArch64::SUBWrr: 3367 case AArch64::ADDSWrr: 3368 case AArch64::ADDSWri: 3369 case AArch64::SUBSWrr: 3370 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi. 3371 case AArch64::SUBWri: 3372 case AArch64::SUBSWri: 3373 return true; 3374 default: 3375 break; 3376 } 3377 return false; 3378 } 3379 3380 // 64b Opcodes that can be combined with a MUL 3381 static bool isCombineInstrCandidate64(unsigned Opc) { 3382 switch (Opc) { 3383 case AArch64::ADDXrr: 3384 case AArch64::ADDXri: 3385 case AArch64::SUBXrr: 3386 case AArch64::ADDSXrr: 3387 case AArch64::ADDSXri: 3388 case AArch64::SUBSXrr: 3389 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi. 3390 case AArch64::SUBXri: 3391 case AArch64::SUBSXri: 3392 return true; 3393 default: 3394 break; 3395 } 3396 return false; 3397 } 3398 3399 // FP Opcodes that can be combined with a FMUL 3400 static bool isCombineInstrCandidateFP(const MachineInstr &Inst) { 3401 switch (Inst.getOpcode()) { 3402 default: 3403 break; 3404 case AArch64::FADDSrr: 3405 case AArch64::FADDDrr: 3406 case AArch64::FADDv2f32: 3407 case AArch64::FADDv2f64: 3408 case AArch64::FADDv4f32: 3409 case AArch64::FSUBSrr: 3410 case AArch64::FSUBDrr: 3411 case AArch64::FSUBv2f32: 3412 case AArch64::FSUBv2f64: 3413 case AArch64::FSUBv4f32: 3414 TargetOptions Options = Inst.getParent()->getParent()->getTarget().Options; 3415 return (Options.UnsafeFPMath || 3416 Options.AllowFPOpFusion == FPOpFusion::Fast); 3417 } 3418 return false; 3419 } 3420 3421 // Opcodes that can be combined with a MUL 3422 static bool isCombineInstrCandidate(unsigned Opc) { 3423 return (isCombineInstrCandidate32(Opc) || isCombineInstrCandidate64(Opc)); 3424 } 3425 3426 // 3427 // Utility routine that checks if \param MO is defined by an 3428 // \param CombineOpc instruction in the basic block \param MBB 3429 static bool canCombine(MachineBasicBlock &MBB, MachineOperand &MO, 3430 unsigned CombineOpc, unsigned ZeroReg = 0, 3431 bool CheckZeroReg = false) { 3432 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 3433 MachineInstr *MI = nullptr; 3434 3435 if (MO.isReg() && TargetRegisterInfo::isVirtualRegister(MO.getReg())) 3436 MI = MRI.getUniqueVRegDef(MO.getReg()); 3437 // And it needs to be in the trace (otherwise, it won't have a depth). 3438 if (!MI || MI->getParent() != &MBB || (unsigned)MI->getOpcode() != CombineOpc) 3439 return false; 3440 // Must only used by the user we combine with. 3441 if (!MRI.hasOneNonDBGUse(MI->getOperand(0).getReg())) 3442 return false; 3443 3444 if (CheckZeroReg) { 3445 assert(MI->getNumOperands() >= 4 && MI->getOperand(0).isReg() && 3446 MI->getOperand(1).isReg() && MI->getOperand(2).isReg() && 3447 MI->getOperand(3).isReg() && "MAdd/MSub must have a least 4 regs"); 3448 // The third input reg must be zero. 3449 if (MI->getOperand(3).getReg() != ZeroReg) 3450 return false; 3451 } 3452 3453 return true; 3454 } 3455 3456 // 3457 // Is \param MO defined by an integer multiply and can be combined? 3458 static bool canCombineWithMUL(MachineBasicBlock &MBB, MachineOperand &MO, 3459 unsigned MulOpc, unsigned ZeroReg) { 3460 return canCombine(MBB, MO, MulOpc, ZeroReg, true); 3461 } 3462 3463 // 3464 // Is \param MO defined by a floating-point multiply and can be combined? 3465 static bool canCombineWithFMUL(MachineBasicBlock &MBB, MachineOperand &MO, 3466 unsigned MulOpc) { 3467 return canCombine(MBB, MO, MulOpc); 3468 } 3469 3470 // TODO: There are many more machine instruction opcodes to match: 3471 // 1. Other data types (integer, vectors) 3472 // 2. Other math / logic operations (xor, or) 3473 // 3. Other forms of the same operation (intrinsics and other variants) 3474 bool AArch64InstrInfo::isAssociativeAndCommutative( 3475 const MachineInstr &Inst) const { 3476 switch (Inst.getOpcode()) { 3477 case AArch64::FADDDrr: 3478 case AArch64::FADDSrr: 3479 case AArch64::FADDv2f32: 3480 case AArch64::FADDv2f64: 3481 case AArch64::FADDv4f32: 3482 case AArch64::FMULDrr: 3483 case AArch64::FMULSrr: 3484 case AArch64::FMULX32: 3485 case AArch64::FMULX64: 3486 case AArch64::FMULXv2f32: 3487 case AArch64::FMULXv2f64: 3488 case AArch64::FMULXv4f32: 3489 case AArch64::FMULv2f32: 3490 case AArch64::FMULv2f64: 3491 case AArch64::FMULv4f32: 3492 return Inst.getParent()->getParent()->getTarget().Options.UnsafeFPMath; 3493 default: 3494 return false; 3495 } 3496 } 3497 3498 /// Find instructions that can be turned into madd. 3499 static bool getMaddPatterns(MachineInstr &Root, 3500 SmallVectorImpl<MachineCombinerPattern> &Patterns) { 3501 unsigned Opc = Root.getOpcode(); 3502 MachineBasicBlock &MBB = *Root.getParent(); 3503 bool Found = false; 3504 3505 if (!isCombineInstrCandidate(Opc)) 3506 return false; 3507 if (isCombineInstrSettingFlag(Opc)) { 3508 int Cmp_NZCV = Root.findRegisterDefOperandIdx(AArch64::NZCV, true); 3509 // When NZCV is live bail out. 3510 if (Cmp_NZCV == -1) 3511 return false; 3512 unsigned NewOpc = convertToNonFlagSettingOpc(Root); 3513 // When opcode can't change bail out. 3514 // CHECKME: do we miss any cases for opcode conversion? 3515 if (NewOpc == Opc) 3516 return false; 3517 Opc = NewOpc; 3518 } 3519 3520 switch (Opc) { 3521 default: 3522 break; 3523 case AArch64::ADDWrr: 3524 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() && 3525 "ADDWrr does not have register operands"); 3526 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr, 3527 AArch64::WZR)) { 3528 Patterns.push_back(MachineCombinerPattern::MULADDW_OP1); 3529 Found = true; 3530 } 3531 if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDWrrr, 3532 AArch64::WZR)) { 3533 Patterns.push_back(MachineCombinerPattern::MULADDW_OP2); 3534 Found = true; 3535 } 3536 break; 3537 case AArch64::ADDXrr: 3538 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr, 3539 AArch64::XZR)) { 3540 Patterns.push_back(MachineCombinerPattern::MULADDX_OP1); 3541 Found = true; 3542 } 3543 if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDXrrr, 3544 AArch64::XZR)) { 3545 Patterns.push_back(MachineCombinerPattern::MULADDX_OP2); 3546 Found = true; 3547 } 3548 break; 3549 case AArch64::SUBWrr: 3550 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr, 3551 AArch64::WZR)) { 3552 Patterns.push_back(MachineCombinerPattern::MULSUBW_OP1); 3553 Found = true; 3554 } 3555 if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDWrrr, 3556 AArch64::WZR)) { 3557 Patterns.push_back(MachineCombinerPattern::MULSUBW_OP2); 3558 Found = true; 3559 } 3560 break; 3561 case AArch64::SUBXrr: 3562 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr, 3563 AArch64::XZR)) { 3564 Patterns.push_back(MachineCombinerPattern::MULSUBX_OP1); 3565 Found = true; 3566 } 3567 if (canCombineWithMUL(MBB, Root.getOperand(2), AArch64::MADDXrrr, 3568 AArch64::XZR)) { 3569 Patterns.push_back(MachineCombinerPattern::MULSUBX_OP2); 3570 Found = true; 3571 } 3572 break; 3573 case AArch64::ADDWri: 3574 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr, 3575 AArch64::WZR)) { 3576 Patterns.push_back(MachineCombinerPattern::MULADDWI_OP1); 3577 Found = true; 3578 } 3579 break; 3580 case AArch64::ADDXri: 3581 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr, 3582 AArch64::XZR)) { 3583 Patterns.push_back(MachineCombinerPattern::MULADDXI_OP1); 3584 Found = true; 3585 } 3586 break; 3587 case AArch64::SUBWri: 3588 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDWrrr, 3589 AArch64::WZR)) { 3590 Patterns.push_back(MachineCombinerPattern::MULSUBWI_OP1); 3591 Found = true; 3592 } 3593 break; 3594 case AArch64::SUBXri: 3595 if (canCombineWithMUL(MBB, Root.getOperand(1), AArch64::MADDXrrr, 3596 AArch64::XZR)) { 3597 Patterns.push_back(MachineCombinerPattern::MULSUBXI_OP1); 3598 Found = true; 3599 } 3600 break; 3601 } 3602 return Found; 3603 } 3604 /// Floating-Point Support 3605 3606 /// Find instructions that can be turned into madd. 3607 static bool getFMAPatterns(MachineInstr &Root, 3608 SmallVectorImpl<MachineCombinerPattern> &Patterns) { 3609 3610 if (!isCombineInstrCandidateFP(Root)) 3611 return false; 3612 3613 MachineBasicBlock &MBB = *Root.getParent(); 3614 bool Found = false; 3615 3616 switch (Root.getOpcode()) { 3617 default: 3618 assert(false && "Unsupported FP instruction in combiner\n"); 3619 break; 3620 case AArch64::FADDSrr: 3621 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() && 3622 "FADDWrr does not have register operands"); 3623 if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULSrr)) { 3624 Patterns.push_back(MachineCombinerPattern::FMULADDS_OP1); 3625 Found = true; 3626 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3627 AArch64::FMULv1i32_indexed)) { 3628 Patterns.push_back(MachineCombinerPattern::FMLAv1i32_indexed_OP1); 3629 Found = true; 3630 } 3631 if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULSrr)) { 3632 Patterns.push_back(MachineCombinerPattern::FMULADDS_OP2); 3633 Found = true; 3634 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3635 AArch64::FMULv1i32_indexed)) { 3636 Patterns.push_back(MachineCombinerPattern::FMLAv1i32_indexed_OP2); 3637 Found = true; 3638 } 3639 break; 3640 case AArch64::FADDDrr: 3641 if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULDrr)) { 3642 Patterns.push_back(MachineCombinerPattern::FMULADDD_OP1); 3643 Found = true; 3644 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3645 AArch64::FMULv1i64_indexed)) { 3646 Patterns.push_back(MachineCombinerPattern::FMLAv1i64_indexed_OP1); 3647 Found = true; 3648 } 3649 if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULDrr)) { 3650 Patterns.push_back(MachineCombinerPattern::FMULADDD_OP2); 3651 Found = true; 3652 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3653 AArch64::FMULv1i64_indexed)) { 3654 Patterns.push_back(MachineCombinerPattern::FMLAv1i64_indexed_OP2); 3655 Found = true; 3656 } 3657 break; 3658 case AArch64::FADDv2f32: 3659 if (canCombineWithFMUL(MBB, Root.getOperand(1), 3660 AArch64::FMULv2i32_indexed)) { 3661 Patterns.push_back(MachineCombinerPattern::FMLAv2i32_indexed_OP1); 3662 Found = true; 3663 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3664 AArch64::FMULv2f32)) { 3665 Patterns.push_back(MachineCombinerPattern::FMLAv2f32_OP1); 3666 Found = true; 3667 } 3668 if (canCombineWithFMUL(MBB, Root.getOperand(2), 3669 AArch64::FMULv2i32_indexed)) { 3670 Patterns.push_back(MachineCombinerPattern::FMLAv2i32_indexed_OP2); 3671 Found = true; 3672 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3673 AArch64::FMULv2f32)) { 3674 Patterns.push_back(MachineCombinerPattern::FMLAv2f32_OP2); 3675 Found = true; 3676 } 3677 break; 3678 case AArch64::FADDv2f64: 3679 if (canCombineWithFMUL(MBB, Root.getOperand(1), 3680 AArch64::FMULv2i64_indexed)) { 3681 Patterns.push_back(MachineCombinerPattern::FMLAv2i64_indexed_OP1); 3682 Found = true; 3683 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3684 AArch64::FMULv2f64)) { 3685 Patterns.push_back(MachineCombinerPattern::FMLAv2f64_OP1); 3686 Found = true; 3687 } 3688 if (canCombineWithFMUL(MBB, Root.getOperand(2), 3689 AArch64::FMULv2i64_indexed)) { 3690 Patterns.push_back(MachineCombinerPattern::FMLAv2i64_indexed_OP2); 3691 Found = true; 3692 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3693 AArch64::FMULv2f64)) { 3694 Patterns.push_back(MachineCombinerPattern::FMLAv2f64_OP2); 3695 Found = true; 3696 } 3697 break; 3698 case AArch64::FADDv4f32: 3699 if (canCombineWithFMUL(MBB, Root.getOperand(1), 3700 AArch64::FMULv4i32_indexed)) { 3701 Patterns.push_back(MachineCombinerPattern::FMLAv4i32_indexed_OP1); 3702 Found = true; 3703 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3704 AArch64::FMULv4f32)) { 3705 Patterns.push_back(MachineCombinerPattern::FMLAv4f32_OP1); 3706 Found = true; 3707 } 3708 if (canCombineWithFMUL(MBB, Root.getOperand(2), 3709 AArch64::FMULv4i32_indexed)) { 3710 Patterns.push_back(MachineCombinerPattern::FMLAv4i32_indexed_OP2); 3711 Found = true; 3712 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3713 AArch64::FMULv4f32)) { 3714 Patterns.push_back(MachineCombinerPattern::FMLAv4f32_OP2); 3715 Found = true; 3716 } 3717 break; 3718 3719 case AArch64::FSUBSrr: 3720 if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULSrr)) { 3721 Patterns.push_back(MachineCombinerPattern::FMULSUBS_OP1); 3722 Found = true; 3723 } 3724 if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULSrr)) { 3725 Patterns.push_back(MachineCombinerPattern::FMULSUBS_OP2); 3726 Found = true; 3727 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3728 AArch64::FMULv1i32_indexed)) { 3729 Patterns.push_back(MachineCombinerPattern::FMLSv1i32_indexed_OP2); 3730 Found = true; 3731 } 3732 if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FNMULSrr)) { 3733 Patterns.push_back(MachineCombinerPattern::FNMULSUBS_OP1); 3734 Found = true; 3735 } 3736 break; 3737 case AArch64::FSUBDrr: 3738 if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FMULDrr)) { 3739 Patterns.push_back(MachineCombinerPattern::FMULSUBD_OP1); 3740 Found = true; 3741 } 3742 if (canCombineWithFMUL(MBB, Root.getOperand(2), AArch64::FMULDrr)) { 3743 Patterns.push_back(MachineCombinerPattern::FMULSUBD_OP2); 3744 Found = true; 3745 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3746 AArch64::FMULv1i64_indexed)) { 3747 Patterns.push_back(MachineCombinerPattern::FMLSv1i64_indexed_OP2); 3748 Found = true; 3749 } 3750 if (canCombineWithFMUL(MBB, Root.getOperand(1), AArch64::FNMULDrr)) { 3751 Patterns.push_back(MachineCombinerPattern::FNMULSUBD_OP1); 3752 Found = true; 3753 } 3754 break; 3755 case AArch64::FSUBv2f32: 3756 if (canCombineWithFMUL(MBB, Root.getOperand(2), 3757 AArch64::FMULv2i32_indexed)) { 3758 Patterns.push_back(MachineCombinerPattern::FMLSv2i32_indexed_OP2); 3759 Found = true; 3760 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3761 AArch64::FMULv2f32)) { 3762 Patterns.push_back(MachineCombinerPattern::FMLSv2f32_OP2); 3763 Found = true; 3764 } 3765 if (canCombineWithFMUL(MBB, Root.getOperand(1), 3766 AArch64::FMULv2i32_indexed)) { 3767 Patterns.push_back(MachineCombinerPattern::FMLSv2i32_indexed_OP1); 3768 Found = true; 3769 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3770 AArch64::FMULv2f32)) { 3771 Patterns.push_back(MachineCombinerPattern::FMLSv2f32_OP1); 3772 Found = true; 3773 } 3774 break; 3775 case AArch64::FSUBv2f64: 3776 if (canCombineWithFMUL(MBB, Root.getOperand(2), 3777 AArch64::FMULv2i64_indexed)) { 3778 Patterns.push_back(MachineCombinerPattern::FMLSv2i64_indexed_OP2); 3779 Found = true; 3780 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3781 AArch64::FMULv2f64)) { 3782 Patterns.push_back(MachineCombinerPattern::FMLSv2f64_OP2); 3783 Found = true; 3784 } 3785 if (canCombineWithFMUL(MBB, Root.getOperand(1), 3786 AArch64::FMULv2i64_indexed)) { 3787 Patterns.push_back(MachineCombinerPattern::FMLSv2i64_indexed_OP1); 3788 Found = true; 3789 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3790 AArch64::FMULv2f64)) { 3791 Patterns.push_back(MachineCombinerPattern::FMLSv2f64_OP1); 3792 Found = true; 3793 } 3794 break; 3795 case AArch64::FSUBv4f32: 3796 if (canCombineWithFMUL(MBB, Root.getOperand(2), 3797 AArch64::FMULv4i32_indexed)) { 3798 Patterns.push_back(MachineCombinerPattern::FMLSv4i32_indexed_OP2); 3799 Found = true; 3800 } else if (canCombineWithFMUL(MBB, Root.getOperand(2), 3801 AArch64::FMULv4f32)) { 3802 Patterns.push_back(MachineCombinerPattern::FMLSv4f32_OP2); 3803 Found = true; 3804 } 3805 if (canCombineWithFMUL(MBB, Root.getOperand(1), 3806 AArch64::FMULv4i32_indexed)) { 3807 Patterns.push_back(MachineCombinerPattern::FMLSv4i32_indexed_OP1); 3808 Found = true; 3809 } else if (canCombineWithFMUL(MBB, Root.getOperand(1), 3810 AArch64::FMULv4f32)) { 3811 Patterns.push_back(MachineCombinerPattern::FMLSv4f32_OP1); 3812 Found = true; 3813 } 3814 break; 3815 } 3816 return Found; 3817 } 3818 3819 /// Return true when a code sequence can improve throughput. It 3820 /// should be called only for instructions in loops. 3821 /// \param Pattern - combiner pattern 3822 bool AArch64InstrInfo::isThroughputPattern( 3823 MachineCombinerPattern Pattern) const { 3824 switch (Pattern) { 3825 default: 3826 break; 3827 case MachineCombinerPattern::FMULADDS_OP1: 3828 case MachineCombinerPattern::FMULADDS_OP2: 3829 case MachineCombinerPattern::FMULSUBS_OP1: 3830 case MachineCombinerPattern::FMULSUBS_OP2: 3831 case MachineCombinerPattern::FMULADDD_OP1: 3832 case MachineCombinerPattern::FMULADDD_OP2: 3833 case MachineCombinerPattern::FMULSUBD_OP1: 3834 case MachineCombinerPattern::FMULSUBD_OP2: 3835 case MachineCombinerPattern::FNMULSUBS_OP1: 3836 case MachineCombinerPattern::FNMULSUBD_OP1: 3837 case MachineCombinerPattern::FMLAv1i32_indexed_OP1: 3838 case MachineCombinerPattern::FMLAv1i32_indexed_OP2: 3839 case MachineCombinerPattern::FMLAv1i64_indexed_OP1: 3840 case MachineCombinerPattern::FMLAv1i64_indexed_OP2: 3841 case MachineCombinerPattern::FMLAv2f32_OP2: 3842 case MachineCombinerPattern::FMLAv2f32_OP1: 3843 case MachineCombinerPattern::FMLAv2f64_OP1: 3844 case MachineCombinerPattern::FMLAv2f64_OP2: 3845 case MachineCombinerPattern::FMLAv2i32_indexed_OP1: 3846 case MachineCombinerPattern::FMLAv2i32_indexed_OP2: 3847 case MachineCombinerPattern::FMLAv2i64_indexed_OP1: 3848 case MachineCombinerPattern::FMLAv2i64_indexed_OP2: 3849 case MachineCombinerPattern::FMLAv4f32_OP1: 3850 case MachineCombinerPattern::FMLAv4f32_OP2: 3851 case MachineCombinerPattern::FMLAv4i32_indexed_OP1: 3852 case MachineCombinerPattern::FMLAv4i32_indexed_OP2: 3853 case MachineCombinerPattern::FMLSv1i32_indexed_OP2: 3854 case MachineCombinerPattern::FMLSv1i64_indexed_OP2: 3855 case MachineCombinerPattern::FMLSv2i32_indexed_OP2: 3856 case MachineCombinerPattern::FMLSv2i64_indexed_OP2: 3857 case MachineCombinerPattern::FMLSv2f32_OP2: 3858 case MachineCombinerPattern::FMLSv2f64_OP2: 3859 case MachineCombinerPattern::FMLSv4i32_indexed_OP2: 3860 case MachineCombinerPattern::FMLSv4f32_OP2: 3861 return true; 3862 } // end switch (Pattern) 3863 return false; 3864 } 3865 /// Return true when there is potentially a faster code sequence for an 3866 /// instruction chain ending in \p Root. All potential patterns are listed in 3867 /// the \p Pattern vector. Pattern should be sorted in priority order since the 3868 /// pattern evaluator stops checking as soon as it finds a faster sequence. 3869 3870 bool AArch64InstrInfo::getMachineCombinerPatterns( 3871 MachineInstr &Root, 3872 SmallVectorImpl<MachineCombinerPattern> &Patterns) const { 3873 // Integer patterns 3874 if (getMaddPatterns(Root, Patterns)) 3875 return true; 3876 // Floating point patterns 3877 if (getFMAPatterns(Root, Patterns)) 3878 return true; 3879 3880 return TargetInstrInfo::getMachineCombinerPatterns(Root, Patterns); 3881 } 3882 3883 enum class FMAInstKind { Default, Indexed, Accumulator }; 3884 /// genFusedMultiply - Generate fused multiply instructions. 3885 /// This function supports both integer and floating point instructions. 3886 /// A typical example: 3887 /// F|MUL I=A,B,0 3888 /// F|ADD R,I,C 3889 /// ==> F|MADD R,A,B,C 3890 /// \param MF Containing MachineFunction 3891 /// \param MRI Register information 3892 /// \param TII Target information 3893 /// \param Root is the F|ADD instruction 3894 /// \param [out] InsInstrs is a vector of machine instructions and will 3895 /// contain the generated madd instruction 3896 /// \param IdxMulOpd is index of operand in Root that is the result of 3897 /// the F|MUL. In the example above IdxMulOpd is 1. 3898 /// \param MaddOpc the opcode fo the f|madd instruction 3899 /// \param RC Register class of operands 3900 /// \param kind of fma instruction (addressing mode) to be generated 3901 /// \param ReplacedAddend is the result register from the instruction 3902 /// replacing the non-combined operand, if any. 3903 static MachineInstr * 3904 genFusedMultiply(MachineFunction &MF, MachineRegisterInfo &MRI, 3905 const TargetInstrInfo *TII, MachineInstr &Root, 3906 SmallVectorImpl<MachineInstr *> &InsInstrs, unsigned IdxMulOpd, 3907 unsigned MaddOpc, const TargetRegisterClass *RC, 3908 FMAInstKind kind = FMAInstKind::Default, 3909 const unsigned *ReplacedAddend = nullptr) { 3910 assert(IdxMulOpd == 1 || IdxMulOpd == 2); 3911 3912 unsigned IdxOtherOpd = IdxMulOpd == 1 ? 2 : 1; 3913 MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg()); 3914 unsigned ResultReg = Root.getOperand(0).getReg(); 3915 unsigned SrcReg0 = MUL->getOperand(1).getReg(); 3916 bool Src0IsKill = MUL->getOperand(1).isKill(); 3917 unsigned SrcReg1 = MUL->getOperand(2).getReg(); 3918 bool Src1IsKill = MUL->getOperand(2).isKill(); 3919 3920 unsigned SrcReg2; 3921 bool Src2IsKill; 3922 if (ReplacedAddend) { 3923 // If we just generated a new addend, we must be it's only use. 3924 SrcReg2 = *ReplacedAddend; 3925 Src2IsKill = true; 3926 } else { 3927 SrcReg2 = Root.getOperand(IdxOtherOpd).getReg(); 3928 Src2IsKill = Root.getOperand(IdxOtherOpd).isKill(); 3929 } 3930 3931 if (TargetRegisterInfo::isVirtualRegister(ResultReg)) 3932 MRI.constrainRegClass(ResultReg, RC); 3933 if (TargetRegisterInfo::isVirtualRegister(SrcReg0)) 3934 MRI.constrainRegClass(SrcReg0, RC); 3935 if (TargetRegisterInfo::isVirtualRegister(SrcReg1)) 3936 MRI.constrainRegClass(SrcReg1, RC); 3937 if (TargetRegisterInfo::isVirtualRegister(SrcReg2)) 3938 MRI.constrainRegClass(SrcReg2, RC); 3939 3940 MachineInstrBuilder MIB; 3941 if (kind == FMAInstKind::Default) 3942 MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 3943 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 3944 .addReg(SrcReg1, getKillRegState(Src1IsKill)) 3945 .addReg(SrcReg2, getKillRegState(Src2IsKill)); 3946 else if (kind == FMAInstKind::Indexed) 3947 MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 3948 .addReg(SrcReg2, getKillRegState(Src2IsKill)) 3949 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 3950 .addReg(SrcReg1, getKillRegState(Src1IsKill)) 3951 .addImm(MUL->getOperand(3).getImm()); 3952 else if (kind == FMAInstKind::Accumulator) 3953 MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 3954 .addReg(SrcReg2, getKillRegState(Src2IsKill)) 3955 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 3956 .addReg(SrcReg1, getKillRegState(Src1IsKill)); 3957 else 3958 assert(false && "Invalid FMA instruction kind \n"); 3959 // Insert the MADD (MADD, FMA, FMS, FMLA, FMSL) 3960 InsInstrs.push_back(MIB); 3961 return MUL; 3962 } 3963 3964 /// genMaddR - Generate madd instruction and combine mul and add using 3965 /// an extra virtual register 3966 /// Example - an ADD intermediate needs to be stored in a register: 3967 /// MUL I=A,B,0 3968 /// ADD R,I,Imm 3969 /// ==> ORR V, ZR, Imm 3970 /// ==> MADD R,A,B,V 3971 /// \param MF Containing MachineFunction 3972 /// \param MRI Register information 3973 /// \param TII Target information 3974 /// \param Root is the ADD instruction 3975 /// \param [out] InsInstrs is a vector of machine instructions and will 3976 /// contain the generated madd instruction 3977 /// \param IdxMulOpd is index of operand in Root that is the result of 3978 /// the MUL. In the example above IdxMulOpd is 1. 3979 /// \param MaddOpc the opcode fo the madd instruction 3980 /// \param VR is a virtual register that holds the value of an ADD operand 3981 /// (V in the example above). 3982 /// \param RC Register class of operands 3983 static MachineInstr *genMaddR(MachineFunction &MF, MachineRegisterInfo &MRI, 3984 const TargetInstrInfo *TII, MachineInstr &Root, 3985 SmallVectorImpl<MachineInstr *> &InsInstrs, 3986 unsigned IdxMulOpd, unsigned MaddOpc, unsigned VR, 3987 const TargetRegisterClass *RC) { 3988 assert(IdxMulOpd == 1 || IdxMulOpd == 2); 3989 3990 MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg()); 3991 unsigned ResultReg = Root.getOperand(0).getReg(); 3992 unsigned SrcReg0 = MUL->getOperand(1).getReg(); 3993 bool Src0IsKill = MUL->getOperand(1).isKill(); 3994 unsigned SrcReg1 = MUL->getOperand(2).getReg(); 3995 bool Src1IsKill = MUL->getOperand(2).isKill(); 3996 3997 if (TargetRegisterInfo::isVirtualRegister(ResultReg)) 3998 MRI.constrainRegClass(ResultReg, RC); 3999 if (TargetRegisterInfo::isVirtualRegister(SrcReg0)) 4000 MRI.constrainRegClass(SrcReg0, RC); 4001 if (TargetRegisterInfo::isVirtualRegister(SrcReg1)) 4002 MRI.constrainRegClass(SrcReg1, RC); 4003 if (TargetRegisterInfo::isVirtualRegister(VR)) 4004 MRI.constrainRegClass(VR, RC); 4005 4006 MachineInstrBuilder MIB = 4007 BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 4008 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 4009 .addReg(SrcReg1, getKillRegState(Src1IsKill)) 4010 .addReg(VR); 4011 // Insert the MADD 4012 InsInstrs.push_back(MIB); 4013 return MUL; 4014 } 4015 4016 /// When getMachineCombinerPatterns() finds potential patterns, 4017 /// this function generates the instructions that could replace the 4018 /// original code sequence 4019 void AArch64InstrInfo::genAlternativeCodeSequence( 4020 MachineInstr &Root, MachineCombinerPattern Pattern, 4021 SmallVectorImpl<MachineInstr *> &InsInstrs, 4022 SmallVectorImpl<MachineInstr *> &DelInstrs, 4023 DenseMap<unsigned, unsigned> &InstrIdxForVirtReg) const { 4024 MachineBasicBlock &MBB = *Root.getParent(); 4025 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 4026 MachineFunction &MF = *MBB.getParent(); 4027 const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo(); 4028 4029 MachineInstr *MUL; 4030 const TargetRegisterClass *RC; 4031 unsigned Opc; 4032 switch (Pattern) { 4033 default: 4034 // Reassociate instructions. 4035 TargetInstrInfo::genAlternativeCodeSequence(Root, Pattern, InsInstrs, 4036 DelInstrs, InstrIdxForVirtReg); 4037 return; 4038 case MachineCombinerPattern::MULADDW_OP1: 4039 case MachineCombinerPattern::MULADDX_OP1: 4040 // MUL I=A,B,0 4041 // ADD R,I,C 4042 // ==> MADD R,A,B,C 4043 // --- Create(MADD); 4044 if (Pattern == MachineCombinerPattern::MULADDW_OP1) { 4045 Opc = AArch64::MADDWrrr; 4046 RC = &AArch64::GPR32RegClass; 4047 } else { 4048 Opc = AArch64::MADDXrrr; 4049 RC = &AArch64::GPR64RegClass; 4050 } 4051 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4052 break; 4053 case MachineCombinerPattern::MULADDW_OP2: 4054 case MachineCombinerPattern::MULADDX_OP2: 4055 // MUL I=A,B,0 4056 // ADD R,C,I 4057 // ==> MADD R,A,B,C 4058 // --- Create(MADD); 4059 if (Pattern == MachineCombinerPattern::MULADDW_OP2) { 4060 Opc = AArch64::MADDWrrr; 4061 RC = &AArch64::GPR32RegClass; 4062 } else { 4063 Opc = AArch64::MADDXrrr; 4064 RC = &AArch64::GPR64RegClass; 4065 } 4066 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4067 break; 4068 case MachineCombinerPattern::MULADDWI_OP1: 4069 case MachineCombinerPattern::MULADDXI_OP1: { 4070 // MUL I=A,B,0 4071 // ADD R,I,Imm 4072 // ==> ORR V, ZR, Imm 4073 // ==> MADD R,A,B,V 4074 // --- Create(MADD); 4075 const TargetRegisterClass *OrrRC; 4076 unsigned BitSize, OrrOpc, ZeroReg; 4077 if (Pattern == MachineCombinerPattern::MULADDWI_OP1) { 4078 OrrOpc = AArch64::ORRWri; 4079 OrrRC = &AArch64::GPR32spRegClass; 4080 BitSize = 32; 4081 ZeroReg = AArch64::WZR; 4082 Opc = AArch64::MADDWrrr; 4083 RC = &AArch64::GPR32RegClass; 4084 } else { 4085 OrrOpc = AArch64::ORRXri; 4086 OrrRC = &AArch64::GPR64spRegClass; 4087 BitSize = 64; 4088 ZeroReg = AArch64::XZR; 4089 Opc = AArch64::MADDXrrr; 4090 RC = &AArch64::GPR64RegClass; 4091 } 4092 unsigned NewVR = MRI.createVirtualRegister(OrrRC); 4093 uint64_t Imm = Root.getOperand(2).getImm(); 4094 4095 if (Root.getOperand(3).isImm()) { 4096 unsigned Val = Root.getOperand(3).getImm(); 4097 Imm = Imm << Val; 4098 } 4099 uint64_t UImm = SignExtend64(Imm, BitSize); 4100 uint64_t Encoding; 4101 if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) { 4102 MachineInstrBuilder MIB1 = 4103 BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR) 4104 .addReg(ZeroReg) 4105 .addImm(Encoding); 4106 InsInstrs.push_back(MIB1); 4107 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4108 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC); 4109 } 4110 break; 4111 } 4112 case MachineCombinerPattern::MULSUBW_OP1: 4113 case MachineCombinerPattern::MULSUBX_OP1: { 4114 // MUL I=A,B,0 4115 // SUB R,I, C 4116 // ==> SUB V, 0, C 4117 // ==> MADD R,A,B,V // = -C + A*B 4118 // --- Create(MADD); 4119 const TargetRegisterClass *SubRC; 4120 unsigned SubOpc, ZeroReg; 4121 if (Pattern == MachineCombinerPattern::MULSUBW_OP1) { 4122 SubOpc = AArch64::SUBWrr; 4123 SubRC = &AArch64::GPR32spRegClass; 4124 ZeroReg = AArch64::WZR; 4125 Opc = AArch64::MADDWrrr; 4126 RC = &AArch64::GPR32RegClass; 4127 } else { 4128 SubOpc = AArch64::SUBXrr; 4129 SubRC = &AArch64::GPR64spRegClass; 4130 ZeroReg = AArch64::XZR; 4131 Opc = AArch64::MADDXrrr; 4132 RC = &AArch64::GPR64RegClass; 4133 } 4134 unsigned NewVR = MRI.createVirtualRegister(SubRC); 4135 // SUB NewVR, 0, C 4136 MachineInstrBuilder MIB1 = 4137 BuildMI(MF, Root.getDebugLoc(), TII->get(SubOpc), NewVR) 4138 .addReg(ZeroReg) 4139 .add(Root.getOperand(2)); 4140 InsInstrs.push_back(MIB1); 4141 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4142 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC); 4143 break; 4144 } 4145 case MachineCombinerPattern::MULSUBW_OP2: 4146 case MachineCombinerPattern::MULSUBX_OP2: 4147 // MUL I=A,B,0 4148 // SUB R,C,I 4149 // ==> MSUB R,A,B,C (computes C - A*B) 4150 // --- Create(MSUB); 4151 if (Pattern == MachineCombinerPattern::MULSUBW_OP2) { 4152 Opc = AArch64::MSUBWrrr; 4153 RC = &AArch64::GPR32RegClass; 4154 } else { 4155 Opc = AArch64::MSUBXrrr; 4156 RC = &AArch64::GPR64RegClass; 4157 } 4158 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4159 break; 4160 case MachineCombinerPattern::MULSUBWI_OP1: 4161 case MachineCombinerPattern::MULSUBXI_OP1: { 4162 // MUL I=A,B,0 4163 // SUB R,I, Imm 4164 // ==> ORR V, ZR, -Imm 4165 // ==> MADD R,A,B,V // = -Imm + A*B 4166 // --- Create(MADD); 4167 const TargetRegisterClass *OrrRC; 4168 unsigned BitSize, OrrOpc, ZeroReg; 4169 if (Pattern == MachineCombinerPattern::MULSUBWI_OP1) { 4170 OrrOpc = AArch64::ORRWri; 4171 OrrRC = &AArch64::GPR32spRegClass; 4172 BitSize = 32; 4173 ZeroReg = AArch64::WZR; 4174 Opc = AArch64::MADDWrrr; 4175 RC = &AArch64::GPR32RegClass; 4176 } else { 4177 OrrOpc = AArch64::ORRXri; 4178 OrrRC = &AArch64::GPR64spRegClass; 4179 BitSize = 64; 4180 ZeroReg = AArch64::XZR; 4181 Opc = AArch64::MADDXrrr; 4182 RC = &AArch64::GPR64RegClass; 4183 } 4184 unsigned NewVR = MRI.createVirtualRegister(OrrRC); 4185 uint64_t Imm = Root.getOperand(2).getImm(); 4186 if (Root.getOperand(3).isImm()) { 4187 unsigned Val = Root.getOperand(3).getImm(); 4188 Imm = Imm << Val; 4189 } 4190 uint64_t UImm = SignExtend64(-Imm, BitSize); 4191 uint64_t Encoding; 4192 if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) { 4193 MachineInstrBuilder MIB1 = 4194 BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR) 4195 .addReg(ZeroReg) 4196 .addImm(Encoding); 4197 InsInstrs.push_back(MIB1); 4198 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4199 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC); 4200 } 4201 break; 4202 } 4203 // Floating Point Support 4204 case MachineCombinerPattern::FMULADDS_OP1: 4205 case MachineCombinerPattern::FMULADDD_OP1: 4206 // MUL I=A,B,0 4207 // ADD R,I,C 4208 // ==> MADD R,A,B,C 4209 // --- Create(MADD); 4210 if (Pattern == MachineCombinerPattern::FMULADDS_OP1) { 4211 Opc = AArch64::FMADDSrrr; 4212 RC = &AArch64::FPR32RegClass; 4213 } else { 4214 Opc = AArch64::FMADDDrrr; 4215 RC = &AArch64::FPR64RegClass; 4216 } 4217 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4218 break; 4219 case MachineCombinerPattern::FMULADDS_OP2: 4220 case MachineCombinerPattern::FMULADDD_OP2: 4221 // FMUL I=A,B,0 4222 // FADD R,C,I 4223 // ==> FMADD R,A,B,C 4224 // --- Create(FMADD); 4225 if (Pattern == MachineCombinerPattern::FMULADDS_OP2) { 4226 Opc = AArch64::FMADDSrrr; 4227 RC = &AArch64::FPR32RegClass; 4228 } else { 4229 Opc = AArch64::FMADDDrrr; 4230 RC = &AArch64::FPR64RegClass; 4231 } 4232 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4233 break; 4234 4235 case MachineCombinerPattern::FMLAv1i32_indexed_OP1: 4236 Opc = AArch64::FMLAv1i32_indexed; 4237 RC = &AArch64::FPR32RegClass; 4238 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4239 FMAInstKind::Indexed); 4240 break; 4241 case MachineCombinerPattern::FMLAv1i32_indexed_OP2: 4242 Opc = AArch64::FMLAv1i32_indexed; 4243 RC = &AArch64::FPR32RegClass; 4244 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4245 FMAInstKind::Indexed); 4246 break; 4247 4248 case MachineCombinerPattern::FMLAv1i64_indexed_OP1: 4249 Opc = AArch64::FMLAv1i64_indexed; 4250 RC = &AArch64::FPR64RegClass; 4251 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4252 FMAInstKind::Indexed); 4253 break; 4254 case MachineCombinerPattern::FMLAv1i64_indexed_OP2: 4255 Opc = AArch64::FMLAv1i64_indexed; 4256 RC = &AArch64::FPR64RegClass; 4257 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4258 FMAInstKind::Indexed); 4259 break; 4260 4261 case MachineCombinerPattern::FMLAv2i32_indexed_OP1: 4262 case MachineCombinerPattern::FMLAv2f32_OP1: 4263 RC = &AArch64::FPR64RegClass; 4264 if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP1) { 4265 Opc = AArch64::FMLAv2i32_indexed; 4266 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4267 FMAInstKind::Indexed); 4268 } else { 4269 Opc = AArch64::FMLAv2f32; 4270 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4271 FMAInstKind::Accumulator); 4272 } 4273 break; 4274 case MachineCombinerPattern::FMLAv2i32_indexed_OP2: 4275 case MachineCombinerPattern::FMLAv2f32_OP2: 4276 RC = &AArch64::FPR64RegClass; 4277 if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP2) { 4278 Opc = AArch64::FMLAv2i32_indexed; 4279 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4280 FMAInstKind::Indexed); 4281 } else { 4282 Opc = AArch64::FMLAv2f32; 4283 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4284 FMAInstKind::Accumulator); 4285 } 4286 break; 4287 4288 case MachineCombinerPattern::FMLAv2i64_indexed_OP1: 4289 case MachineCombinerPattern::FMLAv2f64_OP1: 4290 RC = &AArch64::FPR128RegClass; 4291 if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP1) { 4292 Opc = AArch64::FMLAv2i64_indexed; 4293 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4294 FMAInstKind::Indexed); 4295 } else { 4296 Opc = AArch64::FMLAv2f64; 4297 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4298 FMAInstKind::Accumulator); 4299 } 4300 break; 4301 case MachineCombinerPattern::FMLAv2i64_indexed_OP2: 4302 case MachineCombinerPattern::FMLAv2f64_OP2: 4303 RC = &AArch64::FPR128RegClass; 4304 if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP2) { 4305 Opc = AArch64::FMLAv2i64_indexed; 4306 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4307 FMAInstKind::Indexed); 4308 } else { 4309 Opc = AArch64::FMLAv2f64; 4310 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4311 FMAInstKind::Accumulator); 4312 } 4313 break; 4314 4315 case MachineCombinerPattern::FMLAv4i32_indexed_OP1: 4316 case MachineCombinerPattern::FMLAv4f32_OP1: 4317 RC = &AArch64::FPR128RegClass; 4318 if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP1) { 4319 Opc = AArch64::FMLAv4i32_indexed; 4320 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4321 FMAInstKind::Indexed); 4322 } else { 4323 Opc = AArch64::FMLAv4f32; 4324 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4325 FMAInstKind::Accumulator); 4326 } 4327 break; 4328 4329 case MachineCombinerPattern::FMLAv4i32_indexed_OP2: 4330 case MachineCombinerPattern::FMLAv4f32_OP2: 4331 RC = &AArch64::FPR128RegClass; 4332 if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP2) { 4333 Opc = AArch64::FMLAv4i32_indexed; 4334 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4335 FMAInstKind::Indexed); 4336 } else { 4337 Opc = AArch64::FMLAv4f32; 4338 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4339 FMAInstKind::Accumulator); 4340 } 4341 break; 4342 4343 case MachineCombinerPattern::FMULSUBS_OP1: 4344 case MachineCombinerPattern::FMULSUBD_OP1: { 4345 // FMUL I=A,B,0 4346 // FSUB R,I,C 4347 // ==> FNMSUB R,A,B,C // = -C + A*B 4348 // --- Create(FNMSUB); 4349 if (Pattern == MachineCombinerPattern::FMULSUBS_OP1) { 4350 Opc = AArch64::FNMSUBSrrr; 4351 RC = &AArch64::FPR32RegClass; 4352 } else { 4353 Opc = AArch64::FNMSUBDrrr; 4354 RC = &AArch64::FPR64RegClass; 4355 } 4356 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4357 break; 4358 } 4359 4360 case MachineCombinerPattern::FNMULSUBS_OP1: 4361 case MachineCombinerPattern::FNMULSUBD_OP1: { 4362 // FNMUL I=A,B,0 4363 // FSUB R,I,C 4364 // ==> FNMADD R,A,B,C // = -A*B - C 4365 // --- Create(FNMADD); 4366 if (Pattern == MachineCombinerPattern::FNMULSUBS_OP1) { 4367 Opc = AArch64::FNMADDSrrr; 4368 RC = &AArch64::FPR32RegClass; 4369 } else { 4370 Opc = AArch64::FNMADDDrrr; 4371 RC = &AArch64::FPR64RegClass; 4372 } 4373 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4374 break; 4375 } 4376 4377 case MachineCombinerPattern::FMULSUBS_OP2: 4378 case MachineCombinerPattern::FMULSUBD_OP2: { 4379 // FMUL I=A,B,0 4380 // FSUB R,C,I 4381 // ==> FMSUB R,A,B,C (computes C - A*B) 4382 // --- Create(FMSUB); 4383 if (Pattern == MachineCombinerPattern::FMULSUBS_OP2) { 4384 Opc = AArch64::FMSUBSrrr; 4385 RC = &AArch64::FPR32RegClass; 4386 } else { 4387 Opc = AArch64::FMSUBDrrr; 4388 RC = &AArch64::FPR64RegClass; 4389 } 4390 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4391 break; 4392 } 4393 4394 case MachineCombinerPattern::FMLSv1i32_indexed_OP2: 4395 Opc = AArch64::FMLSv1i32_indexed; 4396 RC = &AArch64::FPR32RegClass; 4397 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4398 FMAInstKind::Indexed); 4399 break; 4400 4401 case MachineCombinerPattern::FMLSv1i64_indexed_OP2: 4402 Opc = AArch64::FMLSv1i64_indexed; 4403 RC = &AArch64::FPR64RegClass; 4404 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4405 FMAInstKind::Indexed); 4406 break; 4407 4408 case MachineCombinerPattern::FMLSv2f32_OP2: 4409 case MachineCombinerPattern::FMLSv2i32_indexed_OP2: 4410 RC = &AArch64::FPR64RegClass; 4411 if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP2) { 4412 Opc = AArch64::FMLSv2i32_indexed; 4413 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4414 FMAInstKind::Indexed); 4415 } else { 4416 Opc = AArch64::FMLSv2f32; 4417 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4418 FMAInstKind::Accumulator); 4419 } 4420 break; 4421 4422 case MachineCombinerPattern::FMLSv2f64_OP2: 4423 case MachineCombinerPattern::FMLSv2i64_indexed_OP2: 4424 RC = &AArch64::FPR128RegClass; 4425 if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP2) { 4426 Opc = AArch64::FMLSv2i64_indexed; 4427 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4428 FMAInstKind::Indexed); 4429 } else { 4430 Opc = AArch64::FMLSv2f64; 4431 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4432 FMAInstKind::Accumulator); 4433 } 4434 break; 4435 4436 case MachineCombinerPattern::FMLSv4f32_OP2: 4437 case MachineCombinerPattern::FMLSv4i32_indexed_OP2: 4438 RC = &AArch64::FPR128RegClass; 4439 if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP2) { 4440 Opc = AArch64::FMLSv4i32_indexed; 4441 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4442 FMAInstKind::Indexed); 4443 } else { 4444 Opc = AArch64::FMLSv4f32; 4445 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4446 FMAInstKind::Accumulator); 4447 } 4448 break; 4449 case MachineCombinerPattern::FMLSv2f32_OP1: 4450 case MachineCombinerPattern::FMLSv2i32_indexed_OP1: { 4451 RC = &AArch64::FPR64RegClass; 4452 unsigned NewVR = MRI.createVirtualRegister(RC); 4453 MachineInstrBuilder MIB1 = 4454 BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f32), NewVR) 4455 .add(Root.getOperand(2)); 4456 InsInstrs.push_back(MIB1); 4457 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4458 if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP1) { 4459 Opc = AArch64::FMLAv2i32_indexed; 4460 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4461 FMAInstKind::Indexed, &NewVR); 4462 } else { 4463 Opc = AArch64::FMLAv2f32; 4464 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4465 FMAInstKind::Accumulator, &NewVR); 4466 } 4467 break; 4468 } 4469 case MachineCombinerPattern::FMLSv4f32_OP1: 4470 case MachineCombinerPattern::FMLSv4i32_indexed_OP1: { 4471 RC = &AArch64::FPR128RegClass; 4472 unsigned NewVR = MRI.createVirtualRegister(RC); 4473 MachineInstrBuilder MIB1 = 4474 BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv4f32), NewVR) 4475 .add(Root.getOperand(2)); 4476 InsInstrs.push_back(MIB1); 4477 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4478 if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP1) { 4479 Opc = AArch64::FMLAv4i32_indexed; 4480 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4481 FMAInstKind::Indexed, &NewVR); 4482 } else { 4483 Opc = AArch64::FMLAv4f32; 4484 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4485 FMAInstKind::Accumulator, &NewVR); 4486 } 4487 break; 4488 } 4489 case MachineCombinerPattern::FMLSv2f64_OP1: 4490 case MachineCombinerPattern::FMLSv2i64_indexed_OP1: { 4491 RC = &AArch64::FPR128RegClass; 4492 unsigned NewVR = MRI.createVirtualRegister(RC); 4493 MachineInstrBuilder MIB1 = 4494 BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f64), NewVR) 4495 .add(Root.getOperand(2)); 4496 InsInstrs.push_back(MIB1); 4497 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4498 if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP1) { 4499 Opc = AArch64::FMLAv2i64_indexed; 4500 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4501 FMAInstKind::Indexed, &NewVR); 4502 } else { 4503 Opc = AArch64::FMLAv2f64; 4504 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4505 FMAInstKind::Accumulator, &NewVR); 4506 } 4507 break; 4508 } 4509 } // end switch (Pattern) 4510 // Record MUL and ADD/SUB for deletion 4511 DelInstrs.push_back(MUL); 4512 DelInstrs.push_back(&Root); 4513 } 4514 4515 /// Replace csincr-branch sequence by simple conditional branch 4516 /// 4517 /// Examples: 4518 /// 1. \code 4519 /// csinc w9, wzr, wzr, <condition code> 4520 /// tbnz w9, #0, 0x44 4521 /// \endcode 4522 /// to 4523 /// \code 4524 /// b.<inverted condition code> 4525 /// \endcode 4526 /// 4527 /// 2. \code 4528 /// csinc w9, wzr, wzr, <condition code> 4529 /// tbz w9, #0, 0x44 4530 /// \endcode 4531 /// to 4532 /// \code 4533 /// b.<condition code> 4534 /// \endcode 4535 /// 4536 /// Replace compare and branch sequence by TBZ/TBNZ instruction when the 4537 /// compare's constant operand is power of 2. 4538 /// 4539 /// Examples: 4540 /// \code 4541 /// and w8, w8, #0x400 4542 /// cbnz w8, L1 4543 /// \endcode 4544 /// to 4545 /// \code 4546 /// tbnz w8, #10, L1 4547 /// \endcode 4548 /// 4549 /// \param MI Conditional Branch 4550 /// \return True when the simple conditional branch is generated 4551 /// 4552 bool AArch64InstrInfo::optimizeCondBranch(MachineInstr &MI) const { 4553 bool IsNegativeBranch = false; 4554 bool IsTestAndBranch = false; 4555 unsigned TargetBBInMI = 0; 4556 switch (MI.getOpcode()) { 4557 default: 4558 llvm_unreachable("Unknown branch instruction?"); 4559 case AArch64::Bcc: 4560 return false; 4561 case AArch64::CBZW: 4562 case AArch64::CBZX: 4563 TargetBBInMI = 1; 4564 break; 4565 case AArch64::CBNZW: 4566 case AArch64::CBNZX: 4567 TargetBBInMI = 1; 4568 IsNegativeBranch = true; 4569 break; 4570 case AArch64::TBZW: 4571 case AArch64::TBZX: 4572 TargetBBInMI = 2; 4573 IsTestAndBranch = true; 4574 break; 4575 case AArch64::TBNZW: 4576 case AArch64::TBNZX: 4577 TargetBBInMI = 2; 4578 IsNegativeBranch = true; 4579 IsTestAndBranch = true; 4580 break; 4581 } 4582 // So we increment a zero register and test for bits other 4583 // than bit 0? Conservatively bail out in case the verifier 4584 // missed this case. 4585 if (IsTestAndBranch && MI.getOperand(1).getImm()) 4586 return false; 4587 4588 // Find Definition. 4589 assert(MI.getParent() && "Incomplete machine instruciton\n"); 4590 MachineBasicBlock *MBB = MI.getParent(); 4591 MachineFunction *MF = MBB->getParent(); 4592 MachineRegisterInfo *MRI = &MF->getRegInfo(); 4593 unsigned VReg = MI.getOperand(0).getReg(); 4594 if (!TargetRegisterInfo::isVirtualRegister(VReg)) 4595 return false; 4596 4597 MachineInstr *DefMI = MRI->getVRegDef(VReg); 4598 4599 // Look through COPY instructions to find definition. 4600 while (DefMI->isCopy()) { 4601 unsigned CopyVReg = DefMI->getOperand(1).getReg(); 4602 if (!MRI->hasOneNonDBGUse(CopyVReg)) 4603 return false; 4604 if (!MRI->hasOneDef(CopyVReg)) 4605 return false; 4606 DefMI = MRI->getVRegDef(CopyVReg); 4607 } 4608 4609 switch (DefMI->getOpcode()) { 4610 default: 4611 return false; 4612 // Fold AND into a TBZ/TBNZ if constant operand is power of 2. 4613 case AArch64::ANDWri: 4614 case AArch64::ANDXri: { 4615 if (IsTestAndBranch) 4616 return false; 4617 if (DefMI->getParent() != MBB) 4618 return false; 4619 if (!MRI->hasOneNonDBGUse(VReg)) 4620 return false; 4621 4622 bool Is32Bit = (DefMI->getOpcode() == AArch64::ANDWri); 4623 uint64_t Mask = AArch64_AM::decodeLogicalImmediate( 4624 DefMI->getOperand(2).getImm(), Is32Bit ? 32 : 64); 4625 if (!isPowerOf2_64(Mask)) 4626 return false; 4627 4628 MachineOperand &MO = DefMI->getOperand(1); 4629 unsigned NewReg = MO.getReg(); 4630 if (!TargetRegisterInfo::isVirtualRegister(NewReg)) 4631 return false; 4632 4633 assert(!MRI->def_empty(NewReg) && "Register must be defined."); 4634 4635 MachineBasicBlock &RefToMBB = *MBB; 4636 MachineBasicBlock *TBB = MI.getOperand(1).getMBB(); 4637 DebugLoc DL = MI.getDebugLoc(); 4638 unsigned Imm = Log2_64(Mask); 4639 unsigned Opc = (Imm < 32) 4640 ? (IsNegativeBranch ? AArch64::TBNZW : AArch64::TBZW) 4641 : (IsNegativeBranch ? AArch64::TBNZX : AArch64::TBZX); 4642 MachineInstr *NewMI = BuildMI(RefToMBB, MI, DL, get(Opc)) 4643 .addReg(NewReg) 4644 .addImm(Imm) 4645 .addMBB(TBB); 4646 // Register lives on to the CBZ now. 4647 MO.setIsKill(false); 4648 4649 // For immediate smaller than 32, we need to use the 32-bit 4650 // variant (W) in all cases. Indeed the 64-bit variant does not 4651 // allow to encode them. 4652 // Therefore, if the input register is 64-bit, we need to take the 4653 // 32-bit sub-part. 4654 if (!Is32Bit && Imm < 32) 4655 NewMI->getOperand(0).setSubReg(AArch64::sub_32); 4656 MI.eraseFromParent(); 4657 return true; 4658 } 4659 // Look for CSINC 4660 case AArch64::CSINCWr: 4661 case AArch64::CSINCXr: { 4662 if (!(DefMI->getOperand(1).getReg() == AArch64::WZR && 4663 DefMI->getOperand(2).getReg() == AArch64::WZR) && 4664 !(DefMI->getOperand(1).getReg() == AArch64::XZR && 4665 DefMI->getOperand(2).getReg() == AArch64::XZR)) 4666 return false; 4667 4668 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) != -1) 4669 return false; 4670 4671 AArch64CC::CondCode CC = (AArch64CC::CondCode)DefMI->getOperand(3).getImm(); 4672 // Convert only when the condition code is not modified between 4673 // the CSINC and the branch. The CC may be used by other 4674 // instructions in between. 4675 if (areCFlagsAccessedBetweenInstrs(DefMI, MI, &getRegisterInfo(), AK_Write)) 4676 return false; 4677 MachineBasicBlock &RefToMBB = *MBB; 4678 MachineBasicBlock *TBB = MI.getOperand(TargetBBInMI).getMBB(); 4679 DebugLoc DL = MI.getDebugLoc(); 4680 if (IsNegativeBranch) 4681 CC = AArch64CC::getInvertedCondCode(CC); 4682 BuildMI(RefToMBB, MI, DL, get(AArch64::Bcc)).addImm(CC).addMBB(TBB); 4683 MI.eraseFromParent(); 4684 return true; 4685 } 4686 } 4687 } 4688 4689 std::pair<unsigned, unsigned> 4690 AArch64InstrInfo::decomposeMachineOperandsTargetFlags(unsigned TF) const { 4691 const unsigned Mask = AArch64II::MO_FRAGMENT; 4692 return std::make_pair(TF & Mask, TF & ~Mask); 4693 } 4694 4695 ArrayRef<std::pair<unsigned, const char *>> 4696 AArch64InstrInfo::getSerializableDirectMachineOperandTargetFlags() const { 4697 using namespace AArch64II; 4698 4699 static const std::pair<unsigned, const char *> TargetFlags[] = { 4700 {MO_PAGE, "aarch64-page"}, {MO_PAGEOFF, "aarch64-pageoff"}, 4701 {MO_G3, "aarch64-g3"}, {MO_G2, "aarch64-g2"}, 4702 {MO_G1, "aarch64-g1"}, {MO_G0, "aarch64-g0"}, 4703 {MO_HI12, "aarch64-hi12"}}; 4704 return makeArrayRef(TargetFlags); 4705 } 4706 4707 ArrayRef<std::pair<unsigned, const char *>> 4708 AArch64InstrInfo::getSerializableBitmaskMachineOperandTargetFlags() const { 4709 using namespace AArch64II; 4710 4711 static const std::pair<unsigned, const char *> TargetFlags[] = { 4712 {MO_COFFSTUB, "aarch64-coffstub"}, 4713 {MO_GOT, "aarch64-got"}, {MO_NC, "aarch64-nc"}, 4714 {MO_S, "aarch64-s"}, {MO_TLS, "aarch64-tls"}, 4715 {MO_DLLIMPORT, "aarch64-dllimport"}}; 4716 return makeArrayRef(TargetFlags); 4717 } 4718 4719 ArrayRef<std::pair<MachineMemOperand::Flags, const char *>> 4720 AArch64InstrInfo::getSerializableMachineMemOperandTargetFlags() const { 4721 static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] = 4722 {{MOSuppressPair, "aarch64-suppress-pair"}, 4723 {MOStridedAccess, "aarch64-strided-access"}}; 4724 return makeArrayRef(TargetFlags); 4725 } 4726 4727 /// Constants defining how certain sequences should be outlined. 4728 /// This encompasses how an outlined function should be called, and what kind of 4729 /// frame should be emitted for that outlined function. 4730 /// 4731 /// \p MachineOutlinerDefault implies that the function should be called with 4732 /// a save and restore of LR to the stack. 4733 /// 4734 /// That is, 4735 /// 4736 /// I1 Save LR OUTLINED_FUNCTION: 4737 /// I2 --> BL OUTLINED_FUNCTION I1 4738 /// I3 Restore LR I2 4739 /// I3 4740 /// RET 4741 /// 4742 /// * Call construction overhead: 3 (save + BL + restore) 4743 /// * Frame construction overhead: 1 (ret) 4744 /// * Requires stack fixups? Yes 4745 /// 4746 /// \p MachineOutlinerTailCall implies that the function is being created from 4747 /// a sequence of instructions ending in a return. 4748 /// 4749 /// That is, 4750 /// 4751 /// I1 OUTLINED_FUNCTION: 4752 /// I2 --> B OUTLINED_FUNCTION I1 4753 /// RET I2 4754 /// RET 4755 /// 4756 /// * Call construction overhead: 1 (B) 4757 /// * Frame construction overhead: 0 (Return included in sequence) 4758 /// * Requires stack fixups? No 4759 /// 4760 /// \p MachineOutlinerNoLRSave implies that the function should be called using 4761 /// a BL instruction, but doesn't require LR to be saved and restored. This 4762 /// happens when LR is known to be dead. 4763 /// 4764 /// That is, 4765 /// 4766 /// I1 OUTLINED_FUNCTION: 4767 /// I2 --> BL OUTLINED_FUNCTION I1 4768 /// I3 I2 4769 /// I3 4770 /// RET 4771 /// 4772 /// * Call construction overhead: 1 (BL) 4773 /// * Frame construction overhead: 1 (RET) 4774 /// * Requires stack fixups? No 4775 /// 4776 /// \p MachineOutlinerThunk implies that the function is being created from 4777 /// a sequence of instructions ending in a call. The outlined function is 4778 /// called with a BL instruction, and the outlined function tail-calls the 4779 /// original call destination. 4780 /// 4781 /// That is, 4782 /// 4783 /// I1 OUTLINED_FUNCTION: 4784 /// I2 --> BL OUTLINED_FUNCTION I1 4785 /// BL f I2 4786 /// B f 4787 /// * Call construction overhead: 1 (BL) 4788 /// * Frame construction overhead: 0 4789 /// * Requires stack fixups? No 4790 /// 4791 /// \p MachineOutlinerRegSave implies that the function should be called with a 4792 /// save and restore of LR to an available register. This allows us to avoid 4793 /// stack fixups. Note that this outlining variant is compatible with the 4794 /// NoLRSave case. 4795 /// 4796 /// That is, 4797 /// 4798 /// I1 Save LR OUTLINED_FUNCTION: 4799 /// I2 --> BL OUTLINED_FUNCTION I1 4800 /// I3 Restore LR I2 4801 /// I3 4802 /// RET 4803 /// 4804 /// * Call construction overhead: 3 (save + BL + restore) 4805 /// * Frame construction overhead: 1 (ret) 4806 /// * Requires stack fixups? No 4807 enum MachineOutlinerClass { 4808 MachineOutlinerDefault, /// Emit a save, restore, call, and return. 4809 MachineOutlinerTailCall, /// Only emit a branch. 4810 MachineOutlinerNoLRSave, /// Emit a call and return. 4811 MachineOutlinerThunk, /// Emit a call and tail-call. 4812 MachineOutlinerRegSave /// Same as default, but save to a register. 4813 }; 4814 4815 enum MachineOutlinerMBBFlags { 4816 LRUnavailableSomewhere = 0x2, 4817 HasCalls = 0x4, 4818 UnsafeRegsDead = 0x8 4819 }; 4820 4821 unsigned 4822 AArch64InstrInfo::findRegisterToSaveLRTo(const outliner::Candidate &C) const { 4823 assert(C.LRUWasSet && "LRU wasn't set?"); 4824 MachineFunction *MF = C.getMF(); 4825 const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>( 4826 MF->getSubtarget().getRegisterInfo()); 4827 4828 // Check if there is an available register across the sequence that we can 4829 // use. 4830 for (unsigned Reg : AArch64::GPR64RegClass) { 4831 if (!ARI->isReservedReg(*MF, Reg) && 4832 Reg != AArch64::LR && // LR is not reserved, but don't use it. 4833 Reg != AArch64::X16 && // X16 is not guaranteed to be preserved. 4834 Reg != AArch64::X17 && // Ditto for X17. 4835 C.LRU.available(Reg) && C.UsedInSequence.available(Reg)) 4836 return Reg; 4837 } 4838 4839 // No suitable register. Return 0. 4840 return 0u; 4841 } 4842 4843 outliner::OutlinedFunction 4844 AArch64InstrInfo::getOutliningCandidateInfo( 4845 std::vector<outliner::Candidate> &RepeatedSequenceLocs) const { 4846 outliner::Candidate &FirstCand = RepeatedSequenceLocs[0]; 4847 unsigned SequenceSize = 4848 std::accumulate(FirstCand.front(), std::next(FirstCand.back()), 0, 4849 [this](unsigned Sum, const MachineInstr &MI) { 4850 return Sum + getInstSizeInBytes(MI); 4851 }); 4852 4853 // Properties about candidate MBBs that hold for all of them. 4854 unsigned FlagsSetInAll = 0xF; 4855 4856 // Compute liveness information for each candidate, and set FlagsSetInAll. 4857 const TargetRegisterInfo &TRI = getRegisterInfo(); 4858 std::for_each(RepeatedSequenceLocs.begin(), RepeatedSequenceLocs.end(), 4859 [&FlagsSetInAll](outliner::Candidate &C) { 4860 FlagsSetInAll &= C.Flags; 4861 }); 4862 4863 // According to the AArch64 Procedure Call Standard, the following are 4864 // undefined on entry/exit from a function call: 4865 // 4866 // * Registers x16, x17, (and thus w16, w17) 4867 // * Condition codes (and thus the NZCV register) 4868 // 4869 // Because if this, we can't outline any sequence of instructions where 4870 // one 4871 // of these registers is live into/across it. Thus, we need to delete 4872 // those 4873 // candidates. 4874 auto CantGuaranteeValueAcrossCall = [&TRI](outliner::Candidate &C) { 4875 // If the unsafe registers in this block are all dead, then we don't need 4876 // to compute liveness here. 4877 if (C.Flags & UnsafeRegsDead) 4878 return false; 4879 C.initLRU(TRI); 4880 LiveRegUnits LRU = C.LRU; 4881 return (!LRU.available(AArch64::W16) || !LRU.available(AArch64::W17) || 4882 !LRU.available(AArch64::NZCV)); 4883 }; 4884 4885 // Are there any candidates where those registers are live? 4886 if (!(FlagsSetInAll & UnsafeRegsDead)) { 4887 // Erase every candidate that violates the restrictions above. (It could be 4888 // true that we have viable candidates, so it's not worth bailing out in 4889 // the case that, say, 1 out of 20 candidates violate the restructions.) 4890 RepeatedSequenceLocs.erase(std::remove_if(RepeatedSequenceLocs.begin(), 4891 RepeatedSequenceLocs.end(), 4892 CantGuaranteeValueAcrossCall), 4893 RepeatedSequenceLocs.end()); 4894 4895 // If the sequence doesn't have enough candidates left, then we're done. 4896 if (RepeatedSequenceLocs.size() < 2) 4897 return outliner::OutlinedFunction(); 4898 } 4899 4900 // At this point, we have only "safe" candidates to outline. Figure out 4901 // frame + call instruction information. 4902 4903 unsigned LastInstrOpcode = RepeatedSequenceLocs[0].back()->getOpcode(); 4904 4905 // Helper lambda which sets call information for every candidate. 4906 auto SetCandidateCallInfo = 4907 [&RepeatedSequenceLocs](unsigned CallID, unsigned NumBytesForCall) { 4908 for (outliner::Candidate &C : RepeatedSequenceLocs) 4909 C.setCallInfo(CallID, NumBytesForCall); 4910 }; 4911 4912 unsigned FrameID = MachineOutlinerDefault; 4913 unsigned NumBytesToCreateFrame = 4; 4914 4915 bool HasBTI = any_of(RepeatedSequenceLocs, [](outliner::Candidate &C) { 4916 return C.getMF()->getFunction().hasFnAttribute("branch-target-enforcement"); 4917 }); 4918 4919 // Returns true if an instructions is safe to fix up, false otherwise. 4920 auto IsSafeToFixup = [this, &TRI](MachineInstr &MI) { 4921 if (MI.isCall()) 4922 return true; 4923 4924 if (!MI.modifiesRegister(AArch64::SP, &TRI) && 4925 !MI.readsRegister(AArch64::SP, &TRI)) 4926 return true; 4927 4928 // Any modification of SP will break our code to save/restore LR. 4929 // FIXME: We could handle some instructions which add a constant 4930 // offset to SP, with a bit more work. 4931 if (MI.modifiesRegister(AArch64::SP, &TRI)) 4932 return false; 4933 4934 // At this point, we have a stack instruction that we might need to 4935 // fix up. We'll handle it if it's a load or store. 4936 if (MI.mayLoadOrStore()) { 4937 const MachineOperand *Base; // Filled with the base operand of MI. 4938 int64_t Offset; // Filled with the offset of MI. 4939 4940 // Does it allow us to offset the base operand and is the base the 4941 // register SP? 4942 if (!getMemOperandWithOffset(MI, Base, Offset, &TRI) || !Base->isReg() || 4943 Base->getReg() != AArch64::SP) 4944 return false; 4945 4946 // Find the minimum/maximum offset for this instruction and check 4947 // if fixing it up would be in range. 4948 int64_t MinOffset, 4949 MaxOffset; // Unscaled offsets for the instruction. 4950 unsigned Scale; // The scale to multiply the offsets by. 4951 unsigned DummyWidth; 4952 getMemOpInfo(MI.getOpcode(), Scale, DummyWidth, MinOffset, MaxOffset); 4953 4954 Offset += 16; // Update the offset to what it would be if we outlined. 4955 if (Offset < MinOffset * Scale || Offset > MaxOffset * Scale) 4956 return false; 4957 4958 // It's in range, so we can outline it. 4959 return true; 4960 } 4961 4962 // FIXME: Add handling for instructions like "add x0, sp, #8". 4963 4964 // We can't fix it up, so don't outline it. 4965 return false; 4966 }; 4967 4968 // True if it's possible to fix up each stack instruction in this sequence. 4969 // Important for frames/call variants that modify the stack. 4970 bool AllStackInstrsSafe = std::all_of( 4971 FirstCand.front(), std::next(FirstCand.back()), IsSafeToFixup); 4972 4973 // If the last instruction in any candidate is a terminator, then we should 4974 // tail call all of the candidates. 4975 if (RepeatedSequenceLocs[0].back()->isTerminator()) { 4976 FrameID = MachineOutlinerTailCall; 4977 NumBytesToCreateFrame = 0; 4978 SetCandidateCallInfo(MachineOutlinerTailCall, 4); 4979 } 4980 4981 else if (LastInstrOpcode == AArch64::BL || 4982 (LastInstrOpcode == AArch64::BLR && !HasBTI)) { 4983 // FIXME: Do we need to check if the code after this uses the value of LR? 4984 FrameID = MachineOutlinerThunk; 4985 NumBytesToCreateFrame = 0; 4986 SetCandidateCallInfo(MachineOutlinerThunk, 4); 4987 } 4988 4989 else { 4990 // We need to decide how to emit calls + frames. We can always emit the same 4991 // frame if we don't need to save to the stack. If we have to save to the 4992 // stack, then we need a different frame. 4993 unsigned NumBytesNoStackCalls = 0; 4994 std::vector<outliner::Candidate> CandidatesWithoutStackFixups; 4995 4996 for (outliner::Candidate &C : RepeatedSequenceLocs) { 4997 C.initLRU(TRI); 4998 4999 // Is LR available? If so, we don't need a save. 5000 if (C.LRU.available(AArch64::LR)) { 5001 NumBytesNoStackCalls += 4; 5002 C.setCallInfo(MachineOutlinerNoLRSave, 4); 5003 CandidatesWithoutStackFixups.push_back(C); 5004 } 5005 5006 // Is an unused register available? If so, we won't modify the stack, so 5007 // we can outline with the same frame type as those that don't save LR. 5008 else if (findRegisterToSaveLRTo(C)) { 5009 NumBytesNoStackCalls += 12; 5010 C.setCallInfo(MachineOutlinerRegSave, 12); 5011 CandidatesWithoutStackFixups.push_back(C); 5012 } 5013 5014 // Is SP used in the sequence at all? If not, we don't have to modify 5015 // the stack, so we are guaranteed to get the same frame. 5016 else if (C.UsedInSequence.available(AArch64::SP)) { 5017 NumBytesNoStackCalls += 12; 5018 C.setCallInfo(MachineOutlinerDefault, 12); 5019 CandidatesWithoutStackFixups.push_back(C); 5020 } 5021 5022 // If we outline this, we need to modify the stack. Pretend we don't 5023 // outline this by saving all of its bytes. 5024 else { 5025 NumBytesNoStackCalls += SequenceSize; 5026 } 5027 } 5028 5029 // If there are no places where we have to save LR, then note that we 5030 // don't have to update the stack. Otherwise, give every candidate the 5031 // default call type, as long as it's safe to do so. 5032 if (!AllStackInstrsSafe || 5033 NumBytesNoStackCalls <= RepeatedSequenceLocs.size() * 12) { 5034 RepeatedSequenceLocs = CandidatesWithoutStackFixups; 5035 FrameID = MachineOutlinerNoLRSave; 5036 } else { 5037 SetCandidateCallInfo(MachineOutlinerDefault, 12); 5038 } 5039 5040 // If we dropped all of the candidates, bail out here. 5041 if (RepeatedSequenceLocs.size() < 2) { 5042 RepeatedSequenceLocs.clear(); 5043 return outliner::OutlinedFunction(); 5044 } 5045 } 5046 5047 // Does every candidate's MBB contain a call? If so, then we might have a call 5048 // in the range. 5049 if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) { 5050 // Check if the range contains a call. These require a save + restore of the 5051 // link register. 5052 bool ModStackToSaveLR = false; 5053 if (std::any_of(FirstCand.front(), FirstCand.back(), 5054 [](const MachineInstr &MI) { return MI.isCall(); })) 5055 ModStackToSaveLR = true; 5056 5057 // Handle the last instruction separately. If this is a tail call, then the 5058 // last instruction is a call. We don't want to save + restore in this case. 5059 // However, it could be possible that the last instruction is a call without 5060 // it being valid to tail call this sequence. We should consider this as 5061 // well. 5062 else if (FrameID != MachineOutlinerThunk && 5063 FrameID != MachineOutlinerTailCall && FirstCand.back()->isCall()) 5064 ModStackToSaveLR = true; 5065 5066 if (ModStackToSaveLR) { 5067 // We can't fix up the stack. Bail out. 5068 if (!AllStackInstrsSafe) { 5069 RepeatedSequenceLocs.clear(); 5070 return outliner::OutlinedFunction(); 5071 } 5072 5073 // Save + restore LR. 5074 NumBytesToCreateFrame += 8; 5075 } 5076 } 5077 5078 return outliner::OutlinedFunction(RepeatedSequenceLocs, SequenceSize, 5079 NumBytesToCreateFrame, FrameID); 5080 } 5081 5082 bool AArch64InstrInfo::isFunctionSafeToOutlineFrom( 5083 MachineFunction &MF, bool OutlineFromLinkOnceODRs) const { 5084 const Function &F = MF.getFunction(); 5085 5086 // Can F be deduplicated by the linker? If it can, don't outline from it. 5087 if (!OutlineFromLinkOnceODRs && F.hasLinkOnceODRLinkage()) 5088 return false; 5089 5090 // Don't outline from functions with section markings; the program could 5091 // expect that all the code is in the named section. 5092 // FIXME: Allow outlining from multiple functions with the same section 5093 // marking. 5094 if (F.hasSection()) 5095 return false; 5096 5097 // Outlining from functions with redzones is unsafe since the outliner may 5098 // modify the stack. Check if hasRedZone is true or unknown; if yes, don't 5099 // outline from it. 5100 AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 5101 if (!AFI || AFI->hasRedZone().getValueOr(true)) 5102 return false; 5103 5104 // It's safe to outline from MF. 5105 return true; 5106 } 5107 5108 bool AArch64InstrInfo::isMBBSafeToOutlineFrom(MachineBasicBlock &MBB, 5109 unsigned &Flags) const { 5110 // Check if LR is available through all of the MBB. If it's not, then set 5111 // a flag. 5112 assert(MBB.getParent()->getRegInfo().tracksLiveness() && 5113 "Suitable Machine Function for outlining must track liveness"); 5114 LiveRegUnits LRU(getRegisterInfo()); 5115 5116 std::for_each(MBB.rbegin(), MBB.rend(), 5117 [&LRU](MachineInstr &MI) { LRU.accumulate(MI); }); 5118 5119 // Check if each of the unsafe registers are available... 5120 bool W16AvailableInBlock = LRU.available(AArch64::W16); 5121 bool W17AvailableInBlock = LRU.available(AArch64::W17); 5122 bool NZCVAvailableInBlock = LRU.available(AArch64::NZCV); 5123 5124 // If all of these are dead (and not live out), we know we don't have to check 5125 // them later. 5126 if (W16AvailableInBlock && W17AvailableInBlock && NZCVAvailableInBlock) 5127 Flags |= MachineOutlinerMBBFlags::UnsafeRegsDead; 5128 5129 // Now, add the live outs to the set. 5130 LRU.addLiveOuts(MBB); 5131 5132 // If any of these registers is available in the MBB, but also a live out of 5133 // the block, then we know outlining is unsafe. 5134 if (W16AvailableInBlock && !LRU.available(AArch64::W16)) 5135 return false; 5136 if (W17AvailableInBlock && !LRU.available(AArch64::W17)) 5137 return false; 5138 if (NZCVAvailableInBlock && !LRU.available(AArch64::NZCV)) 5139 return false; 5140 5141 // Check if there's a call inside this MachineBasicBlock. If there is, then 5142 // set a flag. 5143 if (any_of(MBB, [](MachineInstr &MI) { return MI.isCall(); })) 5144 Flags |= MachineOutlinerMBBFlags::HasCalls; 5145 5146 MachineFunction *MF = MBB.getParent(); 5147 5148 // In the event that we outline, we may have to save LR. If there is an 5149 // available register in the MBB, then we'll always save LR there. Check if 5150 // this is true. 5151 bool CanSaveLR = false; 5152 const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>( 5153 MF->getSubtarget().getRegisterInfo()); 5154 5155 // Check if there is an available register across the sequence that we can 5156 // use. 5157 for (unsigned Reg : AArch64::GPR64RegClass) { 5158 if (!ARI->isReservedReg(*MF, Reg) && Reg != AArch64::LR && 5159 Reg != AArch64::X16 && Reg != AArch64::X17 && LRU.available(Reg)) { 5160 CanSaveLR = true; 5161 break; 5162 } 5163 } 5164 5165 // Check if we have a register we can save LR to, and if LR was used 5166 // somewhere. If both of those things are true, then we need to evaluate the 5167 // safety of outlining stack instructions later. 5168 if (!CanSaveLR && !LRU.available(AArch64::LR)) 5169 Flags |= MachineOutlinerMBBFlags::LRUnavailableSomewhere; 5170 5171 return true; 5172 } 5173 5174 outliner::InstrType 5175 AArch64InstrInfo::getOutliningType(MachineBasicBlock::iterator &MIT, 5176 unsigned Flags) const { 5177 MachineInstr &MI = *MIT; 5178 MachineBasicBlock *MBB = MI.getParent(); 5179 MachineFunction *MF = MBB->getParent(); 5180 AArch64FunctionInfo *FuncInfo = MF->getInfo<AArch64FunctionInfo>(); 5181 5182 // Don't outline LOHs. 5183 if (FuncInfo->getLOHRelated().count(&MI)) 5184 return outliner::InstrType::Illegal; 5185 5186 // Don't allow debug values to impact outlining type. 5187 if (MI.isDebugInstr() || MI.isIndirectDebugValue()) 5188 return outliner::InstrType::Invisible; 5189 5190 // At this point, KILL instructions don't really tell us much so we can go 5191 // ahead and skip over them. 5192 if (MI.isKill()) 5193 return outliner::InstrType::Invisible; 5194 5195 // Is this a terminator for a basic block? 5196 if (MI.isTerminator()) { 5197 5198 // Is this the end of a function? 5199 if (MI.getParent()->succ_empty()) 5200 return outliner::InstrType::Legal; 5201 5202 // It's not, so don't outline it. 5203 return outliner::InstrType::Illegal; 5204 } 5205 5206 // Make sure none of the operands are un-outlinable. 5207 for (const MachineOperand &MOP : MI.operands()) { 5208 if (MOP.isCPI() || MOP.isJTI() || MOP.isCFIIndex() || MOP.isFI() || 5209 MOP.isTargetIndex()) 5210 return outliner::InstrType::Illegal; 5211 5212 // If it uses LR or W30 explicitly, then don't touch it. 5213 if (MOP.isReg() && !MOP.isImplicit() && 5214 (MOP.getReg() == AArch64::LR || MOP.getReg() == AArch64::W30)) 5215 return outliner::InstrType::Illegal; 5216 } 5217 5218 // Special cases for instructions that can always be outlined, but will fail 5219 // the later tests. e.g, ADRPs, which are PC-relative use LR, but can always 5220 // be outlined because they don't require a *specific* value to be in LR. 5221 if (MI.getOpcode() == AArch64::ADRP) 5222 return outliner::InstrType::Legal; 5223 5224 // If MI is a call we might be able to outline it. We don't want to outline 5225 // any calls that rely on the position of items on the stack. When we outline 5226 // something containing a call, we have to emit a save and restore of LR in 5227 // the outlined function. Currently, this always happens by saving LR to the 5228 // stack. Thus, if we outline, say, half the parameters for a function call 5229 // plus the call, then we'll break the callee's expectations for the layout 5230 // of the stack. 5231 // 5232 // FIXME: Allow calls to functions which construct a stack frame, as long 5233 // as they don't access arguments on the stack. 5234 // FIXME: Figure out some way to analyze functions defined in other modules. 5235 // We should be able to compute the memory usage based on the IR calling 5236 // convention, even if we can't see the definition. 5237 if (MI.isCall()) { 5238 // Get the function associated with the call. Look at each operand and find 5239 // the one that represents the callee and get its name. 5240 const Function *Callee = nullptr; 5241 for (const MachineOperand &MOP : MI.operands()) { 5242 if (MOP.isGlobal()) { 5243 Callee = dyn_cast<Function>(MOP.getGlobal()); 5244 break; 5245 } 5246 } 5247 5248 // Never outline calls to mcount. There isn't any rule that would require 5249 // this, but the Linux kernel's "ftrace" feature depends on it. 5250 if (Callee && Callee->getName() == "\01_mcount") 5251 return outliner::InstrType::Illegal; 5252 5253 // If we don't know anything about the callee, assume it depends on the 5254 // stack layout of the caller. In that case, it's only legal to outline 5255 // as a tail-call. Whitelist the call instructions we know about so we 5256 // don't get unexpected results with call pseudo-instructions. 5257 auto UnknownCallOutlineType = outliner::InstrType::Illegal; 5258 if (MI.getOpcode() == AArch64::BLR || MI.getOpcode() == AArch64::BL) 5259 UnknownCallOutlineType = outliner::InstrType::LegalTerminator; 5260 5261 if (!Callee) 5262 return UnknownCallOutlineType; 5263 5264 // We have a function we have information about. Check it if it's something 5265 // can safely outline. 5266 MachineFunction *CalleeMF = MF->getMMI().getMachineFunction(*Callee); 5267 5268 // We don't know what's going on with the callee at all. Don't touch it. 5269 if (!CalleeMF) 5270 return UnknownCallOutlineType; 5271 5272 // Check if we know anything about the callee saves on the function. If we 5273 // don't, then don't touch it, since that implies that we haven't 5274 // computed anything about its stack frame yet. 5275 MachineFrameInfo &MFI = CalleeMF->getFrameInfo(); 5276 if (!MFI.isCalleeSavedInfoValid() || MFI.getStackSize() > 0 || 5277 MFI.getNumObjects() > 0) 5278 return UnknownCallOutlineType; 5279 5280 // At this point, we can say that CalleeMF ought to not pass anything on the 5281 // stack. Therefore, we can outline it. 5282 return outliner::InstrType::Legal; 5283 } 5284 5285 // Don't outline positions. 5286 if (MI.isPosition()) 5287 return outliner::InstrType::Illegal; 5288 5289 // Don't touch the link register or W30. 5290 if (MI.readsRegister(AArch64::W30, &getRegisterInfo()) || 5291 MI.modifiesRegister(AArch64::W30, &getRegisterInfo())) 5292 return outliner::InstrType::Illegal; 5293 5294 // Don't outline BTI instructions, because that will prevent the outlining 5295 // site from being indirectly callable. 5296 if (MI.getOpcode() == AArch64::HINT) { 5297 int64_t Imm = MI.getOperand(0).getImm(); 5298 if (Imm == 32 || Imm == 34 || Imm == 36 || Imm == 38) 5299 return outliner::InstrType::Illegal; 5300 } 5301 5302 return outliner::InstrType::Legal; 5303 } 5304 5305 void AArch64InstrInfo::fixupPostOutline(MachineBasicBlock &MBB) const { 5306 for (MachineInstr &MI : MBB) { 5307 const MachineOperand *Base; 5308 unsigned Width; 5309 int64_t Offset; 5310 5311 // Is this a load or store with an immediate offset with SP as the base? 5312 if (!MI.mayLoadOrStore() || 5313 !getMemOperandWithOffsetWidth(MI, Base, Offset, Width, &RI) || 5314 (Base->isReg() && Base->getReg() != AArch64::SP)) 5315 continue; 5316 5317 // It is, so we have to fix it up. 5318 unsigned Scale; 5319 int64_t Dummy1, Dummy2; 5320 5321 MachineOperand &StackOffsetOperand = getMemOpBaseRegImmOfsOffsetOperand(MI); 5322 assert(StackOffsetOperand.isImm() && "Stack offset wasn't immediate!"); 5323 getMemOpInfo(MI.getOpcode(), Scale, Width, Dummy1, Dummy2); 5324 assert(Scale != 0 && "Unexpected opcode!"); 5325 5326 // We've pushed the return address to the stack, so add 16 to the offset. 5327 // This is safe, since we already checked if it would overflow when we 5328 // checked if this instruction was legal to outline. 5329 int64_t NewImm = (Offset + 16) / Scale; 5330 StackOffsetOperand.setImm(NewImm); 5331 } 5332 } 5333 5334 void AArch64InstrInfo::buildOutlinedFrame( 5335 MachineBasicBlock &MBB, MachineFunction &MF, 5336 const outliner::OutlinedFunction &OF) const { 5337 // For thunk outlining, rewrite the last instruction from a call to a 5338 // tail-call. 5339 if (OF.FrameConstructionID == MachineOutlinerThunk) { 5340 MachineInstr *Call = &*--MBB.instr_end(); 5341 unsigned TailOpcode; 5342 if (Call->getOpcode() == AArch64::BL) { 5343 TailOpcode = AArch64::TCRETURNdi; 5344 } else { 5345 assert(Call->getOpcode() == AArch64::BLR); 5346 TailOpcode = AArch64::TCRETURNriALL; 5347 } 5348 MachineInstr *TC = BuildMI(MF, DebugLoc(), get(TailOpcode)) 5349 .add(Call->getOperand(0)) 5350 .addImm(0); 5351 MBB.insert(MBB.end(), TC); 5352 Call->eraseFromParent(); 5353 } 5354 5355 // Is there a call in the outlined range? 5356 auto IsNonTailCall = [](MachineInstr &MI) { 5357 return MI.isCall() && !MI.isReturn(); 5358 }; 5359 if (std::any_of(MBB.instr_begin(), MBB.instr_end(), IsNonTailCall)) { 5360 // Fix up the instructions in the range, since we're going to modify the 5361 // stack. 5362 assert(OF.FrameConstructionID != MachineOutlinerDefault && 5363 "Can only fix up stack references once"); 5364 fixupPostOutline(MBB); 5365 5366 // LR has to be a live in so that we can save it. 5367 MBB.addLiveIn(AArch64::LR); 5368 5369 MachineBasicBlock::iterator It = MBB.begin(); 5370 MachineBasicBlock::iterator Et = MBB.end(); 5371 5372 if (OF.FrameConstructionID == MachineOutlinerTailCall || 5373 OF.FrameConstructionID == MachineOutlinerThunk) 5374 Et = std::prev(MBB.end()); 5375 5376 // Insert a save before the outlined region 5377 MachineInstr *STRXpre = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre)) 5378 .addReg(AArch64::SP, RegState::Define) 5379 .addReg(AArch64::LR) 5380 .addReg(AArch64::SP) 5381 .addImm(-16); 5382 It = MBB.insert(It, STRXpre); 5383 5384 const TargetSubtargetInfo &STI = MF.getSubtarget(); 5385 const MCRegisterInfo *MRI = STI.getRegisterInfo(); 5386 unsigned DwarfReg = MRI->getDwarfRegNum(AArch64::LR, true); 5387 5388 // Add a CFI saying the stack was moved 16 B down. 5389 int64_t StackPosEntry = 5390 MF.addFrameInst(MCCFIInstruction::createDefCfaOffset(nullptr, 16)); 5391 BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION)) 5392 .addCFIIndex(StackPosEntry) 5393 .setMIFlags(MachineInstr::FrameSetup); 5394 5395 // Add a CFI saying that the LR that we want to find is now 16 B higher than 5396 // before. 5397 int64_t LRPosEntry = 5398 MF.addFrameInst(MCCFIInstruction::createOffset(nullptr, DwarfReg, 16)); 5399 BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION)) 5400 .addCFIIndex(LRPosEntry) 5401 .setMIFlags(MachineInstr::FrameSetup); 5402 5403 // Insert a restore before the terminator for the function. 5404 MachineInstr *LDRXpost = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost)) 5405 .addReg(AArch64::SP, RegState::Define) 5406 .addReg(AArch64::LR, RegState::Define) 5407 .addReg(AArch64::SP) 5408 .addImm(16); 5409 Et = MBB.insert(Et, LDRXpost); 5410 } 5411 5412 // If this is a tail call outlined function, then there's already a return. 5413 if (OF.FrameConstructionID == MachineOutlinerTailCall || 5414 OF.FrameConstructionID == MachineOutlinerThunk) 5415 return; 5416 5417 // It's not a tail call, so we have to insert the return ourselves. 5418 MachineInstr *ret = BuildMI(MF, DebugLoc(), get(AArch64::RET)) 5419 .addReg(AArch64::LR, RegState::Undef); 5420 MBB.insert(MBB.end(), ret); 5421 5422 // Did we have to modify the stack by saving the link register? 5423 if (OF.FrameConstructionID != MachineOutlinerDefault) 5424 return; 5425 5426 // We modified the stack. 5427 // Walk over the basic block and fix up all the stack accesses. 5428 fixupPostOutline(MBB); 5429 } 5430 5431 MachineBasicBlock::iterator AArch64InstrInfo::insertOutlinedCall( 5432 Module &M, MachineBasicBlock &MBB, MachineBasicBlock::iterator &It, 5433 MachineFunction &MF, const outliner::Candidate &C) const { 5434 5435 // Are we tail calling? 5436 if (C.CallConstructionID == MachineOutlinerTailCall) { 5437 // If yes, then we can just branch to the label. 5438 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::TCRETURNdi)) 5439 .addGlobalAddress(M.getNamedValue(MF.getName())) 5440 .addImm(0)); 5441 return It; 5442 } 5443 5444 // Are we saving the link register? 5445 if (C.CallConstructionID == MachineOutlinerNoLRSave || 5446 C.CallConstructionID == MachineOutlinerThunk) { 5447 // No, so just insert the call. 5448 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL)) 5449 .addGlobalAddress(M.getNamedValue(MF.getName()))); 5450 return It; 5451 } 5452 5453 // We want to return the spot where we inserted the call. 5454 MachineBasicBlock::iterator CallPt; 5455 5456 // Instructions for saving and restoring LR around the call instruction we're 5457 // going to insert. 5458 MachineInstr *Save; 5459 MachineInstr *Restore; 5460 // Can we save to a register? 5461 if (C.CallConstructionID == MachineOutlinerRegSave) { 5462 // FIXME: This logic should be sunk into a target-specific interface so that 5463 // we don't have to recompute the register. 5464 unsigned Reg = findRegisterToSaveLRTo(C); 5465 assert(Reg != 0 && "No callee-saved register available?"); 5466 5467 // Save and restore LR from that register. 5468 Save = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), Reg) 5469 .addReg(AArch64::XZR) 5470 .addReg(AArch64::LR) 5471 .addImm(0); 5472 Restore = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), AArch64::LR) 5473 .addReg(AArch64::XZR) 5474 .addReg(Reg) 5475 .addImm(0); 5476 } else { 5477 // We have the default case. Save and restore from SP. 5478 Save = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre)) 5479 .addReg(AArch64::SP, RegState::Define) 5480 .addReg(AArch64::LR) 5481 .addReg(AArch64::SP) 5482 .addImm(-16); 5483 Restore = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost)) 5484 .addReg(AArch64::SP, RegState::Define) 5485 .addReg(AArch64::LR, RegState::Define) 5486 .addReg(AArch64::SP) 5487 .addImm(16); 5488 } 5489 5490 It = MBB.insert(It, Save); 5491 It++; 5492 5493 // Insert the call. 5494 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL)) 5495 .addGlobalAddress(M.getNamedValue(MF.getName()))); 5496 CallPt = It; 5497 It++; 5498 5499 It = MBB.insert(It, Restore); 5500 return CallPt; 5501 } 5502 5503 bool AArch64InstrInfo::shouldOutlineFromFunctionByDefault( 5504 MachineFunction &MF) const { 5505 return MF.getFunction().hasMinSize(); 5506 } 5507 5508 #define GET_INSTRINFO_HELPERS 5509 #include "AArch64GenInstrInfo.inc" 5510