1 //===- AArch64InstrInfo.cpp - AArch64 Instruction Information -------------===// 2 // 3 // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. 4 // See https://llvm.org/LICENSE.txt for license information. 5 // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception 6 // 7 //===----------------------------------------------------------------------===// 8 // 9 // This file contains the AArch64 implementation of the TargetInstrInfo class. 10 // 11 //===----------------------------------------------------------------------===// 12 13 #include "AArch64InstrInfo.h" 14 #include "AArch64MachineFunctionInfo.h" 15 #include "AArch64Subtarget.h" 16 #include "MCTargetDesc/AArch64AddressingModes.h" 17 #include "Utils/AArch64BaseInfo.h" 18 #include "llvm/ADT/ArrayRef.h" 19 #include "llvm/ADT/STLExtras.h" 20 #include "llvm/ADT/SmallVector.h" 21 #include "llvm/CodeGen/MachineBasicBlock.h" 22 #include "llvm/CodeGen/MachineFrameInfo.h" 23 #include "llvm/CodeGen/MachineFunction.h" 24 #include "llvm/CodeGen/MachineInstr.h" 25 #include "llvm/CodeGen/MachineInstrBuilder.h" 26 #include "llvm/CodeGen/MachineMemOperand.h" 27 #include "llvm/CodeGen/MachineOperand.h" 28 #include "llvm/CodeGen/MachineRegisterInfo.h" 29 #include "llvm/CodeGen/MachineModuleInfo.h" 30 #include "llvm/CodeGen/StackMaps.h" 31 #include "llvm/CodeGen/TargetRegisterInfo.h" 32 #include "llvm/CodeGen/TargetSubtargetInfo.h" 33 #include "llvm/IR/DebugLoc.h" 34 #include "llvm/IR/GlobalValue.h" 35 #include "llvm/MC/MCAsmInfo.h" 36 #include "llvm/MC/MCInst.h" 37 #include "llvm/MC/MCInstrDesc.h" 38 #include "llvm/Support/Casting.h" 39 #include "llvm/Support/CodeGen.h" 40 #include "llvm/Support/CommandLine.h" 41 #include "llvm/Support/Compiler.h" 42 #include "llvm/Support/ErrorHandling.h" 43 #include "llvm/Support/MathExtras.h" 44 #include "llvm/Target/TargetMachine.h" 45 #include "llvm/Target/TargetOptions.h" 46 #include <cassert> 47 #include <cstdint> 48 #include <iterator> 49 #include <utility> 50 51 using namespace llvm; 52 53 #define GET_INSTRINFO_CTOR_DTOR 54 #include "AArch64GenInstrInfo.inc" 55 56 static cl::opt<unsigned> TBZDisplacementBits( 57 "aarch64-tbz-offset-bits", cl::Hidden, cl::init(14), 58 cl::desc("Restrict range of TB[N]Z instructions (DEBUG)")); 59 60 static cl::opt<unsigned> CBZDisplacementBits( 61 "aarch64-cbz-offset-bits", cl::Hidden, cl::init(19), 62 cl::desc("Restrict range of CB[N]Z instructions (DEBUG)")); 63 64 static cl::opt<unsigned> 65 BCCDisplacementBits("aarch64-bcc-offset-bits", cl::Hidden, cl::init(19), 66 cl::desc("Restrict range of Bcc instructions (DEBUG)")); 67 68 AArch64InstrInfo::AArch64InstrInfo(const AArch64Subtarget &STI) 69 : AArch64GenInstrInfo(AArch64::ADJCALLSTACKDOWN, AArch64::ADJCALLSTACKUP, 70 AArch64::CATCHRET), 71 RI(STI.getTargetTriple()), Subtarget(STI) {} 72 73 /// GetInstSize - Return the number of bytes of code the specified 74 /// instruction may be. This returns the maximum number of bytes. 75 unsigned AArch64InstrInfo::getInstSizeInBytes(const MachineInstr &MI) const { 76 const MachineBasicBlock &MBB = *MI.getParent(); 77 const MachineFunction *MF = MBB.getParent(); 78 const MCAsmInfo *MAI = MF->getTarget().getMCAsmInfo(); 79 80 { 81 auto Op = MI.getOpcode(); 82 if (Op == AArch64::INLINEASM || Op == AArch64::INLINEASM_BR) 83 return getInlineAsmLength(MI.getOperand(0).getSymbolName(), *MAI); 84 } 85 86 // Meta-instructions emit no code. 87 if (MI.isMetaInstruction()) 88 return 0; 89 90 // FIXME: We currently only handle pseudoinstructions that don't get expanded 91 // before the assembly printer. 92 unsigned NumBytes = 0; 93 const MCInstrDesc &Desc = MI.getDesc(); 94 switch (Desc.getOpcode()) { 95 default: 96 // Anything not explicitly designated otherwise is a normal 4-byte insn. 97 NumBytes = 4; 98 break; 99 case TargetOpcode::STACKMAP: 100 // The upper bound for a stackmap intrinsic is the full length of its shadow 101 NumBytes = StackMapOpers(&MI).getNumPatchBytes(); 102 assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!"); 103 break; 104 case TargetOpcode::PATCHPOINT: 105 // The size of the patchpoint intrinsic is the number of bytes requested 106 NumBytes = PatchPointOpers(&MI).getNumPatchBytes(); 107 assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!"); 108 break; 109 case AArch64::TLSDESC_CALLSEQ: 110 // This gets lowered to an instruction sequence which takes 16 bytes 111 NumBytes = 16; 112 break; 113 case AArch64::JumpTableDest32: 114 case AArch64::JumpTableDest16: 115 case AArch64::JumpTableDest8: 116 NumBytes = 12; 117 break; 118 case AArch64::SPACE: 119 NumBytes = MI.getOperand(1).getImm(); 120 break; 121 } 122 123 return NumBytes; 124 } 125 126 static void parseCondBranch(MachineInstr *LastInst, MachineBasicBlock *&Target, 127 SmallVectorImpl<MachineOperand> &Cond) { 128 // Block ends with fall-through condbranch. 129 switch (LastInst->getOpcode()) { 130 default: 131 llvm_unreachable("Unknown branch instruction?"); 132 case AArch64::Bcc: 133 Target = LastInst->getOperand(1).getMBB(); 134 Cond.push_back(LastInst->getOperand(0)); 135 break; 136 case AArch64::CBZW: 137 case AArch64::CBZX: 138 case AArch64::CBNZW: 139 case AArch64::CBNZX: 140 Target = LastInst->getOperand(1).getMBB(); 141 Cond.push_back(MachineOperand::CreateImm(-1)); 142 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode())); 143 Cond.push_back(LastInst->getOperand(0)); 144 break; 145 case AArch64::TBZW: 146 case AArch64::TBZX: 147 case AArch64::TBNZW: 148 case AArch64::TBNZX: 149 Target = LastInst->getOperand(2).getMBB(); 150 Cond.push_back(MachineOperand::CreateImm(-1)); 151 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode())); 152 Cond.push_back(LastInst->getOperand(0)); 153 Cond.push_back(LastInst->getOperand(1)); 154 } 155 } 156 157 static unsigned getBranchDisplacementBits(unsigned Opc) { 158 switch (Opc) { 159 default: 160 llvm_unreachable("unexpected opcode!"); 161 case AArch64::B: 162 return 64; 163 case AArch64::TBNZW: 164 case AArch64::TBZW: 165 case AArch64::TBNZX: 166 case AArch64::TBZX: 167 return TBZDisplacementBits; 168 case AArch64::CBNZW: 169 case AArch64::CBZW: 170 case AArch64::CBNZX: 171 case AArch64::CBZX: 172 return CBZDisplacementBits; 173 case AArch64::Bcc: 174 return BCCDisplacementBits; 175 } 176 } 177 178 bool AArch64InstrInfo::isBranchOffsetInRange(unsigned BranchOp, 179 int64_t BrOffset) const { 180 unsigned Bits = getBranchDisplacementBits(BranchOp); 181 assert(Bits >= 3 && "max branch displacement must be enough to jump" 182 "over conditional branch expansion"); 183 return isIntN(Bits, BrOffset / 4); 184 } 185 186 MachineBasicBlock * 187 AArch64InstrInfo::getBranchDestBlock(const MachineInstr &MI) const { 188 switch (MI.getOpcode()) { 189 default: 190 llvm_unreachable("unexpected opcode!"); 191 case AArch64::B: 192 return MI.getOperand(0).getMBB(); 193 case AArch64::TBZW: 194 case AArch64::TBNZW: 195 case AArch64::TBZX: 196 case AArch64::TBNZX: 197 return MI.getOperand(2).getMBB(); 198 case AArch64::CBZW: 199 case AArch64::CBNZW: 200 case AArch64::CBZX: 201 case AArch64::CBNZX: 202 case AArch64::Bcc: 203 return MI.getOperand(1).getMBB(); 204 } 205 } 206 207 // Branch analysis. 208 bool AArch64InstrInfo::analyzeBranch(MachineBasicBlock &MBB, 209 MachineBasicBlock *&TBB, 210 MachineBasicBlock *&FBB, 211 SmallVectorImpl<MachineOperand> &Cond, 212 bool AllowModify) const { 213 // If the block has no terminators, it just falls into the block after it. 214 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr(); 215 if (I == MBB.end()) 216 return false; 217 218 if (!isUnpredicatedTerminator(*I)) 219 return false; 220 221 // Get the last instruction in the block. 222 MachineInstr *LastInst = &*I; 223 224 // If there is only one terminator instruction, process it. 225 unsigned LastOpc = LastInst->getOpcode(); 226 if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) { 227 if (isUncondBranchOpcode(LastOpc)) { 228 TBB = LastInst->getOperand(0).getMBB(); 229 return false; 230 } 231 if (isCondBranchOpcode(LastOpc)) { 232 // Block ends with fall-through condbranch. 233 parseCondBranch(LastInst, TBB, Cond); 234 return false; 235 } 236 return true; // Can't handle indirect branch. 237 } 238 239 // Get the instruction before it if it is a terminator. 240 MachineInstr *SecondLastInst = &*I; 241 unsigned SecondLastOpc = SecondLastInst->getOpcode(); 242 243 // If AllowModify is true and the block ends with two or more unconditional 244 // branches, delete all but the first unconditional branch. 245 if (AllowModify && isUncondBranchOpcode(LastOpc)) { 246 while (isUncondBranchOpcode(SecondLastOpc)) { 247 LastInst->eraseFromParent(); 248 LastInst = SecondLastInst; 249 LastOpc = LastInst->getOpcode(); 250 if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) { 251 // Return now the only terminator is an unconditional branch. 252 TBB = LastInst->getOperand(0).getMBB(); 253 return false; 254 } else { 255 SecondLastInst = &*I; 256 SecondLastOpc = SecondLastInst->getOpcode(); 257 } 258 } 259 } 260 261 // If there are three terminators, we don't know what sort of block this is. 262 if (SecondLastInst && I != MBB.begin() && isUnpredicatedTerminator(*--I)) 263 return true; 264 265 // If the block ends with a B and a Bcc, handle it. 266 if (isCondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) { 267 parseCondBranch(SecondLastInst, TBB, Cond); 268 FBB = LastInst->getOperand(0).getMBB(); 269 return false; 270 } 271 272 // If the block ends with two unconditional branches, handle it. The second 273 // one is not executed, so remove it. 274 if (isUncondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) { 275 TBB = SecondLastInst->getOperand(0).getMBB(); 276 I = LastInst; 277 if (AllowModify) 278 I->eraseFromParent(); 279 return false; 280 } 281 282 // ...likewise if it ends with an indirect branch followed by an unconditional 283 // branch. 284 if (isIndirectBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) { 285 I = LastInst; 286 if (AllowModify) 287 I->eraseFromParent(); 288 return true; 289 } 290 291 // Otherwise, can't handle this. 292 return true; 293 } 294 295 bool AArch64InstrInfo::reverseBranchCondition( 296 SmallVectorImpl<MachineOperand> &Cond) const { 297 if (Cond[0].getImm() != -1) { 298 // Regular Bcc 299 AArch64CC::CondCode CC = (AArch64CC::CondCode)(int)Cond[0].getImm(); 300 Cond[0].setImm(AArch64CC::getInvertedCondCode(CC)); 301 } else { 302 // Folded compare-and-branch 303 switch (Cond[1].getImm()) { 304 default: 305 llvm_unreachable("Unknown conditional branch!"); 306 case AArch64::CBZW: 307 Cond[1].setImm(AArch64::CBNZW); 308 break; 309 case AArch64::CBNZW: 310 Cond[1].setImm(AArch64::CBZW); 311 break; 312 case AArch64::CBZX: 313 Cond[1].setImm(AArch64::CBNZX); 314 break; 315 case AArch64::CBNZX: 316 Cond[1].setImm(AArch64::CBZX); 317 break; 318 case AArch64::TBZW: 319 Cond[1].setImm(AArch64::TBNZW); 320 break; 321 case AArch64::TBNZW: 322 Cond[1].setImm(AArch64::TBZW); 323 break; 324 case AArch64::TBZX: 325 Cond[1].setImm(AArch64::TBNZX); 326 break; 327 case AArch64::TBNZX: 328 Cond[1].setImm(AArch64::TBZX); 329 break; 330 } 331 } 332 333 return false; 334 } 335 336 unsigned AArch64InstrInfo::removeBranch(MachineBasicBlock &MBB, 337 int *BytesRemoved) const { 338 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr(); 339 if (I == MBB.end()) 340 return 0; 341 342 if (!isUncondBranchOpcode(I->getOpcode()) && 343 !isCondBranchOpcode(I->getOpcode())) 344 return 0; 345 346 // Remove the branch. 347 I->eraseFromParent(); 348 349 I = MBB.end(); 350 351 if (I == MBB.begin()) { 352 if (BytesRemoved) 353 *BytesRemoved = 4; 354 return 1; 355 } 356 --I; 357 if (!isCondBranchOpcode(I->getOpcode())) { 358 if (BytesRemoved) 359 *BytesRemoved = 4; 360 return 1; 361 } 362 363 // Remove the branch. 364 I->eraseFromParent(); 365 if (BytesRemoved) 366 *BytesRemoved = 8; 367 368 return 2; 369 } 370 371 void AArch64InstrInfo::instantiateCondBranch( 372 MachineBasicBlock &MBB, const DebugLoc &DL, MachineBasicBlock *TBB, 373 ArrayRef<MachineOperand> Cond) const { 374 if (Cond[0].getImm() != -1) { 375 // Regular Bcc 376 BuildMI(&MBB, DL, get(AArch64::Bcc)).addImm(Cond[0].getImm()).addMBB(TBB); 377 } else { 378 // Folded compare-and-branch 379 // Note that we use addOperand instead of addReg to keep the flags. 380 const MachineInstrBuilder MIB = 381 BuildMI(&MBB, DL, get(Cond[1].getImm())).add(Cond[2]); 382 if (Cond.size() > 3) 383 MIB.addImm(Cond[3].getImm()); 384 MIB.addMBB(TBB); 385 } 386 } 387 388 unsigned AArch64InstrInfo::insertBranch( 389 MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, 390 ArrayRef<MachineOperand> Cond, const DebugLoc &DL, int *BytesAdded) const { 391 // Shouldn't be a fall through. 392 assert(TBB && "insertBranch must not be told to insert a fallthrough"); 393 394 if (!FBB) { 395 if (Cond.empty()) // Unconditional branch? 396 BuildMI(&MBB, DL, get(AArch64::B)).addMBB(TBB); 397 else 398 instantiateCondBranch(MBB, DL, TBB, Cond); 399 400 if (BytesAdded) 401 *BytesAdded = 4; 402 403 return 1; 404 } 405 406 // Two-way conditional branch. 407 instantiateCondBranch(MBB, DL, TBB, Cond); 408 BuildMI(&MBB, DL, get(AArch64::B)).addMBB(FBB); 409 410 if (BytesAdded) 411 *BytesAdded = 8; 412 413 return 2; 414 } 415 416 // Find the original register that VReg is copied from. 417 static unsigned removeCopies(const MachineRegisterInfo &MRI, unsigned VReg) { 418 while (Register::isVirtualRegister(VReg)) { 419 const MachineInstr *DefMI = MRI.getVRegDef(VReg); 420 if (!DefMI->isFullCopy()) 421 return VReg; 422 VReg = DefMI->getOperand(1).getReg(); 423 } 424 return VReg; 425 } 426 427 // Determine if VReg is defined by an instruction that can be folded into a 428 // csel instruction. If so, return the folded opcode, and the replacement 429 // register. 430 static unsigned canFoldIntoCSel(const MachineRegisterInfo &MRI, unsigned VReg, 431 unsigned *NewVReg = nullptr) { 432 VReg = removeCopies(MRI, VReg); 433 if (!Register::isVirtualRegister(VReg)) 434 return 0; 435 436 bool Is64Bit = AArch64::GPR64allRegClass.hasSubClassEq(MRI.getRegClass(VReg)); 437 const MachineInstr *DefMI = MRI.getVRegDef(VReg); 438 unsigned Opc = 0; 439 unsigned SrcOpNum = 0; 440 switch (DefMI->getOpcode()) { 441 case AArch64::ADDSXri: 442 case AArch64::ADDSWri: 443 // if NZCV is used, do not fold. 444 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1) 445 return 0; 446 // fall-through to ADDXri and ADDWri. 447 LLVM_FALLTHROUGH; 448 case AArch64::ADDXri: 449 case AArch64::ADDWri: 450 // add x, 1 -> csinc. 451 if (!DefMI->getOperand(2).isImm() || DefMI->getOperand(2).getImm() != 1 || 452 DefMI->getOperand(3).getImm() != 0) 453 return 0; 454 SrcOpNum = 1; 455 Opc = Is64Bit ? AArch64::CSINCXr : AArch64::CSINCWr; 456 break; 457 458 case AArch64::ORNXrr: 459 case AArch64::ORNWrr: { 460 // not x -> csinv, represented as orn dst, xzr, src. 461 unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg()); 462 if (ZReg != AArch64::XZR && ZReg != AArch64::WZR) 463 return 0; 464 SrcOpNum = 2; 465 Opc = Is64Bit ? AArch64::CSINVXr : AArch64::CSINVWr; 466 break; 467 } 468 469 case AArch64::SUBSXrr: 470 case AArch64::SUBSWrr: 471 // if NZCV is used, do not fold. 472 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) == -1) 473 return 0; 474 // fall-through to SUBXrr and SUBWrr. 475 LLVM_FALLTHROUGH; 476 case AArch64::SUBXrr: 477 case AArch64::SUBWrr: { 478 // neg x -> csneg, represented as sub dst, xzr, src. 479 unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg()); 480 if (ZReg != AArch64::XZR && ZReg != AArch64::WZR) 481 return 0; 482 SrcOpNum = 2; 483 Opc = Is64Bit ? AArch64::CSNEGXr : AArch64::CSNEGWr; 484 break; 485 } 486 default: 487 return 0; 488 } 489 assert(Opc && SrcOpNum && "Missing parameters"); 490 491 if (NewVReg) 492 *NewVReg = DefMI->getOperand(SrcOpNum).getReg(); 493 return Opc; 494 } 495 496 bool AArch64InstrInfo::canInsertSelect(const MachineBasicBlock &MBB, 497 ArrayRef<MachineOperand> Cond, 498 unsigned TrueReg, unsigned FalseReg, 499 int &CondCycles, int &TrueCycles, 500 int &FalseCycles) const { 501 // Check register classes. 502 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 503 const TargetRegisterClass *RC = 504 RI.getCommonSubClass(MRI.getRegClass(TrueReg), MRI.getRegClass(FalseReg)); 505 if (!RC) 506 return false; 507 508 // Expanding cbz/tbz requires an extra cycle of latency on the condition. 509 unsigned ExtraCondLat = Cond.size() != 1; 510 511 // GPRs are handled by csel. 512 // FIXME: Fold in x+1, -x, and ~x when applicable. 513 if (AArch64::GPR64allRegClass.hasSubClassEq(RC) || 514 AArch64::GPR32allRegClass.hasSubClassEq(RC)) { 515 // Single-cycle csel, csinc, csinv, and csneg. 516 CondCycles = 1 + ExtraCondLat; 517 TrueCycles = FalseCycles = 1; 518 if (canFoldIntoCSel(MRI, TrueReg)) 519 TrueCycles = 0; 520 else if (canFoldIntoCSel(MRI, FalseReg)) 521 FalseCycles = 0; 522 return true; 523 } 524 525 // Scalar floating point is handled by fcsel. 526 // FIXME: Form fabs, fmin, and fmax when applicable. 527 if (AArch64::FPR64RegClass.hasSubClassEq(RC) || 528 AArch64::FPR32RegClass.hasSubClassEq(RC)) { 529 CondCycles = 5 + ExtraCondLat; 530 TrueCycles = FalseCycles = 2; 531 return true; 532 } 533 534 // Can't do vectors. 535 return false; 536 } 537 538 void AArch64InstrInfo::insertSelect(MachineBasicBlock &MBB, 539 MachineBasicBlock::iterator I, 540 const DebugLoc &DL, unsigned DstReg, 541 ArrayRef<MachineOperand> Cond, 542 unsigned TrueReg, unsigned FalseReg) const { 543 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 544 545 // Parse the condition code, see parseCondBranch() above. 546 AArch64CC::CondCode CC; 547 switch (Cond.size()) { 548 default: 549 llvm_unreachable("Unknown condition opcode in Cond"); 550 case 1: // b.cc 551 CC = AArch64CC::CondCode(Cond[0].getImm()); 552 break; 553 case 3: { // cbz/cbnz 554 // We must insert a compare against 0. 555 bool Is64Bit; 556 switch (Cond[1].getImm()) { 557 default: 558 llvm_unreachable("Unknown branch opcode in Cond"); 559 case AArch64::CBZW: 560 Is64Bit = false; 561 CC = AArch64CC::EQ; 562 break; 563 case AArch64::CBZX: 564 Is64Bit = true; 565 CC = AArch64CC::EQ; 566 break; 567 case AArch64::CBNZW: 568 Is64Bit = false; 569 CC = AArch64CC::NE; 570 break; 571 case AArch64::CBNZX: 572 Is64Bit = true; 573 CC = AArch64CC::NE; 574 break; 575 } 576 Register SrcReg = Cond[2].getReg(); 577 if (Is64Bit) { 578 // cmp reg, #0 is actually subs xzr, reg, #0. 579 MRI.constrainRegClass(SrcReg, &AArch64::GPR64spRegClass); 580 BuildMI(MBB, I, DL, get(AArch64::SUBSXri), AArch64::XZR) 581 .addReg(SrcReg) 582 .addImm(0) 583 .addImm(0); 584 } else { 585 MRI.constrainRegClass(SrcReg, &AArch64::GPR32spRegClass); 586 BuildMI(MBB, I, DL, get(AArch64::SUBSWri), AArch64::WZR) 587 .addReg(SrcReg) 588 .addImm(0) 589 .addImm(0); 590 } 591 break; 592 } 593 case 4: { // tbz/tbnz 594 // We must insert a tst instruction. 595 switch (Cond[1].getImm()) { 596 default: 597 llvm_unreachable("Unknown branch opcode in Cond"); 598 case AArch64::TBZW: 599 case AArch64::TBZX: 600 CC = AArch64CC::EQ; 601 break; 602 case AArch64::TBNZW: 603 case AArch64::TBNZX: 604 CC = AArch64CC::NE; 605 break; 606 } 607 // cmp reg, #foo is actually ands xzr, reg, #1<<foo. 608 if (Cond[1].getImm() == AArch64::TBZW || Cond[1].getImm() == AArch64::TBNZW) 609 BuildMI(MBB, I, DL, get(AArch64::ANDSWri), AArch64::WZR) 610 .addReg(Cond[2].getReg()) 611 .addImm( 612 AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 32)); 613 else 614 BuildMI(MBB, I, DL, get(AArch64::ANDSXri), AArch64::XZR) 615 .addReg(Cond[2].getReg()) 616 .addImm( 617 AArch64_AM::encodeLogicalImmediate(1ull << Cond[3].getImm(), 64)); 618 break; 619 } 620 } 621 622 unsigned Opc = 0; 623 const TargetRegisterClass *RC = nullptr; 624 bool TryFold = false; 625 if (MRI.constrainRegClass(DstReg, &AArch64::GPR64RegClass)) { 626 RC = &AArch64::GPR64RegClass; 627 Opc = AArch64::CSELXr; 628 TryFold = true; 629 } else if (MRI.constrainRegClass(DstReg, &AArch64::GPR32RegClass)) { 630 RC = &AArch64::GPR32RegClass; 631 Opc = AArch64::CSELWr; 632 TryFold = true; 633 } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR64RegClass)) { 634 RC = &AArch64::FPR64RegClass; 635 Opc = AArch64::FCSELDrrr; 636 } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR32RegClass)) { 637 RC = &AArch64::FPR32RegClass; 638 Opc = AArch64::FCSELSrrr; 639 } 640 assert(RC && "Unsupported regclass"); 641 642 // Try folding simple instructions into the csel. 643 if (TryFold) { 644 unsigned NewVReg = 0; 645 unsigned FoldedOpc = canFoldIntoCSel(MRI, TrueReg, &NewVReg); 646 if (FoldedOpc) { 647 // The folded opcodes csinc, csinc and csneg apply the operation to 648 // FalseReg, so we need to invert the condition. 649 CC = AArch64CC::getInvertedCondCode(CC); 650 TrueReg = FalseReg; 651 } else 652 FoldedOpc = canFoldIntoCSel(MRI, FalseReg, &NewVReg); 653 654 // Fold the operation. Leave any dead instructions for DCE to clean up. 655 if (FoldedOpc) { 656 FalseReg = NewVReg; 657 Opc = FoldedOpc; 658 // The extends the live range of NewVReg. 659 MRI.clearKillFlags(NewVReg); 660 } 661 } 662 663 // Pull all virtual register into the appropriate class. 664 MRI.constrainRegClass(TrueReg, RC); 665 MRI.constrainRegClass(FalseReg, RC); 666 667 // Insert the csel. 668 BuildMI(MBB, I, DL, get(Opc), DstReg) 669 .addReg(TrueReg) 670 .addReg(FalseReg) 671 .addImm(CC); 672 } 673 674 /// Returns true if a MOVi32imm or MOVi64imm can be expanded to an ORRxx. 675 static bool canBeExpandedToORR(const MachineInstr &MI, unsigned BitSize) { 676 uint64_t Imm = MI.getOperand(1).getImm(); 677 uint64_t UImm = Imm << (64 - BitSize) >> (64 - BitSize); 678 uint64_t Encoding; 679 return AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding); 680 } 681 682 // FIXME: this implementation should be micro-architecture dependent, so a 683 // micro-architecture target hook should be introduced here in future. 684 bool AArch64InstrInfo::isAsCheapAsAMove(const MachineInstr &MI) const { 685 if (!Subtarget.hasCustomCheapAsMoveHandling()) 686 return MI.isAsCheapAsAMove(); 687 688 const unsigned Opcode = MI.getOpcode(); 689 690 // Firstly, check cases gated by features. 691 692 if (Subtarget.hasZeroCycleZeroingFP()) { 693 if (Opcode == AArch64::FMOVH0 || 694 Opcode == AArch64::FMOVS0 || 695 Opcode == AArch64::FMOVD0) 696 return true; 697 } 698 699 if (Subtarget.hasZeroCycleZeroingGP()) { 700 if (Opcode == TargetOpcode::COPY && 701 (MI.getOperand(1).getReg() == AArch64::WZR || 702 MI.getOperand(1).getReg() == AArch64::XZR)) 703 return true; 704 } 705 706 // Secondly, check cases specific to sub-targets. 707 708 if (Subtarget.hasExynosCheapAsMoveHandling()) { 709 if (isExynosCheapAsMove(MI)) 710 return true; 711 712 return MI.isAsCheapAsAMove(); 713 } 714 715 // Finally, check generic cases. 716 717 switch (Opcode) { 718 default: 719 return false; 720 721 // add/sub on register without shift 722 case AArch64::ADDWri: 723 case AArch64::ADDXri: 724 case AArch64::SUBWri: 725 case AArch64::SUBXri: 726 return (MI.getOperand(3).getImm() == 0); 727 728 // logical ops on immediate 729 case AArch64::ANDWri: 730 case AArch64::ANDXri: 731 case AArch64::EORWri: 732 case AArch64::EORXri: 733 case AArch64::ORRWri: 734 case AArch64::ORRXri: 735 return true; 736 737 // logical ops on register without shift 738 case AArch64::ANDWrr: 739 case AArch64::ANDXrr: 740 case AArch64::BICWrr: 741 case AArch64::BICXrr: 742 case AArch64::EONWrr: 743 case AArch64::EONXrr: 744 case AArch64::EORWrr: 745 case AArch64::EORXrr: 746 case AArch64::ORNWrr: 747 case AArch64::ORNXrr: 748 case AArch64::ORRWrr: 749 case AArch64::ORRXrr: 750 return true; 751 752 // If MOVi32imm or MOVi64imm can be expanded into ORRWri or 753 // ORRXri, it is as cheap as MOV 754 case AArch64::MOVi32imm: 755 return canBeExpandedToORR(MI, 32); 756 case AArch64::MOVi64imm: 757 return canBeExpandedToORR(MI, 64); 758 } 759 760 llvm_unreachable("Unknown opcode to check as cheap as a move!"); 761 } 762 763 bool AArch64InstrInfo::isFalkorShiftExtFast(const MachineInstr &MI) { 764 switch (MI.getOpcode()) { 765 default: 766 return false; 767 768 case AArch64::ADDWrs: 769 case AArch64::ADDXrs: 770 case AArch64::ADDSWrs: 771 case AArch64::ADDSXrs: { 772 unsigned Imm = MI.getOperand(3).getImm(); 773 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm); 774 if (ShiftVal == 0) 775 return true; 776 return AArch64_AM::getShiftType(Imm) == AArch64_AM::LSL && ShiftVal <= 5; 777 } 778 779 case AArch64::ADDWrx: 780 case AArch64::ADDXrx: 781 case AArch64::ADDXrx64: 782 case AArch64::ADDSWrx: 783 case AArch64::ADDSXrx: 784 case AArch64::ADDSXrx64: { 785 unsigned Imm = MI.getOperand(3).getImm(); 786 switch (AArch64_AM::getArithExtendType(Imm)) { 787 default: 788 return false; 789 case AArch64_AM::UXTB: 790 case AArch64_AM::UXTH: 791 case AArch64_AM::UXTW: 792 case AArch64_AM::UXTX: 793 return AArch64_AM::getArithShiftValue(Imm) <= 4; 794 } 795 } 796 797 case AArch64::SUBWrs: 798 case AArch64::SUBSWrs: { 799 unsigned Imm = MI.getOperand(3).getImm(); 800 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm); 801 return ShiftVal == 0 || 802 (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 31); 803 } 804 805 case AArch64::SUBXrs: 806 case AArch64::SUBSXrs: { 807 unsigned Imm = MI.getOperand(3).getImm(); 808 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm); 809 return ShiftVal == 0 || 810 (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 63); 811 } 812 813 case AArch64::SUBWrx: 814 case AArch64::SUBXrx: 815 case AArch64::SUBXrx64: 816 case AArch64::SUBSWrx: 817 case AArch64::SUBSXrx: 818 case AArch64::SUBSXrx64: { 819 unsigned Imm = MI.getOperand(3).getImm(); 820 switch (AArch64_AM::getArithExtendType(Imm)) { 821 default: 822 return false; 823 case AArch64_AM::UXTB: 824 case AArch64_AM::UXTH: 825 case AArch64_AM::UXTW: 826 case AArch64_AM::UXTX: 827 return AArch64_AM::getArithShiftValue(Imm) == 0; 828 } 829 } 830 831 case AArch64::LDRBBroW: 832 case AArch64::LDRBBroX: 833 case AArch64::LDRBroW: 834 case AArch64::LDRBroX: 835 case AArch64::LDRDroW: 836 case AArch64::LDRDroX: 837 case AArch64::LDRHHroW: 838 case AArch64::LDRHHroX: 839 case AArch64::LDRHroW: 840 case AArch64::LDRHroX: 841 case AArch64::LDRQroW: 842 case AArch64::LDRQroX: 843 case AArch64::LDRSBWroW: 844 case AArch64::LDRSBWroX: 845 case AArch64::LDRSBXroW: 846 case AArch64::LDRSBXroX: 847 case AArch64::LDRSHWroW: 848 case AArch64::LDRSHWroX: 849 case AArch64::LDRSHXroW: 850 case AArch64::LDRSHXroX: 851 case AArch64::LDRSWroW: 852 case AArch64::LDRSWroX: 853 case AArch64::LDRSroW: 854 case AArch64::LDRSroX: 855 case AArch64::LDRWroW: 856 case AArch64::LDRWroX: 857 case AArch64::LDRXroW: 858 case AArch64::LDRXroX: 859 case AArch64::PRFMroW: 860 case AArch64::PRFMroX: 861 case AArch64::STRBBroW: 862 case AArch64::STRBBroX: 863 case AArch64::STRBroW: 864 case AArch64::STRBroX: 865 case AArch64::STRDroW: 866 case AArch64::STRDroX: 867 case AArch64::STRHHroW: 868 case AArch64::STRHHroX: 869 case AArch64::STRHroW: 870 case AArch64::STRHroX: 871 case AArch64::STRQroW: 872 case AArch64::STRQroX: 873 case AArch64::STRSroW: 874 case AArch64::STRSroX: 875 case AArch64::STRWroW: 876 case AArch64::STRWroX: 877 case AArch64::STRXroW: 878 case AArch64::STRXroX: { 879 unsigned IsSigned = MI.getOperand(3).getImm(); 880 return !IsSigned; 881 } 882 } 883 } 884 885 bool AArch64InstrInfo::isSEHInstruction(const MachineInstr &MI) { 886 unsigned Opc = MI.getOpcode(); 887 switch (Opc) { 888 default: 889 return false; 890 case AArch64::SEH_StackAlloc: 891 case AArch64::SEH_SaveFPLR: 892 case AArch64::SEH_SaveFPLR_X: 893 case AArch64::SEH_SaveReg: 894 case AArch64::SEH_SaveReg_X: 895 case AArch64::SEH_SaveRegP: 896 case AArch64::SEH_SaveRegP_X: 897 case AArch64::SEH_SaveFReg: 898 case AArch64::SEH_SaveFReg_X: 899 case AArch64::SEH_SaveFRegP: 900 case AArch64::SEH_SaveFRegP_X: 901 case AArch64::SEH_SetFP: 902 case AArch64::SEH_AddFP: 903 case AArch64::SEH_Nop: 904 case AArch64::SEH_PrologEnd: 905 case AArch64::SEH_EpilogStart: 906 case AArch64::SEH_EpilogEnd: 907 return true; 908 } 909 } 910 911 bool AArch64InstrInfo::isCoalescableExtInstr(const MachineInstr &MI, 912 unsigned &SrcReg, unsigned &DstReg, 913 unsigned &SubIdx) const { 914 switch (MI.getOpcode()) { 915 default: 916 return false; 917 case AArch64::SBFMXri: // aka sxtw 918 case AArch64::UBFMXri: // aka uxtw 919 // Check for the 32 -> 64 bit extension case, these instructions can do 920 // much more. 921 if (MI.getOperand(2).getImm() != 0 || MI.getOperand(3).getImm() != 31) 922 return false; 923 // This is a signed or unsigned 32 -> 64 bit extension. 924 SrcReg = MI.getOperand(1).getReg(); 925 DstReg = MI.getOperand(0).getReg(); 926 SubIdx = AArch64::sub_32; 927 return true; 928 } 929 } 930 931 bool AArch64InstrInfo::areMemAccessesTriviallyDisjoint( 932 const MachineInstr &MIa, const MachineInstr &MIb, AliasAnalysis *AA) const { 933 const TargetRegisterInfo *TRI = &getRegisterInfo(); 934 const MachineOperand *BaseOpA = nullptr, *BaseOpB = nullptr; 935 int64_t OffsetA = 0, OffsetB = 0; 936 unsigned WidthA = 0, WidthB = 0; 937 938 assert(MIa.mayLoadOrStore() && "MIa must be a load or store."); 939 assert(MIb.mayLoadOrStore() && "MIb must be a load or store."); 940 941 if (MIa.hasUnmodeledSideEffects() || MIb.hasUnmodeledSideEffects() || 942 MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef()) 943 return false; 944 945 // Retrieve the base, offset from the base and width. Width 946 // is the size of memory that is being loaded/stored (e.g. 1, 2, 4, 8). If 947 // base are identical, and the offset of a lower memory access + 948 // the width doesn't overlap the offset of a higher memory access, 949 // then the memory accesses are different. 950 if (getMemOperandWithOffsetWidth(MIa, BaseOpA, OffsetA, WidthA, TRI) && 951 getMemOperandWithOffsetWidth(MIb, BaseOpB, OffsetB, WidthB, TRI)) { 952 if (BaseOpA->isIdenticalTo(*BaseOpB)) { 953 int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB; 954 int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA; 955 int LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB; 956 if (LowOffset + LowWidth <= HighOffset) 957 return true; 958 } 959 } 960 return false; 961 } 962 963 bool AArch64InstrInfo::isSchedulingBoundary(const MachineInstr &MI, 964 const MachineBasicBlock *MBB, 965 const MachineFunction &MF) const { 966 if (TargetInstrInfo::isSchedulingBoundary(MI, MBB, MF)) 967 return true; 968 switch (MI.getOpcode()) { 969 case AArch64::HINT: 970 // CSDB hints are scheduling barriers. 971 if (MI.getOperand(0).getImm() == 0x14) 972 return true; 973 break; 974 case AArch64::DSB: 975 case AArch64::ISB: 976 // DSB and ISB also are scheduling barriers. 977 return true; 978 default:; 979 } 980 return isSEHInstruction(MI); 981 } 982 983 /// analyzeCompare - For a comparison instruction, return the source registers 984 /// in SrcReg and SrcReg2, and the value it compares against in CmpValue. 985 /// Return true if the comparison instruction can be analyzed. 986 bool AArch64InstrInfo::analyzeCompare(const MachineInstr &MI, unsigned &SrcReg, 987 unsigned &SrcReg2, int &CmpMask, 988 int &CmpValue) const { 989 // The first operand can be a frame index where we'd normally expect a 990 // register. 991 assert(MI.getNumOperands() >= 2 && "All AArch64 cmps should have 2 operands"); 992 if (!MI.getOperand(1).isReg()) 993 return false; 994 995 switch (MI.getOpcode()) { 996 default: 997 break; 998 case AArch64::SUBSWrr: 999 case AArch64::SUBSWrs: 1000 case AArch64::SUBSWrx: 1001 case AArch64::SUBSXrr: 1002 case AArch64::SUBSXrs: 1003 case AArch64::SUBSXrx: 1004 case AArch64::ADDSWrr: 1005 case AArch64::ADDSWrs: 1006 case AArch64::ADDSWrx: 1007 case AArch64::ADDSXrr: 1008 case AArch64::ADDSXrs: 1009 case AArch64::ADDSXrx: 1010 // Replace SUBSWrr with SUBWrr if NZCV is not used. 1011 SrcReg = MI.getOperand(1).getReg(); 1012 SrcReg2 = MI.getOperand(2).getReg(); 1013 CmpMask = ~0; 1014 CmpValue = 0; 1015 return true; 1016 case AArch64::SUBSWri: 1017 case AArch64::ADDSWri: 1018 case AArch64::SUBSXri: 1019 case AArch64::ADDSXri: 1020 SrcReg = MI.getOperand(1).getReg(); 1021 SrcReg2 = 0; 1022 CmpMask = ~0; 1023 // FIXME: In order to convert CmpValue to 0 or 1 1024 CmpValue = MI.getOperand(2).getImm() != 0; 1025 return true; 1026 case AArch64::ANDSWri: 1027 case AArch64::ANDSXri: 1028 // ANDS does not use the same encoding scheme as the others xxxS 1029 // instructions. 1030 SrcReg = MI.getOperand(1).getReg(); 1031 SrcReg2 = 0; 1032 CmpMask = ~0; 1033 // FIXME:The return val type of decodeLogicalImmediate is uint64_t, 1034 // while the type of CmpValue is int. When converting uint64_t to int, 1035 // the high 32 bits of uint64_t will be lost. 1036 // In fact it causes a bug in spec2006-483.xalancbmk 1037 // CmpValue is only used to compare with zero in OptimizeCompareInstr 1038 CmpValue = AArch64_AM::decodeLogicalImmediate( 1039 MI.getOperand(2).getImm(), 1040 MI.getOpcode() == AArch64::ANDSWri ? 32 : 64) != 0; 1041 return true; 1042 } 1043 1044 return false; 1045 } 1046 1047 static bool UpdateOperandRegClass(MachineInstr &Instr) { 1048 MachineBasicBlock *MBB = Instr.getParent(); 1049 assert(MBB && "Can't get MachineBasicBlock here"); 1050 MachineFunction *MF = MBB->getParent(); 1051 assert(MF && "Can't get MachineFunction here"); 1052 const TargetInstrInfo *TII = MF->getSubtarget().getInstrInfo(); 1053 const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo(); 1054 MachineRegisterInfo *MRI = &MF->getRegInfo(); 1055 1056 for (unsigned OpIdx = 0, EndIdx = Instr.getNumOperands(); OpIdx < EndIdx; 1057 ++OpIdx) { 1058 MachineOperand &MO = Instr.getOperand(OpIdx); 1059 const TargetRegisterClass *OpRegCstraints = 1060 Instr.getRegClassConstraint(OpIdx, TII, TRI); 1061 1062 // If there's no constraint, there's nothing to do. 1063 if (!OpRegCstraints) 1064 continue; 1065 // If the operand is a frame index, there's nothing to do here. 1066 // A frame index operand will resolve correctly during PEI. 1067 if (MO.isFI()) 1068 continue; 1069 1070 assert(MO.isReg() && 1071 "Operand has register constraints without being a register!"); 1072 1073 Register Reg = MO.getReg(); 1074 if (Register::isPhysicalRegister(Reg)) { 1075 if (!OpRegCstraints->contains(Reg)) 1076 return false; 1077 } else if (!OpRegCstraints->hasSubClassEq(MRI->getRegClass(Reg)) && 1078 !MRI->constrainRegClass(Reg, OpRegCstraints)) 1079 return false; 1080 } 1081 1082 return true; 1083 } 1084 1085 /// Return the opcode that does not set flags when possible - otherwise 1086 /// return the original opcode. The caller is responsible to do the actual 1087 /// substitution and legality checking. 1088 static unsigned convertToNonFlagSettingOpc(const MachineInstr &MI) { 1089 // Don't convert all compare instructions, because for some the zero register 1090 // encoding becomes the sp register. 1091 bool MIDefinesZeroReg = false; 1092 if (MI.definesRegister(AArch64::WZR) || MI.definesRegister(AArch64::XZR)) 1093 MIDefinesZeroReg = true; 1094 1095 switch (MI.getOpcode()) { 1096 default: 1097 return MI.getOpcode(); 1098 case AArch64::ADDSWrr: 1099 return AArch64::ADDWrr; 1100 case AArch64::ADDSWri: 1101 return MIDefinesZeroReg ? AArch64::ADDSWri : AArch64::ADDWri; 1102 case AArch64::ADDSWrs: 1103 return MIDefinesZeroReg ? AArch64::ADDSWrs : AArch64::ADDWrs; 1104 case AArch64::ADDSWrx: 1105 return AArch64::ADDWrx; 1106 case AArch64::ADDSXrr: 1107 return AArch64::ADDXrr; 1108 case AArch64::ADDSXri: 1109 return MIDefinesZeroReg ? AArch64::ADDSXri : AArch64::ADDXri; 1110 case AArch64::ADDSXrs: 1111 return MIDefinesZeroReg ? AArch64::ADDSXrs : AArch64::ADDXrs; 1112 case AArch64::ADDSXrx: 1113 return AArch64::ADDXrx; 1114 case AArch64::SUBSWrr: 1115 return AArch64::SUBWrr; 1116 case AArch64::SUBSWri: 1117 return MIDefinesZeroReg ? AArch64::SUBSWri : AArch64::SUBWri; 1118 case AArch64::SUBSWrs: 1119 return MIDefinesZeroReg ? AArch64::SUBSWrs : AArch64::SUBWrs; 1120 case AArch64::SUBSWrx: 1121 return AArch64::SUBWrx; 1122 case AArch64::SUBSXrr: 1123 return AArch64::SUBXrr; 1124 case AArch64::SUBSXri: 1125 return MIDefinesZeroReg ? AArch64::SUBSXri : AArch64::SUBXri; 1126 case AArch64::SUBSXrs: 1127 return MIDefinesZeroReg ? AArch64::SUBSXrs : AArch64::SUBXrs; 1128 case AArch64::SUBSXrx: 1129 return AArch64::SUBXrx; 1130 } 1131 } 1132 1133 enum AccessKind { AK_Write = 0x01, AK_Read = 0x10, AK_All = 0x11 }; 1134 1135 /// True when condition flags are accessed (either by writing or reading) 1136 /// on the instruction trace starting at From and ending at To. 1137 /// 1138 /// Note: If From and To are from different blocks it's assumed CC are accessed 1139 /// on the path. 1140 static bool areCFlagsAccessedBetweenInstrs( 1141 MachineBasicBlock::iterator From, MachineBasicBlock::iterator To, 1142 const TargetRegisterInfo *TRI, const AccessKind AccessToCheck = AK_All) { 1143 // Early exit if To is at the beginning of the BB. 1144 if (To == To->getParent()->begin()) 1145 return true; 1146 1147 // Check whether the instructions are in the same basic block 1148 // If not, assume the condition flags might get modified somewhere. 1149 if (To->getParent() != From->getParent()) 1150 return true; 1151 1152 // From must be above To. 1153 assert(std::find_if(++To.getReverse(), To->getParent()->rend(), 1154 [From](MachineInstr &MI) { 1155 return MI.getIterator() == From; 1156 }) != To->getParent()->rend()); 1157 1158 // We iterate backward starting \p To until we hit \p From. 1159 for (--To; To != From; --To) { 1160 const MachineInstr &Instr = *To; 1161 1162 if (((AccessToCheck & AK_Write) && 1163 Instr.modifiesRegister(AArch64::NZCV, TRI)) || 1164 ((AccessToCheck & AK_Read) && Instr.readsRegister(AArch64::NZCV, TRI))) 1165 return true; 1166 } 1167 return false; 1168 } 1169 1170 /// Try to optimize a compare instruction. A compare instruction is an 1171 /// instruction which produces AArch64::NZCV. It can be truly compare 1172 /// instruction 1173 /// when there are no uses of its destination register. 1174 /// 1175 /// The following steps are tried in order: 1176 /// 1. Convert CmpInstr into an unconditional version. 1177 /// 2. Remove CmpInstr if above there is an instruction producing a needed 1178 /// condition code or an instruction which can be converted into such an 1179 /// instruction. 1180 /// Only comparison with zero is supported. 1181 bool AArch64InstrInfo::optimizeCompareInstr( 1182 MachineInstr &CmpInstr, unsigned SrcReg, unsigned SrcReg2, int CmpMask, 1183 int CmpValue, const MachineRegisterInfo *MRI) const { 1184 assert(CmpInstr.getParent()); 1185 assert(MRI); 1186 1187 // Replace SUBSWrr with SUBWrr if NZCV is not used. 1188 int DeadNZCVIdx = CmpInstr.findRegisterDefOperandIdx(AArch64::NZCV, true); 1189 if (DeadNZCVIdx != -1) { 1190 if (CmpInstr.definesRegister(AArch64::WZR) || 1191 CmpInstr.definesRegister(AArch64::XZR)) { 1192 CmpInstr.eraseFromParent(); 1193 return true; 1194 } 1195 unsigned Opc = CmpInstr.getOpcode(); 1196 unsigned NewOpc = convertToNonFlagSettingOpc(CmpInstr); 1197 if (NewOpc == Opc) 1198 return false; 1199 const MCInstrDesc &MCID = get(NewOpc); 1200 CmpInstr.setDesc(MCID); 1201 CmpInstr.RemoveOperand(DeadNZCVIdx); 1202 bool succeeded = UpdateOperandRegClass(CmpInstr); 1203 (void)succeeded; 1204 assert(succeeded && "Some operands reg class are incompatible!"); 1205 return true; 1206 } 1207 1208 // Continue only if we have a "ri" where immediate is zero. 1209 // FIXME:CmpValue has already been converted to 0 or 1 in analyzeCompare 1210 // function. 1211 assert((CmpValue == 0 || CmpValue == 1) && "CmpValue must be 0 or 1!"); 1212 if (CmpValue != 0 || SrcReg2 != 0) 1213 return false; 1214 1215 // CmpInstr is a Compare instruction if destination register is not used. 1216 if (!MRI->use_nodbg_empty(CmpInstr.getOperand(0).getReg())) 1217 return false; 1218 1219 return substituteCmpToZero(CmpInstr, SrcReg, MRI); 1220 } 1221 1222 /// Get opcode of S version of Instr. 1223 /// If Instr is S version its opcode is returned. 1224 /// AArch64::INSTRUCTION_LIST_END is returned if Instr does not have S version 1225 /// or we are not interested in it. 1226 static unsigned sForm(MachineInstr &Instr) { 1227 switch (Instr.getOpcode()) { 1228 default: 1229 return AArch64::INSTRUCTION_LIST_END; 1230 1231 case AArch64::ADDSWrr: 1232 case AArch64::ADDSWri: 1233 case AArch64::ADDSXrr: 1234 case AArch64::ADDSXri: 1235 case AArch64::SUBSWrr: 1236 case AArch64::SUBSWri: 1237 case AArch64::SUBSXrr: 1238 case AArch64::SUBSXri: 1239 return Instr.getOpcode(); 1240 1241 case AArch64::ADDWrr: 1242 return AArch64::ADDSWrr; 1243 case AArch64::ADDWri: 1244 return AArch64::ADDSWri; 1245 case AArch64::ADDXrr: 1246 return AArch64::ADDSXrr; 1247 case AArch64::ADDXri: 1248 return AArch64::ADDSXri; 1249 case AArch64::ADCWr: 1250 return AArch64::ADCSWr; 1251 case AArch64::ADCXr: 1252 return AArch64::ADCSXr; 1253 case AArch64::SUBWrr: 1254 return AArch64::SUBSWrr; 1255 case AArch64::SUBWri: 1256 return AArch64::SUBSWri; 1257 case AArch64::SUBXrr: 1258 return AArch64::SUBSXrr; 1259 case AArch64::SUBXri: 1260 return AArch64::SUBSXri; 1261 case AArch64::SBCWr: 1262 return AArch64::SBCSWr; 1263 case AArch64::SBCXr: 1264 return AArch64::SBCSXr; 1265 case AArch64::ANDWri: 1266 return AArch64::ANDSWri; 1267 case AArch64::ANDXri: 1268 return AArch64::ANDSXri; 1269 } 1270 } 1271 1272 /// Check if AArch64::NZCV should be alive in successors of MBB. 1273 static bool areCFlagsAliveInSuccessors(MachineBasicBlock *MBB) { 1274 for (auto *BB : MBB->successors()) 1275 if (BB->isLiveIn(AArch64::NZCV)) 1276 return true; 1277 return false; 1278 } 1279 1280 namespace { 1281 1282 struct UsedNZCV { 1283 bool N = false; 1284 bool Z = false; 1285 bool C = false; 1286 bool V = false; 1287 1288 UsedNZCV() = default; 1289 1290 UsedNZCV &operator|=(const UsedNZCV &UsedFlags) { 1291 this->N |= UsedFlags.N; 1292 this->Z |= UsedFlags.Z; 1293 this->C |= UsedFlags.C; 1294 this->V |= UsedFlags.V; 1295 return *this; 1296 } 1297 }; 1298 1299 } // end anonymous namespace 1300 1301 /// Find a condition code used by the instruction. 1302 /// Returns AArch64CC::Invalid if either the instruction does not use condition 1303 /// codes or we don't optimize CmpInstr in the presence of such instructions. 1304 static AArch64CC::CondCode findCondCodeUsedByInstr(const MachineInstr &Instr) { 1305 switch (Instr.getOpcode()) { 1306 default: 1307 return AArch64CC::Invalid; 1308 1309 case AArch64::Bcc: { 1310 int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV); 1311 assert(Idx >= 2); 1312 return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 2).getImm()); 1313 } 1314 1315 case AArch64::CSINVWr: 1316 case AArch64::CSINVXr: 1317 case AArch64::CSINCWr: 1318 case AArch64::CSINCXr: 1319 case AArch64::CSELWr: 1320 case AArch64::CSELXr: 1321 case AArch64::CSNEGWr: 1322 case AArch64::CSNEGXr: 1323 case AArch64::FCSELSrrr: 1324 case AArch64::FCSELDrrr: { 1325 int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV); 1326 assert(Idx >= 1); 1327 return static_cast<AArch64CC::CondCode>(Instr.getOperand(Idx - 1).getImm()); 1328 } 1329 } 1330 } 1331 1332 static UsedNZCV getUsedNZCV(AArch64CC::CondCode CC) { 1333 assert(CC != AArch64CC::Invalid); 1334 UsedNZCV UsedFlags; 1335 switch (CC) { 1336 default: 1337 break; 1338 1339 case AArch64CC::EQ: // Z set 1340 case AArch64CC::NE: // Z clear 1341 UsedFlags.Z = true; 1342 break; 1343 1344 case AArch64CC::HI: // Z clear and C set 1345 case AArch64CC::LS: // Z set or C clear 1346 UsedFlags.Z = true; 1347 LLVM_FALLTHROUGH; 1348 case AArch64CC::HS: // C set 1349 case AArch64CC::LO: // C clear 1350 UsedFlags.C = true; 1351 break; 1352 1353 case AArch64CC::MI: // N set 1354 case AArch64CC::PL: // N clear 1355 UsedFlags.N = true; 1356 break; 1357 1358 case AArch64CC::VS: // V set 1359 case AArch64CC::VC: // V clear 1360 UsedFlags.V = true; 1361 break; 1362 1363 case AArch64CC::GT: // Z clear, N and V the same 1364 case AArch64CC::LE: // Z set, N and V differ 1365 UsedFlags.Z = true; 1366 LLVM_FALLTHROUGH; 1367 case AArch64CC::GE: // N and V the same 1368 case AArch64CC::LT: // N and V differ 1369 UsedFlags.N = true; 1370 UsedFlags.V = true; 1371 break; 1372 } 1373 return UsedFlags; 1374 } 1375 1376 static bool isADDSRegImm(unsigned Opcode) { 1377 return Opcode == AArch64::ADDSWri || Opcode == AArch64::ADDSXri; 1378 } 1379 1380 static bool isSUBSRegImm(unsigned Opcode) { 1381 return Opcode == AArch64::SUBSWri || Opcode == AArch64::SUBSXri; 1382 } 1383 1384 /// Check if CmpInstr can be substituted by MI. 1385 /// 1386 /// CmpInstr can be substituted: 1387 /// - CmpInstr is either 'ADDS %vreg, 0' or 'SUBS %vreg, 0' 1388 /// - and, MI and CmpInstr are from the same MachineBB 1389 /// - and, condition flags are not alive in successors of the CmpInstr parent 1390 /// - and, if MI opcode is the S form there must be no defs of flags between 1391 /// MI and CmpInstr 1392 /// or if MI opcode is not the S form there must be neither defs of flags 1393 /// nor uses of flags between MI and CmpInstr. 1394 /// - and C/V flags are not used after CmpInstr 1395 static bool canInstrSubstituteCmpInstr(MachineInstr *MI, MachineInstr *CmpInstr, 1396 const TargetRegisterInfo *TRI) { 1397 assert(MI); 1398 assert(sForm(*MI) != AArch64::INSTRUCTION_LIST_END); 1399 assert(CmpInstr); 1400 1401 const unsigned CmpOpcode = CmpInstr->getOpcode(); 1402 if (!isADDSRegImm(CmpOpcode) && !isSUBSRegImm(CmpOpcode)) 1403 return false; 1404 1405 if (MI->getParent() != CmpInstr->getParent()) 1406 return false; 1407 1408 if (areCFlagsAliveInSuccessors(CmpInstr->getParent())) 1409 return false; 1410 1411 AccessKind AccessToCheck = AK_Write; 1412 if (sForm(*MI) != MI->getOpcode()) 1413 AccessToCheck = AK_All; 1414 if (areCFlagsAccessedBetweenInstrs(MI, CmpInstr, TRI, AccessToCheck)) 1415 return false; 1416 1417 UsedNZCV NZCVUsedAfterCmp; 1418 for (auto I = std::next(CmpInstr->getIterator()), 1419 E = CmpInstr->getParent()->instr_end(); 1420 I != E; ++I) { 1421 const MachineInstr &Instr = *I; 1422 if (Instr.readsRegister(AArch64::NZCV, TRI)) { 1423 AArch64CC::CondCode CC = findCondCodeUsedByInstr(Instr); 1424 if (CC == AArch64CC::Invalid) // Unsupported conditional instruction 1425 return false; 1426 NZCVUsedAfterCmp |= getUsedNZCV(CC); 1427 } 1428 1429 if (Instr.modifiesRegister(AArch64::NZCV, TRI)) 1430 break; 1431 } 1432 1433 return !NZCVUsedAfterCmp.C && !NZCVUsedAfterCmp.V; 1434 } 1435 1436 /// Substitute an instruction comparing to zero with another instruction 1437 /// which produces needed condition flags. 1438 /// 1439 /// Return true on success. 1440 bool AArch64InstrInfo::substituteCmpToZero( 1441 MachineInstr &CmpInstr, unsigned SrcReg, 1442 const MachineRegisterInfo *MRI) const { 1443 assert(MRI); 1444 // Get the unique definition of SrcReg. 1445 MachineInstr *MI = MRI->getUniqueVRegDef(SrcReg); 1446 if (!MI) 1447 return false; 1448 1449 const TargetRegisterInfo *TRI = &getRegisterInfo(); 1450 1451 unsigned NewOpc = sForm(*MI); 1452 if (NewOpc == AArch64::INSTRUCTION_LIST_END) 1453 return false; 1454 1455 if (!canInstrSubstituteCmpInstr(MI, &CmpInstr, TRI)) 1456 return false; 1457 1458 // Update the instruction to set NZCV. 1459 MI->setDesc(get(NewOpc)); 1460 CmpInstr.eraseFromParent(); 1461 bool succeeded = UpdateOperandRegClass(*MI); 1462 (void)succeeded; 1463 assert(succeeded && "Some operands reg class are incompatible!"); 1464 MI->addRegisterDefined(AArch64::NZCV, TRI); 1465 return true; 1466 } 1467 1468 bool AArch64InstrInfo::expandPostRAPseudo(MachineInstr &MI) const { 1469 if (MI.getOpcode() != TargetOpcode::LOAD_STACK_GUARD && 1470 MI.getOpcode() != AArch64::CATCHRET) 1471 return false; 1472 1473 MachineBasicBlock &MBB = *MI.getParent(); 1474 auto &Subtarget = MBB.getParent()->getSubtarget<AArch64Subtarget>(); 1475 auto TRI = Subtarget.getRegisterInfo(); 1476 DebugLoc DL = MI.getDebugLoc(); 1477 1478 if (MI.getOpcode() == AArch64::CATCHRET) { 1479 // Skip to the first instruction before the epilog. 1480 const TargetInstrInfo *TII = 1481 MBB.getParent()->getSubtarget().getInstrInfo(); 1482 MachineBasicBlock *TargetMBB = MI.getOperand(0).getMBB(); 1483 auto MBBI = MachineBasicBlock::iterator(MI); 1484 MachineBasicBlock::iterator FirstEpilogSEH = std::prev(MBBI); 1485 while (FirstEpilogSEH->getFlag(MachineInstr::FrameDestroy) && 1486 FirstEpilogSEH != MBB.begin()) 1487 FirstEpilogSEH = std::prev(FirstEpilogSEH); 1488 if (FirstEpilogSEH != MBB.begin()) 1489 FirstEpilogSEH = std::next(FirstEpilogSEH); 1490 BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADRP)) 1491 .addReg(AArch64::X0, RegState::Define) 1492 .addMBB(TargetMBB); 1493 BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADDXri)) 1494 .addReg(AArch64::X0, RegState::Define) 1495 .addReg(AArch64::X0) 1496 .addMBB(TargetMBB) 1497 .addImm(0); 1498 return true; 1499 } 1500 1501 Register Reg = MI.getOperand(0).getReg(); 1502 const GlobalValue *GV = 1503 cast<GlobalValue>((*MI.memoperands_begin())->getValue()); 1504 const TargetMachine &TM = MBB.getParent()->getTarget(); 1505 unsigned OpFlags = Subtarget.ClassifyGlobalReference(GV, TM); 1506 const unsigned char MO_NC = AArch64II::MO_NC; 1507 1508 if ((OpFlags & AArch64II::MO_GOT) != 0) { 1509 BuildMI(MBB, MI, DL, get(AArch64::LOADgot), Reg) 1510 .addGlobalAddress(GV, 0, OpFlags); 1511 if (Subtarget.isTargetILP32()) { 1512 unsigned Reg32 = TRI->getSubReg(Reg, AArch64::sub_32); 1513 BuildMI(MBB, MI, DL, get(AArch64::LDRWui)) 1514 .addDef(Reg32, RegState::Dead) 1515 .addUse(Reg, RegState::Kill) 1516 .addImm(0) 1517 .addMemOperand(*MI.memoperands_begin()) 1518 .addDef(Reg, RegState::Implicit); 1519 } else { 1520 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg) 1521 .addReg(Reg, RegState::Kill) 1522 .addImm(0) 1523 .addMemOperand(*MI.memoperands_begin()); 1524 } 1525 } else if (TM.getCodeModel() == CodeModel::Large) { 1526 assert(!Subtarget.isTargetILP32() && "how can large exist in ILP32?"); 1527 BuildMI(MBB, MI, DL, get(AArch64::MOVZXi), Reg) 1528 .addGlobalAddress(GV, 0, AArch64II::MO_G0 | MO_NC) 1529 .addImm(0); 1530 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg) 1531 .addReg(Reg, RegState::Kill) 1532 .addGlobalAddress(GV, 0, AArch64II::MO_G1 | MO_NC) 1533 .addImm(16); 1534 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg) 1535 .addReg(Reg, RegState::Kill) 1536 .addGlobalAddress(GV, 0, AArch64II::MO_G2 | MO_NC) 1537 .addImm(32); 1538 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg) 1539 .addReg(Reg, RegState::Kill) 1540 .addGlobalAddress(GV, 0, AArch64II::MO_G3) 1541 .addImm(48); 1542 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg) 1543 .addReg(Reg, RegState::Kill) 1544 .addImm(0) 1545 .addMemOperand(*MI.memoperands_begin()); 1546 } else if (TM.getCodeModel() == CodeModel::Tiny) { 1547 BuildMI(MBB, MI, DL, get(AArch64::ADR), Reg) 1548 .addGlobalAddress(GV, 0, OpFlags); 1549 } else { 1550 BuildMI(MBB, MI, DL, get(AArch64::ADRP), Reg) 1551 .addGlobalAddress(GV, 0, OpFlags | AArch64II::MO_PAGE); 1552 unsigned char LoFlags = OpFlags | AArch64II::MO_PAGEOFF | MO_NC; 1553 if (Subtarget.isTargetILP32()) { 1554 unsigned Reg32 = TRI->getSubReg(Reg, AArch64::sub_32); 1555 BuildMI(MBB, MI, DL, get(AArch64::LDRWui)) 1556 .addDef(Reg32, RegState::Dead) 1557 .addUse(Reg, RegState::Kill) 1558 .addGlobalAddress(GV, 0, LoFlags) 1559 .addMemOperand(*MI.memoperands_begin()) 1560 .addDef(Reg, RegState::Implicit); 1561 } else { 1562 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg) 1563 .addReg(Reg, RegState::Kill) 1564 .addGlobalAddress(GV, 0, LoFlags) 1565 .addMemOperand(*MI.memoperands_begin()); 1566 } 1567 } 1568 1569 MBB.erase(MI); 1570 1571 return true; 1572 } 1573 1574 // Return true if this instruction simply sets its single destination register 1575 // to zero. This is equivalent to a register rename of the zero-register. 1576 bool AArch64InstrInfo::isGPRZero(const MachineInstr &MI) { 1577 switch (MI.getOpcode()) { 1578 default: 1579 break; 1580 case AArch64::MOVZWi: 1581 case AArch64::MOVZXi: // movz Rd, #0 (LSL #0) 1582 if (MI.getOperand(1).isImm() && MI.getOperand(1).getImm() == 0) { 1583 assert(MI.getDesc().getNumOperands() == 3 && 1584 MI.getOperand(2).getImm() == 0 && "invalid MOVZi operands"); 1585 return true; 1586 } 1587 break; 1588 case AArch64::ANDWri: // and Rd, Rzr, #imm 1589 return MI.getOperand(1).getReg() == AArch64::WZR; 1590 case AArch64::ANDXri: 1591 return MI.getOperand(1).getReg() == AArch64::XZR; 1592 case TargetOpcode::COPY: 1593 return MI.getOperand(1).getReg() == AArch64::WZR; 1594 } 1595 return false; 1596 } 1597 1598 // Return true if this instruction simply renames a general register without 1599 // modifying bits. 1600 bool AArch64InstrInfo::isGPRCopy(const MachineInstr &MI) { 1601 switch (MI.getOpcode()) { 1602 default: 1603 break; 1604 case TargetOpcode::COPY: { 1605 // GPR32 copies will by lowered to ORRXrs 1606 Register DstReg = MI.getOperand(0).getReg(); 1607 return (AArch64::GPR32RegClass.contains(DstReg) || 1608 AArch64::GPR64RegClass.contains(DstReg)); 1609 } 1610 case AArch64::ORRXrs: // orr Xd, Xzr, Xm (LSL #0) 1611 if (MI.getOperand(1).getReg() == AArch64::XZR) { 1612 assert(MI.getDesc().getNumOperands() == 4 && 1613 MI.getOperand(3).getImm() == 0 && "invalid ORRrs operands"); 1614 return true; 1615 } 1616 break; 1617 case AArch64::ADDXri: // add Xd, Xn, #0 (LSL #0) 1618 if (MI.getOperand(2).getImm() == 0) { 1619 assert(MI.getDesc().getNumOperands() == 4 && 1620 MI.getOperand(3).getImm() == 0 && "invalid ADDXri operands"); 1621 return true; 1622 } 1623 break; 1624 } 1625 return false; 1626 } 1627 1628 // Return true if this instruction simply renames a general register without 1629 // modifying bits. 1630 bool AArch64InstrInfo::isFPRCopy(const MachineInstr &MI) { 1631 switch (MI.getOpcode()) { 1632 default: 1633 break; 1634 case TargetOpcode::COPY: { 1635 // FPR64 copies will by lowered to ORR.16b 1636 Register DstReg = MI.getOperand(0).getReg(); 1637 return (AArch64::FPR64RegClass.contains(DstReg) || 1638 AArch64::FPR128RegClass.contains(DstReg)); 1639 } 1640 case AArch64::ORRv16i8: 1641 if (MI.getOperand(1).getReg() == MI.getOperand(2).getReg()) { 1642 assert(MI.getDesc().getNumOperands() == 3 && MI.getOperand(0).isReg() && 1643 "invalid ORRv16i8 operands"); 1644 return true; 1645 } 1646 break; 1647 } 1648 return false; 1649 } 1650 1651 unsigned AArch64InstrInfo::isLoadFromStackSlot(const MachineInstr &MI, 1652 int &FrameIndex) const { 1653 switch (MI.getOpcode()) { 1654 default: 1655 break; 1656 case AArch64::LDRWui: 1657 case AArch64::LDRXui: 1658 case AArch64::LDRBui: 1659 case AArch64::LDRHui: 1660 case AArch64::LDRSui: 1661 case AArch64::LDRDui: 1662 case AArch64::LDRQui: 1663 if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() && 1664 MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) { 1665 FrameIndex = MI.getOperand(1).getIndex(); 1666 return MI.getOperand(0).getReg(); 1667 } 1668 break; 1669 } 1670 1671 return 0; 1672 } 1673 1674 unsigned AArch64InstrInfo::isStoreToStackSlot(const MachineInstr &MI, 1675 int &FrameIndex) const { 1676 switch (MI.getOpcode()) { 1677 default: 1678 break; 1679 case AArch64::STRWui: 1680 case AArch64::STRXui: 1681 case AArch64::STRBui: 1682 case AArch64::STRHui: 1683 case AArch64::STRSui: 1684 case AArch64::STRDui: 1685 case AArch64::STRQui: 1686 if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() && 1687 MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) { 1688 FrameIndex = MI.getOperand(1).getIndex(); 1689 return MI.getOperand(0).getReg(); 1690 } 1691 break; 1692 } 1693 return 0; 1694 } 1695 1696 /// Check all MachineMemOperands for a hint to suppress pairing. 1697 bool AArch64InstrInfo::isLdStPairSuppressed(const MachineInstr &MI) { 1698 return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) { 1699 return MMO->getFlags() & MOSuppressPair; 1700 }); 1701 } 1702 1703 /// Set a flag on the first MachineMemOperand to suppress pairing. 1704 void AArch64InstrInfo::suppressLdStPair(MachineInstr &MI) { 1705 if (MI.memoperands_empty()) 1706 return; 1707 (*MI.memoperands_begin())->setFlags(MOSuppressPair); 1708 } 1709 1710 /// Check all MachineMemOperands for a hint that the load/store is strided. 1711 bool AArch64InstrInfo::isStridedAccess(const MachineInstr &MI) { 1712 return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) { 1713 return MMO->getFlags() & MOStridedAccess; 1714 }); 1715 } 1716 1717 bool AArch64InstrInfo::isUnscaledLdSt(unsigned Opc) { 1718 switch (Opc) { 1719 default: 1720 return false; 1721 case AArch64::STURSi: 1722 case AArch64::STURDi: 1723 case AArch64::STURQi: 1724 case AArch64::STURBBi: 1725 case AArch64::STURHHi: 1726 case AArch64::STURWi: 1727 case AArch64::STURXi: 1728 case AArch64::LDURSi: 1729 case AArch64::LDURDi: 1730 case AArch64::LDURQi: 1731 case AArch64::LDURWi: 1732 case AArch64::LDURXi: 1733 case AArch64::LDURSWi: 1734 case AArch64::LDURHHi: 1735 case AArch64::LDURBBi: 1736 case AArch64::LDURSBWi: 1737 case AArch64::LDURSHWi: 1738 return true; 1739 } 1740 } 1741 1742 Optional<unsigned> AArch64InstrInfo::getUnscaledLdSt(unsigned Opc) { 1743 switch (Opc) { 1744 default: return {}; 1745 case AArch64::PRFMui: return AArch64::PRFUMi; 1746 case AArch64::LDRXui: return AArch64::LDURXi; 1747 case AArch64::LDRWui: return AArch64::LDURWi; 1748 case AArch64::LDRBui: return AArch64::LDURBi; 1749 case AArch64::LDRHui: return AArch64::LDURHi; 1750 case AArch64::LDRSui: return AArch64::LDURSi; 1751 case AArch64::LDRDui: return AArch64::LDURDi; 1752 case AArch64::LDRQui: return AArch64::LDURQi; 1753 case AArch64::LDRBBui: return AArch64::LDURBBi; 1754 case AArch64::LDRHHui: return AArch64::LDURHHi; 1755 case AArch64::LDRSBXui: return AArch64::LDURSBXi; 1756 case AArch64::LDRSBWui: return AArch64::LDURSBWi; 1757 case AArch64::LDRSHXui: return AArch64::LDURSHXi; 1758 case AArch64::LDRSHWui: return AArch64::LDURSHWi; 1759 case AArch64::LDRSWui: return AArch64::LDURSWi; 1760 case AArch64::STRXui: return AArch64::STURXi; 1761 case AArch64::STRWui: return AArch64::STURWi; 1762 case AArch64::STRBui: return AArch64::STURBi; 1763 case AArch64::STRHui: return AArch64::STURHi; 1764 case AArch64::STRSui: return AArch64::STURSi; 1765 case AArch64::STRDui: return AArch64::STURDi; 1766 case AArch64::STRQui: return AArch64::STURQi; 1767 case AArch64::STRBBui: return AArch64::STURBBi; 1768 case AArch64::STRHHui: return AArch64::STURHHi; 1769 } 1770 } 1771 1772 unsigned AArch64InstrInfo::getLoadStoreImmIdx(unsigned Opc) { 1773 switch (Opc) { 1774 default: 1775 return 2; 1776 case AArch64::LDPXi: 1777 case AArch64::LDPDi: 1778 case AArch64::STPXi: 1779 case AArch64::STPDi: 1780 case AArch64::LDNPXi: 1781 case AArch64::LDNPDi: 1782 case AArch64::STNPXi: 1783 case AArch64::STNPDi: 1784 case AArch64::LDPQi: 1785 case AArch64::STPQi: 1786 case AArch64::LDNPQi: 1787 case AArch64::STNPQi: 1788 case AArch64::LDPWi: 1789 case AArch64::LDPSi: 1790 case AArch64::STPWi: 1791 case AArch64::STPSi: 1792 case AArch64::LDNPWi: 1793 case AArch64::LDNPSi: 1794 case AArch64::STNPWi: 1795 case AArch64::STNPSi: 1796 case AArch64::LDG: 1797 case AArch64::STGPi: 1798 return 3; 1799 case AArch64::ADDG: 1800 case AArch64::STGOffset: 1801 return 2; 1802 } 1803 } 1804 1805 bool AArch64InstrInfo::isPairableLdStInst(const MachineInstr &MI) { 1806 switch (MI.getOpcode()) { 1807 default: 1808 return false; 1809 // Scaled instructions. 1810 case AArch64::STRSui: 1811 case AArch64::STRDui: 1812 case AArch64::STRQui: 1813 case AArch64::STRXui: 1814 case AArch64::STRWui: 1815 case AArch64::LDRSui: 1816 case AArch64::LDRDui: 1817 case AArch64::LDRQui: 1818 case AArch64::LDRXui: 1819 case AArch64::LDRWui: 1820 case AArch64::LDRSWui: 1821 // Unscaled instructions. 1822 case AArch64::STURSi: 1823 case AArch64::STURDi: 1824 case AArch64::STURQi: 1825 case AArch64::STURWi: 1826 case AArch64::STURXi: 1827 case AArch64::LDURSi: 1828 case AArch64::LDURDi: 1829 case AArch64::LDURQi: 1830 case AArch64::LDURWi: 1831 case AArch64::LDURXi: 1832 case AArch64::LDURSWi: 1833 return true; 1834 } 1835 } 1836 1837 unsigned AArch64InstrInfo::convertToFlagSettingOpc(unsigned Opc, 1838 bool &Is64Bit) { 1839 switch (Opc) { 1840 default: 1841 llvm_unreachable("Opcode has no flag setting equivalent!"); 1842 // 32-bit cases: 1843 case AArch64::ADDWri: 1844 Is64Bit = false; 1845 return AArch64::ADDSWri; 1846 case AArch64::ADDWrr: 1847 Is64Bit = false; 1848 return AArch64::ADDSWrr; 1849 case AArch64::ADDWrs: 1850 Is64Bit = false; 1851 return AArch64::ADDSWrs; 1852 case AArch64::ADDWrx: 1853 Is64Bit = false; 1854 return AArch64::ADDSWrx; 1855 case AArch64::ANDWri: 1856 Is64Bit = false; 1857 return AArch64::ANDSWri; 1858 case AArch64::ANDWrr: 1859 Is64Bit = false; 1860 return AArch64::ANDSWrr; 1861 case AArch64::ANDWrs: 1862 Is64Bit = false; 1863 return AArch64::ANDSWrs; 1864 case AArch64::BICWrr: 1865 Is64Bit = false; 1866 return AArch64::BICSWrr; 1867 case AArch64::BICWrs: 1868 Is64Bit = false; 1869 return AArch64::BICSWrs; 1870 case AArch64::SUBWri: 1871 Is64Bit = false; 1872 return AArch64::SUBSWri; 1873 case AArch64::SUBWrr: 1874 Is64Bit = false; 1875 return AArch64::SUBSWrr; 1876 case AArch64::SUBWrs: 1877 Is64Bit = false; 1878 return AArch64::SUBSWrs; 1879 case AArch64::SUBWrx: 1880 Is64Bit = false; 1881 return AArch64::SUBSWrx; 1882 // 64-bit cases: 1883 case AArch64::ADDXri: 1884 Is64Bit = true; 1885 return AArch64::ADDSXri; 1886 case AArch64::ADDXrr: 1887 Is64Bit = true; 1888 return AArch64::ADDSXrr; 1889 case AArch64::ADDXrs: 1890 Is64Bit = true; 1891 return AArch64::ADDSXrs; 1892 case AArch64::ADDXrx: 1893 Is64Bit = true; 1894 return AArch64::ADDSXrx; 1895 case AArch64::ANDXri: 1896 Is64Bit = true; 1897 return AArch64::ANDSXri; 1898 case AArch64::ANDXrr: 1899 Is64Bit = true; 1900 return AArch64::ANDSXrr; 1901 case AArch64::ANDXrs: 1902 Is64Bit = true; 1903 return AArch64::ANDSXrs; 1904 case AArch64::BICXrr: 1905 Is64Bit = true; 1906 return AArch64::BICSXrr; 1907 case AArch64::BICXrs: 1908 Is64Bit = true; 1909 return AArch64::BICSXrs; 1910 case AArch64::SUBXri: 1911 Is64Bit = true; 1912 return AArch64::SUBSXri; 1913 case AArch64::SUBXrr: 1914 Is64Bit = true; 1915 return AArch64::SUBSXrr; 1916 case AArch64::SUBXrs: 1917 Is64Bit = true; 1918 return AArch64::SUBSXrs; 1919 case AArch64::SUBXrx: 1920 Is64Bit = true; 1921 return AArch64::SUBSXrx; 1922 } 1923 } 1924 1925 // Is this a candidate for ld/st merging or pairing? For example, we don't 1926 // touch volatiles or load/stores that have a hint to avoid pair formation. 1927 bool AArch64InstrInfo::isCandidateToMergeOrPair(const MachineInstr &MI) const { 1928 // If this is a volatile load/store, don't mess with it. 1929 if (MI.hasOrderedMemoryRef()) 1930 return false; 1931 1932 // Make sure this is a reg/fi+imm (as opposed to an address reloc). 1933 assert((MI.getOperand(1).isReg() || MI.getOperand(1).isFI()) && 1934 "Expected a reg or frame index operand."); 1935 if (!MI.getOperand(2).isImm()) 1936 return false; 1937 1938 // Can't merge/pair if the instruction modifies the base register. 1939 // e.g., ldr x0, [x0] 1940 // This case will never occur with an FI base. 1941 if (MI.getOperand(1).isReg()) { 1942 Register BaseReg = MI.getOperand(1).getReg(); 1943 const TargetRegisterInfo *TRI = &getRegisterInfo(); 1944 if (MI.modifiesRegister(BaseReg, TRI)) 1945 return false; 1946 } 1947 1948 // Check if this load/store has a hint to avoid pair formation. 1949 // MachineMemOperands hints are set by the AArch64StorePairSuppress pass. 1950 if (isLdStPairSuppressed(MI)) 1951 return false; 1952 1953 // Do not pair any callee-save store/reload instructions in the 1954 // prologue/epilogue if the CFI information encoded the operations as separate 1955 // instructions, as that will cause the size of the actual prologue to mismatch 1956 // with the prologue size recorded in the Windows CFI. 1957 const MCAsmInfo *MAI = MI.getMF()->getTarget().getMCAsmInfo(); 1958 bool NeedsWinCFI = MAI->usesWindowsCFI() && 1959 MI.getMF()->getFunction().needsUnwindTableEntry(); 1960 if (NeedsWinCFI && (MI.getFlag(MachineInstr::FrameSetup) || 1961 MI.getFlag(MachineInstr::FrameDestroy))) 1962 return false; 1963 1964 // On some CPUs quad load/store pairs are slower than two single load/stores. 1965 if (Subtarget.isPaired128Slow()) { 1966 switch (MI.getOpcode()) { 1967 default: 1968 break; 1969 case AArch64::LDURQi: 1970 case AArch64::STURQi: 1971 case AArch64::LDRQui: 1972 case AArch64::STRQui: 1973 return false; 1974 } 1975 } 1976 1977 return true; 1978 } 1979 1980 bool AArch64InstrInfo::getMemOperandWithOffset(const MachineInstr &LdSt, 1981 const MachineOperand *&BaseOp, 1982 int64_t &Offset, 1983 const TargetRegisterInfo *TRI) const { 1984 unsigned Width; 1985 return getMemOperandWithOffsetWidth(LdSt, BaseOp, Offset, Width, TRI); 1986 } 1987 1988 bool AArch64InstrInfo::getMemOperandWithOffsetWidth( 1989 const MachineInstr &LdSt, const MachineOperand *&BaseOp, int64_t &Offset, 1990 unsigned &Width, const TargetRegisterInfo *TRI) const { 1991 assert(LdSt.mayLoadOrStore() && "Expected a memory operation."); 1992 // Handle only loads/stores with base register followed by immediate offset. 1993 if (LdSt.getNumExplicitOperands() == 3) { 1994 // Non-paired instruction (e.g., ldr x1, [x0, #8]). 1995 if ((!LdSt.getOperand(1).isReg() && !LdSt.getOperand(1).isFI()) || 1996 !LdSt.getOperand(2).isImm()) 1997 return false; 1998 } else if (LdSt.getNumExplicitOperands() == 4) { 1999 // Paired instruction (e.g., ldp x1, x2, [x0, #8]). 2000 if (!LdSt.getOperand(1).isReg() || 2001 (!LdSt.getOperand(2).isReg() && !LdSt.getOperand(2).isFI()) || 2002 !LdSt.getOperand(3).isImm()) 2003 return false; 2004 } else 2005 return false; 2006 2007 // Get the scaling factor for the instruction and set the width for the 2008 // instruction. 2009 unsigned Scale = 0; 2010 int64_t Dummy1, Dummy2; 2011 2012 // If this returns false, then it's an instruction we don't want to handle. 2013 if (!getMemOpInfo(LdSt.getOpcode(), Scale, Width, Dummy1, Dummy2)) 2014 return false; 2015 2016 // Compute the offset. Offset is calculated as the immediate operand 2017 // multiplied by the scaling factor. Unscaled instructions have scaling factor 2018 // set to 1. 2019 if (LdSt.getNumExplicitOperands() == 3) { 2020 BaseOp = &LdSt.getOperand(1); 2021 Offset = LdSt.getOperand(2).getImm() * Scale; 2022 } else { 2023 assert(LdSt.getNumExplicitOperands() == 4 && "invalid number of operands"); 2024 BaseOp = &LdSt.getOperand(2); 2025 Offset = LdSt.getOperand(3).getImm() * Scale; 2026 } 2027 2028 assert((BaseOp->isReg() || BaseOp->isFI()) && 2029 "getMemOperandWithOffset only supports base " 2030 "operands of type register or frame index."); 2031 2032 return true; 2033 } 2034 2035 MachineOperand & 2036 AArch64InstrInfo::getMemOpBaseRegImmOfsOffsetOperand(MachineInstr &LdSt) const { 2037 assert(LdSt.mayLoadOrStore() && "Expected a memory operation."); 2038 MachineOperand &OfsOp = LdSt.getOperand(LdSt.getNumExplicitOperands() - 1); 2039 assert(OfsOp.isImm() && "Offset operand wasn't immediate."); 2040 return OfsOp; 2041 } 2042 2043 bool AArch64InstrInfo::getMemOpInfo(unsigned Opcode, unsigned &Scale, 2044 unsigned &Width, int64_t &MinOffset, 2045 int64_t &MaxOffset) { 2046 switch (Opcode) { 2047 // Not a memory operation or something we want to handle. 2048 default: 2049 Scale = Width = 0; 2050 MinOffset = MaxOffset = 0; 2051 return false; 2052 case AArch64::STRWpost: 2053 case AArch64::LDRWpost: 2054 Width = 32; 2055 Scale = 4; 2056 MinOffset = -256; 2057 MaxOffset = 255; 2058 break; 2059 case AArch64::LDURQi: 2060 case AArch64::STURQi: 2061 Width = 16; 2062 Scale = 1; 2063 MinOffset = -256; 2064 MaxOffset = 255; 2065 break; 2066 case AArch64::PRFUMi: 2067 case AArch64::LDURXi: 2068 case AArch64::LDURDi: 2069 case AArch64::STURXi: 2070 case AArch64::STURDi: 2071 Width = 8; 2072 Scale = 1; 2073 MinOffset = -256; 2074 MaxOffset = 255; 2075 break; 2076 case AArch64::LDURWi: 2077 case AArch64::LDURSi: 2078 case AArch64::LDURSWi: 2079 case AArch64::STURWi: 2080 case AArch64::STURSi: 2081 Width = 4; 2082 Scale = 1; 2083 MinOffset = -256; 2084 MaxOffset = 255; 2085 break; 2086 case AArch64::LDURHi: 2087 case AArch64::LDURHHi: 2088 case AArch64::LDURSHXi: 2089 case AArch64::LDURSHWi: 2090 case AArch64::STURHi: 2091 case AArch64::STURHHi: 2092 Width = 2; 2093 Scale = 1; 2094 MinOffset = -256; 2095 MaxOffset = 255; 2096 break; 2097 case AArch64::LDURBi: 2098 case AArch64::LDURBBi: 2099 case AArch64::LDURSBXi: 2100 case AArch64::LDURSBWi: 2101 case AArch64::STURBi: 2102 case AArch64::STURBBi: 2103 Width = 1; 2104 Scale = 1; 2105 MinOffset = -256; 2106 MaxOffset = 255; 2107 break; 2108 case AArch64::LDPQi: 2109 case AArch64::LDNPQi: 2110 case AArch64::STPQi: 2111 case AArch64::STNPQi: 2112 Scale = 16; 2113 Width = 32; 2114 MinOffset = -64; 2115 MaxOffset = 63; 2116 break; 2117 case AArch64::LDRQui: 2118 case AArch64::STRQui: 2119 Scale = Width = 16; 2120 MinOffset = 0; 2121 MaxOffset = 4095; 2122 break; 2123 case AArch64::LDPXi: 2124 case AArch64::LDPDi: 2125 case AArch64::LDNPXi: 2126 case AArch64::LDNPDi: 2127 case AArch64::STPXi: 2128 case AArch64::STPDi: 2129 case AArch64::STNPXi: 2130 case AArch64::STNPDi: 2131 Scale = 8; 2132 Width = 16; 2133 MinOffset = -64; 2134 MaxOffset = 63; 2135 break; 2136 case AArch64::PRFMui: 2137 case AArch64::LDRXui: 2138 case AArch64::LDRDui: 2139 case AArch64::STRXui: 2140 case AArch64::STRDui: 2141 Scale = Width = 8; 2142 MinOffset = 0; 2143 MaxOffset = 4095; 2144 break; 2145 case AArch64::LDPWi: 2146 case AArch64::LDPSi: 2147 case AArch64::LDNPWi: 2148 case AArch64::LDNPSi: 2149 case AArch64::STPWi: 2150 case AArch64::STPSi: 2151 case AArch64::STNPWi: 2152 case AArch64::STNPSi: 2153 Scale = 4; 2154 Width = 8; 2155 MinOffset = -64; 2156 MaxOffset = 63; 2157 break; 2158 case AArch64::LDRWui: 2159 case AArch64::LDRSui: 2160 case AArch64::LDRSWui: 2161 case AArch64::STRWui: 2162 case AArch64::STRSui: 2163 Scale = Width = 4; 2164 MinOffset = 0; 2165 MaxOffset = 4095; 2166 break; 2167 case AArch64::LDRHui: 2168 case AArch64::LDRHHui: 2169 case AArch64::LDRSHWui: 2170 case AArch64::LDRSHXui: 2171 case AArch64::STRHui: 2172 case AArch64::STRHHui: 2173 Scale = Width = 2; 2174 MinOffset = 0; 2175 MaxOffset = 4095; 2176 break; 2177 case AArch64::LDRBui: 2178 case AArch64::LDRBBui: 2179 case AArch64::LDRSBWui: 2180 case AArch64::LDRSBXui: 2181 case AArch64::STRBui: 2182 case AArch64::STRBBui: 2183 Scale = Width = 1; 2184 MinOffset = 0; 2185 MaxOffset = 4095; 2186 break; 2187 case AArch64::ADDG: 2188 case AArch64::TAGPstack: 2189 Scale = 16; 2190 Width = 0; 2191 MinOffset = 0; 2192 MaxOffset = 63; 2193 break; 2194 case AArch64::LDG: 2195 case AArch64::STGOffset: 2196 case AArch64::STZGOffset: 2197 Scale = Width = 16; 2198 MinOffset = -256; 2199 MaxOffset = 255; 2200 break; 2201 case AArch64::ST2GOffset: 2202 case AArch64::STZ2GOffset: 2203 Scale = 16; 2204 Width = 32; 2205 MinOffset = -256; 2206 MaxOffset = 255; 2207 break; 2208 case AArch64::STGPi: 2209 Scale = Width = 16; 2210 MinOffset = -64; 2211 MaxOffset = 63; 2212 break; 2213 } 2214 2215 return true; 2216 } 2217 2218 static unsigned getOffsetStride(unsigned Opc) { 2219 switch (Opc) { 2220 default: 2221 return 0; 2222 case AArch64::LDURQi: 2223 case AArch64::STURQi: 2224 return 16; 2225 case AArch64::LDURXi: 2226 case AArch64::LDURDi: 2227 case AArch64::STURXi: 2228 case AArch64::STURDi: 2229 return 8; 2230 case AArch64::LDURWi: 2231 case AArch64::LDURSi: 2232 case AArch64::LDURSWi: 2233 case AArch64::STURWi: 2234 case AArch64::STURSi: 2235 return 4; 2236 } 2237 } 2238 2239 // Scale the unscaled offsets. Returns false if the unscaled offset can't be 2240 // scaled. 2241 static bool scaleOffset(unsigned Opc, int64_t &Offset) { 2242 unsigned OffsetStride = getOffsetStride(Opc); 2243 if (OffsetStride == 0) 2244 return false; 2245 // If the byte-offset isn't a multiple of the stride, we can't scale this 2246 // offset. 2247 if (Offset % OffsetStride != 0) 2248 return false; 2249 2250 // Convert the byte-offset used by unscaled into an "element" offset used 2251 // by the scaled pair load/store instructions. 2252 Offset /= OffsetStride; 2253 return true; 2254 } 2255 2256 // Unscale the scaled offsets. Returns false if the scaled offset can't be 2257 // unscaled. 2258 static bool unscaleOffset(unsigned Opc, int64_t &Offset) { 2259 unsigned OffsetStride = getOffsetStride(Opc); 2260 if (OffsetStride == 0) 2261 return false; 2262 2263 // Convert the "element" offset used by scaled pair load/store instructions 2264 // into the byte-offset used by unscaled. 2265 Offset *= OffsetStride; 2266 return true; 2267 } 2268 2269 static bool canPairLdStOpc(unsigned FirstOpc, unsigned SecondOpc) { 2270 if (FirstOpc == SecondOpc) 2271 return true; 2272 // We can also pair sign-ext and zero-ext instructions. 2273 switch (FirstOpc) { 2274 default: 2275 return false; 2276 case AArch64::LDRWui: 2277 case AArch64::LDURWi: 2278 return SecondOpc == AArch64::LDRSWui || SecondOpc == AArch64::LDURSWi; 2279 case AArch64::LDRSWui: 2280 case AArch64::LDURSWi: 2281 return SecondOpc == AArch64::LDRWui || SecondOpc == AArch64::LDURWi; 2282 } 2283 // These instructions can't be paired based on their opcodes. 2284 return false; 2285 } 2286 2287 static bool shouldClusterFI(const MachineFrameInfo &MFI, int FI1, 2288 int64_t Offset1, unsigned Opcode1, int FI2, 2289 int64_t Offset2, unsigned Opcode2) { 2290 // Accesses through fixed stack object frame indices may access a different 2291 // fixed stack slot. Check that the object offsets + offsets match. 2292 if (MFI.isFixedObjectIndex(FI1) && MFI.isFixedObjectIndex(FI2)) { 2293 int64_t ObjectOffset1 = MFI.getObjectOffset(FI1); 2294 int64_t ObjectOffset2 = MFI.getObjectOffset(FI2); 2295 assert(ObjectOffset1 <= ObjectOffset2 && "Object offsets are not ordered."); 2296 // Get the byte-offset from the object offset. 2297 if (!unscaleOffset(Opcode1, Offset1) || !unscaleOffset(Opcode2, Offset2)) 2298 return false; 2299 ObjectOffset1 += Offset1; 2300 ObjectOffset2 += Offset2; 2301 // Get the "element" index in the object. 2302 if (!scaleOffset(Opcode1, ObjectOffset1) || 2303 !scaleOffset(Opcode2, ObjectOffset2)) 2304 return false; 2305 return ObjectOffset1 + 1 == ObjectOffset2; 2306 } 2307 2308 return FI1 == FI2; 2309 } 2310 2311 /// Detect opportunities for ldp/stp formation. 2312 /// 2313 /// Only called for LdSt for which getMemOperandWithOffset returns true. 2314 bool AArch64InstrInfo::shouldClusterMemOps(const MachineOperand &BaseOp1, 2315 const MachineOperand &BaseOp2, 2316 unsigned NumLoads) const { 2317 const MachineInstr &FirstLdSt = *BaseOp1.getParent(); 2318 const MachineInstr &SecondLdSt = *BaseOp2.getParent(); 2319 if (BaseOp1.getType() != BaseOp2.getType()) 2320 return false; 2321 2322 assert((BaseOp1.isReg() || BaseOp1.isFI()) && 2323 "Only base registers and frame indices are supported."); 2324 2325 // Check for both base regs and base FI. 2326 if (BaseOp1.isReg() && BaseOp1.getReg() != BaseOp2.getReg()) 2327 return false; 2328 2329 // Only cluster up to a single pair. 2330 if (NumLoads > 1) 2331 return false; 2332 2333 if (!isPairableLdStInst(FirstLdSt) || !isPairableLdStInst(SecondLdSt)) 2334 return false; 2335 2336 // Can we pair these instructions based on their opcodes? 2337 unsigned FirstOpc = FirstLdSt.getOpcode(); 2338 unsigned SecondOpc = SecondLdSt.getOpcode(); 2339 if (!canPairLdStOpc(FirstOpc, SecondOpc)) 2340 return false; 2341 2342 // Can't merge volatiles or load/stores that have a hint to avoid pair 2343 // formation, for example. 2344 if (!isCandidateToMergeOrPair(FirstLdSt) || 2345 !isCandidateToMergeOrPair(SecondLdSt)) 2346 return false; 2347 2348 // isCandidateToMergeOrPair guarantees that operand 2 is an immediate. 2349 int64_t Offset1 = FirstLdSt.getOperand(2).getImm(); 2350 if (isUnscaledLdSt(FirstOpc) && !scaleOffset(FirstOpc, Offset1)) 2351 return false; 2352 2353 int64_t Offset2 = SecondLdSt.getOperand(2).getImm(); 2354 if (isUnscaledLdSt(SecondOpc) && !scaleOffset(SecondOpc, Offset2)) 2355 return false; 2356 2357 // Pairwise instructions have a 7-bit signed offset field. 2358 if (Offset1 > 63 || Offset1 < -64) 2359 return false; 2360 2361 // The caller should already have ordered First/SecondLdSt by offset. 2362 // Note: except for non-equal frame index bases 2363 if (BaseOp1.isFI()) { 2364 assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 >= Offset2) && 2365 "Caller should have ordered offsets."); 2366 2367 const MachineFrameInfo &MFI = 2368 FirstLdSt.getParent()->getParent()->getFrameInfo(); 2369 return shouldClusterFI(MFI, BaseOp1.getIndex(), Offset1, FirstOpc, 2370 BaseOp2.getIndex(), Offset2, SecondOpc); 2371 } 2372 2373 assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 <= Offset2) && 2374 "Caller should have ordered offsets."); 2375 2376 return Offset1 + 1 == Offset2; 2377 } 2378 2379 static const MachineInstrBuilder &AddSubReg(const MachineInstrBuilder &MIB, 2380 unsigned Reg, unsigned SubIdx, 2381 unsigned State, 2382 const TargetRegisterInfo *TRI) { 2383 if (!SubIdx) 2384 return MIB.addReg(Reg, State); 2385 2386 if (Register::isPhysicalRegister(Reg)) 2387 return MIB.addReg(TRI->getSubReg(Reg, SubIdx), State); 2388 return MIB.addReg(Reg, State, SubIdx); 2389 } 2390 2391 static bool forwardCopyWillClobberTuple(unsigned DestReg, unsigned SrcReg, 2392 unsigned NumRegs) { 2393 // We really want the positive remainder mod 32 here, that happens to be 2394 // easily obtainable with a mask. 2395 return ((DestReg - SrcReg) & 0x1f) < NumRegs; 2396 } 2397 2398 void AArch64InstrInfo::copyPhysRegTuple(MachineBasicBlock &MBB, 2399 MachineBasicBlock::iterator I, 2400 const DebugLoc &DL, unsigned DestReg, 2401 unsigned SrcReg, bool KillSrc, 2402 unsigned Opcode, 2403 ArrayRef<unsigned> Indices) const { 2404 assert(Subtarget.hasNEON() && "Unexpected register copy without NEON"); 2405 const TargetRegisterInfo *TRI = &getRegisterInfo(); 2406 uint16_t DestEncoding = TRI->getEncodingValue(DestReg); 2407 uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg); 2408 unsigned NumRegs = Indices.size(); 2409 2410 int SubReg = 0, End = NumRegs, Incr = 1; 2411 if (forwardCopyWillClobberTuple(DestEncoding, SrcEncoding, NumRegs)) { 2412 SubReg = NumRegs - 1; 2413 End = -1; 2414 Incr = -1; 2415 } 2416 2417 for (; SubReg != End; SubReg += Incr) { 2418 const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode)); 2419 AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI); 2420 AddSubReg(MIB, SrcReg, Indices[SubReg], 0, TRI); 2421 AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI); 2422 } 2423 } 2424 2425 void AArch64InstrInfo::copyGPRRegTuple(MachineBasicBlock &MBB, 2426 MachineBasicBlock::iterator I, 2427 DebugLoc DL, unsigned DestReg, 2428 unsigned SrcReg, bool KillSrc, 2429 unsigned Opcode, unsigned ZeroReg, 2430 llvm::ArrayRef<unsigned> Indices) const { 2431 const TargetRegisterInfo *TRI = &getRegisterInfo(); 2432 unsigned NumRegs = Indices.size(); 2433 2434 #ifndef NDEBUG 2435 uint16_t DestEncoding = TRI->getEncodingValue(DestReg); 2436 uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg); 2437 assert(DestEncoding % NumRegs == 0 && SrcEncoding % NumRegs == 0 && 2438 "GPR reg sequences should not be able to overlap"); 2439 #endif 2440 2441 for (unsigned SubReg = 0; SubReg != NumRegs; ++SubReg) { 2442 const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode)); 2443 AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI); 2444 MIB.addReg(ZeroReg); 2445 AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI); 2446 MIB.addImm(0); 2447 } 2448 } 2449 2450 void AArch64InstrInfo::copyPhysReg(MachineBasicBlock &MBB, 2451 MachineBasicBlock::iterator I, 2452 const DebugLoc &DL, unsigned DestReg, 2453 unsigned SrcReg, bool KillSrc) const { 2454 if (AArch64::GPR32spRegClass.contains(DestReg) && 2455 (AArch64::GPR32spRegClass.contains(SrcReg) || SrcReg == AArch64::WZR)) { 2456 const TargetRegisterInfo *TRI = &getRegisterInfo(); 2457 2458 if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) { 2459 // If either operand is WSP, expand to ADD #0. 2460 if (Subtarget.hasZeroCycleRegMove()) { 2461 // Cyclone recognizes "ADD Xd, Xn, #0" as a zero-cycle register move. 2462 unsigned DestRegX = TRI->getMatchingSuperReg(DestReg, AArch64::sub_32, 2463 &AArch64::GPR64spRegClass); 2464 unsigned SrcRegX = TRI->getMatchingSuperReg(SrcReg, AArch64::sub_32, 2465 &AArch64::GPR64spRegClass); 2466 // This instruction is reading and writing X registers. This may upset 2467 // the register scavenger and machine verifier, so we need to indicate 2468 // that we are reading an undefined value from SrcRegX, but a proper 2469 // value from SrcReg. 2470 BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestRegX) 2471 .addReg(SrcRegX, RegState::Undef) 2472 .addImm(0) 2473 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)) 2474 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc)); 2475 } else { 2476 BuildMI(MBB, I, DL, get(AArch64::ADDWri), DestReg) 2477 .addReg(SrcReg, getKillRegState(KillSrc)) 2478 .addImm(0) 2479 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2480 } 2481 } else if (SrcReg == AArch64::WZR && Subtarget.hasZeroCycleZeroingGP()) { 2482 BuildMI(MBB, I, DL, get(AArch64::MOVZWi), DestReg) 2483 .addImm(0) 2484 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2485 } else { 2486 if (Subtarget.hasZeroCycleRegMove()) { 2487 // Cyclone recognizes "ORR Xd, XZR, Xm" as a zero-cycle register move. 2488 unsigned DestRegX = TRI->getMatchingSuperReg(DestReg, AArch64::sub_32, 2489 &AArch64::GPR64spRegClass); 2490 unsigned SrcRegX = TRI->getMatchingSuperReg(SrcReg, AArch64::sub_32, 2491 &AArch64::GPR64spRegClass); 2492 // This instruction is reading and writing X registers. This may upset 2493 // the register scavenger and machine verifier, so we need to indicate 2494 // that we are reading an undefined value from SrcRegX, but a proper 2495 // value from SrcReg. 2496 BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestRegX) 2497 .addReg(AArch64::XZR) 2498 .addReg(SrcRegX, RegState::Undef) 2499 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc)); 2500 } else { 2501 // Otherwise, expand to ORR WZR. 2502 BuildMI(MBB, I, DL, get(AArch64::ORRWrr), DestReg) 2503 .addReg(AArch64::WZR) 2504 .addReg(SrcReg, getKillRegState(KillSrc)); 2505 } 2506 } 2507 return; 2508 } 2509 2510 // Copy a Z register by ORRing with itself. 2511 if (AArch64::ZPRRegClass.contains(DestReg) && 2512 AArch64::ZPRRegClass.contains(SrcReg)) { 2513 assert(Subtarget.hasSVE() && "Unexpected SVE register."); 2514 BuildMI(MBB, I, DL, get(AArch64::ORR_ZZZ), DestReg) 2515 .addReg(SrcReg) 2516 .addReg(SrcReg, getKillRegState(KillSrc)); 2517 return; 2518 } 2519 2520 if (AArch64::GPR64spRegClass.contains(DestReg) && 2521 (AArch64::GPR64spRegClass.contains(SrcReg) || SrcReg == AArch64::XZR)) { 2522 if (DestReg == AArch64::SP || SrcReg == AArch64::SP) { 2523 // If either operand is SP, expand to ADD #0. 2524 BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestReg) 2525 .addReg(SrcReg, getKillRegState(KillSrc)) 2526 .addImm(0) 2527 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2528 } else if (SrcReg == AArch64::XZR && Subtarget.hasZeroCycleZeroingGP()) { 2529 BuildMI(MBB, I, DL, get(AArch64::MOVZXi), DestReg) 2530 .addImm(0) 2531 .addImm(AArch64_AM::getShifterImm(AArch64_AM::LSL, 0)); 2532 } else { 2533 // Otherwise, expand to ORR XZR. 2534 BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestReg) 2535 .addReg(AArch64::XZR) 2536 .addReg(SrcReg, getKillRegState(KillSrc)); 2537 } 2538 return; 2539 } 2540 2541 // Copy a DDDD register quad by copying the individual sub-registers. 2542 if (AArch64::DDDDRegClass.contains(DestReg) && 2543 AArch64::DDDDRegClass.contains(SrcReg)) { 2544 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1, 2545 AArch64::dsub2, AArch64::dsub3}; 2546 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8, 2547 Indices); 2548 return; 2549 } 2550 2551 // Copy a DDD register triple by copying the individual sub-registers. 2552 if (AArch64::DDDRegClass.contains(DestReg) && 2553 AArch64::DDDRegClass.contains(SrcReg)) { 2554 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1, 2555 AArch64::dsub2}; 2556 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8, 2557 Indices); 2558 return; 2559 } 2560 2561 // Copy a DD register pair by copying the individual sub-registers. 2562 if (AArch64::DDRegClass.contains(DestReg) && 2563 AArch64::DDRegClass.contains(SrcReg)) { 2564 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1}; 2565 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv8i8, 2566 Indices); 2567 return; 2568 } 2569 2570 // Copy a QQQQ register quad by copying the individual sub-registers. 2571 if (AArch64::QQQQRegClass.contains(DestReg) && 2572 AArch64::QQQQRegClass.contains(SrcReg)) { 2573 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1, 2574 AArch64::qsub2, AArch64::qsub3}; 2575 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8, 2576 Indices); 2577 return; 2578 } 2579 2580 // Copy a QQQ register triple by copying the individual sub-registers. 2581 if (AArch64::QQQRegClass.contains(DestReg) && 2582 AArch64::QQQRegClass.contains(SrcReg)) { 2583 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1, 2584 AArch64::qsub2}; 2585 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8, 2586 Indices); 2587 return; 2588 } 2589 2590 // Copy a QQ register pair by copying the individual sub-registers. 2591 if (AArch64::QQRegClass.contains(DestReg) && 2592 AArch64::QQRegClass.contains(SrcReg)) { 2593 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1}; 2594 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRv16i8, 2595 Indices); 2596 return; 2597 } 2598 2599 if (AArch64::XSeqPairsClassRegClass.contains(DestReg) && 2600 AArch64::XSeqPairsClassRegClass.contains(SrcReg)) { 2601 static const unsigned Indices[] = {AArch64::sube64, AArch64::subo64}; 2602 copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRXrs, 2603 AArch64::XZR, Indices); 2604 return; 2605 } 2606 2607 if (AArch64::WSeqPairsClassRegClass.contains(DestReg) && 2608 AArch64::WSeqPairsClassRegClass.contains(SrcReg)) { 2609 static const unsigned Indices[] = {AArch64::sube32, AArch64::subo32}; 2610 copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRWrs, 2611 AArch64::WZR, Indices); 2612 return; 2613 } 2614 2615 if (AArch64::FPR128RegClass.contains(DestReg) && 2616 AArch64::FPR128RegClass.contains(SrcReg)) { 2617 if (Subtarget.hasNEON()) { 2618 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2619 .addReg(SrcReg) 2620 .addReg(SrcReg, getKillRegState(KillSrc)); 2621 } else { 2622 BuildMI(MBB, I, DL, get(AArch64::STRQpre)) 2623 .addReg(AArch64::SP, RegState::Define) 2624 .addReg(SrcReg, getKillRegState(KillSrc)) 2625 .addReg(AArch64::SP) 2626 .addImm(-16); 2627 BuildMI(MBB, I, DL, get(AArch64::LDRQpre)) 2628 .addReg(AArch64::SP, RegState::Define) 2629 .addReg(DestReg, RegState::Define) 2630 .addReg(AArch64::SP) 2631 .addImm(16); 2632 } 2633 return; 2634 } 2635 2636 if (AArch64::FPR64RegClass.contains(DestReg) && 2637 AArch64::FPR64RegClass.contains(SrcReg)) { 2638 if (Subtarget.hasNEON()) { 2639 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::dsub, 2640 &AArch64::FPR128RegClass); 2641 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::dsub, 2642 &AArch64::FPR128RegClass); 2643 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2644 .addReg(SrcReg) 2645 .addReg(SrcReg, getKillRegState(KillSrc)); 2646 } else { 2647 BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestReg) 2648 .addReg(SrcReg, getKillRegState(KillSrc)); 2649 } 2650 return; 2651 } 2652 2653 if (AArch64::FPR32RegClass.contains(DestReg) && 2654 AArch64::FPR32RegClass.contains(SrcReg)) { 2655 if (Subtarget.hasNEON()) { 2656 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::ssub, 2657 &AArch64::FPR128RegClass); 2658 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::ssub, 2659 &AArch64::FPR128RegClass); 2660 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2661 .addReg(SrcReg) 2662 .addReg(SrcReg, getKillRegState(KillSrc)); 2663 } else { 2664 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg) 2665 .addReg(SrcReg, getKillRegState(KillSrc)); 2666 } 2667 return; 2668 } 2669 2670 if (AArch64::FPR16RegClass.contains(DestReg) && 2671 AArch64::FPR16RegClass.contains(SrcReg)) { 2672 if (Subtarget.hasNEON()) { 2673 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub, 2674 &AArch64::FPR128RegClass); 2675 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub, 2676 &AArch64::FPR128RegClass); 2677 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2678 .addReg(SrcReg) 2679 .addReg(SrcReg, getKillRegState(KillSrc)); 2680 } else { 2681 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub, 2682 &AArch64::FPR32RegClass); 2683 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub, 2684 &AArch64::FPR32RegClass); 2685 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg) 2686 .addReg(SrcReg, getKillRegState(KillSrc)); 2687 } 2688 return; 2689 } 2690 2691 if (AArch64::FPR8RegClass.contains(DestReg) && 2692 AArch64::FPR8RegClass.contains(SrcReg)) { 2693 if (Subtarget.hasNEON()) { 2694 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub, 2695 &AArch64::FPR128RegClass); 2696 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub, 2697 &AArch64::FPR128RegClass); 2698 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg) 2699 .addReg(SrcReg) 2700 .addReg(SrcReg, getKillRegState(KillSrc)); 2701 } else { 2702 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub, 2703 &AArch64::FPR32RegClass); 2704 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub, 2705 &AArch64::FPR32RegClass); 2706 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg) 2707 .addReg(SrcReg, getKillRegState(KillSrc)); 2708 } 2709 return; 2710 } 2711 2712 // Copies between GPR64 and FPR64. 2713 if (AArch64::FPR64RegClass.contains(DestReg) && 2714 AArch64::GPR64RegClass.contains(SrcReg)) { 2715 BuildMI(MBB, I, DL, get(AArch64::FMOVXDr), DestReg) 2716 .addReg(SrcReg, getKillRegState(KillSrc)); 2717 return; 2718 } 2719 if (AArch64::GPR64RegClass.contains(DestReg) && 2720 AArch64::FPR64RegClass.contains(SrcReg)) { 2721 BuildMI(MBB, I, DL, get(AArch64::FMOVDXr), DestReg) 2722 .addReg(SrcReg, getKillRegState(KillSrc)); 2723 return; 2724 } 2725 // Copies between GPR32 and FPR32. 2726 if (AArch64::FPR32RegClass.contains(DestReg) && 2727 AArch64::GPR32RegClass.contains(SrcReg)) { 2728 BuildMI(MBB, I, DL, get(AArch64::FMOVWSr), DestReg) 2729 .addReg(SrcReg, getKillRegState(KillSrc)); 2730 return; 2731 } 2732 if (AArch64::GPR32RegClass.contains(DestReg) && 2733 AArch64::FPR32RegClass.contains(SrcReg)) { 2734 BuildMI(MBB, I, DL, get(AArch64::FMOVSWr), DestReg) 2735 .addReg(SrcReg, getKillRegState(KillSrc)); 2736 return; 2737 } 2738 2739 if (DestReg == AArch64::NZCV) { 2740 assert(AArch64::GPR64RegClass.contains(SrcReg) && "Invalid NZCV copy"); 2741 BuildMI(MBB, I, DL, get(AArch64::MSR)) 2742 .addImm(AArch64SysReg::NZCV) 2743 .addReg(SrcReg, getKillRegState(KillSrc)) 2744 .addReg(AArch64::NZCV, RegState::Implicit | RegState::Define); 2745 return; 2746 } 2747 2748 if (SrcReg == AArch64::NZCV) { 2749 assert(AArch64::GPR64RegClass.contains(DestReg) && "Invalid NZCV copy"); 2750 BuildMI(MBB, I, DL, get(AArch64::MRS), DestReg) 2751 .addImm(AArch64SysReg::NZCV) 2752 .addReg(AArch64::NZCV, RegState::Implicit | getKillRegState(KillSrc)); 2753 return; 2754 } 2755 2756 llvm_unreachable("unimplemented reg-to-reg copy"); 2757 } 2758 2759 static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI, 2760 MachineBasicBlock &MBB, 2761 MachineBasicBlock::iterator InsertBefore, 2762 const MCInstrDesc &MCID, 2763 unsigned SrcReg, bool IsKill, 2764 unsigned SubIdx0, unsigned SubIdx1, int FI, 2765 MachineMemOperand *MMO) { 2766 unsigned SrcReg0 = SrcReg; 2767 unsigned SrcReg1 = SrcReg; 2768 if (Register::isPhysicalRegister(SrcReg)) { 2769 SrcReg0 = TRI.getSubReg(SrcReg, SubIdx0); 2770 SubIdx0 = 0; 2771 SrcReg1 = TRI.getSubReg(SrcReg, SubIdx1); 2772 SubIdx1 = 0; 2773 } 2774 BuildMI(MBB, InsertBefore, DebugLoc(), MCID) 2775 .addReg(SrcReg0, getKillRegState(IsKill), SubIdx0) 2776 .addReg(SrcReg1, getKillRegState(IsKill), SubIdx1) 2777 .addFrameIndex(FI) 2778 .addImm(0) 2779 .addMemOperand(MMO); 2780 } 2781 2782 void AArch64InstrInfo::storeRegToStackSlot( 2783 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned SrcReg, 2784 bool isKill, int FI, const TargetRegisterClass *RC, 2785 const TargetRegisterInfo *TRI) const { 2786 MachineFunction &MF = *MBB.getParent(); 2787 MachineFrameInfo &MFI = MF.getFrameInfo(); 2788 unsigned Align = MFI.getObjectAlignment(FI); 2789 2790 MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI); 2791 MachineMemOperand *MMO = MF.getMachineMemOperand( 2792 PtrInfo, MachineMemOperand::MOStore, MFI.getObjectSize(FI), Align); 2793 unsigned Opc = 0; 2794 bool Offset = true; 2795 switch (TRI->getSpillSize(*RC)) { 2796 case 1: 2797 if (AArch64::FPR8RegClass.hasSubClassEq(RC)) 2798 Opc = AArch64::STRBui; 2799 break; 2800 case 2: 2801 if (AArch64::FPR16RegClass.hasSubClassEq(RC)) 2802 Opc = AArch64::STRHui; 2803 break; 2804 case 4: 2805 if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) { 2806 Opc = AArch64::STRWui; 2807 if (Register::isVirtualRegister(SrcReg)) 2808 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR32RegClass); 2809 else 2810 assert(SrcReg != AArch64::WSP); 2811 } else if (AArch64::FPR32RegClass.hasSubClassEq(RC)) 2812 Opc = AArch64::STRSui; 2813 break; 2814 case 8: 2815 if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) { 2816 Opc = AArch64::STRXui; 2817 if (Register::isVirtualRegister(SrcReg)) 2818 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass); 2819 else 2820 assert(SrcReg != AArch64::SP); 2821 } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) { 2822 Opc = AArch64::STRDui; 2823 } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) { 2824 storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI, 2825 get(AArch64::STPWi), SrcReg, isKill, 2826 AArch64::sube32, AArch64::subo32, FI, MMO); 2827 return; 2828 } 2829 break; 2830 case 16: 2831 if (AArch64::FPR128RegClass.hasSubClassEq(RC)) 2832 Opc = AArch64::STRQui; 2833 else if (AArch64::DDRegClass.hasSubClassEq(RC)) { 2834 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2835 Opc = AArch64::ST1Twov1d; 2836 Offset = false; 2837 } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) { 2838 storeRegPairToStackSlot(getRegisterInfo(), MBB, MBBI, 2839 get(AArch64::STPXi), SrcReg, isKill, 2840 AArch64::sube64, AArch64::subo64, FI, MMO); 2841 return; 2842 } 2843 break; 2844 case 24: 2845 if (AArch64::DDDRegClass.hasSubClassEq(RC)) { 2846 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2847 Opc = AArch64::ST1Threev1d; 2848 Offset = false; 2849 } 2850 break; 2851 case 32: 2852 if (AArch64::DDDDRegClass.hasSubClassEq(RC)) { 2853 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2854 Opc = AArch64::ST1Fourv1d; 2855 Offset = false; 2856 } else if (AArch64::QQRegClass.hasSubClassEq(RC)) { 2857 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2858 Opc = AArch64::ST1Twov2d; 2859 Offset = false; 2860 } 2861 break; 2862 case 48: 2863 if (AArch64::QQQRegClass.hasSubClassEq(RC)) { 2864 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2865 Opc = AArch64::ST1Threev2d; 2866 Offset = false; 2867 } 2868 break; 2869 case 64: 2870 if (AArch64::QQQQRegClass.hasSubClassEq(RC)) { 2871 assert(Subtarget.hasNEON() && "Unexpected register store without NEON"); 2872 Opc = AArch64::ST1Fourv2d; 2873 Offset = false; 2874 } 2875 break; 2876 } 2877 assert(Opc && "Unknown register class"); 2878 2879 const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc)) 2880 .addReg(SrcReg, getKillRegState(isKill)) 2881 .addFrameIndex(FI); 2882 2883 if (Offset) 2884 MI.addImm(0); 2885 MI.addMemOperand(MMO); 2886 } 2887 2888 static void loadRegPairFromStackSlot(const TargetRegisterInfo &TRI, 2889 MachineBasicBlock &MBB, 2890 MachineBasicBlock::iterator InsertBefore, 2891 const MCInstrDesc &MCID, 2892 unsigned DestReg, unsigned SubIdx0, 2893 unsigned SubIdx1, int FI, 2894 MachineMemOperand *MMO) { 2895 unsigned DestReg0 = DestReg; 2896 unsigned DestReg1 = DestReg; 2897 bool IsUndef = true; 2898 if (Register::isPhysicalRegister(DestReg)) { 2899 DestReg0 = TRI.getSubReg(DestReg, SubIdx0); 2900 SubIdx0 = 0; 2901 DestReg1 = TRI.getSubReg(DestReg, SubIdx1); 2902 SubIdx1 = 0; 2903 IsUndef = false; 2904 } 2905 BuildMI(MBB, InsertBefore, DebugLoc(), MCID) 2906 .addReg(DestReg0, RegState::Define | getUndefRegState(IsUndef), SubIdx0) 2907 .addReg(DestReg1, RegState::Define | getUndefRegState(IsUndef), SubIdx1) 2908 .addFrameIndex(FI) 2909 .addImm(0) 2910 .addMemOperand(MMO); 2911 } 2912 2913 void AArch64InstrInfo::loadRegFromStackSlot( 2914 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned DestReg, 2915 int FI, const TargetRegisterClass *RC, 2916 const TargetRegisterInfo *TRI) const { 2917 MachineFunction &MF = *MBB.getParent(); 2918 MachineFrameInfo &MFI = MF.getFrameInfo(); 2919 unsigned Align = MFI.getObjectAlignment(FI); 2920 MachinePointerInfo PtrInfo = MachinePointerInfo::getFixedStack(MF, FI); 2921 MachineMemOperand *MMO = MF.getMachineMemOperand( 2922 PtrInfo, MachineMemOperand::MOLoad, MFI.getObjectSize(FI), Align); 2923 2924 unsigned Opc = 0; 2925 bool Offset = true; 2926 switch (TRI->getSpillSize(*RC)) { 2927 case 1: 2928 if (AArch64::FPR8RegClass.hasSubClassEq(RC)) 2929 Opc = AArch64::LDRBui; 2930 break; 2931 case 2: 2932 if (AArch64::FPR16RegClass.hasSubClassEq(RC)) 2933 Opc = AArch64::LDRHui; 2934 break; 2935 case 4: 2936 if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) { 2937 Opc = AArch64::LDRWui; 2938 if (Register::isVirtualRegister(DestReg)) 2939 MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR32RegClass); 2940 else 2941 assert(DestReg != AArch64::WSP); 2942 } else if (AArch64::FPR32RegClass.hasSubClassEq(RC)) 2943 Opc = AArch64::LDRSui; 2944 break; 2945 case 8: 2946 if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) { 2947 Opc = AArch64::LDRXui; 2948 if (Register::isVirtualRegister(DestReg)) 2949 MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR64RegClass); 2950 else 2951 assert(DestReg != AArch64::SP); 2952 } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) { 2953 Opc = AArch64::LDRDui; 2954 } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) { 2955 loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI, 2956 get(AArch64::LDPWi), DestReg, AArch64::sube32, 2957 AArch64::subo32, FI, MMO); 2958 return; 2959 } 2960 break; 2961 case 16: 2962 if (AArch64::FPR128RegClass.hasSubClassEq(RC)) 2963 Opc = AArch64::LDRQui; 2964 else if (AArch64::DDRegClass.hasSubClassEq(RC)) { 2965 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2966 Opc = AArch64::LD1Twov1d; 2967 Offset = false; 2968 } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) { 2969 loadRegPairFromStackSlot(getRegisterInfo(), MBB, MBBI, 2970 get(AArch64::LDPXi), DestReg, AArch64::sube64, 2971 AArch64::subo64, FI, MMO); 2972 return; 2973 } 2974 break; 2975 case 24: 2976 if (AArch64::DDDRegClass.hasSubClassEq(RC)) { 2977 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2978 Opc = AArch64::LD1Threev1d; 2979 Offset = false; 2980 } 2981 break; 2982 case 32: 2983 if (AArch64::DDDDRegClass.hasSubClassEq(RC)) { 2984 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2985 Opc = AArch64::LD1Fourv1d; 2986 Offset = false; 2987 } else if (AArch64::QQRegClass.hasSubClassEq(RC)) { 2988 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2989 Opc = AArch64::LD1Twov2d; 2990 Offset = false; 2991 } 2992 break; 2993 case 48: 2994 if (AArch64::QQQRegClass.hasSubClassEq(RC)) { 2995 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 2996 Opc = AArch64::LD1Threev2d; 2997 Offset = false; 2998 } 2999 break; 3000 case 64: 3001 if (AArch64::QQQQRegClass.hasSubClassEq(RC)) { 3002 assert(Subtarget.hasNEON() && "Unexpected register load without NEON"); 3003 Opc = AArch64::LD1Fourv2d; 3004 Offset = false; 3005 } 3006 break; 3007 } 3008 assert(Opc && "Unknown register class"); 3009 3010 const MachineInstrBuilder MI = BuildMI(MBB, MBBI, DebugLoc(), get(Opc)) 3011 .addReg(DestReg, getDefRegState(true)) 3012 .addFrameIndex(FI); 3013 if (Offset) 3014 MI.addImm(0); 3015 MI.addMemOperand(MMO); 3016 } 3017 3018 // Helper function to emit a frame offset adjustment from a given 3019 // pointer (SrcReg), stored into DestReg. This function is explicit 3020 // in that it requires the opcode. 3021 static void emitFrameOffsetAdj(MachineBasicBlock &MBB, 3022 MachineBasicBlock::iterator MBBI, 3023 const DebugLoc &DL, unsigned DestReg, 3024 unsigned SrcReg, int64_t Offset, unsigned Opc, 3025 const TargetInstrInfo *TII, 3026 MachineInstr::MIFlag Flag, bool NeedsWinCFI, 3027 bool *HasWinCFI) { 3028 int Sign = 1; 3029 unsigned MaxEncoding, ShiftSize; 3030 switch (Opc) { 3031 case AArch64::ADDXri: 3032 case AArch64::ADDSXri: 3033 case AArch64::SUBXri: 3034 case AArch64::SUBSXri: 3035 MaxEncoding = 0xfff; 3036 ShiftSize = 12; 3037 break; 3038 default: 3039 llvm_unreachable("Unsupported opcode"); 3040 } 3041 3042 // FIXME: If the offset won't fit in 24-bits, compute the offset into a 3043 // scratch register. If DestReg is a virtual register, use it as the 3044 // scratch register; otherwise, create a new virtual register (to be 3045 // replaced by the scavenger at the end of PEI). That case can be optimized 3046 // slightly if DestReg is SP which is always 16-byte aligned, so the scratch 3047 // register can be loaded with offset%8 and the add/sub can use an extending 3048 // instruction with LSL#3. 3049 // Currently the function handles any offsets but generates a poor sequence 3050 // of code. 3051 // assert(Offset < (1 << 24) && "unimplemented reg plus immediate"); 3052 3053 const unsigned MaxEncodableValue = MaxEncoding << ShiftSize; 3054 do { 3055 unsigned ThisVal = std::min<unsigned>(Offset, MaxEncodableValue); 3056 unsigned LocalShiftSize = 0; 3057 if (ThisVal > MaxEncoding) { 3058 ThisVal = ThisVal >> ShiftSize; 3059 LocalShiftSize = ShiftSize; 3060 } 3061 assert((ThisVal >> ShiftSize) <= MaxEncoding && 3062 "Encoding cannot handle value that big"); 3063 auto MBI = BuildMI(MBB, MBBI, DL, TII->get(Opc), DestReg) 3064 .addReg(SrcReg) 3065 .addImm(Sign * (int)ThisVal); 3066 if (ShiftSize) 3067 MBI = MBI.addImm( 3068 AArch64_AM::getShifterImm(AArch64_AM::LSL, LocalShiftSize)); 3069 MBI = MBI.setMIFlag(Flag); 3070 3071 if (NeedsWinCFI) { 3072 assert(Sign == 1 && "SEH directives should always have a positive sign"); 3073 int Imm = (int)(ThisVal << LocalShiftSize); 3074 if ((DestReg == AArch64::FP && SrcReg == AArch64::SP) || 3075 (SrcReg == AArch64::FP && DestReg == AArch64::SP)) { 3076 if (HasWinCFI) 3077 *HasWinCFI = true; 3078 if (Imm == 0) 3079 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_SetFP)).setMIFlag(Flag); 3080 else 3081 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_AddFP)) 3082 .addImm(Imm) 3083 .setMIFlag(Flag); 3084 assert((Offset - Imm) == 0 && "Expected remaining offset to be zero to " 3085 "emit a single SEH directive"); 3086 } else if (DestReg == AArch64::SP) { 3087 if (HasWinCFI) 3088 *HasWinCFI = true; 3089 assert(SrcReg == AArch64::SP && "Unexpected SrcReg for SEH_StackAlloc"); 3090 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_StackAlloc)) 3091 .addImm(Imm) 3092 .setMIFlag(Flag); 3093 } 3094 if (HasWinCFI) 3095 *HasWinCFI = true; 3096 } 3097 3098 SrcReg = DestReg; 3099 Offset -= ThisVal << LocalShiftSize; 3100 } while (Offset); 3101 } 3102 3103 void llvm::emitFrameOffset(MachineBasicBlock &MBB, 3104 MachineBasicBlock::iterator MBBI, const DebugLoc &DL, 3105 unsigned DestReg, unsigned SrcReg, 3106 StackOffset Offset, const TargetInstrInfo *TII, 3107 MachineInstr::MIFlag Flag, bool SetNZCV, 3108 bool NeedsWinCFI, bool *HasWinCFI) { 3109 int64_t Bytes; 3110 Offset.getForFrameOffset(Bytes); 3111 3112 // First emit non-scalable frame offsets, or a simple 'mov'. 3113 if (Bytes || (!Offset && SrcReg != DestReg)) { 3114 assert((DestReg != AArch64::SP || Bytes % 16 == 0) && 3115 "SP increment/decrement not 16-byte aligned"); 3116 unsigned Opc = SetNZCV ? AArch64::ADDSXri : AArch64::ADDXri; 3117 if (Bytes < 0) { 3118 Bytes = -Bytes; 3119 Opc = SetNZCV ? AArch64::SUBSXri : AArch64::SUBXri; 3120 } 3121 emitFrameOffsetAdj(MBB, MBBI, DL, DestReg, SrcReg, Bytes, Opc, TII, Flag, 3122 NeedsWinCFI, HasWinCFI); 3123 SrcReg = DestReg; 3124 } 3125 } 3126 3127 MachineInstr *AArch64InstrInfo::foldMemoryOperandImpl( 3128 MachineFunction &MF, MachineInstr &MI, ArrayRef<unsigned> Ops, 3129 MachineBasicBlock::iterator InsertPt, int FrameIndex, 3130 LiveIntervals *LIS, VirtRegMap *VRM) const { 3131 // This is a bit of a hack. Consider this instruction: 3132 // 3133 // %0 = COPY %sp; GPR64all:%0 3134 // 3135 // We explicitly chose GPR64all for the virtual register so such a copy might 3136 // be eliminated by RegisterCoalescer. However, that may not be possible, and 3137 // %0 may even spill. We can't spill %sp, and since it is in the GPR64all 3138 // register class, TargetInstrInfo::foldMemoryOperand() is going to try. 3139 // 3140 // To prevent that, we are going to constrain the %0 register class here. 3141 // 3142 // <rdar://problem/11522048> 3143 // 3144 if (MI.isFullCopy()) { 3145 Register DstReg = MI.getOperand(0).getReg(); 3146 Register SrcReg = MI.getOperand(1).getReg(); 3147 if (SrcReg == AArch64::SP && Register::isVirtualRegister(DstReg)) { 3148 MF.getRegInfo().constrainRegClass(DstReg, &AArch64::GPR64RegClass); 3149 return nullptr; 3150 } 3151 if (DstReg == AArch64::SP && Register::isVirtualRegister(SrcReg)) { 3152 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass); 3153 return nullptr; 3154 } 3155 } 3156 3157 // Handle the case where a copy is being spilled or filled but the source 3158 // and destination register class don't match. For example: 3159 // 3160 // %0 = COPY %xzr; GPR64common:%0 3161 // 3162 // In this case we can still safely fold away the COPY and generate the 3163 // following spill code: 3164 // 3165 // STRXui %xzr, %stack.0 3166 // 3167 // This also eliminates spilled cross register class COPYs (e.g. between x and 3168 // d regs) of the same size. For example: 3169 // 3170 // %0 = COPY %1; GPR64:%0, FPR64:%1 3171 // 3172 // will be filled as 3173 // 3174 // LDRDui %0, fi<#0> 3175 // 3176 // instead of 3177 // 3178 // LDRXui %Temp, fi<#0> 3179 // %0 = FMOV %Temp 3180 // 3181 if (MI.isCopy() && Ops.size() == 1 && 3182 // Make sure we're only folding the explicit COPY defs/uses. 3183 (Ops[0] == 0 || Ops[0] == 1)) { 3184 bool IsSpill = Ops[0] == 0; 3185 bool IsFill = !IsSpill; 3186 const TargetRegisterInfo &TRI = *MF.getSubtarget().getRegisterInfo(); 3187 const MachineRegisterInfo &MRI = MF.getRegInfo(); 3188 MachineBasicBlock &MBB = *MI.getParent(); 3189 const MachineOperand &DstMO = MI.getOperand(0); 3190 const MachineOperand &SrcMO = MI.getOperand(1); 3191 Register DstReg = DstMO.getReg(); 3192 Register SrcReg = SrcMO.getReg(); 3193 // This is slightly expensive to compute for physical regs since 3194 // getMinimalPhysRegClass is slow. 3195 auto getRegClass = [&](unsigned Reg) { 3196 return Register::isVirtualRegister(Reg) ? MRI.getRegClass(Reg) 3197 : TRI.getMinimalPhysRegClass(Reg); 3198 }; 3199 3200 if (DstMO.getSubReg() == 0 && SrcMO.getSubReg() == 0) { 3201 assert(TRI.getRegSizeInBits(*getRegClass(DstReg)) == 3202 TRI.getRegSizeInBits(*getRegClass(SrcReg)) && 3203 "Mismatched register size in non subreg COPY"); 3204 if (IsSpill) 3205 storeRegToStackSlot(MBB, InsertPt, SrcReg, SrcMO.isKill(), FrameIndex, 3206 getRegClass(SrcReg), &TRI); 3207 else 3208 loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex, 3209 getRegClass(DstReg), &TRI); 3210 return &*--InsertPt; 3211 } 3212 3213 // Handle cases like spilling def of: 3214 // 3215 // %0:sub_32<def,read-undef> = COPY %wzr; GPR64common:%0 3216 // 3217 // where the physical register source can be widened and stored to the full 3218 // virtual reg destination stack slot, in this case producing: 3219 // 3220 // STRXui %xzr, %stack.0 3221 // 3222 if (IsSpill && DstMO.isUndef() && Register::isPhysicalRegister(SrcReg)) { 3223 assert(SrcMO.getSubReg() == 0 && 3224 "Unexpected subreg on physical register"); 3225 const TargetRegisterClass *SpillRC; 3226 unsigned SpillSubreg; 3227 switch (DstMO.getSubReg()) { 3228 default: 3229 SpillRC = nullptr; 3230 break; 3231 case AArch64::sub_32: 3232 case AArch64::ssub: 3233 if (AArch64::GPR32RegClass.contains(SrcReg)) { 3234 SpillRC = &AArch64::GPR64RegClass; 3235 SpillSubreg = AArch64::sub_32; 3236 } else if (AArch64::FPR32RegClass.contains(SrcReg)) { 3237 SpillRC = &AArch64::FPR64RegClass; 3238 SpillSubreg = AArch64::ssub; 3239 } else 3240 SpillRC = nullptr; 3241 break; 3242 case AArch64::dsub: 3243 if (AArch64::FPR64RegClass.contains(SrcReg)) { 3244 SpillRC = &AArch64::FPR128RegClass; 3245 SpillSubreg = AArch64::dsub; 3246 } else 3247 SpillRC = nullptr; 3248 break; 3249 } 3250 3251 if (SpillRC) 3252 if (unsigned WidenedSrcReg = 3253 TRI.getMatchingSuperReg(SrcReg, SpillSubreg, SpillRC)) { 3254 storeRegToStackSlot(MBB, InsertPt, WidenedSrcReg, SrcMO.isKill(), 3255 FrameIndex, SpillRC, &TRI); 3256 return &*--InsertPt; 3257 } 3258 } 3259 3260 // Handle cases like filling use of: 3261 // 3262 // %0:sub_32<def,read-undef> = COPY %1; GPR64:%0, GPR32:%1 3263 // 3264 // where we can load the full virtual reg source stack slot, into the subreg 3265 // destination, in this case producing: 3266 // 3267 // LDRWui %0:sub_32<def,read-undef>, %stack.0 3268 // 3269 if (IsFill && SrcMO.getSubReg() == 0 && DstMO.isUndef()) { 3270 const TargetRegisterClass *FillRC; 3271 switch (DstMO.getSubReg()) { 3272 default: 3273 FillRC = nullptr; 3274 break; 3275 case AArch64::sub_32: 3276 FillRC = &AArch64::GPR32RegClass; 3277 break; 3278 case AArch64::ssub: 3279 FillRC = &AArch64::FPR32RegClass; 3280 break; 3281 case AArch64::dsub: 3282 FillRC = &AArch64::FPR64RegClass; 3283 break; 3284 } 3285 3286 if (FillRC) { 3287 assert(TRI.getRegSizeInBits(*getRegClass(SrcReg)) == 3288 TRI.getRegSizeInBits(*FillRC) && 3289 "Mismatched regclass size on folded subreg COPY"); 3290 loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex, FillRC, &TRI); 3291 MachineInstr &LoadMI = *--InsertPt; 3292 MachineOperand &LoadDst = LoadMI.getOperand(0); 3293 assert(LoadDst.getSubReg() == 0 && "unexpected subreg on fill load"); 3294 LoadDst.setSubReg(DstMO.getSubReg()); 3295 LoadDst.setIsUndef(); 3296 return &LoadMI; 3297 } 3298 } 3299 } 3300 3301 // Cannot fold. 3302 return nullptr; 3303 } 3304 3305 int llvm::isAArch64FrameOffsetLegal(const MachineInstr &MI, 3306 StackOffset &SOffset, 3307 bool *OutUseUnscaledOp, 3308 unsigned *OutUnscaledOp, 3309 int *EmittableOffset) { 3310 // Set output values in case of early exit. 3311 if (EmittableOffset) 3312 *EmittableOffset = 0; 3313 if (OutUseUnscaledOp) 3314 *OutUseUnscaledOp = false; 3315 if (OutUnscaledOp) 3316 *OutUnscaledOp = 0; 3317 3318 // Exit early for structured vector spills/fills as they can't take an 3319 // immediate offset. 3320 switch (MI.getOpcode()) { 3321 default: 3322 break; 3323 case AArch64::LD1Twov2d: 3324 case AArch64::LD1Threev2d: 3325 case AArch64::LD1Fourv2d: 3326 case AArch64::LD1Twov1d: 3327 case AArch64::LD1Threev1d: 3328 case AArch64::LD1Fourv1d: 3329 case AArch64::ST1Twov2d: 3330 case AArch64::ST1Threev2d: 3331 case AArch64::ST1Fourv2d: 3332 case AArch64::ST1Twov1d: 3333 case AArch64::ST1Threev1d: 3334 case AArch64::ST1Fourv1d: 3335 case AArch64::IRG: 3336 case AArch64::IRGstack: 3337 return AArch64FrameOffsetCannotUpdate; 3338 } 3339 3340 // Get the min/max offset and the scale. 3341 unsigned Scale, Width; 3342 int64_t MinOff, MaxOff; 3343 if (!AArch64InstrInfo::getMemOpInfo(MI.getOpcode(), Scale, Width, MinOff, 3344 MaxOff)) 3345 llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal"); 3346 3347 // Construct the complete offset. 3348 const MachineOperand &ImmOpnd = 3349 MI.getOperand(AArch64InstrInfo::getLoadStoreImmIdx(MI.getOpcode())); 3350 int Offset = SOffset.getBytes() + ImmOpnd.getImm() * Scale; 3351 3352 // If the offset doesn't match the scale, we rewrite the instruction to 3353 // use the unscaled instruction instead. Likewise, if we have a negative 3354 // offset and there is an unscaled op to use. 3355 Optional<unsigned> UnscaledOp = 3356 AArch64InstrInfo::getUnscaledLdSt(MI.getOpcode()); 3357 bool useUnscaledOp = UnscaledOp && (Offset % Scale || Offset < 0); 3358 if (useUnscaledOp && 3359 !AArch64InstrInfo::getMemOpInfo(*UnscaledOp, Scale, Width, MinOff, MaxOff)) 3360 llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal"); 3361 3362 int64_t Remainder = Offset % Scale; 3363 assert(!(Remainder && useUnscaledOp) && 3364 "Cannot have remainder when using unscaled op"); 3365 3366 assert(MinOff < MaxOff && "Unexpected Min/Max offsets"); 3367 int NewOffset = Offset / Scale; 3368 if (MinOff <= NewOffset && NewOffset <= MaxOff) 3369 Offset = Remainder; 3370 else { 3371 NewOffset = NewOffset < 0 ? MinOff : MaxOff; 3372 Offset = Offset - NewOffset * Scale + Remainder; 3373 } 3374 3375 if (EmittableOffset) 3376 *EmittableOffset = NewOffset; 3377 if (OutUseUnscaledOp) 3378 *OutUseUnscaledOp = useUnscaledOp; 3379 if (OutUnscaledOp && UnscaledOp) 3380 *OutUnscaledOp = *UnscaledOp; 3381 3382 SOffset = StackOffset(Offset, MVT::i8); 3383 return AArch64FrameOffsetCanUpdate | 3384 (Offset == 0 ? AArch64FrameOffsetIsLegal : 0); 3385 } 3386 3387 bool llvm::rewriteAArch64FrameIndex(MachineInstr &MI, unsigned FrameRegIdx, 3388 unsigned FrameReg, StackOffset &Offset, 3389 const AArch64InstrInfo *TII) { 3390 unsigned Opcode = MI.getOpcode(); 3391 unsigned ImmIdx = FrameRegIdx + 1; 3392 3393 if (Opcode == AArch64::ADDSXri || Opcode == AArch64::ADDXri) { 3394 Offset += StackOffset(MI.getOperand(ImmIdx).getImm(), MVT::i8); 3395 emitFrameOffset(*MI.getParent(), MI, MI.getDebugLoc(), 3396 MI.getOperand(0).getReg(), FrameReg, Offset, TII, 3397 MachineInstr::NoFlags, (Opcode == AArch64::ADDSXri)); 3398 MI.eraseFromParent(); 3399 Offset = StackOffset(); 3400 return true; 3401 } 3402 3403 int NewOffset; 3404 unsigned UnscaledOp; 3405 bool UseUnscaledOp; 3406 int Status = isAArch64FrameOffsetLegal(MI, Offset, &UseUnscaledOp, 3407 &UnscaledOp, &NewOffset); 3408 if (Status & AArch64FrameOffsetCanUpdate) { 3409 if (Status & AArch64FrameOffsetIsLegal) 3410 // Replace the FrameIndex with FrameReg. 3411 MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false); 3412 if (UseUnscaledOp) 3413 MI.setDesc(TII->get(UnscaledOp)); 3414 3415 MI.getOperand(ImmIdx).ChangeToImmediate(NewOffset); 3416 return !Offset; 3417 } 3418 3419 return false; 3420 } 3421 3422 void AArch64InstrInfo::getNoop(MCInst &NopInst) const { 3423 NopInst.setOpcode(AArch64::HINT); 3424 NopInst.addOperand(MCOperand::createImm(0)); 3425 } 3426 3427 // AArch64 supports MachineCombiner. 3428 bool AArch64InstrInfo::useMachineCombiner() const { return true; } 3429 3430 // True when Opc sets flag 3431 static bool isCombineInstrSettingFlag(unsigned Opc) { 3432 switch (Opc) { 3433 case AArch64::ADDSWrr: 3434 case AArch64::ADDSWri: 3435 case AArch64::ADDSXrr: 3436 case AArch64::ADDSXri: 3437 case AArch64::SUBSWrr: 3438 case AArch64::SUBSXrr: 3439 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi. 3440 case AArch64::SUBSWri: 3441 case AArch64::SUBSXri: 3442 return true; 3443 default: 3444 break; 3445 } 3446 return false; 3447 } 3448 3449 // 32b Opcodes that can be combined with a MUL 3450 static bool isCombineInstrCandidate32(unsigned Opc) { 3451 switch (Opc) { 3452 case AArch64::ADDWrr: 3453 case AArch64::ADDWri: 3454 case AArch64::SUBWrr: 3455 case AArch64::ADDSWrr: 3456 case AArch64::ADDSWri: 3457 case AArch64::SUBSWrr: 3458 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi. 3459 case AArch64::SUBWri: 3460 case AArch64::SUBSWri: 3461 return true; 3462 default: 3463 break; 3464 } 3465 return false; 3466 } 3467 3468 // 64b Opcodes that can be combined with a MUL 3469 static bool isCombineInstrCandidate64(unsigned Opc) { 3470 switch (Opc) { 3471 case AArch64::ADDXrr: 3472 case AArch64::ADDXri: 3473 case AArch64::SUBXrr: 3474 case AArch64::ADDSXrr: 3475 case AArch64::ADDSXri: 3476 case AArch64::SUBSXrr: 3477 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi. 3478 case AArch64::SUBXri: 3479 case AArch64::SUBSXri: 3480 return true; 3481 default: 3482 break; 3483 } 3484 return false; 3485 } 3486 3487 // FP Opcodes that can be combined with a FMUL 3488 static bool isCombineInstrCandidateFP(const MachineInstr &Inst) { 3489 switch (Inst.getOpcode()) { 3490 default: 3491 break; 3492 case AArch64::FADDHrr: 3493 case AArch64::FADDSrr: 3494 case AArch64::FADDDrr: 3495 case AArch64::FADDv4f16: 3496 case AArch64::FADDv8f16: 3497 case AArch64::FADDv2f32: 3498 case AArch64::FADDv2f64: 3499 case AArch64::FADDv4f32: 3500 case AArch64::FSUBHrr: 3501 case AArch64::FSUBSrr: 3502 case AArch64::FSUBDrr: 3503 case AArch64::FSUBv4f16: 3504 case AArch64::FSUBv8f16: 3505 case AArch64::FSUBv2f32: 3506 case AArch64::FSUBv2f64: 3507 case AArch64::FSUBv4f32: 3508 TargetOptions Options = Inst.getParent()->getParent()->getTarget().Options; 3509 return (Options.UnsafeFPMath || 3510 Options.AllowFPOpFusion == FPOpFusion::Fast); 3511 } 3512 return false; 3513 } 3514 3515 // Opcodes that can be combined with a MUL 3516 static bool isCombineInstrCandidate(unsigned Opc) { 3517 return (isCombineInstrCandidate32(Opc) || isCombineInstrCandidate64(Opc)); 3518 } 3519 3520 // 3521 // Utility routine that checks if \param MO is defined by an 3522 // \param CombineOpc instruction in the basic block \param MBB 3523 static bool canCombine(MachineBasicBlock &MBB, MachineOperand &MO, 3524 unsigned CombineOpc, unsigned ZeroReg = 0, 3525 bool CheckZeroReg = false) { 3526 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 3527 MachineInstr *MI = nullptr; 3528 3529 if (MO.isReg() && Register::isVirtualRegister(MO.getReg())) 3530 MI = MRI.getUniqueVRegDef(MO.getReg()); 3531 // And it needs to be in the trace (otherwise, it won't have a depth). 3532 if (!MI || MI->getParent() != &MBB || (unsigned)MI->getOpcode() != CombineOpc) 3533 return false; 3534 // Must only used by the user we combine with. 3535 if (!MRI.hasOneNonDBGUse(MI->getOperand(0).getReg())) 3536 return false; 3537 3538 if (CheckZeroReg) { 3539 assert(MI->getNumOperands() >= 4 && MI->getOperand(0).isReg() && 3540 MI->getOperand(1).isReg() && MI->getOperand(2).isReg() && 3541 MI->getOperand(3).isReg() && "MAdd/MSub must have a least 4 regs"); 3542 // The third input reg must be zero. 3543 if (MI->getOperand(3).getReg() != ZeroReg) 3544 return false; 3545 } 3546 3547 return true; 3548 } 3549 3550 // 3551 // Is \param MO defined by an integer multiply and can be combined? 3552 static bool canCombineWithMUL(MachineBasicBlock &MBB, MachineOperand &MO, 3553 unsigned MulOpc, unsigned ZeroReg) { 3554 return canCombine(MBB, MO, MulOpc, ZeroReg, true); 3555 } 3556 3557 // 3558 // Is \param MO defined by a floating-point multiply and can be combined? 3559 static bool canCombineWithFMUL(MachineBasicBlock &MBB, MachineOperand &MO, 3560 unsigned MulOpc) { 3561 return canCombine(MBB, MO, MulOpc); 3562 } 3563 3564 // TODO: There are many more machine instruction opcodes to match: 3565 // 1. Other data types (integer, vectors) 3566 // 2. Other math / logic operations (xor, or) 3567 // 3. Other forms of the same operation (intrinsics and other variants) 3568 bool AArch64InstrInfo::isAssociativeAndCommutative( 3569 const MachineInstr &Inst) const { 3570 switch (Inst.getOpcode()) { 3571 case AArch64::FADDDrr: 3572 case AArch64::FADDSrr: 3573 case AArch64::FADDv2f32: 3574 case AArch64::FADDv2f64: 3575 case AArch64::FADDv4f32: 3576 case AArch64::FMULDrr: 3577 case AArch64::FMULSrr: 3578 case AArch64::FMULX32: 3579 case AArch64::FMULX64: 3580 case AArch64::FMULXv2f32: 3581 case AArch64::FMULXv2f64: 3582 case AArch64::FMULXv4f32: 3583 case AArch64::FMULv2f32: 3584 case AArch64::FMULv2f64: 3585 case AArch64::FMULv4f32: 3586 return Inst.getParent()->getParent()->getTarget().Options.UnsafeFPMath; 3587 default: 3588 return false; 3589 } 3590 } 3591 3592 /// Find instructions that can be turned into madd. 3593 static bool getMaddPatterns(MachineInstr &Root, 3594 SmallVectorImpl<MachineCombinerPattern> &Patterns) { 3595 unsigned Opc = Root.getOpcode(); 3596 MachineBasicBlock &MBB = *Root.getParent(); 3597 bool Found = false; 3598 3599 if (!isCombineInstrCandidate(Opc)) 3600 return false; 3601 if (isCombineInstrSettingFlag(Opc)) { 3602 int Cmp_NZCV = Root.findRegisterDefOperandIdx(AArch64::NZCV, true); 3603 // When NZCV is live bail out. 3604 if (Cmp_NZCV == -1) 3605 return false; 3606 unsigned NewOpc = convertToNonFlagSettingOpc(Root); 3607 // When opcode can't change bail out. 3608 // CHECKME: do we miss any cases for opcode conversion? 3609 if (NewOpc == Opc) 3610 return false; 3611 Opc = NewOpc; 3612 } 3613 3614 auto setFound = [&](int Opcode, int Operand, unsigned ZeroReg, 3615 MachineCombinerPattern Pattern) { 3616 if (canCombineWithMUL(MBB, Root.getOperand(Operand), Opcode, ZeroReg)) { 3617 Patterns.push_back(Pattern); 3618 Found = true; 3619 } 3620 }; 3621 3622 typedef MachineCombinerPattern MCP; 3623 3624 switch (Opc) { 3625 default: 3626 break; 3627 case AArch64::ADDWrr: 3628 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() && 3629 "ADDWrr does not have register operands"); 3630 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULADDW_OP1); 3631 setFound(AArch64::MADDWrrr, 2, AArch64::WZR, MCP::MULADDW_OP2); 3632 break; 3633 case AArch64::ADDXrr: 3634 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULADDX_OP1); 3635 setFound(AArch64::MADDXrrr, 2, AArch64::XZR, MCP::MULADDX_OP2); 3636 break; 3637 case AArch64::SUBWrr: 3638 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULSUBW_OP1); 3639 setFound(AArch64::MADDWrrr, 2, AArch64::WZR, MCP::MULSUBW_OP2); 3640 break; 3641 case AArch64::SUBXrr: 3642 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULSUBX_OP1); 3643 setFound(AArch64::MADDXrrr, 2, AArch64::XZR, MCP::MULSUBX_OP2); 3644 break; 3645 case AArch64::ADDWri: 3646 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULADDWI_OP1); 3647 break; 3648 case AArch64::ADDXri: 3649 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULADDXI_OP1); 3650 break; 3651 case AArch64::SUBWri: 3652 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULSUBWI_OP1); 3653 break; 3654 case AArch64::SUBXri: 3655 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULSUBXI_OP1); 3656 break; 3657 } 3658 return Found; 3659 } 3660 /// Floating-Point Support 3661 3662 /// Find instructions that can be turned into madd. 3663 static bool getFMAPatterns(MachineInstr &Root, 3664 SmallVectorImpl<MachineCombinerPattern> &Patterns) { 3665 3666 if (!isCombineInstrCandidateFP(Root)) 3667 return false; 3668 3669 MachineBasicBlock &MBB = *Root.getParent(); 3670 bool Found = false; 3671 3672 auto Match = [&](int Opcode, int Operand, 3673 MachineCombinerPattern Pattern) -> bool { 3674 if (canCombineWithFMUL(MBB, Root.getOperand(Operand), Opcode)) { 3675 Patterns.push_back(Pattern); 3676 return true; 3677 } 3678 return false; 3679 }; 3680 3681 typedef MachineCombinerPattern MCP; 3682 3683 switch (Root.getOpcode()) { 3684 default: 3685 assert(false && "Unsupported FP instruction in combiner\n"); 3686 break; 3687 case AArch64::FADDHrr: 3688 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() && 3689 "FADDHrr does not have register operands"); 3690 3691 Found = Match(AArch64::FMULHrr, 1, MCP::FMULADDH_OP1); 3692 Found |= Match(AArch64::FMULHrr, 2, MCP::FMULADDH_OP2); 3693 break; 3694 case AArch64::FADDSrr: 3695 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() && 3696 "FADDSrr does not have register operands"); 3697 3698 Found |= Match(AArch64::FMULSrr, 1, MCP::FMULADDS_OP1) || 3699 Match(AArch64::FMULv1i32_indexed, 1, MCP::FMLAv1i32_indexed_OP1); 3700 3701 Found |= Match(AArch64::FMULSrr, 2, MCP::FMULADDS_OP2) || 3702 Match(AArch64::FMULv1i32_indexed, 2, MCP::FMLAv1i32_indexed_OP2); 3703 break; 3704 case AArch64::FADDDrr: 3705 Found |= Match(AArch64::FMULDrr, 1, MCP::FMULADDD_OP1) || 3706 Match(AArch64::FMULv1i64_indexed, 1, MCP::FMLAv1i64_indexed_OP1); 3707 3708 Found |= Match(AArch64::FMULDrr, 2, MCP::FMULADDD_OP2) || 3709 Match(AArch64::FMULv1i64_indexed, 2, MCP::FMLAv1i64_indexed_OP2); 3710 break; 3711 case AArch64::FADDv4f16: 3712 Found |= Match(AArch64::FMULv4i16_indexed, 1, MCP::FMLAv4i16_indexed_OP1) || 3713 Match(AArch64::FMULv4f16, 1, MCP::FMLAv4f16_OP1); 3714 3715 Found |= Match(AArch64::FMULv4i16_indexed, 2, MCP::FMLAv4i16_indexed_OP2) || 3716 Match(AArch64::FMULv4f16, 2, MCP::FMLAv4f16_OP2); 3717 break; 3718 case AArch64::FADDv8f16: 3719 Found |= Match(AArch64::FMULv8i16_indexed, 1, MCP::FMLAv8i16_indexed_OP1) || 3720 Match(AArch64::FMULv8f16, 1, MCP::FMLAv8f16_OP1); 3721 3722 Found |= Match(AArch64::FMULv8i16_indexed, 2, MCP::FMLAv8i16_indexed_OP2) || 3723 Match(AArch64::FMULv8f16, 2, MCP::FMLAv8f16_OP2); 3724 break; 3725 case AArch64::FADDv2f32: 3726 Found |= Match(AArch64::FMULv2i32_indexed, 1, MCP::FMLAv2i32_indexed_OP1) || 3727 Match(AArch64::FMULv2f32, 1, MCP::FMLAv2f32_OP1); 3728 3729 Found |= Match(AArch64::FMULv2i32_indexed, 2, MCP::FMLAv2i32_indexed_OP2) || 3730 Match(AArch64::FMULv2f32, 2, MCP::FMLAv2f32_OP2); 3731 break; 3732 case AArch64::FADDv2f64: 3733 Found |= Match(AArch64::FMULv2i64_indexed, 1, MCP::FMLAv2i64_indexed_OP1) || 3734 Match(AArch64::FMULv2f64, 1, MCP::FMLAv2f64_OP1); 3735 3736 Found |= Match(AArch64::FMULv2i64_indexed, 2, MCP::FMLAv2i64_indexed_OP2) || 3737 Match(AArch64::FMULv2f64, 2, MCP::FMLAv2f64_OP2); 3738 break; 3739 case AArch64::FADDv4f32: 3740 Found |= Match(AArch64::FMULv4i32_indexed, 1, MCP::FMLAv4i32_indexed_OP1) || 3741 Match(AArch64::FMULv4f32, 1, MCP::FMLAv4f32_OP1); 3742 3743 Found |= Match(AArch64::FMULv4i32_indexed, 2, MCP::FMLAv4i32_indexed_OP2) || 3744 Match(AArch64::FMULv4f32, 2, MCP::FMLAv4f32_OP2); 3745 break; 3746 case AArch64::FSUBHrr: 3747 Found = Match(AArch64::FMULHrr, 1, MCP::FMULSUBH_OP1); 3748 Found |= Match(AArch64::FMULHrr, 2, MCP::FMULSUBH_OP2); 3749 Found |= Match(AArch64::FNMULHrr, 1, MCP::FNMULSUBH_OP1); 3750 break; 3751 case AArch64::FSUBSrr: 3752 Found = Match(AArch64::FMULSrr, 1, MCP::FMULSUBS_OP1); 3753 3754 Found |= Match(AArch64::FMULSrr, 2, MCP::FMULSUBS_OP2) || 3755 Match(AArch64::FMULv1i32_indexed, 2, MCP::FMLSv1i32_indexed_OP2); 3756 3757 Found |= Match(AArch64::FNMULSrr, 1, MCP::FNMULSUBS_OP1); 3758 break; 3759 case AArch64::FSUBDrr: 3760 Found = Match(AArch64::FMULDrr, 1, MCP::FMULSUBD_OP1); 3761 3762 Found |= Match(AArch64::FMULDrr, 2, MCP::FMULSUBD_OP2) || 3763 Match(AArch64::FMULv1i64_indexed, 2, MCP::FMLSv1i64_indexed_OP2); 3764 3765 Found |= Match(AArch64::FNMULDrr, 1, MCP::FNMULSUBD_OP1); 3766 break; 3767 case AArch64::FSUBv4f16: 3768 Found |= Match(AArch64::FMULv4i16_indexed, 2, MCP::FMLSv4i16_indexed_OP2) || 3769 Match(AArch64::FMULv4f16, 2, MCP::FMLSv4f16_OP2); 3770 3771 Found |= Match(AArch64::FMULv4i16_indexed, 1, MCP::FMLSv2i32_indexed_OP1) || 3772 Match(AArch64::FMULv4f16, 1, MCP::FMLSv2f32_OP1); 3773 break; 3774 case AArch64::FSUBv8f16: 3775 Found |= Match(AArch64::FMULv8i16_indexed, 2, MCP::FMLSv8i16_indexed_OP2) || 3776 Match(AArch64::FMULv8f16, 2, MCP::FMLSv8f16_OP2); 3777 3778 Found |= Match(AArch64::FMULv8i16_indexed, 1, MCP::FMLSv8i16_indexed_OP1) || 3779 Match(AArch64::FMULv8f16, 1, MCP::FMLSv8f16_OP1); 3780 break; 3781 case AArch64::FSUBv2f32: 3782 Found |= Match(AArch64::FMULv2i32_indexed, 2, MCP::FMLSv2i32_indexed_OP2) || 3783 Match(AArch64::FMULv2f32, 2, MCP::FMLSv2f32_OP2); 3784 3785 Found |= Match(AArch64::FMULv2i32_indexed, 1, MCP::FMLSv2i32_indexed_OP1) || 3786 Match(AArch64::FMULv2f32, 1, MCP::FMLSv2f32_OP1); 3787 break; 3788 case AArch64::FSUBv2f64: 3789 Found |= Match(AArch64::FMULv2i64_indexed, 2, MCP::FMLSv2i64_indexed_OP2) || 3790 Match(AArch64::FMULv2f64, 2, MCP::FMLSv2f64_OP2); 3791 3792 Found |= Match(AArch64::FMULv2i64_indexed, 1, MCP::FMLSv2i64_indexed_OP1) || 3793 Match(AArch64::FMULv2f64, 1, MCP::FMLSv2f64_OP1); 3794 break; 3795 case AArch64::FSUBv4f32: 3796 Found |= Match(AArch64::FMULv4i32_indexed, 2, MCP::FMLSv4i32_indexed_OP2) || 3797 Match(AArch64::FMULv4f32, 2, MCP::FMLSv4f32_OP2); 3798 3799 Found |= Match(AArch64::FMULv4i32_indexed, 1, MCP::FMLSv4i32_indexed_OP1) || 3800 Match(AArch64::FMULv4f32, 1, MCP::FMLSv4f32_OP1); 3801 break; 3802 } 3803 return Found; 3804 } 3805 3806 /// Return true when a code sequence can improve throughput. It 3807 /// should be called only for instructions in loops. 3808 /// \param Pattern - combiner pattern 3809 bool AArch64InstrInfo::isThroughputPattern( 3810 MachineCombinerPattern Pattern) const { 3811 switch (Pattern) { 3812 default: 3813 break; 3814 case MachineCombinerPattern::FMULADDH_OP1: 3815 case MachineCombinerPattern::FMULADDH_OP2: 3816 case MachineCombinerPattern::FMULSUBH_OP1: 3817 case MachineCombinerPattern::FMULSUBH_OP2: 3818 case MachineCombinerPattern::FMULADDS_OP1: 3819 case MachineCombinerPattern::FMULADDS_OP2: 3820 case MachineCombinerPattern::FMULSUBS_OP1: 3821 case MachineCombinerPattern::FMULSUBS_OP2: 3822 case MachineCombinerPattern::FMULADDD_OP1: 3823 case MachineCombinerPattern::FMULADDD_OP2: 3824 case MachineCombinerPattern::FMULSUBD_OP1: 3825 case MachineCombinerPattern::FMULSUBD_OP2: 3826 case MachineCombinerPattern::FNMULSUBH_OP1: 3827 case MachineCombinerPattern::FNMULSUBS_OP1: 3828 case MachineCombinerPattern::FNMULSUBD_OP1: 3829 case MachineCombinerPattern::FMLAv4i16_indexed_OP1: 3830 case MachineCombinerPattern::FMLAv4i16_indexed_OP2: 3831 case MachineCombinerPattern::FMLAv8i16_indexed_OP1: 3832 case MachineCombinerPattern::FMLAv8i16_indexed_OP2: 3833 case MachineCombinerPattern::FMLAv1i32_indexed_OP1: 3834 case MachineCombinerPattern::FMLAv1i32_indexed_OP2: 3835 case MachineCombinerPattern::FMLAv1i64_indexed_OP1: 3836 case MachineCombinerPattern::FMLAv1i64_indexed_OP2: 3837 case MachineCombinerPattern::FMLAv4f16_OP2: 3838 case MachineCombinerPattern::FMLAv4f16_OP1: 3839 case MachineCombinerPattern::FMLAv8f16_OP1: 3840 case MachineCombinerPattern::FMLAv8f16_OP2: 3841 case MachineCombinerPattern::FMLAv2f32_OP2: 3842 case MachineCombinerPattern::FMLAv2f32_OP1: 3843 case MachineCombinerPattern::FMLAv2f64_OP1: 3844 case MachineCombinerPattern::FMLAv2f64_OP2: 3845 case MachineCombinerPattern::FMLAv2i32_indexed_OP1: 3846 case MachineCombinerPattern::FMLAv2i32_indexed_OP2: 3847 case MachineCombinerPattern::FMLAv2i64_indexed_OP1: 3848 case MachineCombinerPattern::FMLAv2i64_indexed_OP2: 3849 case MachineCombinerPattern::FMLAv4f32_OP1: 3850 case MachineCombinerPattern::FMLAv4f32_OP2: 3851 case MachineCombinerPattern::FMLAv4i32_indexed_OP1: 3852 case MachineCombinerPattern::FMLAv4i32_indexed_OP2: 3853 case MachineCombinerPattern::FMLSv4i16_indexed_OP2: 3854 case MachineCombinerPattern::FMLSv8i16_indexed_OP1: 3855 case MachineCombinerPattern::FMLSv8i16_indexed_OP2: 3856 case MachineCombinerPattern::FMLSv1i32_indexed_OP2: 3857 case MachineCombinerPattern::FMLSv1i64_indexed_OP2: 3858 case MachineCombinerPattern::FMLSv2i32_indexed_OP2: 3859 case MachineCombinerPattern::FMLSv2i64_indexed_OP2: 3860 case MachineCombinerPattern::FMLSv4f16_OP2: 3861 case MachineCombinerPattern::FMLSv8f16_OP1: 3862 case MachineCombinerPattern::FMLSv8f16_OP2: 3863 case MachineCombinerPattern::FMLSv2f32_OP2: 3864 case MachineCombinerPattern::FMLSv2f64_OP2: 3865 case MachineCombinerPattern::FMLSv4i32_indexed_OP2: 3866 case MachineCombinerPattern::FMLSv4f32_OP2: 3867 return true; 3868 } // end switch (Pattern) 3869 return false; 3870 } 3871 /// Return true when there is potentially a faster code sequence for an 3872 /// instruction chain ending in \p Root. All potential patterns are listed in 3873 /// the \p Pattern vector. Pattern should be sorted in priority order since the 3874 /// pattern evaluator stops checking as soon as it finds a faster sequence. 3875 3876 bool AArch64InstrInfo::getMachineCombinerPatterns( 3877 MachineInstr &Root, 3878 SmallVectorImpl<MachineCombinerPattern> &Patterns) const { 3879 // Integer patterns 3880 if (getMaddPatterns(Root, Patterns)) 3881 return true; 3882 // Floating point patterns 3883 if (getFMAPatterns(Root, Patterns)) 3884 return true; 3885 3886 return TargetInstrInfo::getMachineCombinerPatterns(Root, Patterns); 3887 } 3888 3889 enum class FMAInstKind { Default, Indexed, Accumulator }; 3890 /// genFusedMultiply - Generate fused multiply instructions. 3891 /// This function supports both integer and floating point instructions. 3892 /// A typical example: 3893 /// F|MUL I=A,B,0 3894 /// F|ADD R,I,C 3895 /// ==> F|MADD R,A,B,C 3896 /// \param MF Containing MachineFunction 3897 /// \param MRI Register information 3898 /// \param TII Target information 3899 /// \param Root is the F|ADD instruction 3900 /// \param [out] InsInstrs is a vector of machine instructions and will 3901 /// contain the generated madd instruction 3902 /// \param IdxMulOpd is index of operand in Root that is the result of 3903 /// the F|MUL. In the example above IdxMulOpd is 1. 3904 /// \param MaddOpc the opcode fo the f|madd instruction 3905 /// \param RC Register class of operands 3906 /// \param kind of fma instruction (addressing mode) to be generated 3907 /// \param ReplacedAddend is the result register from the instruction 3908 /// replacing the non-combined operand, if any. 3909 static MachineInstr * 3910 genFusedMultiply(MachineFunction &MF, MachineRegisterInfo &MRI, 3911 const TargetInstrInfo *TII, MachineInstr &Root, 3912 SmallVectorImpl<MachineInstr *> &InsInstrs, unsigned IdxMulOpd, 3913 unsigned MaddOpc, const TargetRegisterClass *RC, 3914 FMAInstKind kind = FMAInstKind::Default, 3915 const Register *ReplacedAddend = nullptr) { 3916 assert(IdxMulOpd == 1 || IdxMulOpd == 2); 3917 3918 unsigned IdxOtherOpd = IdxMulOpd == 1 ? 2 : 1; 3919 MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg()); 3920 Register ResultReg = Root.getOperand(0).getReg(); 3921 Register SrcReg0 = MUL->getOperand(1).getReg(); 3922 bool Src0IsKill = MUL->getOperand(1).isKill(); 3923 Register SrcReg1 = MUL->getOperand(2).getReg(); 3924 bool Src1IsKill = MUL->getOperand(2).isKill(); 3925 3926 unsigned SrcReg2; 3927 bool Src2IsKill; 3928 if (ReplacedAddend) { 3929 // If we just generated a new addend, we must be it's only use. 3930 SrcReg2 = *ReplacedAddend; 3931 Src2IsKill = true; 3932 } else { 3933 SrcReg2 = Root.getOperand(IdxOtherOpd).getReg(); 3934 Src2IsKill = Root.getOperand(IdxOtherOpd).isKill(); 3935 } 3936 3937 if (Register::isVirtualRegister(ResultReg)) 3938 MRI.constrainRegClass(ResultReg, RC); 3939 if (Register::isVirtualRegister(SrcReg0)) 3940 MRI.constrainRegClass(SrcReg0, RC); 3941 if (Register::isVirtualRegister(SrcReg1)) 3942 MRI.constrainRegClass(SrcReg1, RC); 3943 if (Register::isVirtualRegister(SrcReg2)) 3944 MRI.constrainRegClass(SrcReg2, RC); 3945 3946 MachineInstrBuilder MIB; 3947 if (kind == FMAInstKind::Default) 3948 MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 3949 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 3950 .addReg(SrcReg1, getKillRegState(Src1IsKill)) 3951 .addReg(SrcReg2, getKillRegState(Src2IsKill)); 3952 else if (kind == FMAInstKind::Indexed) 3953 MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 3954 .addReg(SrcReg2, getKillRegState(Src2IsKill)) 3955 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 3956 .addReg(SrcReg1, getKillRegState(Src1IsKill)) 3957 .addImm(MUL->getOperand(3).getImm()); 3958 else if (kind == FMAInstKind::Accumulator) 3959 MIB = BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 3960 .addReg(SrcReg2, getKillRegState(Src2IsKill)) 3961 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 3962 .addReg(SrcReg1, getKillRegState(Src1IsKill)); 3963 else 3964 assert(false && "Invalid FMA instruction kind \n"); 3965 // Insert the MADD (MADD, FMA, FMS, FMLA, FMSL) 3966 InsInstrs.push_back(MIB); 3967 return MUL; 3968 } 3969 3970 /// genMaddR - Generate madd instruction and combine mul and add using 3971 /// an extra virtual register 3972 /// Example - an ADD intermediate needs to be stored in a register: 3973 /// MUL I=A,B,0 3974 /// ADD R,I,Imm 3975 /// ==> ORR V, ZR, Imm 3976 /// ==> MADD R,A,B,V 3977 /// \param MF Containing MachineFunction 3978 /// \param MRI Register information 3979 /// \param TII Target information 3980 /// \param Root is the ADD instruction 3981 /// \param [out] InsInstrs is a vector of machine instructions and will 3982 /// contain the generated madd instruction 3983 /// \param IdxMulOpd is index of operand in Root that is the result of 3984 /// the MUL. In the example above IdxMulOpd is 1. 3985 /// \param MaddOpc the opcode fo the madd instruction 3986 /// \param VR is a virtual register that holds the value of an ADD operand 3987 /// (V in the example above). 3988 /// \param RC Register class of operands 3989 static MachineInstr *genMaddR(MachineFunction &MF, MachineRegisterInfo &MRI, 3990 const TargetInstrInfo *TII, MachineInstr &Root, 3991 SmallVectorImpl<MachineInstr *> &InsInstrs, 3992 unsigned IdxMulOpd, unsigned MaddOpc, unsigned VR, 3993 const TargetRegisterClass *RC) { 3994 assert(IdxMulOpd == 1 || IdxMulOpd == 2); 3995 3996 MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg()); 3997 Register ResultReg = Root.getOperand(0).getReg(); 3998 Register SrcReg0 = MUL->getOperand(1).getReg(); 3999 bool Src0IsKill = MUL->getOperand(1).isKill(); 4000 Register SrcReg1 = MUL->getOperand(2).getReg(); 4001 bool Src1IsKill = MUL->getOperand(2).isKill(); 4002 4003 if (Register::isVirtualRegister(ResultReg)) 4004 MRI.constrainRegClass(ResultReg, RC); 4005 if (Register::isVirtualRegister(SrcReg0)) 4006 MRI.constrainRegClass(SrcReg0, RC); 4007 if (Register::isVirtualRegister(SrcReg1)) 4008 MRI.constrainRegClass(SrcReg1, RC); 4009 if (Register::isVirtualRegister(VR)) 4010 MRI.constrainRegClass(VR, RC); 4011 4012 MachineInstrBuilder MIB = 4013 BuildMI(MF, Root.getDebugLoc(), TII->get(MaddOpc), ResultReg) 4014 .addReg(SrcReg0, getKillRegState(Src0IsKill)) 4015 .addReg(SrcReg1, getKillRegState(Src1IsKill)) 4016 .addReg(VR); 4017 // Insert the MADD 4018 InsInstrs.push_back(MIB); 4019 return MUL; 4020 } 4021 4022 /// When getMachineCombinerPatterns() finds potential patterns, 4023 /// this function generates the instructions that could replace the 4024 /// original code sequence 4025 void AArch64InstrInfo::genAlternativeCodeSequence( 4026 MachineInstr &Root, MachineCombinerPattern Pattern, 4027 SmallVectorImpl<MachineInstr *> &InsInstrs, 4028 SmallVectorImpl<MachineInstr *> &DelInstrs, 4029 DenseMap<unsigned, unsigned> &InstrIdxForVirtReg) const { 4030 MachineBasicBlock &MBB = *Root.getParent(); 4031 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo(); 4032 MachineFunction &MF = *MBB.getParent(); 4033 const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo(); 4034 4035 MachineInstr *MUL; 4036 const TargetRegisterClass *RC; 4037 unsigned Opc; 4038 switch (Pattern) { 4039 default: 4040 // Reassociate instructions. 4041 TargetInstrInfo::genAlternativeCodeSequence(Root, Pattern, InsInstrs, 4042 DelInstrs, InstrIdxForVirtReg); 4043 return; 4044 case MachineCombinerPattern::MULADDW_OP1: 4045 case MachineCombinerPattern::MULADDX_OP1: 4046 // MUL I=A,B,0 4047 // ADD R,I,C 4048 // ==> MADD R,A,B,C 4049 // --- Create(MADD); 4050 if (Pattern == MachineCombinerPattern::MULADDW_OP1) { 4051 Opc = AArch64::MADDWrrr; 4052 RC = &AArch64::GPR32RegClass; 4053 } else { 4054 Opc = AArch64::MADDXrrr; 4055 RC = &AArch64::GPR64RegClass; 4056 } 4057 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4058 break; 4059 case MachineCombinerPattern::MULADDW_OP2: 4060 case MachineCombinerPattern::MULADDX_OP2: 4061 // MUL I=A,B,0 4062 // ADD R,C,I 4063 // ==> MADD R,A,B,C 4064 // --- Create(MADD); 4065 if (Pattern == MachineCombinerPattern::MULADDW_OP2) { 4066 Opc = AArch64::MADDWrrr; 4067 RC = &AArch64::GPR32RegClass; 4068 } else { 4069 Opc = AArch64::MADDXrrr; 4070 RC = &AArch64::GPR64RegClass; 4071 } 4072 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4073 break; 4074 case MachineCombinerPattern::MULADDWI_OP1: 4075 case MachineCombinerPattern::MULADDXI_OP1: { 4076 // MUL I=A,B,0 4077 // ADD R,I,Imm 4078 // ==> ORR V, ZR, Imm 4079 // ==> MADD R,A,B,V 4080 // --- Create(MADD); 4081 const TargetRegisterClass *OrrRC; 4082 unsigned BitSize, OrrOpc, ZeroReg; 4083 if (Pattern == MachineCombinerPattern::MULADDWI_OP1) { 4084 OrrOpc = AArch64::ORRWri; 4085 OrrRC = &AArch64::GPR32spRegClass; 4086 BitSize = 32; 4087 ZeroReg = AArch64::WZR; 4088 Opc = AArch64::MADDWrrr; 4089 RC = &AArch64::GPR32RegClass; 4090 } else { 4091 OrrOpc = AArch64::ORRXri; 4092 OrrRC = &AArch64::GPR64spRegClass; 4093 BitSize = 64; 4094 ZeroReg = AArch64::XZR; 4095 Opc = AArch64::MADDXrrr; 4096 RC = &AArch64::GPR64RegClass; 4097 } 4098 Register NewVR = MRI.createVirtualRegister(OrrRC); 4099 uint64_t Imm = Root.getOperand(2).getImm(); 4100 4101 if (Root.getOperand(3).isImm()) { 4102 unsigned Val = Root.getOperand(3).getImm(); 4103 Imm = Imm << Val; 4104 } 4105 uint64_t UImm = SignExtend64(Imm, BitSize); 4106 uint64_t Encoding; 4107 if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) { 4108 MachineInstrBuilder MIB1 = 4109 BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR) 4110 .addReg(ZeroReg) 4111 .addImm(Encoding); 4112 InsInstrs.push_back(MIB1); 4113 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4114 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC); 4115 } 4116 break; 4117 } 4118 case MachineCombinerPattern::MULSUBW_OP1: 4119 case MachineCombinerPattern::MULSUBX_OP1: { 4120 // MUL I=A,B,0 4121 // SUB R,I, C 4122 // ==> SUB V, 0, C 4123 // ==> MADD R,A,B,V // = -C + A*B 4124 // --- Create(MADD); 4125 const TargetRegisterClass *SubRC; 4126 unsigned SubOpc, ZeroReg; 4127 if (Pattern == MachineCombinerPattern::MULSUBW_OP1) { 4128 SubOpc = AArch64::SUBWrr; 4129 SubRC = &AArch64::GPR32spRegClass; 4130 ZeroReg = AArch64::WZR; 4131 Opc = AArch64::MADDWrrr; 4132 RC = &AArch64::GPR32RegClass; 4133 } else { 4134 SubOpc = AArch64::SUBXrr; 4135 SubRC = &AArch64::GPR64spRegClass; 4136 ZeroReg = AArch64::XZR; 4137 Opc = AArch64::MADDXrrr; 4138 RC = &AArch64::GPR64RegClass; 4139 } 4140 Register NewVR = MRI.createVirtualRegister(SubRC); 4141 // SUB NewVR, 0, C 4142 MachineInstrBuilder MIB1 = 4143 BuildMI(MF, Root.getDebugLoc(), TII->get(SubOpc), NewVR) 4144 .addReg(ZeroReg) 4145 .add(Root.getOperand(2)); 4146 InsInstrs.push_back(MIB1); 4147 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4148 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC); 4149 break; 4150 } 4151 case MachineCombinerPattern::MULSUBW_OP2: 4152 case MachineCombinerPattern::MULSUBX_OP2: 4153 // MUL I=A,B,0 4154 // SUB R,C,I 4155 // ==> MSUB R,A,B,C (computes C - A*B) 4156 // --- Create(MSUB); 4157 if (Pattern == MachineCombinerPattern::MULSUBW_OP2) { 4158 Opc = AArch64::MSUBWrrr; 4159 RC = &AArch64::GPR32RegClass; 4160 } else { 4161 Opc = AArch64::MSUBXrrr; 4162 RC = &AArch64::GPR64RegClass; 4163 } 4164 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4165 break; 4166 case MachineCombinerPattern::MULSUBWI_OP1: 4167 case MachineCombinerPattern::MULSUBXI_OP1: { 4168 // MUL I=A,B,0 4169 // SUB R,I, Imm 4170 // ==> ORR V, ZR, -Imm 4171 // ==> MADD R,A,B,V // = -Imm + A*B 4172 // --- Create(MADD); 4173 const TargetRegisterClass *OrrRC; 4174 unsigned BitSize, OrrOpc, ZeroReg; 4175 if (Pattern == MachineCombinerPattern::MULSUBWI_OP1) { 4176 OrrOpc = AArch64::ORRWri; 4177 OrrRC = &AArch64::GPR32spRegClass; 4178 BitSize = 32; 4179 ZeroReg = AArch64::WZR; 4180 Opc = AArch64::MADDWrrr; 4181 RC = &AArch64::GPR32RegClass; 4182 } else { 4183 OrrOpc = AArch64::ORRXri; 4184 OrrRC = &AArch64::GPR64spRegClass; 4185 BitSize = 64; 4186 ZeroReg = AArch64::XZR; 4187 Opc = AArch64::MADDXrrr; 4188 RC = &AArch64::GPR64RegClass; 4189 } 4190 Register NewVR = MRI.createVirtualRegister(OrrRC); 4191 uint64_t Imm = Root.getOperand(2).getImm(); 4192 if (Root.getOperand(3).isImm()) { 4193 unsigned Val = Root.getOperand(3).getImm(); 4194 Imm = Imm << Val; 4195 } 4196 uint64_t UImm = SignExtend64(-Imm, BitSize); 4197 uint64_t Encoding; 4198 if (AArch64_AM::processLogicalImmediate(UImm, BitSize, Encoding)) { 4199 MachineInstrBuilder MIB1 = 4200 BuildMI(MF, Root.getDebugLoc(), TII->get(OrrOpc), NewVR) 4201 .addReg(ZeroReg) 4202 .addImm(Encoding); 4203 InsInstrs.push_back(MIB1); 4204 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4205 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC); 4206 } 4207 break; 4208 } 4209 // Floating Point Support 4210 case MachineCombinerPattern::FMULADDH_OP1: 4211 Opc = AArch64::FMADDHrrr; 4212 RC = &AArch64::FPR16RegClass; 4213 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4214 break; 4215 case MachineCombinerPattern::FMULADDS_OP1: 4216 Opc = AArch64::FMADDSrrr; 4217 RC = &AArch64::FPR32RegClass; 4218 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4219 break; 4220 case MachineCombinerPattern::FMULADDD_OP1: 4221 Opc = AArch64::FMADDDrrr; 4222 RC = &AArch64::FPR64RegClass; 4223 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4224 break; 4225 4226 case MachineCombinerPattern::FMULADDH_OP2: 4227 Opc = AArch64::FMADDHrrr; 4228 RC = &AArch64::FPR16RegClass; 4229 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4230 break; 4231 case MachineCombinerPattern::FMULADDS_OP2: 4232 Opc = AArch64::FMADDSrrr; 4233 RC = &AArch64::FPR32RegClass; 4234 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4235 break; 4236 case MachineCombinerPattern::FMULADDD_OP2: 4237 Opc = AArch64::FMADDDrrr; 4238 RC = &AArch64::FPR64RegClass; 4239 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4240 break; 4241 4242 case MachineCombinerPattern::FMLAv1i32_indexed_OP1: 4243 Opc = AArch64::FMLAv1i32_indexed; 4244 RC = &AArch64::FPR32RegClass; 4245 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4246 FMAInstKind::Indexed); 4247 break; 4248 case MachineCombinerPattern::FMLAv1i32_indexed_OP2: 4249 Opc = AArch64::FMLAv1i32_indexed; 4250 RC = &AArch64::FPR32RegClass; 4251 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4252 FMAInstKind::Indexed); 4253 break; 4254 4255 case MachineCombinerPattern::FMLAv1i64_indexed_OP1: 4256 Opc = AArch64::FMLAv1i64_indexed; 4257 RC = &AArch64::FPR64RegClass; 4258 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4259 FMAInstKind::Indexed); 4260 break; 4261 case MachineCombinerPattern::FMLAv1i64_indexed_OP2: 4262 Opc = AArch64::FMLAv1i64_indexed; 4263 RC = &AArch64::FPR64RegClass; 4264 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4265 FMAInstKind::Indexed); 4266 break; 4267 4268 case MachineCombinerPattern::FMLAv4i16_indexed_OP1: 4269 RC = &AArch64::FPR64RegClass; 4270 Opc = AArch64::FMLAv4i16_indexed; 4271 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4272 FMAInstKind::Indexed); 4273 break; 4274 case MachineCombinerPattern::FMLAv4f16_OP1: 4275 RC = &AArch64::FPR64RegClass; 4276 Opc = AArch64::FMLAv4f16; 4277 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4278 FMAInstKind::Accumulator); 4279 break; 4280 case MachineCombinerPattern::FMLAv4i16_indexed_OP2: 4281 RC = &AArch64::FPR64RegClass; 4282 Opc = AArch64::FMLAv4i16_indexed; 4283 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4284 FMAInstKind::Indexed); 4285 break; 4286 case MachineCombinerPattern::FMLAv4f16_OP2: 4287 RC = &AArch64::FPR64RegClass; 4288 Opc = AArch64::FMLAv4f16; 4289 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4290 FMAInstKind::Accumulator); 4291 break; 4292 4293 case MachineCombinerPattern::FMLAv2i32_indexed_OP1: 4294 case MachineCombinerPattern::FMLAv2f32_OP1: 4295 RC = &AArch64::FPR64RegClass; 4296 if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP1) { 4297 Opc = AArch64::FMLAv2i32_indexed; 4298 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4299 FMAInstKind::Indexed); 4300 } else { 4301 Opc = AArch64::FMLAv2f32; 4302 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4303 FMAInstKind::Accumulator); 4304 } 4305 break; 4306 case MachineCombinerPattern::FMLAv2i32_indexed_OP2: 4307 case MachineCombinerPattern::FMLAv2f32_OP2: 4308 RC = &AArch64::FPR64RegClass; 4309 if (Pattern == MachineCombinerPattern::FMLAv2i32_indexed_OP2) { 4310 Opc = AArch64::FMLAv2i32_indexed; 4311 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4312 FMAInstKind::Indexed); 4313 } else { 4314 Opc = AArch64::FMLAv2f32; 4315 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4316 FMAInstKind::Accumulator); 4317 } 4318 break; 4319 4320 case MachineCombinerPattern::FMLAv8i16_indexed_OP1: 4321 RC = &AArch64::FPR128RegClass; 4322 Opc = AArch64::FMLAv8i16_indexed; 4323 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4324 FMAInstKind::Indexed); 4325 break; 4326 case MachineCombinerPattern::FMLAv8f16_OP1: 4327 RC = &AArch64::FPR128RegClass; 4328 Opc = AArch64::FMLAv8f16; 4329 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4330 FMAInstKind::Accumulator); 4331 break; 4332 case MachineCombinerPattern::FMLAv8i16_indexed_OP2: 4333 RC = &AArch64::FPR128RegClass; 4334 Opc = AArch64::FMLAv8i16_indexed; 4335 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4336 FMAInstKind::Indexed); 4337 break; 4338 case MachineCombinerPattern::FMLAv8f16_OP2: 4339 RC = &AArch64::FPR128RegClass; 4340 Opc = AArch64::FMLAv8f16; 4341 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4342 FMAInstKind::Accumulator); 4343 break; 4344 4345 case MachineCombinerPattern::FMLAv2i64_indexed_OP1: 4346 case MachineCombinerPattern::FMLAv2f64_OP1: 4347 RC = &AArch64::FPR128RegClass; 4348 if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP1) { 4349 Opc = AArch64::FMLAv2i64_indexed; 4350 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4351 FMAInstKind::Indexed); 4352 } else { 4353 Opc = AArch64::FMLAv2f64; 4354 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4355 FMAInstKind::Accumulator); 4356 } 4357 break; 4358 case MachineCombinerPattern::FMLAv2i64_indexed_OP2: 4359 case MachineCombinerPattern::FMLAv2f64_OP2: 4360 RC = &AArch64::FPR128RegClass; 4361 if (Pattern == MachineCombinerPattern::FMLAv2i64_indexed_OP2) { 4362 Opc = AArch64::FMLAv2i64_indexed; 4363 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4364 FMAInstKind::Indexed); 4365 } else { 4366 Opc = AArch64::FMLAv2f64; 4367 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4368 FMAInstKind::Accumulator); 4369 } 4370 break; 4371 4372 case MachineCombinerPattern::FMLAv4i32_indexed_OP1: 4373 case MachineCombinerPattern::FMLAv4f32_OP1: 4374 RC = &AArch64::FPR128RegClass; 4375 if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP1) { 4376 Opc = AArch64::FMLAv4i32_indexed; 4377 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4378 FMAInstKind::Indexed); 4379 } else { 4380 Opc = AArch64::FMLAv4f32; 4381 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4382 FMAInstKind::Accumulator); 4383 } 4384 break; 4385 4386 case MachineCombinerPattern::FMLAv4i32_indexed_OP2: 4387 case MachineCombinerPattern::FMLAv4f32_OP2: 4388 RC = &AArch64::FPR128RegClass; 4389 if (Pattern == MachineCombinerPattern::FMLAv4i32_indexed_OP2) { 4390 Opc = AArch64::FMLAv4i32_indexed; 4391 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4392 FMAInstKind::Indexed); 4393 } else { 4394 Opc = AArch64::FMLAv4f32; 4395 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4396 FMAInstKind::Accumulator); 4397 } 4398 break; 4399 4400 case MachineCombinerPattern::FMULSUBH_OP1: 4401 Opc = AArch64::FNMSUBHrrr; 4402 RC = &AArch64::FPR16RegClass; 4403 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4404 break; 4405 case MachineCombinerPattern::FMULSUBS_OP1: 4406 Opc = AArch64::FNMSUBSrrr; 4407 RC = &AArch64::FPR32RegClass; 4408 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4409 break; 4410 case MachineCombinerPattern::FMULSUBD_OP1: 4411 Opc = AArch64::FNMSUBDrrr; 4412 RC = &AArch64::FPR64RegClass; 4413 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4414 break; 4415 4416 case MachineCombinerPattern::FNMULSUBH_OP1: 4417 Opc = AArch64::FNMADDHrrr; 4418 RC = &AArch64::FPR16RegClass; 4419 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4420 break; 4421 case MachineCombinerPattern::FNMULSUBS_OP1: 4422 Opc = AArch64::FNMADDSrrr; 4423 RC = &AArch64::FPR32RegClass; 4424 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4425 break; 4426 case MachineCombinerPattern::FNMULSUBD_OP1: 4427 Opc = AArch64::FNMADDDrrr; 4428 RC = &AArch64::FPR64RegClass; 4429 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC); 4430 break; 4431 4432 case MachineCombinerPattern::FMULSUBH_OP2: 4433 Opc = AArch64::FMSUBHrrr; 4434 RC = &AArch64::FPR16RegClass; 4435 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4436 break; 4437 case MachineCombinerPattern::FMULSUBS_OP2: 4438 Opc = AArch64::FMSUBSrrr; 4439 RC = &AArch64::FPR32RegClass; 4440 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4441 break; 4442 case MachineCombinerPattern::FMULSUBD_OP2: 4443 Opc = AArch64::FMSUBDrrr; 4444 RC = &AArch64::FPR64RegClass; 4445 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC); 4446 break; 4447 4448 case MachineCombinerPattern::FMLSv1i32_indexed_OP2: 4449 Opc = AArch64::FMLSv1i32_indexed; 4450 RC = &AArch64::FPR32RegClass; 4451 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4452 FMAInstKind::Indexed); 4453 break; 4454 4455 case MachineCombinerPattern::FMLSv1i64_indexed_OP2: 4456 Opc = AArch64::FMLSv1i64_indexed; 4457 RC = &AArch64::FPR64RegClass; 4458 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4459 FMAInstKind::Indexed); 4460 break; 4461 4462 case MachineCombinerPattern::FMLSv4f16_OP2: 4463 RC = &AArch64::FPR64RegClass; 4464 Opc = AArch64::FMLSv4f16; 4465 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4466 FMAInstKind::Accumulator); 4467 break; 4468 case MachineCombinerPattern::FMLSv4i16_indexed_OP2: 4469 RC = &AArch64::FPR64RegClass; 4470 Opc = AArch64::FMLSv4i16_indexed; 4471 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4472 FMAInstKind::Indexed); 4473 break; 4474 4475 case MachineCombinerPattern::FMLSv2f32_OP2: 4476 case MachineCombinerPattern::FMLSv2i32_indexed_OP2: 4477 RC = &AArch64::FPR64RegClass; 4478 if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP2) { 4479 Opc = AArch64::FMLSv2i32_indexed; 4480 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4481 FMAInstKind::Indexed); 4482 } else { 4483 Opc = AArch64::FMLSv2f32; 4484 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4485 FMAInstKind::Accumulator); 4486 } 4487 break; 4488 4489 case MachineCombinerPattern::FMLSv8f16_OP1: 4490 RC = &AArch64::FPR128RegClass; 4491 Opc = AArch64::FMLSv8f16; 4492 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4493 FMAInstKind::Accumulator); 4494 break; 4495 case MachineCombinerPattern::FMLSv8i16_indexed_OP1: 4496 RC = &AArch64::FPR128RegClass; 4497 Opc = AArch64::FMLSv8i16_indexed; 4498 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4499 FMAInstKind::Indexed); 4500 break; 4501 4502 case MachineCombinerPattern::FMLSv8f16_OP2: 4503 RC = &AArch64::FPR128RegClass; 4504 Opc = AArch64::FMLSv8f16; 4505 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4506 FMAInstKind::Accumulator); 4507 break; 4508 case MachineCombinerPattern::FMLSv8i16_indexed_OP2: 4509 RC = &AArch64::FPR128RegClass; 4510 Opc = AArch64::FMLSv8i16_indexed; 4511 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4512 FMAInstKind::Indexed); 4513 break; 4514 4515 case MachineCombinerPattern::FMLSv2f64_OP2: 4516 case MachineCombinerPattern::FMLSv2i64_indexed_OP2: 4517 RC = &AArch64::FPR128RegClass; 4518 if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP2) { 4519 Opc = AArch64::FMLSv2i64_indexed; 4520 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4521 FMAInstKind::Indexed); 4522 } else { 4523 Opc = AArch64::FMLSv2f64; 4524 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4525 FMAInstKind::Accumulator); 4526 } 4527 break; 4528 4529 case MachineCombinerPattern::FMLSv4f32_OP2: 4530 case MachineCombinerPattern::FMLSv4i32_indexed_OP2: 4531 RC = &AArch64::FPR128RegClass; 4532 if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP2) { 4533 Opc = AArch64::FMLSv4i32_indexed; 4534 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4535 FMAInstKind::Indexed); 4536 } else { 4537 Opc = AArch64::FMLSv4f32; 4538 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC, 4539 FMAInstKind::Accumulator); 4540 } 4541 break; 4542 case MachineCombinerPattern::FMLSv2f32_OP1: 4543 case MachineCombinerPattern::FMLSv2i32_indexed_OP1: { 4544 RC = &AArch64::FPR64RegClass; 4545 Register NewVR = MRI.createVirtualRegister(RC); 4546 MachineInstrBuilder MIB1 = 4547 BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f32), NewVR) 4548 .add(Root.getOperand(2)); 4549 InsInstrs.push_back(MIB1); 4550 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4551 if (Pattern == MachineCombinerPattern::FMLSv2i32_indexed_OP1) { 4552 Opc = AArch64::FMLAv2i32_indexed; 4553 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4554 FMAInstKind::Indexed, &NewVR); 4555 } else { 4556 Opc = AArch64::FMLAv2f32; 4557 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4558 FMAInstKind::Accumulator, &NewVR); 4559 } 4560 break; 4561 } 4562 case MachineCombinerPattern::FMLSv4f32_OP1: 4563 case MachineCombinerPattern::FMLSv4i32_indexed_OP1: { 4564 RC = &AArch64::FPR128RegClass; 4565 Register NewVR = MRI.createVirtualRegister(RC); 4566 MachineInstrBuilder MIB1 = 4567 BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv4f32), NewVR) 4568 .add(Root.getOperand(2)); 4569 InsInstrs.push_back(MIB1); 4570 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4571 if (Pattern == MachineCombinerPattern::FMLSv4i32_indexed_OP1) { 4572 Opc = AArch64::FMLAv4i32_indexed; 4573 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4574 FMAInstKind::Indexed, &NewVR); 4575 } else { 4576 Opc = AArch64::FMLAv4f32; 4577 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4578 FMAInstKind::Accumulator, &NewVR); 4579 } 4580 break; 4581 } 4582 case MachineCombinerPattern::FMLSv2f64_OP1: 4583 case MachineCombinerPattern::FMLSv2i64_indexed_OP1: { 4584 RC = &AArch64::FPR128RegClass; 4585 Register NewVR = MRI.createVirtualRegister(RC); 4586 MachineInstrBuilder MIB1 = 4587 BuildMI(MF, Root.getDebugLoc(), TII->get(AArch64::FNEGv2f64), NewVR) 4588 .add(Root.getOperand(2)); 4589 InsInstrs.push_back(MIB1); 4590 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0)); 4591 if (Pattern == MachineCombinerPattern::FMLSv2i64_indexed_OP1) { 4592 Opc = AArch64::FMLAv2i64_indexed; 4593 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4594 FMAInstKind::Indexed, &NewVR); 4595 } else { 4596 Opc = AArch64::FMLAv2f64; 4597 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC, 4598 FMAInstKind::Accumulator, &NewVR); 4599 } 4600 break; 4601 } 4602 } // end switch (Pattern) 4603 // Record MUL and ADD/SUB for deletion 4604 DelInstrs.push_back(MUL); 4605 DelInstrs.push_back(&Root); 4606 } 4607 4608 /// Replace csincr-branch sequence by simple conditional branch 4609 /// 4610 /// Examples: 4611 /// 1. \code 4612 /// csinc w9, wzr, wzr, <condition code> 4613 /// tbnz w9, #0, 0x44 4614 /// \endcode 4615 /// to 4616 /// \code 4617 /// b.<inverted condition code> 4618 /// \endcode 4619 /// 4620 /// 2. \code 4621 /// csinc w9, wzr, wzr, <condition code> 4622 /// tbz w9, #0, 0x44 4623 /// \endcode 4624 /// to 4625 /// \code 4626 /// b.<condition code> 4627 /// \endcode 4628 /// 4629 /// Replace compare and branch sequence by TBZ/TBNZ instruction when the 4630 /// compare's constant operand is power of 2. 4631 /// 4632 /// Examples: 4633 /// \code 4634 /// and w8, w8, #0x400 4635 /// cbnz w8, L1 4636 /// \endcode 4637 /// to 4638 /// \code 4639 /// tbnz w8, #10, L1 4640 /// \endcode 4641 /// 4642 /// \param MI Conditional Branch 4643 /// \return True when the simple conditional branch is generated 4644 /// 4645 bool AArch64InstrInfo::optimizeCondBranch(MachineInstr &MI) const { 4646 bool IsNegativeBranch = false; 4647 bool IsTestAndBranch = false; 4648 unsigned TargetBBInMI = 0; 4649 switch (MI.getOpcode()) { 4650 default: 4651 llvm_unreachable("Unknown branch instruction?"); 4652 case AArch64::Bcc: 4653 return false; 4654 case AArch64::CBZW: 4655 case AArch64::CBZX: 4656 TargetBBInMI = 1; 4657 break; 4658 case AArch64::CBNZW: 4659 case AArch64::CBNZX: 4660 TargetBBInMI = 1; 4661 IsNegativeBranch = true; 4662 break; 4663 case AArch64::TBZW: 4664 case AArch64::TBZX: 4665 TargetBBInMI = 2; 4666 IsTestAndBranch = true; 4667 break; 4668 case AArch64::TBNZW: 4669 case AArch64::TBNZX: 4670 TargetBBInMI = 2; 4671 IsNegativeBranch = true; 4672 IsTestAndBranch = true; 4673 break; 4674 } 4675 // So we increment a zero register and test for bits other 4676 // than bit 0? Conservatively bail out in case the verifier 4677 // missed this case. 4678 if (IsTestAndBranch && MI.getOperand(1).getImm()) 4679 return false; 4680 4681 // Find Definition. 4682 assert(MI.getParent() && "Incomplete machine instruciton\n"); 4683 MachineBasicBlock *MBB = MI.getParent(); 4684 MachineFunction *MF = MBB->getParent(); 4685 MachineRegisterInfo *MRI = &MF->getRegInfo(); 4686 Register VReg = MI.getOperand(0).getReg(); 4687 if (!Register::isVirtualRegister(VReg)) 4688 return false; 4689 4690 MachineInstr *DefMI = MRI->getVRegDef(VReg); 4691 4692 // Look through COPY instructions to find definition. 4693 while (DefMI->isCopy()) { 4694 Register CopyVReg = DefMI->getOperand(1).getReg(); 4695 if (!MRI->hasOneNonDBGUse(CopyVReg)) 4696 return false; 4697 if (!MRI->hasOneDef(CopyVReg)) 4698 return false; 4699 DefMI = MRI->getVRegDef(CopyVReg); 4700 } 4701 4702 switch (DefMI->getOpcode()) { 4703 default: 4704 return false; 4705 // Fold AND into a TBZ/TBNZ if constant operand is power of 2. 4706 case AArch64::ANDWri: 4707 case AArch64::ANDXri: { 4708 if (IsTestAndBranch) 4709 return false; 4710 if (DefMI->getParent() != MBB) 4711 return false; 4712 if (!MRI->hasOneNonDBGUse(VReg)) 4713 return false; 4714 4715 bool Is32Bit = (DefMI->getOpcode() == AArch64::ANDWri); 4716 uint64_t Mask = AArch64_AM::decodeLogicalImmediate( 4717 DefMI->getOperand(2).getImm(), Is32Bit ? 32 : 64); 4718 if (!isPowerOf2_64(Mask)) 4719 return false; 4720 4721 MachineOperand &MO = DefMI->getOperand(1); 4722 Register NewReg = MO.getReg(); 4723 if (!Register::isVirtualRegister(NewReg)) 4724 return false; 4725 4726 assert(!MRI->def_empty(NewReg) && "Register must be defined."); 4727 4728 MachineBasicBlock &RefToMBB = *MBB; 4729 MachineBasicBlock *TBB = MI.getOperand(1).getMBB(); 4730 DebugLoc DL = MI.getDebugLoc(); 4731 unsigned Imm = Log2_64(Mask); 4732 unsigned Opc = (Imm < 32) 4733 ? (IsNegativeBranch ? AArch64::TBNZW : AArch64::TBZW) 4734 : (IsNegativeBranch ? AArch64::TBNZX : AArch64::TBZX); 4735 MachineInstr *NewMI = BuildMI(RefToMBB, MI, DL, get(Opc)) 4736 .addReg(NewReg) 4737 .addImm(Imm) 4738 .addMBB(TBB); 4739 // Register lives on to the CBZ now. 4740 MO.setIsKill(false); 4741 4742 // For immediate smaller than 32, we need to use the 32-bit 4743 // variant (W) in all cases. Indeed the 64-bit variant does not 4744 // allow to encode them. 4745 // Therefore, if the input register is 64-bit, we need to take the 4746 // 32-bit sub-part. 4747 if (!Is32Bit && Imm < 32) 4748 NewMI->getOperand(0).setSubReg(AArch64::sub_32); 4749 MI.eraseFromParent(); 4750 return true; 4751 } 4752 // Look for CSINC 4753 case AArch64::CSINCWr: 4754 case AArch64::CSINCXr: { 4755 if (!(DefMI->getOperand(1).getReg() == AArch64::WZR && 4756 DefMI->getOperand(2).getReg() == AArch64::WZR) && 4757 !(DefMI->getOperand(1).getReg() == AArch64::XZR && 4758 DefMI->getOperand(2).getReg() == AArch64::XZR)) 4759 return false; 4760 4761 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, true) != -1) 4762 return false; 4763 4764 AArch64CC::CondCode CC = (AArch64CC::CondCode)DefMI->getOperand(3).getImm(); 4765 // Convert only when the condition code is not modified between 4766 // the CSINC and the branch. The CC may be used by other 4767 // instructions in between. 4768 if (areCFlagsAccessedBetweenInstrs(DefMI, MI, &getRegisterInfo(), AK_Write)) 4769 return false; 4770 MachineBasicBlock &RefToMBB = *MBB; 4771 MachineBasicBlock *TBB = MI.getOperand(TargetBBInMI).getMBB(); 4772 DebugLoc DL = MI.getDebugLoc(); 4773 if (IsNegativeBranch) 4774 CC = AArch64CC::getInvertedCondCode(CC); 4775 BuildMI(RefToMBB, MI, DL, get(AArch64::Bcc)).addImm(CC).addMBB(TBB); 4776 MI.eraseFromParent(); 4777 return true; 4778 } 4779 } 4780 } 4781 4782 std::pair<unsigned, unsigned> 4783 AArch64InstrInfo::decomposeMachineOperandsTargetFlags(unsigned TF) const { 4784 const unsigned Mask = AArch64II::MO_FRAGMENT; 4785 return std::make_pair(TF & Mask, TF & ~Mask); 4786 } 4787 4788 ArrayRef<std::pair<unsigned, const char *>> 4789 AArch64InstrInfo::getSerializableDirectMachineOperandTargetFlags() const { 4790 using namespace AArch64II; 4791 4792 static const std::pair<unsigned, const char *> TargetFlags[] = { 4793 {MO_PAGE, "aarch64-page"}, {MO_PAGEOFF, "aarch64-pageoff"}, 4794 {MO_G3, "aarch64-g3"}, {MO_G2, "aarch64-g2"}, 4795 {MO_G1, "aarch64-g1"}, {MO_G0, "aarch64-g0"}, 4796 {MO_HI12, "aarch64-hi12"}}; 4797 return makeArrayRef(TargetFlags); 4798 } 4799 4800 ArrayRef<std::pair<unsigned, const char *>> 4801 AArch64InstrInfo::getSerializableBitmaskMachineOperandTargetFlags() const { 4802 using namespace AArch64II; 4803 4804 static const std::pair<unsigned, const char *> TargetFlags[] = { 4805 {MO_COFFSTUB, "aarch64-coffstub"}, 4806 {MO_GOT, "aarch64-got"}, 4807 {MO_NC, "aarch64-nc"}, 4808 {MO_S, "aarch64-s"}, 4809 {MO_TLS, "aarch64-tls"}, 4810 {MO_DLLIMPORT, "aarch64-dllimport"}, 4811 {MO_PREL, "aarch64-prel"}, 4812 {MO_TAGGED, "aarch64-tagged"}}; 4813 return makeArrayRef(TargetFlags); 4814 } 4815 4816 ArrayRef<std::pair<MachineMemOperand::Flags, const char *>> 4817 AArch64InstrInfo::getSerializableMachineMemOperandTargetFlags() const { 4818 static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] = 4819 {{MOSuppressPair, "aarch64-suppress-pair"}, 4820 {MOStridedAccess, "aarch64-strided-access"}}; 4821 return makeArrayRef(TargetFlags); 4822 } 4823 4824 /// Constants defining how certain sequences should be outlined. 4825 /// This encompasses how an outlined function should be called, and what kind of 4826 /// frame should be emitted for that outlined function. 4827 /// 4828 /// \p MachineOutlinerDefault implies that the function should be called with 4829 /// a save and restore of LR to the stack. 4830 /// 4831 /// That is, 4832 /// 4833 /// I1 Save LR OUTLINED_FUNCTION: 4834 /// I2 --> BL OUTLINED_FUNCTION I1 4835 /// I3 Restore LR I2 4836 /// I3 4837 /// RET 4838 /// 4839 /// * Call construction overhead: 3 (save + BL + restore) 4840 /// * Frame construction overhead: 1 (ret) 4841 /// * Requires stack fixups? Yes 4842 /// 4843 /// \p MachineOutlinerTailCall implies that the function is being created from 4844 /// a sequence of instructions ending in a return. 4845 /// 4846 /// That is, 4847 /// 4848 /// I1 OUTLINED_FUNCTION: 4849 /// I2 --> B OUTLINED_FUNCTION I1 4850 /// RET I2 4851 /// RET 4852 /// 4853 /// * Call construction overhead: 1 (B) 4854 /// * Frame construction overhead: 0 (Return included in sequence) 4855 /// * Requires stack fixups? No 4856 /// 4857 /// \p MachineOutlinerNoLRSave implies that the function should be called using 4858 /// a BL instruction, but doesn't require LR to be saved and restored. This 4859 /// happens when LR is known to be dead. 4860 /// 4861 /// That is, 4862 /// 4863 /// I1 OUTLINED_FUNCTION: 4864 /// I2 --> BL OUTLINED_FUNCTION I1 4865 /// I3 I2 4866 /// I3 4867 /// RET 4868 /// 4869 /// * Call construction overhead: 1 (BL) 4870 /// * Frame construction overhead: 1 (RET) 4871 /// * Requires stack fixups? No 4872 /// 4873 /// \p MachineOutlinerThunk implies that the function is being created from 4874 /// a sequence of instructions ending in a call. The outlined function is 4875 /// called with a BL instruction, and the outlined function tail-calls the 4876 /// original call destination. 4877 /// 4878 /// That is, 4879 /// 4880 /// I1 OUTLINED_FUNCTION: 4881 /// I2 --> BL OUTLINED_FUNCTION I1 4882 /// BL f I2 4883 /// B f 4884 /// * Call construction overhead: 1 (BL) 4885 /// * Frame construction overhead: 0 4886 /// * Requires stack fixups? No 4887 /// 4888 /// \p MachineOutlinerRegSave implies that the function should be called with a 4889 /// save and restore of LR to an available register. This allows us to avoid 4890 /// stack fixups. Note that this outlining variant is compatible with the 4891 /// NoLRSave case. 4892 /// 4893 /// That is, 4894 /// 4895 /// I1 Save LR OUTLINED_FUNCTION: 4896 /// I2 --> BL OUTLINED_FUNCTION I1 4897 /// I3 Restore LR I2 4898 /// I3 4899 /// RET 4900 /// 4901 /// * Call construction overhead: 3 (save + BL + restore) 4902 /// * Frame construction overhead: 1 (ret) 4903 /// * Requires stack fixups? No 4904 enum MachineOutlinerClass { 4905 MachineOutlinerDefault, /// Emit a save, restore, call, and return. 4906 MachineOutlinerTailCall, /// Only emit a branch. 4907 MachineOutlinerNoLRSave, /// Emit a call and return. 4908 MachineOutlinerThunk, /// Emit a call and tail-call. 4909 MachineOutlinerRegSave /// Same as default, but save to a register. 4910 }; 4911 4912 enum MachineOutlinerMBBFlags { 4913 LRUnavailableSomewhere = 0x2, 4914 HasCalls = 0x4, 4915 UnsafeRegsDead = 0x8 4916 }; 4917 4918 unsigned 4919 AArch64InstrInfo::findRegisterToSaveLRTo(const outliner::Candidate &C) const { 4920 assert(C.LRUWasSet && "LRU wasn't set?"); 4921 MachineFunction *MF = C.getMF(); 4922 const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>( 4923 MF->getSubtarget().getRegisterInfo()); 4924 4925 // Check if there is an available register across the sequence that we can 4926 // use. 4927 for (unsigned Reg : AArch64::GPR64RegClass) { 4928 if (!ARI->isReservedReg(*MF, Reg) && 4929 Reg != AArch64::LR && // LR is not reserved, but don't use it. 4930 Reg != AArch64::X16 && // X16 is not guaranteed to be preserved. 4931 Reg != AArch64::X17 && // Ditto for X17. 4932 C.LRU.available(Reg) && C.UsedInSequence.available(Reg)) 4933 return Reg; 4934 } 4935 4936 // No suitable register. Return 0. 4937 return 0u; 4938 } 4939 4940 outliner::OutlinedFunction 4941 AArch64InstrInfo::getOutliningCandidateInfo( 4942 std::vector<outliner::Candidate> &RepeatedSequenceLocs) const { 4943 outliner::Candidate &FirstCand = RepeatedSequenceLocs[0]; 4944 unsigned SequenceSize = 4945 std::accumulate(FirstCand.front(), std::next(FirstCand.back()), 0, 4946 [this](unsigned Sum, const MachineInstr &MI) { 4947 return Sum + getInstSizeInBytes(MI); 4948 }); 4949 4950 // Properties about candidate MBBs that hold for all of them. 4951 unsigned FlagsSetInAll = 0xF; 4952 4953 // Compute liveness information for each candidate, and set FlagsSetInAll. 4954 const TargetRegisterInfo &TRI = getRegisterInfo(); 4955 std::for_each(RepeatedSequenceLocs.begin(), RepeatedSequenceLocs.end(), 4956 [&FlagsSetInAll](outliner::Candidate &C) { 4957 FlagsSetInAll &= C.Flags; 4958 }); 4959 4960 // According to the AArch64 Procedure Call Standard, the following are 4961 // undefined on entry/exit from a function call: 4962 // 4963 // * Registers x16, x17, (and thus w16, w17) 4964 // * Condition codes (and thus the NZCV register) 4965 // 4966 // Because if this, we can't outline any sequence of instructions where 4967 // one 4968 // of these registers is live into/across it. Thus, we need to delete 4969 // those 4970 // candidates. 4971 auto CantGuaranteeValueAcrossCall = [&TRI](outliner::Candidate &C) { 4972 // If the unsafe registers in this block are all dead, then we don't need 4973 // to compute liveness here. 4974 if (C.Flags & UnsafeRegsDead) 4975 return false; 4976 C.initLRU(TRI); 4977 LiveRegUnits LRU = C.LRU; 4978 return (!LRU.available(AArch64::W16) || !LRU.available(AArch64::W17) || 4979 !LRU.available(AArch64::NZCV)); 4980 }; 4981 4982 // Are there any candidates where those registers are live? 4983 if (!(FlagsSetInAll & UnsafeRegsDead)) { 4984 // Erase every candidate that violates the restrictions above. (It could be 4985 // true that we have viable candidates, so it's not worth bailing out in 4986 // the case that, say, 1 out of 20 candidates violate the restructions.) 4987 RepeatedSequenceLocs.erase(std::remove_if(RepeatedSequenceLocs.begin(), 4988 RepeatedSequenceLocs.end(), 4989 CantGuaranteeValueAcrossCall), 4990 RepeatedSequenceLocs.end()); 4991 4992 // If the sequence doesn't have enough candidates left, then we're done. 4993 if (RepeatedSequenceLocs.size() < 2) 4994 return outliner::OutlinedFunction(); 4995 } 4996 4997 // At this point, we have only "safe" candidates to outline. Figure out 4998 // frame + call instruction information. 4999 5000 unsigned LastInstrOpcode = RepeatedSequenceLocs[0].back()->getOpcode(); 5001 5002 // Helper lambda which sets call information for every candidate. 5003 auto SetCandidateCallInfo = 5004 [&RepeatedSequenceLocs](unsigned CallID, unsigned NumBytesForCall) { 5005 for (outliner::Candidate &C : RepeatedSequenceLocs) 5006 C.setCallInfo(CallID, NumBytesForCall); 5007 }; 5008 5009 unsigned FrameID = MachineOutlinerDefault; 5010 unsigned NumBytesToCreateFrame = 4; 5011 5012 bool HasBTI = any_of(RepeatedSequenceLocs, [](outliner::Candidate &C) { 5013 return C.getMF()->getFunction().hasFnAttribute("branch-target-enforcement"); 5014 }); 5015 5016 // Returns true if an instructions is safe to fix up, false otherwise. 5017 auto IsSafeToFixup = [this, &TRI](MachineInstr &MI) { 5018 if (MI.isCall()) 5019 return true; 5020 5021 if (!MI.modifiesRegister(AArch64::SP, &TRI) && 5022 !MI.readsRegister(AArch64::SP, &TRI)) 5023 return true; 5024 5025 // Any modification of SP will break our code to save/restore LR. 5026 // FIXME: We could handle some instructions which add a constant 5027 // offset to SP, with a bit more work. 5028 if (MI.modifiesRegister(AArch64::SP, &TRI)) 5029 return false; 5030 5031 // At this point, we have a stack instruction that we might need to 5032 // fix up. We'll handle it if it's a load or store. 5033 if (MI.mayLoadOrStore()) { 5034 const MachineOperand *Base; // Filled with the base operand of MI. 5035 int64_t Offset; // Filled with the offset of MI. 5036 5037 // Does it allow us to offset the base operand and is the base the 5038 // register SP? 5039 if (!getMemOperandWithOffset(MI, Base, Offset, &TRI) || !Base->isReg() || 5040 Base->getReg() != AArch64::SP) 5041 return false; 5042 5043 // Find the minimum/maximum offset for this instruction and check 5044 // if fixing it up would be in range. 5045 int64_t MinOffset, 5046 MaxOffset; // Unscaled offsets for the instruction. 5047 unsigned Scale; // The scale to multiply the offsets by. 5048 unsigned DummyWidth; 5049 getMemOpInfo(MI.getOpcode(), Scale, DummyWidth, MinOffset, MaxOffset); 5050 5051 Offset += 16; // Update the offset to what it would be if we outlined. 5052 if (Offset < MinOffset * Scale || Offset > MaxOffset * Scale) 5053 return false; 5054 5055 // It's in range, so we can outline it. 5056 return true; 5057 } 5058 5059 // FIXME: Add handling for instructions like "add x0, sp, #8". 5060 5061 // We can't fix it up, so don't outline it. 5062 return false; 5063 }; 5064 5065 // True if it's possible to fix up each stack instruction in this sequence. 5066 // Important for frames/call variants that modify the stack. 5067 bool AllStackInstrsSafe = std::all_of( 5068 FirstCand.front(), std::next(FirstCand.back()), IsSafeToFixup); 5069 5070 // If the last instruction in any candidate is a terminator, then we should 5071 // tail call all of the candidates. 5072 if (RepeatedSequenceLocs[0].back()->isTerminator()) { 5073 FrameID = MachineOutlinerTailCall; 5074 NumBytesToCreateFrame = 0; 5075 SetCandidateCallInfo(MachineOutlinerTailCall, 4); 5076 } 5077 5078 else if (LastInstrOpcode == AArch64::BL || 5079 (LastInstrOpcode == AArch64::BLR && !HasBTI)) { 5080 // FIXME: Do we need to check if the code after this uses the value of LR? 5081 FrameID = MachineOutlinerThunk; 5082 NumBytesToCreateFrame = 0; 5083 SetCandidateCallInfo(MachineOutlinerThunk, 4); 5084 } 5085 5086 else { 5087 // We need to decide how to emit calls + frames. We can always emit the same 5088 // frame if we don't need to save to the stack. If we have to save to the 5089 // stack, then we need a different frame. 5090 unsigned NumBytesNoStackCalls = 0; 5091 std::vector<outliner::Candidate> CandidatesWithoutStackFixups; 5092 5093 for (outliner::Candidate &C : RepeatedSequenceLocs) { 5094 C.initLRU(TRI); 5095 5096 // Is LR available? If so, we don't need a save. 5097 if (C.LRU.available(AArch64::LR)) { 5098 NumBytesNoStackCalls += 4; 5099 C.setCallInfo(MachineOutlinerNoLRSave, 4); 5100 CandidatesWithoutStackFixups.push_back(C); 5101 } 5102 5103 // Is an unused register available? If so, we won't modify the stack, so 5104 // we can outline with the same frame type as those that don't save LR. 5105 else if (findRegisterToSaveLRTo(C)) { 5106 NumBytesNoStackCalls += 12; 5107 C.setCallInfo(MachineOutlinerRegSave, 12); 5108 CandidatesWithoutStackFixups.push_back(C); 5109 } 5110 5111 // Is SP used in the sequence at all? If not, we don't have to modify 5112 // the stack, so we are guaranteed to get the same frame. 5113 else if (C.UsedInSequence.available(AArch64::SP)) { 5114 NumBytesNoStackCalls += 12; 5115 C.setCallInfo(MachineOutlinerDefault, 12); 5116 CandidatesWithoutStackFixups.push_back(C); 5117 } 5118 5119 // If we outline this, we need to modify the stack. Pretend we don't 5120 // outline this by saving all of its bytes. 5121 else { 5122 NumBytesNoStackCalls += SequenceSize; 5123 } 5124 } 5125 5126 // If there are no places where we have to save LR, then note that we 5127 // don't have to update the stack. Otherwise, give every candidate the 5128 // default call type, as long as it's safe to do so. 5129 if (!AllStackInstrsSafe || 5130 NumBytesNoStackCalls <= RepeatedSequenceLocs.size() * 12) { 5131 RepeatedSequenceLocs = CandidatesWithoutStackFixups; 5132 FrameID = MachineOutlinerNoLRSave; 5133 } else { 5134 SetCandidateCallInfo(MachineOutlinerDefault, 12); 5135 } 5136 5137 // If we dropped all of the candidates, bail out here. 5138 if (RepeatedSequenceLocs.size() < 2) { 5139 RepeatedSequenceLocs.clear(); 5140 return outliner::OutlinedFunction(); 5141 } 5142 } 5143 5144 // Does every candidate's MBB contain a call? If so, then we might have a call 5145 // in the range. 5146 if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) { 5147 // Check if the range contains a call. These require a save + restore of the 5148 // link register. 5149 bool ModStackToSaveLR = false; 5150 if (std::any_of(FirstCand.front(), FirstCand.back(), 5151 [](const MachineInstr &MI) { return MI.isCall(); })) 5152 ModStackToSaveLR = true; 5153 5154 // Handle the last instruction separately. If this is a tail call, then the 5155 // last instruction is a call. We don't want to save + restore in this case. 5156 // However, it could be possible that the last instruction is a call without 5157 // it being valid to tail call this sequence. We should consider this as 5158 // well. 5159 else if (FrameID != MachineOutlinerThunk && 5160 FrameID != MachineOutlinerTailCall && FirstCand.back()->isCall()) 5161 ModStackToSaveLR = true; 5162 5163 if (ModStackToSaveLR) { 5164 // We can't fix up the stack. Bail out. 5165 if (!AllStackInstrsSafe) { 5166 RepeatedSequenceLocs.clear(); 5167 return outliner::OutlinedFunction(); 5168 } 5169 5170 // Save + restore LR. 5171 NumBytesToCreateFrame += 8; 5172 } 5173 } 5174 5175 return outliner::OutlinedFunction(RepeatedSequenceLocs, SequenceSize, 5176 NumBytesToCreateFrame, FrameID); 5177 } 5178 5179 bool AArch64InstrInfo::isFunctionSafeToOutlineFrom( 5180 MachineFunction &MF, bool OutlineFromLinkOnceODRs) const { 5181 const Function &F = MF.getFunction(); 5182 5183 // Can F be deduplicated by the linker? If it can, don't outline from it. 5184 if (!OutlineFromLinkOnceODRs && F.hasLinkOnceODRLinkage()) 5185 return false; 5186 5187 // Don't outline from functions with section markings; the program could 5188 // expect that all the code is in the named section. 5189 // FIXME: Allow outlining from multiple functions with the same section 5190 // marking. 5191 if (F.hasSection()) 5192 return false; 5193 5194 // Outlining from functions with redzones is unsafe since the outliner may 5195 // modify the stack. Check if hasRedZone is true or unknown; if yes, don't 5196 // outline from it. 5197 AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>(); 5198 if (!AFI || AFI->hasRedZone().getValueOr(true)) 5199 return false; 5200 5201 // It's safe to outline from MF. 5202 return true; 5203 } 5204 5205 bool AArch64InstrInfo::isMBBSafeToOutlineFrom(MachineBasicBlock &MBB, 5206 unsigned &Flags) const { 5207 // Check if LR is available through all of the MBB. If it's not, then set 5208 // a flag. 5209 assert(MBB.getParent()->getRegInfo().tracksLiveness() && 5210 "Suitable Machine Function for outlining must track liveness"); 5211 LiveRegUnits LRU(getRegisterInfo()); 5212 5213 std::for_each(MBB.rbegin(), MBB.rend(), 5214 [&LRU](MachineInstr &MI) { LRU.accumulate(MI); }); 5215 5216 // Check if each of the unsafe registers are available... 5217 bool W16AvailableInBlock = LRU.available(AArch64::W16); 5218 bool W17AvailableInBlock = LRU.available(AArch64::W17); 5219 bool NZCVAvailableInBlock = LRU.available(AArch64::NZCV); 5220 5221 // If all of these are dead (and not live out), we know we don't have to check 5222 // them later. 5223 if (W16AvailableInBlock && W17AvailableInBlock && NZCVAvailableInBlock) 5224 Flags |= MachineOutlinerMBBFlags::UnsafeRegsDead; 5225 5226 // Now, add the live outs to the set. 5227 LRU.addLiveOuts(MBB); 5228 5229 // If any of these registers is available in the MBB, but also a live out of 5230 // the block, then we know outlining is unsafe. 5231 if (W16AvailableInBlock && !LRU.available(AArch64::W16)) 5232 return false; 5233 if (W17AvailableInBlock && !LRU.available(AArch64::W17)) 5234 return false; 5235 if (NZCVAvailableInBlock && !LRU.available(AArch64::NZCV)) 5236 return false; 5237 5238 // Check if there's a call inside this MachineBasicBlock. If there is, then 5239 // set a flag. 5240 if (any_of(MBB, [](MachineInstr &MI) { return MI.isCall(); })) 5241 Flags |= MachineOutlinerMBBFlags::HasCalls; 5242 5243 MachineFunction *MF = MBB.getParent(); 5244 5245 // In the event that we outline, we may have to save LR. If there is an 5246 // available register in the MBB, then we'll always save LR there. Check if 5247 // this is true. 5248 bool CanSaveLR = false; 5249 const AArch64RegisterInfo *ARI = static_cast<const AArch64RegisterInfo *>( 5250 MF->getSubtarget().getRegisterInfo()); 5251 5252 // Check if there is an available register across the sequence that we can 5253 // use. 5254 for (unsigned Reg : AArch64::GPR64RegClass) { 5255 if (!ARI->isReservedReg(*MF, Reg) && Reg != AArch64::LR && 5256 Reg != AArch64::X16 && Reg != AArch64::X17 && LRU.available(Reg)) { 5257 CanSaveLR = true; 5258 break; 5259 } 5260 } 5261 5262 // Check if we have a register we can save LR to, and if LR was used 5263 // somewhere. If both of those things are true, then we need to evaluate the 5264 // safety of outlining stack instructions later. 5265 if (!CanSaveLR && !LRU.available(AArch64::LR)) 5266 Flags |= MachineOutlinerMBBFlags::LRUnavailableSomewhere; 5267 5268 return true; 5269 } 5270 5271 outliner::InstrType 5272 AArch64InstrInfo::getOutliningType(MachineBasicBlock::iterator &MIT, 5273 unsigned Flags) const { 5274 MachineInstr &MI = *MIT; 5275 MachineBasicBlock *MBB = MI.getParent(); 5276 MachineFunction *MF = MBB->getParent(); 5277 AArch64FunctionInfo *FuncInfo = MF->getInfo<AArch64FunctionInfo>(); 5278 5279 // Don't outline LOHs. 5280 if (FuncInfo->getLOHRelated().count(&MI)) 5281 return outliner::InstrType::Illegal; 5282 5283 // Don't allow debug values to impact outlining type. 5284 if (MI.isDebugInstr() || MI.isIndirectDebugValue()) 5285 return outliner::InstrType::Invisible; 5286 5287 // At this point, KILL instructions don't really tell us much so we can go 5288 // ahead and skip over them. 5289 if (MI.isKill()) 5290 return outliner::InstrType::Invisible; 5291 5292 // Is this a terminator for a basic block? 5293 if (MI.isTerminator()) { 5294 5295 // Is this the end of a function? 5296 if (MI.getParent()->succ_empty()) 5297 return outliner::InstrType::Legal; 5298 5299 // It's not, so don't outline it. 5300 return outliner::InstrType::Illegal; 5301 } 5302 5303 // Make sure none of the operands are un-outlinable. 5304 for (const MachineOperand &MOP : MI.operands()) { 5305 if (MOP.isCPI() || MOP.isJTI() || MOP.isCFIIndex() || MOP.isFI() || 5306 MOP.isTargetIndex()) 5307 return outliner::InstrType::Illegal; 5308 5309 // If it uses LR or W30 explicitly, then don't touch it. 5310 if (MOP.isReg() && !MOP.isImplicit() && 5311 (MOP.getReg() == AArch64::LR || MOP.getReg() == AArch64::W30)) 5312 return outliner::InstrType::Illegal; 5313 } 5314 5315 // Special cases for instructions that can always be outlined, but will fail 5316 // the later tests. e.g, ADRPs, which are PC-relative use LR, but can always 5317 // be outlined because they don't require a *specific* value to be in LR. 5318 if (MI.getOpcode() == AArch64::ADRP) 5319 return outliner::InstrType::Legal; 5320 5321 // If MI is a call we might be able to outline it. We don't want to outline 5322 // any calls that rely on the position of items on the stack. When we outline 5323 // something containing a call, we have to emit a save and restore of LR in 5324 // the outlined function. Currently, this always happens by saving LR to the 5325 // stack. Thus, if we outline, say, half the parameters for a function call 5326 // plus the call, then we'll break the callee's expectations for the layout 5327 // of the stack. 5328 // 5329 // FIXME: Allow calls to functions which construct a stack frame, as long 5330 // as they don't access arguments on the stack. 5331 // FIXME: Figure out some way to analyze functions defined in other modules. 5332 // We should be able to compute the memory usage based on the IR calling 5333 // convention, even if we can't see the definition. 5334 if (MI.isCall()) { 5335 // Get the function associated with the call. Look at each operand and find 5336 // the one that represents the callee and get its name. 5337 const Function *Callee = nullptr; 5338 for (const MachineOperand &MOP : MI.operands()) { 5339 if (MOP.isGlobal()) { 5340 Callee = dyn_cast<Function>(MOP.getGlobal()); 5341 break; 5342 } 5343 } 5344 5345 // Never outline calls to mcount. There isn't any rule that would require 5346 // this, but the Linux kernel's "ftrace" feature depends on it. 5347 if (Callee && Callee->getName() == "\01_mcount") 5348 return outliner::InstrType::Illegal; 5349 5350 // If we don't know anything about the callee, assume it depends on the 5351 // stack layout of the caller. In that case, it's only legal to outline 5352 // as a tail-call. Whitelist the call instructions we know about so we 5353 // don't get unexpected results with call pseudo-instructions. 5354 auto UnknownCallOutlineType = outliner::InstrType::Illegal; 5355 if (MI.getOpcode() == AArch64::BLR || MI.getOpcode() == AArch64::BL) 5356 UnknownCallOutlineType = outliner::InstrType::LegalTerminator; 5357 5358 if (!Callee) 5359 return UnknownCallOutlineType; 5360 5361 // We have a function we have information about. Check it if it's something 5362 // can safely outline. 5363 MachineFunction *CalleeMF = MF->getMMI().getMachineFunction(*Callee); 5364 5365 // We don't know what's going on with the callee at all. Don't touch it. 5366 if (!CalleeMF) 5367 return UnknownCallOutlineType; 5368 5369 // Check if we know anything about the callee saves on the function. If we 5370 // don't, then don't touch it, since that implies that we haven't 5371 // computed anything about its stack frame yet. 5372 MachineFrameInfo &MFI = CalleeMF->getFrameInfo(); 5373 if (!MFI.isCalleeSavedInfoValid() || MFI.getStackSize() > 0 || 5374 MFI.getNumObjects() > 0) 5375 return UnknownCallOutlineType; 5376 5377 // At this point, we can say that CalleeMF ought to not pass anything on the 5378 // stack. Therefore, we can outline it. 5379 return outliner::InstrType::Legal; 5380 } 5381 5382 // Don't outline positions. 5383 if (MI.isPosition()) 5384 return outliner::InstrType::Illegal; 5385 5386 // Don't touch the link register or W30. 5387 if (MI.readsRegister(AArch64::W30, &getRegisterInfo()) || 5388 MI.modifiesRegister(AArch64::W30, &getRegisterInfo())) 5389 return outliner::InstrType::Illegal; 5390 5391 // Don't outline BTI instructions, because that will prevent the outlining 5392 // site from being indirectly callable. 5393 if (MI.getOpcode() == AArch64::HINT) { 5394 int64_t Imm = MI.getOperand(0).getImm(); 5395 if (Imm == 32 || Imm == 34 || Imm == 36 || Imm == 38) 5396 return outliner::InstrType::Illegal; 5397 } 5398 5399 return outliner::InstrType::Legal; 5400 } 5401 5402 void AArch64InstrInfo::fixupPostOutline(MachineBasicBlock &MBB) const { 5403 for (MachineInstr &MI : MBB) { 5404 const MachineOperand *Base; 5405 unsigned Width; 5406 int64_t Offset; 5407 5408 // Is this a load or store with an immediate offset with SP as the base? 5409 if (!MI.mayLoadOrStore() || 5410 !getMemOperandWithOffsetWidth(MI, Base, Offset, Width, &RI) || 5411 (Base->isReg() && Base->getReg() != AArch64::SP)) 5412 continue; 5413 5414 // It is, so we have to fix it up. 5415 unsigned Scale; 5416 int64_t Dummy1, Dummy2; 5417 5418 MachineOperand &StackOffsetOperand = getMemOpBaseRegImmOfsOffsetOperand(MI); 5419 assert(StackOffsetOperand.isImm() && "Stack offset wasn't immediate!"); 5420 getMemOpInfo(MI.getOpcode(), Scale, Width, Dummy1, Dummy2); 5421 assert(Scale != 0 && "Unexpected opcode!"); 5422 5423 // We've pushed the return address to the stack, so add 16 to the offset. 5424 // This is safe, since we already checked if it would overflow when we 5425 // checked if this instruction was legal to outline. 5426 int64_t NewImm = (Offset + 16) / Scale; 5427 StackOffsetOperand.setImm(NewImm); 5428 } 5429 } 5430 5431 void AArch64InstrInfo::buildOutlinedFrame( 5432 MachineBasicBlock &MBB, MachineFunction &MF, 5433 const outliner::OutlinedFunction &OF) const { 5434 // For thunk outlining, rewrite the last instruction from a call to a 5435 // tail-call. 5436 if (OF.FrameConstructionID == MachineOutlinerThunk) { 5437 MachineInstr *Call = &*--MBB.instr_end(); 5438 unsigned TailOpcode; 5439 if (Call->getOpcode() == AArch64::BL) { 5440 TailOpcode = AArch64::TCRETURNdi; 5441 } else { 5442 assert(Call->getOpcode() == AArch64::BLR); 5443 TailOpcode = AArch64::TCRETURNriALL; 5444 } 5445 MachineInstr *TC = BuildMI(MF, DebugLoc(), get(TailOpcode)) 5446 .add(Call->getOperand(0)) 5447 .addImm(0); 5448 MBB.insert(MBB.end(), TC); 5449 Call->eraseFromParent(); 5450 } 5451 5452 // Is there a call in the outlined range? 5453 auto IsNonTailCall = [](MachineInstr &MI) { 5454 return MI.isCall() && !MI.isReturn(); 5455 }; 5456 if (std::any_of(MBB.instr_begin(), MBB.instr_end(), IsNonTailCall)) { 5457 // Fix up the instructions in the range, since we're going to modify the 5458 // stack. 5459 assert(OF.FrameConstructionID != MachineOutlinerDefault && 5460 "Can only fix up stack references once"); 5461 fixupPostOutline(MBB); 5462 5463 // LR has to be a live in so that we can save it. 5464 MBB.addLiveIn(AArch64::LR); 5465 5466 MachineBasicBlock::iterator It = MBB.begin(); 5467 MachineBasicBlock::iterator Et = MBB.end(); 5468 5469 if (OF.FrameConstructionID == MachineOutlinerTailCall || 5470 OF.FrameConstructionID == MachineOutlinerThunk) 5471 Et = std::prev(MBB.end()); 5472 5473 // Insert a save before the outlined region 5474 MachineInstr *STRXpre = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre)) 5475 .addReg(AArch64::SP, RegState::Define) 5476 .addReg(AArch64::LR) 5477 .addReg(AArch64::SP) 5478 .addImm(-16); 5479 It = MBB.insert(It, STRXpre); 5480 5481 const TargetSubtargetInfo &STI = MF.getSubtarget(); 5482 const MCRegisterInfo *MRI = STI.getRegisterInfo(); 5483 unsigned DwarfReg = MRI->getDwarfRegNum(AArch64::LR, true); 5484 5485 // Add a CFI saying the stack was moved 16 B down. 5486 int64_t StackPosEntry = 5487 MF.addFrameInst(MCCFIInstruction::createDefCfaOffset(nullptr, 16)); 5488 BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION)) 5489 .addCFIIndex(StackPosEntry) 5490 .setMIFlags(MachineInstr::FrameSetup); 5491 5492 // Add a CFI saying that the LR that we want to find is now 16 B higher than 5493 // before. 5494 int64_t LRPosEntry = 5495 MF.addFrameInst(MCCFIInstruction::createOffset(nullptr, DwarfReg, 16)); 5496 BuildMI(MBB, It, DebugLoc(), get(AArch64::CFI_INSTRUCTION)) 5497 .addCFIIndex(LRPosEntry) 5498 .setMIFlags(MachineInstr::FrameSetup); 5499 5500 // Insert a restore before the terminator for the function. 5501 MachineInstr *LDRXpost = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost)) 5502 .addReg(AArch64::SP, RegState::Define) 5503 .addReg(AArch64::LR, RegState::Define) 5504 .addReg(AArch64::SP) 5505 .addImm(16); 5506 Et = MBB.insert(Et, LDRXpost); 5507 } 5508 5509 // If this is a tail call outlined function, then there's already a return. 5510 if (OF.FrameConstructionID == MachineOutlinerTailCall || 5511 OF.FrameConstructionID == MachineOutlinerThunk) 5512 return; 5513 5514 // It's not a tail call, so we have to insert the return ourselves. 5515 MachineInstr *ret = BuildMI(MF, DebugLoc(), get(AArch64::RET)) 5516 .addReg(AArch64::LR, RegState::Undef); 5517 MBB.insert(MBB.end(), ret); 5518 5519 // Did we have to modify the stack by saving the link register? 5520 if (OF.FrameConstructionID != MachineOutlinerDefault) 5521 return; 5522 5523 // We modified the stack. 5524 // Walk over the basic block and fix up all the stack accesses. 5525 fixupPostOutline(MBB); 5526 } 5527 5528 MachineBasicBlock::iterator AArch64InstrInfo::insertOutlinedCall( 5529 Module &M, MachineBasicBlock &MBB, MachineBasicBlock::iterator &It, 5530 MachineFunction &MF, const outliner::Candidate &C) const { 5531 5532 // Are we tail calling? 5533 if (C.CallConstructionID == MachineOutlinerTailCall) { 5534 // If yes, then we can just branch to the label. 5535 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::TCRETURNdi)) 5536 .addGlobalAddress(M.getNamedValue(MF.getName())) 5537 .addImm(0)); 5538 return It; 5539 } 5540 5541 // Are we saving the link register? 5542 if (C.CallConstructionID == MachineOutlinerNoLRSave || 5543 C.CallConstructionID == MachineOutlinerThunk) { 5544 // No, so just insert the call. 5545 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL)) 5546 .addGlobalAddress(M.getNamedValue(MF.getName()))); 5547 return It; 5548 } 5549 5550 // We want to return the spot where we inserted the call. 5551 MachineBasicBlock::iterator CallPt; 5552 5553 // Instructions for saving and restoring LR around the call instruction we're 5554 // going to insert. 5555 MachineInstr *Save; 5556 MachineInstr *Restore; 5557 // Can we save to a register? 5558 if (C.CallConstructionID == MachineOutlinerRegSave) { 5559 // FIXME: This logic should be sunk into a target-specific interface so that 5560 // we don't have to recompute the register. 5561 unsigned Reg = findRegisterToSaveLRTo(C); 5562 assert(Reg != 0 && "No callee-saved register available?"); 5563 5564 // Save and restore LR from that register. 5565 Save = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), Reg) 5566 .addReg(AArch64::XZR) 5567 .addReg(AArch64::LR) 5568 .addImm(0); 5569 Restore = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), AArch64::LR) 5570 .addReg(AArch64::XZR) 5571 .addReg(Reg) 5572 .addImm(0); 5573 } else { 5574 // We have the default case. Save and restore from SP. 5575 Save = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre)) 5576 .addReg(AArch64::SP, RegState::Define) 5577 .addReg(AArch64::LR) 5578 .addReg(AArch64::SP) 5579 .addImm(-16); 5580 Restore = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost)) 5581 .addReg(AArch64::SP, RegState::Define) 5582 .addReg(AArch64::LR, RegState::Define) 5583 .addReg(AArch64::SP) 5584 .addImm(16); 5585 } 5586 5587 It = MBB.insert(It, Save); 5588 It++; 5589 5590 // Insert the call. 5591 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL)) 5592 .addGlobalAddress(M.getNamedValue(MF.getName()))); 5593 CallPt = It; 5594 It++; 5595 5596 It = MBB.insert(It, Restore); 5597 return CallPt; 5598 } 5599 5600 bool AArch64InstrInfo::shouldOutlineFromFunctionByDefault( 5601 MachineFunction &MF) const { 5602 return MF.getFunction().hasMinSize(); 5603 } 5604 5605 bool AArch64InstrInfo::isCopyInstrImpl( 5606 const MachineInstr &MI, const MachineOperand *&Source, 5607 const MachineOperand *&Destination) const { 5608 5609 // AArch64::ORRWrs and AArch64::ORRXrs with WZR/XZR reg 5610 // and zero immediate operands used as an alias for mov instruction. 5611 if (MI.getOpcode() == AArch64::ORRWrs && 5612 MI.getOperand(1).getReg() == AArch64::WZR && 5613 MI.getOperand(3).getImm() == 0x0) { 5614 Destination = &MI.getOperand(0); 5615 Source = &MI.getOperand(2); 5616 return true; 5617 } 5618 5619 if (MI.getOpcode() == AArch64::ORRXrs && 5620 MI.getOperand(1).getReg() == AArch64::XZR && 5621 MI.getOperand(3).getImm() == 0x0) { 5622 Destination = &MI.getOperand(0); 5623 Source = &MI.getOperand(2); 5624 return true; 5625 } 5626 5627 return false; 5628 } 5629 5630 #define GET_INSTRINFO_HELPERS 5631 #include "AArch64GenInstrInfo.inc" 5632