1 //===- ARMLegalizerInfo.cpp --------------------------------------*- C++ -*-==//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 /// \file
10 /// This file implements the targeting of the Machinelegalizer class for ARM.
11 /// \todo This should be generated by TableGen.
12 //===----------------------------------------------------------------------===//
13 
14 #include "ARMLegalizerInfo.h"
15 #include "ARMCallLowering.h"
16 #include "ARMSubtarget.h"
17 #include "llvm/CodeGen/GlobalISel/LegalizerHelper.h"
18 #include "llvm/CodeGen/LowLevelType.h"
19 #include "llvm/CodeGen/MachineRegisterInfo.h"
20 #include "llvm/CodeGen/TargetOpcodes.h"
21 #include "llvm/CodeGen/ValueTypes.h"
22 #include "llvm/IR/DerivedTypes.h"
23 #include "llvm/IR/Type.h"
24 
25 using namespace llvm;
26 
27 /// FIXME: The following static functions are SizeChangeStrategy functions
28 /// that are meant to temporarily mimic the behaviour of the old legalization
29 /// based on doubling/halving non-legal types as closely as possible. This is
30 /// not entirly possible as only legalizing the types that are exactly a power
31 /// of 2 times the size of the legal types would require specifying all those
32 /// sizes explicitly.
33 /// In practice, not specifying those isn't a problem, and the below functions
34 /// should disappear quickly as we add support for legalizing non-power-of-2
35 /// sized types further.
36 static void
37 addAndInterleaveWithUnsupported(LegalizerInfo::SizeAndActionsVec &result,
38                                 const LegalizerInfo::SizeAndActionsVec &v) {
39   for (unsigned i = 0; i < v.size(); ++i) {
40     result.push_back(v[i]);
41     if (i + 1 < v[i].first && i + 1 < v.size() &&
42         v[i + 1].first != v[i].first + 1)
43       result.push_back({v[i].first + 1, LegalizerInfo::Unsupported});
44   }
45 }
46 
47 static LegalizerInfo::SizeAndActionsVec
48 widen_8_16(const LegalizerInfo::SizeAndActionsVec &v) {
49   assert(v.size() >= 1);
50   assert(v[0].first > 17);
51   LegalizerInfo::SizeAndActionsVec result = {
52       {1, LegalizerInfo::Unsupported},
53       {8, LegalizerInfo::WidenScalar},  {9, LegalizerInfo::Unsupported},
54       {16, LegalizerInfo::WidenScalar}, {17, LegalizerInfo::Unsupported}};
55   addAndInterleaveWithUnsupported(result, v);
56   auto Largest = result.back().first;
57   result.push_back({Largest + 1, LegalizerInfo::Unsupported});
58   return result;
59 }
60 
61 static LegalizerInfo::SizeAndActionsVec
62 widen_1_8_16(const LegalizerInfo::SizeAndActionsVec &v) {
63   assert(v.size() >= 1);
64   assert(v[0].first > 17);
65   LegalizerInfo::SizeAndActionsVec result = {
66       {1, LegalizerInfo::WidenScalar},  {2, LegalizerInfo::Unsupported},
67       {8, LegalizerInfo::WidenScalar},  {9, LegalizerInfo::Unsupported},
68       {16, LegalizerInfo::WidenScalar}, {17, LegalizerInfo::Unsupported}};
69   addAndInterleaveWithUnsupported(result, v);
70   auto Largest = result.back().first;
71   result.push_back({Largest + 1, LegalizerInfo::Unsupported});
72   return result;
73 }
74 
75 static bool AEABI(const ARMSubtarget &ST) {
76   return ST.isTargetAEABI() || ST.isTargetGNUAEABI() || ST.isTargetMuslAEABI();
77 }
78 
79 ARMLegalizerInfo::ARMLegalizerInfo(const ARMSubtarget &ST) {
80   using namespace TargetOpcode;
81 
82   const LLT p0 = LLT::pointer(0, 32);
83 
84   const LLT s1 = LLT::scalar(1);
85   const LLT s8 = LLT::scalar(8);
86   const LLT s16 = LLT::scalar(16);
87   const LLT s32 = LLT::scalar(32);
88   const LLT s64 = LLT::scalar(64);
89 
90   setAction({G_GLOBAL_VALUE, p0}, Legal);
91   setAction({G_FRAME_INDEX, p0}, Legal);
92 
93   for (unsigned Op : {G_LOAD, G_STORE}) {
94     for (auto Ty : {s1, s8, s16, s32, p0})
95       setAction({Op, Ty}, Legal);
96     setAction({Op, 1, p0}, Legal);
97   }
98 
99   for (unsigned Op : {G_ADD, G_SUB, G_MUL, G_AND, G_OR, G_XOR}) {
100     if (Op != G_ADD)
101       setLegalizeScalarToDifferentSizeStrategy(
102           Op, 0, widenToLargerTypesUnsupportedOtherwise);
103     setAction({Op, s32}, Legal);
104   }
105 
106   for (unsigned Op : {G_SDIV, G_UDIV}) {
107     setLegalizeScalarToDifferentSizeStrategy(Op, 0,
108         widenToLargerTypesUnsupportedOtherwise);
109     if (ST.hasDivideInARMMode())
110       setAction({Op, s32}, Legal);
111     else
112       setAction({Op, s32}, Libcall);
113   }
114 
115   for (unsigned Op : {G_SREM, G_UREM}) {
116     setLegalizeScalarToDifferentSizeStrategy(Op, 0, widen_8_16);
117     if (ST.hasDivideInARMMode())
118       setAction({Op, s32}, Lower);
119     else if (AEABI(ST))
120       setAction({Op, s32}, Custom);
121     else
122       setAction({Op, s32}, Libcall);
123   }
124 
125   for (unsigned Op : {G_SEXT, G_ZEXT, G_ANYEXT}) {
126     setAction({Op, s32}, Legal);
127   }
128 
129   for (unsigned Op : {G_ASHR, G_LSHR, G_SHL})
130     setAction({Op, s32}, Legal);
131 
132   setAction({G_GEP, p0}, Legal);
133   setAction({G_GEP, 1, s32}, Legal);
134 
135   setAction({G_SELECT, s32}, Legal);
136   setAction({G_SELECT, p0}, Legal);
137   setAction({G_SELECT, 1, s1}, Legal);
138 
139   setAction({G_BRCOND, s1}, Legal);
140 
141   setAction({G_CONSTANT, s32}, Legal);
142   setLegalizeScalarToDifferentSizeStrategy(G_CONSTANT, 0, widen_1_8_16);
143 
144   setAction({G_ICMP, s1}, Legal);
145   setLegalizeScalarToDifferentSizeStrategy(G_ICMP, 1,
146       widenToLargerTypesUnsupportedOtherwise);
147   for (auto Ty : {s32, p0})
148     setAction({G_ICMP, 1, Ty}, Legal);
149 
150   if (!ST.useSoftFloat() && ST.hasVFP2()) {
151     for (unsigned BinOp : {G_FADD, G_FSUB, G_FMUL, G_FDIV})
152       for (auto Ty : {s32, s64})
153         setAction({BinOp, Ty}, Legal);
154 
155     setAction({G_LOAD, s64}, Legal);
156     setAction({G_STORE, s64}, Legal);
157 
158     setAction({G_FCMP, s1}, Legal);
159     setAction({G_FCMP, 1, s32}, Legal);
160     setAction({G_FCMP, 1, s64}, Legal);
161 
162     setAction({G_MERGE_VALUES, s64}, Legal);
163     setAction({G_MERGE_VALUES, 1, s32}, Legal);
164     setAction({G_UNMERGE_VALUES, s32}, Legal);
165     setAction({G_UNMERGE_VALUES, 1, s64}, Legal);
166   } else {
167     for (unsigned BinOp : {G_FADD, G_FSUB, G_FMUL, G_FDIV})
168       for (auto Ty : {s32, s64})
169         setAction({BinOp, Ty}, Libcall);
170 
171     setAction({G_FCMP, s1}, Legal);
172     setAction({G_FCMP, 1, s32}, Custom);
173     setAction({G_FCMP, 1, s64}, Custom);
174 
175     if (AEABI(ST))
176       setFCmpLibcallsAEABI();
177     else
178       setFCmpLibcallsGNU();
179   }
180 
181   for (unsigned Op : {G_FREM, G_FPOW})
182     for (auto Ty : {s32, s64})
183       setAction({Op, Ty}, Libcall);
184 
185   computeTables();
186 }
187 
188 void ARMLegalizerInfo::setFCmpLibcallsAEABI() {
189   // FCMP_TRUE and FCMP_FALSE don't need libcalls, they should be
190   // default-initialized.
191   FCmp32Libcalls.resize(CmpInst::LAST_FCMP_PREDICATE + 1);
192   FCmp32Libcalls[CmpInst::FCMP_OEQ] = {
193       {RTLIB::OEQ_F32, CmpInst::BAD_ICMP_PREDICATE}};
194   FCmp32Libcalls[CmpInst::FCMP_OGE] = {
195       {RTLIB::OGE_F32, CmpInst::BAD_ICMP_PREDICATE}};
196   FCmp32Libcalls[CmpInst::FCMP_OGT] = {
197       {RTLIB::OGT_F32, CmpInst::BAD_ICMP_PREDICATE}};
198   FCmp32Libcalls[CmpInst::FCMP_OLE] = {
199       {RTLIB::OLE_F32, CmpInst::BAD_ICMP_PREDICATE}};
200   FCmp32Libcalls[CmpInst::FCMP_OLT] = {
201       {RTLIB::OLT_F32, CmpInst::BAD_ICMP_PREDICATE}};
202   FCmp32Libcalls[CmpInst::FCMP_ORD] = {{RTLIB::O_F32, CmpInst::ICMP_EQ}};
203   FCmp32Libcalls[CmpInst::FCMP_UGE] = {{RTLIB::OLT_F32, CmpInst::ICMP_EQ}};
204   FCmp32Libcalls[CmpInst::FCMP_UGT] = {{RTLIB::OLE_F32, CmpInst::ICMP_EQ}};
205   FCmp32Libcalls[CmpInst::FCMP_ULE] = {{RTLIB::OGT_F32, CmpInst::ICMP_EQ}};
206   FCmp32Libcalls[CmpInst::FCMP_ULT] = {{RTLIB::OGE_F32, CmpInst::ICMP_EQ}};
207   FCmp32Libcalls[CmpInst::FCMP_UNE] = {{RTLIB::UNE_F32, CmpInst::ICMP_EQ}};
208   FCmp32Libcalls[CmpInst::FCMP_UNO] = {
209       {RTLIB::UO_F32, CmpInst::BAD_ICMP_PREDICATE}};
210   FCmp32Libcalls[CmpInst::FCMP_ONE] = {
211       {RTLIB::OGT_F32, CmpInst::BAD_ICMP_PREDICATE},
212       {RTLIB::OLT_F32, CmpInst::BAD_ICMP_PREDICATE}};
213   FCmp32Libcalls[CmpInst::FCMP_UEQ] = {
214       {RTLIB::OEQ_F32, CmpInst::BAD_ICMP_PREDICATE},
215       {RTLIB::UO_F32, CmpInst::BAD_ICMP_PREDICATE}};
216 
217   FCmp64Libcalls.resize(CmpInst::LAST_FCMP_PREDICATE + 1);
218   FCmp64Libcalls[CmpInst::FCMP_OEQ] = {
219       {RTLIB::OEQ_F64, CmpInst::BAD_ICMP_PREDICATE}};
220   FCmp64Libcalls[CmpInst::FCMP_OGE] = {
221       {RTLIB::OGE_F64, CmpInst::BAD_ICMP_PREDICATE}};
222   FCmp64Libcalls[CmpInst::FCMP_OGT] = {
223       {RTLIB::OGT_F64, CmpInst::BAD_ICMP_PREDICATE}};
224   FCmp64Libcalls[CmpInst::FCMP_OLE] = {
225       {RTLIB::OLE_F64, CmpInst::BAD_ICMP_PREDICATE}};
226   FCmp64Libcalls[CmpInst::FCMP_OLT] = {
227       {RTLIB::OLT_F64, CmpInst::BAD_ICMP_PREDICATE}};
228   FCmp64Libcalls[CmpInst::FCMP_ORD] = {{RTLIB::O_F64, CmpInst::ICMP_EQ}};
229   FCmp64Libcalls[CmpInst::FCMP_UGE] = {{RTLIB::OLT_F64, CmpInst::ICMP_EQ}};
230   FCmp64Libcalls[CmpInst::FCMP_UGT] = {{RTLIB::OLE_F64, CmpInst::ICMP_EQ}};
231   FCmp64Libcalls[CmpInst::FCMP_ULE] = {{RTLIB::OGT_F64, CmpInst::ICMP_EQ}};
232   FCmp64Libcalls[CmpInst::FCMP_ULT] = {{RTLIB::OGE_F64, CmpInst::ICMP_EQ}};
233   FCmp64Libcalls[CmpInst::FCMP_UNE] = {{RTLIB::UNE_F64, CmpInst::ICMP_EQ}};
234   FCmp64Libcalls[CmpInst::FCMP_UNO] = {
235       {RTLIB::UO_F64, CmpInst::BAD_ICMP_PREDICATE}};
236   FCmp64Libcalls[CmpInst::FCMP_ONE] = {
237       {RTLIB::OGT_F64, CmpInst::BAD_ICMP_PREDICATE},
238       {RTLIB::OLT_F64, CmpInst::BAD_ICMP_PREDICATE}};
239   FCmp64Libcalls[CmpInst::FCMP_UEQ] = {
240       {RTLIB::OEQ_F64, CmpInst::BAD_ICMP_PREDICATE},
241       {RTLIB::UO_F64, CmpInst::BAD_ICMP_PREDICATE}};
242 }
243 
244 void ARMLegalizerInfo::setFCmpLibcallsGNU() {
245   // FCMP_TRUE and FCMP_FALSE don't need libcalls, they should be
246   // default-initialized.
247   FCmp32Libcalls.resize(CmpInst::LAST_FCMP_PREDICATE + 1);
248   FCmp32Libcalls[CmpInst::FCMP_OEQ] = {{RTLIB::OEQ_F32, CmpInst::ICMP_EQ}};
249   FCmp32Libcalls[CmpInst::FCMP_OGE] = {{RTLIB::OGE_F32, CmpInst::ICMP_SGE}};
250   FCmp32Libcalls[CmpInst::FCMP_OGT] = {{RTLIB::OGT_F32, CmpInst::ICMP_SGT}};
251   FCmp32Libcalls[CmpInst::FCMP_OLE] = {{RTLIB::OLE_F32, CmpInst::ICMP_SLE}};
252   FCmp32Libcalls[CmpInst::FCMP_OLT] = {{RTLIB::OLT_F32, CmpInst::ICMP_SLT}};
253   FCmp32Libcalls[CmpInst::FCMP_ORD] = {{RTLIB::O_F32, CmpInst::ICMP_EQ}};
254   FCmp32Libcalls[CmpInst::FCMP_UGE] = {{RTLIB::OLT_F32, CmpInst::ICMP_SGE}};
255   FCmp32Libcalls[CmpInst::FCMP_UGT] = {{RTLIB::OLE_F32, CmpInst::ICMP_SGT}};
256   FCmp32Libcalls[CmpInst::FCMP_ULE] = {{RTLIB::OGT_F32, CmpInst::ICMP_SLE}};
257   FCmp32Libcalls[CmpInst::FCMP_ULT] = {{RTLIB::OGE_F32, CmpInst::ICMP_SLT}};
258   FCmp32Libcalls[CmpInst::FCMP_UNE] = {{RTLIB::UNE_F32, CmpInst::ICMP_NE}};
259   FCmp32Libcalls[CmpInst::FCMP_UNO] = {{RTLIB::UO_F32, CmpInst::ICMP_NE}};
260   FCmp32Libcalls[CmpInst::FCMP_ONE] = {{RTLIB::OGT_F32, CmpInst::ICMP_SGT},
261                                        {RTLIB::OLT_F32, CmpInst::ICMP_SLT}};
262   FCmp32Libcalls[CmpInst::FCMP_UEQ] = {{RTLIB::OEQ_F32, CmpInst::ICMP_EQ},
263                                        {RTLIB::UO_F32, CmpInst::ICMP_NE}};
264 
265   FCmp64Libcalls.resize(CmpInst::LAST_FCMP_PREDICATE + 1);
266   FCmp64Libcalls[CmpInst::FCMP_OEQ] = {{RTLIB::OEQ_F64, CmpInst::ICMP_EQ}};
267   FCmp64Libcalls[CmpInst::FCMP_OGE] = {{RTLIB::OGE_F64, CmpInst::ICMP_SGE}};
268   FCmp64Libcalls[CmpInst::FCMP_OGT] = {{RTLIB::OGT_F64, CmpInst::ICMP_SGT}};
269   FCmp64Libcalls[CmpInst::FCMP_OLE] = {{RTLIB::OLE_F64, CmpInst::ICMP_SLE}};
270   FCmp64Libcalls[CmpInst::FCMP_OLT] = {{RTLIB::OLT_F64, CmpInst::ICMP_SLT}};
271   FCmp64Libcalls[CmpInst::FCMP_ORD] = {{RTLIB::O_F64, CmpInst::ICMP_EQ}};
272   FCmp64Libcalls[CmpInst::FCMP_UGE] = {{RTLIB::OLT_F64, CmpInst::ICMP_SGE}};
273   FCmp64Libcalls[CmpInst::FCMP_UGT] = {{RTLIB::OLE_F64, CmpInst::ICMP_SGT}};
274   FCmp64Libcalls[CmpInst::FCMP_ULE] = {{RTLIB::OGT_F64, CmpInst::ICMP_SLE}};
275   FCmp64Libcalls[CmpInst::FCMP_ULT] = {{RTLIB::OGE_F64, CmpInst::ICMP_SLT}};
276   FCmp64Libcalls[CmpInst::FCMP_UNE] = {{RTLIB::UNE_F64, CmpInst::ICMP_NE}};
277   FCmp64Libcalls[CmpInst::FCMP_UNO] = {{RTLIB::UO_F64, CmpInst::ICMP_NE}};
278   FCmp64Libcalls[CmpInst::FCMP_ONE] = {{RTLIB::OGT_F64, CmpInst::ICMP_SGT},
279                                        {RTLIB::OLT_F64, CmpInst::ICMP_SLT}};
280   FCmp64Libcalls[CmpInst::FCMP_UEQ] = {{RTLIB::OEQ_F64, CmpInst::ICMP_EQ},
281                                        {RTLIB::UO_F64, CmpInst::ICMP_NE}};
282 }
283 
284 ARMLegalizerInfo::FCmpLibcallsList
285 ARMLegalizerInfo::getFCmpLibcalls(CmpInst::Predicate Predicate,
286                                   unsigned Size) const {
287   assert(CmpInst::isFPPredicate(Predicate) && "Unsupported FCmp predicate");
288   if (Size == 32)
289     return FCmp32Libcalls[Predicate];
290   if (Size == 64)
291     return FCmp64Libcalls[Predicate];
292   llvm_unreachable("Unsupported size for FCmp predicate");
293 }
294 
295 bool ARMLegalizerInfo::legalizeCustom(MachineInstr &MI,
296                                       MachineRegisterInfo &MRI,
297                                       MachineIRBuilder &MIRBuilder) const {
298   using namespace TargetOpcode;
299 
300   MIRBuilder.setInstr(MI);
301 
302   switch (MI.getOpcode()) {
303   default:
304     return false;
305   case G_SREM:
306   case G_UREM: {
307     unsigned OriginalResult = MI.getOperand(0).getReg();
308     auto Size = MRI.getType(OriginalResult).getSizeInBits();
309     if (Size != 32)
310       return false;
311 
312     auto Libcall =
313         MI.getOpcode() == G_SREM ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32;
314 
315     // Our divmod libcalls return a struct containing the quotient and the
316     // remainder. We need to create a virtual register for it.
317     auto &Ctx = MIRBuilder.getMF().getFunction().getContext();
318     Type *ArgTy = Type::getInt32Ty(Ctx);
319     StructType *RetTy = StructType::get(Ctx, {ArgTy, ArgTy}, /* Packed */ true);
320     auto RetVal = MRI.createGenericVirtualRegister(
321         getLLTForType(*RetTy, MIRBuilder.getMF().getDataLayout()));
322 
323     auto Status = createLibcall(MIRBuilder, Libcall, {RetVal, RetTy},
324                                 {{MI.getOperand(1).getReg(), ArgTy},
325                                  {MI.getOperand(2).getReg(), ArgTy}});
326     if (Status != LegalizerHelper::Legalized)
327       return false;
328 
329     // The remainder is the second result of divmod. Split the return value into
330     // a new, unused register for the quotient and the destination of the
331     // original instruction for the remainder.
332     MIRBuilder.buildUnmerge(
333         {MRI.createGenericVirtualRegister(LLT::scalar(32)), OriginalResult},
334         RetVal);
335     break;
336   }
337   case G_FCMP: {
338     assert(MRI.getType(MI.getOperand(2).getReg()) ==
339                MRI.getType(MI.getOperand(3).getReg()) &&
340            "Mismatched operands for G_FCMP");
341     auto OpSize = MRI.getType(MI.getOperand(2).getReg()).getSizeInBits();
342 
343     auto OriginalResult = MI.getOperand(0).getReg();
344     auto Predicate =
345         static_cast<CmpInst::Predicate>(MI.getOperand(1).getPredicate());
346     auto Libcalls = getFCmpLibcalls(Predicate, OpSize);
347 
348     if (Libcalls.empty()) {
349       assert((Predicate == CmpInst::FCMP_TRUE ||
350               Predicate == CmpInst::FCMP_FALSE) &&
351              "Predicate needs libcalls, but none specified");
352       MIRBuilder.buildConstant(OriginalResult,
353                                Predicate == CmpInst::FCMP_TRUE ? 1 : 0);
354       MI.eraseFromParent();
355       return true;
356     }
357 
358     auto &Ctx = MIRBuilder.getMF().getFunction().getContext();
359     assert((OpSize == 32 || OpSize == 64) && "Unsupported operand size");
360     auto *ArgTy = OpSize == 32 ? Type::getFloatTy(Ctx) : Type::getDoubleTy(Ctx);
361     auto *RetTy = Type::getInt32Ty(Ctx);
362 
363     SmallVector<unsigned, 2> Results;
364     for (auto Libcall : Libcalls) {
365       auto LibcallResult = MRI.createGenericVirtualRegister(LLT::scalar(32));
366       auto Status =
367           createLibcall(MIRBuilder, Libcall.LibcallID, {LibcallResult, RetTy},
368                         {{MI.getOperand(2).getReg(), ArgTy},
369                          {MI.getOperand(3).getReg(), ArgTy}});
370 
371       if (Status != LegalizerHelper::Legalized)
372         return false;
373 
374       auto ProcessedResult =
375           Libcalls.size() == 1
376               ? OriginalResult
377               : MRI.createGenericVirtualRegister(MRI.getType(OriginalResult));
378 
379       // We have a result, but we need to transform it into a proper 1-bit 0 or
380       // 1, taking into account the different peculiarities of the values
381       // returned by the comparison functions.
382       CmpInst::Predicate ResultPred = Libcall.Predicate;
383       if (ResultPred == CmpInst::BAD_ICMP_PREDICATE) {
384         // We have a nice 0 or 1, and we just need to truncate it back to 1 bit
385         // to keep the types consistent.
386         MIRBuilder.buildTrunc(ProcessedResult, LibcallResult);
387       } else {
388         // We need to compare against 0.
389         assert(CmpInst::isIntPredicate(ResultPred) && "Unsupported predicate");
390         auto Zero = MRI.createGenericVirtualRegister(LLT::scalar(32));
391         MIRBuilder.buildConstant(Zero, 0);
392         MIRBuilder.buildICmp(ResultPred, ProcessedResult, LibcallResult, Zero);
393       }
394       Results.push_back(ProcessedResult);
395     }
396 
397     if (Results.size() != 1) {
398       assert(Results.size() == 2 && "Unexpected number of results");
399       MIRBuilder.buildOr(OriginalResult, Results[0], Results[1]);
400     }
401     break;
402   }
403   }
404 
405   MI.eraseFromParent();
406   return true;
407 }
408