1 //===-- SIISelLowering.cpp - SI DAG Lowering Implementation ---------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 //
10 /// \file
11 /// \brief Custom DAG lowering for SI
12 //
13 //===----------------------------------------------------------------------===//
14 
15 #ifdef _MSC_VER
16 // Provide M_PI.
17 #define _USE_MATH_DEFINES
18 #endif
19 
20 #include "SIISelLowering.h"
21 #include "AMDGPU.h"
22 #include "AMDGPUIntrinsicInfo.h"
23 #include "AMDGPUSubtarget.h"
24 #include "AMDGPUTargetMachine.h"
25 #include "SIDefines.h"
26 #include "SIInstrInfo.h"
27 #include "SIMachineFunctionInfo.h"
28 #include "SIRegisterInfo.h"
29 #include "Utils/AMDGPUBaseInfo.h"
30 #include "llvm/ADT/APFloat.h"
31 #include "llvm/ADT/APInt.h"
32 #include "llvm/ADT/ArrayRef.h"
33 #include "llvm/ADT/BitVector.h"
34 #include "llvm/ADT/SmallVector.h"
35 #include "llvm/ADT/StringRef.h"
36 #include "llvm/ADT/StringSwitch.h"
37 #include "llvm/ADT/Twine.h"
38 #include "llvm/CodeGen/Analysis.h"
39 #include "llvm/CodeGen/CallingConvLower.h"
40 #include "llvm/CodeGen/DAGCombine.h"
41 #include "llvm/CodeGen/ISDOpcodes.h"
42 #include "llvm/CodeGen/MachineBasicBlock.h"
43 #include "llvm/CodeGen/MachineFrameInfo.h"
44 #include "llvm/CodeGen/MachineFunction.h"
45 #include "llvm/CodeGen/MachineInstr.h"
46 #include "llvm/CodeGen/MachineInstrBuilder.h"
47 #include "llvm/CodeGen/MachineMemOperand.h"
48 #include "llvm/CodeGen/MachineOperand.h"
49 #include "llvm/CodeGen/MachineRegisterInfo.h"
50 #include "llvm/CodeGen/MachineValueType.h"
51 #include "llvm/CodeGen/SelectionDAG.h"
52 #include "llvm/CodeGen/SelectionDAGNodes.h"
53 #include "llvm/CodeGen/ValueTypes.h"
54 #include "llvm/IR/Constants.h"
55 #include "llvm/IR/DataLayout.h"
56 #include "llvm/IR/DebugLoc.h"
57 #include "llvm/IR/DerivedTypes.h"
58 #include "llvm/IR/DiagnosticInfo.h"
59 #include "llvm/IR/Function.h"
60 #include "llvm/IR/GlobalValue.h"
61 #include "llvm/IR/InstrTypes.h"
62 #include "llvm/IR/Instruction.h"
63 #include "llvm/IR/Instructions.h"
64 #include "llvm/IR/IntrinsicInst.h"
65 #include "llvm/IR/Type.h"
66 #include "llvm/Support/Casting.h"
67 #include "llvm/Support/CodeGen.h"
68 #include "llvm/Support/CommandLine.h"
69 #include "llvm/Support/Compiler.h"
70 #include "llvm/Support/ErrorHandling.h"
71 #include "llvm/Support/KnownBits.h"
72 #include "llvm/Support/MathExtras.h"
73 #include "llvm/Target/TargetCallingConv.h"
74 #include "llvm/Target/TargetOptions.h"
75 #include "llvm/Target/TargetRegisterInfo.h"
76 #include <cassert>
77 #include <cmath>
78 #include <cstdint>
79 #include <iterator>
80 #include <tuple>
81 #include <utility>
82 #include <vector>
83 
84 using namespace llvm;
85 
86 static cl::opt<bool> EnableVGPRIndexMode(
87   "amdgpu-vgpr-index-mode",
88   cl::desc("Use GPR indexing mode instead of movrel for vector indexing"),
89   cl::init(false));
90 
91 static unsigned findFirstFreeSGPR(CCState &CCInfo) {
92   unsigned NumSGPRs = AMDGPU::SGPR_32RegClass.getNumRegs();
93   for (unsigned Reg = 0; Reg < NumSGPRs; ++Reg) {
94     if (!CCInfo.isAllocated(AMDGPU::SGPR0 + Reg)) {
95       return AMDGPU::SGPR0 + Reg;
96     }
97   }
98   llvm_unreachable("Cannot allocate sgpr");
99 }
100 
101 SITargetLowering::SITargetLowering(const TargetMachine &TM,
102                                    const SISubtarget &STI)
103     : AMDGPUTargetLowering(TM, STI) {
104   addRegisterClass(MVT::i1, &AMDGPU::VReg_1RegClass);
105   addRegisterClass(MVT::i64, &AMDGPU::SReg_64RegClass);
106 
107   addRegisterClass(MVT::i32, &AMDGPU::SReg_32_XM0RegClass);
108   addRegisterClass(MVT::f32, &AMDGPU::VGPR_32RegClass);
109 
110   addRegisterClass(MVT::f64, &AMDGPU::VReg_64RegClass);
111   addRegisterClass(MVT::v2i32, &AMDGPU::SReg_64RegClass);
112   addRegisterClass(MVT::v2f32, &AMDGPU::VReg_64RegClass);
113 
114   addRegisterClass(MVT::v2i64, &AMDGPU::SReg_128RegClass);
115   addRegisterClass(MVT::v2f64, &AMDGPU::SReg_128RegClass);
116 
117   addRegisterClass(MVT::v4i32, &AMDGPU::SReg_128RegClass);
118   addRegisterClass(MVT::v4f32, &AMDGPU::VReg_128RegClass);
119 
120   addRegisterClass(MVT::v8i32, &AMDGPU::SReg_256RegClass);
121   addRegisterClass(MVT::v8f32, &AMDGPU::VReg_256RegClass);
122 
123   addRegisterClass(MVT::v16i32, &AMDGPU::SReg_512RegClass);
124   addRegisterClass(MVT::v16f32, &AMDGPU::VReg_512RegClass);
125 
126   if (Subtarget->has16BitInsts()) {
127     addRegisterClass(MVT::i16, &AMDGPU::SReg_32_XM0RegClass);
128     addRegisterClass(MVT::f16, &AMDGPU::SReg_32_XM0RegClass);
129   }
130 
131   if (Subtarget->hasVOP3PInsts()) {
132     addRegisterClass(MVT::v2i16, &AMDGPU::SReg_32_XM0RegClass);
133     addRegisterClass(MVT::v2f16, &AMDGPU::SReg_32_XM0RegClass);
134   }
135 
136   computeRegisterProperties(STI.getRegisterInfo());
137 
138   // We need to custom lower vector stores from local memory
139   setOperationAction(ISD::LOAD, MVT::v2i32, Custom);
140   setOperationAction(ISD::LOAD, MVT::v4i32, Custom);
141   setOperationAction(ISD::LOAD, MVT::v8i32, Custom);
142   setOperationAction(ISD::LOAD, MVT::v16i32, Custom);
143   setOperationAction(ISD::LOAD, MVT::i1, Custom);
144 
145   setOperationAction(ISD::STORE, MVT::v2i32, Custom);
146   setOperationAction(ISD::STORE, MVT::v4i32, Custom);
147   setOperationAction(ISD::STORE, MVT::v8i32, Custom);
148   setOperationAction(ISD::STORE, MVT::v16i32, Custom);
149   setOperationAction(ISD::STORE, MVT::i1, Custom);
150 
151   setTruncStoreAction(MVT::v2i32, MVT::v2i16, Expand);
152   setTruncStoreAction(MVT::v4i32, MVT::v4i16, Expand);
153   setTruncStoreAction(MVT::v8i32, MVT::v8i16, Expand);
154   setTruncStoreAction(MVT::v16i32, MVT::v16i16, Expand);
155   setTruncStoreAction(MVT::v32i32, MVT::v32i16, Expand);
156   setTruncStoreAction(MVT::v2i32, MVT::v2i8, Expand);
157   setTruncStoreAction(MVT::v4i32, MVT::v4i8, Expand);
158   setTruncStoreAction(MVT::v8i32, MVT::v8i8, Expand);
159   setTruncStoreAction(MVT::v16i32, MVT::v16i8, Expand);
160   setTruncStoreAction(MVT::v32i32, MVT::v32i8, Expand);
161 
162   setOperationAction(ISD::GlobalAddress, MVT::i32, Custom);
163   setOperationAction(ISD::GlobalAddress, MVT::i64, Custom);
164   setOperationAction(ISD::ConstantPool, MVT::v2i64, Expand);
165 
166   setOperationAction(ISD::SELECT, MVT::i1, Promote);
167   setOperationAction(ISD::SELECT, MVT::i64, Custom);
168   setOperationAction(ISD::SELECT, MVT::f64, Promote);
169   AddPromotedToType(ISD::SELECT, MVT::f64, MVT::i64);
170 
171   setOperationAction(ISD::SELECT_CC, MVT::f32, Expand);
172   setOperationAction(ISD::SELECT_CC, MVT::i32, Expand);
173   setOperationAction(ISD::SELECT_CC, MVT::i64, Expand);
174   setOperationAction(ISD::SELECT_CC, MVT::f64, Expand);
175   setOperationAction(ISD::SELECT_CC, MVT::i1, Expand);
176 
177   setOperationAction(ISD::SETCC, MVT::i1, Promote);
178   setOperationAction(ISD::SETCC, MVT::v2i1, Expand);
179   setOperationAction(ISD::SETCC, MVT::v4i1, Expand);
180   AddPromotedToType(ISD::SETCC, MVT::i1, MVT::i32);
181 
182   setOperationAction(ISD::TRUNCATE, MVT::v2i32, Expand);
183   setOperationAction(ISD::FP_ROUND, MVT::v2f32, Expand);
184 
185   setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::v2i1, Custom);
186   setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::v4i1, Custom);
187   setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::v2i8, Custom);
188   setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::v4i8, Custom);
189   setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::v2i16, Custom);
190   setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::v4i16, Custom);
191   setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::Other, Custom);
192 
193   setOperationAction(ISD::INTRINSIC_WO_CHAIN, MVT::Other, Custom);
194   setOperationAction(ISD::INTRINSIC_WO_CHAIN, MVT::f32, Custom);
195   setOperationAction(ISD::INTRINSIC_WO_CHAIN, MVT::v4f32, Custom);
196   setOperationAction(ISD::INTRINSIC_WO_CHAIN, MVT::v2f16, Custom);
197 
198   setOperationAction(ISD::INTRINSIC_W_CHAIN, MVT::Other, Custom);
199 
200   setOperationAction(ISD::INTRINSIC_VOID, MVT::Other, Custom);
201   setOperationAction(ISD::INTRINSIC_VOID, MVT::v2i16, Custom);
202   setOperationAction(ISD::INTRINSIC_VOID, MVT::v2f16, Custom);
203 
204   setOperationAction(ISD::BRCOND, MVT::Other, Custom);
205   setOperationAction(ISD::BR_CC, MVT::i1, Expand);
206   setOperationAction(ISD::BR_CC, MVT::i32, Expand);
207   setOperationAction(ISD::BR_CC, MVT::i64, Expand);
208   setOperationAction(ISD::BR_CC, MVT::f32, Expand);
209   setOperationAction(ISD::BR_CC, MVT::f64, Expand);
210 
211   setOperationAction(ISD::UADDO, MVT::i32, Legal);
212   setOperationAction(ISD::USUBO, MVT::i32, Legal);
213 
214   setOperationAction(ISD::ADDCARRY, MVT::i32, Legal);
215   setOperationAction(ISD::SUBCARRY, MVT::i32, Legal);
216 
217   // We only support LOAD/STORE and vector manipulation ops for vectors
218   // with > 4 elements.
219   for (MVT VT : {MVT::v8i32, MVT::v8f32, MVT::v16i32, MVT::v16f32,
220         MVT::v2i64, MVT::v2f64}) {
221     for (unsigned Op = 0; Op < ISD::BUILTIN_OP_END; ++Op) {
222       switch (Op) {
223       case ISD::LOAD:
224       case ISD::STORE:
225       case ISD::BUILD_VECTOR:
226       case ISD::BITCAST:
227       case ISD::EXTRACT_VECTOR_ELT:
228       case ISD::INSERT_VECTOR_ELT:
229       case ISD::INSERT_SUBVECTOR:
230       case ISD::EXTRACT_SUBVECTOR:
231       case ISD::SCALAR_TO_VECTOR:
232         break;
233       case ISD::CONCAT_VECTORS:
234         setOperationAction(Op, VT, Custom);
235         break;
236       default:
237         setOperationAction(Op, VT, Expand);
238         break;
239       }
240     }
241   }
242 
243   // TODO: For dynamic 64-bit vector inserts/extracts, should emit a pseudo that
244   // is expanded to avoid having two separate loops in case the index is a VGPR.
245 
246   // Most operations are naturally 32-bit vector operations. We only support
247   // load and store of i64 vectors, so promote v2i64 vector operations to v4i32.
248   for (MVT Vec64 : { MVT::v2i64, MVT::v2f64 }) {
249     setOperationAction(ISD::BUILD_VECTOR, Vec64, Promote);
250     AddPromotedToType(ISD::BUILD_VECTOR, Vec64, MVT::v4i32);
251 
252     setOperationAction(ISD::EXTRACT_VECTOR_ELT, Vec64, Promote);
253     AddPromotedToType(ISD::EXTRACT_VECTOR_ELT, Vec64, MVT::v4i32);
254 
255     setOperationAction(ISD::INSERT_VECTOR_ELT, Vec64, Promote);
256     AddPromotedToType(ISD::INSERT_VECTOR_ELT, Vec64, MVT::v4i32);
257 
258     setOperationAction(ISD::SCALAR_TO_VECTOR, Vec64, Promote);
259     AddPromotedToType(ISD::SCALAR_TO_VECTOR, Vec64, MVT::v4i32);
260   }
261 
262   setOperationAction(ISD::VECTOR_SHUFFLE, MVT::v8i32, Expand);
263   setOperationAction(ISD::VECTOR_SHUFFLE, MVT::v8f32, Expand);
264   setOperationAction(ISD::VECTOR_SHUFFLE, MVT::v16i32, Expand);
265   setOperationAction(ISD::VECTOR_SHUFFLE, MVT::v16f32, Expand);
266 
267   // Avoid stack access for these.
268   // TODO: Generalize to more vector types.
269   setOperationAction(ISD::INSERT_VECTOR_ELT, MVT::v2i16, Custom);
270   setOperationAction(ISD::INSERT_VECTOR_ELT, MVT::v2f16, Custom);
271   setOperationAction(ISD::EXTRACT_VECTOR_ELT, MVT::v2i16, Custom);
272   setOperationAction(ISD::EXTRACT_VECTOR_ELT, MVT::v2f16, Custom);
273 
274   // BUFFER/FLAT_ATOMIC_CMP_SWAP on GCN GPUs needs input marshalling,
275   // and output demarshalling
276   setOperationAction(ISD::ATOMIC_CMP_SWAP, MVT::i32, Custom);
277   setOperationAction(ISD::ATOMIC_CMP_SWAP, MVT::i64, Custom);
278 
279   // We can't return success/failure, only the old value,
280   // let LLVM add the comparison
281   setOperationAction(ISD::ATOMIC_CMP_SWAP_WITH_SUCCESS, MVT::i32, Expand);
282   setOperationAction(ISD::ATOMIC_CMP_SWAP_WITH_SUCCESS, MVT::i64, Expand);
283 
284   if (getSubtarget()->hasFlatAddressSpace()) {
285     setOperationAction(ISD::ADDRSPACECAST, MVT::i32, Custom);
286     setOperationAction(ISD::ADDRSPACECAST, MVT::i64, Custom);
287   }
288 
289   setOperationAction(ISD::BSWAP, MVT::i32, Legal);
290   setOperationAction(ISD::BITREVERSE, MVT::i32, Legal);
291 
292   // On SI this is s_memtime and s_memrealtime on VI.
293   setOperationAction(ISD::READCYCLECOUNTER, MVT::i64, Legal);
294   setOperationAction(ISD::TRAP, MVT::Other, Custom);
295   setOperationAction(ISD::DEBUGTRAP, MVT::Other, Custom);
296 
297   setOperationAction(ISD::FMINNUM, MVT::f64, Legal);
298   setOperationAction(ISD::FMAXNUM, MVT::f64, Legal);
299 
300   if (Subtarget->getGeneration() >= SISubtarget::SEA_ISLANDS) {
301     setOperationAction(ISD::FTRUNC, MVT::f64, Legal);
302     setOperationAction(ISD::FCEIL, MVT::f64, Legal);
303     setOperationAction(ISD::FRINT, MVT::f64, Legal);
304   }
305 
306   setOperationAction(ISD::FFLOOR, MVT::f64, Legal);
307 
308   setOperationAction(ISD::FSIN, MVT::f32, Custom);
309   setOperationAction(ISD::FCOS, MVT::f32, Custom);
310   setOperationAction(ISD::FDIV, MVT::f32, Custom);
311   setOperationAction(ISD::FDIV, MVT::f64, Custom);
312 
313   if (Subtarget->has16BitInsts()) {
314     setOperationAction(ISD::Constant, MVT::i16, Legal);
315 
316     setOperationAction(ISD::SMIN, MVT::i16, Legal);
317     setOperationAction(ISD::SMAX, MVT::i16, Legal);
318 
319     setOperationAction(ISD::UMIN, MVT::i16, Legal);
320     setOperationAction(ISD::UMAX, MVT::i16, Legal);
321 
322     setOperationAction(ISD::SIGN_EXTEND, MVT::i16, Promote);
323     AddPromotedToType(ISD::SIGN_EXTEND, MVT::i16, MVT::i32);
324 
325     setOperationAction(ISD::ROTR, MVT::i16, Promote);
326     setOperationAction(ISD::ROTL, MVT::i16, Promote);
327 
328     setOperationAction(ISD::SDIV, MVT::i16, Promote);
329     setOperationAction(ISD::UDIV, MVT::i16, Promote);
330     setOperationAction(ISD::SREM, MVT::i16, Promote);
331     setOperationAction(ISD::UREM, MVT::i16, Promote);
332 
333     setOperationAction(ISD::BSWAP, MVT::i16, Promote);
334     setOperationAction(ISD::BITREVERSE, MVT::i16, Promote);
335 
336     setOperationAction(ISD::CTTZ, MVT::i16, Promote);
337     setOperationAction(ISD::CTTZ_ZERO_UNDEF, MVT::i16, Promote);
338     setOperationAction(ISD::CTLZ, MVT::i16, Promote);
339     setOperationAction(ISD::CTLZ_ZERO_UNDEF, MVT::i16, Promote);
340 
341     setOperationAction(ISD::SELECT_CC, MVT::i16, Expand);
342 
343     setOperationAction(ISD::BR_CC, MVT::i16, Expand);
344 
345     setOperationAction(ISD::LOAD, MVT::i16, Custom);
346 
347     setTruncStoreAction(MVT::i64, MVT::i16, Expand);
348 
349     setOperationAction(ISD::FP16_TO_FP, MVT::i16, Promote);
350     AddPromotedToType(ISD::FP16_TO_FP, MVT::i16, MVT::i32);
351     setOperationAction(ISD::FP_TO_FP16, MVT::i16, Promote);
352     AddPromotedToType(ISD::FP_TO_FP16, MVT::i16, MVT::i32);
353 
354     setOperationAction(ISD::FP_TO_SINT, MVT::i16, Promote);
355     setOperationAction(ISD::FP_TO_UINT, MVT::i16, Promote);
356     setOperationAction(ISD::SINT_TO_FP, MVT::i16, Promote);
357     setOperationAction(ISD::UINT_TO_FP, MVT::i16, Promote);
358 
359     // F16 - Constant Actions.
360     setOperationAction(ISD::ConstantFP, MVT::f16, Legal);
361 
362     // F16 - Load/Store Actions.
363     setOperationAction(ISD::LOAD, MVT::f16, Promote);
364     AddPromotedToType(ISD::LOAD, MVT::f16, MVT::i16);
365     setOperationAction(ISD::STORE, MVT::f16, Promote);
366     AddPromotedToType(ISD::STORE, MVT::f16, MVT::i16);
367 
368     // F16 - VOP1 Actions.
369     setOperationAction(ISD::FP_ROUND, MVT::f16, Custom);
370     setOperationAction(ISD::FCOS, MVT::f16, Promote);
371     setOperationAction(ISD::FSIN, MVT::f16, Promote);
372     setOperationAction(ISD::FP_TO_SINT, MVT::f16, Promote);
373     setOperationAction(ISD::FP_TO_UINT, MVT::f16, Promote);
374     setOperationAction(ISD::SINT_TO_FP, MVT::f16, Promote);
375     setOperationAction(ISD::UINT_TO_FP, MVT::f16, Promote);
376     setOperationAction(ISD::FROUND, MVT::f16, Custom);
377 
378     // F16 - VOP2 Actions.
379     setOperationAction(ISD::BR_CC, MVT::f16, Expand);
380     setOperationAction(ISD::SELECT_CC, MVT::f16, Expand);
381     setOperationAction(ISD::FMAXNUM, MVT::f16, Legal);
382     setOperationAction(ISD::FMINNUM, MVT::f16, Legal);
383     setOperationAction(ISD::FDIV, MVT::f16, Custom);
384 
385     // F16 - VOP3 Actions.
386     setOperationAction(ISD::FMA, MVT::f16, Legal);
387     if (!Subtarget->hasFP16Denormals())
388       setOperationAction(ISD::FMAD, MVT::f16, Legal);
389   }
390 
391   if (Subtarget->hasVOP3PInsts()) {
392     for (MVT VT : {MVT::v2i16, MVT::v2f16}) {
393       for (unsigned Op = 0; Op < ISD::BUILTIN_OP_END; ++Op) {
394         switch (Op) {
395         case ISD::LOAD:
396         case ISD::STORE:
397         case ISD::BUILD_VECTOR:
398         case ISD::BITCAST:
399         case ISD::EXTRACT_VECTOR_ELT:
400         case ISD::INSERT_VECTOR_ELT:
401         case ISD::INSERT_SUBVECTOR:
402         case ISD::EXTRACT_SUBVECTOR:
403         case ISD::SCALAR_TO_VECTOR:
404           break;
405         case ISD::CONCAT_VECTORS:
406           setOperationAction(Op, VT, Custom);
407           break;
408         default:
409           setOperationAction(Op, VT, Expand);
410           break;
411         }
412       }
413     }
414 
415     // XXX - Do these do anything? Vector constants turn into build_vector.
416     setOperationAction(ISD::Constant, MVT::v2i16, Legal);
417     setOperationAction(ISD::ConstantFP, MVT::v2f16, Legal);
418 
419     setOperationAction(ISD::STORE, MVT::v2i16, Promote);
420     AddPromotedToType(ISD::STORE, MVT::v2i16, MVT::i32);
421     setOperationAction(ISD::STORE, MVT::v2f16, Promote);
422     AddPromotedToType(ISD::STORE, MVT::v2f16, MVT::i32);
423 
424     setOperationAction(ISD::LOAD, MVT::v2i16, Promote);
425     AddPromotedToType(ISD::LOAD, MVT::v2i16, MVT::i32);
426     setOperationAction(ISD::LOAD, MVT::v2f16, Promote);
427     AddPromotedToType(ISD::LOAD, MVT::v2f16, MVT::i32);
428 
429     setOperationAction(ISD::AND, MVT::v2i16, Promote);
430     AddPromotedToType(ISD::AND, MVT::v2i16, MVT::i32);
431     setOperationAction(ISD::OR, MVT::v2i16, Promote);
432     AddPromotedToType(ISD::OR, MVT::v2i16, MVT::i32);
433     setOperationAction(ISD::XOR, MVT::v2i16, Promote);
434     AddPromotedToType(ISD::XOR, MVT::v2i16, MVT::i32);
435     setOperationAction(ISD::SELECT, MVT::v2i16, Promote);
436     AddPromotedToType(ISD::SELECT, MVT::v2i16, MVT::i32);
437     setOperationAction(ISD::SELECT, MVT::v2f16, Promote);
438     AddPromotedToType(ISD::SELECT, MVT::v2f16, MVT::i32);
439 
440     setOperationAction(ISD::ADD, MVT::v2i16, Legal);
441     setOperationAction(ISD::SUB, MVT::v2i16, Legal);
442     setOperationAction(ISD::MUL, MVT::v2i16, Legal);
443     setOperationAction(ISD::SHL, MVT::v2i16, Legal);
444     setOperationAction(ISD::SRL, MVT::v2i16, Legal);
445     setOperationAction(ISD::SRA, MVT::v2i16, Legal);
446     setOperationAction(ISD::SMIN, MVT::v2i16, Legal);
447     setOperationAction(ISD::UMIN, MVT::v2i16, Legal);
448     setOperationAction(ISD::SMAX, MVT::v2i16, Legal);
449     setOperationAction(ISD::UMAX, MVT::v2i16, Legal);
450 
451     setOperationAction(ISD::FADD, MVT::v2f16, Legal);
452     setOperationAction(ISD::FNEG, MVT::v2f16, Legal);
453     setOperationAction(ISD::FMUL, MVT::v2f16, Legal);
454     setOperationAction(ISD::FMA, MVT::v2f16, Legal);
455     setOperationAction(ISD::FMINNUM, MVT::v2f16, Legal);
456     setOperationAction(ISD::FMAXNUM, MVT::v2f16, Legal);
457 
458     // This isn't really legal, but this avoids the legalizer unrolling it (and
459     // allows matching fneg (fabs x) patterns)
460     setOperationAction(ISD::FABS, MVT::v2f16, Legal);
461 
462     setOperationAction(ISD::EXTRACT_VECTOR_ELT, MVT::v2i16, Custom);
463     setOperationAction(ISD::EXTRACT_VECTOR_ELT, MVT::v2f16, Custom);
464 
465     setOperationAction(ISD::ZERO_EXTEND, MVT::v2i32, Expand);
466     setOperationAction(ISD::SIGN_EXTEND, MVT::v2i32, Expand);
467     setOperationAction(ISD::FP_EXTEND, MVT::v2f32, Expand);
468   } else {
469     setOperationAction(ISD::SELECT, MVT::v2i16, Custom);
470     setOperationAction(ISD::SELECT, MVT::v2f16, Custom);
471   }
472 
473   for (MVT VT : { MVT::v4i16, MVT::v4f16, MVT::v2i8, MVT::v4i8, MVT::v8i8 }) {
474     setOperationAction(ISD::SELECT, VT, Custom);
475   }
476 
477   setTargetDAGCombine(ISD::ADD);
478   setTargetDAGCombine(ISD::ADDCARRY);
479   setTargetDAGCombine(ISD::SUB);
480   setTargetDAGCombine(ISD::SUBCARRY);
481   setTargetDAGCombine(ISD::FADD);
482   setTargetDAGCombine(ISD::FSUB);
483   setTargetDAGCombine(ISD::FMINNUM);
484   setTargetDAGCombine(ISD::FMAXNUM);
485   setTargetDAGCombine(ISD::SMIN);
486   setTargetDAGCombine(ISD::SMAX);
487   setTargetDAGCombine(ISD::UMIN);
488   setTargetDAGCombine(ISD::UMAX);
489   setTargetDAGCombine(ISD::SETCC);
490   setTargetDAGCombine(ISD::AND);
491   setTargetDAGCombine(ISD::OR);
492   setTargetDAGCombine(ISD::XOR);
493   setTargetDAGCombine(ISD::SINT_TO_FP);
494   setTargetDAGCombine(ISD::UINT_TO_FP);
495   setTargetDAGCombine(ISD::FCANONICALIZE);
496   setTargetDAGCombine(ISD::SCALAR_TO_VECTOR);
497   setTargetDAGCombine(ISD::ZERO_EXTEND);
498   setTargetDAGCombine(ISD::EXTRACT_VECTOR_ELT);
499 
500   // All memory operations. Some folding on the pointer operand is done to help
501   // matching the constant offsets in the addressing modes.
502   setTargetDAGCombine(ISD::LOAD);
503   setTargetDAGCombine(ISD::STORE);
504   setTargetDAGCombine(ISD::ATOMIC_LOAD);
505   setTargetDAGCombine(ISD::ATOMIC_STORE);
506   setTargetDAGCombine(ISD::ATOMIC_CMP_SWAP);
507   setTargetDAGCombine(ISD::ATOMIC_CMP_SWAP_WITH_SUCCESS);
508   setTargetDAGCombine(ISD::ATOMIC_SWAP);
509   setTargetDAGCombine(ISD::ATOMIC_LOAD_ADD);
510   setTargetDAGCombine(ISD::ATOMIC_LOAD_SUB);
511   setTargetDAGCombine(ISD::ATOMIC_LOAD_AND);
512   setTargetDAGCombine(ISD::ATOMIC_LOAD_OR);
513   setTargetDAGCombine(ISD::ATOMIC_LOAD_XOR);
514   setTargetDAGCombine(ISD::ATOMIC_LOAD_NAND);
515   setTargetDAGCombine(ISD::ATOMIC_LOAD_MIN);
516   setTargetDAGCombine(ISD::ATOMIC_LOAD_MAX);
517   setTargetDAGCombine(ISD::ATOMIC_LOAD_UMIN);
518   setTargetDAGCombine(ISD::ATOMIC_LOAD_UMAX);
519 
520   setSchedulingPreference(Sched::RegPressure);
521 }
522 
523 const SISubtarget *SITargetLowering::getSubtarget() const {
524   return static_cast<const SISubtarget *>(Subtarget);
525 }
526 
527 //===----------------------------------------------------------------------===//
528 // TargetLowering queries
529 //===----------------------------------------------------------------------===//
530 
531 bool SITargetLowering::isShuffleMaskLegal(ArrayRef<int>, EVT) const {
532   // SI has some legal vector types, but no legal vector operations. Say no
533   // shuffles are legal in order to prefer scalarizing some vector operations.
534   return false;
535 }
536 
537 bool SITargetLowering::getTgtMemIntrinsic(IntrinsicInfo &Info,
538                                           const CallInst &CI,
539                                           unsigned IntrID) const {
540   switch (IntrID) {
541   case Intrinsic::amdgcn_atomic_inc:
542   case Intrinsic::amdgcn_atomic_dec: {
543     Info.opc = ISD::INTRINSIC_W_CHAIN;
544     Info.memVT = MVT::getVT(CI.getType());
545     Info.ptrVal = CI.getOperand(0);
546     Info.align = 0;
547 
548     const ConstantInt *Vol = dyn_cast<ConstantInt>(CI.getOperand(4));
549     Info.vol = !Vol || !Vol->isZero();
550     Info.readMem = true;
551     Info.writeMem = true;
552     return true;
553   }
554   default:
555     return false;
556   }
557 }
558 
559 bool SITargetLowering::getAddrModeArguments(IntrinsicInst *II,
560                                             SmallVectorImpl<Value*> &Ops,
561                                             Type *&AccessTy) const {
562   switch (II->getIntrinsicID()) {
563   case Intrinsic::amdgcn_atomic_inc:
564   case Intrinsic::amdgcn_atomic_dec: {
565     Value *Ptr = II->getArgOperand(0);
566     AccessTy = II->getType();
567     Ops.push_back(Ptr);
568     return true;
569   }
570   default:
571     return false;
572   }
573 }
574 
575 bool SITargetLowering::isLegalFlatAddressingMode(const AddrMode &AM) const {
576   if (!Subtarget->hasFlatInstOffsets()) {
577     // Flat instructions do not have offsets, and only have the register
578     // address.
579     return AM.BaseOffs == 0 && AM.Scale == 0;
580   }
581 
582   // GFX9 added a 13-bit signed offset. When using regular flat instructions,
583   // the sign bit is ignored and is treated as a 12-bit unsigned offset.
584 
585   // Just r + i
586   return isUInt<12>(AM.BaseOffs) && AM.Scale == 0;
587 }
588 
589 bool SITargetLowering::isLegalMUBUFAddressingMode(const AddrMode &AM) const {
590   // MUBUF / MTBUF instructions have a 12-bit unsigned byte offset, and
591   // additionally can do r + r + i with addr64. 32-bit has more addressing
592   // mode options. Depending on the resource constant, it can also do
593   // (i64 r0) + (i32 r1) * (i14 i).
594   //
595   // Private arrays end up using a scratch buffer most of the time, so also
596   // assume those use MUBUF instructions. Scratch loads / stores are currently
597   // implemented as mubuf instructions with offen bit set, so slightly
598   // different than the normal addr64.
599   if (!isUInt<12>(AM.BaseOffs))
600     return false;
601 
602   // FIXME: Since we can split immediate into soffset and immediate offset,
603   // would it make sense to allow any immediate?
604 
605   switch (AM.Scale) {
606   case 0: // r + i or just i, depending on HasBaseReg.
607     return true;
608   case 1:
609     return true; // We have r + r or r + i.
610   case 2:
611     if (AM.HasBaseReg) {
612       // Reject 2 * r + r.
613       return false;
614     }
615 
616     // Allow 2 * r as r + r
617     // Or  2 * r + i is allowed as r + r + i.
618     return true;
619   default: // Don't allow n * r
620     return false;
621   }
622 }
623 
624 bool SITargetLowering::isLegalAddressingMode(const DataLayout &DL,
625                                              const AddrMode &AM, Type *Ty,
626                                              unsigned AS, Instruction *I) const {
627   // No global is ever allowed as a base.
628   if (AM.BaseGV)
629     return false;
630 
631   if (AS == AMDGPUASI.GLOBAL_ADDRESS) {
632     if (Subtarget->getGeneration() >= SISubtarget::VOLCANIC_ISLANDS) {
633       // Assume the we will use FLAT for all global memory accesses
634       // on VI.
635       // FIXME: This assumption is currently wrong.  On VI we still use
636       // MUBUF instructions for the r + i addressing mode.  As currently
637       // implemented, the MUBUF instructions only work on buffer < 4GB.
638       // It may be possible to support > 4GB buffers with MUBUF instructions,
639       // by setting the stride value in the resource descriptor which would
640       // increase the size limit to (stride * 4GB).  However, this is risky,
641       // because it has never been validated.
642       return isLegalFlatAddressingMode(AM);
643     }
644 
645     return isLegalMUBUFAddressingMode(AM);
646   } else if (AS == AMDGPUASI.CONSTANT_ADDRESS) {
647     // If the offset isn't a multiple of 4, it probably isn't going to be
648     // correctly aligned.
649     // FIXME: Can we get the real alignment here?
650     if (AM.BaseOffs % 4 != 0)
651       return isLegalMUBUFAddressingMode(AM);
652 
653     // There are no SMRD extloads, so if we have to do a small type access we
654     // will use a MUBUF load.
655     // FIXME?: We also need to do this if unaligned, but we don't know the
656     // alignment here.
657     if (DL.getTypeStoreSize(Ty) < 4)
658       return isLegalMUBUFAddressingMode(AM);
659 
660     if (Subtarget->getGeneration() == SISubtarget::SOUTHERN_ISLANDS) {
661       // SMRD instructions have an 8-bit, dword offset on SI.
662       if (!isUInt<8>(AM.BaseOffs / 4))
663         return false;
664     } else if (Subtarget->getGeneration() == SISubtarget::SEA_ISLANDS) {
665       // On CI+, this can also be a 32-bit literal constant offset. If it fits
666       // in 8-bits, it can use a smaller encoding.
667       if (!isUInt<32>(AM.BaseOffs / 4))
668         return false;
669     } else if (Subtarget->getGeneration() >= SISubtarget::VOLCANIC_ISLANDS) {
670       // On VI, these use the SMEM format and the offset is 20-bit in bytes.
671       if (!isUInt<20>(AM.BaseOffs))
672         return false;
673     } else
674       llvm_unreachable("unhandled generation");
675 
676     if (AM.Scale == 0) // r + i or just i, depending on HasBaseReg.
677       return true;
678 
679     if (AM.Scale == 1 && AM.HasBaseReg)
680       return true;
681 
682     return false;
683 
684   } else if (AS == AMDGPUASI.PRIVATE_ADDRESS) {
685     return isLegalMUBUFAddressingMode(AM);
686   } else if (AS == AMDGPUASI.LOCAL_ADDRESS ||
687              AS == AMDGPUASI.REGION_ADDRESS) {
688     // Basic, single offset DS instructions allow a 16-bit unsigned immediate
689     // field.
690     // XXX - If doing a 4-byte aligned 8-byte type access, we effectively have
691     // an 8-bit dword offset but we don't know the alignment here.
692     if (!isUInt<16>(AM.BaseOffs))
693       return false;
694 
695     if (AM.Scale == 0) // r + i or just i, depending on HasBaseReg.
696       return true;
697 
698     if (AM.Scale == 1 && AM.HasBaseReg)
699       return true;
700 
701     return false;
702   } else if (AS == AMDGPUASI.FLAT_ADDRESS ||
703              AS == AMDGPUASI.UNKNOWN_ADDRESS_SPACE) {
704     // For an unknown address space, this usually means that this is for some
705     // reason being used for pure arithmetic, and not based on some addressing
706     // computation. We don't have instructions that compute pointers with any
707     // addressing modes, so treat them as having no offset like flat
708     // instructions.
709     return isLegalFlatAddressingMode(AM);
710   } else {
711     llvm_unreachable("unhandled address space");
712   }
713 }
714 
715 bool SITargetLowering::canMergeStoresTo(unsigned AS, EVT MemVT,
716                                         const SelectionDAG &DAG) const {
717   if (AS == AMDGPUASI.GLOBAL_ADDRESS || AS == AMDGPUASI.FLAT_ADDRESS) {
718     return (MemVT.getSizeInBits() <= 4 * 32);
719   } else if (AS == AMDGPUASI.PRIVATE_ADDRESS) {
720     unsigned MaxPrivateBits = 8 * getSubtarget()->getMaxPrivateElementSize();
721     return (MemVT.getSizeInBits() <= MaxPrivateBits);
722   } else if (AS == AMDGPUASI.LOCAL_ADDRESS) {
723     return (MemVT.getSizeInBits() <= 2 * 32);
724   }
725   return true;
726 }
727 
728 bool SITargetLowering::allowsMisalignedMemoryAccesses(EVT VT,
729                                                       unsigned AddrSpace,
730                                                       unsigned Align,
731                                                       bool *IsFast) const {
732   if (IsFast)
733     *IsFast = false;
734 
735   // TODO: I think v3i32 should allow unaligned accesses on CI with DS_READ_B96,
736   // which isn't a simple VT.
737   // Until MVT is extended to handle this, simply check for the size and
738   // rely on the condition below: allow accesses if the size is a multiple of 4.
739   if (VT == MVT::Other || (VT != MVT::Other && VT.getSizeInBits() > 1024 &&
740                            VT.getStoreSize() > 16)) {
741     return false;
742   }
743 
744   if (AddrSpace == AMDGPUASI.LOCAL_ADDRESS ||
745       AddrSpace == AMDGPUASI.REGION_ADDRESS) {
746     // ds_read/write_b64 require 8-byte alignment, but we can do a 4 byte
747     // aligned, 8 byte access in a single operation using ds_read2/write2_b32
748     // with adjacent offsets.
749     bool AlignedBy4 = (Align % 4 == 0);
750     if (IsFast)
751       *IsFast = AlignedBy4;
752 
753     return AlignedBy4;
754   }
755 
756   // FIXME: We have to be conservative here and assume that flat operations
757   // will access scratch.  If we had access to the IR function, then we
758   // could determine if any private memory was used in the function.
759   if (!Subtarget->hasUnalignedScratchAccess() &&
760       (AddrSpace == AMDGPUASI.PRIVATE_ADDRESS ||
761        AddrSpace == AMDGPUASI.FLAT_ADDRESS)) {
762     return false;
763   }
764 
765   if (Subtarget->hasUnalignedBufferAccess()) {
766     // If we have an uniform constant load, it still requires using a slow
767     // buffer instruction if unaligned.
768     if (IsFast) {
769       *IsFast = (AddrSpace == AMDGPUASI.CONSTANT_ADDRESS) ?
770         (Align % 4 == 0) : true;
771     }
772 
773     return true;
774   }
775 
776   // Smaller than dword value must be aligned.
777   if (VT.bitsLT(MVT::i32))
778     return false;
779 
780   // 8.1.6 - For Dword or larger reads or writes, the two LSBs of the
781   // byte-address are ignored, thus forcing Dword alignment.
782   // This applies to private, global, and constant memory.
783   if (IsFast)
784     *IsFast = true;
785 
786   return VT.bitsGT(MVT::i32) && Align % 4 == 0;
787 }
788 
789 EVT SITargetLowering::getOptimalMemOpType(uint64_t Size, unsigned DstAlign,
790                                           unsigned SrcAlign, bool IsMemset,
791                                           bool ZeroMemset,
792                                           bool MemcpyStrSrc,
793                                           MachineFunction &MF) const {
794   // FIXME: Should account for address space here.
795 
796   // The default fallback uses the private pointer size as a guess for a type to
797   // use. Make sure we switch these to 64-bit accesses.
798 
799   if (Size >= 16 && DstAlign >= 4) // XXX: Should only do for global
800     return MVT::v4i32;
801 
802   if (Size >= 8 && DstAlign >= 4)
803     return MVT::v2i32;
804 
805   // Use the default.
806   return MVT::Other;
807 }
808 
809 static bool isFlatGlobalAddrSpace(unsigned AS, AMDGPUAS AMDGPUASI) {
810   return AS == AMDGPUASI.GLOBAL_ADDRESS ||
811          AS == AMDGPUASI.FLAT_ADDRESS ||
812          AS == AMDGPUASI.CONSTANT_ADDRESS;
813 }
814 
815 bool SITargetLowering::isNoopAddrSpaceCast(unsigned SrcAS,
816                                            unsigned DestAS) const {
817   return isFlatGlobalAddrSpace(SrcAS, AMDGPUASI) &&
818          isFlatGlobalAddrSpace(DestAS, AMDGPUASI);
819 }
820 
821 bool SITargetLowering::isMemOpHasNoClobberedMemOperand(const SDNode *N) const {
822   const MemSDNode *MemNode = cast<MemSDNode>(N);
823   const Value *Ptr = MemNode->getMemOperand()->getValue();
824   const Instruction *I = dyn_cast<Instruction>(Ptr);
825   return I && I->getMetadata("amdgpu.noclobber");
826 }
827 
828 bool SITargetLowering::isCheapAddrSpaceCast(unsigned SrcAS,
829                                             unsigned DestAS) const {
830   // Flat -> private/local is a simple truncate.
831   // Flat -> global is no-op
832   if (SrcAS == AMDGPUASI.FLAT_ADDRESS)
833     return true;
834 
835   return isNoopAddrSpaceCast(SrcAS, DestAS);
836 }
837 
838 bool SITargetLowering::isMemOpUniform(const SDNode *N) const {
839   const MemSDNode *MemNode = cast<MemSDNode>(N);
840 
841   return AMDGPU::isUniformMMO(MemNode->getMemOperand());
842 }
843 
844 TargetLoweringBase::LegalizeTypeAction
845 SITargetLowering::getPreferredVectorAction(EVT VT) const {
846   if (VT.getVectorNumElements() != 1 && VT.getScalarType().bitsLE(MVT::i16))
847     return TypeSplitVector;
848 
849   return TargetLoweringBase::getPreferredVectorAction(VT);
850 }
851 
852 bool SITargetLowering::shouldConvertConstantLoadToIntImm(const APInt &Imm,
853                                                          Type *Ty) const {
854   // FIXME: Could be smarter if called for vector constants.
855   return true;
856 }
857 
858 bool SITargetLowering::isTypeDesirableForOp(unsigned Op, EVT VT) const {
859   if (Subtarget->has16BitInsts() && VT == MVT::i16) {
860     switch (Op) {
861     case ISD::LOAD:
862     case ISD::STORE:
863 
864     // These operations are done with 32-bit instructions anyway.
865     case ISD::AND:
866     case ISD::OR:
867     case ISD::XOR:
868     case ISD::SELECT:
869       // TODO: Extensions?
870       return true;
871     default:
872       return false;
873     }
874   }
875 
876   // SimplifySetCC uses this function to determine whether or not it should
877   // create setcc with i1 operands.  We don't have instructions for i1 setcc.
878   if (VT == MVT::i1 && Op == ISD::SETCC)
879     return false;
880 
881   return TargetLowering::isTypeDesirableForOp(Op, VT);
882 }
883 
884 SDValue SITargetLowering::lowerKernArgParameterPtr(SelectionDAG &DAG,
885                                                    const SDLoc &SL,
886                                                    SDValue Chain,
887                                                    uint64_t Offset) const {
888   const DataLayout &DL = DAG.getDataLayout();
889   MachineFunction &MF = DAG.getMachineFunction();
890   const SIRegisterInfo *TRI = getSubtarget()->getRegisterInfo();
891   unsigned InputPtrReg = TRI->getPreloadedValue(MF,
892                                                 SIRegisterInfo::KERNARG_SEGMENT_PTR);
893 
894   MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
895   MVT PtrVT = getPointerTy(DL, AMDGPUASI.CONSTANT_ADDRESS);
896   SDValue BasePtr = DAG.getCopyFromReg(Chain, SL,
897                                        MRI.getLiveInVirtReg(InputPtrReg), PtrVT);
898   return DAG.getNode(ISD::ADD, SL, PtrVT, BasePtr,
899                      DAG.getConstant(Offset, SL, PtrVT));
900 }
901 
902 SDValue SITargetLowering::getImplicitArgPtr(SelectionDAG &DAG,
903                                             const SDLoc &SL) const {
904   auto MFI = DAG.getMachineFunction().getInfo<SIMachineFunctionInfo>();
905   uint64_t Offset = getImplicitParameterOffset(MFI, FIRST_IMPLICIT);
906   return lowerKernArgParameterPtr(DAG, SL, DAG.getEntryNode(), Offset);
907 }
908 
909 SDValue SITargetLowering::convertArgType(SelectionDAG &DAG, EVT VT, EVT MemVT,
910                                          const SDLoc &SL, SDValue Val,
911                                          bool Signed,
912                                          const ISD::InputArg *Arg) const {
913   if (Arg && (Arg->Flags.isSExt() || Arg->Flags.isZExt()) &&
914       VT.bitsLT(MemVT)) {
915     unsigned Opc = Arg->Flags.isZExt() ? ISD::AssertZext : ISD::AssertSext;
916     Val = DAG.getNode(Opc, SL, MemVT, Val, DAG.getValueType(VT));
917   }
918 
919   if (MemVT.isFloatingPoint())
920     Val = getFPExtOrFPTrunc(DAG, Val, SL, VT);
921   else if (Signed)
922     Val = DAG.getSExtOrTrunc(Val, SL, VT);
923   else
924     Val = DAG.getZExtOrTrunc(Val, SL, VT);
925 
926   return Val;
927 }
928 
929 SDValue SITargetLowering::lowerKernargMemParameter(
930   SelectionDAG &DAG, EVT VT, EVT MemVT,
931   const SDLoc &SL, SDValue Chain,
932   uint64_t Offset, bool Signed,
933   const ISD::InputArg *Arg) const {
934   const DataLayout &DL = DAG.getDataLayout();
935   Type *Ty = MemVT.getTypeForEVT(*DAG.getContext());
936   PointerType *PtrTy = PointerType::get(Ty, AMDGPUASI.CONSTANT_ADDRESS);
937   MachinePointerInfo PtrInfo(UndefValue::get(PtrTy));
938 
939   unsigned Align = DL.getABITypeAlignment(Ty);
940 
941   SDValue Ptr = lowerKernArgParameterPtr(DAG, SL, Chain, Offset);
942   SDValue Load = DAG.getLoad(MemVT, SL, Chain, Ptr, PtrInfo, Align,
943                              MachineMemOperand::MONonTemporal |
944                              MachineMemOperand::MODereferenceable |
945                              MachineMemOperand::MOInvariant);
946 
947   SDValue Val = convertArgType(DAG, VT, MemVT, SL, Load, Signed, Arg);
948   return DAG.getMergeValues({ Val, Load.getValue(1) }, SL);
949 }
950 
951 SDValue SITargetLowering::lowerStackParameter(SelectionDAG &DAG, CCValAssign &VA,
952                                               const SDLoc &SL, SDValue Chain,
953                                               const ISD::InputArg &Arg) const {
954   MachineFunction &MF = DAG.getMachineFunction();
955   MachineFrameInfo &MFI = MF.getFrameInfo();
956 
957   if (Arg.Flags.isByVal()) {
958     unsigned Size = Arg.Flags.getByValSize();
959     int FrameIdx = MFI.CreateFixedObject(Size, VA.getLocMemOffset(), false);
960     return DAG.getFrameIndex(FrameIdx, MVT::i32);
961   }
962 
963   unsigned ArgOffset = VA.getLocMemOffset();
964   unsigned ArgSize = VA.getValVT().getStoreSize();
965 
966   int FI = MFI.CreateFixedObject(ArgSize, ArgOffset, true);
967 
968   // Create load nodes to retrieve arguments from the stack.
969   SDValue FIN = DAG.getFrameIndex(FI, MVT::i32);
970   SDValue ArgValue;
971 
972   // For NON_EXTLOAD, generic code in getLoad assert(ValVT == MemVT)
973   ISD::LoadExtType ExtType = ISD::NON_EXTLOAD;
974   MVT MemVT = VA.getValVT();
975 
976   switch (VA.getLocInfo()) {
977   default:
978     break;
979   case CCValAssign::BCvt:
980     MemVT = VA.getLocVT();
981     break;
982   case CCValAssign::SExt:
983     ExtType = ISD::SEXTLOAD;
984     break;
985   case CCValAssign::ZExt:
986     ExtType = ISD::ZEXTLOAD;
987     break;
988   case CCValAssign::AExt:
989     ExtType = ISD::EXTLOAD;
990     break;
991   }
992 
993   ArgValue = DAG.getExtLoad(
994     ExtType, SL, VA.getLocVT(), Chain, FIN,
995     MachinePointerInfo::getFixedStack(DAG.getMachineFunction(), FI),
996     MemVT);
997   return ArgValue;
998 }
999 
1000 static void processShaderInputArgs(SmallVectorImpl<ISD::InputArg> &Splits,
1001                                    CallingConv::ID CallConv,
1002                                    ArrayRef<ISD::InputArg> Ins,
1003                                    BitVector &Skipped,
1004                                    FunctionType *FType,
1005                                    SIMachineFunctionInfo *Info) {
1006   for (unsigned I = 0, E = Ins.size(), PSInputNum = 0; I != E; ++I) {
1007     const ISD::InputArg &Arg = Ins[I];
1008 
1009     // First check if it's a PS input addr.
1010     if (CallConv == CallingConv::AMDGPU_PS && !Arg.Flags.isInReg() &&
1011         !Arg.Flags.isByVal() && PSInputNum <= 15) {
1012 
1013       if (!Arg.Used && !Info->isPSInputAllocated(PSInputNum)) {
1014         // We can safely skip PS inputs.
1015         Skipped.set(I);
1016         ++PSInputNum;
1017         continue;
1018       }
1019 
1020       Info->markPSInputAllocated(PSInputNum);
1021       if (Arg.Used)
1022         Info->markPSInputEnabled(PSInputNum);
1023 
1024       ++PSInputNum;
1025     }
1026 
1027     // Second split vertices into their elements.
1028     if (Arg.VT.isVector()) {
1029       ISD::InputArg NewArg = Arg;
1030       NewArg.Flags.setSplit();
1031       NewArg.VT = Arg.VT.getVectorElementType();
1032 
1033       // We REALLY want the ORIGINAL number of vertex elements here, e.g. a
1034       // three or five element vertex only needs three or five registers,
1035       // NOT four or eight.
1036       Type *ParamType = FType->getParamType(Arg.getOrigArgIndex());
1037       unsigned NumElements = ParamType->getVectorNumElements();
1038 
1039       for (unsigned J = 0; J != NumElements; ++J) {
1040         Splits.push_back(NewArg);
1041         NewArg.PartOffset += NewArg.VT.getStoreSize();
1042       }
1043     } else {
1044       Splits.push_back(Arg);
1045     }
1046   }
1047 }
1048 
1049 // Allocate special inputs passed in VGPRs.
1050 static void allocateSpecialInputVGPRs(CCState &CCInfo,
1051                                       MachineFunction &MF,
1052                                       const SIRegisterInfo &TRI,
1053                                       SIMachineFunctionInfo &Info) {
1054   if (Info.hasWorkItemIDX()) {
1055     unsigned Reg = TRI.getPreloadedValue(MF, SIRegisterInfo::WORKITEM_ID_X);
1056     MF.addLiveIn(Reg, &AMDGPU::VGPR_32RegClass);
1057     CCInfo.AllocateReg(Reg);
1058   }
1059 
1060   if (Info.hasWorkItemIDY()) {
1061     unsigned Reg = TRI.getPreloadedValue(MF, SIRegisterInfo::WORKITEM_ID_Y);
1062     MF.addLiveIn(Reg, &AMDGPU::VGPR_32RegClass);
1063     CCInfo.AllocateReg(Reg);
1064   }
1065 
1066   if (Info.hasWorkItemIDZ()) {
1067     unsigned Reg = TRI.getPreloadedValue(MF, SIRegisterInfo::WORKITEM_ID_Z);
1068     MF.addLiveIn(Reg, &AMDGPU::VGPR_32RegClass);
1069     CCInfo.AllocateReg(Reg);
1070   }
1071 }
1072 
1073 // Allocate special inputs passed in user SGPRs.
1074 static void allocateHSAUserSGPRs(CCState &CCInfo,
1075                                  MachineFunction &MF,
1076                                  const SIRegisterInfo &TRI,
1077                                  SIMachineFunctionInfo &Info) {
1078   if (Info.hasImplicitBufferPtr()) {
1079     unsigned ImplicitBufferPtrReg = Info.addImplicitBufferPtr(TRI);
1080     MF.addLiveIn(ImplicitBufferPtrReg, &AMDGPU::SGPR_64RegClass);
1081     CCInfo.AllocateReg(ImplicitBufferPtrReg);
1082   }
1083 
1084   // FIXME: How should these inputs interact with inreg / custom SGPR inputs?
1085   if (Info.hasPrivateSegmentBuffer()) {
1086     unsigned PrivateSegmentBufferReg = Info.addPrivateSegmentBuffer(TRI);
1087     MF.addLiveIn(PrivateSegmentBufferReg, &AMDGPU::SGPR_128RegClass);
1088     CCInfo.AllocateReg(PrivateSegmentBufferReg);
1089   }
1090 
1091   if (Info.hasDispatchPtr()) {
1092     unsigned DispatchPtrReg = Info.addDispatchPtr(TRI);
1093     MF.addLiveIn(DispatchPtrReg, &AMDGPU::SGPR_64RegClass);
1094     CCInfo.AllocateReg(DispatchPtrReg);
1095   }
1096 
1097   if (Info.hasQueuePtr()) {
1098     unsigned QueuePtrReg = Info.addQueuePtr(TRI);
1099     MF.addLiveIn(QueuePtrReg, &AMDGPU::SGPR_64RegClass);
1100     CCInfo.AllocateReg(QueuePtrReg);
1101   }
1102 
1103   if (Info.hasKernargSegmentPtr()) {
1104     unsigned InputPtrReg = Info.addKernargSegmentPtr(TRI);
1105     MF.addLiveIn(InputPtrReg, &AMDGPU::SGPR_64RegClass);
1106     CCInfo.AllocateReg(InputPtrReg);
1107   }
1108 
1109   if (Info.hasDispatchID()) {
1110     unsigned DispatchIDReg = Info.addDispatchID(TRI);
1111     MF.addLiveIn(DispatchIDReg, &AMDGPU::SGPR_64RegClass);
1112     CCInfo.AllocateReg(DispatchIDReg);
1113   }
1114 
1115   if (Info.hasFlatScratchInit()) {
1116     unsigned FlatScratchInitReg = Info.addFlatScratchInit(TRI);
1117     MF.addLiveIn(FlatScratchInitReg, &AMDGPU::SGPR_64RegClass);
1118     CCInfo.AllocateReg(FlatScratchInitReg);
1119   }
1120 
1121   // TODO: Add GridWorkGroupCount user SGPRs when used. For now with HSA we read
1122   // these from the dispatch pointer.
1123 }
1124 
1125 // Allocate special input registers that are initialized per-wave.
1126 static void allocateSystemSGPRs(CCState &CCInfo,
1127                                 MachineFunction &MF,
1128                                 SIMachineFunctionInfo &Info,
1129                                 CallingConv::ID CallConv,
1130                                 bool IsShader) {
1131   if (Info.hasWorkGroupIDX()) {
1132     unsigned Reg = Info.addWorkGroupIDX();
1133     MF.addLiveIn(Reg, &AMDGPU::SReg_32_XM0RegClass);
1134     CCInfo.AllocateReg(Reg);
1135   }
1136 
1137   if (Info.hasWorkGroupIDY()) {
1138     unsigned Reg = Info.addWorkGroupIDY();
1139     MF.addLiveIn(Reg, &AMDGPU::SReg_32_XM0RegClass);
1140     CCInfo.AllocateReg(Reg);
1141   }
1142 
1143   if (Info.hasWorkGroupIDZ()) {
1144     unsigned Reg = Info.addWorkGroupIDZ();
1145     MF.addLiveIn(Reg, &AMDGPU::SReg_32_XM0RegClass);
1146     CCInfo.AllocateReg(Reg);
1147   }
1148 
1149   if (Info.hasWorkGroupInfo()) {
1150     unsigned Reg = Info.addWorkGroupInfo();
1151     MF.addLiveIn(Reg, &AMDGPU::SReg_32_XM0RegClass);
1152     CCInfo.AllocateReg(Reg);
1153   }
1154 
1155   if (Info.hasPrivateSegmentWaveByteOffset()) {
1156     // Scratch wave offset passed in system SGPR.
1157     unsigned PrivateSegmentWaveByteOffsetReg;
1158 
1159     if (IsShader) {
1160       PrivateSegmentWaveByteOffsetReg =
1161         Info.getPrivateSegmentWaveByteOffsetSystemSGPR();
1162 
1163       // This is true if the scratch wave byte offset doesn't have a fixed
1164       // location.
1165       if (PrivateSegmentWaveByteOffsetReg == AMDGPU::NoRegister) {
1166         PrivateSegmentWaveByteOffsetReg = findFirstFreeSGPR(CCInfo);
1167         Info.setPrivateSegmentWaveByteOffset(PrivateSegmentWaveByteOffsetReg);
1168       }
1169     } else
1170       PrivateSegmentWaveByteOffsetReg = Info.addPrivateSegmentWaveByteOffset();
1171 
1172     MF.addLiveIn(PrivateSegmentWaveByteOffsetReg, &AMDGPU::SGPR_32RegClass);
1173     CCInfo.AllocateReg(PrivateSegmentWaveByteOffsetReg);
1174   }
1175 }
1176 
1177 static void reservePrivateMemoryRegs(const TargetMachine &TM,
1178                                      MachineFunction &MF,
1179                                      const SIRegisterInfo &TRI,
1180                                      SIMachineFunctionInfo &Info) {
1181   // Now that we've figured out where the scratch register inputs are, see if
1182   // should reserve the arguments and use them directly.
1183   MachineFrameInfo &MFI = MF.getFrameInfo();
1184   bool HasStackObjects = MFI.hasStackObjects();
1185 
1186   // Record that we know we have non-spill stack objects so we don't need to
1187   // check all stack objects later.
1188   if (HasStackObjects)
1189     Info.setHasNonSpillStackObjects(true);
1190 
1191   // Everything live out of a block is spilled with fast regalloc, so it's
1192   // almost certain that spilling will be required.
1193   if (TM.getOptLevel() == CodeGenOpt::None)
1194     HasStackObjects = true;
1195 
1196   const SISubtarget &ST = MF.getSubtarget<SISubtarget>();
1197   if (ST.isAmdCodeObjectV2(MF)) {
1198     if (HasStackObjects) {
1199       // If we have stack objects, we unquestionably need the private buffer
1200       // resource. For the Code Object V2 ABI, this will be the first 4 user
1201       // SGPR inputs. We can reserve those and use them directly.
1202 
1203       unsigned PrivateSegmentBufferReg = TRI.getPreloadedValue(
1204         MF, SIRegisterInfo::PRIVATE_SEGMENT_BUFFER);
1205       Info.setScratchRSrcReg(PrivateSegmentBufferReg);
1206 
1207       unsigned PrivateSegmentWaveByteOffsetReg = TRI.getPreloadedValue(
1208         MF, SIRegisterInfo::PRIVATE_SEGMENT_WAVE_BYTE_OFFSET);
1209       Info.setScratchWaveOffsetReg(PrivateSegmentWaveByteOffsetReg);
1210     } else {
1211       unsigned ReservedBufferReg
1212         = TRI.reservedPrivateSegmentBufferReg(MF);
1213       unsigned ReservedOffsetReg
1214         = TRI.reservedPrivateSegmentWaveByteOffsetReg(MF);
1215 
1216       // We tentatively reserve the last registers (skipping the last two
1217       // which may contain VCC). After register allocation, we'll replace
1218       // these with the ones immediately after those which were really
1219       // allocated. In the prologue copies will be inserted from the argument
1220       // to these reserved registers.
1221       Info.setScratchRSrcReg(ReservedBufferReg);
1222       Info.setScratchWaveOffsetReg(ReservedOffsetReg);
1223     }
1224   } else {
1225     unsigned ReservedBufferReg = TRI.reservedPrivateSegmentBufferReg(MF);
1226 
1227     // Without HSA, relocations are used for the scratch pointer and the
1228     // buffer resource setup is always inserted in the prologue. Scratch wave
1229     // offset is still in an input SGPR.
1230     Info.setScratchRSrcReg(ReservedBufferReg);
1231 
1232     if (HasStackObjects) {
1233       unsigned ScratchWaveOffsetReg = TRI.getPreloadedValue(
1234         MF, SIRegisterInfo::PRIVATE_SEGMENT_WAVE_BYTE_OFFSET);
1235       Info.setScratchWaveOffsetReg(ScratchWaveOffsetReg);
1236     } else {
1237       unsigned ReservedOffsetReg
1238         = TRI.reservedPrivateSegmentWaveByteOffsetReg(MF);
1239       Info.setScratchWaveOffsetReg(ReservedOffsetReg);
1240     }
1241   }
1242 }
1243 
1244 SDValue SITargetLowering::LowerFormalArguments(
1245     SDValue Chain, CallingConv::ID CallConv, bool isVarArg,
1246     const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
1247     SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
1248   const SIRegisterInfo *TRI = getSubtarget()->getRegisterInfo();
1249 
1250   MachineFunction &MF = DAG.getMachineFunction();
1251   FunctionType *FType = MF.getFunction()->getFunctionType();
1252   SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
1253   const SISubtarget &ST = MF.getSubtarget<SISubtarget>();
1254 
1255   if (Subtarget->isAmdHsaOS() && AMDGPU::isShader(CallConv)) {
1256     const Function *Fn = MF.getFunction();
1257     DiagnosticInfoUnsupported NoGraphicsHSA(
1258         *Fn, "unsupported non-compute shaders with HSA", DL.getDebugLoc());
1259     DAG.getContext()->diagnose(NoGraphicsHSA);
1260     return DAG.getEntryNode();
1261   }
1262 
1263   // Create stack objects that are used for emitting debugger prologue if
1264   // "amdgpu-debugger-emit-prologue" attribute was specified.
1265   if (ST.debuggerEmitPrologue())
1266     createDebuggerPrologueStackObjects(MF);
1267 
1268   SmallVector<ISD::InputArg, 16> Splits;
1269   SmallVector<CCValAssign, 16> ArgLocs;
1270   BitVector Skipped(Ins.size());
1271   CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), ArgLocs,
1272                  *DAG.getContext());
1273 
1274   bool IsShader = AMDGPU::isShader(CallConv);
1275   bool IsKernel = AMDGPU::isKernel(CallConv);
1276   bool IsEntryFunc = AMDGPU::isEntryFunctionCC(CallConv);
1277 
1278   if (IsShader) {
1279     processShaderInputArgs(Splits, CallConv, Ins, Skipped, FType, Info);
1280 
1281     // At least one interpolation mode must be enabled or else the GPU will
1282     // hang.
1283     //
1284     // Check PSInputAddr instead of PSInputEnable. The idea is that if the user
1285     // set PSInputAddr, the user wants to enable some bits after the compilation
1286     // based on run-time states. Since we can't know what the final PSInputEna
1287     // will look like, so we shouldn't do anything here and the user should take
1288     // responsibility for the correct programming.
1289     //
1290     // Otherwise, the following restrictions apply:
1291     // - At least one of PERSP_* (0xF) or LINEAR_* (0x70) must be enabled.
1292     // - If POS_W_FLOAT (11) is enabled, at least one of PERSP_* must be
1293     //   enabled too.
1294     if (CallConv == CallingConv::AMDGPU_PS &&
1295         ((Info->getPSInputAddr() & 0x7F) == 0 ||
1296          ((Info->getPSInputAddr() & 0xF) == 0 &&
1297           Info->isPSInputAllocated(11)))) {
1298       CCInfo.AllocateReg(AMDGPU::VGPR0);
1299       CCInfo.AllocateReg(AMDGPU::VGPR1);
1300       Info->markPSInputAllocated(0);
1301       Info->markPSInputEnabled(0);
1302     }
1303 
1304     assert(!Info->hasDispatchPtr() &&
1305            !Info->hasKernargSegmentPtr() && !Info->hasFlatScratchInit() &&
1306            !Info->hasWorkGroupIDX() && !Info->hasWorkGroupIDY() &&
1307            !Info->hasWorkGroupIDZ() && !Info->hasWorkGroupInfo() &&
1308            !Info->hasWorkItemIDX() && !Info->hasWorkItemIDY() &&
1309            !Info->hasWorkItemIDZ());
1310   } else if (IsKernel) {
1311     assert(Info->hasWorkGroupIDX() && Info->hasWorkItemIDX());
1312   } else {
1313     Splits.append(Ins.begin(), Ins.end());
1314   }
1315 
1316   if (IsEntryFunc) {
1317     allocateSpecialInputVGPRs(CCInfo, MF, *TRI, *Info);
1318     allocateHSAUserSGPRs(CCInfo, MF, *TRI, *Info);
1319   }
1320 
1321   if (IsKernel) {
1322     analyzeFormalArgumentsCompute(CCInfo, Ins);
1323   } else {
1324     CCAssignFn *AssignFn = CCAssignFnForCall(CallConv, isVarArg);
1325     CCInfo.AnalyzeFormalArguments(Splits, AssignFn);
1326   }
1327 
1328   SmallVector<SDValue, 16> Chains;
1329 
1330   for (unsigned i = 0, e = Ins.size(), ArgIdx = 0; i != e; ++i) {
1331     const ISD::InputArg &Arg = Ins[i];
1332     if (Skipped[i]) {
1333       InVals.push_back(DAG.getUNDEF(Arg.VT));
1334       continue;
1335     }
1336 
1337     CCValAssign &VA = ArgLocs[ArgIdx++];
1338     MVT VT = VA.getLocVT();
1339 
1340     if (IsEntryFunc && VA.isMemLoc()) {
1341       VT = Ins[i].VT;
1342       EVT MemVT = VA.getLocVT();
1343 
1344       const uint64_t Offset = Subtarget->getExplicitKernelArgOffset(MF) +
1345         VA.getLocMemOffset();
1346       Info->setABIArgOffset(Offset + MemVT.getStoreSize());
1347 
1348       // The first 36 bytes of the input buffer contains information about
1349       // thread group and global sizes.
1350       SDValue Arg = lowerKernargMemParameter(
1351         DAG, VT, MemVT, DL, Chain, Offset, Ins[i].Flags.isSExt(), &Ins[i]);
1352       Chains.push_back(Arg.getValue(1));
1353 
1354       auto *ParamTy =
1355         dyn_cast<PointerType>(FType->getParamType(Ins[i].getOrigArgIndex()));
1356       if (Subtarget->getGeneration() == SISubtarget::SOUTHERN_ISLANDS &&
1357           ParamTy && ParamTy->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS) {
1358         // On SI local pointers are just offsets into LDS, so they are always
1359         // less than 16-bits.  On CI and newer they could potentially be
1360         // real pointers, so we can't guarantee their size.
1361         Arg = DAG.getNode(ISD::AssertZext, DL, Arg.getValueType(), Arg,
1362                           DAG.getValueType(MVT::i16));
1363       }
1364 
1365       InVals.push_back(Arg);
1366       continue;
1367     } else if (!IsEntryFunc && VA.isMemLoc()) {
1368       SDValue Val = lowerStackParameter(DAG, VA, DL, Chain, Arg);
1369       InVals.push_back(Val);
1370       if (!Arg.Flags.isByVal())
1371         Chains.push_back(Val.getValue(1));
1372       continue;
1373     }
1374 
1375     assert(VA.isRegLoc() && "Parameter must be in a register!");
1376 
1377     unsigned Reg = VA.getLocReg();
1378     const TargetRegisterClass *RC = TRI->getMinimalPhysRegClass(Reg, VT);
1379     EVT ValVT = VA.getValVT();
1380 
1381     Reg = MF.addLiveIn(Reg, RC);
1382     SDValue Val = DAG.getCopyFromReg(Chain, DL, Reg, VT);
1383 
1384     // If this is an 8 or 16-bit value, it is really passed promoted
1385     // to 32 bits. Insert an assert[sz]ext to capture this, then
1386     // truncate to the right size.
1387     switch (VA.getLocInfo()) {
1388     case CCValAssign::Full:
1389       break;
1390     case CCValAssign::BCvt:
1391       Val = DAG.getNode(ISD::BITCAST, DL, ValVT, Val);
1392       break;
1393     case CCValAssign::SExt:
1394       Val = DAG.getNode(ISD::AssertSext, DL, VT, Val,
1395                         DAG.getValueType(ValVT));
1396       Val = DAG.getNode(ISD::TRUNCATE, DL, ValVT, Val);
1397       break;
1398     case CCValAssign::ZExt:
1399       Val = DAG.getNode(ISD::AssertZext, DL, VT, Val,
1400                         DAG.getValueType(ValVT));
1401       Val = DAG.getNode(ISD::TRUNCATE, DL, ValVT, Val);
1402       break;
1403     case CCValAssign::AExt:
1404       Val = DAG.getNode(ISD::TRUNCATE, DL, ValVT, Val);
1405       break;
1406     default:
1407       llvm_unreachable("Unknown loc info!");
1408     }
1409 
1410     if (IsShader && Arg.VT.isVector()) {
1411       // Build a vector from the registers
1412       Type *ParamType = FType->getParamType(Arg.getOrigArgIndex());
1413       unsigned NumElements = ParamType->getVectorNumElements();
1414 
1415       SmallVector<SDValue, 4> Regs;
1416       Regs.push_back(Val);
1417       for (unsigned j = 1; j != NumElements; ++j) {
1418         Reg = ArgLocs[ArgIdx++].getLocReg();
1419         Reg = MF.addLiveIn(Reg, RC);
1420 
1421         SDValue Copy = DAG.getCopyFromReg(Chain, DL, Reg, VT);
1422         Regs.push_back(Copy);
1423       }
1424 
1425       // Fill up the missing vector elements
1426       NumElements = Arg.VT.getVectorNumElements() - NumElements;
1427       Regs.append(NumElements, DAG.getUNDEF(VT));
1428 
1429       InVals.push_back(DAG.getBuildVector(Arg.VT, DL, Regs));
1430       continue;
1431     }
1432 
1433     InVals.push_back(Val);
1434   }
1435 
1436   // Start adding system SGPRs.
1437   if (IsEntryFunc) {
1438     allocateSystemSGPRs(CCInfo, MF, *Info, CallConv, IsShader);
1439   } else {
1440     CCInfo.AllocateReg(Info->getScratchRSrcReg());
1441     CCInfo.AllocateReg(Info->getScratchWaveOffsetReg());
1442     CCInfo.AllocateReg(Info->getFrameOffsetReg());
1443   }
1444 
1445   return Chains.empty() ? Chain :
1446     DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
1447 }
1448 
1449 // TODO: If return values can't fit in registers, we should return as many as
1450 // possible in registers before passing on stack.
1451 bool SITargetLowering::CanLowerReturn(
1452   CallingConv::ID CallConv,
1453   MachineFunction &MF, bool IsVarArg,
1454   const SmallVectorImpl<ISD::OutputArg> &Outs,
1455   LLVMContext &Context) const {
1456   // Replacing returns with sret/stack usage doesn't make sense for shaders.
1457   // FIXME: Also sort of a workaround for custom vector splitting in LowerReturn
1458   // for shaders. Vector types should be explicitly handled by CC.
1459   if (AMDGPU::isEntryFunctionCC(CallConv))
1460     return true;
1461 
1462   SmallVector<CCValAssign, 16> RVLocs;
1463   CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context);
1464   return CCInfo.CheckReturn(Outs, CCAssignFnForReturn(CallConv, IsVarArg));
1465 }
1466 
1467 SDValue
1468 SITargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv,
1469                               bool isVarArg,
1470                               const SmallVectorImpl<ISD::OutputArg> &Outs,
1471                               const SmallVectorImpl<SDValue> &OutVals,
1472                               const SDLoc &DL, SelectionDAG &DAG) const {
1473   MachineFunction &MF = DAG.getMachineFunction();
1474   SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
1475 
1476   if (AMDGPU::isKernel(CallConv)) {
1477     return AMDGPUTargetLowering::LowerReturn(Chain, CallConv, isVarArg, Outs,
1478                                              OutVals, DL, DAG);
1479   }
1480 
1481   bool IsShader = AMDGPU::isShader(CallConv);
1482 
1483   Info->setIfReturnsVoid(Outs.size() == 0);
1484   bool IsWaveEnd = Info->returnsVoid() && IsShader;
1485 
1486   SmallVector<ISD::OutputArg, 48> Splits;
1487   SmallVector<SDValue, 48> SplitVals;
1488 
1489   // Split vectors into their elements.
1490   for (unsigned i = 0, e = Outs.size(); i != e; ++i) {
1491     const ISD::OutputArg &Out = Outs[i];
1492 
1493     if (IsShader && Out.VT.isVector()) {
1494       MVT VT = Out.VT.getVectorElementType();
1495       ISD::OutputArg NewOut = Out;
1496       NewOut.Flags.setSplit();
1497       NewOut.VT = VT;
1498 
1499       // We want the original number of vector elements here, e.g.
1500       // three or five, not four or eight.
1501       unsigned NumElements = Out.ArgVT.getVectorNumElements();
1502 
1503       for (unsigned j = 0; j != NumElements; ++j) {
1504         SDValue Elem = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, OutVals[i],
1505                                    DAG.getConstant(j, DL, MVT::i32));
1506         SplitVals.push_back(Elem);
1507         Splits.push_back(NewOut);
1508         NewOut.PartOffset += NewOut.VT.getStoreSize();
1509       }
1510     } else {
1511       SplitVals.push_back(OutVals[i]);
1512       Splits.push_back(Out);
1513     }
1514   }
1515 
1516   // CCValAssign - represent the assignment of the return value to a location.
1517   SmallVector<CCValAssign, 48> RVLocs;
1518 
1519   // CCState - Info about the registers and stack slots.
1520   CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), RVLocs,
1521                  *DAG.getContext());
1522 
1523   // Analyze outgoing return values.
1524   CCInfo.AnalyzeReturn(Splits, CCAssignFnForReturn(CallConv, isVarArg));
1525 
1526   SDValue Flag;
1527   SmallVector<SDValue, 48> RetOps;
1528   RetOps.push_back(Chain); // Operand #0 = Chain (updated below)
1529 
1530   // Add return address for callable functions.
1531   if (!Info->isEntryFunction()) {
1532     const SIRegisterInfo *TRI = getSubtarget()->getRegisterInfo();
1533     SDValue ReturnAddrReg = CreateLiveInRegister(
1534       DAG, &AMDGPU::SReg_64RegClass, TRI->getReturnAddressReg(MF), MVT::i64);
1535 
1536     // FIXME: Should be able to use a vreg here, but need a way to prevent it
1537     // from being allcoated to a CSR.
1538 
1539     SDValue PhysReturnAddrReg = DAG.getRegister(TRI->getReturnAddressReg(MF),
1540                                                 MVT::i64);
1541 
1542     Chain = DAG.getCopyToReg(Chain, DL, PhysReturnAddrReg, ReturnAddrReg, Flag);
1543     Flag = Chain.getValue(1);
1544 
1545     RetOps.push_back(PhysReturnAddrReg);
1546   }
1547 
1548   // Copy the result values into the output registers.
1549   for (unsigned i = 0, realRVLocIdx = 0;
1550        i != RVLocs.size();
1551        ++i, ++realRVLocIdx) {
1552     CCValAssign &VA = RVLocs[i];
1553     assert(VA.isRegLoc() && "Can only return in registers!");
1554     // TODO: Partially return in registers if return values don't fit.
1555 
1556     SDValue Arg = SplitVals[realRVLocIdx];
1557 
1558     // Copied from other backends.
1559     switch (VA.getLocInfo()) {
1560     case CCValAssign::Full:
1561       break;
1562     case CCValAssign::BCvt:
1563       Arg = DAG.getNode(ISD::BITCAST, DL, VA.getLocVT(), Arg);
1564       break;
1565     case CCValAssign::SExt:
1566       Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Arg);
1567       break;
1568     case CCValAssign::ZExt:
1569       Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Arg);
1570       break;
1571     case CCValAssign::AExt:
1572       Arg = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Arg);
1573       break;
1574     default:
1575       llvm_unreachable("Unknown loc info!");
1576     }
1577 
1578     Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), Arg, Flag);
1579     Flag = Chain.getValue(1);
1580     RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
1581   }
1582 
1583   // FIXME: Does sret work properly?
1584 
1585   // Update chain and glue.
1586   RetOps[0] = Chain;
1587   if (Flag.getNode())
1588     RetOps.push_back(Flag);
1589 
1590   unsigned Opc = AMDGPUISD::ENDPGM;
1591   if (!IsWaveEnd)
1592     Opc = IsShader ? AMDGPUISD::RETURN_TO_EPILOG : AMDGPUISD::RET_FLAG;
1593   return DAG.getNode(Opc, DL, MVT::Other, RetOps);
1594 }
1595 
1596 unsigned SITargetLowering::getRegisterByName(const char* RegName, EVT VT,
1597                                              SelectionDAG &DAG) const {
1598   unsigned Reg = StringSwitch<unsigned>(RegName)
1599     .Case("m0", AMDGPU::M0)
1600     .Case("exec", AMDGPU::EXEC)
1601     .Case("exec_lo", AMDGPU::EXEC_LO)
1602     .Case("exec_hi", AMDGPU::EXEC_HI)
1603     .Case("flat_scratch", AMDGPU::FLAT_SCR)
1604     .Case("flat_scratch_lo", AMDGPU::FLAT_SCR_LO)
1605     .Case("flat_scratch_hi", AMDGPU::FLAT_SCR_HI)
1606     .Default(AMDGPU::NoRegister);
1607 
1608   if (Reg == AMDGPU::NoRegister) {
1609     report_fatal_error(Twine("invalid register name \""
1610                              + StringRef(RegName)  + "\"."));
1611 
1612   }
1613 
1614   if (Subtarget->getGeneration() == SISubtarget::SOUTHERN_ISLANDS &&
1615       Subtarget->getRegisterInfo()->regsOverlap(Reg, AMDGPU::FLAT_SCR)) {
1616     report_fatal_error(Twine("invalid register \""
1617                              + StringRef(RegName)  + "\" for subtarget."));
1618   }
1619 
1620   switch (Reg) {
1621   case AMDGPU::M0:
1622   case AMDGPU::EXEC_LO:
1623   case AMDGPU::EXEC_HI:
1624   case AMDGPU::FLAT_SCR_LO:
1625   case AMDGPU::FLAT_SCR_HI:
1626     if (VT.getSizeInBits() == 32)
1627       return Reg;
1628     break;
1629   case AMDGPU::EXEC:
1630   case AMDGPU::FLAT_SCR:
1631     if (VT.getSizeInBits() == 64)
1632       return Reg;
1633     break;
1634   default:
1635     llvm_unreachable("missing register type checking");
1636   }
1637 
1638   report_fatal_error(Twine("invalid type for register \""
1639                            + StringRef(RegName) + "\"."));
1640 }
1641 
1642 // If kill is not the last instruction, split the block so kill is always a
1643 // proper terminator.
1644 MachineBasicBlock *SITargetLowering::splitKillBlock(MachineInstr &MI,
1645                                                     MachineBasicBlock *BB) const {
1646   const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
1647 
1648   MachineBasicBlock::iterator SplitPoint(&MI);
1649   ++SplitPoint;
1650 
1651   if (SplitPoint == BB->end()) {
1652     // Don't bother with a new block.
1653     MI.setDesc(TII->get(AMDGPU::SI_KILL_TERMINATOR));
1654     return BB;
1655   }
1656 
1657   MachineFunction *MF = BB->getParent();
1658   MachineBasicBlock *SplitBB
1659     = MF->CreateMachineBasicBlock(BB->getBasicBlock());
1660 
1661   MF->insert(++MachineFunction::iterator(BB), SplitBB);
1662   SplitBB->splice(SplitBB->begin(), BB, SplitPoint, BB->end());
1663 
1664   SplitBB->transferSuccessorsAndUpdatePHIs(BB);
1665   BB->addSuccessor(SplitBB);
1666 
1667   MI.setDesc(TII->get(AMDGPU::SI_KILL_TERMINATOR));
1668   return SplitBB;
1669 }
1670 
1671 // Do a v_movrels_b32 or v_movreld_b32 for each unique value of \p IdxReg in the
1672 // wavefront. If the value is uniform and just happens to be in a VGPR, this
1673 // will only do one iteration. In the worst case, this will loop 64 times.
1674 //
1675 // TODO: Just use v_readlane_b32 if we know the VGPR has a uniform value.
1676 static MachineBasicBlock::iterator emitLoadM0FromVGPRLoop(
1677   const SIInstrInfo *TII,
1678   MachineRegisterInfo &MRI,
1679   MachineBasicBlock &OrigBB,
1680   MachineBasicBlock &LoopBB,
1681   const DebugLoc &DL,
1682   const MachineOperand &IdxReg,
1683   unsigned InitReg,
1684   unsigned ResultReg,
1685   unsigned PhiReg,
1686   unsigned InitSaveExecReg,
1687   int Offset,
1688   bool UseGPRIdxMode) {
1689   MachineBasicBlock::iterator I = LoopBB.begin();
1690 
1691   unsigned PhiExec = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
1692   unsigned NewExec = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
1693   unsigned CurrentIdxReg = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
1694   unsigned CondReg = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
1695 
1696   BuildMI(LoopBB, I, DL, TII->get(TargetOpcode::PHI), PhiReg)
1697     .addReg(InitReg)
1698     .addMBB(&OrigBB)
1699     .addReg(ResultReg)
1700     .addMBB(&LoopBB);
1701 
1702   BuildMI(LoopBB, I, DL, TII->get(TargetOpcode::PHI), PhiExec)
1703     .addReg(InitSaveExecReg)
1704     .addMBB(&OrigBB)
1705     .addReg(NewExec)
1706     .addMBB(&LoopBB);
1707 
1708   // Read the next variant <- also loop target.
1709   BuildMI(LoopBB, I, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), CurrentIdxReg)
1710     .addReg(IdxReg.getReg(), getUndefRegState(IdxReg.isUndef()));
1711 
1712   // Compare the just read M0 value to all possible Idx values.
1713   BuildMI(LoopBB, I, DL, TII->get(AMDGPU::V_CMP_EQ_U32_e64), CondReg)
1714     .addReg(CurrentIdxReg)
1715     .addReg(IdxReg.getReg(), 0, IdxReg.getSubReg());
1716 
1717   if (UseGPRIdxMode) {
1718     unsigned IdxReg;
1719     if (Offset == 0) {
1720       IdxReg = CurrentIdxReg;
1721     } else {
1722       IdxReg = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
1723       BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_ADD_I32), IdxReg)
1724         .addReg(CurrentIdxReg, RegState::Kill)
1725         .addImm(Offset);
1726     }
1727 
1728     MachineInstr *SetIdx =
1729       BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_SET_GPR_IDX_IDX))
1730       .addReg(IdxReg, RegState::Kill);
1731     SetIdx->getOperand(2).setIsUndef();
1732   } else {
1733     // Move index from VCC into M0
1734     if (Offset == 0) {
1735       BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_MOV_B32), AMDGPU::M0)
1736         .addReg(CurrentIdxReg, RegState::Kill);
1737     } else {
1738       BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_ADD_I32), AMDGPU::M0)
1739         .addReg(CurrentIdxReg, RegState::Kill)
1740         .addImm(Offset);
1741     }
1742   }
1743 
1744   // Update EXEC, save the original EXEC value to VCC.
1745   BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_AND_SAVEEXEC_B64), NewExec)
1746     .addReg(CondReg, RegState::Kill);
1747 
1748   MRI.setSimpleHint(NewExec, CondReg);
1749 
1750   // Update EXEC, switch all done bits to 0 and all todo bits to 1.
1751   MachineInstr *InsertPt =
1752     BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_XOR_B64), AMDGPU::EXEC)
1753     .addReg(AMDGPU::EXEC)
1754     .addReg(NewExec);
1755 
1756   // XXX - s_xor_b64 sets scc to 1 if the result is nonzero, so can we use
1757   // s_cbranch_scc0?
1758 
1759   // Loop back to V_READFIRSTLANE_B32 if there are still variants to cover.
1760   BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_CBRANCH_EXECNZ))
1761     .addMBB(&LoopBB);
1762 
1763   return InsertPt->getIterator();
1764 }
1765 
1766 // This has slightly sub-optimal regalloc when the source vector is killed by
1767 // the read. The register allocator does not understand that the kill is
1768 // per-workitem, so is kept alive for the whole loop so we end up not re-using a
1769 // subregister from it, using 1 more VGPR than necessary. This was saved when
1770 // this was expanded after register allocation.
1771 static MachineBasicBlock::iterator loadM0FromVGPR(const SIInstrInfo *TII,
1772                                                   MachineBasicBlock &MBB,
1773                                                   MachineInstr &MI,
1774                                                   unsigned InitResultReg,
1775                                                   unsigned PhiReg,
1776                                                   int Offset,
1777                                                   bool UseGPRIdxMode) {
1778   MachineFunction *MF = MBB.getParent();
1779   MachineRegisterInfo &MRI = MF->getRegInfo();
1780   const DebugLoc &DL = MI.getDebugLoc();
1781   MachineBasicBlock::iterator I(&MI);
1782 
1783   unsigned DstReg = MI.getOperand(0).getReg();
1784   unsigned SaveExec = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
1785   unsigned TmpExec = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
1786 
1787   BuildMI(MBB, I, DL, TII->get(TargetOpcode::IMPLICIT_DEF), TmpExec);
1788 
1789   // Save the EXEC mask
1790   BuildMI(MBB, I, DL, TII->get(AMDGPU::S_MOV_B64), SaveExec)
1791     .addReg(AMDGPU::EXEC);
1792 
1793   // To insert the loop we need to split the block. Move everything after this
1794   // point to a new block, and insert a new empty block between the two.
1795   MachineBasicBlock *LoopBB = MF->CreateMachineBasicBlock();
1796   MachineBasicBlock *RemainderBB = MF->CreateMachineBasicBlock();
1797   MachineFunction::iterator MBBI(MBB);
1798   ++MBBI;
1799 
1800   MF->insert(MBBI, LoopBB);
1801   MF->insert(MBBI, RemainderBB);
1802 
1803   LoopBB->addSuccessor(LoopBB);
1804   LoopBB->addSuccessor(RemainderBB);
1805 
1806   // Move the rest of the block into a new block.
1807   RemainderBB->transferSuccessorsAndUpdatePHIs(&MBB);
1808   RemainderBB->splice(RemainderBB->begin(), &MBB, I, MBB.end());
1809 
1810   MBB.addSuccessor(LoopBB);
1811 
1812   const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
1813 
1814   auto InsPt = emitLoadM0FromVGPRLoop(TII, MRI, MBB, *LoopBB, DL, *Idx,
1815                                       InitResultReg, DstReg, PhiReg, TmpExec,
1816                                       Offset, UseGPRIdxMode);
1817 
1818   MachineBasicBlock::iterator First = RemainderBB->begin();
1819   BuildMI(*RemainderBB, First, DL, TII->get(AMDGPU::S_MOV_B64), AMDGPU::EXEC)
1820     .addReg(SaveExec);
1821 
1822   return InsPt;
1823 }
1824 
1825 // Returns subreg index, offset
1826 static std::pair<unsigned, int>
1827 computeIndirectRegAndOffset(const SIRegisterInfo &TRI,
1828                             const TargetRegisterClass *SuperRC,
1829                             unsigned VecReg,
1830                             int Offset) {
1831   int NumElts = TRI.getRegSizeInBits(*SuperRC) / 32;
1832 
1833   // Skip out of bounds offsets, or else we would end up using an undefined
1834   // register.
1835   if (Offset >= NumElts || Offset < 0)
1836     return std::make_pair(AMDGPU::sub0, Offset);
1837 
1838   return std::make_pair(AMDGPU::sub0 + Offset, 0);
1839 }
1840 
1841 // Return true if the index is an SGPR and was set.
1842 static bool setM0ToIndexFromSGPR(const SIInstrInfo *TII,
1843                                  MachineRegisterInfo &MRI,
1844                                  MachineInstr &MI,
1845                                  int Offset,
1846                                  bool UseGPRIdxMode,
1847                                  bool IsIndirectSrc) {
1848   MachineBasicBlock *MBB = MI.getParent();
1849   const DebugLoc &DL = MI.getDebugLoc();
1850   MachineBasicBlock::iterator I(&MI);
1851 
1852   const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
1853   const TargetRegisterClass *IdxRC = MRI.getRegClass(Idx->getReg());
1854 
1855   assert(Idx->getReg() != AMDGPU::NoRegister);
1856 
1857   if (!TII->getRegisterInfo().isSGPRClass(IdxRC))
1858     return false;
1859 
1860   if (UseGPRIdxMode) {
1861     unsigned IdxMode = IsIndirectSrc ?
1862       VGPRIndexMode::SRC0_ENABLE : VGPRIndexMode::DST_ENABLE;
1863     if (Offset == 0) {
1864       MachineInstr *SetOn =
1865           BuildMI(*MBB, I, DL, TII->get(AMDGPU::S_SET_GPR_IDX_ON))
1866               .add(*Idx)
1867               .addImm(IdxMode);
1868 
1869       SetOn->getOperand(3).setIsUndef();
1870     } else {
1871       unsigned Tmp = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
1872       BuildMI(*MBB, I, DL, TII->get(AMDGPU::S_ADD_I32), Tmp)
1873           .add(*Idx)
1874           .addImm(Offset);
1875       MachineInstr *SetOn =
1876         BuildMI(*MBB, I, DL, TII->get(AMDGPU::S_SET_GPR_IDX_ON))
1877         .addReg(Tmp, RegState::Kill)
1878         .addImm(IdxMode);
1879 
1880       SetOn->getOperand(3).setIsUndef();
1881     }
1882 
1883     return true;
1884   }
1885 
1886   if (Offset == 0) {
1887     BuildMI(*MBB, I, DL, TII->get(AMDGPU::S_MOV_B32), AMDGPU::M0)
1888       .add(*Idx);
1889   } else {
1890     BuildMI(*MBB, I, DL, TII->get(AMDGPU::S_ADD_I32), AMDGPU::M0)
1891       .add(*Idx)
1892       .addImm(Offset);
1893   }
1894 
1895   return true;
1896 }
1897 
1898 // Control flow needs to be inserted if indexing with a VGPR.
1899 static MachineBasicBlock *emitIndirectSrc(MachineInstr &MI,
1900                                           MachineBasicBlock &MBB,
1901                                           const SISubtarget &ST) {
1902   const SIInstrInfo *TII = ST.getInstrInfo();
1903   const SIRegisterInfo &TRI = TII->getRegisterInfo();
1904   MachineFunction *MF = MBB.getParent();
1905   MachineRegisterInfo &MRI = MF->getRegInfo();
1906 
1907   unsigned Dst = MI.getOperand(0).getReg();
1908   unsigned SrcReg = TII->getNamedOperand(MI, AMDGPU::OpName::src)->getReg();
1909   int Offset = TII->getNamedOperand(MI, AMDGPU::OpName::offset)->getImm();
1910 
1911   const TargetRegisterClass *VecRC = MRI.getRegClass(SrcReg);
1912 
1913   unsigned SubReg;
1914   std::tie(SubReg, Offset)
1915     = computeIndirectRegAndOffset(TRI, VecRC, SrcReg, Offset);
1916 
1917   bool UseGPRIdxMode = ST.useVGPRIndexMode(EnableVGPRIndexMode);
1918 
1919   if (setM0ToIndexFromSGPR(TII, MRI, MI, Offset, UseGPRIdxMode, true)) {
1920     MachineBasicBlock::iterator I(&MI);
1921     const DebugLoc &DL = MI.getDebugLoc();
1922 
1923     if (UseGPRIdxMode) {
1924       // TODO: Look at the uses to avoid the copy. This may require rescheduling
1925       // to avoid interfering with other uses, so probably requires a new
1926       // optimization pass.
1927       BuildMI(MBB, I, DL, TII->get(AMDGPU::V_MOV_B32_e32), Dst)
1928         .addReg(SrcReg, RegState::Undef, SubReg)
1929         .addReg(SrcReg, RegState::Implicit)
1930         .addReg(AMDGPU::M0, RegState::Implicit);
1931       BuildMI(MBB, I, DL, TII->get(AMDGPU::S_SET_GPR_IDX_OFF));
1932     } else {
1933       BuildMI(MBB, I, DL, TII->get(AMDGPU::V_MOVRELS_B32_e32), Dst)
1934         .addReg(SrcReg, RegState::Undef, SubReg)
1935         .addReg(SrcReg, RegState::Implicit);
1936     }
1937 
1938     MI.eraseFromParent();
1939 
1940     return &MBB;
1941   }
1942 
1943   const DebugLoc &DL = MI.getDebugLoc();
1944   MachineBasicBlock::iterator I(&MI);
1945 
1946   unsigned PhiReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1947   unsigned InitReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1948 
1949   BuildMI(MBB, I, DL, TII->get(TargetOpcode::IMPLICIT_DEF), InitReg);
1950 
1951   if (UseGPRIdxMode) {
1952     MachineInstr *SetOn = BuildMI(MBB, I, DL, TII->get(AMDGPU::S_SET_GPR_IDX_ON))
1953       .addImm(0) // Reset inside loop.
1954       .addImm(VGPRIndexMode::SRC0_ENABLE);
1955     SetOn->getOperand(3).setIsUndef();
1956 
1957     // Disable again after the loop.
1958     BuildMI(MBB, std::next(I), DL, TII->get(AMDGPU::S_SET_GPR_IDX_OFF));
1959   }
1960 
1961   auto InsPt = loadM0FromVGPR(TII, MBB, MI, InitReg, PhiReg, Offset, UseGPRIdxMode);
1962   MachineBasicBlock *LoopBB = InsPt->getParent();
1963 
1964   if (UseGPRIdxMode) {
1965     BuildMI(*LoopBB, InsPt, DL, TII->get(AMDGPU::V_MOV_B32_e32), Dst)
1966       .addReg(SrcReg, RegState::Undef, SubReg)
1967       .addReg(SrcReg, RegState::Implicit)
1968       .addReg(AMDGPU::M0, RegState::Implicit);
1969   } else {
1970     BuildMI(*LoopBB, InsPt, DL, TII->get(AMDGPU::V_MOVRELS_B32_e32), Dst)
1971       .addReg(SrcReg, RegState::Undef, SubReg)
1972       .addReg(SrcReg, RegState::Implicit);
1973   }
1974 
1975   MI.eraseFromParent();
1976 
1977   return LoopBB;
1978 }
1979 
1980 static unsigned getMOVRELDPseudo(const SIRegisterInfo &TRI,
1981                                  const TargetRegisterClass *VecRC) {
1982   switch (TRI.getRegSizeInBits(*VecRC)) {
1983   case 32: // 4 bytes
1984     return AMDGPU::V_MOVRELD_B32_V1;
1985   case 64: // 8 bytes
1986     return AMDGPU::V_MOVRELD_B32_V2;
1987   case 128: // 16 bytes
1988     return AMDGPU::V_MOVRELD_B32_V4;
1989   case 256: // 32 bytes
1990     return AMDGPU::V_MOVRELD_B32_V8;
1991   case 512: // 64 bytes
1992     return AMDGPU::V_MOVRELD_B32_V16;
1993   default:
1994     llvm_unreachable("unsupported size for MOVRELD pseudos");
1995   }
1996 }
1997 
1998 static MachineBasicBlock *emitIndirectDst(MachineInstr &MI,
1999                                           MachineBasicBlock &MBB,
2000                                           const SISubtarget &ST) {
2001   const SIInstrInfo *TII = ST.getInstrInfo();
2002   const SIRegisterInfo &TRI = TII->getRegisterInfo();
2003   MachineFunction *MF = MBB.getParent();
2004   MachineRegisterInfo &MRI = MF->getRegInfo();
2005 
2006   unsigned Dst = MI.getOperand(0).getReg();
2007   const MachineOperand *SrcVec = TII->getNamedOperand(MI, AMDGPU::OpName::src);
2008   const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
2009   const MachineOperand *Val = TII->getNamedOperand(MI, AMDGPU::OpName::val);
2010   int Offset = TII->getNamedOperand(MI, AMDGPU::OpName::offset)->getImm();
2011   const TargetRegisterClass *VecRC = MRI.getRegClass(SrcVec->getReg());
2012 
2013   // This can be an immediate, but will be folded later.
2014   assert(Val->getReg());
2015 
2016   unsigned SubReg;
2017   std::tie(SubReg, Offset) = computeIndirectRegAndOffset(TRI, VecRC,
2018                                                          SrcVec->getReg(),
2019                                                          Offset);
2020   bool UseGPRIdxMode = ST.useVGPRIndexMode(EnableVGPRIndexMode);
2021 
2022   if (Idx->getReg() == AMDGPU::NoRegister) {
2023     MachineBasicBlock::iterator I(&MI);
2024     const DebugLoc &DL = MI.getDebugLoc();
2025 
2026     assert(Offset == 0);
2027 
2028     BuildMI(MBB, I, DL, TII->get(TargetOpcode::INSERT_SUBREG), Dst)
2029         .add(*SrcVec)
2030         .add(*Val)
2031         .addImm(SubReg);
2032 
2033     MI.eraseFromParent();
2034     return &MBB;
2035   }
2036 
2037   if (setM0ToIndexFromSGPR(TII, MRI, MI, Offset, UseGPRIdxMode, false)) {
2038     MachineBasicBlock::iterator I(&MI);
2039     const DebugLoc &DL = MI.getDebugLoc();
2040 
2041     if (UseGPRIdxMode) {
2042       BuildMI(MBB, I, DL, TII->get(AMDGPU::V_MOV_B32_indirect))
2043           .addReg(SrcVec->getReg(), RegState::Undef, SubReg) // vdst
2044           .add(*Val)
2045           .addReg(Dst, RegState::ImplicitDefine)
2046           .addReg(SrcVec->getReg(), RegState::Implicit)
2047           .addReg(AMDGPU::M0, RegState::Implicit);
2048 
2049       BuildMI(MBB, I, DL, TII->get(AMDGPU::S_SET_GPR_IDX_OFF));
2050     } else {
2051       const MCInstrDesc &MovRelDesc = TII->get(getMOVRELDPseudo(TRI, VecRC));
2052 
2053       BuildMI(MBB, I, DL, MovRelDesc)
2054           .addReg(Dst, RegState::Define)
2055           .addReg(SrcVec->getReg())
2056           .add(*Val)
2057           .addImm(SubReg - AMDGPU::sub0);
2058     }
2059 
2060     MI.eraseFromParent();
2061     return &MBB;
2062   }
2063 
2064   if (Val->isReg())
2065     MRI.clearKillFlags(Val->getReg());
2066 
2067   const DebugLoc &DL = MI.getDebugLoc();
2068 
2069   if (UseGPRIdxMode) {
2070     MachineBasicBlock::iterator I(&MI);
2071 
2072     MachineInstr *SetOn = BuildMI(MBB, I, DL, TII->get(AMDGPU::S_SET_GPR_IDX_ON))
2073       .addImm(0) // Reset inside loop.
2074       .addImm(VGPRIndexMode::DST_ENABLE);
2075     SetOn->getOperand(3).setIsUndef();
2076 
2077     // Disable again after the loop.
2078     BuildMI(MBB, std::next(I), DL, TII->get(AMDGPU::S_SET_GPR_IDX_OFF));
2079   }
2080 
2081   unsigned PhiReg = MRI.createVirtualRegister(VecRC);
2082 
2083   auto InsPt = loadM0FromVGPR(TII, MBB, MI, SrcVec->getReg(), PhiReg,
2084                               Offset, UseGPRIdxMode);
2085   MachineBasicBlock *LoopBB = InsPt->getParent();
2086 
2087   if (UseGPRIdxMode) {
2088     BuildMI(*LoopBB, InsPt, DL, TII->get(AMDGPU::V_MOV_B32_indirect))
2089         .addReg(PhiReg, RegState::Undef, SubReg) // vdst
2090         .add(*Val)                               // src0
2091         .addReg(Dst, RegState::ImplicitDefine)
2092         .addReg(PhiReg, RegState::Implicit)
2093         .addReg(AMDGPU::M0, RegState::Implicit);
2094   } else {
2095     const MCInstrDesc &MovRelDesc = TII->get(getMOVRELDPseudo(TRI, VecRC));
2096 
2097     BuildMI(*LoopBB, InsPt, DL, MovRelDesc)
2098         .addReg(Dst, RegState::Define)
2099         .addReg(PhiReg)
2100         .add(*Val)
2101         .addImm(SubReg - AMDGPU::sub0);
2102   }
2103 
2104   MI.eraseFromParent();
2105 
2106   return LoopBB;
2107 }
2108 
2109 MachineBasicBlock *SITargetLowering::EmitInstrWithCustomInserter(
2110   MachineInstr &MI, MachineBasicBlock *BB) const {
2111 
2112   const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
2113   MachineFunction *MF = BB->getParent();
2114   SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
2115 
2116   if (TII->isMIMG(MI)) {
2117       if (!MI.memoperands_empty())
2118         return BB;
2119     // Add a memoperand for mimg instructions so that they aren't assumed to
2120     // be ordered memory instuctions.
2121 
2122     MachinePointerInfo PtrInfo(MFI->getImagePSV());
2123     MachineMemOperand::Flags Flags = MachineMemOperand::MODereferenceable;
2124     if (MI.mayStore())
2125       Flags |= MachineMemOperand::MOStore;
2126 
2127     if (MI.mayLoad())
2128       Flags |= MachineMemOperand::MOLoad;
2129 
2130     auto MMO = MF->getMachineMemOperand(PtrInfo, Flags, 0, 0);
2131     MI.addMemOperand(*MF, MMO);
2132     return BB;
2133   }
2134 
2135   switch (MI.getOpcode()) {
2136   case AMDGPU::SI_INIT_M0:
2137     BuildMI(*BB, MI.getIterator(), MI.getDebugLoc(),
2138             TII->get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2139         .add(MI.getOperand(0));
2140     MI.eraseFromParent();
2141     return BB;
2142 
2143   case AMDGPU::SI_INIT_EXEC:
2144     // This should be before all vector instructions.
2145     BuildMI(*BB, &*BB->begin(), MI.getDebugLoc(), TII->get(AMDGPU::S_MOV_B64),
2146             AMDGPU::EXEC)
2147         .addImm(MI.getOperand(0).getImm());
2148     MI.eraseFromParent();
2149     return BB;
2150 
2151   case AMDGPU::SI_INIT_EXEC_FROM_INPUT: {
2152     // Extract the thread count from an SGPR input and set EXEC accordingly.
2153     // Since BFM can't shift by 64, handle that case with CMP + CMOV.
2154     //
2155     // S_BFE_U32 count, input, {shift, 7}
2156     // S_BFM_B64 exec, count, 0
2157     // S_CMP_EQ_U32 count, 64
2158     // S_CMOV_B64 exec, -1
2159     MachineInstr *FirstMI = &*BB->begin();
2160     MachineRegisterInfo &MRI = MF->getRegInfo();
2161     unsigned InputReg = MI.getOperand(0).getReg();
2162     unsigned CountReg = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
2163     bool Found = false;
2164 
2165     // Move the COPY of the input reg to the beginning, so that we can use it.
2166     for (auto I = BB->begin(); I != &MI; I++) {
2167       if (I->getOpcode() != TargetOpcode::COPY ||
2168           I->getOperand(0).getReg() != InputReg)
2169         continue;
2170 
2171       if (I == FirstMI) {
2172         FirstMI = &*++BB->begin();
2173       } else {
2174         I->removeFromParent();
2175         BB->insert(FirstMI, &*I);
2176       }
2177       Found = true;
2178       break;
2179     }
2180     assert(Found);
2181     (void)Found;
2182 
2183     // This should be before all vector instructions.
2184     BuildMI(*BB, FirstMI, DebugLoc(), TII->get(AMDGPU::S_BFE_U32), CountReg)
2185         .addReg(InputReg)
2186         .addImm((MI.getOperand(1).getImm() & 0x7f) | 0x70000);
2187     BuildMI(*BB, FirstMI, DebugLoc(), TII->get(AMDGPU::S_BFM_B64),
2188             AMDGPU::EXEC)
2189         .addReg(CountReg)
2190         .addImm(0);
2191     BuildMI(*BB, FirstMI, DebugLoc(), TII->get(AMDGPU::S_CMP_EQ_U32))
2192         .addReg(CountReg, RegState::Kill)
2193         .addImm(64);
2194     BuildMI(*BB, FirstMI, DebugLoc(), TII->get(AMDGPU::S_CMOV_B64),
2195             AMDGPU::EXEC)
2196         .addImm(-1);
2197     MI.eraseFromParent();
2198     return BB;
2199   }
2200 
2201   case AMDGPU::GET_GROUPSTATICSIZE: {
2202     DebugLoc DL = MI.getDebugLoc();
2203     BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_MOV_B32))
2204         .add(MI.getOperand(0))
2205         .addImm(MFI->getLDSSize());
2206     MI.eraseFromParent();
2207     return BB;
2208   }
2209   case AMDGPU::SI_INDIRECT_SRC_V1:
2210   case AMDGPU::SI_INDIRECT_SRC_V2:
2211   case AMDGPU::SI_INDIRECT_SRC_V4:
2212   case AMDGPU::SI_INDIRECT_SRC_V8:
2213   case AMDGPU::SI_INDIRECT_SRC_V16:
2214     return emitIndirectSrc(MI, *BB, *getSubtarget());
2215   case AMDGPU::SI_INDIRECT_DST_V1:
2216   case AMDGPU::SI_INDIRECT_DST_V2:
2217   case AMDGPU::SI_INDIRECT_DST_V4:
2218   case AMDGPU::SI_INDIRECT_DST_V8:
2219   case AMDGPU::SI_INDIRECT_DST_V16:
2220     return emitIndirectDst(MI, *BB, *getSubtarget());
2221   case AMDGPU::SI_KILL:
2222     return splitKillBlock(MI, BB);
2223   case AMDGPU::V_CNDMASK_B64_PSEUDO: {
2224     MachineRegisterInfo &MRI = BB->getParent()->getRegInfo();
2225 
2226     unsigned Dst = MI.getOperand(0).getReg();
2227     unsigned Src0 = MI.getOperand(1).getReg();
2228     unsigned Src1 = MI.getOperand(2).getReg();
2229     const DebugLoc &DL = MI.getDebugLoc();
2230     unsigned SrcCond = MI.getOperand(3).getReg();
2231 
2232     unsigned DstLo = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
2233     unsigned DstHi = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
2234 
2235     BuildMI(*BB, MI, DL, TII->get(AMDGPU::V_CNDMASK_B32_e64), DstLo)
2236       .addReg(Src0, 0, AMDGPU::sub0)
2237       .addReg(Src1, 0, AMDGPU::sub0)
2238       .addReg(SrcCond);
2239     BuildMI(*BB, MI, DL, TII->get(AMDGPU::V_CNDMASK_B32_e64), DstHi)
2240       .addReg(Src0, 0, AMDGPU::sub1)
2241       .addReg(Src1, 0, AMDGPU::sub1)
2242       .addReg(SrcCond);
2243 
2244     BuildMI(*BB, MI, DL, TII->get(AMDGPU::REG_SEQUENCE), Dst)
2245       .addReg(DstLo)
2246       .addImm(AMDGPU::sub0)
2247       .addReg(DstHi)
2248       .addImm(AMDGPU::sub1);
2249     MI.eraseFromParent();
2250     return BB;
2251   }
2252   case AMDGPU::SI_BR_UNDEF: {
2253     const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
2254     const DebugLoc &DL = MI.getDebugLoc();
2255     MachineInstr *Br = BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_CBRANCH_SCC1))
2256                            .add(MI.getOperand(0));
2257     Br->getOperand(1).setIsUndef(true); // read undef SCC
2258     MI.eraseFromParent();
2259     return BB;
2260   }
2261   default:
2262     return AMDGPUTargetLowering::EmitInstrWithCustomInserter(MI, BB);
2263   }
2264 }
2265 
2266 bool SITargetLowering::enableAggressiveFMAFusion(EVT VT) const {
2267   // This currently forces unfolding various combinations of fsub into fma with
2268   // free fneg'd operands. As long as we have fast FMA (controlled by
2269   // isFMAFasterThanFMulAndFAdd), we should perform these.
2270 
2271   // When fma is quarter rate, for f64 where add / sub are at best half rate,
2272   // most of these combines appear to be cycle neutral but save on instruction
2273   // count / code size.
2274   return true;
2275 }
2276 
2277 EVT SITargetLowering::getSetCCResultType(const DataLayout &DL, LLVMContext &Ctx,
2278                                          EVT VT) const {
2279   if (!VT.isVector()) {
2280     return MVT::i1;
2281   }
2282   return EVT::getVectorVT(Ctx, MVT::i1, VT.getVectorNumElements());
2283 }
2284 
2285 MVT SITargetLowering::getScalarShiftAmountTy(const DataLayout &, EVT VT) const {
2286   // TODO: Should i16 be used always if legal? For now it would force VALU
2287   // shifts.
2288   return (VT == MVT::i16) ? MVT::i16 : MVT::i32;
2289 }
2290 
2291 // Answering this is somewhat tricky and depends on the specific device which
2292 // have different rates for fma or all f64 operations.
2293 //
2294 // v_fma_f64 and v_mul_f64 always take the same number of cycles as each other
2295 // regardless of which device (although the number of cycles differs between
2296 // devices), so it is always profitable for f64.
2297 //
2298 // v_fma_f32 takes 4 or 16 cycles depending on the device, so it is profitable
2299 // only on full rate devices. Normally, we should prefer selecting v_mad_f32
2300 // which we can always do even without fused FP ops since it returns the same
2301 // result as the separate operations and since it is always full
2302 // rate. Therefore, we lie and report that it is not faster for f32. v_mad_f32
2303 // however does not support denormals, so we do report fma as faster if we have
2304 // a fast fma device and require denormals.
2305 //
2306 bool SITargetLowering::isFMAFasterThanFMulAndFAdd(EVT VT) const {
2307   VT = VT.getScalarType();
2308 
2309   switch (VT.getSimpleVT().SimpleTy) {
2310   case MVT::f32:
2311     // This is as fast on some subtargets. However, we always have full rate f32
2312     // mad available which returns the same result as the separate operations
2313     // which we should prefer over fma. We can't use this if we want to support
2314     // denormals, so only report this in these cases.
2315     return Subtarget->hasFP32Denormals() && Subtarget->hasFastFMAF32();
2316   case MVT::f64:
2317     return true;
2318   case MVT::f16:
2319     return Subtarget->has16BitInsts() && Subtarget->hasFP16Denormals();
2320   default:
2321     break;
2322   }
2323 
2324   return false;
2325 }
2326 
2327 //===----------------------------------------------------------------------===//
2328 // Custom DAG Lowering Operations
2329 //===----------------------------------------------------------------------===//
2330 
2331 SDValue SITargetLowering::LowerOperation(SDValue Op, SelectionDAG &DAG) const {
2332   switch (Op.getOpcode()) {
2333   default: return AMDGPUTargetLowering::LowerOperation(Op, DAG);
2334   case ISD::BRCOND: return LowerBRCOND(Op, DAG);
2335   case ISD::LOAD: {
2336     SDValue Result = LowerLOAD(Op, DAG);
2337     assert((!Result.getNode() ||
2338             Result.getNode()->getNumValues() == 2) &&
2339            "Load should return a value and a chain");
2340     return Result;
2341   }
2342 
2343   case ISD::FSIN:
2344   case ISD::FCOS:
2345     return LowerTrig(Op, DAG);
2346   case ISD::SELECT: return LowerSELECT(Op, DAG);
2347   case ISD::FDIV: return LowerFDIV(Op, DAG);
2348   case ISD::ATOMIC_CMP_SWAP: return LowerATOMIC_CMP_SWAP(Op, DAG);
2349   case ISD::STORE: return LowerSTORE(Op, DAG);
2350   case ISD::GlobalAddress: {
2351     MachineFunction &MF = DAG.getMachineFunction();
2352     SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
2353     return LowerGlobalAddress(MFI, Op, DAG);
2354   }
2355   case ISD::INTRINSIC_WO_CHAIN: return LowerINTRINSIC_WO_CHAIN(Op, DAG);
2356   case ISD::INTRINSIC_W_CHAIN: return LowerINTRINSIC_W_CHAIN(Op, DAG);
2357   case ISD::INTRINSIC_VOID: return LowerINTRINSIC_VOID(Op, DAG);
2358   case ISD::ADDRSPACECAST: return lowerADDRSPACECAST(Op, DAG);
2359   case ISD::INSERT_VECTOR_ELT:
2360     return lowerINSERT_VECTOR_ELT(Op, DAG);
2361   case ISD::EXTRACT_VECTOR_ELT:
2362     return lowerEXTRACT_VECTOR_ELT(Op, DAG);
2363   case ISD::FP_ROUND:
2364     return lowerFP_ROUND(Op, DAG);
2365 
2366   case ISD::TRAP:
2367   case ISD::DEBUGTRAP:
2368     return lowerTRAP(Op, DAG);
2369   }
2370   return SDValue();
2371 }
2372 
2373 void SITargetLowering::ReplaceNodeResults(SDNode *N,
2374                                           SmallVectorImpl<SDValue> &Results,
2375                                           SelectionDAG &DAG) const {
2376   switch (N->getOpcode()) {
2377   case ISD::INSERT_VECTOR_ELT: {
2378     if (SDValue Res = lowerINSERT_VECTOR_ELT(SDValue(N, 0), DAG))
2379       Results.push_back(Res);
2380     return;
2381   }
2382   case ISD::EXTRACT_VECTOR_ELT: {
2383     if (SDValue Res = lowerEXTRACT_VECTOR_ELT(SDValue(N, 0), DAG))
2384       Results.push_back(Res);
2385     return;
2386   }
2387   case ISD::INTRINSIC_WO_CHAIN: {
2388     unsigned IID = cast<ConstantSDNode>(N->getOperand(0))->getZExtValue();
2389     if (IID == Intrinsic::amdgcn_cvt_pkrtz) {
2390       SDValue Src0 = N->getOperand(1);
2391       SDValue Src1 = N->getOperand(2);
2392       SDLoc SL(N);
2393       SDValue Cvt = DAG.getNode(AMDGPUISD::CVT_PKRTZ_F16_F32, SL, MVT::i32,
2394                                 Src0, Src1);
2395       Results.push_back(DAG.getNode(ISD::BITCAST, SL, MVT::v2f16, Cvt));
2396       return;
2397     }
2398     break;
2399   }
2400   case ISD::SELECT: {
2401     SDLoc SL(N);
2402     EVT VT = N->getValueType(0);
2403     EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
2404     SDValue LHS = DAG.getNode(ISD::BITCAST, SL, NewVT, N->getOperand(1));
2405     SDValue RHS = DAG.getNode(ISD::BITCAST, SL, NewVT, N->getOperand(2));
2406 
2407     EVT SelectVT = NewVT;
2408     if (NewVT.bitsLT(MVT::i32)) {
2409       LHS = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, LHS);
2410       RHS = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, RHS);
2411       SelectVT = MVT::i32;
2412     }
2413 
2414     SDValue NewSelect = DAG.getNode(ISD::SELECT, SL, SelectVT,
2415                                     N->getOperand(0), LHS, RHS);
2416 
2417     if (NewVT != SelectVT)
2418       NewSelect = DAG.getNode(ISD::TRUNCATE, SL, NewVT, NewSelect);
2419     Results.push_back(DAG.getNode(ISD::BITCAST, SL, VT, NewSelect));
2420     return;
2421   }
2422   default:
2423     break;
2424   }
2425 }
2426 
2427 /// \brief Helper function for LowerBRCOND
2428 static SDNode *findUser(SDValue Value, unsigned Opcode) {
2429 
2430   SDNode *Parent = Value.getNode();
2431   for (SDNode::use_iterator I = Parent->use_begin(), E = Parent->use_end();
2432        I != E; ++I) {
2433 
2434     if (I.getUse().get() != Value)
2435       continue;
2436 
2437     if (I->getOpcode() == Opcode)
2438       return *I;
2439   }
2440   return nullptr;
2441 }
2442 
2443 unsigned SITargetLowering::isCFIntrinsic(const SDNode *Intr) const {
2444   if (Intr->getOpcode() == ISD::INTRINSIC_W_CHAIN) {
2445     switch (cast<ConstantSDNode>(Intr->getOperand(1))->getZExtValue()) {
2446     case Intrinsic::amdgcn_if:
2447       return AMDGPUISD::IF;
2448     case Intrinsic::amdgcn_else:
2449       return AMDGPUISD::ELSE;
2450     case Intrinsic::amdgcn_loop:
2451       return AMDGPUISD::LOOP;
2452     case Intrinsic::amdgcn_end_cf:
2453       llvm_unreachable("should not occur");
2454     default:
2455       return 0;
2456     }
2457   }
2458 
2459   // break, if_break, else_break are all only used as inputs to loop, not
2460   // directly as branch conditions.
2461   return 0;
2462 }
2463 
2464 void SITargetLowering::createDebuggerPrologueStackObjects(
2465     MachineFunction &MF) const {
2466   // Create stack objects that are used for emitting debugger prologue.
2467   //
2468   // Debugger prologue writes work group IDs and work item IDs to scratch memory
2469   // at fixed location in the following format:
2470   //   offset 0:  work group ID x
2471   //   offset 4:  work group ID y
2472   //   offset 8:  work group ID z
2473   //   offset 16: work item ID x
2474   //   offset 20: work item ID y
2475   //   offset 24: work item ID z
2476   SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
2477   int ObjectIdx = 0;
2478 
2479   // For each dimension:
2480   for (unsigned i = 0; i < 3; ++i) {
2481     // Create fixed stack object for work group ID.
2482     ObjectIdx = MF.getFrameInfo().CreateFixedObject(4, i * 4, true);
2483     Info->setDebuggerWorkGroupIDStackObjectIndex(i, ObjectIdx);
2484     // Create fixed stack object for work item ID.
2485     ObjectIdx = MF.getFrameInfo().CreateFixedObject(4, i * 4 + 16, true);
2486     Info->setDebuggerWorkItemIDStackObjectIndex(i, ObjectIdx);
2487   }
2488 }
2489 
2490 bool SITargetLowering::shouldEmitFixup(const GlobalValue *GV) const {
2491   const Triple &TT = getTargetMachine().getTargetTriple();
2492   return GV->getType()->getAddressSpace() == AMDGPUASI.CONSTANT_ADDRESS &&
2493          AMDGPU::shouldEmitConstantsToTextSection(TT);
2494 }
2495 
2496 bool SITargetLowering::shouldEmitGOTReloc(const GlobalValue *GV) const {
2497   return (GV->getType()->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS ||
2498               GV->getType()->getAddressSpace() == AMDGPUASI.CONSTANT_ADDRESS) &&
2499          !shouldEmitFixup(GV) &&
2500          !getTargetMachine().shouldAssumeDSOLocal(*GV->getParent(), GV);
2501 }
2502 
2503 bool SITargetLowering::shouldEmitPCReloc(const GlobalValue *GV) const {
2504   return !shouldEmitFixup(GV) && !shouldEmitGOTReloc(GV);
2505 }
2506 
2507 /// This transforms the control flow intrinsics to get the branch destination as
2508 /// last parameter, also switches branch target with BR if the need arise
2509 SDValue SITargetLowering::LowerBRCOND(SDValue BRCOND,
2510                                       SelectionDAG &DAG) const {
2511   SDLoc DL(BRCOND);
2512 
2513   SDNode *Intr = BRCOND.getOperand(1).getNode();
2514   SDValue Target = BRCOND.getOperand(2);
2515   SDNode *BR = nullptr;
2516   SDNode *SetCC = nullptr;
2517 
2518   if (Intr->getOpcode() == ISD::SETCC) {
2519     // As long as we negate the condition everything is fine
2520     SetCC = Intr;
2521     Intr = SetCC->getOperand(0).getNode();
2522 
2523   } else {
2524     // Get the target from BR if we don't negate the condition
2525     BR = findUser(BRCOND, ISD::BR);
2526     Target = BR->getOperand(1);
2527   }
2528 
2529   // FIXME: This changes the types of the intrinsics instead of introducing new
2530   // nodes with the correct types.
2531   // e.g. llvm.amdgcn.loop
2532 
2533   // eg: i1,ch = llvm.amdgcn.loop t0, TargetConstant:i32<6271>, t3
2534   // =>     t9: ch = llvm.amdgcn.loop t0, TargetConstant:i32<6271>, t3, BasicBlock:ch<bb1 0x7fee5286d088>
2535 
2536   unsigned CFNode = isCFIntrinsic(Intr);
2537   if (CFNode == 0) {
2538     // This is a uniform branch so we don't need to legalize.
2539     return BRCOND;
2540   }
2541 
2542   bool HaveChain = Intr->getOpcode() == ISD::INTRINSIC_VOID ||
2543                    Intr->getOpcode() == ISD::INTRINSIC_W_CHAIN;
2544 
2545   assert(!SetCC ||
2546         (SetCC->getConstantOperandVal(1) == 1 &&
2547          cast<CondCodeSDNode>(SetCC->getOperand(2).getNode())->get() ==
2548                                                              ISD::SETNE));
2549 
2550   // operands of the new intrinsic call
2551   SmallVector<SDValue, 4> Ops;
2552   if (HaveChain)
2553     Ops.push_back(BRCOND.getOperand(0));
2554 
2555   Ops.append(Intr->op_begin() + (HaveChain ?  2 : 1), Intr->op_end());
2556   Ops.push_back(Target);
2557 
2558   ArrayRef<EVT> Res(Intr->value_begin() + 1, Intr->value_end());
2559 
2560   // build the new intrinsic call
2561   SDNode *Result = DAG.getNode(CFNode, DL, DAG.getVTList(Res), Ops).getNode();
2562 
2563   if (!HaveChain) {
2564     SDValue Ops[] =  {
2565       SDValue(Result, 0),
2566       BRCOND.getOperand(0)
2567     };
2568 
2569     Result = DAG.getMergeValues(Ops, DL).getNode();
2570   }
2571 
2572   if (BR) {
2573     // Give the branch instruction our target
2574     SDValue Ops[] = {
2575       BR->getOperand(0),
2576       BRCOND.getOperand(2)
2577     };
2578     SDValue NewBR = DAG.getNode(ISD::BR, DL, BR->getVTList(), Ops);
2579     DAG.ReplaceAllUsesWith(BR, NewBR.getNode());
2580     BR = NewBR.getNode();
2581   }
2582 
2583   SDValue Chain = SDValue(Result, Result->getNumValues() - 1);
2584 
2585   // Copy the intrinsic results to registers
2586   for (unsigned i = 1, e = Intr->getNumValues() - 1; i != e; ++i) {
2587     SDNode *CopyToReg = findUser(SDValue(Intr, i), ISD::CopyToReg);
2588     if (!CopyToReg)
2589       continue;
2590 
2591     Chain = DAG.getCopyToReg(
2592       Chain, DL,
2593       CopyToReg->getOperand(1),
2594       SDValue(Result, i - 1),
2595       SDValue());
2596 
2597     DAG.ReplaceAllUsesWith(SDValue(CopyToReg, 0), CopyToReg->getOperand(0));
2598   }
2599 
2600   // Remove the old intrinsic from the chain
2601   DAG.ReplaceAllUsesOfValueWith(
2602     SDValue(Intr, Intr->getNumValues() - 1),
2603     Intr->getOperand(0));
2604 
2605   return Chain;
2606 }
2607 
2608 SDValue SITargetLowering::getFPExtOrFPTrunc(SelectionDAG &DAG,
2609                                             SDValue Op,
2610                                             const SDLoc &DL,
2611                                             EVT VT) const {
2612   return Op.getValueType().bitsLE(VT) ?
2613       DAG.getNode(ISD::FP_EXTEND, DL, VT, Op) :
2614       DAG.getNode(ISD::FTRUNC, DL, VT, Op);
2615 }
2616 
2617 SDValue SITargetLowering::lowerFP_ROUND(SDValue Op, SelectionDAG &DAG) const {
2618   assert(Op.getValueType() == MVT::f16 &&
2619          "Do not know how to custom lower FP_ROUND for non-f16 type");
2620 
2621   SDValue Src = Op.getOperand(0);
2622   EVT SrcVT = Src.getValueType();
2623   if (SrcVT != MVT::f64)
2624     return Op;
2625 
2626   SDLoc DL(Op);
2627 
2628   SDValue FpToFp16 = DAG.getNode(ISD::FP_TO_FP16, DL, MVT::i32, Src);
2629   SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, FpToFp16);
2630   return DAG.getNode(ISD::BITCAST, DL, MVT::f16, Trunc);
2631 }
2632 
2633 SDValue SITargetLowering::lowerTRAP(SDValue Op, SelectionDAG &DAG) const {
2634   SDLoc SL(Op);
2635   MachineFunction &MF = DAG.getMachineFunction();
2636   SDValue Chain = Op.getOperand(0);
2637 
2638   unsigned TrapID = Op.getOpcode() == ISD::DEBUGTRAP ?
2639     SISubtarget::TrapIDLLVMDebugTrap : SISubtarget::TrapIDLLVMTrap;
2640 
2641   if (Subtarget->getTrapHandlerAbi() == SISubtarget::TrapHandlerAbiHsa &&
2642       Subtarget->isTrapHandlerEnabled()) {
2643     SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
2644     unsigned UserSGPR = Info->getQueuePtrUserSGPR();
2645     assert(UserSGPR != AMDGPU::NoRegister);
2646 
2647     SDValue QueuePtr = CreateLiveInRegister(
2648       DAG, &AMDGPU::SReg_64RegClass, UserSGPR, MVT::i64);
2649 
2650     SDValue SGPR01 = DAG.getRegister(AMDGPU::SGPR0_SGPR1, MVT::i64);
2651 
2652     SDValue ToReg = DAG.getCopyToReg(Chain, SL, SGPR01,
2653                                      QueuePtr, SDValue());
2654 
2655     SDValue Ops[] = {
2656       ToReg,
2657       DAG.getTargetConstant(TrapID, SL, MVT::i16),
2658       SGPR01,
2659       ToReg.getValue(1)
2660     };
2661 
2662     return DAG.getNode(AMDGPUISD::TRAP, SL, MVT::Other, Ops);
2663   }
2664 
2665   switch (TrapID) {
2666   case SISubtarget::TrapIDLLVMTrap:
2667     return DAG.getNode(AMDGPUISD::ENDPGM, SL, MVT::Other, Chain);
2668   case SISubtarget::TrapIDLLVMDebugTrap: {
2669     DiagnosticInfoUnsupported NoTrap(*MF.getFunction(),
2670                                      "debugtrap handler not supported",
2671                                      Op.getDebugLoc(),
2672                                      DS_Warning);
2673     LLVMContext &Ctx = MF.getFunction()->getContext();
2674     Ctx.diagnose(NoTrap);
2675     return Chain;
2676   }
2677   default:
2678     llvm_unreachable("unsupported trap handler type!");
2679   }
2680 
2681   return Chain;
2682 }
2683 
2684 SDValue SITargetLowering::getSegmentAperture(unsigned AS, const SDLoc &DL,
2685                                              SelectionDAG &DAG) const {
2686   // FIXME: Use inline constants (src_{shared, private}_base) instead.
2687   if (Subtarget->hasApertureRegs()) {
2688     unsigned Offset = AS == AMDGPUASI.LOCAL_ADDRESS ?
2689         AMDGPU::Hwreg::OFFSET_SRC_SHARED_BASE :
2690         AMDGPU::Hwreg::OFFSET_SRC_PRIVATE_BASE;
2691     unsigned WidthM1 = AS == AMDGPUASI.LOCAL_ADDRESS ?
2692         AMDGPU::Hwreg::WIDTH_M1_SRC_SHARED_BASE :
2693         AMDGPU::Hwreg::WIDTH_M1_SRC_PRIVATE_BASE;
2694     unsigned Encoding =
2695         AMDGPU::Hwreg::ID_MEM_BASES << AMDGPU::Hwreg::ID_SHIFT_ |
2696         Offset << AMDGPU::Hwreg::OFFSET_SHIFT_ |
2697         WidthM1 << AMDGPU::Hwreg::WIDTH_M1_SHIFT_;
2698 
2699     SDValue EncodingImm = DAG.getTargetConstant(Encoding, DL, MVT::i16);
2700     SDValue ApertureReg = SDValue(
2701         DAG.getMachineNode(AMDGPU::S_GETREG_B32, DL, MVT::i32, EncodingImm), 0);
2702     SDValue ShiftAmount = DAG.getTargetConstant(WidthM1 + 1, DL, MVT::i32);
2703     return DAG.getNode(ISD::SHL, DL, MVT::i32, ApertureReg, ShiftAmount);
2704   }
2705 
2706   MachineFunction &MF = DAG.getMachineFunction();
2707   SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
2708   unsigned UserSGPR = Info->getQueuePtrUserSGPR();
2709   assert(UserSGPR != AMDGPU::NoRegister);
2710 
2711   SDValue QueuePtr = CreateLiveInRegister(
2712     DAG, &AMDGPU::SReg_64RegClass, UserSGPR, MVT::i64);
2713 
2714   // Offset into amd_queue_t for group_segment_aperture_base_hi /
2715   // private_segment_aperture_base_hi.
2716   uint32_t StructOffset = (AS == AMDGPUASI.LOCAL_ADDRESS) ? 0x40 : 0x44;
2717 
2718   SDValue Ptr = DAG.getNode(ISD::ADD, DL, MVT::i64, QueuePtr,
2719                             DAG.getConstant(StructOffset, DL, MVT::i64));
2720 
2721   // TODO: Use custom target PseudoSourceValue.
2722   // TODO: We should use the value from the IR intrinsic call, but it might not
2723   // be available and how do we get it?
2724   Value *V = UndefValue::get(PointerType::get(Type::getInt8Ty(*DAG.getContext()),
2725                                               AMDGPUASI.CONSTANT_ADDRESS));
2726 
2727   MachinePointerInfo PtrInfo(V, StructOffset);
2728   return DAG.getLoad(MVT::i32, DL, QueuePtr.getValue(1), Ptr, PtrInfo,
2729                      MinAlign(64, StructOffset),
2730                      MachineMemOperand::MODereferenceable |
2731                          MachineMemOperand::MOInvariant);
2732 }
2733 
2734 SDValue SITargetLowering::lowerADDRSPACECAST(SDValue Op,
2735                                              SelectionDAG &DAG) const {
2736   SDLoc SL(Op);
2737   const AddrSpaceCastSDNode *ASC = cast<AddrSpaceCastSDNode>(Op);
2738 
2739   SDValue Src = ASC->getOperand(0);
2740   SDValue FlatNullPtr = DAG.getConstant(0, SL, MVT::i64);
2741 
2742   const AMDGPUTargetMachine &TM =
2743     static_cast<const AMDGPUTargetMachine &>(getTargetMachine());
2744 
2745   // flat -> local/private
2746   if (ASC->getSrcAddressSpace() == AMDGPUASI.FLAT_ADDRESS) {
2747     unsigned DestAS = ASC->getDestAddressSpace();
2748 
2749     if (DestAS == AMDGPUASI.LOCAL_ADDRESS ||
2750         DestAS == AMDGPUASI.PRIVATE_ADDRESS) {
2751       unsigned NullVal = TM.getNullPointerValue(DestAS);
2752       SDValue SegmentNullPtr = DAG.getConstant(NullVal, SL, MVT::i32);
2753       SDValue NonNull = DAG.getSetCC(SL, MVT::i1, Src, FlatNullPtr, ISD::SETNE);
2754       SDValue Ptr = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Src);
2755 
2756       return DAG.getNode(ISD::SELECT, SL, MVT::i32,
2757                          NonNull, Ptr, SegmentNullPtr);
2758     }
2759   }
2760 
2761   // local/private -> flat
2762   if (ASC->getDestAddressSpace() == AMDGPUASI.FLAT_ADDRESS) {
2763     unsigned SrcAS = ASC->getSrcAddressSpace();
2764 
2765     if (SrcAS == AMDGPUASI.LOCAL_ADDRESS ||
2766         SrcAS == AMDGPUASI.PRIVATE_ADDRESS) {
2767       unsigned NullVal = TM.getNullPointerValue(SrcAS);
2768       SDValue SegmentNullPtr = DAG.getConstant(NullVal, SL, MVT::i32);
2769 
2770       SDValue NonNull
2771         = DAG.getSetCC(SL, MVT::i1, Src, SegmentNullPtr, ISD::SETNE);
2772 
2773       SDValue Aperture = getSegmentAperture(ASC->getSrcAddressSpace(), SL, DAG);
2774       SDValue CvtPtr
2775         = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32, Src, Aperture);
2776 
2777       return DAG.getNode(ISD::SELECT, SL, MVT::i64, NonNull,
2778                          DAG.getNode(ISD::BITCAST, SL, MVT::i64, CvtPtr),
2779                          FlatNullPtr);
2780     }
2781   }
2782 
2783   // global <-> flat are no-ops and never emitted.
2784 
2785   const MachineFunction &MF = DAG.getMachineFunction();
2786   DiagnosticInfoUnsupported InvalidAddrSpaceCast(
2787     *MF.getFunction(), "invalid addrspacecast", SL.getDebugLoc());
2788   DAG.getContext()->diagnose(InvalidAddrSpaceCast);
2789 
2790   return DAG.getUNDEF(ASC->getValueType(0));
2791 }
2792 
2793 SDValue SITargetLowering::lowerINSERT_VECTOR_ELT(SDValue Op,
2794                                                  SelectionDAG &DAG) const {
2795   SDValue Idx = Op.getOperand(2);
2796   if (isa<ConstantSDNode>(Idx))
2797     return SDValue();
2798 
2799   // Avoid stack access for dynamic indexing.
2800   SDLoc SL(Op);
2801   SDValue Vec = Op.getOperand(0);
2802   SDValue Val = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Op.getOperand(1));
2803 
2804   // v_bfi_b32 (v_bfm_b32 16, (shl idx, 16)), val, vec
2805   SDValue ExtVal = DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i32, Val);
2806 
2807   // Convert vector index to bit-index.
2808   SDValue ScaledIdx = DAG.getNode(ISD::SHL, SL, MVT::i32, Idx,
2809                                   DAG.getConstant(16, SL, MVT::i32));
2810 
2811   SDValue BCVec = DAG.getNode(ISD::BITCAST, SL, MVT::i32, Vec);
2812 
2813   SDValue BFM = DAG.getNode(ISD::SHL, SL, MVT::i32,
2814                             DAG.getConstant(0xffff, SL, MVT::i32),
2815                             ScaledIdx);
2816 
2817   SDValue LHS = DAG.getNode(ISD::AND, SL, MVT::i32, BFM, ExtVal);
2818   SDValue RHS = DAG.getNode(ISD::AND, SL, MVT::i32,
2819                             DAG.getNOT(SL, BFM, MVT::i32), BCVec);
2820 
2821   SDValue BFI = DAG.getNode(ISD::OR, SL, MVT::i32, LHS, RHS);
2822   return DAG.getNode(ISD::BITCAST, SL, Op.getValueType(), BFI);
2823 }
2824 
2825 SDValue SITargetLowering::lowerEXTRACT_VECTOR_ELT(SDValue Op,
2826                                                   SelectionDAG &DAG) const {
2827   SDLoc SL(Op);
2828 
2829   EVT ResultVT = Op.getValueType();
2830   SDValue Vec = Op.getOperand(0);
2831   SDValue Idx = Op.getOperand(1);
2832 
2833   DAGCombinerInfo DCI(DAG, AfterLegalizeVectorOps, true, nullptr);
2834 
2835   // Make sure we we do any optimizations that will make it easier to fold
2836   // source modifiers before obscuring it with bit operations.
2837 
2838   // XXX - Why doesn't this get called when vector_shuffle is expanded?
2839   if (SDValue Combined = performExtractVectorEltCombine(Op.getNode(), DCI))
2840     return Combined;
2841 
2842   if (const ConstantSDNode *CIdx = dyn_cast<ConstantSDNode>(Idx)) {
2843     SDValue Result = DAG.getNode(ISD::BITCAST, SL, MVT::i32, Vec);
2844 
2845     if (CIdx->getZExtValue() == 1) {
2846       Result = DAG.getNode(ISD::SRL, SL, MVT::i32, Result,
2847                            DAG.getConstant(16, SL, MVT::i32));
2848     } else {
2849       assert(CIdx->getZExtValue() == 0);
2850     }
2851 
2852     if (ResultVT.bitsLT(MVT::i32))
2853       Result = DAG.getNode(ISD::TRUNCATE, SL, MVT::i16, Result);
2854     return DAG.getNode(ISD::BITCAST, SL, ResultVT, Result);
2855   }
2856 
2857   SDValue Sixteen = DAG.getConstant(16, SL, MVT::i32);
2858 
2859   // Convert vector index to bit-index.
2860   SDValue ScaledIdx = DAG.getNode(ISD::SHL, SL, MVT::i32, Idx, Sixteen);
2861 
2862   SDValue BC = DAG.getNode(ISD::BITCAST, SL, MVT::i32, Vec);
2863   SDValue Elt = DAG.getNode(ISD::SRL, SL, MVT::i32, BC, ScaledIdx);
2864 
2865   SDValue Result = Elt;
2866   if (ResultVT.bitsLT(MVT::i32))
2867     Result = DAG.getNode(ISD::TRUNCATE, SL, MVT::i16, Result);
2868 
2869   return DAG.getNode(ISD::BITCAST, SL, ResultVT, Result);
2870 }
2871 
2872 bool
2873 SITargetLowering::isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const {
2874   // We can fold offsets for anything that doesn't require a GOT relocation.
2875   return (GA->getAddressSpace() == AMDGPUASI.GLOBAL_ADDRESS ||
2876               GA->getAddressSpace() == AMDGPUASI.CONSTANT_ADDRESS) &&
2877          !shouldEmitGOTReloc(GA->getGlobal());
2878 }
2879 
2880 static SDValue
2881 buildPCRelGlobalAddress(SelectionDAG &DAG, const GlobalValue *GV,
2882                         const SDLoc &DL, unsigned Offset, EVT PtrVT,
2883                         unsigned GAFlags = SIInstrInfo::MO_NONE) {
2884   // In order to support pc-relative addressing, the PC_ADD_REL_OFFSET SDNode is
2885   // lowered to the following code sequence:
2886   //
2887   // For constant address space:
2888   //   s_getpc_b64 s[0:1]
2889   //   s_add_u32 s0, s0, $symbol
2890   //   s_addc_u32 s1, s1, 0
2891   //
2892   //   s_getpc_b64 returns the address of the s_add_u32 instruction and then
2893   //   a fixup or relocation is emitted to replace $symbol with a literal
2894   //   constant, which is a pc-relative offset from the encoding of the $symbol
2895   //   operand to the global variable.
2896   //
2897   // For global address space:
2898   //   s_getpc_b64 s[0:1]
2899   //   s_add_u32 s0, s0, $symbol@{gotpc}rel32@lo
2900   //   s_addc_u32 s1, s1, $symbol@{gotpc}rel32@hi
2901   //
2902   //   s_getpc_b64 returns the address of the s_add_u32 instruction and then
2903   //   fixups or relocations are emitted to replace $symbol@*@lo and
2904   //   $symbol@*@hi with lower 32 bits and higher 32 bits of a literal constant,
2905   //   which is a 64-bit pc-relative offset from the encoding of the $symbol
2906   //   operand to the global variable.
2907   //
2908   // What we want here is an offset from the value returned by s_getpc
2909   // (which is the address of the s_add_u32 instruction) to the global
2910   // variable, but since the encoding of $symbol starts 4 bytes after the start
2911   // of the s_add_u32 instruction, we end up with an offset that is 4 bytes too
2912   // small. This requires us to add 4 to the global variable offset in order to
2913   // compute the correct address.
2914   SDValue PtrLo = DAG.getTargetGlobalAddress(GV, DL, MVT::i32, Offset + 4,
2915                                              GAFlags);
2916   SDValue PtrHi = DAG.getTargetGlobalAddress(GV, DL, MVT::i32, Offset + 4,
2917                                              GAFlags == SIInstrInfo::MO_NONE ?
2918                                              GAFlags : GAFlags + 1);
2919   return DAG.getNode(AMDGPUISD::PC_ADD_REL_OFFSET, DL, PtrVT, PtrLo, PtrHi);
2920 }
2921 
2922 SDValue SITargetLowering::LowerGlobalAddress(AMDGPUMachineFunction *MFI,
2923                                              SDValue Op,
2924                                              SelectionDAG &DAG) const {
2925   GlobalAddressSDNode *GSD = cast<GlobalAddressSDNode>(Op);
2926 
2927   if (GSD->getAddressSpace() != AMDGPUASI.CONSTANT_ADDRESS &&
2928       GSD->getAddressSpace() != AMDGPUASI.GLOBAL_ADDRESS)
2929     return AMDGPUTargetLowering::LowerGlobalAddress(MFI, Op, DAG);
2930 
2931   SDLoc DL(GSD);
2932   const GlobalValue *GV = GSD->getGlobal();
2933   EVT PtrVT = Op.getValueType();
2934 
2935   if (shouldEmitFixup(GV))
2936     return buildPCRelGlobalAddress(DAG, GV, DL, GSD->getOffset(), PtrVT);
2937   else if (shouldEmitPCReloc(GV))
2938     return buildPCRelGlobalAddress(DAG, GV, DL, GSD->getOffset(), PtrVT,
2939                                    SIInstrInfo::MO_REL32);
2940 
2941   SDValue GOTAddr = buildPCRelGlobalAddress(DAG, GV, DL, 0, PtrVT,
2942                                             SIInstrInfo::MO_GOTPCREL32);
2943 
2944   Type *Ty = PtrVT.getTypeForEVT(*DAG.getContext());
2945   PointerType *PtrTy = PointerType::get(Ty, AMDGPUASI.CONSTANT_ADDRESS);
2946   const DataLayout &DataLayout = DAG.getDataLayout();
2947   unsigned Align = DataLayout.getABITypeAlignment(PtrTy);
2948   // FIXME: Use a PseudoSourceValue once those can be assigned an address space.
2949   MachinePointerInfo PtrInfo(UndefValue::get(PtrTy));
2950 
2951   return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), GOTAddr, PtrInfo, Align,
2952                      MachineMemOperand::MODereferenceable |
2953                          MachineMemOperand::MOInvariant);
2954 }
2955 
2956 SDValue SITargetLowering::copyToM0(SelectionDAG &DAG, SDValue Chain,
2957                                    const SDLoc &DL, SDValue V) const {
2958   // We can't use S_MOV_B32 directly, because there is no way to specify m0 as
2959   // the destination register.
2960   //
2961   // We can't use CopyToReg, because MachineCSE won't combine COPY instructions,
2962   // so we will end up with redundant moves to m0.
2963   //
2964   // We use a pseudo to ensure we emit s_mov_b32 with m0 as the direct result.
2965 
2966   // A Null SDValue creates a glue result.
2967   SDNode *M0 = DAG.getMachineNode(AMDGPU::SI_INIT_M0, DL, MVT::Other, MVT::Glue,
2968                                   V, Chain);
2969   return SDValue(M0, 0);
2970 }
2971 
2972 SDValue SITargetLowering::lowerImplicitZextParam(SelectionDAG &DAG,
2973                                                  SDValue Op,
2974                                                  MVT VT,
2975                                                  unsigned Offset) const {
2976   SDLoc SL(Op);
2977   SDValue Param = lowerKernargMemParameter(DAG, MVT::i32, MVT::i32, SL,
2978                                            DAG.getEntryNode(), Offset, false);
2979   // The local size values will have the hi 16-bits as zero.
2980   return DAG.getNode(ISD::AssertZext, SL, MVT::i32, Param,
2981                      DAG.getValueType(VT));
2982 }
2983 
2984 static SDValue emitNonHSAIntrinsicError(SelectionDAG &DAG, const SDLoc &DL,
2985                                         EVT VT) {
2986   DiagnosticInfoUnsupported BadIntrin(*DAG.getMachineFunction().getFunction(),
2987                                       "non-hsa intrinsic with hsa target",
2988                                       DL.getDebugLoc());
2989   DAG.getContext()->diagnose(BadIntrin);
2990   return DAG.getUNDEF(VT);
2991 }
2992 
2993 static SDValue emitRemovedIntrinsicError(SelectionDAG &DAG, const SDLoc &DL,
2994                                          EVT VT) {
2995   DiagnosticInfoUnsupported BadIntrin(*DAG.getMachineFunction().getFunction(),
2996                                       "intrinsic not supported on subtarget",
2997                                       DL.getDebugLoc());
2998   DAG.getContext()->diagnose(BadIntrin);
2999   return DAG.getUNDEF(VT);
3000 }
3001 
3002 SDValue SITargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
3003                                                   SelectionDAG &DAG) const {
3004   MachineFunction &MF = DAG.getMachineFunction();
3005   auto MFI = MF.getInfo<SIMachineFunctionInfo>();
3006   const SIRegisterInfo *TRI = getSubtarget()->getRegisterInfo();
3007 
3008   EVT VT = Op.getValueType();
3009   SDLoc DL(Op);
3010   unsigned IntrinsicID = cast<ConstantSDNode>(Op.getOperand(0))->getZExtValue();
3011 
3012   // TODO: Should this propagate fast-math-flags?
3013 
3014   switch (IntrinsicID) {
3015   case Intrinsic::amdgcn_implicit_buffer_ptr: {
3016     if (getSubtarget()->isAmdCodeObjectV2(MF))
3017       return emitNonHSAIntrinsicError(DAG, DL, VT);
3018 
3019     unsigned Reg = TRI->getPreloadedValue(MF,
3020                                           SIRegisterInfo::IMPLICIT_BUFFER_PTR);
3021     return CreateLiveInRegister(DAG, &AMDGPU::SReg_64RegClass, Reg, VT);
3022   }
3023   case Intrinsic::amdgcn_dispatch_ptr:
3024   case Intrinsic::amdgcn_queue_ptr: {
3025     if (!Subtarget->isAmdCodeObjectV2(MF)) {
3026       DiagnosticInfoUnsupported BadIntrin(
3027           *MF.getFunction(), "unsupported hsa intrinsic without hsa target",
3028           DL.getDebugLoc());
3029       DAG.getContext()->diagnose(BadIntrin);
3030       return DAG.getUNDEF(VT);
3031     }
3032 
3033     auto Reg = IntrinsicID == Intrinsic::amdgcn_dispatch_ptr ?
3034       SIRegisterInfo::DISPATCH_PTR : SIRegisterInfo::QUEUE_PTR;
3035     return CreateLiveInRegister(DAG, &AMDGPU::SReg_64RegClass,
3036                                 TRI->getPreloadedValue(MF, Reg), VT);
3037   }
3038   case Intrinsic::amdgcn_implicitarg_ptr: {
3039     if (MFI->isEntryFunction())
3040       return getImplicitArgPtr(DAG, DL);
3041     report_fatal_error("amdgcn.implicitarg.ptr not implemented for functions");
3042   }
3043   case Intrinsic::amdgcn_kernarg_segment_ptr: {
3044     unsigned Reg
3045       = TRI->getPreloadedValue(MF, SIRegisterInfo::KERNARG_SEGMENT_PTR);
3046     return CreateLiveInRegister(DAG, &AMDGPU::SReg_64RegClass, Reg, VT);
3047   }
3048   case Intrinsic::amdgcn_dispatch_id: {
3049     unsigned Reg = TRI->getPreloadedValue(MF, SIRegisterInfo::DISPATCH_ID);
3050     return CreateLiveInRegister(DAG, &AMDGPU::SReg_64RegClass, Reg, VT);
3051   }
3052   case Intrinsic::amdgcn_rcp:
3053     return DAG.getNode(AMDGPUISD::RCP, DL, VT, Op.getOperand(1));
3054   case Intrinsic::amdgcn_rsq:
3055     return DAG.getNode(AMDGPUISD::RSQ, DL, VT, Op.getOperand(1));
3056   case Intrinsic::amdgcn_rsq_legacy:
3057     if (Subtarget->getGeneration() >= SISubtarget::VOLCANIC_ISLANDS)
3058       return emitRemovedIntrinsicError(DAG, DL, VT);
3059 
3060     return DAG.getNode(AMDGPUISD::RSQ_LEGACY, DL, VT, Op.getOperand(1));
3061   case Intrinsic::amdgcn_rcp_legacy:
3062     if (Subtarget->getGeneration() >= SISubtarget::VOLCANIC_ISLANDS)
3063       return emitRemovedIntrinsicError(DAG, DL, VT);
3064     return DAG.getNode(AMDGPUISD::RCP_LEGACY, DL, VT, Op.getOperand(1));
3065   case Intrinsic::amdgcn_rsq_clamp: {
3066     if (Subtarget->getGeneration() < SISubtarget::VOLCANIC_ISLANDS)
3067       return DAG.getNode(AMDGPUISD::RSQ_CLAMP, DL, VT, Op.getOperand(1));
3068 
3069     Type *Type = VT.getTypeForEVT(*DAG.getContext());
3070     APFloat Max = APFloat::getLargest(Type->getFltSemantics());
3071     APFloat Min = APFloat::getLargest(Type->getFltSemantics(), true);
3072 
3073     SDValue Rsq = DAG.getNode(AMDGPUISD::RSQ, DL, VT, Op.getOperand(1));
3074     SDValue Tmp = DAG.getNode(ISD::FMINNUM, DL, VT, Rsq,
3075                               DAG.getConstantFP(Max, DL, VT));
3076     return DAG.getNode(ISD::FMAXNUM, DL, VT, Tmp,
3077                        DAG.getConstantFP(Min, DL, VT));
3078   }
3079   case Intrinsic::r600_read_ngroups_x:
3080     if (Subtarget->isAmdHsaOS())
3081       return emitNonHSAIntrinsicError(DAG, DL, VT);
3082 
3083     return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
3084                                     SI::KernelInputOffsets::NGROUPS_X, false);
3085   case Intrinsic::r600_read_ngroups_y:
3086     if (Subtarget->isAmdHsaOS())
3087       return emitNonHSAIntrinsicError(DAG, DL, VT);
3088 
3089     return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
3090                                     SI::KernelInputOffsets::NGROUPS_Y, false);
3091   case Intrinsic::r600_read_ngroups_z:
3092     if (Subtarget->isAmdHsaOS())
3093       return emitNonHSAIntrinsicError(DAG, DL, VT);
3094 
3095     return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
3096                                     SI::KernelInputOffsets::NGROUPS_Z, false);
3097   case Intrinsic::r600_read_global_size_x:
3098     if (Subtarget->isAmdHsaOS())
3099       return emitNonHSAIntrinsicError(DAG, DL, VT);
3100 
3101     return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
3102                                     SI::KernelInputOffsets::GLOBAL_SIZE_X, false);
3103   case Intrinsic::r600_read_global_size_y:
3104     if (Subtarget->isAmdHsaOS())
3105       return emitNonHSAIntrinsicError(DAG, DL, VT);
3106 
3107     return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
3108                                     SI::KernelInputOffsets::GLOBAL_SIZE_Y, false);
3109   case Intrinsic::r600_read_global_size_z:
3110     if (Subtarget->isAmdHsaOS())
3111       return emitNonHSAIntrinsicError(DAG, DL, VT);
3112 
3113     return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
3114                                     SI::KernelInputOffsets::GLOBAL_SIZE_Z, false);
3115   case Intrinsic::r600_read_local_size_x:
3116     if (Subtarget->isAmdHsaOS())
3117       return emitNonHSAIntrinsicError(DAG, DL, VT);
3118 
3119     return lowerImplicitZextParam(DAG, Op, MVT::i16,
3120                                   SI::KernelInputOffsets::LOCAL_SIZE_X);
3121   case Intrinsic::r600_read_local_size_y:
3122     if (Subtarget->isAmdHsaOS())
3123       return emitNonHSAIntrinsicError(DAG, DL, VT);
3124 
3125     return lowerImplicitZextParam(DAG, Op, MVT::i16,
3126                                   SI::KernelInputOffsets::LOCAL_SIZE_Y);
3127   case Intrinsic::r600_read_local_size_z:
3128     if (Subtarget->isAmdHsaOS())
3129       return emitNonHSAIntrinsicError(DAG, DL, VT);
3130 
3131     return lowerImplicitZextParam(DAG, Op, MVT::i16,
3132                                   SI::KernelInputOffsets::LOCAL_SIZE_Z);
3133   case Intrinsic::amdgcn_workgroup_id_x:
3134   case Intrinsic::r600_read_tgid_x:
3135     return CreateLiveInRegister(DAG, &AMDGPU::SReg_32_XM0RegClass,
3136       TRI->getPreloadedValue(MF, SIRegisterInfo::WORKGROUP_ID_X), VT);
3137   case Intrinsic::amdgcn_workgroup_id_y:
3138   case Intrinsic::r600_read_tgid_y:
3139     return CreateLiveInRegister(DAG, &AMDGPU::SReg_32_XM0RegClass,
3140       TRI->getPreloadedValue(MF, SIRegisterInfo::WORKGROUP_ID_Y), VT);
3141   case Intrinsic::amdgcn_workgroup_id_z:
3142   case Intrinsic::r600_read_tgid_z:
3143     return CreateLiveInRegister(DAG, &AMDGPU::SReg_32_XM0RegClass,
3144       TRI->getPreloadedValue(MF, SIRegisterInfo::WORKGROUP_ID_Z), VT);
3145   case Intrinsic::amdgcn_workitem_id_x:
3146   case Intrinsic::r600_read_tidig_x:
3147     return CreateLiveInRegister(DAG, &AMDGPU::VGPR_32RegClass,
3148       TRI->getPreloadedValue(MF, SIRegisterInfo::WORKITEM_ID_X), VT);
3149   case Intrinsic::amdgcn_workitem_id_y:
3150   case Intrinsic::r600_read_tidig_y:
3151     return CreateLiveInRegister(DAG, &AMDGPU::VGPR_32RegClass,
3152       TRI->getPreloadedValue(MF, SIRegisterInfo::WORKITEM_ID_Y), VT);
3153   case Intrinsic::amdgcn_workitem_id_z:
3154   case Intrinsic::r600_read_tidig_z:
3155     return CreateLiveInRegister(DAG, &AMDGPU::VGPR_32RegClass,
3156       TRI->getPreloadedValue(MF, SIRegisterInfo::WORKITEM_ID_Z), VT);
3157   case AMDGPUIntrinsic::SI_load_const: {
3158     SDValue Ops[] = {
3159       Op.getOperand(1),
3160       Op.getOperand(2)
3161     };
3162 
3163     MachineMemOperand *MMO = MF.getMachineMemOperand(
3164         MachinePointerInfo(),
3165         MachineMemOperand::MOLoad | MachineMemOperand::MODereferenceable |
3166             MachineMemOperand::MOInvariant,
3167         VT.getStoreSize(), 4);
3168     return DAG.getMemIntrinsicNode(AMDGPUISD::LOAD_CONSTANT, DL,
3169                                    Op->getVTList(), Ops, VT, MMO);
3170   }
3171   case Intrinsic::amdgcn_fdiv_fast:
3172     return lowerFDIV_FAST(Op, DAG);
3173   case Intrinsic::amdgcn_interp_mov: {
3174     SDValue M0 = copyToM0(DAG, DAG.getEntryNode(), DL, Op.getOperand(4));
3175     SDValue Glue = M0.getValue(1);
3176     return DAG.getNode(AMDGPUISD::INTERP_MOV, DL, MVT::f32, Op.getOperand(1),
3177                        Op.getOperand(2), Op.getOperand(3), Glue);
3178   }
3179   case Intrinsic::amdgcn_interp_p1: {
3180     SDValue M0 = copyToM0(DAG, DAG.getEntryNode(), DL, Op.getOperand(4));
3181     SDValue Glue = M0.getValue(1);
3182     return DAG.getNode(AMDGPUISD::INTERP_P1, DL, MVT::f32, Op.getOperand(1),
3183                        Op.getOperand(2), Op.getOperand(3), Glue);
3184   }
3185   case Intrinsic::amdgcn_interp_p2: {
3186     SDValue M0 = copyToM0(DAG, DAG.getEntryNode(), DL, Op.getOperand(5));
3187     SDValue Glue = SDValue(M0.getNode(), 1);
3188     return DAG.getNode(AMDGPUISD::INTERP_P2, DL, MVT::f32, Op.getOperand(1),
3189                        Op.getOperand(2), Op.getOperand(3), Op.getOperand(4),
3190                        Glue);
3191   }
3192   case Intrinsic::amdgcn_sin:
3193     return DAG.getNode(AMDGPUISD::SIN_HW, DL, VT, Op.getOperand(1));
3194 
3195   case Intrinsic::amdgcn_cos:
3196     return DAG.getNode(AMDGPUISD::COS_HW, DL, VT, Op.getOperand(1));
3197 
3198   case Intrinsic::amdgcn_log_clamp: {
3199     if (Subtarget->getGeneration() < SISubtarget::VOLCANIC_ISLANDS)
3200       return SDValue();
3201 
3202     DiagnosticInfoUnsupported BadIntrin(
3203       *MF.getFunction(), "intrinsic not supported on subtarget",
3204       DL.getDebugLoc());
3205       DAG.getContext()->diagnose(BadIntrin);
3206       return DAG.getUNDEF(VT);
3207   }
3208   case Intrinsic::amdgcn_ldexp:
3209     return DAG.getNode(AMDGPUISD::LDEXP, DL, VT,
3210                        Op.getOperand(1), Op.getOperand(2));
3211 
3212   case Intrinsic::amdgcn_fract:
3213     return DAG.getNode(AMDGPUISD::FRACT, DL, VT, Op.getOperand(1));
3214 
3215   case Intrinsic::amdgcn_class:
3216     return DAG.getNode(AMDGPUISD::FP_CLASS, DL, VT,
3217                        Op.getOperand(1), Op.getOperand(2));
3218   case Intrinsic::amdgcn_div_fmas:
3219     return DAG.getNode(AMDGPUISD::DIV_FMAS, DL, VT,
3220                        Op.getOperand(1), Op.getOperand(2), Op.getOperand(3),
3221                        Op.getOperand(4));
3222 
3223   case Intrinsic::amdgcn_div_fixup:
3224     return DAG.getNode(AMDGPUISD::DIV_FIXUP, DL, VT,
3225                        Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
3226 
3227   case Intrinsic::amdgcn_trig_preop:
3228     return DAG.getNode(AMDGPUISD::TRIG_PREOP, DL, VT,
3229                        Op.getOperand(1), Op.getOperand(2));
3230   case Intrinsic::amdgcn_div_scale: {
3231     // 3rd parameter required to be a constant.
3232     const ConstantSDNode *Param = dyn_cast<ConstantSDNode>(Op.getOperand(3));
3233     if (!Param)
3234       return DAG.getUNDEF(VT);
3235 
3236     // Translate to the operands expected by the machine instruction. The
3237     // first parameter must be the same as the first instruction.
3238     SDValue Numerator = Op.getOperand(1);
3239     SDValue Denominator = Op.getOperand(2);
3240 
3241     // Note this order is opposite of the machine instruction's operations,
3242     // which is s0.f = Quotient, s1.f = Denominator, s2.f = Numerator. The
3243     // intrinsic has the numerator as the first operand to match a normal
3244     // division operation.
3245 
3246     SDValue Src0 = Param->isAllOnesValue() ? Numerator : Denominator;
3247 
3248     return DAG.getNode(AMDGPUISD::DIV_SCALE, DL, Op->getVTList(), Src0,
3249                        Denominator, Numerator);
3250   }
3251   case Intrinsic::amdgcn_icmp: {
3252     const auto *CD = dyn_cast<ConstantSDNode>(Op.getOperand(3));
3253     if (!CD)
3254       return DAG.getUNDEF(VT);
3255 
3256     int CondCode = CD->getSExtValue();
3257     if (CondCode < ICmpInst::Predicate::FIRST_ICMP_PREDICATE ||
3258         CondCode > ICmpInst::Predicate::LAST_ICMP_PREDICATE)
3259       return DAG.getUNDEF(VT);
3260 
3261     ICmpInst::Predicate IcInput = static_cast<ICmpInst::Predicate>(CondCode);
3262     ISD::CondCode CCOpcode = getICmpCondCode(IcInput);
3263     return DAG.getNode(AMDGPUISD::SETCC, DL, VT, Op.getOperand(1),
3264                        Op.getOperand(2), DAG.getCondCode(CCOpcode));
3265   }
3266   case Intrinsic::amdgcn_fcmp: {
3267     const auto *CD = dyn_cast<ConstantSDNode>(Op.getOperand(3));
3268     if (!CD)
3269       return DAG.getUNDEF(VT);
3270 
3271     int CondCode = CD->getSExtValue();
3272     if (CondCode < FCmpInst::Predicate::FIRST_FCMP_PREDICATE ||
3273         CondCode > FCmpInst::Predicate::LAST_FCMP_PREDICATE)
3274       return DAG.getUNDEF(VT);
3275 
3276     FCmpInst::Predicate IcInput = static_cast<FCmpInst::Predicate>(CondCode);
3277     ISD::CondCode CCOpcode = getFCmpCondCode(IcInput);
3278     return DAG.getNode(AMDGPUISD::SETCC, DL, VT, Op.getOperand(1),
3279                        Op.getOperand(2), DAG.getCondCode(CCOpcode));
3280   }
3281   case Intrinsic::amdgcn_fmed3:
3282     return DAG.getNode(AMDGPUISD::FMED3, DL, VT,
3283                        Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
3284   case Intrinsic::amdgcn_fmul_legacy:
3285     return DAG.getNode(AMDGPUISD::FMUL_LEGACY, DL, VT,
3286                        Op.getOperand(1), Op.getOperand(2));
3287   case Intrinsic::amdgcn_sffbh:
3288     return DAG.getNode(AMDGPUISD::FFBH_I32, DL, VT, Op.getOperand(1));
3289   case Intrinsic::amdgcn_sbfe:
3290     return DAG.getNode(AMDGPUISD::BFE_I32, DL, VT,
3291                        Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
3292   case Intrinsic::amdgcn_ubfe:
3293     return DAG.getNode(AMDGPUISD::BFE_U32, DL, VT,
3294                        Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
3295   case Intrinsic::amdgcn_cvt_pkrtz: {
3296     // FIXME: Stop adding cast if v2f16 legal.
3297     EVT VT = Op.getValueType();
3298     SDValue Node = DAG.getNode(AMDGPUISD::CVT_PKRTZ_F16_F32, DL, MVT::i32,
3299                                Op.getOperand(1), Op.getOperand(2));
3300     return DAG.getNode(ISD::BITCAST, DL, VT, Node);
3301   }
3302   default:
3303     return Op;
3304   }
3305 }
3306 
3307 SDValue SITargetLowering::LowerINTRINSIC_W_CHAIN(SDValue Op,
3308                                                  SelectionDAG &DAG) const {
3309   unsigned IntrID = cast<ConstantSDNode>(Op.getOperand(1))->getZExtValue();
3310   SDLoc DL(Op);
3311   MachineFunction &MF = DAG.getMachineFunction();
3312 
3313   switch (IntrID) {
3314   case Intrinsic::amdgcn_atomic_inc:
3315   case Intrinsic::amdgcn_atomic_dec: {
3316     MemSDNode *M = cast<MemSDNode>(Op);
3317     unsigned Opc = (IntrID == Intrinsic::amdgcn_atomic_inc) ?
3318       AMDGPUISD::ATOMIC_INC : AMDGPUISD::ATOMIC_DEC;
3319     SDValue Ops[] = {
3320       M->getOperand(0), // Chain
3321       M->getOperand(2), // Ptr
3322       M->getOperand(3)  // Value
3323     };
3324 
3325     return DAG.getMemIntrinsicNode(Opc, SDLoc(Op), M->getVTList(), Ops,
3326                                    M->getMemoryVT(), M->getMemOperand());
3327   }
3328   case Intrinsic::amdgcn_buffer_load:
3329   case Intrinsic::amdgcn_buffer_load_format: {
3330     SDValue Ops[] = {
3331       Op.getOperand(0), // Chain
3332       Op.getOperand(2), // rsrc
3333       Op.getOperand(3), // vindex
3334       Op.getOperand(4), // offset
3335       Op.getOperand(5), // glc
3336       Op.getOperand(6)  // slc
3337     };
3338     SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
3339 
3340     unsigned Opc = (IntrID == Intrinsic::amdgcn_buffer_load) ?
3341         AMDGPUISD::BUFFER_LOAD : AMDGPUISD::BUFFER_LOAD_FORMAT;
3342     EVT VT = Op.getValueType();
3343     EVT IntVT = VT.changeTypeToInteger();
3344 
3345     MachineMemOperand *MMO = MF.getMachineMemOperand(
3346       MachinePointerInfo(MFI->getBufferPSV()),
3347       MachineMemOperand::MOLoad,
3348       VT.getStoreSize(), VT.getStoreSize());
3349 
3350     return DAG.getMemIntrinsicNode(Opc, DL, Op->getVTList(), Ops, IntVT, MMO);
3351   }
3352   case Intrinsic::amdgcn_tbuffer_load: {
3353     SDValue Ops[] = {
3354       Op.getOperand(0),  // Chain
3355       Op.getOperand(2),  // rsrc
3356       Op.getOperand(3),  // vindex
3357       Op.getOperand(4),  // voffset
3358       Op.getOperand(5),  // soffset
3359       Op.getOperand(6),  // offset
3360       Op.getOperand(7),  // dfmt
3361       Op.getOperand(8),  // nfmt
3362       Op.getOperand(9),  // glc
3363       Op.getOperand(10)   // slc
3364     };
3365 
3366     EVT VT = Op.getOperand(2).getValueType();
3367 
3368     MachineMemOperand *MMO = MF.getMachineMemOperand(
3369       MachinePointerInfo(),
3370       MachineMemOperand::MOLoad,
3371       VT.getStoreSize(), VT.getStoreSize());
3372     return DAG.getMemIntrinsicNode(AMDGPUISD::TBUFFER_LOAD_FORMAT, DL,
3373                                    Op->getVTList(), Ops, VT, MMO);
3374   }
3375   // Basic sample.
3376   case Intrinsic::amdgcn_image_sample:
3377   case Intrinsic::amdgcn_image_sample_cl:
3378   case Intrinsic::amdgcn_image_sample_d:
3379   case Intrinsic::amdgcn_image_sample_d_cl:
3380   case Intrinsic::amdgcn_image_sample_l:
3381   case Intrinsic::amdgcn_image_sample_b:
3382   case Intrinsic::amdgcn_image_sample_b_cl:
3383   case Intrinsic::amdgcn_image_sample_lz:
3384   case Intrinsic::amdgcn_image_sample_cd:
3385   case Intrinsic::amdgcn_image_sample_cd_cl:
3386 
3387   // Sample with comparison.
3388   case Intrinsic::amdgcn_image_sample_c:
3389   case Intrinsic::amdgcn_image_sample_c_cl:
3390   case Intrinsic::amdgcn_image_sample_c_d:
3391   case Intrinsic::amdgcn_image_sample_c_d_cl:
3392   case Intrinsic::amdgcn_image_sample_c_l:
3393   case Intrinsic::amdgcn_image_sample_c_b:
3394   case Intrinsic::amdgcn_image_sample_c_b_cl:
3395   case Intrinsic::amdgcn_image_sample_c_lz:
3396   case Intrinsic::amdgcn_image_sample_c_cd:
3397   case Intrinsic::amdgcn_image_sample_c_cd_cl:
3398 
3399   // Sample with offsets.
3400   case Intrinsic::amdgcn_image_sample_o:
3401   case Intrinsic::amdgcn_image_sample_cl_o:
3402   case Intrinsic::amdgcn_image_sample_d_o:
3403   case Intrinsic::amdgcn_image_sample_d_cl_o:
3404   case Intrinsic::amdgcn_image_sample_l_o:
3405   case Intrinsic::amdgcn_image_sample_b_o:
3406   case Intrinsic::amdgcn_image_sample_b_cl_o:
3407   case Intrinsic::amdgcn_image_sample_lz_o:
3408   case Intrinsic::amdgcn_image_sample_cd_o:
3409   case Intrinsic::amdgcn_image_sample_cd_cl_o:
3410 
3411   // Sample with comparison and offsets.
3412   case Intrinsic::amdgcn_image_sample_c_o:
3413   case Intrinsic::amdgcn_image_sample_c_cl_o:
3414   case Intrinsic::amdgcn_image_sample_c_d_o:
3415   case Intrinsic::amdgcn_image_sample_c_d_cl_o:
3416   case Intrinsic::amdgcn_image_sample_c_l_o:
3417   case Intrinsic::amdgcn_image_sample_c_b_o:
3418   case Intrinsic::amdgcn_image_sample_c_b_cl_o:
3419   case Intrinsic::amdgcn_image_sample_c_lz_o:
3420   case Intrinsic::amdgcn_image_sample_c_cd_o:
3421   case Intrinsic::amdgcn_image_sample_c_cd_cl_o:
3422 
3423   case Intrinsic::amdgcn_image_getlod: {
3424     // Replace dmask with everything disabled with undef.
3425     const ConstantSDNode *DMask = dyn_cast<ConstantSDNode>(Op.getOperand(5));
3426     if (!DMask || DMask->isNullValue()) {
3427       SDValue Undef = DAG.getUNDEF(Op.getValueType());
3428       return DAG.getMergeValues({ Undef, Op.getOperand(0) }, SDLoc(Op));
3429     }
3430 
3431     return SDValue();
3432   }
3433   default:
3434     return SDValue();
3435   }
3436 }
3437 
3438 SDValue SITargetLowering::LowerINTRINSIC_VOID(SDValue Op,
3439                                               SelectionDAG &DAG) const {
3440   SDLoc DL(Op);
3441   SDValue Chain = Op.getOperand(0);
3442   unsigned IntrinsicID = cast<ConstantSDNode>(Op.getOperand(1))->getZExtValue();
3443   MachineFunction &MF = DAG.getMachineFunction();
3444 
3445   switch (IntrinsicID) {
3446   case Intrinsic::amdgcn_exp: {
3447     const ConstantSDNode *Tgt = cast<ConstantSDNode>(Op.getOperand(2));
3448     const ConstantSDNode *En = cast<ConstantSDNode>(Op.getOperand(3));
3449     const ConstantSDNode *Done = cast<ConstantSDNode>(Op.getOperand(8));
3450     const ConstantSDNode *VM = cast<ConstantSDNode>(Op.getOperand(9));
3451 
3452     const SDValue Ops[] = {
3453       Chain,
3454       DAG.getTargetConstant(Tgt->getZExtValue(), DL, MVT::i8), // tgt
3455       DAG.getTargetConstant(En->getZExtValue(), DL, MVT::i8),  // en
3456       Op.getOperand(4), // src0
3457       Op.getOperand(5), // src1
3458       Op.getOperand(6), // src2
3459       Op.getOperand(7), // src3
3460       DAG.getTargetConstant(0, DL, MVT::i1), // compr
3461       DAG.getTargetConstant(VM->getZExtValue(), DL, MVT::i1)
3462     };
3463 
3464     unsigned Opc = Done->isNullValue() ?
3465       AMDGPUISD::EXPORT : AMDGPUISD::EXPORT_DONE;
3466     return DAG.getNode(Opc, DL, Op->getVTList(), Ops);
3467   }
3468   case Intrinsic::amdgcn_exp_compr: {
3469     const ConstantSDNode *Tgt = cast<ConstantSDNode>(Op.getOperand(2));
3470     const ConstantSDNode *En = cast<ConstantSDNode>(Op.getOperand(3));
3471     SDValue Src0 = Op.getOperand(4);
3472     SDValue Src1 = Op.getOperand(5);
3473     const ConstantSDNode *Done = cast<ConstantSDNode>(Op.getOperand(6));
3474     const ConstantSDNode *VM = cast<ConstantSDNode>(Op.getOperand(7));
3475 
3476     SDValue Undef = DAG.getUNDEF(MVT::f32);
3477     const SDValue Ops[] = {
3478       Chain,
3479       DAG.getTargetConstant(Tgt->getZExtValue(), DL, MVT::i8), // tgt
3480       DAG.getTargetConstant(En->getZExtValue(), DL, MVT::i8),  // en
3481       DAG.getNode(ISD::BITCAST, DL, MVT::f32, Src0),
3482       DAG.getNode(ISD::BITCAST, DL, MVT::f32, Src1),
3483       Undef, // src2
3484       Undef, // src3
3485       DAG.getTargetConstant(1, DL, MVT::i1), // compr
3486       DAG.getTargetConstant(VM->getZExtValue(), DL, MVT::i1)
3487     };
3488 
3489     unsigned Opc = Done->isNullValue() ?
3490       AMDGPUISD::EXPORT : AMDGPUISD::EXPORT_DONE;
3491     return DAG.getNode(Opc, DL, Op->getVTList(), Ops);
3492   }
3493   case Intrinsic::amdgcn_s_sendmsg:
3494   case Intrinsic::amdgcn_s_sendmsghalt: {
3495     unsigned NodeOp = (IntrinsicID == Intrinsic::amdgcn_s_sendmsg) ?
3496       AMDGPUISD::SENDMSG : AMDGPUISD::SENDMSGHALT;
3497     Chain = copyToM0(DAG, Chain, DL, Op.getOperand(3));
3498     SDValue Glue = Chain.getValue(1);
3499     return DAG.getNode(NodeOp, DL, MVT::Other, Chain,
3500                        Op.getOperand(2), Glue);
3501   }
3502   case Intrinsic::amdgcn_init_exec: {
3503     return DAG.getNode(AMDGPUISD::INIT_EXEC, DL, MVT::Other, Chain,
3504                        Op.getOperand(2));
3505   }
3506   case Intrinsic::amdgcn_init_exec_from_input: {
3507     return DAG.getNode(AMDGPUISD::INIT_EXEC_FROM_INPUT, DL, MVT::Other, Chain,
3508                        Op.getOperand(2), Op.getOperand(3));
3509   }
3510   case AMDGPUIntrinsic::AMDGPU_kill: {
3511     SDValue Src = Op.getOperand(2);
3512     if (const ConstantFPSDNode *K = dyn_cast<ConstantFPSDNode>(Src)) {
3513       if (!K->isNegative())
3514         return Chain;
3515 
3516       SDValue NegOne = DAG.getTargetConstant(FloatToBits(-1.0f), DL, MVT::i32);
3517       return DAG.getNode(AMDGPUISD::KILL, DL, MVT::Other, Chain, NegOne);
3518     }
3519 
3520     SDValue Cast = DAG.getNode(ISD::BITCAST, DL, MVT::i32, Src);
3521     return DAG.getNode(AMDGPUISD::KILL, DL, MVT::Other, Chain, Cast);
3522   }
3523   case Intrinsic::amdgcn_s_barrier: {
3524     if (getTargetMachine().getOptLevel() > CodeGenOpt::None) {
3525       const SISubtarget &ST = MF.getSubtarget<SISubtarget>();
3526       unsigned WGSize = ST.getFlatWorkGroupSizes(*MF.getFunction()).second;
3527       if (WGSize <= ST.getWavefrontSize())
3528         return SDValue(DAG.getMachineNode(AMDGPU::WAVE_BARRIER, DL, MVT::Other,
3529                                           Op.getOperand(0)), 0);
3530     }
3531     return SDValue();
3532   };
3533   case AMDGPUIntrinsic::SI_tbuffer_store: {
3534 
3535     // Extract vindex and voffset from vaddr as appropriate
3536     const ConstantSDNode *OffEn = cast<ConstantSDNode>(Op.getOperand(10));
3537     const ConstantSDNode *IdxEn = cast<ConstantSDNode>(Op.getOperand(11));
3538     SDValue VAddr = Op.getOperand(5);
3539 
3540     SDValue Zero = DAG.getTargetConstant(0, DL, MVT::i32);
3541 
3542     assert(!(OffEn->isOne() && IdxEn->isOne()) &&
3543            "Legacy intrinsic doesn't support both offset and index - use new version");
3544 
3545     SDValue VIndex = IdxEn->isOne() ? VAddr : Zero;
3546     SDValue VOffset = OffEn->isOne() ? VAddr : Zero;
3547 
3548     // Deal with the vec-3 case
3549     const ConstantSDNode *NumChannels = cast<ConstantSDNode>(Op.getOperand(4));
3550     auto Opcode = NumChannels->getZExtValue() == 3 ?
3551       AMDGPUISD::TBUFFER_STORE_FORMAT_X3 : AMDGPUISD::TBUFFER_STORE_FORMAT;
3552 
3553     SDValue Ops[] = {
3554      Chain,
3555      Op.getOperand(3),  // vdata
3556      Op.getOperand(2),  // rsrc
3557      VIndex,
3558      VOffset,
3559      Op.getOperand(6),  // soffset
3560      Op.getOperand(7),  // inst_offset
3561      Op.getOperand(8),  // dfmt
3562      Op.getOperand(9),  // nfmt
3563      Op.getOperand(12), // glc
3564      Op.getOperand(13), // slc
3565     };
3566 
3567     assert((cast<ConstantSDNode>(Op.getOperand(14)))->getZExtValue() == 0 &&
3568            "Value of tfe other than zero is unsupported");
3569 
3570     EVT VT = Op.getOperand(3).getValueType();
3571     MachineMemOperand *MMO = MF.getMachineMemOperand(
3572       MachinePointerInfo(),
3573       MachineMemOperand::MOStore,
3574       VT.getStoreSize(), 4);
3575     return DAG.getMemIntrinsicNode(Opcode, DL,
3576                                    Op->getVTList(), Ops, VT, MMO);
3577   }
3578 
3579   case Intrinsic::amdgcn_tbuffer_store: {
3580     SDValue Ops[] = {
3581       Chain,
3582       Op.getOperand(2),  // vdata
3583       Op.getOperand(3),  // rsrc
3584       Op.getOperand(4),  // vindex
3585       Op.getOperand(5),  // voffset
3586       Op.getOperand(6),  // soffset
3587       Op.getOperand(7),  // offset
3588       Op.getOperand(8),  // dfmt
3589       Op.getOperand(9),  // nfmt
3590       Op.getOperand(10), // glc
3591       Op.getOperand(11)  // slc
3592     };
3593     EVT VT = Op.getOperand(3).getValueType();
3594     MachineMemOperand *MMO = MF.getMachineMemOperand(
3595       MachinePointerInfo(),
3596       MachineMemOperand::MOStore,
3597       VT.getStoreSize(), 4);
3598     return DAG.getMemIntrinsicNode(AMDGPUISD::TBUFFER_STORE_FORMAT, DL,
3599                                    Op->getVTList(), Ops, VT, MMO);
3600   }
3601 
3602   default:
3603     return Op;
3604   }
3605 }
3606 
3607 SDValue SITargetLowering::LowerLOAD(SDValue Op, SelectionDAG &DAG) const {
3608   SDLoc DL(Op);
3609   LoadSDNode *Load = cast<LoadSDNode>(Op);
3610   ISD::LoadExtType ExtType = Load->getExtensionType();
3611   EVT MemVT = Load->getMemoryVT();
3612 
3613   if (ExtType == ISD::NON_EXTLOAD && MemVT.getSizeInBits() < 32) {
3614     // FIXME: Copied from PPC
3615     // First, load into 32 bits, then truncate to 1 bit.
3616 
3617     SDValue Chain = Load->getChain();
3618     SDValue BasePtr = Load->getBasePtr();
3619     MachineMemOperand *MMO = Load->getMemOperand();
3620 
3621     EVT RealMemVT = (MemVT == MVT::i1) ? MVT::i8 : MVT::i16;
3622 
3623     SDValue NewLD = DAG.getExtLoad(ISD::EXTLOAD, DL, MVT::i32, Chain,
3624                                    BasePtr, RealMemVT, MMO);
3625 
3626     SDValue Ops[] = {
3627       DAG.getNode(ISD::TRUNCATE, DL, MemVT, NewLD),
3628       NewLD.getValue(1)
3629     };
3630 
3631     return DAG.getMergeValues(Ops, DL);
3632   }
3633 
3634   if (!MemVT.isVector())
3635     return SDValue();
3636 
3637   assert(Op.getValueType().getVectorElementType() == MVT::i32 &&
3638          "Custom lowering for non-i32 vectors hasn't been implemented.");
3639 
3640   unsigned AS = Load->getAddressSpace();
3641   if (!allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), MemVT,
3642                           AS, Load->getAlignment())) {
3643     SDValue Ops[2];
3644     std::tie(Ops[0], Ops[1]) = expandUnalignedLoad(Load, DAG);
3645     return DAG.getMergeValues(Ops, DL);
3646   }
3647 
3648   MachineFunction &MF = DAG.getMachineFunction();
3649   SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
3650   // If there is a possibilty that flat instruction access scratch memory
3651   // then we need to use the same legalization rules we use for private.
3652   if (AS == AMDGPUASI.FLAT_ADDRESS)
3653     AS = MFI->hasFlatScratchInit() ?
3654          AMDGPUASI.PRIVATE_ADDRESS : AMDGPUASI.GLOBAL_ADDRESS;
3655 
3656   unsigned NumElements = MemVT.getVectorNumElements();
3657   if (AS == AMDGPUASI.CONSTANT_ADDRESS) {
3658     if (isMemOpUniform(Load))
3659       return SDValue();
3660     // Non-uniform loads will be selected to MUBUF instructions, so they
3661     // have the same legalization requirements as global and private
3662     // loads.
3663     //
3664   }
3665   if (AS == AMDGPUASI.CONSTANT_ADDRESS || AS == AMDGPUASI.GLOBAL_ADDRESS) {
3666     if (Subtarget->getScalarizeGlobalBehavior() && isMemOpUniform(Load) &&
3667         !Load->isVolatile() && isMemOpHasNoClobberedMemOperand(Load))
3668       return SDValue();
3669     // Non-uniform loads will be selected to MUBUF instructions, so they
3670     // have the same legalization requirements as global and private
3671     // loads.
3672     //
3673   }
3674   if (AS == AMDGPUASI.CONSTANT_ADDRESS || AS == AMDGPUASI.GLOBAL_ADDRESS ||
3675       AS == AMDGPUASI.FLAT_ADDRESS) {
3676     if (NumElements > 4)
3677       return SplitVectorLoad(Op, DAG);
3678     // v4 loads are supported for private and global memory.
3679     return SDValue();
3680   }
3681   if (AS == AMDGPUASI.PRIVATE_ADDRESS) {
3682     // Depending on the setting of the private_element_size field in the
3683     // resource descriptor, we can only make private accesses up to a certain
3684     // size.
3685     switch (Subtarget->getMaxPrivateElementSize()) {
3686     case 4:
3687       return scalarizeVectorLoad(Load, DAG);
3688     case 8:
3689       if (NumElements > 2)
3690         return SplitVectorLoad(Op, DAG);
3691       return SDValue();
3692     case 16:
3693       // Same as global/flat
3694       if (NumElements > 4)
3695         return SplitVectorLoad(Op, DAG);
3696       return SDValue();
3697     default:
3698       llvm_unreachable("unsupported private_element_size");
3699     }
3700   } else if (AS == AMDGPUASI.LOCAL_ADDRESS) {
3701     if (NumElements > 2)
3702       return SplitVectorLoad(Op, DAG);
3703 
3704     if (NumElements == 2)
3705       return SDValue();
3706 
3707     // If properly aligned, if we split we might be able to use ds_read_b64.
3708     return SplitVectorLoad(Op, DAG);
3709   }
3710   return SDValue();
3711 }
3712 
3713 SDValue SITargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const {
3714   if (Op.getValueType() != MVT::i64)
3715     return SDValue();
3716 
3717   SDLoc DL(Op);
3718   SDValue Cond = Op.getOperand(0);
3719 
3720   SDValue Zero = DAG.getConstant(0, DL, MVT::i32);
3721   SDValue One = DAG.getConstant(1, DL, MVT::i32);
3722 
3723   SDValue LHS = DAG.getNode(ISD::BITCAST, DL, MVT::v2i32, Op.getOperand(1));
3724   SDValue RHS = DAG.getNode(ISD::BITCAST, DL, MVT::v2i32, Op.getOperand(2));
3725 
3726   SDValue Lo0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, LHS, Zero);
3727   SDValue Lo1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, RHS, Zero);
3728 
3729   SDValue Lo = DAG.getSelect(DL, MVT::i32, Cond, Lo0, Lo1);
3730 
3731   SDValue Hi0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, LHS, One);
3732   SDValue Hi1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, RHS, One);
3733 
3734   SDValue Hi = DAG.getSelect(DL, MVT::i32, Cond, Hi0, Hi1);
3735 
3736   SDValue Res = DAG.getBuildVector(MVT::v2i32, DL, {Lo, Hi});
3737   return DAG.getNode(ISD::BITCAST, DL, MVT::i64, Res);
3738 }
3739 
3740 // Catch division cases where we can use shortcuts with rcp and rsq
3741 // instructions.
3742 SDValue SITargetLowering::lowerFastUnsafeFDIV(SDValue Op,
3743                                               SelectionDAG &DAG) const {
3744   SDLoc SL(Op);
3745   SDValue LHS = Op.getOperand(0);
3746   SDValue RHS = Op.getOperand(1);
3747   EVT VT = Op.getValueType();
3748   const SDNodeFlags Flags = Op->getFlags();
3749   bool Unsafe = DAG.getTarget().Options.UnsafeFPMath ||
3750                 Flags.hasUnsafeAlgebra() || Flags.hasAllowReciprocal();
3751 
3752   if (!Unsafe && VT == MVT::f32 && Subtarget->hasFP32Denormals())
3753     return SDValue();
3754 
3755   if (const ConstantFPSDNode *CLHS = dyn_cast<ConstantFPSDNode>(LHS)) {
3756     if (Unsafe || VT == MVT::f32 || VT == MVT::f16) {
3757       if (CLHS->isExactlyValue(1.0)) {
3758         // v_rcp_f32 and v_rsq_f32 do not support denormals, and according to
3759         // the CI documentation has a worst case error of 1 ulp.
3760         // OpenCL requires <= 2.5 ulp for 1.0 / x, so it should always be OK to
3761         // use it as long as we aren't trying to use denormals.
3762         //
3763         // v_rcp_f16 and v_rsq_f16 DO support denormals.
3764 
3765         // 1.0 / sqrt(x) -> rsq(x)
3766 
3767         // XXX - Is UnsafeFPMath sufficient to do this for f64? The maximum ULP
3768         // error seems really high at 2^29 ULP.
3769         if (RHS.getOpcode() == ISD::FSQRT)
3770           return DAG.getNode(AMDGPUISD::RSQ, SL, VT, RHS.getOperand(0));
3771 
3772         // 1.0 / x -> rcp(x)
3773         return DAG.getNode(AMDGPUISD::RCP, SL, VT, RHS);
3774       }
3775 
3776       // Same as for 1.0, but expand the sign out of the constant.
3777       if (CLHS->isExactlyValue(-1.0)) {
3778         // -1.0 / x -> rcp (fneg x)
3779         SDValue FNegRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
3780         return DAG.getNode(AMDGPUISD::RCP, SL, VT, FNegRHS);
3781       }
3782     }
3783   }
3784 
3785   if (Unsafe) {
3786     // Turn into multiply by the reciprocal.
3787     // x / y -> x * (1.0 / y)
3788     SDValue Recip = DAG.getNode(AMDGPUISD::RCP, SL, VT, RHS);
3789     return DAG.getNode(ISD::FMUL, SL, VT, LHS, Recip, Flags);
3790   }
3791 
3792   return SDValue();
3793 }
3794 
3795 static SDValue getFPBinOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL,
3796                           EVT VT, SDValue A, SDValue B, SDValue GlueChain) {
3797   if (GlueChain->getNumValues() <= 1) {
3798     return DAG.getNode(Opcode, SL, VT, A, B);
3799   }
3800 
3801   assert(GlueChain->getNumValues() == 3);
3802 
3803   SDVTList VTList = DAG.getVTList(VT, MVT::Other, MVT::Glue);
3804   switch (Opcode) {
3805   default: llvm_unreachable("no chain equivalent for opcode");
3806   case ISD::FMUL:
3807     Opcode = AMDGPUISD::FMUL_W_CHAIN;
3808     break;
3809   }
3810 
3811   return DAG.getNode(Opcode, SL, VTList, GlueChain.getValue(1), A, B,
3812                      GlueChain.getValue(2));
3813 }
3814 
3815 static SDValue getFPTernOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL,
3816                            EVT VT, SDValue A, SDValue B, SDValue C,
3817                            SDValue GlueChain) {
3818   if (GlueChain->getNumValues() <= 1) {
3819     return DAG.getNode(Opcode, SL, VT, A, B, C);
3820   }
3821 
3822   assert(GlueChain->getNumValues() == 3);
3823 
3824   SDVTList VTList = DAG.getVTList(VT, MVT::Other, MVT::Glue);
3825   switch (Opcode) {
3826   default: llvm_unreachable("no chain equivalent for opcode");
3827   case ISD::FMA:
3828     Opcode = AMDGPUISD::FMA_W_CHAIN;
3829     break;
3830   }
3831 
3832   return DAG.getNode(Opcode, SL, VTList, GlueChain.getValue(1), A, B, C,
3833                      GlueChain.getValue(2));
3834 }
3835 
3836 SDValue SITargetLowering::LowerFDIV16(SDValue Op, SelectionDAG &DAG) const {
3837   if (SDValue FastLowered = lowerFastUnsafeFDIV(Op, DAG))
3838     return FastLowered;
3839 
3840   SDLoc SL(Op);
3841   SDValue Src0 = Op.getOperand(0);
3842   SDValue Src1 = Op.getOperand(1);
3843 
3844   SDValue CvtSrc0 = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src0);
3845   SDValue CvtSrc1 = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src1);
3846 
3847   SDValue RcpSrc1 = DAG.getNode(AMDGPUISD::RCP, SL, MVT::f32, CvtSrc1);
3848   SDValue Quot = DAG.getNode(ISD::FMUL, SL, MVT::f32, CvtSrc0, RcpSrc1);
3849 
3850   SDValue FPRoundFlag = DAG.getTargetConstant(0, SL, MVT::i32);
3851   SDValue BestQuot = DAG.getNode(ISD::FP_ROUND, SL, MVT::f16, Quot, FPRoundFlag);
3852 
3853   return DAG.getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f16, BestQuot, Src1, Src0);
3854 }
3855 
3856 // Faster 2.5 ULP division that does not support denormals.
3857 SDValue SITargetLowering::lowerFDIV_FAST(SDValue Op, SelectionDAG &DAG) const {
3858   SDLoc SL(Op);
3859   SDValue LHS = Op.getOperand(1);
3860   SDValue RHS = Op.getOperand(2);
3861 
3862   SDValue r1 = DAG.getNode(ISD::FABS, SL, MVT::f32, RHS);
3863 
3864   const APFloat K0Val(BitsToFloat(0x6f800000));
3865   const SDValue K0 = DAG.getConstantFP(K0Val, SL, MVT::f32);
3866 
3867   const APFloat K1Val(BitsToFloat(0x2f800000));
3868   const SDValue K1 = DAG.getConstantFP(K1Val, SL, MVT::f32);
3869 
3870   const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f32);
3871 
3872   EVT SetCCVT =
3873     getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f32);
3874 
3875   SDValue r2 = DAG.getSetCC(SL, SetCCVT, r1, K0, ISD::SETOGT);
3876 
3877   SDValue r3 = DAG.getNode(ISD::SELECT, SL, MVT::f32, r2, K1, One);
3878 
3879   // TODO: Should this propagate fast-math-flags?
3880   r1 = DAG.getNode(ISD::FMUL, SL, MVT::f32, RHS, r3);
3881 
3882   // rcp does not support denormals.
3883   SDValue r0 = DAG.getNode(AMDGPUISD::RCP, SL, MVT::f32, r1);
3884 
3885   SDValue Mul = DAG.getNode(ISD::FMUL, SL, MVT::f32, LHS, r0);
3886 
3887   return DAG.getNode(ISD::FMUL, SL, MVT::f32, r3, Mul);
3888 }
3889 
3890 SDValue SITargetLowering::LowerFDIV32(SDValue Op, SelectionDAG &DAG) const {
3891   if (SDValue FastLowered = lowerFastUnsafeFDIV(Op, DAG))
3892     return FastLowered;
3893 
3894   SDLoc SL(Op);
3895   SDValue LHS = Op.getOperand(0);
3896   SDValue RHS = Op.getOperand(1);
3897 
3898   const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f32);
3899 
3900   SDVTList ScaleVT = DAG.getVTList(MVT::f32, MVT::i1);
3901 
3902   SDValue DenominatorScaled = DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT,
3903                                           RHS, RHS, LHS);
3904   SDValue NumeratorScaled = DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT,
3905                                         LHS, RHS, LHS);
3906 
3907   // Denominator is scaled to not be denormal, so using rcp is ok.
3908   SDValue ApproxRcp = DAG.getNode(AMDGPUISD::RCP, SL, MVT::f32,
3909                                   DenominatorScaled);
3910   SDValue NegDivScale0 = DAG.getNode(ISD::FNEG, SL, MVT::f32,
3911                                      DenominatorScaled);
3912 
3913   const unsigned Denorm32Reg = AMDGPU::Hwreg::ID_MODE |
3914                                (4 << AMDGPU::Hwreg::OFFSET_SHIFT_) |
3915                                (1 << AMDGPU::Hwreg::WIDTH_M1_SHIFT_);
3916 
3917   const SDValue BitField = DAG.getTargetConstant(Denorm32Reg, SL, MVT::i16);
3918 
3919   if (!Subtarget->hasFP32Denormals()) {
3920     SDVTList BindParamVTs = DAG.getVTList(MVT::Other, MVT::Glue);
3921     const SDValue EnableDenormValue = DAG.getConstant(FP_DENORM_FLUSH_NONE,
3922                                                       SL, MVT::i32);
3923     SDValue EnableDenorm = DAG.getNode(AMDGPUISD::SETREG, SL, BindParamVTs,
3924                                        DAG.getEntryNode(),
3925                                        EnableDenormValue, BitField);
3926     SDValue Ops[3] = {
3927       NegDivScale0,
3928       EnableDenorm.getValue(0),
3929       EnableDenorm.getValue(1)
3930     };
3931 
3932     NegDivScale0 = DAG.getMergeValues(Ops, SL);
3933   }
3934 
3935   SDValue Fma0 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, NegDivScale0,
3936                              ApproxRcp, One, NegDivScale0);
3937 
3938   SDValue Fma1 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, Fma0, ApproxRcp,
3939                              ApproxRcp, Fma0);
3940 
3941   SDValue Mul = getFPBinOp(DAG, ISD::FMUL, SL, MVT::f32, NumeratorScaled,
3942                            Fma1, Fma1);
3943 
3944   SDValue Fma2 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, NegDivScale0, Mul,
3945                              NumeratorScaled, Mul);
3946 
3947   SDValue Fma3 = getFPTernOp(DAG, ISD::FMA,SL, MVT::f32, Fma2, Fma1, Mul, Fma2);
3948 
3949   SDValue Fma4 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, NegDivScale0, Fma3,
3950                              NumeratorScaled, Fma3);
3951 
3952   if (!Subtarget->hasFP32Denormals()) {
3953     const SDValue DisableDenormValue =
3954         DAG.getConstant(FP_DENORM_FLUSH_IN_FLUSH_OUT, SL, MVT::i32);
3955     SDValue DisableDenorm = DAG.getNode(AMDGPUISD::SETREG, SL, MVT::Other,
3956                                         Fma4.getValue(1),
3957                                         DisableDenormValue,
3958                                         BitField,
3959                                         Fma4.getValue(2));
3960 
3961     SDValue OutputChain = DAG.getNode(ISD::TokenFactor, SL, MVT::Other,
3962                                       DisableDenorm, DAG.getRoot());
3963     DAG.setRoot(OutputChain);
3964   }
3965 
3966   SDValue Scale = NumeratorScaled.getValue(1);
3967   SDValue Fmas = DAG.getNode(AMDGPUISD::DIV_FMAS, SL, MVT::f32,
3968                              Fma4, Fma1, Fma3, Scale);
3969 
3970   return DAG.getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f32, Fmas, RHS, LHS);
3971 }
3972 
3973 SDValue SITargetLowering::LowerFDIV64(SDValue Op, SelectionDAG &DAG) const {
3974   if (DAG.getTarget().Options.UnsafeFPMath)
3975     return lowerFastUnsafeFDIV(Op, DAG);
3976 
3977   SDLoc SL(Op);
3978   SDValue X = Op.getOperand(0);
3979   SDValue Y = Op.getOperand(1);
3980 
3981   const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f64);
3982 
3983   SDVTList ScaleVT = DAG.getVTList(MVT::f64, MVT::i1);
3984 
3985   SDValue DivScale0 = DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT, Y, Y, X);
3986 
3987   SDValue NegDivScale0 = DAG.getNode(ISD::FNEG, SL, MVT::f64, DivScale0);
3988 
3989   SDValue Rcp = DAG.getNode(AMDGPUISD::RCP, SL, MVT::f64, DivScale0);
3990 
3991   SDValue Fma0 = DAG.getNode(ISD::FMA, SL, MVT::f64, NegDivScale0, Rcp, One);
3992 
3993   SDValue Fma1 = DAG.getNode(ISD::FMA, SL, MVT::f64, Rcp, Fma0, Rcp);
3994 
3995   SDValue Fma2 = DAG.getNode(ISD::FMA, SL, MVT::f64, NegDivScale0, Fma1, One);
3996 
3997   SDValue DivScale1 = DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT, X, Y, X);
3998 
3999   SDValue Fma3 = DAG.getNode(ISD::FMA, SL, MVT::f64, Fma1, Fma2, Fma1);
4000   SDValue Mul = DAG.getNode(ISD::FMUL, SL, MVT::f64, DivScale1, Fma3);
4001 
4002   SDValue Fma4 = DAG.getNode(ISD::FMA, SL, MVT::f64,
4003                              NegDivScale0, Mul, DivScale1);
4004 
4005   SDValue Scale;
4006 
4007   if (Subtarget->getGeneration() == SISubtarget::SOUTHERN_ISLANDS) {
4008     // Workaround a hardware bug on SI where the condition output from div_scale
4009     // is not usable.
4010 
4011     const SDValue Hi = DAG.getConstant(1, SL, MVT::i32);
4012 
4013     // Figure out if the scale to use for div_fmas.
4014     SDValue NumBC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, X);
4015     SDValue DenBC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Y);
4016     SDValue Scale0BC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, DivScale0);
4017     SDValue Scale1BC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, DivScale1);
4018 
4019     SDValue NumHi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, NumBC, Hi);
4020     SDValue DenHi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, DenBC, Hi);
4021 
4022     SDValue Scale0Hi
4023       = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Scale0BC, Hi);
4024     SDValue Scale1Hi
4025       = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Scale1BC, Hi);
4026 
4027     SDValue CmpDen = DAG.getSetCC(SL, MVT::i1, DenHi, Scale0Hi, ISD::SETEQ);
4028     SDValue CmpNum = DAG.getSetCC(SL, MVT::i1, NumHi, Scale1Hi, ISD::SETEQ);
4029     Scale = DAG.getNode(ISD::XOR, SL, MVT::i1, CmpNum, CmpDen);
4030   } else {
4031     Scale = DivScale1.getValue(1);
4032   }
4033 
4034   SDValue Fmas = DAG.getNode(AMDGPUISD::DIV_FMAS, SL, MVT::f64,
4035                              Fma4, Fma3, Mul, Scale);
4036 
4037   return DAG.getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f64, Fmas, Y, X);
4038 }
4039 
4040 SDValue SITargetLowering::LowerFDIV(SDValue Op, SelectionDAG &DAG) const {
4041   EVT VT = Op.getValueType();
4042 
4043   if (VT == MVT::f32)
4044     return LowerFDIV32(Op, DAG);
4045 
4046   if (VT == MVT::f64)
4047     return LowerFDIV64(Op, DAG);
4048 
4049   if (VT == MVT::f16)
4050     return LowerFDIV16(Op, DAG);
4051 
4052   llvm_unreachable("Unexpected type for fdiv");
4053 }
4054 
4055 SDValue SITargetLowering::LowerSTORE(SDValue Op, SelectionDAG &DAG) const {
4056   SDLoc DL(Op);
4057   StoreSDNode *Store = cast<StoreSDNode>(Op);
4058   EVT VT = Store->getMemoryVT();
4059 
4060   if (VT == MVT::i1) {
4061     return DAG.getTruncStore(Store->getChain(), DL,
4062        DAG.getSExtOrTrunc(Store->getValue(), DL, MVT::i32),
4063        Store->getBasePtr(), MVT::i1, Store->getMemOperand());
4064   }
4065 
4066   assert(VT.isVector() &&
4067          Store->getValue().getValueType().getScalarType() == MVT::i32);
4068 
4069   unsigned AS = Store->getAddressSpace();
4070   if (!allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT,
4071                           AS, Store->getAlignment())) {
4072     return expandUnalignedStore(Store, DAG);
4073   }
4074 
4075   MachineFunction &MF = DAG.getMachineFunction();
4076   SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
4077   // If there is a possibilty that flat instruction access scratch memory
4078   // then we need to use the same legalization rules we use for private.
4079   if (AS == AMDGPUASI.FLAT_ADDRESS)
4080     AS = MFI->hasFlatScratchInit() ?
4081          AMDGPUASI.PRIVATE_ADDRESS : AMDGPUASI.GLOBAL_ADDRESS;
4082 
4083   unsigned NumElements = VT.getVectorNumElements();
4084   if (AS == AMDGPUASI.GLOBAL_ADDRESS ||
4085       AS == AMDGPUASI.FLAT_ADDRESS) {
4086     if (NumElements > 4)
4087       return SplitVectorStore(Op, DAG);
4088     return SDValue();
4089   } else if (AS == AMDGPUASI.PRIVATE_ADDRESS) {
4090     switch (Subtarget->getMaxPrivateElementSize()) {
4091     case 4:
4092       return scalarizeVectorStore(Store, DAG);
4093     case 8:
4094       if (NumElements > 2)
4095         return SplitVectorStore(Op, DAG);
4096       return SDValue();
4097     case 16:
4098       if (NumElements > 4)
4099         return SplitVectorStore(Op, DAG);
4100       return SDValue();
4101     default:
4102       llvm_unreachable("unsupported private_element_size");
4103     }
4104   } else if (AS == AMDGPUASI.LOCAL_ADDRESS) {
4105     if (NumElements > 2)
4106       return SplitVectorStore(Op, DAG);
4107 
4108     if (NumElements == 2)
4109       return Op;
4110 
4111     // If properly aligned, if we split we might be able to use ds_write_b64.
4112     return SplitVectorStore(Op, DAG);
4113   } else {
4114     llvm_unreachable("unhandled address space");
4115   }
4116 }
4117 
4118 SDValue SITargetLowering::LowerTrig(SDValue Op, SelectionDAG &DAG) const {
4119   SDLoc DL(Op);
4120   EVT VT = Op.getValueType();
4121   SDValue Arg = Op.getOperand(0);
4122   // TODO: Should this propagate fast-math-flags?
4123   SDValue FractPart = DAG.getNode(AMDGPUISD::FRACT, DL, VT,
4124                                   DAG.getNode(ISD::FMUL, DL, VT, Arg,
4125                                               DAG.getConstantFP(0.5/M_PI, DL,
4126                                                                 VT)));
4127 
4128   switch (Op.getOpcode()) {
4129   case ISD::FCOS:
4130     return DAG.getNode(AMDGPUISD::COS_HW, SDLoc(Op), VT, FractPart);
4131   case ISD::FSIN:
4132     return DAG.getNode(AMDGPUISD::SIN_HW, SDLoc(Op), VT, FractPart);
4133   default:
4134     llvm_unreachable("Wrong trig opcode");
4135   }
4136 }
4137 
4138 SDValue SITargetLowering::LowerATOMIC_CMP_SWAP(SDValue Op, SelectionDAG &DAG) const {
4139   AtomicSDNode *AtomicNode = cast<AtomicSDNode>(Op);
4140   assert(AtomicNode->isCompareAndSwap());
4141   unsigned AS = AtomicNode->getAddressSpace();
4142 
4143   // No custom lowering required for local address space
4144   if (!isFlatGlobalAddrSpace(AS, AMDGPUASI))
4145     return Op;
4146 
4147   // Non-local address space requires custom lowering for atomic compare
4148   // and swap; cmp and swap should be in a v2i32 or v2i64 in case of _X2
4149   SDLoc DL(Op);
4150   SDValue ChainIn = Op.getOperand(0);
4151   SDValue Addr = Op.getOperand(1);
4152   SDValue Old = Op.getOperand(2);
4153   SDValue New = Op.getOperand(3);
4154   EVT VT = Op.getValueType();
4155   MVT SimpleVT = VT.getSimpleVT();
4156   MVT VecType = MVT::getVectorVT(SimpleVT, 2);
4157 
4158   SDValue NewOld = DAG.getBuildVector(VecType, DL, {New, Old});
4159   SDValue Ops[] = { ChainIn, Addr, NewOld };
4160 
4161   return DAG.getMemIntrinsicNode(AMDGPUISD::ATOMIC_CMP_SWAP, DL, Op->getVTList(),
4162                                  Ops, VT, AtomicNode->getMemOperand());
4163 }
4164 
4165 //===----------------------------------------------------------------------===//
4166 // Custom DAG optimizations
4167 //===----------------------------------------------------------------------===//
4168 
4169 SDValue SITargetLowering::performUCharToFloatCombine(SDNode *N,
4170                                                      DAGCombinerInfo &DCI) const {
4171   EVT VT = N->getValueType(0);
4172   EVT ScalarVT = VT.getScalarType();
4173   if (ScalarVT != MVT::f32)
4174     return SDValue();
4175 
4176   SelectionDAG &DAG = DCI.DAG;
4177   SDLoc DL(N);
4178 
4179   SDValue Src = N->getOperand(0);
4180   EVT SrcVT = Src.getValueType();
4181 
4182   // TODO: We could try to match extracting the higher bytes, which would be
4183   // easier if i8 vectors weren't promoted to i32 vectors, particularly after
4184   // types are legalized. v4i8 -> v4f32 is probably the only case to worry
4185   // about in practice.
4186   if (DCI.isAfterLegalizeVectorOps() && SrcVT == MVT::i32) {
4187     if (DAG.MaskedValueIsZero(Src, APInt::getHighBitsSet(32, 24))) {
4188       SDValue Cvt = DAG.getNode(AMDGPUISD::CVT_F32_UBYTE0, DL, VT, Src);
4189       DCI.AddToWorklist(Cvt.getNode());
4190       return Cvt;
4191     }
4192   }
4193 
4194   return SDValue();
4195 }
4196 
4197 /// \brief Return true if the given offset Size in bytes can be folded into
4198 /// the immediate offsets of a memory instruction for the given address space.
4199 static bool canFoldOffset(unsigned OffsetSize, unsigned AS,
4200                           const SISubtarget &STI) {
4201   auto AMDGPUASI = STI.getAMDGPUAS();
4202   if (AS == AMDGPUASI.GLOBAL_ADDRESS) {
4203     // MUBUF instructions a 12-bit offset in bytes.
4204     return isUInt<12>(OffsetSize);
4205   }
4206   if (AS == AMDGPUASI.CONSTANT_ADDRESS) {
4207     // SMRD instructions have an 8-bit offset in dwords on SI and
4208     // a 20-bit offset in bytes on VI.
4209     if (STI.getGeneration() >= SISubtarget::VOLCANIC_ISLANDS)
4210       return isUInt<20>(OffsetSize);
4211     else
4212       return (OffsetSize % 4 == 0) && isUInt<8>(OffsetSize / 4);
4213   }
4214   if (AS == AMDGPUASI.LOCAL_ADDRESS ||
4215       AS == AMDGPUASI.REGION_ADDRESS) {
4216     // The single offset versions have a 16-bit offset in bytes.
4217     return isUInt<16>(OffsetSize);
4218   }
4219   // Indirect register addressing does not use any offsets.
4220   return false;
4221 }
4222 
4223 // (shl (add x, c1), c2) -> add (shl x, c2), (shl c1, c2)
4224 
4225 // This is a variant of
4226 // (mul (add x, c1), c2) -> add (mul x, c2), (mul c1, c2),
4227 //
4228 // The normal DAG combiner will do this, but only if the add has one use since
4229 // that would increase the number of instructions.
4230 //
4231 // This prevents us from seeing a constant offset that can be folded into a
4232 // memory instruction's addressing mode. If we know the resulting add offset of
4233 // a pointer can be folded into an addressing offset, we can replace the pointer
4234 // operand with the add of new constant offset. This eliminates one of the uses,
4235 // and may allow the remaining use to also be simplified.
4236 //
4237 SDValue SITargetLowering::performSHLPtrCombine(SDNode *N,
4238                                                unsigned AddrSpace,
4239                                                DAGCombinerInfo &DCI) const {
4240   SDValue N0 = N->getOperand(0);
4241   SDValue N1 = N->getOperand(1);
4242 
4243   if (N0.getOpcode() != ISD::ADD)
4244     return SDValue();
4245 
4246   const ConstantSDNode *CN1 = dyn_cast<ConstantSDNode>(N1);
4247   if (!CN1)
4248     return SDValue();
4249 
4250   const ConstantSDNode *CAdd = dyn_cast<ConstantSDNode>(N0.getOperand(1));
4251   if (!CAdd)
4252     return SDValue();
4253 
4254   // If the resulting offset is too large, we can't fold it into the addressing
4255   // mode offset.
4256   APInt Offset = CAdd->getAPIntValue() << CN1->getAPIntValue();
4257   if (!canFoldOffset(Offset.getZExtValue(), AddrSpace, *getSubtarget()))
4258     return SDValue();
4259 
4260   SelectionDAG &DAG = DCI.DAG;
4261   SDLoc SL(N);
4262   EVT VT = N->getValueType(0);
4263 
4264   SDValue ShlX = DAG.getNode(ISD::SHL, SL, VT, N0.getOperand(0), N1);
4265   SDValue COffset = DAG.getConstant(Offset, SL, MVT::i32);
4266 
4267   return DAG.getNode(ISD::ADD, SL, VT, ShlX, COffset);
4268 }
4269 
4270 SDValue SITargetLowering::performMemSDNodeCombine(MemSDNode *N,
4271                                                   DAGCombinerInfo &DCI) const {
4272   SDValue Ptr = N->getBasePtr();
4273   SelectionDAG &DAG = DCI.DAG;
4274   SDLoc SL(N);
4275 
4276   // TODO: We could also do this for multiplies.
4277   unsigned AS = N->getAddressSpace();
4278   if (Ptr.getOpcode() == ISD::SHL && AS != AMDGPUASI.PRIVATE_ADDRESS) {
4279     SDValue NewPtr = performSHLPtrCombine(Ptr.getNode(), AS, DCI);
4280     if (NewPtr) {
4281       SmallVector<SDValue, 8> NewOps(N->op_begin(), N->op_end());
4282 
4283       NewOps[N->getOpcode() == ISD::STORE ? 2 : 1] = NewPtr;
4284       return SDValue(DAG.UpdateNodeOperands(N, NewOps), 0);
4285     }
4286   }
4287 
4288   return SDValue();
4289 }
4290 
4291 static bool bitOpWithConstantIsReducible(unsigned Opc, uint32_t Val) {
4292   return (Opc == ISD::AND && (Val == 0 || Val == 0xffffffff)) ||
4293          (Opc == ISD::OR && (Val == 0xffffffff || Val == 0)) ||
4294          (Opc == ISD::XOR && Val == 0);
4295 }
4296 
4297 // Break up 64-bit bit operation of a constant into two 32-bit and/or/xor. This
4298 // will typically happen anyway for a VALU 64-bit and. This exposes other 32-bit
4299 // integer combine opportunities since most 64-bit operations are decomposed
4300 // this way.  TODO: We won't want this for SALU especially if it is an inline
4301 // immediate.
4302 SDValue SITargetLowering::splitBinaryBitConstantOp(
4303   DAGCombinerInfo &DCI,
4304   const SDLoc &SL,
4305   unsigned Opc, SDValue LHS,
4306   const ConstantSDNode *CRHS) const {
4307   uint64_t Val = CRHS->getZExtValue();
4308   uint32_t ValLo = Lo_32(Val);
4309   uint32_t ValHi = Hi_32(Val);
4310   const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
4311 
4312     if ((bitOpWithConstantIsReducible(Opc, ValLo) ||
4313          bitOpWithConstantIsReducible(Opc, ValHi)) ||
4314         (CRHS->hasOneUse() && !TII->isInlineConstant(CRHS->getAPIntValue()))) {
4315     // If we need to materialize a 64-bit immediate, it will be split up later
4316     // anyway. Avoid creating the harder to understand 64-bit immediate
4317     // materialization.
4318     return splitBinaryBitConstantOpImpl(DCI, SL, Opc, LHS, ValLo, ValHi);
4319   }
4320 
4321   return SDValue();
4322 }
4323 
4324 // Returns true if argument is a boolean value which is not serialized into
4325 // memory or argument and does not require v_cmdmask_b32 to be deserialized.
4326 static bool isBoolSGPR(SDValue V) {
4327   if (V.getValueType() != MVT::i1)
4328     return false;
4329   switch (V.getOpcode()) {
4330   default: break;
4331   case ISD::SETCC:
4332   case ISD::AND:
4333   case ISD::OR:
4334   case ISD::XOR:
4335   case AMDGPUISD::FP_CLASS:
4336     return true;
4337   }
4338   return false;
4339 }
4340 
4341 SDValue SITargetLowering::performAndCombine(SDNode *N,
4342                                             DAGCombinerInfo &DCI) const {
4343   if (DCI.isBeforeLegalize())
4344     return SDValue();
4345 
4346   SelectionDAG &DAG = DCI.DAG;
4347   EVT VT = N->getValueType(0);
4348   SDValue LHS = N->getOperand(0);
4349   SDValue RHS = N->getOperand(1);
4350 
4351 
4352   const ConstantSDNode *CRHS = dyn_cast<ConstantSDNode>(RHS);
4353   if (VT == MVT::i64 && CRHS) {
4354     if (SDValue Split
4355         = splitBinaryBitConstantOp(DCI, SDLoc(N), ISD::AND, LHS, CRHS))
4356       return Split;
4357   }
4358 
4359   if (CRHS && VT == MVT::i32) {
4360     // and (srl x, c), mask => shl (bfe x, nb + c, mask >> nb), nb
4361     // nb = number of trailing zeroes in mask
4362     // It can be optimized out using SDWA for GFX8+ in the SDWA peephole pass,
4363     // given that we are selecting 8 or 16 bit fields starting at byte boundary.
4364     uint64_t Mask = CRHS->getZExtValue();
4365     unsigned Bits = countPopulation(Mask);
4366     if (getSubtarget()->hasSDWA() && LHS->getOpcode() == ISD::SRL &&
4367         (Bits == 8 || Bits == 16) && isShiftedMask_64(Mask) && !(Mask & 1)) {
4368       if (auto *CShift = dyn_cast<ConstantSDNode>(LHS->getOperand(1))) {
4369         unsigned Shift = CShift->getZExtValue();
4370         unsigned NB = CRHS->getAPIntValue().countTrailingZeros();
4371         unsigned Offset = NB + Shift;
4372         if ((Offset & (Bits - 1)) == 0) { // Starts at a byte or word boundary.
4373           SDLoc SL(N);
4374           SDValue BFE = DAG.getNode(AMDGPUISD::BFE_U32, SL, MVT::i32,
4375                                     LHS->getOperand(0),
4376                                     DAG.getConstant(Offset, SL, MVT::i32),
4377                                     DAG.getConstant(Bits, SL, MVT::i32));
4378           EVT NarrowVT = EVT::getIntegerVT(*DAG.getContext(), Bits);
4379           SDValue Ext = DAG.getNode(ISD::AssertZext, SL, VT, BFE,
4380                                     DAG.getValueType(NarrowVT));
4381           SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(LHS), VT, Ext,
4382                                     DAG.getConstant(NB, SDLoc(CRHS), MVT::i32));
4383           return Shl;
4384         }
4385       }
4386     }
4387   }
4388 
4389   // (and (fcmp ord x, x), (fcmp une (fabs x), inf)) ->
4390   // fp_class x, ~(s_nan | q_nan | n_infinity | p_infinity)
4391   if (LHS.getOpcode() == ISD::SETCC && RHS.getOpcode() == ISD::SETCC) {
4392     ISD::CondCode LCC = cast<CondCodeSDNode>(LHS.getOperand(2))->get();
4393     ISD::CondCode RCC = cast<CondCodeSDNode>(RHS.getOperand(2))->get();
4394 
4395     SDValue X = LHS.getOperand(0);
4396     SDValue Y = RHS.getOperand(0);
4397     if (Y.getOpcode() != ISD::FABS || Y.getOperand(0) != X)
4398       return SDValue();
4399 
4400     if (LCC == ISD::SETO) {
4401       if (X != LHS.getOperand(1))
4402         return SDValue();
4403 
4404       if (RCC == ISD::SETUNE) {
4405         const ConstantFPSDNode *C1 = dyn_cast<ConstantFPSDNode>(RHS.getOperand(1));
4406         if (!C1 || !C1->isInfinity() || C1->isNegative())
4407           return SDValue();
4408 
4409         const uint32_t Mask = SIInstrFlags::N_NORMAL |
4410                               SIInstrFlags::N_SUBNORMAL |
4411                               SIInstrFlags::N_ZERO |
4412                               SIInstrFlags::P_ZERO |
4413                               SIInstrFlags::P_SUBNORMAL |
4414                               SIInstrFlags::P_NORMAL;
4415 
4416         static_assert(((~(SIInstrFlags::S_NAN |
4417                           SIInstrFlags::Q_NAN |
4418                           SIInstrFlags::N_INFINITY |
4419                           SIInstrFlags::P_INFINITY)) & 0x3ff) == Mask,
4420                       "mask not equal");
4421 
4422         SDLoc DL(N);
4423         return DAG.getNode(AMDGPUISD::FP_CLASS, DL, MVT::i1,
4424                            X, DAG.getConstant(Mask, DL, MVT::i32));
4425       }
4426     }
4427   }
4428 
4429   if (VT == MVT::i32 &&
4430       (RHS.getOpcode() == ISD::SIGN_EXTEND || LHS.getOpcode() == ISD::SIGN_EXTEND)) {
4431     // and x, (sext cc from i1) => select cc, x, 0
4432     if (RHS.getOpcode() != ISD::SIGN_EXTEND)
4433       std::swap(LHS, RHS);
4434     if (isBoolSGPR(RHS.getOperand(0)))
4435       return DAG.getSelect(SDLoc(N), MVT::i32, RHS.getOperand(0),
4436                            LHS, DAG.getConstant(0, SDLoc(N), MVT::i32));
4437   }
4438 
4439   return SDValue();
4440 }
4441 
4442 SDValue SITargetLowering::performOrCombine(SDNode *N,
4443                                            DAGCombinerInfo &DCI) const {
4444   SelectionDAG &DAG = DCI.DAG;
4445   SDValue LHS = N->getOperand(0);
4446   SDValue RHS = N->getOperand(1);
4447 
4448   EVT VT = N->getValueType(0);
4449   if (VT == MVT::i1) {
4450     // or (fp_class x, c1), (fp_class x, c2) -> fp_class x, (c1 | c2)
4451     if (LHS.getOpcode() == AMDGPUISD::FP_CLASS &&
4452         RHS.getOpcode() == AMDGPUISD::FP_CLASS) {
4453       SDValue Src = LHS.getOperand(0);
4454       if (Src != RHS.getOperand(0))
4455         return SDValue();
4456 
4457       const ConstantSDNode *CLHS = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
4458       const ConstantSDNode *CRHS = dyn_cast<ConstantSDNode>(RHS.getOperand(1));
4459       if (!CLHS || !CRHS)
4460         return SDValue();
4461 
4462       // Only 10 bits are used.
4463       static const uint32_t MaxMask = 0x3ff;
4464 
4465       uint32_t NewMask = (CLHS->getZExtValue() | CRHS->getZExtValue()) & MaxMask;
4466       SDLoc DL(N);
4467       return DAG.getNode(AMDGPUISD::FP_CLASS, DL, MVT::i1,
4468                          Src, DAG.getConstant(NewMask, DL, MVT::i32));
4469     }
4470 
4471     return SDValue();
4472   }
4473 
4474   if (VT != MVT::i64)
4475     return SDValue();
4476 
4477   // TODO: This could be a generic combine with a predicate for extracting the
4478   // high half of an integer being free.
4479 
4480   // (or i64:x, (zero_extend i32:y)) ->
4481   //   i64 (bitcast (v2i32 build_vector (or i32:y, lo_32(x)), hi_32(x)))
4482   if (LHS.getOpcode() == ISD::ZERO_EXTEND &&
4483       RHS.getOpcode() != ISD::ZERO_EXTEND)
4484     std::swap(LHS, RHS);
4485 
4486   if (RHS.getOpcode() == ISD::ZERO_EXTEND) {
4487     SDValue ExtSrc = RHS.getOperand(0);
4488     EVT SrcVT = ExtSrc.getValueType();
4489     if (SrcVT == MVT::i32) {
4490       SDLoc SL(N);
4491       SDValue LowLHS, HiBits;
4492       std::tie(LowLHS, HiBits) = split64BitValue(LHS, DAG);
4493       SDValue LowOr = DAG.getNode(ISD::OR, SL, MVT::i32, LowLHS, ExtSrc);
4494 
4495       DCI.AddToWorklist(LowOr.getNode());
4496       DCI.AddToWorklist(HiBits.getNode());
4497 
4498       SDValue Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32,
4499                                 LowOr, HiBits);
4500       return DAG.getNode(ISD::BITCAST, SL, MVT::i64, Vec);
4501     }
4502   }
4503 
4504   const ConstantSDNode *CRHS = dyn_cast<ConstantSDNode>(N->getOperand(1));
4505   if (CRHS) {
4506     if (SDValue Split
4507           = splitBinaryBitConstantOp(DCI, SDLoc(N), ISD::OR, LHS, CRHS))
4508       return Split;
4509   }
4510 
4511   return SDValue();
4512 }
4513 
4514 SDValue SITargetLowering::performXorCombine(SDNode *N,
4515                                             DAGCombinerInfo &DCI) const {
4516   EVT VT = N->getValueType(0);
4517   if (VT != MVT::i64)
4518     return SDValue();
4519 
4520   SDValue LHS = N->getOperand(0);
4521   SDValue RHS = N->getOperand(1);
4522 
4523   const ConstantSDNode *CRHS = dyn_cast<ConstantSDNode>(RHS);
4524   if (CRHS) {
4525     if (SDValue Split
4526           = splitBinaryBitConstantOp(DCI, SDLoc(N), ISD::XOR, LHS, CRHS))
4527       return Split;
4528   }
4529 
4530   return SDValue();
4531 }
4532 
4533 // Instructions that will be lowered with a final instruction that zeros the
4534 // high result bits.
4535 // XXX - probably only need to list legal operations.
4536 static bool fp16SrcZerosHighBits(unsigned Opc) {
4537   switch (Opc) {
4538   case ISD::FADD:
4539   case ISD::FSUB:
4540   case ISD::FMUL:
4541   case ISD::FDIV:
4542   case ISD::FREM:
4543   case ISD::FMA:
4544   case ISD::FMAD:
4545   case ISD::FCANONICALIZE:
4546   case ISD::FP_ROUND:
4547   case ISD::UINT_TO_FP:
4548   case ISD::SINT_TO_FP:
4549   case ISD::FABS:
4550     // Fabs is lowered to a bit operation, but it's an and which will clear the
4551     // high bits anyway.
4552   case ISD::FSQRT:
4553   case ISD::FSIN:
4554   case ISD::FCOS:
4555   case ISD::FPOWI:
4556   case ISD::FPOW:
4557   case ISD::FLOG:
4558   case ISD::FLOG2:
4559   case ISD::FLOG10:
4560   case ISD::FEXP:
4561   case ISD::FEXP2:
4562   case ISD::FCEIL:
4563   case ISD::FTRUNC:
4564   case ISD::FRINT:
4565   case ISD::FNEARBYINT:
4566   case ISD::FROUND:
4567   case ISD::FFLOOR:
4568   case ISD::FMINNUM:
4569   case ISD::FMAXNUM:
4570   case AMDGPUISD::FRACT:
4571   case AMDGPUISD::CLAMP:
4572   case AMDGPUISD::COS_HW:
4573   case AMDGPUISD::SIN_HW:
4574   case AMDGPUISD::FMIN3:
4575   case AMDGPUISD::FMAX3:
4576   case AMDGPUISD::FMED3:
4577   case AMDGPUISD::FMAD_FTZ:
4578   case AMDGPUISD::RCP:
4579   case AMDGPUISD::RSQ:
4580   case AMDGPUISD::LDEXP:
4581     return true;
4582   default:
4583     // fcopysign, select and others may be lowered to 32-bit bit operations
4584     // which don't zero the high bits.
4585     return false;
4586   }
4587 }
4588 
4589 SDValue SITargetLowering::performZeroExtendCombine(SDNode *N,
4590                                                    DAGCombinerInfo &DCI) const {
4591   if (!Subtarget->has16BitInsts() ||
4592       DCI.getDAGCombineLevel() < AfterLegalizeDAG)
4593     return SDValue();
4594 
4595   EVT VT = N->getValueType(0);
4596   if (VT != MVT::i32)
4597     return SDValue();
4598 
4599   SDValue Src = N->getOperand(0);
4600   if (Src.getValueType() != MVT::i16)
4601     return SDValue();
4602 
4603   // (i32 zext (i16 (bitcast f16:$src))) -> fp16_zext $src
4604   // FIXME: It is not universally true that the high bits are zeroed on gfx9.
4605   if (Src.getOpcode() == ISD::BITCAST) {
4606     SDValue BCSrc = Src.getOperand(0);
4607     if (BCSrc.getValueType() == MVT::f16 &&
4608         fp16SrcZerosHighBits(BCSrc.getOpcode()))
4609       return DCI.DAG.getNode(AMDGPUISD::FP16_ZEXT, SDLoc(N), VT, BCSrc);
4610   }
4611 
4612   return SDValue();
4613 }
4614 
4615 SDValue SITargetLowering::performClassCombine(SDNode *N,
4616                                               DAGCombinerInfo &DCI) const {
4617   SelectionDAG &DAG = DCI.DAG;
4618   SDValue Mask = N->getOperand(1);
4619 
4620   // fp_class x, 0 -> false
4621   if (const ConstantSDNode *CMask = dyn_cast<ConstantSDNode>(Mask)) {
4622     if (CMask->isNullValue())
4623       return DAG.getConstant(0, SDLoc(N), MVT::i1);
4624   }
4625 
4626   if (N->getOperand(0).isUndef())
4627     return DAG.getUNDEF(MVT::i1);
4628 
4629   return SDValue();
4630 }
4631 
4632 static bool isKnownNeverSNan(SelectionDAG &DAG, SDValue Op) {
4633   if (!DAG.getTargetLoweringInfo().hasFloatingPointExceptions())
4634     return true;
4635 
4636   return DAG.isKnownNeverNaN(Op);
4637 }
4638 
4639 static bool isCanonicalized(SelectionDAG &DAG, SDValue Op,
4640                             const SISubtarget *ST, unsigned MaxDepth=5) {
4641   // If source is a result of another standard FP operation it is already in
4642   // canonical form.
4643 
4644   switch (Op.getOpcode()) {
4645   default:
4646     break;
4647 
4648   // These will flush denorms if required.
4649   case ISD::FADD:
4650   case ISD::FSUB:
4651   case ISD::FMUL:
4652   case ISD::FSQRT:
4653   case ISD::FCEIL:
4654   case ISD::FFLOOR:
4655   case ISD::FMA:
4656   case ISD::FMAD:
4657 
4658   case ISD::FCANONICALIZE:
4659     return true;
4660 
4661   case ISD::FP_ROUND:
4662     return Op.getValueType().getScalarType() != MVT::f16 ||
4663            ST->hasFP16Denormals();
4664 
4665   case ISD::FP_EXTEND:
4666     return Op.getOperand(0).getValueType().getScalarType() != MVT::f16 ||
4667            ST->hasFP16Denormals();
4668 
4669   case ISD::FP16_TO_FP:
4670   case ISD::FP_TO_FP16:
4671     return ST->hasFP16Denormals();
4672 
4673   // It can/will be lowered or combined as a bit operation.
4674   // Need to check their input recursively to handle.
4675   case ISD::FNEG:
4676   case ISD::FABS:
4677     return (MaxDepth > 0) &&
4678            isCanonicalized(DAG, Op.getOperand(0), ST, MaxDepth - 1);
4679 
4680   case ISD::FSIN:
4681   case ISD::FCOS:
4682   case ISD::FSINCOS:
4683     return Op.getValueType().getScalarType() != MVT::f16;
4684 
4685   // In pre-GFX9 targets V_MIN_F32 and others do not flush denorms.
4686   // For such targets need to check their input recursively.
4687   case ISD::FMINNUM:
4688   case ISD::FMAXNUM:
4689   case ISD::FMINNAN:
4690   case ISD::FMAXNAN:
4691 
4692     if (ST->supportsMinMaxDenormModes() &&
4693         DAG.isKnownNeverNaN(Op.getOperand(0)) &&
4694         DAG.isKnownNeverNaN(Op.getOperand(1)))
4695       return true;
4696 
4697     return (MaxDepth > 0) &&
4698            isCanonicalized(DAG, Op.getOperand(0), ST, MaxDepth - 1) &&
4699            isCanonicalized(DAG, Op.getOperand(1), ST, MaxDepth - 1);
4700 
4701   case ISD::ConstantFP: {
4702     auto F = cast<ConstantFPSDNode>(Op)->getValueAPF();
4703     return !F.isDenormal() && !(F.isNaN() && F.isSignaling());
4704   }
4705   }
4706   return false;
4707 }
4708 
4709 // Constant fold canonicalize.
4710 SDValue SITargetLowering::performFCanonicalizeCombine(
4711   SDNode *N,
4712   DAGCombinerInfo &DCI) const {
4713   SelectionDAG &DAG = DCI.DAG;
4714   ConstantFPSDNode *CFP = isConstOrConstSplatFP(N->getOperand(0));
4715 
4716   if (!CFP) {
4717     SDValue N0 = N->getOperand(0);
4718     EVT VT = N0.getValueType().getScalarType();
4719     auto ST = getSubtarget();
4720 
4721     if (((VT == MVT::f32 && ST->hasFP32Denormals()) ||
4722          (VT == MVT::f64 && ST->hasFP64Denormals()) ||
4723          (VT == MVT::f16 && ST->hasFP16Denormals())) &&
4724         DAG.isKnownNeverNaN(N0))
4725       return N0;
4726 
4727     bool IsIEEEMode = Subtarget->enableIEEEBit(DAG.getMachineFunction());
4728 
4729     if ((IsIEEEMode || isKnownNeverSNan(DAG, N0)) &&
4730         isCanonicalized(DAG, N0, ST))
4731       return N0;
4732 
4733     return SDValue();
4734   }
4735 
4736   const APFloat &C = CFP->getValueAPF();
4737 
4738   // Flush denormals to 0 if not enabled.
4739   if (C.isDenormal()) {
4740     EVT VT = N->getValueType(0);
4741     EVT SVT = VT.getScalarType();
4742     if (SVT == MVT::f32 && !Subtarget->hasFP32Denormals())
4743       return DAG.getConstantFP(0.0, SDLoc(N), VT);
4744 
4745     if (SVT == MVT::f64 && !Subtarget->hasFP64Denormals())
4746       return DAG.getConstantFP(0.0, SDLoc(N), VT);
4747 
4748     if (SVT == MVT::f16 && !Subtarget->hasFP16Denormals())
4749       return DAG.getConstantFP(0.0, SDLoc(N), VT);
4750   }
4751 
4752   if (C.isNaN()) {
4753     EVT VT = N->getValueType(0);
4754     APFloat CanonicalQNaN = APFloat::getQNaN(C.getSemantics());
4755     if (C.isSignaling()) {
4756       // Quiet a signaling NaN.
4757       return DAG.getConstantFP(CanonicalQNaN, SDLoc(N), VT);
4758     }
4759 
4760     // Make sure it is the canonical NaN bitpattern.
4761     //
4762     // TODO: Can we use -1 as the canonical NaN value since it's an inline
4763     // immediate?
4764     if (C.bitcastToAPInt() != CanonicalQNaN.bitcastToAPInt())
4765       return DAG.getConstantFP(CanonicalQNaN, SDLoc(N), VT);
4766   }
4767 
4768   return N->getOperand(0);
4769 }
4770 
4771 static unsigned minMaxOpcToMin3Max3Opc(unsigned Opc) {
4772   switch (Opc) {
4773   case ISD::FMAXNUM:
4774     return AMDGPUISD::FMAX3;
4775   case ISD::SMAX:
4776     return AMDGPUISD::SMAX3;
4777   case ISD::UMAX:
4778     return AMDGPUISD::UMAX3;
4779   case ISD::FMINNUM:
4780     return AMDGPUISD::FMIN3;
4781   case ISD::SMIN:
4782     return AMDGPUISD::SMIN3;
4783   case ISD::UMIN:
4784     return AMDGPUISD::UMIN3;
4785   default:
4786     llvm_unreachable("Not a min/max opcode");
4787   }
4788 }
4789 
4790 SDValue SITargetLowering::performIntMed3ImmCombine(
4791   SelectionDAG &DAG, const SDLoc &SL,
4792   SDValue Op0, SDValue Op1, bool Signed) const {
4793   ConstantSDNode *K1 = dyn_cast<ConstantSDNode>(Op1);
4794   if (!K1)
4795     return SDValue();
4796 
4797   ConstantSDNode *K0 = dyn_cast<ConstantSDNode>(Op0.getOperand(1));
4798   if (!K0)
4799     return SDValue();
4800 
4801   if (Signed) {
4802     if (K0->getAPIntValue().sge(K1->getAPIntValue()))
4803       return SDValue();
4804   } else {
4805     if (K0->getAPIntValue().uge(K1->getAPIntValue()))
4806       return SDValue();
4807   }
4808 
4809   EVT VT = K0->getValueType(0);
4810   unsigned Med3Opc = Signed ? AMDGPUISD::SMED3 : AMDGPUISD::UMED3;
4811   if (VT == MVT::i32 || (VT == MVT::i16 && Subtarget->hasMed3_16())) {
4812     return DAG.getNode(Med3Opc, SL, VT,
4813                        Op0.getOperand(0), SDValue(K0, 0), SDValue(K1, 0));
4814   }
4815 
4816   // If there isn't a 16-bit med3 operation, convert to 32-bit.
4817   MVT NVT = MVT::i32;
4818   unsigned ExtOp = Signed ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
4819 
4820   SDValue Tmp1 = DAG.getNode(ExtOp, SL, NVT, Op0->getOperand(0));
4821   SDValue Tmp2 = DAG.getNode(ExtOp, SL, NVT, Op0->getOperand(1));
4822   SDValue Tmp3 = DAG.getNode(ExtOp, SL, NVT, Op1);
4823 
4824   SDValue Med3 = DAG.getNode(Med3Opc, SL, NVT, Tmp1, Tmp2, Tmp3);
4825   return DAG.getNode(ISD::TRUNCATE, SL, VT, Med3);
4826 }
4827 
4828 SDValue SITargetLowering::performFPMed3ImmCombine(SelectionDAG &DAG,
4829                                                   const SDLoc &SL,
4830                                                   SDValue Op0,
4831                                                   SDValue Op1) const {
4832   ConstantFPSDNode *K1 = dyn_cast<ConstantFPSDNode>(Op1);
4833   if (!K1)
4834     return SDValue();
4835 
4836   ConstantFPSDNode *K0 = dyn_cast<ConstantFPSDNode>(Op0.getOperand(1));
4837   if (!K0)
4838     return SDValue();
4839 
4840   // Ordered >= (although NaN inputs should have folded away by now).
4841   APFloat::cmpResult Cmp = K0->getValueAPF().compare(K1->getValueAPF());
4842   if (Cmp == APFloat::cmpGreaterThan)
4843     return SDValue();
4844 
4845   // TODO: Check IEEE bit enabled?
4846   EVT VT = K0->getValueType(0);
4847   if (Subtarget->enableDX10Clamp()) {
4848     // If dx10_clamp is enabled, NaNs clamp to 0.0. This is the same as the
4849     // hardware fmed3 behavior converting to a min.
4850     // FIXME: Should this be allowing -0.0?
4851     if (K1->isExactlyValue(1.0) && K0->isExactlyValue(0.0))
4852       return DAG.getNode(AMDGPUISD::CLAMP, SL, VT, Op0.getOperand(0));
4853   }
4854 
4855   // med3 for f16 is only available on gfx9+.
4856   if (VT == MVT::f64 || (VT == MVT::f16 && !Subtarget->hasMed3_16()))
4857     return SDValue();
4858 
4859   // This isn't safe with signaling NaNs because in IEEE mode, min/max on a
4860   // signaling NaN gives a quiet NaN. The quiet NaN input to the min would then
4861   // give the other result, which is different from med3 with a NaN input.
4862   SDValue Var = Op0.getOperand(0);
4863   if (!isKnownNeverSNan(DAG, Var))
4864     return SDValue();
4865 
4866   return DAG.getNode(AMDGPUISD::FMED3, SL, K0->getValueType(0),
4867                      Var, SDValue(K0, 0), SDValue(K1, 0));
4868 }
4869 
4870 SDValue SITargetLowering::performMinMaxCombine(SDNode *N,
4871                                                DAGCombinerInfo &DCI) const {
4872   SelectionDAG &DAG = DCI.DAG;
4873 
4874   EVT VT = N->getValueType(0);
4875   unsigned Opc = N->getOpcode();
4876   SDValue Op0 = N->getOperand(0);
4877   SDValue Op1 = N->getOperand(1);
4878 
4879   // Only do this if the inner op has one use since this will just increases
4880   // register pressure for no benefit.
4881 
4882 
4883   if (Opc != AMDGPUISD::FMIN_LEGACY && Opc != AMDGPUISD::FMAX_LEGACY &&
4884       VT != MVT::f64 &&
4885       ((VT != MVT::f16 && VT != MVT::i16) || Subtarget->hasMin3Max3_16())) {
4886     // max(max(a, b), c) -> max3(a, b, c)
4887     // min(min(a, b), c) -> min3(a, b, c)
4888     if (Op0.getOpcode() == Opc && Op0.hasOneUse()) {
4889       SDLoc DL(N);
4890       return DAG.getNode(minMaxOpcToMin3Max3Opc(Opc),
4891                          DL,
4892                          N->getValueType(0),
4893                          Op0.getOperand(0),
4894                          Op0.getOperand(1),
4895                          Op1);
4896     }
4897 
4898     // Try commuted.
4899     // max(a, max(b, c)) -> max3(a, b, c)
4900     // min(a, min(b, c)) -> min3(a, b, c)
4901     if (Op1.getOpcode() == Opc && Op1.hasOneUse()) {
4902       SDLoc DL(N);
4903       return DAG.getNode(minMaxOpcToMin3Max3Opc(Opc),
4904                          DL,
4905                          N->getValueType(0),
4906                          Op0,
4907                          Op1.getOperand(0),
4908                          Op1.getOperand(1));
4909     }
4910   }
4911 
4912   // min(max(x, K0), K1), K0 < K1 -> med3(x, K0, K1)
4913   if (Opc == ISD::SMIN && Op0.getOpcode() == ISD::SMAX && Op0.hasOneUse()) {
4914     if (SDValue Med3 = performIntMed3ImmCombine(DAG, SDLoc(N), Op0, Op1, true))
4915       return Med3;
4916   }
4917 
4918   if (Opc == ISD::UMIN && Op0.getOpcode() == ISD::UMAX && Op0.hasOneUse()) {
4919     if (SDValue Med3 = performIntMed3ImmCombine(DAG, SDLoc(N), Op0, Op1, false))
4920       return Med3;
4921   }
4922 
4923   // fminnum(fmaxnum(x, K0), K1), K0 < K1 && !is_snan(x) -> fmed3(x, K0, K1)
4924   if (((Opc == ISD::FMINNUM && Op0.getOpcode() == ISD::FMAXNUM) ||
4925        (Opc == AMDGPUISD::FMIN_LEGACY &&
4926         Op0.getOpcode() == AMDGPUISD::FMAX_LEGACY)) &&
4927       (VT == MVT::f32 || VT == MVT::f64 ||
4928        (VT == MVT::f16 && Subtarget->has16BitInsts())) &&
4929       Op0.hasOneUse()) {
4930     if (SDValue Res = performFPMed3ImmCombine(DAG, SDLoc(N), Op0, Op1))
4931       return Res;
4932   }
4933 
4934   return SDValue();
4935 }
4936 
4937 static bool isClampZeroToOne(SDValue A, SDValue B) {
4938   if (ConstantFPSDNode *CA = dyn_cast<ConstantFPSDNode>(A)) {
4939     if (ConstantFPSDNode *CB = dyn_cast<ConstantFPSDNode>(B)) {
4940       // FIXME: Should this be allowing -0.0?
4941       return (CA->isExactlyValue(0.0) && CB->isExactlyValue(1.0)) ||
4942              (CA->isExactlyValue(1.0) && CB->isExactlyValue(0.0));
4943     }
4944   }
4945 
4946   return false;
4947 }
4948 
4949 // FIXME: Should only worry about snans for version with chain.
4950 SDValue SITargetLowering::performFMed3Combine(SDNode *N,
4951                                               DAGCombinerInfo &DCI) const {
4952   EVT VT = N->getValueType(0);
4953   // v_med3_f32 and v_max_f32 behave identically wrt denorms, exceptions and
4954   // NaNs. With a NaN input, the order of the operands may change the result.
4955 
4956   SelectionDAG &DAG = DCI.DAG;
4957   SDLoc SL(N);
4958 
4959   SDValue Src0 = N->getOperand(0);
4960   SDValue Src1 = N->getOperand(1);
4961   SDValue Src2 = N->getOperand(2);
4962 
4963   if (isClampZeroToOne(Src0, Src1)) {
4964     // const_a, const_b, x -> clamp is safe in all cases including signaling
4965     // nans.
4966     // FIXME: Should this be allowing -0.0?
4967     return DAG.getNode(AMDGPUISD::CLAMP, SL, VT, Src2);
4968   }
4969 
4970   // FIXME: dx10_clamp behavior assumed in instcombine. Should we really bother
4971   // handling no dx10-clamp?
4972   if (Subtarget->enableDX10Clamp()) {
4973     // If NaNs is clamped to 0, we are free to reorder the inputs.
4974 
4975     if (isa<ConstantFPSDNode>(Src0) && !isa<ConstantFPSDNode>(Src1))
4976       std::swap(Src0, Src1);
4977 
4978     if (isa<ConstantFPSDNode>(Src1) && !isa<ConstantFPSDNode>(Src2))
4979       std::swap(Src1, Src2);
4980 
4981     if (isa<ConstantFPSDNode>(Src0) && !isa<ConstantFPSDNode>(Src1))
4982       std::swap(Src0, Src1);
4983 
4984     if (isClampZeroToOne(Src1, Src2))
4985       return DAG.getNode(AMDGPUISD::CLAMP, SL, VT, Src0);
4986   }
4987 
4988   return SDValue();
4989 }
4990 
4991 SDValue SITargetLowering::performCvtPkRTZCombine(SDNode *N,
4992                                                  DAGCombinerInfo &DCI) const {
4993   SDValue Src0 = N->getOperand(0);
4994   SDValue Src1 = N->getOperand(1);
4995   if (Src0.isUndef() && Src1.isUndef())
4996     return DCI.DAG.getUNDEF(N->getValueType(0));
4997   return SDValue();
4998 }
4999 
5000 SDValue SITargetLowering::performExtractVectorEltCombine(
5001   SDNode *N, DAGCombinerInfo &DCI) const {
5002   SDValue Vec = N->getOperand(0);
5003 
5004   SelectionDAG &DAG= DCI.DAG;
5005   if (Vec.getOpcode() == ISD::FNEG && allUsesHaveSourceMods(N)) {
5006     SDLoc SL(N);
5007     EVT EltVT = N->getValueType(0);
5008     SDValue Idx = N->getOperand(1);
5009     SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, EltVT,
5010                               Vec.getOperand(0), Idx);
5011     return DAG.getNode(ISD::FNEG, SL, EltVT, Elt);
5012   }
5013 
5014   return SDValue();
5015 }
5016 
5017 
5018 unsigned SITargetLowering::getFusedOpcode(const SelectionDAG &DAG,
5019                                           const SDNode *N0,
5020                                           const SDNode *N1) const {
5021   EVT VT = N0->getValueType(0);
5022 
5023   // Only do this if we are not trying to support denormals. v_mad_f32 does not
5024   // support denormals ever.
5025   if ((VT == MVT::f32 && !Subtarget->hasFP32Denormals()) ||
5026       (VT == MVT::f16 && !Subtarget->hasFP16Denormals()))
5027     return ISD::FMAD;
5028 
5029   const TargetOptions &Options = DAG.getTarget().Options;
5030   if ((Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath ||
5031        (N0->getFlags().hasUnsafeAlgebra() &&
5032         N1->getFlags().hasUnsafeAlgebra())) &&
5033       isFMAFasterThanFMulAndFAdd(VT)) {
5034     return ISD::FMA;
5035   }
5036 
5037   return 0;
5038 }
5039 
5040 SDValue SITargetLowering::performAddCombine(SDNode *N,
5041                                             DAGCombinerInfo &DCI) const {
5042   SelectionDAG &DAG = DCI.DAG;
5043   EVT VT = N->getValueType(0);
5044 
5045   if (VT != MVT::i32)
5046     return SDValue();
5047 
5048   SDLoc SL(N);
5049   SDValue LHS = N->getOperand(0);
5050   SDValue RHS = N->getOperand(1);
5051 
5052   // add x, zext (setcc) => addcarry x, 0, setcc
5053   // add x, sext (setcc) => subcarry x, 0, setcc
5054   unsigned Opc = LHS.getOpcode();
5055   if (Opc == ISD::ZERO_EXTEND || Opc == ISD::SIGN_EXTEND ||
5056       Opc == ISD::ANY_EXTEND || Opc == ISD::ADDCARRY)
5057     std::swap(RHS, LHS);
5058 
5059   Opc = RHS.getOpcode();
5060   switch (Opc) {
5061   default: break;
5062   case ISD::ZERO_EXTEND:
5063   case ISD::SIGN_EXTEND:
5064   case ISD::ANY_EXTEND: {
5065     auto Cond = RHS.getOperand(0);
5066     if (!isBoolSGPR(Cond))
5067       break;
5068     SDVTList VTList = DAG.getVTList(MVT::i32, MVT::i1);
5069     SDValue Args[] = { LHS, DAG.getConstant(0, SL, MVT::i32), Cond };
5070     Opc = (Opc == ISD::SIGN_EXTEND) ? ISD::SUBCARRY : ISD::ADDCARRY;
5071     return DAG.getNode(Opc, SL, VTList, Args);
5072   }
5073   case ISD::ADDCARRY: {
5074     // add x, (addcarry y, 0, cc) => addcarry x, y, cc
5075     auto C = dyn_cast<ConstantSDNode>(RHS.getOperand(1));
5076     if (!C || C->getZExtValue() != 0) break;
5077     SDValue Args[] = { LHS, RHS.getOperand(0), RHS.getOperand(2) };
5078     return DAG.getNode(ISD::ADDCARRY, SDLoc(N), RHS->getVTList(), Args);
5079   }
5080   }
5081   return SDValue();
5082 }
5083 
5084 SDValue SITargetLowering::performSubCombine(SDNode *N,
5085                                             DAGCombinerInfo &DCI) const {
5086   SelectionDAG &DAG = DCI.DAG;
5087   EVT VT = N->getValueType(0);
5088 
5089   if (VT != MVT::i32)
5090     return SDValue();
5091 
5092   SDLoc SL(N);
5093   SDValue LHS = N->getOperand(0);
5094   SDValue RHS = N->getOperand(1);
5095 
5096   unsigned Opc = LHS.getOpcode();
5097   if (Opc != ISD::SUBCARRY)
5098     std::swap(RHS, LHS);
5099 
5100   if (LHS.getOpcode() == ISD::SUBCARRY) {
5101     // sub (subcarry x, 0, cc), y => subcarry x, y, cc
5102     auto C = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
5103     if (!C || C->getZExtValue() != 0)
5104       return SDValue();
5105     SDValue Args[] = { LHS.getOperand(0), RHS, LHS.getOperand(2) };
5106     return DAG.getNode(ISD::SUBCARRY, SDLoc(N), LHS->getVTList(), Args);
5107   }
5108   return SDValue();
5109 }
5110 
5111 SDValue SITargetLowering::performAddCarrySubCarryCombine(SDNode *N,
5112   DAGCombinerInfo &DCI) const {
5113 
5114   if (N->getValueType(0) != MVT::i32)
5115     return SDValue();
5116 
5117   auto C = dyn_cast<ConstantSDNode>(N->getOperand(1));
5118   if (!C || C->getZExtValue() != 0)
5119     return SDValue();
5120 
5121   SelectionDAG &DAG = DCI.DAG;
5122   SDValue LHS = N->getOperand(0);
5123 
5124   // addcarry (add x, y), 0, cc => addcarry x, y, cc
5125   // subcarry (sub x, y), 0, cc => subcarry x, y, cc
5126   unsigned LHSOpc = LHS.getOpcode();
5127   unsigned Opc = N->getOpcode();
5128   if ((LHSOpc == ISD::ADD && Opc == ISD::ADDCARRY) ||
5129       (LHSOpc == ISD::SUB && Opc == ISD::SUBCARRY)) {
5130     SDValue Args[] = { LHS.getOperand(0), LHS.getOperand(1), N->getOperand(2) };
5131     return DAG.getNode(Opc, SDLoc(N), N->getVTList(), Args);
5132   }
5133   return SDValue();
5134 }
5135 
5136 SDValue SITargetLowering::performFAddCombine(SDNode *N,
5137                                              DAGCombinerInfo &DCI) const {
5138   if (DCI.getDAGCombineLevel() < AfterLegalizeDAG)
5139     return SDValue();
5140 
5141   SelectionDAG &DAG = DCI.DAG;
5142   EVT VT = N->getValueType(0);
5143 
5144   SDLoc SL(N);
5145   SDValue LHS = N->getOperand(0);
5146   SDValue RHS = N->getOperand(1);
5147 
5148   // These should really be instruction patterns, but writing patterns with
5149   // source modiifiers is a pain.
5150 
5151   // fadd (fadd (a, a), b) -> mad 2.0, a, b
5152   if (LHS.getOpcode() == ISD::FADD) {
5153     SDValue A = LHS.getOperand(0);
5154     if (A == LHS.getOperand(1)) {
5155       unsigned FusedOp = getFusedOpcode(DAG, N, LHS.getNode());
5156       if (FusedOp != 0) {
5157         const SDValue Two = DAG.getConstantFP(2.0, SL, VT);
5158         return DAG.getNode(FusedOp, SL, VT, A, Two, RHS);
5159       }
5160     }
5161   }
5162 
5163   // fadd (b, fadd (a, a)) -> mad 2.0, a, b
5164   if (RHS.getOpcode() == ISD::FADD) {
5165     SDValue A = RHS.getOperand(0);
5166     if (A == RHS.getOperand(1)) {
5167       unsigned FusedOp = getFusedOpcode(DAG, N, RHS.getNode());
5168       if (FusedOp != 0) {
5169         const SDValue Two = DAG.getConstantFP(2.0, SL, VT);
5170         return DAG.getNode(FusedOp, SL, VT, A, Two, LHS);
5171       }
5172     }
5173   }
5174 
5175   return SDValue();
5176 }
5177 
5178 SDValue SITargetLowering::performFSubCombine(SDNode *N,
5179                                              DAGCombinerInfo &DCI) const {
5180   if (DCI.getDAGCombineLevel() < AfterLegalizeDAG)
5181     return SDValue();
5182 
5183   SelectionDAG &DAG = DCI.DAG;
5184   SDLoc SL(N);
5185   EVT VT = N->getValueType(0);
5186   assert(!VT.isVector());
5187 
5188   // Try to get the fneg to fold into the source modifier. This undoes generic
5189   // DAG combines and folds them into the mad.
5190   //
5191   // Only do this if we are not trying to support denormals. v_mad_f32 does
5192   // not support denormals ever.
5193   SDValue LHS = N->getOperand(0);
5194   SDValue RHS = N->getOperand(1);
5195   if (LHS.getOpcode() == ISD::FADD) {
5196     // (fsub (fadd a, a), c) -> mad 2.0, a, (fneg c)
5197     SDValue A = LHS.getOperand(0);
5198     if (A == LHS.getOperand(1)) {
5199       unsigned FusedOp = getFusedOpcode(DAG, N, LHS.getNode());
5200       if (FusedOp != 0){
5201         const SDValue Two = DAG.getConstantFP(2.0, SL, VT);
5202         SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5203 
5204         return DAG.getNode(FusedOp, SL, VT, A, Two, NegRHS);
5205       }
5206     }
5207   }
5208 
5209   if (RHS.getOpcode() == ISD::FADD) {
5210     // (fsub c, (fadd a, a)) -> mad -2.0, a, c
5211 
5212     SDValue A = RHS.getOperand(0);
5213     if (A == RHS.getOperand(1)) {
5214       unsigned FusedOp = getFusedOpcode(DAG, N, RHS.getNode());
5215       if (FusedOp != 0){
5216         const SDValue NegTwo = DAG.getConstantFP(-2.0, SL, VT);
5217         return DAG.getNode(FusedOp, SL, VT, A, NegTwo, LHS);
5218       }
5219     }
5220   }
5221 
5222   return SDValue();
5223 }
5224 
5225 SDValue SITargetLowering::performSetCCCombine(SDNode *N,
5226                                               DAGCombinerInfo &DCI) const {
5227   SelectionDAG &DAG = DCI.DAG;
5228   SDLoc SL(N);
5229 
5230   SDValue LHS = N->getOperand(0);
5231   SDValue RHS = N->getOperand(1);
5232   EVT VT = LHS.getValueType();
5233   ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
5234 
5235   auto CRHS = dyn_cast<ConstantSDNode>(RHS);
5236   if (!CRHS) {
5237     CRHS = dyn_cast<ConstantSDNode>(LHS);
5238     if (CRHS) {
5239       std::swap(LHS, RHS);
5240       CC = getSetCCSwappedOperands(CC);
5241     }
5242   }
5243 
5244   if (CRHS && VT == MVT::i32 && LHS.getOpcode() == ISD::SIGN_EXTEND &&
5245       isBoolSGPR(LHS.getOperand(0))) {
5246     // setcc (sext from i1 cc), -1, ne|sgt|ult) => not cc => xor cc, -1
5247     // setcc (sext from i1 cc), -1, eq|sle|uge) => cc
5248     // setcc (sext from i1 cc),  0, eq|sge|ule) => not cc => xor cc, -1
5249     // setcc (sext from i1 cc),  0, ne|ugt|slt) => cc
5250     if ((CRHS->isAllOnesValue() &&
5251          (CC == ISD::SETNE || CC == ISD::SETGT || CC == ISD::SETULT)) ||
5252         (CRHS->isNullValue() &&
5253          (CC == ISD::SETEQ || CC == ISD::SETGE || CC == ISD::SETULE)))
5254       return DAG.getNode(ISD::XOR, SL, MVT::i1, LHS.getOperand(0),
5255                          DAG.getConstant(-1, SL, MVT::i1));
5256     if ((CRHS->isAllOnesValue() &&
5257          (CC == ISD::SETEQ || CC == ISD::SETLE || CC == ISD::SETUGE)) ||
5258         (CRHS->isNullValue() &&
5259          (CC == ISD::SETNE || CC == ISD::SETUGT || CC == ISD::SETLT)))
5260       return LHS.getOperand(0);
5261   }
5262 
5263   if (VT != MVT::f32 && VT != MVT::f64 && (Subtarget->has16BitInsts() &&
5264                                            VT != MVT::f16))
5265     return SDValue();
5266 
5267   // Match isinf pattern
5268   // (fcmp oeq (fabs x), inf) -> (fp_class x, (p_infinity | n_infinity))
5269   if (CC == ISD::SETOEQ && LHS.getOpcode() == ISD::FABS) {
5270     const ConstantFPSDNode *CRHS = dyn_cast<ConstantFPSDNode>(RHS);
5271     if (!CRHS)
5272       return SDValue();
5273 
5274     const APFloat &APF = CRHS->getValueAPF();
5275     if (APF.isInfinity() && !APF.isNegative()) {
5276       unsigned Mask = SIInstrFlags::P_INFINITY | SIInstrFlags::N_INFINITY;
5277       return DAG.getNode(AMDGPUISD::FP_CLASS, SL, MVT::i1, LHS.getOperand(0),
5278                          DAG.getConstant(Mask, SL, MVT::i32));
5279     }
5280   }
5281 
5282   return SDValue();
5283 }
5284 
5285 SDValue SITargetLowering::performCvtF32UByteNCombine(SDNode *N,
5286                                                      DAGCombinerInfo &DCI) const {
5287   SelectionDAG &DAG = DCI.DAG;
5288   SDLoc SL(N);
5289   unsigned Offset = N->getOpcode() - AMDGPUISD::CVT_F32_UBYTE0;
5290 
5291   SDValue Src = N->getOperand(0);
5292   SDValue Srl = N->getOperand(0);
5293   if (Srl.getOpcode() == ISD::ZERO_EXTEND)
5294     Srl = Srl.getOperand(0);
5295 
5296   // TODO: Handle (or x, (srl y, 8)) pattern when known bits are zero.
5297   if (Srl.getOpcode() == ISD::SRL) {
5298     // cvt_f32_ubyte0 (srl x, 16) -> cvt_f32_ubyte2 x
5299     // cvt_f32_ubyte1 (srl x, 16) -> cvt_f32_ubyte3 x
5300     // cvt_f32_ubyte0 (srl x, 8) -> cvt_f32_ubyte1 x
5301 
5302     if (const ConstantSDNode *C =
5303         dyn_cast<ConstantSDNode>(Srl.getOperand(1))) {
5304       Srl = DAG.getZExtOrTrunc(Srl.getOperand(0), SDLoc(Srl.getOperand(0)),
5305                                EVT(MVT::i32));
5306 
5307       unsigned SrcOffset = C->getZExtValue() + 8 * Offset;
5308       if (SrcOffset < 32 && SrcOffset % 8 == 0) {
5309         return DAG.getNode(AMDGPUISD::CVT_F32_UBYTE0 + SrcOffset / 8, SL,
5310                            MVT::f32, Srl);
5311       }
5312     }
5313   }
5314 
5315   APInt Demanded = APInt::getBitsSet(32, 8 * Offset, 8 * Offset + 8);
5316 
5317   KnownBits Known;
5318   TargetLowering::TargetLoweringOpt TLO(DAG, !DCI.isBeforeLegalize(),
5319                                         !DCI.isBeforeLegalizeOps());
5320   const TargetLowering &TLI = DAG.getTargetLoweringInfo();
5321   if (TLI.ShrinkDemandedConstant(Src, Demanded, TLO) ||
5322       TLI.SimplifyDemandedBits(Src, Demanded, Known, TLO)) {
5323     DCI.CommitTargetLoweringOpt(TLO);
5324   }
5325 
5326   return SDValue();
5327 }
5328 
5329 SDValue SITargetLowering::PerformDAGCombine(SDNode *N,
5330                                             DAGCombinerInfo &DCI) const {
5331   switch (N->getOpcode()) {
5332   default:
5333     return AMDGPUTargetLowering::PerformDAGCombine(N, DCI);
5334   case ISD::ADD:
5335     return performAddCombine(N, DCI);
5336   case ISD::SUB:
5337     return performSubCombine(N, DCI);
5338   case ISD::ADDCARRY:
5339   case ISD::SUBCARRY:
5340     return performAddCarrySubCarryCombine(N, DCI);
5341   case ISD::FADD:
5342     return performFAddCombine(N, DCI);
5343   case ISD::FSUB:
5344     return performFSubCombine(N, DCI);
5345   case ISD::SETCC:
5346     return performSetCCCombine(N, DCI);
5347   case ISD::FMAXNUM:
5348   case ISD::FMINNUM:
5349   case ISD::SMAX:
5350   case ISD::SMIN:
5351   case ISD::UMAX:
5352   case ISD::UMIN:
5353   case AMDGPUISD::FMIN_LEGACY:
5354   case AMDGPUISD::FMAX_LEGACY: {
5355     if (DCI.getDAGCombineLevel() >= AfterLegalizeDAG &&
5356         getTargetMachine().getOptLevel() > CodeGenOpt::None)
5357       return performMinMaxCombine(N, DCI);
5358     break;
5359   }
5360   case ISD::LOAD:
5361   case ISD::STORE:
5362   case ISD::ATOMIC_LOAD:
5363   case ISD::ATOMIC_STORE:
5364   case ISD::ATOMIC_CMP_SWAP:
5365   case ISD::ATOMIC_CMP_SWAP_WITH_SUCCESS:
5366   case ISD::ATOMIC_SWAP:
5367   case ISD::ATOMIC_LOAD_ADD:
5368   case ISD::ATOMIC_LOAD_SUB:
5369   case ISD::ATOMIC_LOAD_AND:
5370   case ISD::ATOMIC_LOAD_OR:
5371   case ISD::ATOMIC_LOAD_XOR:
5372   case ISD::ATOMIC_LOAD_NAND:
5373   case ISD::ATOMIC_LOAD_MIN:
5374   case ISD::ATOMIC_LOAD_MAX:
5375   case ISD::ATOMIC_LOAD_UMIN:
5376   case ISD::ATOMIC_LOAD_UMAX:
5377   case AMDGPUISD::ATOMIC_INC:
5378   case AMDGPUISD::ATOMIC_DEC: // TODO: Target mem intrinsics.
5379     if (DCI.isBeforeLegalize())
5380       break;
5381     return performMemSDNodeCombine(cast<MemSDNode>(N), DCI);
5382   case ISD::AND:
5383     return performAndCombine(N, DCI);
5384   case ISD::OR:
5385     return performOrCombine(N, DCI);
5386   case ISD::XOR:
5387     return performXorCombine(N, DCI);
5388   case ISD::ZERO_EXTEND:
5389     return performZeroExtendCombine(N, DCI);
5390   case AMDGPUISD::FP_CLASS:
5391     return performClassCombine(N, DCI);
5392   case ISD::FCANONICALIZE:
5393     return performFCanonicalizeCombine(N, DCI);
5394   case AMDGPUISD::FRACT:
5395   case AMDGPUISD::RCP:
5396   case AMDGPUISD::RSQ:
5397   case AMDGPUISD::RCP_LEGACY:
5398   case AMDGPUISD::RSQ_LEGACY:
5399   case AMDGPUISD::RSQ_CLAMP:
5400   case AMDGPUISD::LDEXP: {
5401     SDValue Src = N->getOperand(0);
5402     if (Src.isUndef())
5403       return Src;
5404     break;
5405   }
5406   case ISD::SINT_TO_FP:
5407   case ISD::UINT_TO_FP:
5408     return performUCharToFloatCombine(N, DCI);
5409   case AMDGPUISD::CVT_F32_UBYTE0:
5410   case AMDGPUISD::CVT_F32_UBYTE1:
5411   case AMDGPUISD::CVT_F32_UBYTE2:
5412   case AMDGPUISD::CVT_F32_UBYTE3:
5413     return performCvtF32UByteNCombine(N, DCI);
5414   case AMDGPUISD::FMED3:
5415     return performFMed3Combine(N, DCI);
5416   case AMDGPUISD::CVT_PKRTZ_F16_F32:
5417     return performCvtPkRTZCombine(N, DCI);
5418   case ISD::SCALAR_TO_VECTOR: {
5419     SelectionDAG &DAG = DCI.DAG;
5420     EVT VT = N->getValueType(0);
5421 
5422     // v2i16 (scalar_to_vector i16:x) -> v2i16 (bitcast (any_extend i16:x))
5423     if (VT == MVT::v2i16 || VT == MVT::v2f16) {
5424       SDLoc SL(N);
5425       SDValue Src = N->getOperand(0);
5426       EVT EltVT = Src.getValueType();
5427       if (EltVT == MVT::f16)
5428         Src = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Src);
5429 
5430       SDValue Ext = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, Src);
5431       return DAG.getNode(ISD::BITCAST, SL, VT, Ext);
5432     }
5433 
5434     break;
5435   }
5436   case ISD::EXTRACT_VECTOR_ELT:
5437     return performExtractVectorEltCombine(N, DCI);
5438   }
5439   return AMDGPUTargetLowering::PerformDAGCombine(N, DCI);
5440 }
5441 
5442 /// \brief Helper function for adjustWritemask
5443 static unsigned SubIdx2Lane(unsigned Idx) {
5444   switch (Idx) {
5445   default: return 0;
5446   case AMDGPU::sub0: return 0;
5447   case AMDGPU::sub1: return 1;
5448   case AMDGPU::sub2: return 2;
5449   case AMDGPU::sub3: return 3;
5450   }
5451 }
5452 
5453 /// \brief Adjust the writemask of MIMG instructions
5454 void SITargetLowering::adjustWritemask(MachineSDNode *&Node,
5455                                        SelectionDAG &DAG) const {
5456   SDNode *Users[4] = { };
5457   unsigned Lane = 0;
5458   unsigned DmaskIdx = (Node->getNumOperands() - Node->getNumValues() == 9) ? 2 : 3;
5459   unsigned OldDmask = Node->getConstantOperandVal(DmaskIdx);
5460   unsigned NewDmask = 0;
5461 
5462   // Try to figure out the used register components
5463   for (SDNode::use_iterator I = Node->use_begin(), E = Node->use_end();
5464        I != E; ++I) {
5465 
5466     // Don't look at users of the chain.
5467     if (I.getUse().getResNo() != 0)
5468       continue;
5469 
5470     // Abort if we can't understand the usage
5471     if (!I->isMachineOpcode() ||
5472         I->getMachineOpcode() != TargetOpcode::EXTRACT_SUBREG)
5473       return;
5474 
5475     // Lane means which subreg of %VGPRa_VGPRb_VGPRc_VGPRd is used.
5476     // Note that subregs are packed, i.e. Lane==0 is the first bit set
5477     // in OldDmask, so it can be any of X,Y,Z,W; Lane==1 is the second bit
5478     // set, etc.
5479     Lane = SubIdx2Lane(I->getConstantOperandVal(1));
5480 
5481     // Set which texture component corresponds to the lane.
5482     unsigned Comp;
5483     for (unsigned i = 0, Dmask = OldDmask; i <= Lane; i++) {
5484       assert(Dmask);
5485       Comp = countTrailingZeros(Dmask);
5486       Dmask &= ~(1 << Comp);
5487     }
5488 
5489     // Abort if we have more than one user per component
5490     if (Users[Lane])
5491       return;
5492 
5493     Users[Lane] = *I;
5494     NewDmask |= 1 << Comp;
5495   }
5496 
5497   // Abort if there's no change
5498   if (NewDmask == OldDmask)
5499     return;
5500 
5501   // Adjust the writemask in the node
5502   std::vector<SDValue> Ops;
5503   Ops.insert(Ops.end(), Node->op_begin(), Node->op_begin() + DmaskIdx);
5504   Ops.push_back(DAG.getTargetConstant(NewDmask, SDLoc(Node), MVT::i32));
5505   Ops.insert(Ops.end(), Node->op_begin() + DmaskIdx + 1, Node->op_end());
5506   Node = (MachineSDNode*)DAG.UpdateNodeOperands(Node, Ops);
5507 
5508   // If we only got one lane, replace it with a copy
5509   // (if NewDmask has only one bit set...)
5510   if (NewDmask && (NewDmask & (NewDmask-1)) == 0) {
5511     SDValue RC = DAG.getTargetConstant(AMDGPU::VGPR_32RegClassID, SDLoc(),
5512                                        MVT::i32);
5513     SDNode *Copy = DAG.getMachineNode(TargetOpcode::COPY_TO_REGCLASS,
5514                                       SDLoc(), Users[Lane]->getValueType(0),
5515                                       SDValue(Node, 0), RC);
5516     DAG.ReplaceAllUsesWith(Users[Lane], Copy);
5517     return;
5518   }
5519 
5520   // Update the users of the node with the new indices
5521   for (unsigned i = 0, Idx = AMDGPU::sub0; i < 4; ++i) {
5522     SDNode *User = Users[i];
5523     if (!User)
5524       continue;
5525 
5526     SDValue Op = DAG.getTargetConstant(Idx, SDLoc(User), MVT::i32);
5527     DAG.UpdateNodeOperands(User, User->getOperand(0), Op);
5528 
5529     switch (Idx) {
5530     default: break;
5531     case AMDGPU::sub0: Idx = AMDGPU::sub1; break;
5532     case AMDGPU::sub1: Idx = AMDGPU::sub2; break;
5533     case AMDGPU::sub2: Idx = AMDGPU::sub3; break;
5534     }
5535   }
5536 }
5537 
5538 static bool isFrameIndexOp(SDValue Op) {
5539   if (Op.getOpcode() == ISD::AssertZext)
5540     Op = Op.getOperand(0);
5541 
5542   return isa<FrameIndexSDNode>(Op);
5543 }
5544 
5545 /// \brief Legalize target independent instructions (e.g. INSERT_SUBREG)
5546 /// with frame index operands.
5547 /// LLVM assumes that inputs are to these instructions are registers.
5548 SDNode *SITargetLowering::legalizeTargetIndependentNode(SDNode *Node,
5549                                                         SelectionDAG &DAG) const {
5550   if (Node->getOpcode() == ISD::CopyToReg) {
5551     RegisterSDNode *DestReg = cast<RegisterSDNode>(Node->getOperand(1));
5552     SDValue SrcVal = Node->getOperand(2);
5553 
5554     // Insert a copy to a VReg_1 virtual register so LowerI1Copies doesn't have
5555     // to try understanding copies to physical registers.
5556     if (SrcVal.getValueType() == MVT::i1 &&
5557         TargetRegisterInfo::isPhysicalRegister(DestReg->getReg())) {
5558       SDLoc SL(Node);
5559       MachineRegisterInfo &MRI = DAG.getMachineFunction().getRegInfo();
5560       SDValue VReg = DAG.getRegister(
5561         MRI.createVirtualRegister(&AMDGPU::VReg_1RegClass), MVT::i1);
5562 
5563       SDNode *Glued = Node->getGluedNode();
5564       SDValue ToVReg
5565         = DAG.getCopyToReg(Node->getOperand(0), SL, VReg, SrcVal,
5566                          SDValue(Glued, Glued ? Glued->getNumValues() - 1 : 0));
5567       SDValue ToResultReg
5568         = DAG.getCopyToReg(ToVReg, SL, SDValue(DestReg, 0),
5569                            VReg, ToVReg.getValue(1));
5570       DAG.ReplaceAllUsesWith(Node, ToResultReg.getNode());
5571       DAG.RemoveDeadNode(Node);
5572       return ToResultReg.getNode();
5573     }
5574   }
5575 
5576   SmallVector<SDValue, 8> Ops;
5577   for (unsigned i = 0; i < Node->getNumOperands(); ++i) {
5578     if (!isFrameIndexOp(Node->getOperand(i))) {
5579       Ops.push_back(Node->getOperand(i));
5580       continue;
5581     }
5582 
5583     SDLoc DL(Node);
5584     Ops.push_back(SDValue(DAG.getMachineNode(AMDGPU::S_MOV_B32, DL,
5585                                      Node->getOperand(i).getValueType(),
5586                                      Node->getOperand(i)), 0));
5587   }
5588 
5589   DAG.UpdateNodeOperands(Node, Ops);
5590   return Node;
5591 }
5592 
5593 /// \brief Fold the instructions after selecting them.
5594 SDNode *SITargetLowering::PostISelFolding(MachineSDNode *Node,
5595                                           SelectionDAG &DAG) const {
5596   const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
5597   unsigned Opcode = Node->getMachineOpcode();
5598 
5599   if (TII->isMIMG(Opcode) && !TII->get(Opcode).mayStore() &&
5600       !TII->isGather4(Opcode))
5601     adjustWritemask(Node, DAG);
5602 
5603   if (Opcode == AMDGPU::INSERT_SUBREG ||
5604       Opcode == AMDGPU::REG_SEQUENCE) {
5605     legalizeTargetIndependentNode(Node, DAG);
5606     return Node;
5607   }
5608   return Node;
5609 }
5610 
5611 /// \brief Assign the register class depending on the number of
5612 /// bits set in the writemask
5613 void SITargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI,
5614                                                      SDNode *Node) const {
5615   const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
5616 
5617   MachineRegisterInfo &MRI = MI.getParent()->getParent()->getRegInfo();
5618 
5619   if (TII->isVOP3(MI.getOpcode())) {
5620     // Make sure constant bus requirements are respected.
5621     TII->legalizeOperandsVOP3(MRI, MI);
5622     return;
5623   }
5624 
5625   if (TII->isMIMG(MI)) {
5626     unsigned VReg = MI.getOperand(0).getReg();
5627     const TargetRegisterClass *RC = MRI.getRegClass(VReg);
5628     // TODO: Need mapping tables to handle other cases (register classes).
5629     if (RC != &AMDGPU::VReg_128RegClass)
5630       return;
5631 
5632     unsigned DmaskIdx = MI.getNumOperands() == 12 ? 3 : 4;
5633     unsigned Writemask = MI.getOperand(DmaskIdx).getImm();
5634     unsigned BitsSet = 0;
5635     for (unsigned i = 0; i < 4; ++i)
5636       BitsSet += Writemask & (1 << i) ? 1 : 0;
5637     switch (BitsSet) {
5638     default: return;
5639     case 1:  RC = &AMDGPU::VGPR_32RegClass; break;
5640     case 2:  RC = &AMDGPU::VReg_64RegClass; break;
5641     case 3:  RC = &AMDGPU::VReg_96RegClass; break;
5642     }
5643 
5644     unsigned NewOpcode = TII->getMaskedMIMGOp(MI.getOpcode(), BitsSet);
5645     MI.setDesc(TII->get(NewOpcode));
5646     MRI.setRegClass(VReg, RC);
5647     return;
5648   }
5649 
5650   // Replace unused atomics with the no return version.
5651   int NoRetAtomicOp = AMDGPU::getAtomicNoRetOp(MI.getOpcode());
5652   if (NoRetAtomicOp != -1) {
5653     if (!Node->hasAnyUseOfValue(0)) {
5654       MI.setDesc(TII->get(NoRetAtomicOp));
5655       MI.RemoveOperand(0);
5656       return;
5657     }
5658 
5659     // For mubuf_atomic_cmpswap, we need to have tablegen use an extract_subreg
5660     // instruction, because the return type of these instructions is a vec2 of
5661     // the memory type, so it can be tied to the input operand.
5662     // This means these instructions always have a use, so we need to add a
5663     // special case to check if the atomic has only one extract_subreg use,
5664     // which itself has no uses.
5665     if ((Node->hasNUsesOfValue(1, 0) &&
5666          Node->use_begin()->isMachineOpcode() &&
5667          Node->use_begin()->getMachineOpcode() == AMDGPU::EXTRACT_SUBREG &&
5668          !Node->use_begin()->hasAnyUseOfValue(0))) {
5669       unsigned Def = MI.getOperand(0).getReg();
5670 
5671       // Change this into a noret atomic.
5672       MI.setDesc(TII->get(NoRetAtomicOp));
5673       MI.RemoveOperand(0);
5674 
5675       // If we only remove the def operand from the atomic instruction, the
5676       // extract_subreg will be left with a use of a vreg without a def.
5677       // So we need to insert an implicit_def to avoid machine verifier
5678       // errors.
5679       BuildMI(*MI.getParent(), MI, MI.getDebugLoc(),
5680               TII->get(AMDGPU::IMPLICIT_DEF), Def);
5681     }
5682     return;
5683   }
5684 }
5685 
5686 static SDValue buildSMovImm32(SelectionDAG &DAG, const SDLoc &DL,
5687                               uint64_t Val) {
5688   SDValue K = DAG.getTargetConstant(Val, DL, MVT::i32);
5689   return SDValue(DAG.getMachineNode(AMDGPU::S_MOV_B32, DL, MVT::i32, K), 0);
5690 }
5691 
5692 MachineSDNode *SITargetLowering::wrapAddr64Rsrc(SelectionDAG &DAG,
5693                                                 const SDLoc &DL,
5694                                                 SDValue Ptr) const {
5695   const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
5696 
5697   // Build the half of the subregister with the constants before building the
5698   // full 128-bit register. If we are building multiple resource descriptors,
5699   // this will allow CSEing of the 2-component register.
5700   const SDValue Ops0[] = {
5701     DAG.getTargetConstant(AMDGPU::SGPR_64RegClassID, DL, MVT::i32),
5702     buildSMovImm32(DAG, DL, 0),
5703     DAG.getTargetConstant(AMDGPU::sub0, DL, MVT::i32),
5704     buildSMovImm32(DAG, DL, TII->getDefaultRsrcDataFormat() >> 32),
5705     DAG.getTargetConstant(AMDGPU::sub1, DL, MVT::i32)
5706   };
5707 
5708   SDValue SubRegHi = SDValue(DAG.getMachineNode(AMDGPU::REG_SEQUENCE, DL,
5709                                                 MVT::v2i32, Ops0), 0);
5710 
5711   // Combine the constants and the pointer.
5712   const SDValue Ops1[] = {
5713     DAG.getTargetConstant(AMDGPU::SReg_128RegClassID, DL, MVT::i32),
5714     Ptr,
5715     DAG.getTargetConstant(AMDGPU::sub0_sub1, DL, MVT::i32),
5716     SubRegHi,
5717     DAG.getTargetConstant(AMDGPU::sub2_sub3, DL, MVT::i32)
5718   };
5719 
5720   return DAG.getMachineNode(AMDGPU::REG_SEQUENCE, DL, MVT::v4i32, Ops1);
5721 }
5722 
5723 /// \brief Return a resource descriptor with the 'Add TID' bit enabled
5724 ///        The TID (Thread ID) is multiplied by the stride value (bits [61:48]
5725 ///        of the resource descriptor) to create an offset, which is added to
5726 ///        the resource pointer.
5727 MachineSDNode *SITargetLowering::buildRSRC(SelectionDAG &DAG, const SDLoc &DL,
5728                                            SDValue Ptr, uint32_t RsrcDword1,
5729                                            uint64_t RsrcDword2And3) const {
5730   SDValue PtrLo = DAG.getTargetExtractSubreg(AMDGPU::sub0, DL, MVT::i32, Ptr);
5731   SDValue PtrHi = DAG.getTargetExtractSubreg(AMDGPU::sub1, DL, MVT::i32, Ptr);
5732   if (RsrcDword1) {
5733     PtrHi = SDValue(DAG.getMachineNode(AMDGPU::S_OR_B32, DL, MVT::i32, PtrHi,
5734                                      DAG.getConstant(RsrcDword1, DL, MVT::i32)),
5735                     0);
5736   }
5737 
5738   SDValue DataLo = buildSMovImm32(DAG, DL,
5739                                   RsrcDword2And3 & UINT64_C(0xFFFFFFFF));
5740   SDValue DataHi = buildSMovImm32(DAG, DL, RsrcDword2And3 >> 32);
5741 
5742   const SDValue Ops[] = {
5743     DAG.getTargetConstant(AMDGPU::SReg_128RegClassID, DL, MVT::i32),
5744     PtrLo,
5745     DAG.getTargetConstant(AMDGPU::sub0, DL, MVT::i32),
5746     PtrHi,
5747     DAG.getTargetConstant(AMDGPU::sub1, DL, MVT::i32),
5748     DataLo,
5749     DAG.getTargetConstant(AMDGPU::sub2, DL, MVT::i32),
5750     DataHi,
5751     DAG.getTargetConstant(AMDGPU::sub3, DL, MVT::i32)
5752   };
5753 
5754   return DAG.getMachineNode(AMDGPU::REG_SEQUENCE, DL, MVT::v4i32, Ops);
5755 }
5756 
5757 //===----------------------------------------------------------------------===//
5758 //                         SI Inline Assembly Support
5759 //===----------------------------------------------------------------------===//
5760 
5761 std::pair<unsigned, const TargetRegisterClass *>
5762 SITargetLowering::getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI,
5763                                                StringRef Constraint,
5764                                                MVT VT) const {
5765   if (!isTypeLegal(VT))
5766     return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
5767 
5768   if (Constraint.size() == 1) {
5769     switch (Constraint[0]) {
5770     case 's':
5771     case 'r':
5772       switch (VT.getSizeInBits()) {
5773       default:
5774         return std::make_pair(0U, nullptr);
5775       case 32:
5776       case 16:
5777         return std::make_pair(0U, &AMDGPU::SReg_32_XM0RegClass);
5778       case 64:
5779         return std::make_pair(0U, &AMDGPU::SGPR_64RegClass);
5780       case 128:
5781         return std::make_pair(0U, &AMDGPU::SReg_128RegClass);
5782       case 256:
5783         return std::make_pair(0U, &AMDGPU::SReg_256RegClass);
5784       case 512:
5785         return std::make_pair(0U, &AMDGPU::SReg_512RegClass);
5786       }
5787 
5788     case 'v':
5789       switch (VT.getSizeInBits()) {
5790       default:
5791         return std::make_pair(0U, nullptr);
5792       case 32:
5793       case 16:
5794         return std::make_pair(0U, &AMDGPU::VGPR_32RegClass);
5795       case 64:
5796         return std::make_pair(0U, &AMDGPU::VReg_64RegClass);
5797       case 96:
5798         return std::make_pair(0U, &AMDGPU::VReg_96RegClass);
5799       case 128:
5800         return std::make_pair(0U, &AMDGPU::VReg_128RegClass);
5801       case 256:
5802         return std::make_pair(0U, &AMDGPU::VReg_256RegClass);
5803       case 512:
5804         return std::make_pair(0U, &AMDGPU::VReg_512RegClass);
5805       }
5806     }
5807   }
5808 
5809   if (Constraint.size() > 1) {
5810     const TargetRegisterClass *RC = nullptr;
5811     if (Constraint[1] == 'v') {
5812       RC = &AMDGPU::VGPR_32RegClass;
5813     } else if (Constraint[1] == 's') {
5814       RC = &AMDGPU::SGPR_32RegClass;
5815     }
5816 
5817     if (RC) {
5818       uint32_t Idx;
5819       bool Failed = Constraint.substr(2).getAsInteger(10, Idx);
5820       if (!Failed && Idx < RC->getNumRegs())
5821         return std::make_pair(RC->getRegister(Idx), RC);
5822     }
5823   }
5824   return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
5825 }
5826 
5827 SITargetLowering::ConstraintType
5828 SITargetLowering::getConstraintType(StringRef Constraint) const {
5829   if (Constraint.size() == 1) {
5830     switch (Constraint[0]) {
5831     default: break;
5832     case 's':
5833     case 'v':
5834       return C_RegisterClass;
5835     }
5836   }
5837   return TargetLowering::getConstraintType(Constraint);
5838 }
5839 
5840 // Figure out which registers should be reserved for stack access. Only after
5841 // the function is legalized do we know all of the non-spill stack objects or if
5842 // calls are present.
5843 void SITargetLowering::finalizeLowering(MachineFunction &MF) const {
5844   MachineRegisterInfo &MRI = MF.getRegInfo();
5845   SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
5846   const MachineFrameInfo &MFI = MF.getFrameInfo();
5847   const SISubtarget &ST = MF.getSubtarget<SISubtarget>();
5848   const SIRegisterInfo *TRI = ST.getRegisterInfo();
5849 
5850   if (Info->isEntryFunction()) {
5851     // Callable functions have fixed registers used for stack access.
5852     reservePrivateMemoryRegs(getTargetMachine(), MF, *TRI, *Info);
5853   }
5854 
5855   // We have to assume the SP is needed in case there are calls in the function
5856   // during lowering. Calls are only detected after the function is
5857   // lowered. We're about to reserve registers, so don't bother using it if we
5858   // aren't really going to use it.
5859   bool NeedSP = !Info->isEntryFunction() ||
5860     MFI.hasVarSizedObjects() ||
5861     MFI.hasCalls();
5862 
5863   if (NeedSP) {
5864     unsigned ReservedStackPtrOffsetReg = TRI->reservedStackPtrOffsetReg(MF);
5865     Info->setStackPtrOffsetReg(ReservedStackPtrOffsetReg);
5866 
5867     assert(Info->getStackPtrOffsetReg() != Info->getFrameOffsetReg());
5868     assert(!TRI->isSubRegister(Info->getScratchRSrcReg(),
5869                                Info->getStackPtrOffsetReg()));
5870     MRI.replaceRegWith(AMDGPU::SP_REG, Info->getStackPtrOffsetReg());
5871   }
5872 
5873   MRI.replaceRegWith(AMDGPU::PRIVATE_RSRC_REG, Info->getScratchRSrcReg());
5874   MRI.replaceRegWith(AMDGPU::FP_REG, Info->getFrameOffsetReg());
5875   MRI.replaceRegWith(AMDGPU::SCRATCH_WAVE_OFFSET_REG,
5876                      Info->getScratchWaveOffsetReg());
5877 
5878   TargetLoweringBase::finalizeLowering(MF);
5879 }
5880