1//===-- AMDGPUInstructions.td - Common instruction defs ---*- tablegen -*-===//
2//
3//                     The LLVM Compiler Infrastructure
4//
5// This file is distributed under the University of Illinois Open Source
6// License. See LICENSE.TXT for details.
7//
8//===----------------------------------------------------------------------===//
9//
10// This file contains instruction defs that are common to all hw codegen
11// targets.
12//
13//===----------------------------------------------------------------------===//
14
15class AMDGPUInst <dag outs, dag ins, string asm = "",
16  list<dag> pattern = []> : Instruction {
17  field bit isRegisterLoad = 0;
18  field bit isRegisterStore = 0;
19
20  let Namespace = "AMDGPU";
21  let OutOperandList = outs;
22  let InOperandList = ins;
23  let AsmString = asm;
24  let Pattern = pattern;
25  let Itinerary = NullALU;
26
27  // SoftFail is a field the disassembler can use to provide a way for
28  // instructions to not match without killing the whole decode process. It is
29  // mainly used for ARM, but Tablegen expects this field to exist or it fails
30  // to build the decode table.
31  field bits<64> SoftFail = 0;
32
33  let DecoderNamespace = Namespace;
34
35  let TSFlags{63} = isRegisterLoad;
36  let TSFlags{62} = isRegisterStore;
37}
38
39class AMDGPUShaderInst <dag outs, dag ins, string asm = "",
40  list<dag> pattern = []> : AMDGPUInst<outs, ins, asm, pattern> {
41
42  field bits<32> Inst = 0xffffffff;
43}
44
45def FP16Denormals : Predicate<"Subtarget.hasFP16Denormals()">;
46def FP32Denormals : Predicate<"Subtarget.hasFP32Denormals()">;
47def FP64Denormals : Predicate<"Subtarget.hasFP64Denormals()">;
48def UnsafeFPMath : Predicate<"TM.Options.UnsafeFPMath">;
49
50def InstFlag : OperandWithDefaultOps <i32, (ops (i32 0))>;
51def ADDRIndirect : ComplexPattern<iPTR, 2, "SelectADDRIndirect", [], []>;
52
53let OperandType = "OPERAND_IMMEDIATE" in {
54
55def u32imm : Operand<i32> {
56  let PrintMethod = "printU32ImmOperand";
57}
58
59def u16imm : Operand<i16> {
60  let PrintMethod = "printU16ImmOperand";
61}
62
63def u8imm : Operand<i8> {
64  let PrintMethod = "printU8ImmOperand";
65}
66
67} // End OperandType = "OPERAND_IMMEDIATE"
68
69//===--------------------------------------------------------------------===//
70// Custom Operands
71//===--------------------------------------------------------------------===//
72def brtarget   : Operand<OtherVT>;
73
74//===----------------------------------------------------------------------===//
75// Misc. PatFrags
76//===----------------------------------------------------------------------===//
77
78class HasOneUseBinOp<SDPatternOperator op> : PatFrag<
79  (ops node:$src0, node:$src1),
80  (op $src0, $src1),
81  [{ return N->hasOneUse(); }]
82>;
83
84class HasOneUseTernaryOp<SDPatternOperator op> : PatFrag<
85  (ops node:$src0, node:$src1, node:$src2),
86  (op $src0, $src1, $src2),
87  [{ return N->hasOneUse(); }]
88>;
89
90
91let Properties = [SDNPCommutative, SDNPAssociative] in {
92def smax_oneuse : HasOneUseBinOp<smax>;
93def smin_oneuse : HasOneUseBinOp<smin>;
94def umax_oneuse : HasOneUseBinOp<umax>;
95def umin_oneuse : HasOneUseBinOp<umin>;
96def fminnum_oneuse : HasOneUseBinOp<fminnum>;
97def fmaxnum_oneuse : HasOneUseBinOp<fmaxnum>;
98def and_oneuse : HasOneUseBinOp<and>;
99def or_oneuse : HasOneUseBinOp<or>;
100def xor_oneuse : HasOneUseBinOp<xor>;
101} // Properties = [SDNPCommutative, SDNPAssociative]
102
103def sub_oneuse : HasOneUseBinOp<sub>;
104def shl_oneuse : HasOneUseBinOp<shl>;
105
106def select_oneuse : HasOneUseTernaryOp<select>;
107
108//===----------------------------------------------------------------------===//
109// PatLeafs for floating-point comparisons
110//===----------------------------------------------------------------------===//
111
112def COND_OEQ : PatLeaf <
113  (cond),
114  [{return N->get() == ISD::SETOEQ || N->get() == ISD::SETEQ;}]
115>;
116
117def COND_ONE : PatLeaf <
118  (cond),
119  [{return N->get() == ISD::SETONE || N->get() == ISD::SETNE;}]
120>;
121
122def COND_OGT : PatLeaf <
123  (cond),
124  [{return N->get() == ISD::SETOGT || N->get() == ISD::SETGT;}]
125>;
126
127def COND_OGE : PatLeaf <
128  (cond),
129  [{return N->get() == ISD::SETOGE || N->get() == ISD::SETGE;}]
130>;
131
132def COND_OLT : PatLeaf <
133  (cond),
134  [{return N->get() == ISD::SETOLT || N->get() == ISD::SETLT;}]
135>;
136
137def COND_OLE : PatLeaf <
138  (cond),
139  [{return N->get() == ISD::SETOLE || N->get() == ISD::SETLE;}]
140>;
141
142
143def COND_O : PatLeaf <(cond), [{return N->get() == ISD::SETO;}]>;
144def COND_UO : PatLeaf <(cond), [{return N->get() == ISD::SETUO;}]>;
145
146//===----------------------------------------------------------------------===//
147// PatLeafs for unsigned / unordered comparisons
148//===----------------------------------------------------------------------===//
149
150def COND_UEQ : PatLeaf <(cond), [{return N->get() == ISD::SETUEQ;}]>;
151def COND_UNE : PatLeaf <(cond), [{return N->get() == ISD::SETUNE;}]>;
152def COND_UGT : PatLeaf <(cond), [{return N->get() == ISD::SETUGT;}]>;
153def COND_UGE : PatLeaf <(cond), [{return N->get() == ISD::SETUGE;}]>;
154def COND_ULT : PatLeaf <(cond), [{return N->get() == ISD::SETULT;}]>;
155def COND_ULE : PatLeaf <(cond), [{return N->get() == ISD::SETULE;}]>;
156
157// XXX - For some reason R600 version is preferring to use unordered
158// for setne?
159def COND_UNE_NE : PatLeaf <
160  (cond),
161  [{return N->get() == ISD::SETUNE || N->get() == ISD::SETNE;}]
162>;
163
164//===----------------------------------------------------------------------===//
165// PatLeafs for signed comparisons
166//===----------------------------------------------------------------------===//
167
168def COND_SGT : PatLeaf <(cond), [{return N->get() == ISD::SETGT;}]>;
169def COND_SGE : PatLeaf <(cond), [{return N->get() == ISD::SETGE;}]>;
170def COND_SLT : PatLeaf <(cond), [{return N->get() == ISD::SETLT;}]>;
171def COND_SLE : PatLeaf <(cond), [{return N->get() == ISD::SETLE;}]>;
172
173//===----------------------------------------------------------------------===//
174// PatLeafs for integer equality
175//===----------------------------------------------------------------------===//
176
177def COND_EQ : PatLeaf <
178  (cond),
179  [{return N->get() == ISD::SETEQ || N->get() == ISD::SETUEQ;}]
180>;
181
182def COND_NE : PatLeaf <
183  (cond),
184  [{return N->get() == ISD::SETNE || N->get() == ISD::SETUNE;}]
185>;
186
187def COND_NULL : PatLeaf <
188  (cond),
189  [{(void)N; return false;}]
190>;
191
192
193//===----------------------------------------------------------------------===//
194// Load/Store Pattern Fragments
195//===----------------------------------------------------------------------===//
196
197class PrivateMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{
198  return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS;
199}]>;
200
201class PrivateLoad <SDPatternOperator op> : PrivateMemOp <
202  (ops node:$ptr), (op node:$ptr)
203>;
204
205class PrivateStore <SDPatternOperator op> : PrivateMemOp <
206  (ops node:$value, node:$ptr), (op node:$value, node:$ptr)
207>;
208
209def load_private : PrivateLoad <load>;
210
211def truncstorei8_private : PrivateStore <truncstorei8>;
212def truncstorei16_private : PrivateStore <truncstorei16>;
213def store_private : PrivateStore <store>;
214
215class GlobalMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{
216  return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;
217}]>;
218
219// Global address space loads
220class GlobalLoad <SDPatternOperator op> : GlobalMemOp <
221  (ops node:$ptr), (op node:$ptr)
222>;
223
224def global_load : GlobalLoad <load>;
225
226// Global address space stores
227class GlobalStore <SDPatternOperator op> : GlobalMemOp <
228  (ops node:$value, node:$ptr), (op node:$value, node:$ptr)
229>;
230
231def global_store : GlobalStore <store>;
232def global_store_atomic : GlobalStore<atomic_store>;
233
234
235class ConstantMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{
236  return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::CONSTANT_ADDRESS;
237}]>;
238
239// Constant address space loads
240class ConstantLoad <SDPatternOperator op> : ConstantMemOp <
241  (ops node:$ptr), (op node:$ptr)
242>;
243
244def constant_load : ConstantLoad<load>;
245
246class LocalMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{
247  return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
248}]>;
249
250// Local address space loads
251class LocalLoad <SDPatternOperator op> : LocalMemOp <
252  (ops node:$ptr), (op node:$ptr)
253>;
254
255class LocalStore <SDPatternOperator op> : LocalMemOp <
256  (ops node:$value, node:$ptr), (op node:$value, node:$ptr)
257>;
258
259class FlatMemOp <dag ops, dag frag> : PatFrag <ops, frag, [{
260  return cast<MemSDNode>(N)->getAddressSPace() == AMDGPUAS::FLAT_ADDRESS;
261}]>;
262
263class FlatLoad <SDPatternOperator op> : FlatMemOp <
264  (ops node:$ptr), (op node:$ptr)
265>;
266
267class AZExtLoadBase <SDPatternOperator ld_node>: PatFrag<(ops node:$ptr),
268                                              (ld_node node:$ptr), [{
269  LoadSDNode *L = cast<LoadSDNode>(N);
270  return L->getExtensionType() == ISD::ZEXTLOAD ||
271         L->getExtensionType() == ISD::EXTLOAD;
272}]>;
273
274def az_extload : AZExtLoadBase <unindexedload>;
275
276def az_extloadi8 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{
277  return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i8;
278}]>;
279
280def az_extloadi8_global : GlobalLoad <az_extloadi8>;
281def sextloadi8_global : GlobalLoad <sextloadi8>;
282
283def az_extloadi8_constant : ConstantLoad <az_extloadi8>;
284def sextloadi8_constant : ConstantLoad <sextloadi8>;
285
286def az_extloadi8_local : LocalLoad <az_extloadi8>;
287def sextloadi8_local : LocalLoad <sextloadi8>;
288
289def extloadi8_private : PrivateLoad <az_extloadi8>;
290def sextloadi8_private : PrivateLoad <sextloadi8>;
291
292def az_extloadi16 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{
293  return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i16;
294}]>;
295
296def az_extloadi16_global : GlobalLoad <az_extloadi16>;
297def sextloadi16_global : GlobalLoad <sextloadi16>;
298
299def az_extloadi16_constant : ConstantLoad <az_extloadi16>;
300def sextloadi16_constant : ConstantLoad <sextloadi16>;
301
302def az_extloadi16_local : LocalLoad <az_extloadi16>;
303def sextloadi16_local : LocalLoad <sextloadi16>;
304
305def extloadi16_private : PrivateLoad <az_extloadi16>;
306def sextloadi16_private : PrivateLoad <sextloadi16>;
307
308def az_extloadi32 : PatFrag<(ops node:$ptr), (az_extload node:$ptr), [{
309  return cast<LoadSDNode>(N)->getMemoryVT() == MVT::i32;
310}]>;
311
312def az_extloadi32_global : GlobalLoad <az_extloadi32>;
313
314def az_extloadi32_flat : FlatLoad <az_extloadi32>;
315
316def az_extloadi32_constant : ConstantLoad <az_extloadi32>;
317
318def truncstorei8_global : GlobalStore <truncstorei8>;
319def truncstorei16_global : GlobalStore <truncstorei16>;
320
321def local_store : LocalStore <store>;
322def truncstorei8_local : LocalStore <truncstorei8>;
323def truncstorei16_local : LocalStore <truncstorei16>;
324
325def local_load : LocalLoad <load>;
326
327class Aligned8Bytes <dag ops, dag frag> : PatFrag <ops, frag, [{
328    return cast<MemSDNode>(N)->getAlignment() % 8 == 0;
329}]>;
330
331def local_load_aligned8bytes : Aligned8Bytes <
332  (ops node:$ptr), (local_load node:$ptr)
333>;
334
335def local_store_aligned8bytes : Aligned8Bytes <
336  (ops node:$val, node:$ptr), (local_store node:$val, node:$ptr)
337>;
338
339class local_binary_atomic_op<SDNode atomic_op> :
340  PatFrag<(ops node:$ptr, node:$value),
341    (atomic_op node:$ptr, node:$value), [{
342  return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
343}]>;
344
345
346def atomic_swap_local : local_binary_atomic_op<atomic_swap>;
347def atomic_load_add_local : local_binary_atomic_op<atomic_load_add>;
348def atomic_load_sub_local : local_binary_atomic_op<atomic_load_sub>;
349def atomic_load_and_local : local_binary_atomic_op<atomic_load_and>;
350def atomic_load_or_local : local_binary_atomic_op<atomic_load_or>;
351def atomic_load_xor_local : local_binary_atomic_op<atomic_load_xor>;
352def atomic_load_nand_local : local_binary_atomic_op<atomic_load_nand>;
353def atomic_load_min_local : local_binary_atomic_op<atomic_load_min>;
354def atomic_load_max_local : local_binary_atomic_op<atomic_load_max>;
355def atomic_load_umin_local : local_binary_atomic_op<atomic_load_umin>;
356def atomic_load_umax_local : local_binary_atomic_op<atomic_load_umax>;
357
358def mskor_global : PatFrag<(ops node:$val, node:$ptr),
359                            (AMDGPUstore_mskor node:$val, node:$ptr), [{
360  return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;
361}]>;
362
363multiclass AtomicCmpSwapLocal <SDNode cmp_swap_node> {
364
365  def _32_local : PatFrag <
366    (ops node:$ptr, node:$cmp, node:$swap),
367    (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{
368      AtomicSDNode *AN = cast<AtomicSDNode>(N);
369      return AN->getMemoryVT() == MVT::i32 &&
370             AN->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
371  }]>;
372
373  def _64_local : PatFrag<
374    (ops node:$ptr, node:$cmp, node:$swap),
375    (cmp_swap_node node:$ptr, node:$cmp, node:$swap), [{
376      AtomicSDNode *AN = cast<AtomicSDNode>(N);
377      return AN->getMemoryVT() == MVT::i64 &&
378             AN->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS;
379  }]>;
380}
381
382defm atomic_cmp_swap : AtomicCmpSwapLocal <atomic_cmp_swap>;
383
384multiclass global_binary_atomic_op<SDNode atomic_op> {
385  def "" : PatFrag<
386        (ops node:$ptr, node:$value),
387        (atomic_op node:$ptr, node:$value),
388        [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}]>;
389
390  def _noret : PatFrag<
391        (ops node:$ptr, node:$value),
392        (atomic_op node:$ptr, node:$value),
393        [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>;
394
395  def _ret : PatFrag<
396        (ops node:$ptr, node:$value),
397        (atomic_op node:$ptr, node:$value),
398        [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>;
399}
400
401defm atomic_swap_global : global_binary_atomic_op<atomic_swap>;
402defm atomic_add_global : global_binary_atomic_op<atomic_load_add>;
403defm atomic_and_global : global_binary_atomic_op<atomic_load_and>;
404defm atomic_max_global : global_binary_atomic_op<atomic_load_max>;
405defm atomic_min_global : global_binary_atomic_op<atomic_load_min>;
406defm atomic_or_global : global_binary_atomic_op<atomic_load_or>;
407defm atomic_sub_global : global_binary_atomic_op<atomic_load_sub>;
408defm atomic_umax_global : global_binary_atomic_op<atomic_load_umax>;
409defm atomic_umin_global : global_binary_atomic_op<atomic_load_umin>;
410defm atomic_xor_global : global_binary_atomic_op<atomic_load_xor>;
411
412//legacy
413def AMDGPUatomic_cmp_swap_global : PatFrag<
414        (ops node:$ptr, node:$value),
415        (AMDGPUatomic_cmp_swap node:$ptr, node:$value),
416        [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}]>;
417
418def atomic_cmp_swap_global : PatFrag<
419      (ops node:$ptr, node:$cmp, node:$value),
420      (atomic_cmp_swap node:$ptr, node:$cmp, node:$value),
421      [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS;}]>;
422
423def atomic_cmp_swap_global_noret : PatFrag<
424      (ops node:$ptr, node:$cmp, node:$value),
425      (atomic_cmp_swap node:$ptr, node:$cmp, node:$value),
426      [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (SDValue(N, 0).use_empty());}]>;
427
428def atomic_cmp_swap_global_ret : PatFrag<
429      (ops node:$ptr, node:$cmp, node:$value),
430      (atomic_cmp_swap node:$ptr, node:$cmp, node:$value),
431      [{return cast<MemSDNode>(N)->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS && (!SDValue(N, 0).use_empty());}]>;
432
433//===----------------------------------------------------------------------===//
434// Misc Pattern Fragments
435//===----------------------------------------------------------------------===//
436
437class Constants {
438int TWO_PI = 0x40c90fdb;
439int PI = 0x40490fdb;
440int TWO_PI_INV = 0x3e22f983;
441int FP_UINT_MAX_PLUS_1 = 0x4f800000;    // 1 << 32 in floating point encoding
442int FP16_ONE = 0x3C00;
443int FP32_ONE = 0x3f800000;
444int FP32_NEG_ONE = 0xbf800000;
445int FP64_ONE = 0x3ff0000000000000;
446int FP64_NEG_ONE = 0xbff0000000000000;
447}
448def CONST : Constants;
449
450def FP_ZERO : PatLeaf <
451  (fpimm),
452  [{return N->getValueAPF().isZero();}]
453>;
454
455def FP_ONE : PatLeaf <
456  (fpimm),
457  [{return N->isExactlyValue(1.0);}]
458>;
459
460def FP_HALF : PatLeaf <
461  (fpimm),
462  [{return N->isExactlyValue(0.5);}]
463>;
464
465let isCodeGenOnly = 1, isPseudo = 1 in {
466
467let usesCustomInserter = 1  in {
468
469class CLAMP <RegisterClass rc> : AMDGPUShaderInst <
470  (outs rc:$dst),
471  (ins rc:$src0),
472  "CLAMP $dst, $src0",
473  [(set f32:$dst, (AMDGPUclamp f32:$src0))]
474>;
475
476class FABS <RegisterClass rc> : AMDGPUShaderInst <
477  (outs rc:$dst),
478  (ins rc:$src0),
479  "FABS $dst, $src0",
480  [(set f32:$dst, (fabs f32:$src0))]
481>;
482
483class FNEG <RegisterClass rc> : AMDGPUShaderInst <
484  (outs rc:$dst),
485  (ins rc:$src0),
486  "FNEG $dst, $src0",
487  [(set f32:$dst, (fneg f32:$src0))]
488>;
489
490} // usesCustomInserter = 1
491
492multiclass RegisterLoadStore <RegisterClass dstClass, Operand addrClass,
493                    ComplexPattern addrPat> {
494let UseNamedOperandTable = 1 in {
495
496  def RegisterLoad : AMDGPUShaderInst <
497    (outs dstClass:$dst),
498    (ins addrClass:$addr, i32imm:$chan),
499    "RegisterLoad $dst, $addr",
500    [(set i32:$dst, (AMDGPUregister_load addrPat:$addr, (i32 timm:$chan)))]
501  > {
502    let isRegisterLoad = 1;
503  }
504
505  def RegisterStore : AMDGPUShaderInst <
506    (outs),
507    (ins dstClass:$val, addrClass:$addr, i32imm:$chan),
508    "RegisterStore $val, $addr",
509    [(AMDGPUregister_store i32:$val, addrPat:$addr, (i32 timm:$chan))]
510  > {
511    let isRegisterStore = 1;
512  }
513}
514}
515
516} // End isCodeGenOnly = 1, isPseudo = 1
517
518/* Generic helper patterns for intrinsics */
519/* -------------------------------------- */
520
521class POW_Common <AMDGPUInst log_ieee, AMDGPUInst exp_ieee, AMDGPUInst mul>
522  : Pat <
523  (fpow f32:$src0, f32:$src1),
524  (exp_ieee (mul f32:$src1, (log_ieee f32:$src0)))
525>;
526
527/* Other helper patterns */
528/* --------------------- */
529
530/* Extract element pattern */
531class Extract_Element <ValueType sub_type, ValueType vec_type, int sub_idx,
532                       SubRegIndex sub_reg>
533  : Pat<
534  (sub_type (extractelt vec_type:$src, sub_idx)),
535  (EXTRACT_SUBREG $src, sub_reg)
536>;
537
538/* Insert element pattern */
539class Insert_Element <ValueType elem_type, ValueType vec_type,
540                      int sub_idx, SubRegIndex sub_reg>
541  : Pat <
542  (insertelt vec_type:$vec, elem_type:$elem, sub_idx),
543  (INSERT_SUBREG $vec, $elem, sub_reg)
544>;
545
546// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer
547// can handle COPY instructions.
548// bitconvert pattern
549class BitConvert <ValueType dt, ValueType st, RegisterClass rc> : Pat <
550  (dt (bitconvert (st rc:$src0))),
551  (dt rc:$src0)
552>;
553
554// XXX: Convert to new syntax and use COPY_TO_REG, once the DFAPacketizer
555// can handle COPY instructions.
556class DwordAddrPat<ValueType vt, RegisterClass rc> : Pat <
557  (vt (AMDGPUdwordaddr (vt rc:$addr))),
558  (vt rc:$addr)
559>;
560
561// BFI_INT patterns
562
563multiclass BFIPatterns <Instruction BFI_INT,
564                        Instruction LoadImm32,
565                        RegisterClass RC64> {
566  // Definition from ISA doc:
567  // (y & x) | (z & ~x)
568  def : Pat <
569    (or (and i32:$y, i32:$x), (and i32:$z, (not i32:$x))),
570    (BFI_INT $x, $y, $z)
571  >;
572
573  // SHA-256 Ch function
574  // z ^ (x & (y ^ z))
575  def : Pat <
576    (xor i32:$z, (and i32:$x, (xor i32:$y, i32:$z))),
577    (BFI_INT $x, $y, $z)
578  >;
579
580  def : Pat <
581    (fcopysign f32:$src0, f32:$src1),
582    (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0, $src1)
583  >;
584
585  def : Pat <
586    (f32 (fcopysign f32:$src0, f64:$src1)),
587    (BFI_INT (LoadImm32 (i32 0x7fffffff)), $src0,
588             (i32 (EXTRACT_SUBREG $src1, sub1)))
589  >;
590
591  def : Pat <
592    (f64 (fcopysign f64:$src0, f64:$src1)),
593    (REG_SEQUENCE RC64,
594      (i32 (EXTRACT_SUBREG $src0, sub0)), sub0,
595      (BFI_INT (LoadImm32 (i32 0x7fffffff)),
596               (i32 (EXTRACT_SUBREG $src0, sub1)),
597               (i32 (EXTRACT_SUBREG $src1, sub1))), sub1)
598  >;
599
600  def : Pat <
601    (f64 (fcopysign f64:$src0, f32:$src1)),
602    (REG_SEQUENCE RC64,
603      (i32 (EXTRACT_SUBREG $src0, sub0)), sub0,
604      (BFI_INT (LoadImm32 (i32 0x7fffffff)),
605               (i32 (EXTRACT_SUBREG $src0, sub1)),
606               $src1), sub1)
607  >;
608}
609
610// SHA-256 Ma patterns
611
612// ((x & z) | (y & (x | z))) -> BFI_INT (XOR x, y), z, y
613class SHA256MaPattern <Instruction BFI_INT, Instruction XOR> : Pat <
614  (or (and i32:$x, i32:$z), (and i32:$y, (or i32:$x, i32:$z))),
615  (BFI_INT (XOR i32:$x, i32:$y), i32:$z, i32:$y)
616>;
617
618// Bitfield extract patterns
619
620def IMMZeroBasedBitfieldMask : PatLeaf <(imm), [{
621  return isMask_32(N->getZExtValue());
622}]>;
623
624def IMMPopCount : SDNodeXForm<imm, [{
625  return CurDAG->getTargetConstant(countPopulation(N->getZExtValue()), SDLoc(N),
626                                   MVT::i32);
627}]>;
628
629multiclass BFEPattern <Instruction UBFE, Instruction SBFE, Instruction MOV> {
630  def : Pat <
631    (i32 (and (i32 (srl i32:$src, i32:$rshift)), IMMZeroBasedBitfieldMask:$mask)),
632    (UBFE $src, $rshift, (MOV (i32 (IMMPopCount $mask))))
633  >;
634
635  def : Pat <
636    (srl (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)),
637    (UBFE $src, (i32 0), $width)
638  >;
639
640  def : Pat <
641    (sra (shl_oneuse i32:$src, (sub 32, i32:$width)), (sub 32, i32:$width)),
642    (SBFE $src, (i32 0), $width)
643  >;
644}
645
646// rotr pattern
647class ROTRPattern <Instruction BIT_ALIGN> : Pat <
648  (rotr i32:$src0, i32:$src1),
649  (BIT_ALIGN $src0, $src0, $src1)
650>;
651
652// This matches 16 permutations of
653// max(min(x, y), min(max(x, y), z))
654class IntMed3Pat<Instruction med3Inst,
655                 SDPatternOperator max,
656                 SDPatternOperator max_oneuse,
657                 SDPatternOperator min_oneuse> : Pat<
658  (max (min_oneuse i32:$src0, i32:$src1),
659       (min_oneuse (max_oneuse i32:$src0, i32:$src1), i32:$src2)),
660  (med3Inst $src0, $src1, $src2)
661>;
662
663// Special conversion patterns
664
665def cvt_rpi_i32_f32 : PatFrag <
666  (ops node:$src),
667  (fp_to_sint (ffloor (fadd $src, FP_HALF))),
668  [{ (void) N; return TM.Options.NoNaNsFPMath; }]
669>;
670
671def cvt_flr_i32_f32 : PatFrag <
672  (ops node:$src),
673  (fp_to_sint (ffloor $src)),
674  [{ (void)N; return TM.Options.NoNaNsFPMath; }]
675>;
676
677class IMad24Pat<Instruction Inst> : Pat <
678  (add (AMDGPUmul_i24 i32:$src0, i32:$src1), i32:$src2),
679  (Inst $src0, $src1, $src2)
680>;
681
682class UMad24Pat<Instruction Inst> : Pat <
683  (add (AMDGPUmul_u24 i32:$src0, i32:$src1), i32:$src2),
684  (Inst $src0, $src1, $src2)
685>;
686
687class RcpPat<Instruction RcpInst, ValueType vt> : Pat <
688  (fdiv FP_ONE, vt:$src),
689  (RcpInst $src)
690>;
691
692class RsqPat<Instruction RsqInst, ValueType vt> : Pat <
693  (AMDGPUrcp (fsqrt vt:$src)),
694  (RsqInst $src)
695>;
696
697include "R600Instructions.td"
698include "R700Instructions.td"
699include "EvergreenInstructions.td"
700include "CaymanInstructions.td"
701
702include "SIInstrInfo.td"
703
704