xref: /wasmtime-44.0.1/winch/codegen/src/masm.rs (revision e7f43b8d)
1 use crate::abi::{self, align_to, scratch, LocalSlot};
2 use crate::codegen::{CodeGenContext, Emission, FuncEnv};
3 use crate::isa::{
4     reg::{writable, Reg, WritableReg},
5     CallingConvention,
6 };
7 use anyhow::Result;
8 use cranelift_codegen::{
9     binemit::CodeOffset,
10     ir::{Endianness, LibCall, MemFlags, RelSourceLoc, SourceLoc, UserExternalNameRef},
11     Final, MachBufferFinalized, MachLabel,
12 };
13 use std::{fmt::Debug, ops::Range};
14 use wasmtime_environ::PtrSize;
15 
16 pub(crate) use cranelift_codegen::ir::TrapCode;
17 
18 #[derive(Eq, PartialEq)]
19 pub(crate) enum DivKind {
20     /// Signed division.
21     Signed,
22     /// Unsigned division.
23     Unsigned,
24 }
25 
26 /// Remainder kind.
27 #[derive(Copy, Clone)]
28 pub(crate) enum RemKind {
29     /// Signed remainder.
30     Signed,
31     /// Unsigned remainder.
32     Unsigned,
33 }
34 
35 impl RemKind {
36     pub fn is_signed(&self) -> bool {
37         matches!(self, Self::Signed)
38     }
39 }
40 
41 #[derive(Copy, Clone, PartialEq, Eq)]
42 pub(crate) enum MemOpKind {
43     /// An atomic memory operation with SeqCst memory ordering.
44     Atomic,
45     /// A memory operation with no memory ordering constraint.
46     Normal,
47 }
48 
49 #[derive(Eq, PartialEq)]
50 pub(crate) enum MulWideKind {
51     Signed,
52     Unsigned,
53 }
54 
55 /// Type of operation for a read-modify-write instruction.
56 pub(crate) enum RmwOp {
57     Add,
58     Sub,
59 }
60 
61 /// The direction to perform the memory move.
62 #[derive(Debug, Clone, Eq, PartialEq)]
63 pub(crate) enum MemMoveDirection {
64     /// From high memory addresses to low memory addresses.
65     /// Invariant: the source location is closer to the FP than the destination
66     /// location, which will be closer to the SP.
67     HighToLow,
68     /// From low memory addresses to high memory addresses.
69     /// Invariant: the source location is closer to the SP than the destination
70     /// location, which will be closer to the FP.
71     LowToHigh,
72 }
73 
74 /// Classifies how to treat float-to-int conversions.
75 #[derive(Debug, Copy, Clone, Eq, PartialEq)]
76 pub(crate) enum TruncKind {
77     /// Saturating conversion. If the source value is greater than the maximum
78     /// value of the destination type, the result is clamped to the
79     /// destination maximum value.
80     Checked,
81     /// An exception is raised if the source value is greater than the maximum
82     /// value of the destination type.
83     Unchecked,
84 }
85 
86 impl TruncKind {
87     /// Returns true if the truncation kind is checked.
88     pub(crate) fn is_checked(&self) -> bool {
89         *self == TruncKind::Checked
90     }
91 
92     /// Returns `true` if the trunc kind is [`Unchecked`].
93     ///
94     /// [`Unchecked`]: TruncKind::Unchecked
95     #[must_use]
96     pub(crate) fn is_unchecked(&self) -> bool {
97         matches!(self, Self::Unchecked)
98     }
99 }
100 
101 /// Representation of the stack pointer offset.
102 #[derive(Copy, Clone, Eq, PartialEq, Debug, PartialOrd, Ord, Default)]
103 pub struct SPOffset(u32);
104 
105 impl SPOffset {
106     pub fn from_u32(offs: u32) -> Self {
107         Self(offs)
108     }
109 
110     pub fn as_u32(&self) -> u32 {
111         self.0
112     }
113 }
114 
115 /// A stack slot.
116 #[derive(Debug, Clone, Copy, Eq, PartialEq)]
117 pub struct StackSlot {
118     /// The location of the slot, relative to the stack pointer.
119     pub offset: SPOffset,
120     /// The size of the slot, in bytes.
121     pub size: u32,
122 }
123 
124 impl StackSlot {
125     pub fn new(offs: SPOffset, size: u32) -> Self {
126         Self { offset: offs, size }
127     }
128 }
129 
130 /// Kinds of integer binary comparison in WebAssembly. The [`MacroAssembler`]
131 /// implementation for each ISA is responsible for emitting the correct
132 /// sequence of instructions when lowering to machine code.
133 #[derive(Debug, Clone, Copy, Eq, PartialEq)]
134 pub(crate) enum IntCmpKind {
135     /// Equal.
136     Eq,
137     /// Not equal.
138     Ne,
139     /// Signed less than.
140     LtS,
141     /// Unsigned less than.
142     LtU,
143     /// Signed greater than.
144     GtS,
145     /// Unsigned greater than.
146     GtU,
147     /// Signed less than or equal.
148     LeS,
149     /// Unsigned less than or equal.
150     LeU,
151     /// Signed greater than or equal.
152     GeS,
153     /// Unsigned greater than or equal.
154     GeU,
155 }
156 
157 /// Kinds of float binary comparison in WebAssembly. The [`MacroAssembler`]
158 /// implementation for each ISA is responsible for emitting the correct
159 /// sequence of instructions when lowering code.
160 #[derive(Debug)]
161 pub(crate) enum FloatCmpKind {
162     /// Equal.
163     Eq,
164     /// Not equal.
165     Ne,
166     /// Less than.
167     Lt,
168     /// Greater than.
169     Gt,
170     /// Less than or equal.
171     Le,
172     /// Greater than or equal.
173     Ge,
174 }
175 
176 /// Kinds of shifts in WebAssembly.The [`masm`] implementation for each ISA is
177 /// responsible for emitting the correct sequence of instructions when
178 /// lowering to machine code.
179 #[derive(Debug, Clone, Copy, Eq, PartialEq)]
180 pub(crate) enum ShiftKind {
181     /// Left shift.
182     Shl,
183     /// Signed right shift.
184     ShrS,
185     /// Unsigned right shift.
186     ShrU,
187     /// Left rotate.
188     Rotl,
189     /// Right rotate.
190     Rotr,
191 }
192 
193 /// Kinds of extends in WebAssembly. Each MacroAssembler implementation
194 /// is responsible for emitting the correct sequence of instructions when
195 /// lowering to machine code.
196 #[derive(Copy, Clone)]
197 pub(crate) enum ExtendKind {
198     /// 8 to 32 bit signed extend.
199     I32Extend8S,
200     /// 8 to 32 bit unsigned extend.
201     I32Extend8U,
202 
203     /// 16 to 32 bit signed extend.
204     I32Extend16S,
205     /// 16 to 32 bit unsigned extend.
206     I32Extend16U,
207 
208     /// 8 to 64 bit signed extend.
209     I64Extend8S,
210     /// 8 to 64 bit unsigned extend.
211     I64Extend8U,
212 
213     /// 16 to 64 bit signed extend.
214     I64Extend16S,
215     /// 16 to 64 bit unsigned extend.
216     I64Extend16U,
217 
218     /// 32 to 64 bit signed extend.
219     I64Extend32S,
220     /// 32 to 64 bit unsigned extend.
221     I64Extend32U,
222 }
223 
224 impl ExtendKind {
225     pub fn signed(&self) -> bool {
226         match self {
227             Self::I32Extend8S
228             | Self::I32Extend16S
229             | Self::I64Extend8S
230             | Self::I64Extend16S
231             | Self::I64Extend32S => true,
232             _ => false,
233         }
234     }
235 
236     pub fn from_bits(&self) -> u8 {
237         match self {
238             Self::I64Extend32S | Self::I64Extend32U => 32,
239             Self::I32Extend8S | Self::I32Extend8U | Self::I64Extend8S | Self::I64Extend8U => 8,
240             Self::I32Extend16S | Self::I64Extend16S | Self::I32Extend16U | Self::I64Extend16U => 16,
241         }
242     }
243 
244     pub fn to_bits(&self) -> u8 {
245         match self {
246             Self::I64Extend32S
247             | Self::I64Extend32U
248             | Self::I64Extend8S
249             | Self::I64Extend8U
250             | Self::I64Extend16S
251             | Self::I64Extend16U => 64,
252             Self::I32Extend8S | Self::I32Extend8U | Self::I32Extend16U | Self::I32Extend16S => 32,
253         }
254     }
255 }
256 
257 /// Kinds of vector extends in WebAssembly. Each MacroAssembler implementation
258 /// is responsible for emitting the correct sequence of instructions when
259 /// lowering to machine code.
260 pub(crate) enum VectorExtendKind {
261     /// Sign extends eight 8 bit integers to eight 16 bit lanes.
262     V128Extend8x8S,
263     /// Zero extends eight 8 bit integers to eight 16 bit lanes.
264     V128Extend8x8U,
265     /// Sign extends four 16 bit integers to four 32 bit lanes.
266     V128Extend16x4S,
267     /// Zero extends four 16 bit integers to four 32 bit lanes.
268     V128Extend16x4U,
269     /// Sign extends two 32 bit integers to two 64 bit lanes.
270     V128Extend32x2S,
271     /// Zero extends two 32 bit integers to two 64 bit lanes.
272     V128Extend32x2U,
273 }
274 
275 /// Kinds of splat loads supported by WebAssembly.
276 pub(crate) enum SplatLoadKind {
277     /// 8 bits.
278     S8,
279     /// 16 bits.
280     S16,
281     /// 32 bits.
282     S32,
283     /// 64 bits.
284     S64,
285 }
286 
287 /// Kinds of splat supported by WebAssembly.
288 #[derive(Copy, Debug, Clone, Eq, PartialEq)]
289 pub(crate) enum SplatKind {
290     /// 8 bit integer.
291     I8x16,
292     /// 16 bit integer.
293     I16x8,
294     /// 32 bit integer.
295     I32x4,
296     /// 64 bit integer.
297     I64x2,
298     /// 32 bit float.
299     F32x4,
300     /// 64 bit float.
301     F64x2,
302 }
303 
304 impl SplatKind {
305     /// The lane size to use for different kinds of splats.
306     pub(crate) fn lane_size(&self) -> OperandSize {
307         match self {
308             SplatKind::I8x16 => OperandSize::S8,
309             SplatKind::I16x8 => OperandSize::S16,
310             SplatKind::I32x4 | SplatKind::F32x4 => OperandSize::S32,
311             SplatKind::I64x2 | SplatKind::F64x2 => OperandSize::S64,
312         }
313     }
314 }
315 
316 /// Kinds of behavior supported by Wasm loads.
317 pub(crate) enum LoadKind {
318     /// Load the entire bytes of the operand size without any modifications.
319     Operand(OperandSize),
320     /// Duplicate value into vector lanes.
321     Splat(SplatLoadKind),
322     /// Scalar (non-vector) extend.
323     ScalarExtend(ExtendKind),
324     /// Vector extend.
325     VectorExtend(VectorExtendKind),
326 }
327 
328 impl LoadKind {
329     /// Returns the [`OperandSize`] used in the load operation.
330     pub(crate) fn derive_operand_size(&self) -> OperandSize {
331         match self {
332             Self::ScalarExtend(scalar) => Self::operand_size_for_scalar(scalar),
333             Self::VectorExtend(vector) => Self::operand_size_for_vector(vector),
334             Self::Splat(kind) => Self::operand_size_for_splat(kind),
335             Self::Operand(op) => *op,
336         }
337     }
338 
339     fn operand_size_for_vector(vector: &VectorExtendKind) -> OperandSize {
340         match vector {
341             VectorExtendKind::V128Extend8x8S | VectorExtendKind::V128Extend8x8U => OperandSize::S8,
342             VectorExtendKind::V128Extend16x4S | VectorExtendKind::V128Extend16x4U => {
343                 OperandSize::S16
344             }
345             VectorExtendKind::V128Extend32x2S | VectorExtendKind::V128Extend32x2U => {
346                 OperandSize::S32
347             }
348         }
349     }
350 
351     fn operand_size_for_scalar(extend_kind: &ExtendKind) -> OperandSize {
352         match extend_kind {
353             ExtendKind::I32Extend8S
354             | ExtendKind::I32Extend8U
355             | ExtendKind::I64Extend8S
356             | ExtendKind::I64Extend8U => OperandSize::S8,
357             ExtendKind::I32Extend16S
358             | ExtendKind::I32Extend16U
359             | ExtendKind::I64Extend16U
360             | ExtendKind::I64Extend16S => OperandSize::S16,
361             ExtendKind::I64Extend32U | ExtendKind::I64Extend32S => OperandSize::S32,
362         }
363     }
364 
365     fn operand_size_for_splat(kind: &SplatLoadKind) -> OperandSize {
366         match kind {
367             SplatLoadKind::S8 => OperandSize::S8,
368             SplatLoadKind::S16 => OperandSize::S16,
369             SplatLoadKind::S32 => OperandSize::S32,
370             SplatLoadKind::S64 => OperandSize::S64,
371         }
372     }
373 }
374 
375 /// Operand size, in bits.
376 #[derive(Copy, Debug, Clone, Eq, PartialEq)]
377 pub(crate) enum OperandSize {
378     /// 8 bits.
379     S8,
380     /// 16 bits.
381     S16,
382     /// 32 bits.
383     S32,
384     /// 64 bits.
385     S64,
386     /// 128 bits.
387     S128,
388 }
389 
390 impl OperandSize {
391     /// The number of bits in the operand.
392     pub fn num_bits(&self) -> u8 {
393         match self {
394             OperandSize::S8 => 8,
395             OperandSize::S16 => 16,
396             OperandSize::S32 => 32,
397             OperandSize::S64 => 64,
398             OperandSize::S128 => 128,
399         }
400     }
401 
402     /// The number of bytes in the operand.
403     pub fn bytes(&self) -> u32 {
404         match self {
405             Self::S8 => 1,
406             Self::S16 => 2,
407             Self::S32 => 4,
408             Self::S64 => 8,
409             Self::S128 => 16,
410         }
411     }
412 
413     /// The binary logarithm of the number of bits in the operand.
414     pub fn log2(&self) -> u8 {
415         match self {
416             OperandSize::S8 => 3,
417             OperandSize::S16 => 4,
418             OperandSize::S32 => 5,
419             OperandSize::S64 => 6,
420             OperandSize::S128 => 7,
421         }
422     }
423 
424     /// Create an [`OperandSize`]  from the given number of bytes.
425     pub fn from_bytes(bytes: u8) -> Self {
426         use OperandSize::*;
427         match bytes {
428             4 => S32,
429             8 => S64,
430             16 => S128,
431             _ => panic!("Invalid bytes {bytes} for OperandSize"),
432         }
433     }
434 }
435 
436 /// An abstraction over a register or immediate.
437 #[derive(Copy, Clone, Debug, PartialEq, Eq)]
438 pub(crate) enum RegImm {
439     /// A register.
440     Reg(Reg),
441     /// A tagged immediate argument.
442     Imm(Imm),
443 }
444 
445 /// An tagged representation of an immediate.
446 #[derive(Copy, Clone, Debug, PartialEq, Eq)]
447 pub(crate) enum Imm {
448     /// I32 immediate.
449     I32(u32),
450     /// I64 immediate.
451     I64(u64),
452     /// F32 immediate.
453     F32(u32),
454     /// F64 immediate.
455     F64(u64),
456     /// V128 immediate.
457     V128(i128),
458 }
459 
460 impl Imm {
461     /// Create a new I64 immediate.
462     pub fn i64(val: i64) -> Self {
463         Self::I64(val as u64)
464     }
465 
466     /// Create a new I32 immediate.
467     pub fn i32(val: i32) -> Self {
468         Self::I32(val as u32)
469     }
470 
471     /// Create a new F32 immediate.
472     pub fn f32(bits: u32) -> Self {
473         Self::F32(bits)
474     }
475 
476     /// Create a new F64 immediate.
477     pub fn f64(bits: u64) -> Self {
478         Self::F64(bits)
479     }
480 
481     /// Create a new V128 immediate.
482     pub fn v128(bits: i128) -> Self {
483         Self::V128(bits)
484     }
485 
486     /// Convert the immediate to i32, if possible.
487     pub fn to_i32(&self) -> Option<i32> {
488         match self {
489             Self::I32(v) => Some(*v as i32),
490             Self::I64(v) => i32::try_from(*v as i64).ok(),
491             _ => None,
492         }
493     }
494 
495     /// Returns true if the [`Imm`] is float.
496     pub fn is_float(&self) -> bool {
497         match self {
498             Self::F32(_) | Self::F64(_) => true,
499             _ => false,
500         }
501     }
502 
503     /// Get the operand size of the immediate.
504     pub fn size(&self) -> OperandSize {
505         match self {
506             Self::I32(_) | Self::F32(_) => OperandSize::S32,
507             Self::I64(_) | Self::F64(_) => OperandSize::S64,
508             Self::V128(_) => OperandSize::S128,
509         }
510     }
511 
512     /// Get a little endian representation of the immediate.
513     ///
514     /// This method heap allocates and is intended to be used when adding
515     /// values to the constant pool.
516     pub fn to_bytes(&self) -> Vec<u8> {
517         match self {
518             Imm::I32(n) => n.to_le_bytes().to_vec(),
519             Imm::I64(n) => n.to_le_bytes().to_vec(),
520             Imm::F32(n) => n.to_le_bytes().to_vec(),
521             Imm::F64(n) => n.to_le_bytes().to_vec(),
522             Imm::V128(n) => n.to_le_bytes().to_vec(),
523         }
524     }
525 }
526 
527 /// The location of the [VMcontext] used for function calls.
528 #[derive(Copy, Clone, Debug, Eq, PartialEq)]
529 pub(crate) enum VMContextLoc {
530     /// Dynamic, stored in the given register.
531     Reg(Reg),
532     /// The pinned [VMContext] register.
533     Pinned,
534 }
535 
536 /// The maximum number of context arguments currently used across the compiler.
537 pub(crate) const MAX_CONTEXT_ARGS: usize = 2;
538 
539 /// Out-of-band special purpose arguments used for function call emission.
540 ///
541 /// We cannot rely on the value stack for these values given that inserting
542 /// register or memory values at arbitrary locations of the value stack has the
543 /// potential to break the stack ordering principle, which states that older
544 /// values must always precede newer values, effectively simulating the order of
545 /// values in the machine stack.
546 /// The [ContextArgs] are meant to be resolved at every callsite; in some cases
547 /// it might be possible to construct it early on, but given that it might
548 /// contain allocatable registers, it's preferred to construct it in
549 /// [FnCall::emit].
550 #[derive(Clone, Debug)]
551 pub(crate) enum ContextArgs {
552     /// No context arguments required. This is used for libcalls that don't
553     /// require any special context arguments. For example builtin functions
554     /// that perform float calculations.
555     None,
556     /// A single context argument is required; the current pinned [VMcontext]
557     /// register must be passed as the first argument of the function call.
558     VMContext([VMContextLoc; 1]),
559     /// The callee and caller context arguments are required. In this case, the
560     /// callee context argument is usually stored into an allocatable register
561     /// and the caller is always the current pinned [VMContext] pointer.
562     CalleeAndCallerVMContext([VMContextLoc; MAX_CONTEXT_ARGS]),
563 }
564 
565 impl ContextArgs {
566     /// Construct an empty [ContextArgs].
567     pub fn none() -> Self {
568         Self::None
569     }
570 
571     /// Construct a [ContextArgs] declaring the usage of the pinned [VMContext]
572     /// register as both the caller and callee context arguments.
573     pub fn pinned_callee_and_caller_vmctx() -> Self {
574         Self::CalleeAndCallerVMContext([VMContextLoc::Pinned, VMContextLoc::Pinned])
575     }
576 
577     /// Construct a [ContextArgs] that declares the usage of the pinned
578     /// [VMContext] register as the only context argument.
579     pub fn pinned_vmctx() -> Self {
580         Self::VMContext([VMContextLoc::Pinned])
581     }
582 
583     /// Construct a [ContextArgs] that declares a dynamic callee context and the
584     /// pinned [VMContext] register as the context arguments.
585     pub fn with_callee_and_pinned_caller(callee_vmctx: Reg) -> Self {
586         Self::CalleeAndCallerVMContext([VMContextLoc::Reg(callee_vmctx), VMContextLoc::Pinned])
587     }
588 
589     /// Get the length of the [ContextArgs].
590     pub fn len(&self) -> usize {
591         self.as_slice().len()
592     }
593 
594     /// Get a slice of the context arguments.
595     pub fn as_slice(&self) -> &[VMContextLoc] {
596         match self {
597             Self::None => &[],
598             Self::VMContext(a) => a.as_slice(),
599             Self::CalleeAndCallerVMContext(a) => a.as_slice(),
600         }
601     }
602 }
603 
604 #[derive(Copy, Clone, Debug)]
605 pub(crate) enum CalleeKind {
606     /// A function call to a raw address.
607     Indirect(Reg),
608     /// A function call to a local function.
609     Direct(UserExternalNameRef),
610     /// Call to a well known LibCall.
611     LibCall(LibCall),
612 }
613 
614 impl CalleeKind {
615     /// Creates a callee kind from a register.
616     pub fn indirect(reg: Reg) -> Self {
617         Self::Indirect(reg)
618     }
619 
620     /// Creates a direct callee kind from a function name.
621     pub fn direct(name: UserExternalNameRef) -> Self {
622         Self::Direct(name)
623     }
624 
625     /// Creates a known callee kind from a libcall.
626     pub fn libcall(call: LibCall) -> Self {
627         Self::LibCall(call)
628     }
629 }
630 
631 impl RegImm {
632     /// Register constructor.
633     pub fn reg(r: Reg) -> Self {
634         RegImm::Reg(r)
635     }
636 
637     /// I64 immediate constructor.
638     pub fn i64(val: i64) -> Self {
639         RegImm::Imm(Imm::i64(val))
640     }
641 
642     /// I32 immediate constructor.
643     pub fn i32(val: i32) -> Self {
644         RegImm::Imm(Imm::i32(val))
645     }
646 
647     /// F32 immediate, stored using its bits representation.
648     pub fn f32(bits: u32) -> Self {
649         RegImm::Imm(Imm::f32(bits))
650     }
651 
652     /// F64 immediate, stored using its bits representation.
653     pub fn f64(bits: u64) -> Self {
654         RegImm::Imm(Imm::f64(bits))
655     }
656 
657     /// V128 immediate.
658     pub fn v128(bits: i128) -> Self {
659         RegImm::Imm(Imm::v128(bits))
660     }
661 }
662 
663 impl From<Reg> for RegImm {
664     fn from(r: Reg) -> Self {
665         Self::Reg(r)
666     }
667 }
668 
669 #[derive(Debug)]
670 pub enum RoundingMode {
671     Nearest,
672     Up,
673     Down,
674     Zero,
675 }
676 
677 /// Memory flags for trusted loads/stores.
678 pub const TRUSTED_FLAGS: MemFlags = MemFlags::trusted();
679 
680 /// Flags used for WebAssembly loads / stores.
681 /// Untrusted by default so we don't set `no_trap`.
682 /// We also ensure that the endianness is the right one for WebAssembly.
683 pub const UNTRUSTED_FLAGS: MemFlags = MemFlags::new().with_endianness(Endianness::Little);
684 
685 /// Generic MacroAssembler interface used by the code generation.
686 ///
687 /// The MacroAssembler trait aims to expose an interface, high-level enough,
688 /// so that each ISA can provide its own lowering to machine code. For example,
689 /// for WebAssembly operators that don't have a direct mapping to a machine
690 /// a instruction, the interface defines a signature matching the WebAssembly
691 /// operator, allowing each implementation to lower such operator entirely.
692 /// This approach attributes more responsibility to the MacroAssembler, but frees
693 /// the caller from concerning about assembling the right sequence of
694 /// instructions at the operator callsite.
695 ///
696 /// The interface defaults to a three-argument form for binary operations;
697 /// this allows a natural mapping to instructions for RISC architectures,
698 /// that use three-argument form.
699 /// This approach allows for a more general interface that can be restricted
700 /// where needed, in the case of architectures that use a two-argument form.
701 
702 pub(crate) trait MacroAssembler {
703     /// The addressing mode.
704     type Address: Copy + Debug;
705 
706     /// The pointer representation of the target ISA,
707     /// used to access information from [`VMOffsets`].
708     type Ptr: PtrSize;
709 
710     /// The ABI details of the target.
711     type ABI: abi::ABI;
712 
713     /// Emit the function prologue.
714     fn prologue(&mut self, vmctx: Reg) -> Result<()> {
715         self.frame_setup()?;
716         self.check_stack(vmctx)
717     }
718 
719     /// Generate the frame setup sequence.
720     fn frame_setup(&mut self) -> Result<()>;
721 
722     /// Generate the frame restore sequence.
723     fn frame_restore(&mut self) -> Result<()>;
724 
725     /// Emit a stack check.
726     fn check_stack(&mut self, vmctx: Reg) -> Result<()>;
727 
728     /// Emit the function epilogue.
729     fn epilogue(&mut self) -> Result<()> {
730         self.frame_restore()
731     }
732 
733     /// Reserve stack space.
734     fn reserve_stack(&mut self, bytes: u32) -> Result<()>;
735 
736     /// Free stack space.
737     fn free_stack(&mut self, bytes: u32) -> Result<()>;
738 
739     /// Reset the stack pointer to the given offset;
740     ///
741     /// Used to reset the stack pointer to a given offset
742     /// when dealing with unreachable code.
743     fn reset_stack_pointer(&mut self, offset: SPOffset) -> Result<()>;
744 
745     /// Get the address of a local slot.
746     fn local_address(&mut self, local: &LocalSlot) -> Result<Self::Address>;
747 
748     /// Constructs an address with an offset that is relative to the
749     /// current position of the stack pointer (e.g. [sp + (sp_offset -
750     /// offset)].
751     fn address_from_sp(&self, offset: SPOffset) -> Result<Self::Address>;
752 
753     /// Constructs an address with an offset that is absolute to the
754     /// current position of the stack pointer (e.g. [sp + offset].
755     fn address_at_sp(&self, offset: SPOffset) -> Result<Self::Address>;
756 
757     /// Alias for [`Self::address_at_reg`] using the VMContext register as
758     /// a base. The VMContext register is derived from the ABI type that is
759     /// associated to the MacroAssembler.
760     fn address_at_vmctx(&self, offset: u32) -> Result<Self::Address>;
761 
762     /// Construct an address that is absolute to the current position
763     /// of the given register.
764     fn address_at_reg(&self, reg: Reg, offset: u32) -> Result<Self::Address>;
765 
766     /// Emit a function call to either a local or external function.
767     fn call(
768         &mut self,
769         stack_args_size: u32,
770         f: impl FnMut(&mut Self) -> Result<(CalleeKind, CallingConvention)>,
771     ) -> Result<u32>;
772 
773     /// Get stack pointer offset.
774     fn sp_offset(&self) -> Result<SPOffset>;
775 
776     /// Perform a stack store.
777     fn store(&mut self, src: RegImm, dst: Self::Address, size: OperandSize) -> Result<()>;
778 
779     /// Alias for `MacroAssembler::store` with the operand size corresponding
780     /// to the pointer size of the target.
781     fn store_ptr(&mut self, src: Reg, dst: Self::Address) -> Result<()>;
782 
783     /// Perform a WebAssembly store.
784     /// A WebAssembly store introduces several additional invariants compared to
785     /// [Self::store], more precisely, it can implicitly trap, in certain
786     /// circumstances, even if explicit bounds checks are elided, in that sense,
787     /// we consider this type of load as untrusted. It can also differ with
788     /// regards to the endianness depending on the target ISA. For this reason,
789     /// [Self::wasm_store], should be explicitly used when emitting WebAssembly
790     /// stores.
791     fn wasm_store(
792         &mut self,
793         src: Reg,
794         dst: Self::Address,
795         size: OperandSize,
796         op_kind: MemOpKind,
797     ) -> Result<()>;
798 
799     /// Perform a zero-extended stack load.
800     fn load(&mut self, src: Self::Address, dst: WritableReg, size: OperandSize) -> Result<()>;
801 
802     /// Perform a WebAssembly load.
803     /// A WebAssembly load introduces several additional invariants compared to
804     /// [Self::load], more precisely, it can implicitly trap, in certain
805     /// circumstances, even if explicit bounds checks are elided, in that sense,
806     /// we consider this type of load as untrusted. It can also differ with
807     /// regards to the endianness depending on the target ISA. For this reason,
808     /// [Self::wasm_load], should be explicitly used when emitting WebAssembly
809     /// loads.
810     fn wasm_load(
811         &mut self,
812         src: Self::Address,
813         dst: WritableReg,
814         kind: LoadKind,
815         op_kind: MemOpKind,
816     ) -> Result<()>;
817 
818     /// Alias for `MacroAssembler::load` with the operand size corresponding
819     /// to the pointer size of the target.
820     fn load_ptr(&mut self, src: Self::Address, dst: WritableReg) -> Result<()>;
821 
822     /// Loads the effective address into destination.
823     fn load_addr(
824         &mut self,
825         _src: Self::Address,
826         _dst: WritableReg,
827         _size: OperandSize,
828     ) -> Result<()>;
829 
830     /// Pop a value from the machine stack into the given register.
831     fn pop(&mut self, dst: WritableReg, size: OperandSize) -> Result<()>;
832 
833     /// Perform a move.
834     fn mov(&mut self, dst: WritableReg, src: RegImm, size: OperandSize) -> Result<()>;
835 
836     /// Perform a conditional move.
837     fn cmov(&mut self, dst: WritableReg, src: Reg, cc: IntCmpKind, size: OperandSize)
838         -> Result<()>;
839 
840     /// Performs a memory move of bytes from src to dest.
841     /// Bytes are moved in blocks of 8 bytes, where possible.
842     fn memmove(
843         &mut self,
844         src: SPOffset,
845         dst: SPOffset,
846         bytes: u32,
847         direction: MemMoveDirection,
848     ) -> Result<()> {
849         match direction {
850             MemMoveDirection::LowToHigh => debug_assert!(dst.as_u32() < src.as_u32()),
851             MemMoveDirection::HighToLow => debug_assert!(dst.as_u32() > src.as_u32()),
852         }
853         // At least 4 byte aligned.
854         debug_assert!(bytes % 4 == 0);
855         let mut remaining = bytes;
856         let word_bytes = <Self::ABI as abi::ABI>::word_bytes();
857         let scratch = scratch!(Self);
858 
859         let mut dst_offs = dst.as_u32() - bytes;
860         let mut src_offs = src.as_u32() - bytes;
861 
862         let word_bytes = word_bytes as u32;
863         while remaining >= word_bytes {
864             remaining -= word_bytes;
865             dst_offs += word_bytes;
866             src_offs += word_bytes;
867 
868             self.load_ptr(
869                 self.address_from_sp(SPOffset::from_u32(src_offs))?,
870                 writable!(scratch),
871             )?;
872             self.store_ptr(
873                 scratch.into(),
874                 self.address_from_sp(SPOffset::from_u32(dst_offs))?,
875             )?;
876         }
877 
878         if remaining > 0 {
879             let half_word = word_bytes / 2;
880             let ptr_size = OperandSize::from_bytes(half_word as u8);
881             debug_assert!(remaining == half_word);
882             dst_offs += half_word;
883             src_offs += half_word;
884 
885             self.load(
886                 self.address_from_sp(SPOffset::from_u32(src_offs))?,
887                 writable!(scratch),
888                 ptr_size,
889             )?;
890             self.store(
891                 scratch.into(),
892                 self.address_from_sp(SPOffset::from_u32(dst_offs))?,
893                 ptr_size,
894             )?;
895         }
896         Ok(())
897     }
898 
899     /// Perform add operation.
900     fn add(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>;
901 
902     /// Perform a checked unsigned integer addition, emitting the provided trap
903     /// if the addition overflows.
904     fn checked_uadd(
905         &mut self,
906         dst: WritableReg,
907         lhs: Reg,
908         rhs: RegImm,
909         size: OperandSize,
910         trap: TrapCode,
911     ) -> Result<()>;
912 
913     /// Perform subtraction operation.
914     fn sub(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>;
915 
916     /// Perform multiplication operation.
917     fn mul(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>;
918 
919     /// Perform a floating point add operation.
920     fn float_add(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>;
921 
922     /// Perform a floating point subtraction operation.
923     fn float_sub(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>;
924 
925     /// Perform a floating point multiply operation.
926     fn float_mul(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>;
927 
928     /// Perform a floating point divide operation.
929     fn float_div(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>;
930 
931     /// Perform a floating point minimum operation. In x86, this will emit
932     /// multiple instructions.
933     fn float_min(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>;
934 
935     /// Perform a floating point maximum operation. In x86, this will emit
936     /// multiple instructions.
937     fn float_max(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>;
938 
939     /// Perform a floating point copysign operation. In x86, this will emit
940     /// multiple instructions.
941     fn float_copysign(
942         &mut self,
943         dst: WritableReg,
944         lhs: Reg,
945         rhs: Reg,
946         size: OperandSize,
947     ) -> Result<()>;
948 
949     /// Perform a floating point abs operation.
950     fn float_abs(&mut self, dst: WritableReg, size: OperandSize) -> Result<()>;
951 
952     /// Perform a floating point negation operation.
953     fn float_neg(&mut self, dst: WritableReg, size: OperandSize) -> Result<()>;
954 
955     /// Perform a floating point floor operation.
956     fn float_round<
957         F: FnMut(&mut FuncEnv<Self::Ptr>, &mut CodeGenContext<Emission>, &mut Self) -> Result<()>,
958     >(
959         &mut self,
960         mode: RoundingMode,
961         env: &mut FuncEnv<Self::Ptr>,
962         context: &mut CodeGenContext<Emission>,
963         size: OperandSize,
964         fallback: F,
965     ) -> Result<()>;
966 
967     /// Perform a floating point square root operation.
968     fn float_sqrt(&mut self, dst: WritableReg, src: Reg, size: OperandSize) -> Result<()>;
969 
970     /// Perform logical and operation.
971     fn and(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>;
972 
973     /// Perform logical or operation.
974     fn or(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>;
975 
976     /// Perform logical exclusive or operation.
977     fn xor(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>;
978 
979     /// Perform a shift operation between a register and an immediate.
980     fn shift_ir(
981         &mut self,
982         dst: WritableReg,
983         imm: u64,
984         lhs: Reg,
985         kind: ShiftKind,
986         size: OperandSize,
987     ) -> Result<()>;
988 
989     /// Perform a shift operation between two registers.
990     /// This case is special in that some architectures have specific expectations
991     /// regarding the location of the instruction arguments. To free the
992     /// caller from having to deal with the architecture specific constraints
993     /// we give this function access to the code generation context, allowing
994     /// each implementation to decide the lowering path.
995     fn shift(
996         &mut self,
997         context: &mut CodeGenContext<Emission>,
998         kind: ShiftKind,
999         size: OperandSize,
1000     ) -> Result<()>;
1001 
1002     /// Perform division operation.
1003     /// Division is special in that some architectures have specific
1004     /// expectations regarding the location of the instruction
1005     /// arguments and regarding the location of the quotient /
1006     /// remainder. To free the caller from having to deal with the
1007     /// architecture specific constraints we give this function access
1008     /// to the code generation context, allowing each implementation
1009     /// to decide the lowering path.  For cases in which division is a
1010     /// unconstrained binary operation, the caller can decide to use
1011     /// the `CodeGenContext::i32_binop` or `CodeGenContext::i64_binop`
1012     /// functions.
1013     fn div(
1014         &mut self,
1015         context: &mut CodeGenContext<Emission>,
1016         kind: DivKind,
1017         size: OperandSize,
1018     ) -> Result<()>;
1019 
1020     /// Calculate remainder.
1021     fn rem(
1022         &mut self,
1023         context: &mut CodeGenContext<Emission>,
1024         kind: RemKind,
1025         size: OperandSize,
1026     ) -> Result<()>;
1027 
1028     /// Compares `src1` against `src2` for the side effect of setting processor
1029     /// flags.
1030     ///
1031     /// Note that `src1` is the left-hand-side of the comparison and `src2` is
1032     /// the right-hand-side, so if testing `a < b` then `src1 == a` and
1033     /// `src2 == b`
1034     fn cmp(&mut self, src1: Reg, src2: RegImm, size: OperandSize) -> Result<()>;
1035 
1036     /// Compare src and dst and put the result in dst.
1037     /// This function will potentially emit a series of instructions.
1038     ///
1039     /// The initial value in `dst` is the left-hand-side of the comparison and
1040     /// the initial value in `src` is the right-hand-side of the comparison.
1041     /// That means for `a < b` then `dst == a` and `src == b`.
1042     fn cmp_with_set(
1043         &mut self,
1044         dst: WritableReg,
1045         src: RegImm,
1046         kind: IntCmpKind,
1047         size: OperandSize,
1048     ) -> Result<()>;
1049 
1050     /// Compare floats in src1 and src2 and put the result in dst.
1051     /// In x86, this will emit multiple instructions.
1052     fn float_cmp_with_set(
1053         &mut self,
1054         dst: WritableReg,
1055         src1: Reg,
1056         src2: Reg,
1057         kind: FloatCmpKind,
1058         size: OperandSize,
1059     ) -> Result<()>;
1060 
1061     /// Count the number of leading zeroes in src and put the result in dst.
1062     /// In x64, this will emit multiple instructions if the `has_lzcnt` flag is
1063     /// false.
1064     fn clz(&mut self, dst: WritableReg, src: Reg, size: OperandSize) -> Result<()>;
1065 
1066     /// Count the number of trailing zeroes in src and put the result in dst.masm
1067     /// In x64, this will emit multiple instructions if the `has_tzcnt` flag is
1068     /// false.
1069     fn ctz(&mut self, dst: WritableReg, src: Reg, size: OperandSize) -> Result<()>;
1070 
1071     /// Push the register to the stack, returning the stack slot metadata.
1072     // NB
1073     // The stack alignment should not be assumed after any call to `push`,
1074     // unless explicitly aligned otherwise.  Typically, stack alignment is
1075     // maintained at call sites and during the execution of
1076     // epilogues.
1077     fn push(&mut self, src: Reg, size: OperandSize) -> Result<StackSlot>;
1078 
1079     /// Finalize the assembly and return the result.
1080     fn finalize(self, base: Option<SourceLoc>) -> Result<MachBufferFinalized<Final>>;
1081 
1082     /// Zero a particular register.
1083     fn zero(&mut self, reg: WritableReg) -> Result<()>;
1084 
1085     /// Count the number of 1 bits in src and put the result in dst. In x64,
1086     /// this will emit multiple instructions if the `has_popcnt` flag is false.
1087     fn popcnt(&mut self, context: &mut CodeGenContext<Emission>, size: OperandSize) -> Result<()>;
1088 
1089     /// Converts an i64 to an i32 by discarding the high 32 bits.
1090     fn wrap(&mut self, dst: WritableReg, src: Reg) -> Result<()>;
1091 
1092     /// Extends an integer of a given size to a larger size.
1093     fn extend(&mut self, dst: WritableReg, src: Reg, kind: ExtendKind) -> Result<()>;
1094 
1095     /// Emits one or more instructions to perform a signed truncation of a
1096     /// float into an integer.
1097     fn signed_truncate(
1098         &mut self,
1099         dst: WritableReg,
1100         src: Reg,
1101         src_size: OperandSize,
1102         dst_size: OperandSize,
1103         kind: TruncKind,
1104     ) -> Result<()>;
1105 
1106     /// Emits one or more instructions to perform an unsigned truncation of a
1107     /// float into an integer.
1108     fn unsigned_truncate(
1109         &mut self,
1110         context: &mut CodeGenContext<Emission>,
1111         src_size: OperandSize,
1112         dst_size: OperandSize,
1113         kind: TruncKind,
1114     ) -> Result<()>;
1115 
1116     /// Emits one or more instructions to perform a signed convert of an
1117     /// integer into a float.
1118     fn signed_convert(
1119         &mut self,
1120         dst: WritableReg,
1121         src: Reg,
1122         src_size: OperandSize,
1123         dst_size: OperandSize,
1124     ) -> Result<()>;
1125 
1126     /// Emits one or more instructions to perform an unsigned convert of an
1127     /// integer into a float.
1128     fn unsigned_convert(
1129         &mut self,
1130         dst: WritableReg,
1131         src: Reg,
1132         tmp_gpr: Reg,
1133         src_size: OperandSize,
1134         dst_size: OperandSize,
1135     ) -> Result<()>;
1136 
1137     /// Reinterpret a float as an integer.
1138     fn reinterpret_float_as_int(
1139         &mut self,
1140         dst: WritableReg,
1141         src: Reg,
1142         size: OperandSize,
1143     ) -> Result<()>;
1144 
1145     /// Reinterpret an integer as a float.
1146     fn reinterpret_int_as_float(
1147         &mut self,
1148         dst: WritableReg,
1149         src: Reg,
1150         size: OperandSize,
1151     ) -> Result<()>;
1152 
1153     /// Demote an f64 to an f32.
1154     fn demote(&mut self, dst: WritableReg, src: Reg) -> Result<()>;
1155 
1156     /// Promote an f32 to an f64.
1157     fn promote(&mut self, dst: WritableReg, src: Reg) -> Result<()>;
1158 
1159     /// Zero a given memory range.
1160     ///
1161     /// The default implementation divides the given memory range
1162     /// into word-sized slots. Then it unrolls a series of store
1163     /// instructions, effectively assigning zero to each slot.
1164     fn zero_mem_range(&mut self, mem: &Range<u32>) -> Result<()> {
1165         let word_size = <Self::ABI as abi::ABI>::word_bytes() as u32;
1166         if mem.is_empty() {
1167             return Ok(());
1168         }
1169 
1170         let start = if mem.start % word_size == 0 {
1171             mem.start
1172         } else {
1173             // Ensure that the start of the range is at least 4-byte aligned.
1174             assert!(mem.start % 4 == 0);
1175             let start = align_to(mem.start, word_size);
1176             let addr: Self::Address = self.local_address(&LocalSlot::i32(start))?;
1177             self.store(RegImm::i32(0), addr, OperandSize::S32)?;
1178             // Ensure that the new start of the range, is word-size aligned.
1179             assert!(start % word_size == 0);
1180             start
1181         };
1182 
1183         let end = align_to(mem.end, word_size);
1184         let slots = (end - start) / word_size;
1185 
1186         if slots == 1 {
1187             let slot = LocalSlot::i64(start + word_size);
1188             let addr: Self::Address = self.local_address(&slot)?;
1189             self.store(RegImm::i64(0), addr, OperandSize::S64)?;
1190         } else {
1191             // TODO
1192             // Add an upper bound to this generation;
1193             // given a considerably large amount of slots
1194             // this will be inefficient.
1195             let zero = scratch!(Self);
1196             self.zero(writable!(zero))?;
1197             let zero = RegImm::reg(zero);
1198 
1199             for step in (start..end).into_iter().step_by(word_size as usize) {
1200                 let slot = LocalSlot::i64(step + word_size);
1201                 let addr: Self::Address = self.local_address(&slot)?;
1202                 self.store(zero, addr, OperandSize::S64)?;
1203             }
1204         }
1205 
1206         Ok(())
1207     }
1208 
1209     /// Generate a label.
1210     fn get_label(&mut self) -> Result<MachLabel>;
1211 
1212     /// Bind the given label at the current code offset.
1213     fn bind(&mut self, label: MachLabel) -> Result<()>;
1214 
1215     /// Conditional branch.
1216     ///
1217     /// Performs a comparison between the two operands,
1218     /// and immediately after emits a jump to the given
1219     /// label destination if the condition is met.
1220     fn branch(
1221         &mut self,
1222         kind: IntCmpKind,
1223         lhs: Reg,
1224         rhs: RegImm,
1225         taken: MachLabel,
1226         size: OperandSize,
1227     ) -> Result<()>;
1228 
1229     /// Emits and unconditional jump to the given label.
1230     fn jmp(&mut self, target: MachLabel) -> Result<()>;
1231 
1232     /// Emits a jump table sequence. The default label is specified as
1233     /// the last element of the targets slice.
1234     fn jmp_table(&mut self, targets: &[MachLabel], index: Reg, tmp: Reg) -> Result<()>;
1235 
1236     /// Emit an unreachable code trap.
1237     fn unreachable(&mut self) -> Result<()>;
1238 
1239     /// Emit an unconditional trap.
1240     fn trap(&mut self, code: TrapCode) -> Result<()>;
1241 
1242     /// Traps if the condition code is met.
1243     fn trapif(&mut self, cc: IntCmpKind, code: TrapCode) -> Result<()>;
1244 
1245     /// Trap if the source register is zero.
1246     fn trapz(&mut self, src: Reg, code: TrapCode) -> Result<()>;
1247 
1248     /// Ensures that the stack pointer is correctly positioned before an unconditional
1249     /// jump according to the requirements of the destination target.
1250     fn ensure_sp_for_jump(&mut self, target: SPOffset) -> Result<()> {
1251         let bytes = self
1252             .sp_offset()?
1253             .as_u32()
1254             .checked_sub(target.as_u32())
1255             .unwrap_or(0);
1256 
1257         if bytes > 0 {
1258             self.free_stack(bytes)?;
1259         }
1260 
1261         Ok(())
1262     }
1263 
1264     /// Mark the start of a source location returning the machine code offset
1265     /// and the relative source code location.
1266     fn start_source_loc(&mut self, loc: RelSourceLoc) -> Result<(CodeOffset, RelSourceLoc)>;
1267 
1268     /// Mark the end of a source location.
1269     fn end_source_loc(&mut self) -> Result<()>;
1270 
1271     /// The current offset, in bytes from the beginning of the function.
1272     fn current_code_offset(&self) -> Result<CodeOffset>;
1273 
1274     /// Performs a 128-bit addition
1275     fn add128(
1276         &mut self,
1277         dst_lo: WritableReg,
1278         dst_hi: WritableReg,
1279         lhs_lo: Reg,
1280         lhs_hi: Reg,
1281         rhs_lo: Reg,
1282         rhs_hi: Reg,
1283     ) -> Result<()>;
1284 
1285     /// Performs a 128-bit subtraction
1286     fn sub128(
1287         &mut self,
1288         dst_lo: WritableReg,
1289         dst_hi: WritableReg,
1290         lhs_lo: Reg,
1291         lhs_hi: Reg,
1292         rhs_lo: Reg,
1293         rhs_hi: Reg,
1294     ) -> Result<()>;
1295 
1296     /// Performs a widening multiplication from two 64-bit operands into a
1297     /// 128-bit result.
1298     ///
1299     /// Note that some platforms require special handling of registers in this
1300     /// instruction (e.g. x64) so full access to `CodeGenContext` is provided.
1301     fn mul_wide(&mut self, context: &mut CodeGenContext<Emission>, kind: MulWideKind)
1302         -> Result<()>;
1303 
1304     /// Takes the value in a src operand and replicates it across lanes of
1305     /// `size` in a destination result.
1306     fn splat(&mut self, context: &mut CodeGenContext<Emission>, size: SplatKind) -> Result<()>;
1307 
1308     /// Performs a shuffle between two 128-bit vectors into a 128-bit result
1309     /// using lanes as a mask to select which indexes to copy.
1310     fn shuffle(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, lanes: [u8; 16]) -> Result<()>;
1311 
1312     /// Performs the RMW `op` operation on the passed `addr`.
1313     ///
1314     /// The value *before* the operation was performed is written back to the `operand` register.
1315     fn atomic_rmw(
1316         &mut self,
1317         addr: Self::Address,
1318         operand: WritableReg,
1319         size: OperandSize,
1320         op: RmwOp,
1321         flags: MemFlags,
1322         extend: Option<ExtendKind>,
1323     ) -> Result<()>;
1324 }
1325