xref: /wasmtime-44.0.1/winch/codegen/src/masm.rs (revision a0442ea0)
1 use crate::abi::{self, align_to, scratch, LocalSlot};
2 use crate::codegen::{CodeGenContext, FuncEnv};
3 use crate::isa::reg::Reg;
4 use cranelift_codegen::{
5     binemit::CodeOffset,
6     ir::{Endianness, LibCall, MemFlags, RelSourceLoc, SourceLoc, UserExternalNameRef},
7     Final, MachBufferFinalized, MachLabel,
8 };
9 use std::{fmt::Debug, ops::Range};
10 use wasmtime_environ::PtrSize;
11 
12 pub(crate) use cranelift_codegen::ir::TrapCode;
13 
14 #[derive(Eq, PartialEq)]
15 pub(crate) enum DivKind {
16     /// Signed division.
17     Signed,
18     /// Unsigned division.
19     Unsigned,
20 }
21 
22 /// Remainder kind.
23 pub(crate) enum RemKind {
24     /// Signed remainder.
25     Signed,
26     /// Unsigned remainder.
27     Unsigned,
28 }
29 
30 /// The direction to perform the memory move.
31 #[derive(Debug, Clone, Eq, PartialEq)]
32 pub(crate) enum MemMoveDirection {
33     /// From high memory addresses to low memory addresses.
34     /// Invariant: the source location is closer to the FP than the destination
35     /// location, which will be closer to the SP.
36     HighToLow,
37     /// From low memory addresses to high memory addresses.
38     /// Invariant: the source location is closer to the SP than the destination
39     /// location, which will be closer to the FP.
40     LowToHigh,
41 }
42 
43 /// Classifies how to treat float-to-int conversions.
44 #[derive(Debug, Copy, Clone, Eq, PartialEq)]
45 pub(crate) enum TruncKind {
46     /// Saturating conversion. If the source value is greater than the maximum
47     /// value of the destination type, the result is clamped to the
48     /// destination maximum value.
49     Checked,
50     /// An exception is raised if the source value is greater than the maximum
51     /// value of the destination type.
52     Unchecked,
53 }
54 
55 impl TruncKind {
56     /// Returns true if the truncation kind is checked.
57     pub(crate) fn is_checked(&self) -> bool {
58         *self == TruncKind::Checked
59     }
60 }
61 
62 /// Representation of the stack pointer offset.
63 #[derive(Copy, Clone, Eq, PartialEq, Debug, PartialOrd, Ord, Default)]
64 pub struct SPOffset(u32);
65 
66 impl SPOffset {
67     pub fn from_u32(offs: u32) -> Self {
68         Self(offs)
69     }
70 
71     pub fn as_u32(&self) -> u32 {
72         self.0
73     }
74 }
75 
76 /// A stack slot.
77 #[derive(Debug, Clone, Copy, Eq, PartialEq)]
78 pub struct StackSlot {
79     /// The location of the slot, relative to the stack pointer.
80     pub offset: SPOffset,
81     /// The size of the slot, in bytes.
82     pub size: u32,
83 }
84 
85 impl StackSlot {
86     pub fn new(offs: SPOffset, size: u32) -> Self {
87         Self { offset: offs, size }
88     }
89 }
90 
91 /// Kinds of integer binary comparison in WebAssembly. The [`MacroAssembler`]
92 /// implementation for each ISA is responsible for emitting the correct
93 /// sequence of instructions when lowering to machine code.
94 #[derive(Debug, Clone, Copy, Eq, PartialEq)]
95 pub(crate) enum IntCmpKind {
96     /// Equal.
97     Eq,
98     /// Not equal.
99     Ne,
100     /// Signed less than.
101     LtS,
102     /// Unsigned less than.
103     LtU,
104     /// Signed greater than.
105     GtS,
106     /// Unsigned greater than.
107     GtU,
108     /// Signed less than or equal.
109     LeS,
110     /// Unsigned less than or equal.
111     LeU,
112     /// Signed greater than or equal.
113     GeS,
114     /// Unsigned greater than or equal.
115     GeU,
116 }
117 
118 /// Kinds of float binary comparison in WebAssembly. The [`MacroAssembler`]
119 /// implementation for each ISA is responsible for emitting the correct
120 /// sequence of instructions when lowering code.
121 #[derive(Debug)]
122 pub(crate) enum FloatCmpKind {
123     /// Equal.
124     Eq,
125     /// Not equal.
126     Ne,
127     /// Less than.
128     Lt,
129     /// Greater than.
130     Gt,
131     /// Less than or equal.
132     Le,
133     /// Greater than or equal.
134     Ge,
135 }
136 
137 /// Kinds of shifts in WebAssembly.The [`masm`] implementation for each ISA is
138 /// responsible for emitting the correct sequence of instructions when
139 /// lowering to machine code.
140 #[derive(Debug, Clone, Copy, Eq, PartialEq)]
141 pub(crate) enum ShiftKind {
142     /// Left shift.
143     Shl,
144     /// Signed right shift.
145     ShrS,
146     /// Unsigned right shift.
147     ShrU,
148     /// Left rotate.
149     Rotl,
150     /// Right rotate.
151     Rotr,
152 }
153 
154 /// Kinds of extends in WebAssembly. Each MacroAssembler implementation
155 /// is responsible for emitting the correct sequence of instructions when
156 /// lowering to machine code.
157 pub(crate) enum ExtendKind {
158     /// Sign extends i32 to i64.
159     I64ExtendI32S,
160     /// Zero extends i32 to i64.
161     I64ExtendI32U,
162     // Sign extends the 8 least significant bits to 32 bits.
163     I32Extend8S,
164     // Sign extends the 16 least significant bits to 32 bits.
165     I32Extend16S,
166     /// Sign extends the 8 least significant bits to 64 bits.
167     I64Extend8S,
168     /// Sign extends the 16 least significant bits to 64 bits.
169     I64Extend16S,
170     /// Sign extends the 32 least significant bits to 64 bits.
171     I64Extend32S,
172 }
173 
174 /// Operand size, in bits.
175 #[derive(Copy, Debug, Clone, Eq, PartialEq)]
176 pub(crate) enum OperandSize {
177     /// 8 bits.
178     S8,
179     /// 16 bits.
180     S16,
181     /// 32 bits.
182     S32,
183     /// 64 bits.
184     S64,
185     /// 128 bits.
186     S128,
187 }
188 
189 impl OperandSize {
190     /// The number of bits in the operand.
191     pub fn num_bits(&self) -> u8 {
192         match self {
193             OperandSize::S8 => 8,
194             OperandSize::S16 => 16,
195             OperandSize::S32 => 32,
196             OperandSize::S64 => 64,
197             OperandSize::S128 => 128,
198         }
199     }
200 
201     /// The number of bytes in the operand.
202     pub fn bytes(&self) -> u32 {
203         match self {
204             Self::S8 => 1,
205             Self::S16 => 2,
206             Self::S32 => 4,
207             Self::S64 => 8,
208             Self::S128 => 16,
209         }
210     }
211 
212     /// The binary logarithm of the number of bits in the operand.
213     pub fn log2(&self) -> u8 {
214         match self {
215             OperandSize::S8 => 3,
216             OperandSize::S16 => 4,
217             OperandSize::S32 => 5,
218             OperandSize::S64 => 6,
219             OperandSize::S128 => 7,
220         }
221     }
222 
223     /// Create an [`OperandSize`]  from the given number of bytes.
224     pub fn from_bytes(bytes: u8) -> Self {
225         use OperandSize::*;
226         match bytes {
227             4 => S32,
228             8 => S64,
229             16 => S128,
230             _ => panic!("Invalid bytes {bytes} for OperandSize"),
231         }
232     }
233 }
234 
235 /// An abstraction over a register or immediate.
236 #[derive(Copy, Clone, Debug, PartialEq, Eq)]
237 pub(crate) enum RegImm {
238     /// A register.
239     Reg(Reg),
240     /// A tagged immediate argument.
241     Imm(Imm),
242 }
243 
244 /// An tagged representation of an immediate.
245 #[derive(Copy, Clone, Debug, PartialEq, Eq)]
246 pub(crate) enum Imm {
247     /// I32 immediate.
248     I32(u32),
249     /// I64 immediate.
250     I64(u64),
251     /// F32 immediate.
252     F32(u32),
253     /// F64 immediate.
254     F64(u64),
255     /// V128 immediate.
256     V128(i128),
257 }
258 
259 impl Imm {
260     /// Create a new I64 immediate.
261     pub fn i64(val: i64) -> Self {
262         Self::I64(val as u64)
263     }
264 
265     /// Create a new I32 immediate.
266     pub fn i32(val: i32) -> Self {
267         Self::I32(val as u32)
268     }
269 
270     /// Create a new F32 immediate.
271     pub fn f32(bits: u32) -> Self {
272         Self::F32(bits)
273     }
274 
275     /// Create a new F64 immediate.
276     pub fn f64(bits: u64) -> Self {
277         Self::F64(bits)
278     }
279 
280     /// Create a new V128 immediate.
281     pub fn v128(bits: i128) -> Self {
282         Self::V128(bits)
283     }
284 
285     /// Convert the immediate to i32, if possible.
286     pub fn to_i32(&self) -> Option<i32> {
287         match self {
288             Self::I32(v) => Some(*v as i32),
289             Self::I64(v) => i32::try_from(*v as i64).ok(),
290             _ => None,
291         }
292     }
293 }
294 
295 /// The location of the [VMcontext] used for function calls.
296 #[derive(Copy, Clone, Debug, Eq, PartialEq)]
297 pub(crate) enum VMContextLoc {
298     /// Dynamic, stored in the given register.
299     Reg(Reg),
300     /// The pinned [VMContext] register.
301     Pinned,
302 }
303 
304 /// The maximum number of context arguments currently used across the compiler.
305 pub(crate) const MAX_CONTEXT_ARGS: usize = 2;
306 
307 /// Out-of-band special purpose arguments used for function call emission.
308 ///
309 /// We cannot rely on the value stack for these values given that inserting
310 /// register or memory values at arbitrary locations of the value stack has the
311 /// potential to break the stack ordering principle, which states that older
312 /// values must always precede newer values, effectively simulating the order of
313 /// values in the machine stack.
314 /// The [ContextArgs] are meant to be resolved at every callsite; in some cases
315 /// it might be possible to construct it early on, but given that it might
316 /// contain allocatable registers, it's preferred to construct it in
317 /// [FnCall::emit].
318 #[derive(Clone, Debug)]
319 pub(crate) enum ContextArgs {
320     /// No context arguments required. This is used for libcalls that don't
321     /// require any special context arguments. For example builtin functions
322     /// that perform float calculations.
323     None,
324     /// A single context argument is required; the current pinned [VMcontext]
325     /// register must be passed as the first argument of the function call.
326     VMContext([VMContextLoc; 1]),
327     /// The callee and caller context arguments are required. In this case, the
328     /// callee context argument is usually stored into an allocatable register
329     /// and the caller is always the current pinned [VMContext] pointer.
330     CalleeAndCallerVMContext([VMContextLoc; MAX_CONTEXT_ARGS]),
331 }
332 
333 impl ContextArgs {
334     /// Construct an empty [ContextArgs].
335     pub fn none() -> Self {
336         Self::None
337     }
338 
339     /// Construct a [ContextArgs] declaring the usage of the pinned [VMContext]
340     /// register as both the caller and callee context arguments.
341     pub fn pinned_callee_and_caller_vmctx() -> Self {
342         Self::CalleeAndCallerVMContext([VMContextLoc::Pinned, VMContextLoc::Pinned])
343     }
344 
345     /// Construct a [ContextArgs] that declares the usage of the pinned
346     /// [VMContext] register as the only context argument.
347     pub fn pinned_vmctx() -> Self {
348         Self::VMContext([VMContextLoc::Pinned])
349     }
350 
351     /// Construct a [ContextArgs] that declares a dynamic callee context and the
352     /// pinned [VMContext] register as the context arguments.
353     pub fn with_callee_and_pinned_caller(callee_vmctx: Reg) -> Self {
354         Self::CalleeAndCallerVMContext([VMContextLoc::Reg(callee_vmctx), VMContextLoc::Pinned])
355     }
356 
357     /// Get the length of the [ContextArgs].
358     pub fn len(&self) -> usize {
359         self.as_slice().len()
360     }
361 
362     /// Get a slice of the context arguments.
363     pub fn as_slice(&self) -> &[VMContextLoc] {
364         match self {
365             Self::None => &[],
366             Self::VMContext(a) => a.as_slice(),
367             Self::CalleeAndCallerVMContext(a) => a.as_slice(),
368         }
369     }
370 }
371 
372 #[derive(Copy, Clone, Debug)]
373 pub(crate) enum CalleeKind {
374     /// A function call to a raw address.
375     Indirect(Reg),
376     /// A function call to a local function.
377     Direct(UserExternalNameRef),
378     /// Call to a well known LibCall.
379     LibCall(LibCall),
380 }
381 
382 impl CalleeKind {
383     /// Creates a callee kind from a register.
384     pub fn indirect(reg: Reg) -> Self {
385         Self::Indirect(reg)
386     }
387 
388     /// Creates a direct callee kind from a function name.
389     pub fn direct(name: UserExternalNameRef) -> Self {
390         Self::Direct(name)
391     }
392 
393     /// Creates a known callee kind from a libcall.
394     pub fn libcall(call: LibCall) -> Self {
395         Self::LibCall(call)
396     }
397 }
398 
399 impl RegImm {
400     /// Register constructor.
401     pub fn reg(r: Reg) -> Self {
402         RegImm::Reg(r)
403     }
404 
405     /// I64 immediate constructor.
406     pub fn i64(val: i64) -> Self {
407         RegImm::Imm(Imm::i64(val))
408     }
409 
410     /// I32 immediate constructor.
411     pub fn i32(val: i32) -> Self {
412         RegImm::Imm(Imm::i32(val))
413     }
414 
415     /// F32 immediate, stored using its bits representation.
416     // Temporary until support for f32.const is added.
417     #[allow(dead_code)]
418     pub fn f32(bits: u32) -> Self {
419         RegImm::Imm(Imm::f32(bits))
420     }
421 
422     /// F64 immediate, stored using its bits representation.
423     // Temporary until support for f64.const is added.
424     #[allow(dead_code)]
425     pub fn f64(bits: u64) -> Self {
426         RegImm::Imm(Imm::f64(bits))
427     }
428 
429     /// V128 immediate.
430     pub fn v128(bits: i128) -> Self {
431         RegImm::Imm(Imm::v128(bits))
432     }
433 }
434 
435 impl From<Reg> for RegImm {
436     fn from(r: Reg) -> Self {
437         Self::Reg(r)
438     }
439 }
440 
441 #[derive(Debug)]
442 pub enum RoundingMode {
443     Nearest,
444     Up,
445     Down,
446     Zero,
447 }
448 
449 /// Memory flags for trusted loads/stores.
450 pub const TRUSTED_FLAGS: MemFlags = MemFlags::trusted();
451 
452 /// Flags used for WebAssembly loads / stores.
453 /// Untrusted by default so we don't set `no_trap`.
454 /// We also ensure that the endianness is the right one for WebAssembly.
455 pub const UNTRUSTED_FLAGS: MemFlags = MemFlags::new().with_endianness(Endianness::Little);
456 
457 /// Generic MacroAssembler interface used by the code generation.
458 ///
459 /// The MacroAssembler trait aims to expose an interface, high-level enough,
460 /// so that each ISA can provide its own lowering to machine code. For example,
461 /// for WebAssembly operators that don't have a direct mapping to a machine
462 /// a instruction, the interface defines a signature matching the WebAssembly
463 /// operator, allowing each implementation to lower such operator entirely.
464 /// This approach attributes more responsibility to the MacroAssembler, but frees
465 /// the caller from concerning about assembling the right sequence of
466 /// instructions at the operator callsite.
467 ///
468 /// The interface defaults to a three-argument form for binary operations;
469 /// this allows a natural mapping to instructions for RISC architectures,
470 /// that use three-argument form.
471 /// This approach allows for a more general interface that can be restricted
472 /// where needed, in the case of architectures that use a two-argument form.
473 
474 pub(crate) trait MacroAssembler {
475     /// The addressing mode.
476     type Address: Copy + Debug;
477 
478     /// The pointer representation of the target ISA,
479     /// used to access information from [`VMOffsets`].
480     type Ptr: PtrSize;
481 
482     /// The ABI details of the target.
483     type ABI: abi::ABI;
484 
485     /// Emit the function prologue.
486     fn prologue(&mut self, vmctx: Reg) {
487         self.frame_setup();
488         self.check_stack(vmctx);
489     }
490 
491     /// Generate the frame setup sequence.
492     fn frame_setup(&mut self);
493 
494     /// Generate the frame restore sequence.
495     fn frame_restore(&mut self);
496 
497     /// Emit a stack check.
498     fn check_stack(&mut self, vmctx: Reg);
499 
500     /// Emit the function epilogue.
501     fn epilogue(&mut self) {
502         self.frame_restore();
503     }
504 
505     /// Reserve stack space.
506     fn reserve_stack(&mut self, bytes: u32);
507 
508     /// Free stack space.
509     fn free_stack(&mut self, bytes: u32);
510 
511     /// Reset the stack pointer to the given offset;
512     ///
513     /// Used to reset the stack pointer to a given offset
514     /// when dealing with unreachable code.
515     fn reset_stack_pointer(&mut self, offset: SPOffset);
516 
517     /// Get the address of a local slot.
518     fn local_address(&mut self, local: &LocalSlot) -> Self::Address;
519 
520     /// Constructs an address with an offset that is relative to the
521     /// current position of the stack pointer (e.g. [sp + (sp_offset -
522     /// offset)].
523     fn address_from_sp(&self, offset: SPOffset) -> Self::Address;
524 
525     /// Constructs an address with an offset that is absolute to the
526     /// current position of the stack pointer (e.g. [sp + offset].
527     fn address_at_sp(&self, offset: SPOffset) -> Self::Address;
528 
529     /// Alias for [`Self::address_at_reg`] using the VMContext register as
530     /// a base. The VMContext register is derived from the ABI type that is
531     /// associated to the MacroAssembler.
532     fn address_at_vmctx(&self, offset: u32) -> Self::Address;
533 
534     /// Construct an address that is absolute to the current position
535     /// of the given register.
536     fn address_at_reg(&self, reg: Reg, offset: u32) -> Self::Address;
537 
538     /// Emit a function call to either a local or external function.
539     fn call(&mut self, stack_args_size: u32, f: impl FnMut(&mut Self) -> CalleeKind) -> u32;
540 
541     /// Get stack pointer offset.
542     fn sp_offset(&self) -> SPOffset;
543 
544     /// Perform a stack store.
545     fn store(&mut self, src: RegImm, dst: Self::Address, size: OperandSize);
546 
547     /// Alias for `MacroAssembler::store` with the operand size corresponding
548     /// to the pointer size of the target.
549     fn store_ptr(&mut self, src: Reg, dst: Self::Address);
550 
551     /// Perform a WebAssembly store.
552     /// A WebAssembly store introduces several additional invariants compared to
553     /// [Self::store], more precisely, it can implicitly trap, in certain
554     /// circumstances, even if explicit bounds checks are elided, in that sense,
555     /// we consider this type of load as untrusted. It can also differ with
556     /// regards to the endianness depending on the target ISA. For this reason,
557     /// [Self::wasm_store], should be explicitly used when emitting WebAssembly
558     /// stores.
559     fn wasm_store(&mut self, src: Reg, dst: Self::Address, size: OperandSize);
560 
561     /// Perform a zero-extended stack load.
562     fn load(&mut self, src: Self::Address, dst: Reg, size: OperandSize);
563 
564     /// Perform a WebAssembly load.
565     /// A WebAssembly load introduces several additional invariants compared to
566     /// [Self::load], more precisely, it can implicitly trap, in certain
567     /// circumstances, even if explicit bounds checks are elided, in that sense,
568     /// we consider this type of load as untrusted. It can also differ with
569     /// regards to the endianness depending on the target ISA. For this reason,
570     /// [Self::wasm_load], should be explicitly used when emitting WebAssembly
571     /// loads.
572     fn wasm_load(
573         &mut self,
574         src: Self::Address,
575         dst: Reg,
576         size: OperandSize,
577         kind: Option<ExtendKind>,
578     );
579 
580     /// Alias for `MacroAssembler::load` with the operand size corresponding
581     /// to the pointer size of the target.
582     fn load_ptr(&mut self, src: Self::Address, dst: Reg);
583 
584     /// Loads the effective address into destination.
585     fn load_addr(&mut self, _src: Self::Address, _dst: Reg, _size: OperandSize);
586 
587     /// Pop a value from the machine stack into the given register.
588     fn pop(&mut self, dst: Reg, size: OperandSize);
589 
590     /// Perform a move.
591     fn mov(&mut self, src: RegImm, dst: Reg, size: OperandSize);
592 
593     /// Perform a conditional move.
594     fn cmov(&mut self, src: Reg, dst: Reg, cc: IntCmpKind, size: OperandSize);
595 
596     /// Performs a memory move of bytes from src to dest.
597     /// Bytes are moved in blocks of 8 bytes, where possible.
598     fn memmove(&mut self, src: SPOffset, dst: SPOffset, bytes: u32, direction: MemMoveDirection) {
599         match direction {
600             MemMoveDirection::LowToHigh => debug_assert!(dst.as_u32() < src.as_u32()),
601             MemMoveDirection::HighToLow => debug_assert!(dst.as_u32() > src.as_u32()),
602         }
603         // At least 4 byte aligned.
604         debug_assert!(bytes % 4 == 0);
605         let mut remaining = bytes;
606         let word_bytes = <Self::ABI as abi::ABI>::word_bytes();
607         let scratch = scratch!(Self);
608 
609         let mut dst_offs = dst.as_u32() - bytes;
610         let mut src_offs = src.as_u32() - bytes;
611 
612         let word_bytes = word_bytes as u32;
613         while remaining >= word_bytes {
614             remaining -= word_bytes;
615             dst_offs += word_bytes;
616             src_offs += word_bytes;
617 
618             self.load_ptr(self.address_from_sp(SPOffset::from_u32(src_offs)), scratch);
619             self.store_ptr(
620                 scratch.into(),
621                 self.address_from_sp(SPOffset::from_u32(dst_offs)),
622             );
623         }
624 
625         if remaining > 0 {
626             let half_word = word_bytes / 2;
627             let ptr_size = OperandSize::from_bytes(half_word as u8);
628             debug_assert!(remaining == half_word);
629             dst_offs += half_word;
630             src_offs += half_word;
631 
632             self.load(
633                 self.address_from_sp(SPOffset::from_u32(src_offs)),
634                 scratch,
635                 ptr_size,
636             );
637             self.store(
638                 scratch.into(),
639                 self.address_from_sp(SPOffset::from_u32(dst_offs)),
640                 ptr_size,
641             );
642         }
643     }
644 
645     /// Perform add operation.
646     fn add(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize);
647 
648     /// Perform a checked unsigned integer addition, emitting the provided trap
649     /// if the addition overflows.
650     fn checked_uadd(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize, trap: TrapCode);
651 
652     /// Perform subtraction operation.
653     fn sub(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize);
654 
655     /// Perform multiplication operation.
656     fn mul(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize);
657 
658     /// Perform a floating point add operation.
659     fn float_add(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize);
660 
661     /// Perform a floating point subtraction operation.
662     fn float_sub(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize);
663 
664     /// Perform a floating point multiply operation.
665     fn float_mul(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize);
666 
667     /// Perform a floating point divide operation.
668     fn float_div(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize);
669 
670     /// Perform a floating point minimum operation. In x86, this will emit
671     /// multiple instructions.
672     fn float_min(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize);
673 
674     /// Perform a floating point maximum operation. In x86, this will emit
675     /// multiple instructions.
676     fn float_max(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize);
677 
678     /// Perform a floating point copysign operation. In x86, this will emit
679     /// multiple instructions.
680     fn float_copysign(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize);
681 
682     /// Perform a floating point abs operation.
683     fn float_abs(&mut self, dst: Reg, size: OperandSize);
684 
685     /// Perform a floating point negation operation.
686     fn float_neg(&mut self, dst: Reg, size: OperandSize);
687 
688     /// Perform a floating point floor operation.
689     fn float_round<F: FnMut(&mut FuncEnv<Self::Ptr>, &mut CodeGenContext, &mut Self)>(
690         &mut self,
691         mode: RoundingMode,
692         env: &mut FuncEnv<Self::Ptr>,
693         context: &mut CodeGenContext,
694         size: OperandSize,
695         fallback: F,
696     );
697 
698     /// Perform a floating point square root operation.
699     fn float_sqrt(&mut self, dst: Reg, src: Reg, size: OperandSize);
700 
701     /// Perform logical and operation.
702     fn and(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize);
703 
704     /// Perform logical or operation.
705     fn or(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize);
706 
707     /// Perform logical exclusive or operation.
708     fn xor(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize);
709 
710     /// Perform a shift operation between a register and an immediate.
711     fn shift_ir(&mut self, dst: Reg, imm: u64, lhs: Reg, kind: ShiftKind, size: OperandSize);
712 
713     /// Perform a shift operation between two registers.
714     /// This case is special in that some architectures have specific expectations
715     /// regarding the location of the instruction arguments. To free the
716     /// caller from having to deal with the architecture specific constraints
717     /// we give this function access to the code generation context, allowing
718     /// each implementation to decide the lowering path.
719     fn shift(&mut self, context: &mut CodeGenContext, kind: ShiftKind, size: OperandSize);
720 
721     /// Perform division operation.
722     /// Division is special in that some architectures have specific
723     /// expectations regarding the location of the instruction
724     /// arguments and regarding the location of the quotient /
725     /// remainder. To free the caller from having to deal with the
726     /// architecture specific constraints we give this function access
727     /// to the code generation context, allowing each implementation
728     /// to decide the lowering path.  For cases in which division is a
729     /// unconstrained binary operation, the caller can decide to use
730     /// the `CodeGenContext::i32_binop` or `CodeGenContext::i64_binop`
731     /// functions.
732     fn div(&mut self, context: &mut CodeGenContext, kind: DivKind, size: OperandSize);
733 
734     /// Calculate remainder.
735     fn rem(&mut self, context: &mut CodeGenContext, kind: RemKind, size: OperandSize);
736 
737     /// Compares `src1` against `src2` for the side effect of setting processor
738     /// flags.
739     ///
740     /// Note that `src1` is the left-hand-side of the comparison and `src2` is
741     /// the right-hand-side, so if testing `a < b` then `src1 == a` and
742     /// `src2 == b`
743     fn cmp(&mut self, src1: Reg, src2: RegImm, size: OperandSize);
744 
745     /// Compare src and dst and put the result in dst.
746     /// This function will potentially emit a series of instructions.
747     ///
748     /// The initial value in `dst` is the left-hand-side of the comparison and
749     /// the initial value in `src` is the right-hand-side of the comparison.
750     /// That means for `a < b` then `dst == a` and `src == b`.
751     fn cmp_with_set(&mut self, src: RegImm, dst: Reg, kind: IntCmpKind, size: OperandSize);
752 
753     /// Compare floats in src1 and src2 and put the result in dst.
754     /// In x86, this will emit multiple instructions.
755     fn float_cmp_with_set(
756         &mut self,
757         src1: Reg,
758         src2: Reg,
759         dst: Reg,
760         kind: FloatCmpKind,
761         size: OperandSize,
762     );
763 
764     /// Count the number of leading zeroes in src and put the result in dst.
765     /// In x64, this will emit multiple instructions if the `has_lzcnt` flag is
766     /// false.
767     fn clz(&mut self, src: Reg, dst: Reg, size: OperandSize);
768 
769     /// Count the number of trailing zeroes in src and put the result in dst.masm
770     /// In x64, this will emit multiple instructions if the `has_tzcnt` flag is
771     /// false.
772     fn ctz(&mut self, src: Reg, dst: Reg, size: OperandSize);
773 
774     /// Push the register to the stack, returning the stack slot metadata.
775     // NB
776     // The stack alignment should not be assumed after any call to `push`,
777     // unless explicitly aligned otherwise.  Typically, stack alignment is
778     // maintained at call sites and during the execution of
779     // epilogues.
780     fn push(&mut self, src: Reg, size: OperandSize) -> StackSlot;
781 
782     /// Finalize the assembly and return the result.
783     fn finalize(self, base: Option<SourceLoc>) -> MachBufferFinalized<Final>;
784 
785     /// Zero a particular register.
786     fn zero(&mut self, reg: Reg);
787 
788     /// Count the number of 1 bits in src and put the result in dst. In x64,
789     /// this will emit multiple instructions if the `has_popcnt` flag is false.
790     fn popcnt(&mut self, context: &mut CodeGenContext, size: OperandSize);
791 
792     /// Converts an i64 to an i32 by discarding the high 32 bits.
793     fn wrap(&mut self, src: Reg, dst: Reg);
794 
795     /// Extends an integer of a given size to a larger size.
796     fn extend(&mut self, src: Reg, dst: Reg, kind: ExtendKind);
797 
798     /// Emits one or more instructions to perform a signed truncation of a
799     /// float into an integer.
800     fn signed_truncate(
801         &mut self,
802         src: Reg,
803         dst: Reg,
804         src_size: OperandSize,
805         dst_size: OperandSize,
806         kind: TruncKind,
807     );
808 
809     /// Emits one or more instructions to perform an unsigned truncation of a
810     /// float into an integer.
811     fn unsigned_truncate(
812         &mut self,
813         src: Reg,
814         dst: Reg,
815         tmp_fpr: Reg,
816         src_size: OperandSize,
817         dst_size: OperandSize,
818         kind: TruncKind,
819     );
820 
821     /// Emits one or more instructions to perform a signed convert of an
822     /// integer into a float.
823     fn signed_convert(&mut self, src: Reg, dst: Reg, src_size: OperandSize, dst_size: OperandSize);
824 
825     /// Emits one or more instructions to perform an unsigned convert of an
826     /// integer into a float.
827     fn unsigned_convert(
828         &mut self,
829         src: Reg,
830         dst: Reg,
831         tmp_gpr: Reg,
832         src_size: OperandSize,
833         dst_size: OperandSize,
834     );
835 
836     /// Reinterpret a float as an integer.
837     fn reinterpret_float_as_int(&mut self, src: Reg, dst: Reg, size: OperandSize);
838 
839     /// Reinterpret an integer as a float.
840     fn reinterpret_int_as_float(&mut self, src: Reg, dst: Reg, size: OperandSize);
841 
842     /// Demote an f64 to an f32.
843     fn demote(&mut self, src: Reg, dst: Reg);
844 
845     /// Promote an f32 to an f64.
846     fn promote(&mut self, src: Reg, dst: Reg);
847 
848     /// Zero a given memory range.
849     ///
850     /// The default implementation divides the given memory range
851     /// into word-sized slots. Then it unrolls a series of store
852     /// instructions, effectively assigning zero to each slot.
853     fn zero_mem_range(&mut self, mem: &Range<u32>) {
854         let word_size = <Self::ABI as abi::ABI>::word_bytes() as u32;
855         if mem.is_empty() {
856             return;
857         }
858 
859         let start = if mem.start % word_size == 0 {
860             mem.start
861         } else {
862             // Ensure that the start of the range is at least 4-byte aligned.
863             assert!(mem.start % 4 == 0);
864             let start = align_to(mem.start, word_size);
865             let addr: Self::Address = self.local_address(&LocalSlot::i32(start));
866             self.store(RegImm::i32(0), addr, OperandSize::S32);
867             // Ensure that the new start of the range, is word-size aligned.
868             assert!(start % word_size == 0);
869             start
870         };
871 
872         let end = align_to(mem.end, word_size);
873         let slots = (end - start) / word_size;
874 
875         if slots == 1 {
876             let slot = LocalSlot::i64(start + word_size);
877             let addr: Self::Address = self.local_address(&slot);
878             self.store(RegImm::i64(0), addr, OperandSize::S64);
879         } else {
880             // TODO
881             // Add an upper bound to this generation;
882             // given a considerably large amount of slots
883             // this will be inefficient.
884             let zero = scratch!(Self);
885             self.zero(zero);
886             let zero = RegImm::reg(zero);
887 
888             for step in (start..end).into_iter().step_by(word_size as usize) {
889                 let slot = LocalSlot::i64(step + word_size);
890                 let addr: Self::Address = self.local_address(&slot);
891                 self.store(zero, addr, OperandSize::S64);
892             }
893         }
894     }
895 
896     /// Generate a label.
897     fn get_label(&mut self) -> MachLabel;
898 
899     /// Bind the given label at the current code offset.
900     fn bind(&mut self, label: MachLabel);
901 
902     /// Conditional branch.
903     ///
904     /// Performs a comparison between the two operands,
905     /// and immediately after emits a jump to the given
906     /// label destination if the condition is met.
907     fn branch(
908         &mut self,
909         kind: IntCmpKind,
910         lhs: Reg,
911         rhs: RegImm,
912         taken: MachLabel,
913         size: OperandSize,
914     );
915 
916     /// Emits and unconditional jump to the given label.
917     fn jmp(&mut self, target: MachLabel);
918 
919     /// Emits a jump table sequence. The default label is specified as
920     /// the last element of the targets slice.
921     fn jmp_table(&mut self, targets: &[MachLabel], index: Reg, tmp: Reg);
922 
923     /// Emit an unreachable code trap.
924     fn unreachable(&mut self);
925 
926     /// Emit an unconditional trap.
927     fn trap(&mut self, code: TrapCode);
928 
929     /// Traps if the condition code is met.
930     fn trapif(&mut self, cc: IntCmpKind, code: TrapCode);
931 
932     /// Trap if the source register is zero.
933     fn trapz(&mut self, src: Reg, code: TrapCode);
934 
935     /// Ensures that the stack pointer is correctly positioned before an unconditional
936     /// jump according to the requirements of the destination target.
937     fn ensure_sp_for_jump(&mut self, target: SPOffset) {
938         let bytes = self
939             .sp_offset()
940             .as_u32()
941             .checked_sub(target.as_u32())
942             .unwrap_or(0);
943         if bytes > 0 {
944             self.free_stack(bytes);
945         }
946     }
947 
948     /// Mark the start of a source location returning the machine code offset
949     /// and the relative source code location.
950     fn start_source_loc(&mut self, loc: RelSourceLoc) -> (CodeOffset, RelSourceLoc);
951 
952     /// Mark the end of a source location.
953     fn end_source_loc(&mut self);
954 
955     /// The current offset, in bytes from the beginning of the function.
956     fn current_code_offset(&self) -> CodeOffset;
957 }
958