1 use crate::abi::{self, align_to, scratch, LocalSlot}; 2 use crate::codegen::{CodeGenContext, FuncEnv}; 3 use crate::isa::reg::Reg; 4 use cranelift_codegen::{ 5 binemit::CodeOffset, 6 ir::{Endianness, LibCall, MemFlags, RelSourceLoc, SourceLoc, UserExternalNameRef}, 7 Final, MachBufferFinalized, MachLabel, 8 }; 9 use std::{fmt::Debug, ops::Range}; 10 use wasmtime_environ::PtrSize; 11 12 pub(crate) use cranelift_codegen::ir::TrapCode; 13 14 #[derive(Eq, PartialEq)] 15 pub(crate) enum DivKind { 16 /// Signed division. 17 Signed, 18 /// Unsigned division. 19 Unsigned, 20 } 21 22 /// Remainder kind. 23 pub(crate) enum RemKind { 24 /// Signed remainder. 25 Signed, 26 /// Unsigned remainder. 27 Unsigned, 28 } 29 30 /// The direction to perform the memory move. 31 #[derive(Debug, Clone, Eq, PartialEq)] 32 pub(crate) enum MemMoveDirection { 33 /// From high memory addresses to low memory addresses. 34 /// Invariant: the source location is closer to the FP than the destination 35 /// location, which will be closer to the SP. 36 HighToLow, 37 /// From low memory addresses to high memory addresses. 38 /// Invariant: the source location is closer to the SP than the destination 39 /// location, which will be closer to the FP. 40 LowToHigh, 41 } 42 43 /// Classifies how to treat float-to-int conversions. 44 #[derive(Debug, Copy, Clone, Eq, PartialEq)] 45 pub(crate) enum TruncKind { 46 /// Saturating conversion. If the source value is greater than the maximum 47 /// value of the destination type, the result is clamped to the 48 /// destination maximum value. 49 Checked, 50 /// An exception is raised if the source value is greater than the maximum 51 /// value of the destination type. 52 Unchecked, 53 } 54 55 impl TruncKind { 56 /// Returns true if the truncation kind is checked. 57 pub(crate) fn is_checked(&self) -> bool { 58 *self == TruncKind::Checked 59 } 60 } 61 62 /// Representation of the stack pointer offset. 63 #[derive(Copy, Clone, Eq, PartialEq, Debug, PartialOrd, Ord, Default)] 64 pub struct SPOffset(u32); 65 66 impl SPOffset { 67 pub fn from_u32(offs: u32) -> Self { 68 Self(offs) 69 } 70 71 pub fn as_u32(&self) -> u32 { 72 self.0 73 } 74 } 75 76 /// A stack slot. 77 #[derive(Debug, Clone, Copy, Eq, PartialEq)] 78 pub struct StackSlot { 79 /// The location of the slot, relative to the stack pointer. 80 pub offset: SPOffset, 81 /// The size of the slot, in bytes. 82 pub size: u32, 83 } 84 85 impl StackSlot { 86 pub fn new(offs: SPOffset, size: u32) -> Self { 87 Self { offset: offs, size } 88 } 89 } 90 91 /// Kinds of integer binary comparison in WebAssembly. The [`MacroAssembler`] 92 /// implementation for each ISA is responsible for emitting the correct 93 /// sequence of instructions when lowering to machine code. 94 #[derive(Debug, Clone, Copy, Eq, PartialEq)] 95 pub(crate) enum IntCmpKind { 96 /// Equal. 97 Eq, 98 /// Not equal. 99 Ne, 100 /// Signed less than. 101 LtS, 102 /// Unsigned less than. 103 LtU, 104 /// Signed greater than. 105 GtS, 106 /// Unsigned greater than. 107 GtU, 108 /// Signed less than or equal. 109 LeS, 110 /// Unsigned less than or equal. 111 LeU, 112 /// Signed greater than or equal. 113 GeS, 114 /// Unsigned greater than or equal. 115 GeU, 116 } 117 118 /// Kinds of float binary comparison in WebAssembly. The [`MacroAssembler`] 119 /// implementation for each ISA is responsible for emitting the correct 120 /// sequence of instructions when lowering code. 121 #[derive(Debug)] 122 pub(crate) enum FloatCmpKind { 123 /// Equal. 124 Eq, 125 /// Not equal. 126 Ne, 127 /// Less than. 128 Lt, 129 /// Greater than. 130 Gt, 131 /// Less than or equal. 132 Le, 133 /// Greater than or equal. 134 Ge, 135 } 136 137 /// Kinds of shifts in WebAssembly.The [`masm`] implementation for each ISA is 138 /// responsible for emitting the correct sequence of instructions when 139 /// lowering to machine code. 140 #[derive(Debug, Clone, Copy, Eq, PartialEq)] 141 pub(crate) enum ShiftKind { 142 /// Left shift. 143 Shl, 144 /// Signed right shift. 145 ShrS, 146 /// Unsigned right shift. 147 ShrU, 148 /// Left rotate. 149 Rotl, 150 /// Right rotate. 151 Rotr, 152 } 153 154 /// Kinds of extends in WebAssembly. Each MacroAssembler implementation 155 /// is responsible for emitting the correct sequence of instructions when 156 /// lowering to machine code. 157 pub(crate) enum ExtendKind { 158 /// Sign extends i32 to i64. 159 I64ExtendI32S, 160 /// Zero extends i32 to i64. 161 I64ExtendI32U, 162 // Sign extends the 8 least significant bits to 32 bits. 163 I32Extend8S, 164 // Sign extends the 16 least significant bits to 32 bits. 165 I32Extend16S, 166 /// Sign extends the 8 least significant bits to 64 bits. 167 I64Extend8S, 168 /// Sign extends the 16 least significant bits to 64 bits. 169 I64Extend16S, 170 /// Sign extends the 32 least significant bits to 64 bits. 171 I64Extend32S, 172 } 173 174 /// Operand size, in bits. 175 #[derive(Copy, Debug, Clone, Eq, PartialEq)] 176 pub(crate) enum OperandSize { 177 /// 8 bits. 178 S8, 179 /// 16 bits. 180 S16, 181 /// 32 bits. 182 S32, 183 /// 64 bits. 184 S64, 185 /// 128 bits. 186 S128, 187 } 188 189 impl OperandSize { 190 /// The number of bits in the operand. 191 pub fn num_bits(&self) -> u8 { 192 match self { 193 OperandSize::S8 => 8, 194 OperandSize::S16 => 16, 195 OperandSize::S32 => 32, 196 OperandSize::S64 => 64, 197 OperandSize::S128 => 128, 198 } 199 } 200 201 /// The number of bytes in the operand. 202 pub fn bytes(&self) -> u32 { 203 match self { 204 Self::S8 => 1, 205 Self::S16 => 2, 206 Self::S32 => 4, 207 Self::S64 => 8, 208 Self::S128 => 16, 209 } 210 } 211 212 /// The binary logarithm of the number of bits in the operand. 213 pub fn log2(&self) -> u8 { 214 match self { 215 OperandSize::S8 => 3, 216 OperandSize::S16 => 4, 217 OperandSize::S32 => 5, 218 OperandSize::S64 => 6, 219 OperandSize::S128 => 7, 220 } 221 } 222 223 /// Create an [`OperandSize`] from the given number of bytes. 224 pub fn from_bytes(bytes: u8) -> Self { 225 use OperandSize::*; 226 match bytes { 227 4 => S32, 228 8 => S64, 229 16 => S128, 230 _ => panic!("Invalid bytes {bytes} for OperandSize"), 231 } 232 } 233 } 234 235 /// An abstraction over a register or immediate. 236 #[derive(Copy, Clone, Debug, PartialEq, Eq)] 237 pub(crate) enum RegImm { 238 /// A register. 239 Reg(Reg), 240 /// A tagged immediate argument. 241 Imm(Imm), 242 } 243 244 /// An tagged representation of an immediate. 245 #[derive(Copy, Clone, Debug, PartialEq, Eq)] 246 pub(crate) enum Imm { 247 /// I32 immediate. 248 I32(u32), 249 /// I64 immediate. 250 I64(u64), 251 /// F32 immediate. 252 F32(u32), 253 /// F64 immediate. 254 F64(u64), 255 /// V128 immediate. 256 V128(i128), 257 } 258 259 impl Imm { 260 /// Create a new I64 immediate. 261 pub fn i64(val: i64) -> Self { 262 Self::I64(val as u64) 263 } 264 265 /// Create a new I32 immediate. 266 pub fn i32(val: i32) -> Self { 267 Self::I32(val as u32) 268 } 269 270 /// Create a new F32 immediate. 271 pub fn f32(bits: u32) -> Self { 272 Self::F32(bits) 273 } 274 275 /// Create a new F64 immediate. 276 pub fn f64(bits: u64) -> Self { 277 Self::F64(bits) 278 } 279 280 /// Create a new V128 immediate. 281 pub fn v128(bits: i128) -> Self { 282 Self::V128(bits) 283 } 284 285 /// Convert the immediate to i32, if possible. 286 pub fn to_i32(&self) -> Option<i32> { 287 match self { 288 Self::I32(v) => Some(*v as i32), 289 Self::I64(v) => i32::try_from(*v as i64).ok(), 290 _ => None, 291 } 292 } 293 } 294 295 /// The location of the [VMcontext] used for function calls. 296 #[derive(Copy, Clone, Debug, Eq, PartialEq)] 297 pub(crate) enum VMContextLoc { 298 /// Dynamic, stored in the given register. 299 Reg(Reg), 300 /// The pinned [VMContext] register. 301 Pinned, 302 } 303 304 /// The maximum number of context arguments currently used across the compiler. 305 pub(crate) const MAX_CONTEXT_ARGS: usize = 2; 306 307 /// Out-of-band special purpose arguments used for function call emission. 308 /// 309 /// We cannot rely on the value stack for these values given that inserting 310 /// register or memory values at arbitrary locations of the value stack has the 311 /// potential to break the stack ordering principle, which states that older 312 /// values must always precede newer values, effectively simulating the order of 313 /// values in the machine stack. 314 /// The [ContextArgs] are meant to be resolved at every callsite; in some cases 315 /// it might be possible to construct it early on, but given that it might 316 /// contain allocatable registers, it's preferred to construct it in 317 /// [FnCall::emit]. 318 #[derive(Clone, Debug)] 319 pub(crate) enum ContextArgs { 320 /// No context arguments required. This is used for libcalls that don't 321 /// require any special context arguments. For example builtin functions 322 /// that perform float calculations. 323 None, 324 /// A single context argument is required; the current pinned [VMcontext] 325 /// register must be passed as the first argument of the function call. 326 VMContext([VMContextLoc; 1]), 327 /// The callee and caller context arguments are required. In this case, the 328 /// callee context argument is usually stored into an allocatable register 329 /// and the caller is always the current pinned [VMContext] pointer. 330 CalleeAndCallerVMContext([VMContextLoc; MAX_CONTEXT_ARGS]), 331 } 332 333 impl ContextArgs { 334 /// Construct an empty [ContextArgs]. 335 pub fn none() -> Self { 336 Self::None 337 } 338 339 /// Construct a [ContextArgs] declaring the usage of the pinned [VMContext] 340 /// register as both the caller and callee context arguments. 341 pub fn pinned_callee_and_caller_vmctx() -> Self { 342 Self::CalleeAndCallerVMContext([VMContextLoc::Pinned, VMContextLoc::Pinned]) 343 } 344 345 /// Construct a [ContextArgs] that declares the usage of the pinned 346 /// [VMContext] register as the only context argument. 347 pub fn pinned_vmctx() -> Self { 348 Self::VMContext([VMContextLoc::Pinned]) 349 } 350 351 /// Construct a [ContextArgs] that declares a dynamic callee context and the 352 /// pinned [VMContext] register as the context arguments. 353 pub fn with_callee_and_pinned_caller(callee_vmctx: Reg) -> Self { 354 Self::CalleeAndCallerVMContext([VMContextLoc::Reg(callee_vmctx), VMContextLoc::Pinned]) 355 } 356 357 /// Get the length of the [ContextArgs]. 358 pub fn len(&self) -> usize { 359 self.as_slice().len() 360 } 361 362 /// Get a slice of the context arguments. 363 pub fn as_slice(&self) -> &[VMContextLoc] { 364 match self { 365 Self::None => &[], 366 Self::VMContext(a) => a.as_slice(), 367 Self::CalleeAndCallerVMContext(a) => a.as_slice(), 368 } 369 } 370 } 371 372 #[derive(Copy, Clone, Debug)] 373 pub(crate) enum CalleeKind { 374 /// A function call to a raw address. 375 Indirect(Reg), 376 /// A function call to a local function. 377 Direct(UserExternalNameRef), 378 /// Call to a well known LibCall. 379 LibCall(LibCall), 380 } 381 382 impl CalleeKind { 383 /// Creates a callee kind from a register. 384 pub fn indirect(reg: Reg) -> Self { 385 Self::Indirect(reg) 386 } 387 388 /// Creates a direct callee kind from a function name. 389 pub fn direct(name: UserExternalNameRef) -> Self { 390 Self::Direct(name) 391 } 392 393 /// Creates a known callee kind from a libcall. 394 pub fn libcall(call: LibCall) -> Self { 395 Self::LibCall(call) 396 } 397 } 398 399 impl RegImm { 400 /// Register constructor. 401 pub fn reg(r: Reg) -> Self { 402 RegImm::Reg(r) 403 } 404 405 /// I64 immediate constructor. 406 pub fn i64(val: i64) -> Self { 407 RegImm::Imm(Imm::i64(val)) 408 } 409 410 /// I32 immediate constructor. 411 pub fn i32(val: i32) -> Self { 412 RegImm::Imm(Imm::i32(val)) 413 } 414 415 /// F32 immediate, stored using its bits representation. 416 // Temporary until support for f32.const is added. 417 #[allow(dead_code)] 418 pub fn f32(bits: u32) -> Self { 419 RegImm::Imm(Imm::f32(bits)) 420 } 421 422 /// F64 immediate, stored using its bits representation. 423 // Temporary until support for f64.const is added. 424 #[allow(dead_code)] 425 pub fn f64(bits: u64) -> Self { 426 RegImm::Imm(Imm::f64(bits)) 427 } 428 429 /// V128 immediate. 430 pub fn v128(bits: i128) -> Self { 431 RegImm::Imm(Imm::v128(bits)) 432 } 433 } 434 435 impl From<Reg> for RegImm { 436 fn from(r: Reg) -> Self { 437 Self::Reg(r) 438 } 439 } 440 441 #[derive(Debug)] 442 pub enum RoundingMode { 443 Nearest, 444 Up, 445 Down, 446 Zero, 447 } 448 449 /// Memory flags for trusted loads/stores. 450 pub const TRUSTED_FLAGS: MemFlags = MemFlags::trusted(); 451 452 /// Flags used for WebAssembly loads / stores. 453 /// Untrusted by default so we don't set `no_trap`. 454 /// We also ensure that the endianness is the right one for WebAssembly. 455 pub const UNTRUSTED_FLAGS: MemFlags = MemFlags::new().with_endianness(Endianness::Little); 456 457 /// Generic MacroAssembler interface used by the code generation. 458 /// 459 /// The MacroAssembler trait aims to expose an interface, high-level enough, 460 /// so that each ISA can provide its own lowering to machine code. For example, 461 /// for WebAssembly operators that don't have a direct mapping to a machine 462 /// a instruction, the interface defines a signature matching the WebAssembly 463 /// operator, allowing each implementation to lower such operator entirely. 464 /// This approach attributes more responsibility to the MacroAssembler, but frees 465 /// the caller from concerning about assembling the right sequence of 466 /// instructions at the operator callsite. 467 /// 468 /// The interface defaults to a three-argument form for binary operations; 469 /// this allows a natural mapping to instructions for RISC architectures, 470 /// that use three-argument form. 471 /// This approach allows for a more general interface that can be restricted 472 /// where needed, in the case of architectures that use a two-argument form. 473 474 pub(crate) trait MacroAssembler { 475 /// The addressing mode. 476 type Address: Copy + Debug; 477 478 /// The pointer representation of the target ISA, 479 /// used to access information from [`VMOffsets`]. 480 type Ptr: PtrSize; 481 482 /// The ABI details of the target. 483 type ABI: abi::ABI; 484 485 /// Emit the function prologue. 486 fn prologue(&mut self, vmctx: Reg) { 487 self.frame_setup(); 488 self.check_stack(vmctx); 489 } 490 491 /// Generate the frame setup sequence. 492 fn frame_setup(&mut self); 493 494 /// Generate the frame restore sequence. 495 fn frame_restore(&mut self); 496 497 /// Emit a stack check. 498 fn check_stack(&mut self, vmctx: Reg); 499 500 /// Emit the function epilogue. 501 fn epilogue(&mut self) { 502 self.frame_restore(); 503 } 504 505 /// Reserve stack space. 506 fn reserve_stack(&mut self, bytes: u32); 507 508 /// Free stack space. 509 fn free_stack(&mut self, bytes: u32); 510 511 /// Reset the stack pointer to the given offset; 512 /// 513 /// Used to reset the stack pointer to a given offset 514 /// when dealing with unreachable code. 515 fn reset_stack_pointer(&mut self, offset: SPOffset); 516 517 /// Get the address of a local slot. 518 fn local_address(&mut self, local: &LocalSlot) -> Self::Address; 519 520 /// Constructs an address with an offset that is relative to the 521 /// current position of the stack pointer (e.g. [sp + (sp_offset - 522 /// offset)]. 523 fn address_from_sp(&self, offset: SPOffset) -> Self::Address; 524 525 /// Constructs an address with an offset that is absolute to the 526 /// current position of the stack pointer (e.g. [sp + offset]. 527 fn address_at_sp(&self, offset: SPOffset) -> Self::Address; 528 529 /// Alias for [`Self::address_at_reg`] using the VMContext register as 530 /// a base. The VMContext register is derived from the ABI type that is 531 /// associated to the MacroAssembler. 532 fn address_at_vmctx(&self, offset: u32) -> Self::Address; 533 534 /// Construct an address that is absolute to the current position 535 /// of the given register. 536 fn address_at_reg(&self, reg: Reg, offset: u32) -> Self::Address; 537 538 /// Emit a function call to either a local or external function. 539 fn call(&mut self, stack_args_size: u32, f: impl FnMut(&mut Self) -> CalleeKind) -> u32; 540 541 /// Get stack pointer offset. 542 fn sp_offset(&self) -> SPOffset; 543 544 /// Perform a stack store. 545 fn store(&mut self, src: RegImm, dst: Self::Address, size: OperandSize); 546 547 /// Alias for `MacroAssembler::store` with the operand size corresponding 548 /// to the pointer size of the target. 549 fn store_ptr(&mut self, src: Reg, dst: Self::Address); 550 551 /// Perform a WebAssembly store. 552 /// A WebAssembly store introduces several additional invariants compared to 553 /// [Self::store], more precisely, it can implicitly trap, in certain 554 /// circumstances, even if explicit bounds checks are elided, in that sense, 555 /// we consider this type of load as untrusted. It can also differ with 556 /// regards to the endianness depending on the target ISA. For this reason, 557 /// [Self::wasm_store], should be explicitly used when emitting WebAssembly 558 /// stores. 559 fn wasm_store(&mut self, src: Reg, dst: Self::Address, size: OperandSize); 560 561 /// Perform a zero-extended stack load. 562 fn load(&mut self, src: Self::Address, dst: Reg, size: OperandSize); 563 564 /// Perform a WebAssembly load. 565 /// A WebAssembly load introduces several additional invariants compared to 566 /// [Self::load], more precisely, it can implicitly trap, in certain 567 /// circumstances, even if explicit bounds checks are elided, in that sense, 568 /// we consider this type of load as untrusted. It can also differ with 569 /// regards to the endianness depending on the target ISA. For this reason, 570 /// [Self::wasm_load], should be explicitly used when emitting WebAssembly 571 /// loads. 572 fn wasm_load( 573 &mut self, 574 src: Self::Address, 575 dst: Reg, 576 size: OperandSize, 577 kind: Option<ExtendKind>, 578 ); 579 580 /// Alias for `MacroAssembler::load` with the operand size corresponding 581 /// to the pointer size of the target. 582 fn load_ptr(&mut self, src: Self::Address, dst: Reg); 583 584 /// Loads the effective address into destination. 585 fn load_addr(&mut self, _src: Self::Address, _dst: Reg, _size: OperandSize); 586 587 /// Pop a value from the machine stack into the given register. 588 fn pop(&mut self, dst: Reg, size: OperandSize); 589 590 /// Perform a move. 591 fn mov(&mut self, src: RegImm, dst: Reg, size: OperandSize); 592 593 /// Perform a conditional move. 594 fn cmov(&mut self, src: Reg, dst: Reg, cc: IntCmpKind, size: OperandSize); 595 596 /// Performs a memory move of bytes from src to dest. 597 /// Bytes are moved in blocks of 8 bytes, where possible. 598 fn memmove(&mut self, src: SPOffset, dst: SPOffset, bytes: u32, direction: MemMoveDirection) { 599 match direction { 600 MemMoveDirection::LowToHigh => debug_assert!(dst.as_u32() < src.as_u32()), 601 MemMoveDirection::HighToLow => debug_assert!(dst.as_u32() > src.as_u32()), 602 } 603 // At least 4 byte aligned. 604 debug_assert!(bytes % 4 == 0); 605 let mut remaining = bytes; 606 let word_bytes = <Self::ABI as abi::ABI>::word_bytes(); 607 let scratch = scratch!(Self); 608 609 let mut dst_offs = dst.as_u32() - bytes; 610 let mut src_offs = src.as_u32() - bytes; 611 612 let word_bytes = word_bytes as u32; 613 while remaining >= word_bytes { 614 remaining -= word_bytes; 615 dst_offs += word_bytes; 616 src_offs += word_bytes; 617 618 self.load_ptr(self.address_from_sp(SPOffset::from_u32(src_offs)), scratch); 619 self.store_ptr( 620 scratch.into(), 621 self.address_from_sp(SPOffset::from_u32(dst_offs)), 622 ); 623 } 624 625 if remaining > 0 { 626 let half_word = word_bytes / 2; 627 let ptr_size = OperandSize::from_bytes(half_word as u8); 628 debug_assert!(remaining == half_word); 629 dst_offs += half_word; 630 src_offs += half_word; 631 632 self.load( 633 self.address_from_sp(SPOffset::from_u32(src_offs)), 634 scratch, 635 ptr_size, 636 ); 637 self.store( 638 scratch.into(), 639 self.address_from_sp(SPOffset::from_u32(dst_offs)), 640 ptr_size, 641 ); 642 } 643 } 644 645 /// Perform add operation. 646 fn add(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize); 647 648 /// Perform a checked unsigned integer addition, emitting the provided trap 649 /// if the addition overflows. 650 fn checked_uadd(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize, trap: TrapCode); 651 652 /// Perform subtraction operation. 653 fn sub(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize); 654 655 /// Perform multiplication operation. 656 fn mul(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize); 657 658 /// Perform a floating point add operation. 659 fn float_add(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize); 660 661 /// Perform a floating point subtraction operation. 662 fn float_sub(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize); 663 664 /// Perform a floating point multiply operation. 665 fn float_mul(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize); 666 667 /// Perform a floating point divide operation. 668 fn float_div(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize); 669 670 /// Perform a floating point minimum operation. In x86, this will emit 671 /// multiple instructions. 672 fn float_min(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize); 673 674 /// Perform a floating point maximum operation. In x86, this will emit 675 /// multiple instructions. 676 fn float_max(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize); 677 678 /// Perform a floating point copysign operation. In x86, this will emit 679 /// multiple instructions. 680 fn float_copysign(&mut self, dst: Reg, lhs: Reg, rhs: Reg, size: OperandSize); 681 682 /// Perform a floating point abs operation. 683 fn float_abs(&mut self, dst: Reg, size: OperandSize); 684 685 /// Perform a floating point negation operation. 686 fn float_neg(&mut self, dst: Reg, size: OperandSize); 687 688 /// Perform a floating point floor operation. 689 fn float_round<F: FnMut(&mut FuncEnv<Self::Ptr>, &mut CodeGenContext, &mut Self)>( 690 &mut self, 691 mode: RoundingMode, 692 env: &mut FuncEnv<Self::Ptr>, 693 context: &mut CodeGenContext, 694 size: OperandSize, 695 fallback: F, 696 ); 697 698 /// Perform a floating point square root operation. 699 fn float_sqrt(&mut self, dst: Reg, src: Reg, size: OperandSize); 700 701 /// Perform logical and operation. 702 fn and(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize); 703 704 /// Perform logical or operation. 705 fn or(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize); 706 707 /// Perform logical exclusive or operation. 708 fn xor(&mut self, dst: Reg, lhs: Reg, rhs: RegImm, size: OperandSize); 709 710 /// Perform a shift operation between a register and an immediate. 711 fn shift_ir(&mut self, dst: Reg, imm: u64, lhs: Reg, kind: ShiftKind, size: OperandSize); 712 713 /// Perform a shift operation between two registers. 714 /// This case is special in that some architectures have specific expectations 715 /// regarding the location of the instruction arguments. To free the 716 /// caller from having to deal with the architecture specific constraints 717 /// we give this function access to the code generation context, allowing 718 /// each implementation to decide the lowering path. 719 fn shift(&mut self, context: &mut CodeGenContext, kind: ShiftKind, size: OperandSize); 720 721 /// Perform division operation. 722 /// Division is special in that some architectures have specific 723 /// expectations regarding the location of the instruction 724 /// arguments and regarding the location of the quotient / 725 /// remainder. To free the caller from having to deal with the 726 /// architecture specific constraints we give this function access 727 /// to the code generation context, allowing each implementation 728 /// to decide the lowering path. For cases in which division is a 729 /// unconstrained binary operation, the caller can decide to use 730 /// the `CodeGenContext::i32_binop` or `CodeGenContext::i64_binop` 731 /// functions. 732 fn div(&mut self, context: &mut CodeGenContext, kind: DivKind, size: OperandSize); 733 734 /// Calculate remainder. 735 fn rem(&mut self, context: &mut CodeGenContext, kind: RemKind, size: OperandSize); 736 737 /// Compares `src1` against `src2` for the side effect of setting processor 738 /// flags. 739 /// 740 /// Note that `src1` is the left-hand-side of the comparison and `src2` is 741 /// the right-hand-side, so if testing `a < b` then `src1 == a` and 742 /// `src2 == b` 743 fn cmp(&mut self, src1: Reg, src2: RegImm, size: OperandSize); 744 745 /// Compare src and dst and put the result in dst. 746 /// This function will potentially emit a series of instructions. 747 /// 748 /// The initial value in `dst` is the left-hand-side of the comparison and 749 /// the initial value in `src` is the right-hand-side of the comparison. 750 /// That means for `a < b` then `dst == a` and `src == b`. 751 fn cmp_with_set(&mut self, src: RegImm, dst: Reg, kind: IntCmpKind, size: OperandSize); 752 753 /// Compare floats in src1 and src2 and put the result in dst. 754 /// In x86, this will emit multiple instructions. 755 fn float_cmp_with_set( 756 &mut self, 757 src1: Reg, 758 src2: Reg, 759 dst: Reg, 760 kind: FloatCmpKind, 761 size: OperandSize, 762 ); 763 764 /// Count the number of leading zeroes in src and put the result in dst. 765 /// In x64, this will emit multiple instructions if the `has_lzcnt` flag is 766 /// false. 767 fn clz(&mut self, src: Reg, dst: Reg, size: OperandSize); 768 769 /// Count the number of trailing zeroes in src and put the result in dst.masm 770 /// In x64, this will emit multiple instructions if the `has_tzcnt` flag is 771 /// false. 772 fn ctz(&mut self, src: Reg, dst: Reg, size: OperandSize); 773 774 /// Push the register to the stack, returning the stack slot metadata. 775 // NB 776 // The stack alignment should not be assumed after any call to `push`, 777 // unless explicitly aligned otherwise. Typically, stack alignment is 778 // maintained at call sites and during the execution of 779 // epilogues. 780 fn push(&mut self, src: Reg, size: OperandSize) -> StackSlot; 781 782 /// Finalize the assembly and return the result. 783 fn finalize(self, base: Option<SourceLoc>) -> MachBufferFinalized<Final>; 784 785 /// Zero a particular register. 786 fn zero(&mut self, reg: Reg); 787 788 /// Count the number of 1 bits in src and put the result in dst. In x64, 789 /// this will emit multiple instructions if the `has_popcnt` flag is false. 790 fn popcnt(&mut self, context: &mut CodeGenContext, size: OperandSize); 791 792 /// Converts an i64 to an i32 by discarding the high 32 bits. 793 fn wrap(&mut self, src: Reg, dst: Reg); 794 795 /// Extends an integer of a given size to a larger size. 796 fn extend(&mut self, src: Reg, dst: Reg, kind: ExtendKind); 797 798 /// Emits one or more instructions to perform a signed truncation of a 799 /// float into an integer. 800 fn signed_truncate( 801 &mut self, 802 src: Reg, 803 dst: Reg, 804 src_size: OperandSize, 805 dst_size: OperandSize, 806 kind: TruncKind, 807 ); 808 809 /// Emits one or more instructions to perform an unsigned truncation of a 810 /// float into an integer. 811 fn unsigned_truncate( 812 &mut self, 813 src: Reg, 814 dst: Reg, 815 tmp_fpr: Reg, 816 src_size: OperandSize, 817 dst_size: OperandSize, 818 kind: TruncKind, 819 ); 820 821 /// Emits one or more instructions to perform a signed convert of an 822 /// integer into a float. 823 fn signed_convert(&mut self, src: Reg, dst: Reg, src_size: OperandSize, dst_size: OperandSize); 824 825 /// Emits one or more instructions to perform an unsigned convert of an 826 /// integer into a float. 827 fn unsigned_convert( 828 &mut self, 829 src: Reg, 830 dst: Reg, 831 tmp_gpr: Reg, 832 src_size: OperandSize, 833 dst_size: OperandSize, 834 ); 835 836 /// Reinterpret a float as an integer. 837 fn reinterpret_float_as_int(&mut self, src: Reg, dst: Reg, size: OperandSize); 838 839 /// Reinterpret an integer as a float. 840 fn reinterpret_int_as_float(&mut self, src: Reg, dst: Reg, size: OperandSize); 841 842 /// Demote an f64 to an f32. 843 fn demote(&mut self, src: Reg, dst: Reg); 844 845 /// Promote an f32 to an f64. 846 fn promote(&mut self, src: Reg, dst: Reg); 847 848 /// Zero a given memory range. 849 /// 850 /// The default implementation divides the given memory range 851 /// into word-sized slots. Then it unrolls a series of store 852 /// instructions, effectively assigning zero to each slot. 853 fn zero_mem_range(&mut self, mem: &Range<u32>) { 854 let word_size = <Self::ABI as abi::ABI>::word_bytes() as u32; 855 if mem.is_empty() { 856 return; 857 } 858 859 let start = if mem.start % word_size == 0 { 860 mem.start 861 } else { 862 // Ensure that the start of the range is at least 4-byte aligned. 863 assert!(mem.start % 4 == 0); 864 let start = align_to(mem.start, word_size); 865 let addr: Self::Address = self.local_address(&LocalSlot::i32(start)); 866 self.store(RegImm::i32(0), addr, OperandSize::S32); 867 // Ensure that the new start of the range, is word-size aligned. 868 assert!(start % word_size == 0); 869 start 870 }; 871 872 let end = align_to(mem.end, word_size); 873 let slots = (end - start) / word_size; 874 875 if slots == 1 { 876 let slot = LocalSlot::i64(start + word_size); 877 let addr: Self::Address = self.local_address(&slot); 878 self.store(RegImm::i64(0), addr, OperandSize::S64); 879 } else { 880 // TODO 881 // Add an upper bound to this generation; 882 // given a considerably large amount of slots 883 // this will be inefficient. 884 let zero = scratch!(Self); 885 self.zero(zero); 886 let zero = RegImm::reg(zero); 887 888 for step in (start..end).into_iter().step_by(word_size as usize) { 889 let slot = LocalSlot::i64(step + word_size); 890 let addr: Self::Address = self.local_address(&slot); 891 self.store(zero, addr, OperandSize::S64); 892 } 893 } 894 } 895 896 /// Generate a label. 897 fn get_label(&mut self) -> MachLabel; 898 899 /// Bind the given label at the current code offset. 900 fn bind(&mut self, label: MachLabel); 901 902 /// Conditional branch. 903 /// 904 /// Performs a comparison between the two operands, 905 /// and immediately after emits a jump to the given 906 /// label destination if the condition is met. 907 fn branch( 908 &mut self, 909 kind: IntCmpKind, 910 lhs: Reg, 911 rhs: RegImm, 912 taken: MachLabel, 913 size: OperandSize, 914 ); 915 916 /// Emits and unconditional jump to the given label. 917 fn jmp(&mut self, target: MachLabel); 918 919 /// Emits a jump table sequence. The default label is specified as 920 /// the last element of the targets slice. 921 fn jmp_table(&mut self, targets: &[MachLabel], index: Reg, tmp: Reg); 922 923 /// Emit an unreachable code trap. 924 fn unreachable(&mut self); 925 926 /// Emit an unconditional trap. 927 fn trap(&mut self, code: TrapCode); 928 929 /// Traps if the condition code is met. 930 fn trapif(&mut self, cc: IntCmpKind, code: TrapCode); 931 932 /// Trap if the source register is zero. 933 fn trapz(&mut self, src: Reg, code: TrapCode); 934 935 /// Ensures that the stack pointer is correctly positioned before an unconditional 936 /// jump according to the requirements of the destination target. 937 fn ensure_sp_for_jump(&mut self, target: SPOffset) { 938 let bytes = self 939 .sp_offset() 940 .as_u32() 941 .checked_sub(target.as_u32()) 942 .unwrap_or(0); 943 if bytes > 0 { 944 self.free_stack(bytes); 945 } 946 } 947 948 /// Mark the start of a source location returning the machine code offset 949 /// and the relative source code location. 950 fn start_source_loc(&mut self, loc: RelSourceLoc) -> (CodeOffset, RelSourceLoc); 951 952 /// Mark the end of a source location. 953 fn end_source_loc(&mut self); 954 955 /// The current offset, in bytes from the beginning of the function. 956 fn current_code_offset(&self) -> CodeOffset; 957 } 958