1 use crate::abi::{self, align_to, scratch, LocalSlot}; 2 use crate::codegen::{CodeGenContext, Emission, FuncEnv}; 3 use crate::isa::{ 4 reg::{writable, Reg, WritableReg}, 5 CallingConvention, 6 }; 7 use anyhow::Result; 8 use cranelift_codegen::{ 9 binemit::CodeOffset, 10 ir::{Endianness, LibCall, MemFlags, RelSourceLoc, SourceLoc, UserExternalNameRef}, 11 Final, MachBufferFinalized, MachLabel, 12 }; 13 use std::{fmt::Debug, ops::Range}; 14 use wasmtime_environ::PtrSize; 15 16 pub(crate) use cranelift_codegen::ir::TrapCode; 17 18 #[derive(Eq, PartialEq)] 19 pub(crate) enum DivKind { 20 /// Signed division. 21 Signed, 22 /// Unsigned division. 23 Unsigned, 24 } 25 26 /// Remainder kind. 27 #[derive(Copy, Clone)] 28 pub(crate) enum RemKind { 29 /// Signed remainder. 30 Signed, 31 /// Unsigned remainder. 32 Unsigned, 33 } 34 35 impl RemKind { 36 pub fn is_signed(&self) -> bool { 37 matches!(self, Self::Signed) 38 } 39 } 40 41 #[derive(Copy, Clone, PartialEq, Eq)] 42 pub(crate) enum MemOpKind { 43 /// An atomic memory operation with SeqCst memory ordering. 44 Atomic, 45 /// A memory operation with no memory ordering constraint. 46 Normal, 47 } 48 49 #[derive(Eq, PartialEq)] 50 pub(crate) enum MulWideKind { 51 Signed, 52 Unsigned, 53 } 54 55 /// Type of operation for a read-modify-write instruction. 56 pub(crate) enum RmwOp { 57 Add, 58 Sub, 59 } 60 61 /// The direction to perform the memory move. 62 #[derive(Debug, Clone, Eq, PartialEq)] 63 pub(crate) enum MemMoveDirection { 64 /// From high memory addresses to low memory addresses. 65 /// Invariant: the source location is closer to the FP than the destination 66 /// location, which will be closer to the SP. 67 HighToLow, 68 /// From low memory addresses to high memory addresses. 69 /// Invariant: the source location is closer to the SP than the destination 70 /// location, which will be closer to the FP. 71 LowToHigh, 72 } 73 74 /// Classifies how to treat float-to-int conversions. 75 #[derive(Debug, Copy, Clone, Eq, PartialEq)] 76 pub(crate) enum TruncKind { 77 /// Saturating conversion. If the source value is greater than the maximum 78 /// value of the destination type, the result is clamped to the 79 /// destination maximum value. 80 Checked, 81 /// An exception is raised if the source value is greater than the maximum 82 /// value of the destination type. 83 Unchecked, 84 } 85 86 impl TruncKind { 87 /// Returns true if the truncation kind is checked. 88 pub(crate) fn is_checked(&self) -> bool { 89 *self == TruncKind::Checked 90 } 91 92 /// Returns `true` if the trunc kind is [`Unchecked`]. 93 /// 94 /// [`Unchecked`]: TruncKind::Unchecked 95 #[must_use] 96 pub(crate) fn is_unchecked(&self) -> bool { 97 matches!(self, Self::Unchecked) 98 } 99 } 100 101 /// Representation of the stack pointer offset. 102 #[derive(Copy, Clone, Eq, PartialEq, Debug, PartialOrd, Ord, Default)] 103 pub struct SPOffset(u32); 104 105 impl SPOffset { 106 pub fn from_u32(offs: u32) -> Self { 107 Self(offs) 108 } 109 110 pub fn as_u32(&self) -> u32 { 111 self.0 112 } 113 } 114 115 /// A stack slot. 116 #[derive(Debug, Clone, Copy, Eq, PartialEq)] 117 pub struct StackSlot { 118 /// The location of the slot, relative to the stack pointer. 119 pub offset: SPOffset, 120 /// The size of the slot, in bytes. 121 pub size: u32, 122 } 123 124 impl StackSlot { 125 pub fn new(offs: SPOffset, size: u32) -> Self { 126 Self { offset: offs, size } 127 } 128 } 129 130 /// Kinds of integer binary comparison in WebAssembly. The [`MacroAssembler`] 131 /// implementation for each ISA is responsible for emitting the correct 132 /// sequence of instructions when lowering to machine code. 133 #[derive(Debug, Clone, Copy, Eq, PartialEq)] 134 pub(crate) enum IntCmpKind { 135 /// Equal. 136 Eq, 137 /// Not equal. 138 Ne, 139 /// Signed less than. 140 LtS, 141 /// Unsigned less than. 142 LtU, 143 /// Signed greater than. 144 GtS, 145 /// Unsigned greater than. 146 GtU, 147 /// Signed less than or equal. 148 LeS, 149 /// Unsigned less than or equal. 150 LeU, 151 /// Signed greater than or equal. 152 GeS, 153 /// Unsigned greater than or equal. 154 GeU, 155 } 156 157 /// Kinds of float binary comparison in WebAssembly. The [`MacroAssembler`] 158 /// implementation for each ISA is responsible for emitting the correct 159 /// sequence of instructions when lowering code. 160 #[derive(Debug)] 161 pub(crate) enum FloatCmpKind { 162 /// Equal. 163 Eq, 164 /// Not equal. 165 Ne, 166 /// Less than. 167 Lt, 168 /// Greater than. 169 Gt, 170 /// Less than or equal. 171 Le, 172 /// Greater than or equal. 173 Ge, 174 } 175 176 /// Kinds of shifts in WebAssembly.The [`masm`] implementation for each ISA is 177 /// responsible for emitting the correct sequence of instructions when 178 /// lowering to machine code. 179 #[derive(Debug, Clone, Copy, Eq, PartialEq)] 180 pub(crate) enum ShiftKind { 181 /// Left shift. 182 Shl, 183 /// Signed right shift. 184 ShrS, 185 /// Unsigned right shift. 186 ShrU, 187 /// Left rotate. 188 Rotl, 189 /// Right rotate. 190 Rotr, 191 } 192 193 /// Kinds of extends in WebAssembly. Each MacroAssembler implementation 194 /// is responsible for emitting the correct sequence of instructions when 195 /// lowering to machine code. 196 #[derive(Copy, Clone)] 197 pub(crate) enum ExtendKind { 198 /// 8 to 32 bit signed extend. 199 I32Extend8S, 200 /// 8 to 32 bit unsigned extend. 201 I32Extend8U, 202 203 /// 16 to 32 bit signed extend. 204 I32Extend16S, 205 /// 16 to 32 bit unsigned extend. 206 I32Extend16U, 207 208 /// 8 to 64 bit signed extend. 209 I64Extend8S, 210 /// 8 to 64 bit unsigned extend. 211 I64Extend8U, 212 213 /// 16 to 64 bit signed extend. 214 I64Extend16S, 215 /// 16 to 64 bit unsigned extend. 216 I64Extend16U, 217 218 /// 32 to 64 bit signed extend. 219 I64Extend32S, 220 /// 32 to 64 bit unsigned extend. 221 I64Extend32U, 222 } 223 224 impl ExtendKind { 225 pub fn signed(&self) -> bool { 226 match self { 227 Self::I32Extend8S 228 | Self::I32Extend16S 229 | Self::I64Extend8S 230 | Self::I64Extend16S 231 | Self::I64Extend32S => true, 232 _ => false, 233 } 234 } 235 236 pub fn from_bits(&self) -> u8 { 237 match self { 238 Self::I64Extend32S | Self::I64Extend32U => 32, 239 Self::I32Extend8S | Self::I32Extend8U | Self::I64Extend8S | Self::I64Extend8U => 8, 240 Self::I32Extend16S | Self::I64Extend16S | Self::I32Extend16U | Self::I64Extend16U => 16, 241 } 242 } 243 244 pub fn to_bits(&self) -> u8 { 245 match self { 246 Self::I64Extend32S 247 | Self::I64Extend32U 248 | Self::I64Extend8S 249 | Self::I64Extend8U 250 | Self::I64Extend16S 251 | Self::I64Extend16U => 64, 252 Self::I32Extend8S | Self::I32Extend8U | Self::I32Extend16U | Self::I32Extend16S => 32, 253 } 254 } 255 } 256 257 /// Kinds of vector extends in WebAssembly. Each MacroAssembler implementation 258 /// is responsible for emitting the correct sequence of instructions when 259 /// lowering to machine code. 260 pub(crate) enum VectorExtendKind { 261 /// Sign extends eight 8 bit integers to eight 16 bit lanes. 262 V128Extend8x8S, 263 /// Zero extends eight 8 bit integers to eight 16 bit lanes. 264 V128Extend8x8U, 265 /// Sign extends four 16 bit integers to four 32 bit lanes. 266 V128Extend16x4S, 267 /// Zero extends four 16 bit integers to four 32 bit lanes. 268 V128Extend16x4U, 269 /// Sign extends two 32 bit integers to two 64 bit lanes. 270 V128Extend32x2S, 271 /// Zero extends two 32 bit integers to two 64 bit lanes. 272 V128Extend32x2U, 273 } 274 275 /// Kinds of splat loads supported by WebAssembly. 276 pub(crate) enum SplatLoadKind { 277 /// 8 bits. 278 S8, 279 /// 16 bits. 280 S16, 281 /// 32 bits. 282 S32, 283 /// 64 bits. 284 S64, 285 } 286 287 /// Kinds of splat supported by WebAssembly. 288 #[derive(Copy, Debug, Clone, Eq, PartialEq)] 289 pub(crate) enum SplatKind { 290 /// 8 bit integer. 291 I8x16, 292 /// 16 bit integer. 293 I16x8, 294 /// 32 bit integer. 295 I32x4, 296 /// 64 bit integer. 297 I64x2, 298 /// 32 bit float. 299 F32x4, 300 /// 64 bit float. 301 F64x2, 302 } 303 304 impl SplatKind { 305 /// The lane size to use for different kinds of splats. 306 pub(crate) fn lane_size(&self) -> OperandSize { 307 match self { 308 SplatKind::I8x16 => OperandSize::S8, 309 SplatKind::I16x8 => OperandSize::S16, 310 SplatKind::I32x4 | SplatKind::F32x4 => OperandSize::S32, 311 SplatKind::I64x2 | SplatKind::F64x2 => OperandSize::S64, 312 } 313 } 314 } 315 316 /// Kinds of behavior supported by Wasm loads. 317 pub(crate) enum LoadKind { 318 /// Load the entire bytes of the operand size without any modifications. 319 Operand(OperandSize), 320 /// Duplicate value into vector lanes. 321 Splat(SplatLoadKind), 322 /// Scalar (non-vector) extend. 323 ScalarExtend(ExtendKind), 324 /// Vector extend. 325 VectorExtend(VectorExtendKind), 326 } 327 328 impl LoadKind { 329 /// Returns the [`OperandSize`] used in the load operation. 330 pub(crate) fn derive_operand_size(&self) -> OperandSize { 331 match self { 332 Self::ScalarExtend(scalar) => Self::operand_size_for_scalar(scalar), 333 Self::VectorExtend(vector) => Self::operand_size_for_vector(vector), 334 Self::Splat(kind) => Self::operand_size_for_splat(kind), 335 Self::Operand(op) => *op, 336 } 337 } 338 339 fn operand_size_for_vector(vector: &VectorExtendKind) -> OperandSize { 340 match vector { 341 VectorExtendKind::V128Extend8x8S | VectorExtendKind::V128Extend8x8U => OperandSize::S8, 342 VectorExtendKind::V128Extend16x4S | VectorExtendKind::V128Extend16x4U => { 343 OperandSize::S16 344 } 345 VectorExtendKind::V128Extend32x2S | VectorExtendKind::V128Extend32x2U => { 346 OperandSize::S32 347 } 348 } 349 } 350 351 fn operand_size_for_scalar(extend_kind: &ExtendKind) -> OperandSize { 352 match extend_kind { 353 ExtendKind::I32Extend8S 354 | ExtendKind::I32Extend8U 355 | ExtendKind::I64Extend8S 356 | ExtendKind::I64Extend8U => OperandSize::S8, 357 ExtendKind::I32Extend16S 358 | ExtendKind::I32Extend16U 359 | ExtendKind::I64Extend16U 360 | ExtendKind::I64Extend16S => OperandSize::S16, 361 ExtendKind::I64Extend32U | ExtendKind::I64Extend32S => OperandSize::S32, 362 } 363 } 364 365 fn operand_size_for_splat(kind: &SplatLoadKind) -> OperandSize { 366 match kind { 367 SplatLoadKind::S8 => OperandSize::S8, 368 SplatLoadKind::S16 => OperandSize::S16, 369 SplatLoadKind::S32 => OperandSize::S32, 370 SplatLoadKind::S64 => OperandSize::S64, 371 } 372 } 373 } 374 375 /// Operand size, in bits. 376 #[derive(Copy, Debug, Clone, Eq, PartialEq)] 377 pub(crate) enum OperandSize { 378 /// 8 bits. 379 S8, 380 /// 16 bits. 381 S16, 382 /// 32 bits. 383 S32, 384 /// 64 bits. 385 S64, 386 /// 128 bits. 387 S128, 388 } 389 390 impl OperandSize { 391 /// The number of bits in the operand. 392 pub fn num_bits(&self) -> u8 { 393 match self { 394 OperandSize::S8 => 8, 395 OperandSize::S16 => 16, 396 OperandSize::S32 => 32, 397 OperandSize::S64 => 64, 398 OperandSize::S128 => 128, 399 } 400 } 401 402 /// The number of bytes in the operand. 403 pub fn bytes(&self) -> u32 { 404 match self { 405 Self::S8 => 1, 406 Self::S16 => 2, 407 Self::S32 => 4, 408 Self::S64 => 8, 409 Self::S128 => 16, 410 } 411 } 412 413 /// The binary logarithm of the number of bits in the operand. 414 pub fn log2(&self) -> u8 { 415 match self { 416 OperandSize::S8 => 3, 417 OperandSize::S16 => 4, 418 OperandSize::S32 => 5, 419 OperandSize::S64 => 6, 420 OperandSize::S128 => 7, 421 } 422 } 423 424 /// Create an [`OperandSize`] from the given number of bytes. 425 pub fn from_bytes(bytes: u8) -> Self { 426 use OperandSize::*; 427 match bytes { 428 4 => S32, 429 8 => S64, 430 16 => S128, 431 _ => panic!("Invalid bytes {bytes} for OperandSize"), 432 } 433 } 434 } 435 436 /// An abstraction over a register or immediate. 437 #[derive(Copy, Clone, Debug, PartialEq, Eq)] 438 pub(crate) enum RegImm { 439 /// A register. 440 Reg(Reg), 441 /// A tagged immediate argument. 442 Imm(Imm), 443 } 444 445 /// An tagged representation of an immediate. 446 #[derive(Copy, Clone, Debug, PartialEq, Eq)] 447 pub(crate) enum Imm { 448 /// I32 immediate. 449 I32(u32), 450 /// I64 immediate. 451 I64(u64), 452 /// F32 immediate. 453 F32(u32), 454 /// F64 immediate. 455 F64(u64), 456 /// V128 immediate. 457 V128(i128), 458 } 459 460 impl Imm { 461 /// Create a new I64 immediate. 462 pub fn i64(val: i64) -> Self { 463 Self::I64(val as u64) 464 } 465 466 /// Create a new I32 immediate. 467 pub fn i32(val: i32) -> Self { 468 Self::I32(val as u32) 469 } 470 471 /// Create a new F32 immediate. 472 pub fn f32(bits: u32) -> Self { 473 Self::F32(bits) 474 } 475 476 /// Create a new F64 immediate. 477 pub fn f64(bits: u64) -> Self { 478 Self::F64(bits) 479 } 480 481 /// Create a new V128 immediate. 482 pub fn v128(bits: i128) -> Self { 483 Self::V128(bits) 484 } 485 486 /// Convert the immediate to i32, if possible. 487 pub fn to_i32(&self) -> Option<i32> { 488 match self { 489 Self::I32(v) => Some(*v as i32), 490 Self::I64(v) => i32::try_from(*v as i64).ok(), 491 _ => None, 492 } 493 } 494 495 /// Returns true if the [`Imm`] is float. 496 pub fn is_float(&self) -> bool { 497 match self { 498 Self::F32(_) | Self::F64(_) => true, 499 _ => false, 500 } 501 } 502 503 /// Get the operand size of the immediate. 504 pub fn size(&self) -> OperandSize { 505 match self { 506 Self::I32(_) | Self::F32(_) => OperandSize::S32, 507 Self::I64(_) | Self::F64(_) => OperandSize::S64, 508 Self::V128(_) => OperandSize::S128, 509 } 510 } 511 512 /// Get a little endian representation of the immediate. 513 /// 514 /// This method heap allocates and is intended to be used when adding 515 /// values to the constant pool. 516 pub fn to_bytes(&self) -> Vec<u8> { 517 match self { 518 Imm::I32(n) => n.to_le_bytes().to_vec(), 519 Imm::I64(n) => n.to_le_bytes().to_vec(), 520 Imm::F32(n) => n.to_le_bytes().to_vec(), 521 Imm::F64(n) => n.to_le_bytes().to_vec(), 522 Imm::V128(n) => n.to_le_bytes().to_vec(), 523 } 524 } 525 } 526 527 /// The location of the [VMcontext] used for function calls. 528 #[derive(Copy, Clone, Debug, Eq, PartialEq)] 529 pub(crate) enum VMContextLoc { 530 /// Dynamic, stored in the given register. 531 Reg(Reg), 532 /// The pinned [VMContext] register. 533 Pinned, 534 } 535 536 /// The maximum number of context arguments currently used across the compiler. 537 pub(crate) const MAX_CONTEXT_ARGS: usize = 2; 538 539 /// Out-of-band special purpose arguments used for function call emission. 540 /// 541 /// We cannot rely on the value stack for these values given that inserting 542 /// register or memory values at arbitrary locations of the value stack has the 543 /// potential to break the stack ordering principle, which states that older 544 /// values must always precede newer values, effectively simulating the order of 545 /// values in the machine stack. 546 /// The [ContextArgs] are meant to be resolved at every callsite; in some cases 547 /// it might be possible to construct it early on, but given that it might 548 /// contain allocatable registers, it's preferred to construct it in 549 /// [FnCall::emit]. 550 #[derive(Clone, Debug)] 551 pub(crate) enum ContextArgs { 552 /// No context arguments required. This is used for libcalls that don't 553 /// require any special context arguments. For example builtin functions 554 /// that perform float calculations. 555 None, 556 /// A single context argument is required; the current pinned [VMcontext] 557 /// register must be passed as the first argument of the function call. 558 VMContext([VMContextLoc; 1]), 559 /// The callee and caller context arguments are required. In this case, the 560 /// callee context argument is usually stored into an allocatable register 561 /// and the caller is always the current pinned [VMContext] pointer. 562 CalleeAndCallerVMContext([VMContextLoc; MAX_CONTEXT_ARGS]), 563 } 564 565 impl ContextArgs { 566 /// Construct an empty [ContextArgs]. 567 pub fn none() -> Self { 568 Self::None 569 } 570 571 /// Construct a [ContextArgs] declaring the usage of the pinned [VMContext] 572 /// register as both the caller and callee context arguments. 573 pub fn pinned_callee_and_caller_vmctx() -> Self { 574 Self::CalleeAndCallerVMContext([VMContextLoc::Pinned, VMContextLoc::Pinned]) 575 } 576 577 /// Construct a [ContextArgs] that declares the usage of the pinned 578 /// [VMContext] register as the only context argument. 579 pub fn pinned_vmctx() -> Self { 580 Self::VMContext([VMContextLoc::Pinned]) 581 } 582 583 /// Construct a [ContextArgs] that declares a dynamic callee context and the 584 /// pinned [VMContext] register as the context arguments. 585 pub fn with_callee_and_pinned_caller(callee_vmctx: Reg) -> Self { 586 Self::CalleeAndCallerVMContext([VMContextLoc::Reg(callee_vmctx), VMContextLoc::Pinned]) 587 } 588 589 /// Get the length of the [ContextArgs]. 590 pub fn len(&self) -> usize { 591 self.as_slice().len() 592 } 593 594 /// Get a slice of the context arguments. 595 pub fn as_slice(&self) -> &[VMContextLoc] { 596 match self { 597 Self::None => &[], 598 Self::VMContext(a) => a.as_slice(), 599 Self::CalleeAndCallerVMContext(a) => a.as_slice(), 600 } 601 } 602 } 603 604 #[derive(Copy, Clone, Debug)] 605 pub(crate) enum CalleeKind { 606 /// A function call to a raw address. 607 Indirect(Reg), 608 /// A function call to a local function. 609 Direct(UserExternalNameRef), 610 /// Call to a well known LibCall. 611 LibCall(LibCall), 612 } 613 614 impl CalleeKind { 615 /// Creates a callee kind from a register. 616 pub fn indirect(reg: Reg) -> Self { 617 Self::Indirect(reg) 618 } 619 620 /// Creates a direct callee kind from a function name. 621 pub fn direct(name: UserExternalNameRef) -> Self { 622 Self::Direct(name) 623 } 624 625 /// Creates a known callee kind from a libcall. 626 pub fn libcall(call: LibCall) -> Self { 627 Self::LibCall(call) 628 } 629 } 630 631 impl RegImm { 632 /// Register constructor. 633 pub fn reg(r: Reg) -> Self { 634 RegImm::Reg(r) 635 } 636 637 /// I64 immediate constructor. 638 pub fn i64(val: i64) -> Self { 639 RegImm::Imm(Imm::i64(val)) 640 } 641 642 /// I32 immediate constructor. 643 pub fn i32(val: i32) -> Self { 644 RegImm::Imm(Imm::i32(val)) 645 } 646 647 /// F32 immediate, stored using its bits representation. 648 pub fn f32(bits: u32) -> Self { 649 RegImm::Imm(Imm::f32(bits)) 650 } 651 652 /// F64 immediate, stored using its bits representation. 653 pub fn f64(bits: u64) -> Self { 654 RegImm::Imm(Imm::f64(bits)) 655 } 656 657 /// V128 immediate. 658 pub fn v128(bits: i128) -> Self { 659 RegImm::Imm(Imm::v128(bits)) 660 } 661 } 662 663 impl From<Reg> for RegImm { 664 fn from(r: Reg) -> Self { 665 Self::Reg(r) 666 } 667 } 668 669 #[derive(Debug)] 670 pub enum RoundingMode { 671 Nearest, 672 Up, 673 Down, 674 Zero, 675 } 676 677 /// Memory flags for trusted loads/stores. 678 pub const TRUSTED_FLAGS: MemFlags = MemFlags::trusted(); 679 680 /// Flags used for WebAssembly loads / stores. 681 /// Untrusted by default so we don't set `no_trap`. 682 /// We also ensure that the endianness is the right one for WebAssembly. 683 pub const UNTRUSTED_FLAGS: MemFlags = MemFlags::new().with_endianness(Endianness::Little); 684 685 /// Generic MacroAssembler interface used by the code generation. 686 /// 687 /// The MacroAssembler trait aims to expose an interface, high-level enough, 688 /// so that each ISA can provide its own lowering to machine code. For example, 689 /// for WebAssembly operators that don't have a direct mapping to a machine 690 /// a instruction, the interface defines a signature matching the WebAssembly 691 /// operator, allowing each implementation to lower such operator entirely. 692 /// This approach attributes more responsibility to the MacroAssembler, but frees 693 /// the caller from concerning about assembling the right sequence of 694 /// instructions at the operator callsite. 695 /// 696 /// The interface defaults to a three-argument form for binary operations; 697 /// this allows a natural mapping to instructions for RISC architectures, 698 /// that use three-argument form. 699 /// This approach allows for a more general interface that can be restricted 700 /// where needed, in the case of architectures that use a two-argument form. 701 702 pub(crate) trait MacroAssembler { 703 /// The addressing mode. 704 type Address: Copy + Debug; 705 706 /// The pointer representation of the target ISA, 707 /// used to access information from [`VMOffsets`]. 708 type Ptr: PtrSize; 709 710 /// The ABI details of the target. 711 type ABI: abi::ABI; 712 713 /// Emit the function prologue. 714 fn prologue(&mut self, vmctx: Reg) -> Result<()> { 715 self.frame_setup()?; 716 self.check_stack(vmctx) 717 } 718 719 /// Generate the frame setup sequence. 720 fn frame_setup(&mut self) -> Result<()>; 721 722 /// Generate the frame restore sequence. 723 fn frame_restore(&mut self) -> Result<()>; 724 725 /// Emit a stack check. 726 fn check_stack(&mut self, vmctx: Reg) -> Result<()>; 727 728 /// Emit the function epilogue. 729 fn epilogue(&mut self) -> Result<()> { 730 self.frame_restore() 731 } 732 733 /// Reserve stack space. 734 fn reserve_stack(&mut self, bytes: u32) -> Result<()>; 735 736 /// Free stack space. 737 fn free_stack(&mut self, bytes: u32) -> Result<()>; 738 739 /// Reset the stack pointer to the given offset; 740 /// 741 /// Used to reset the stack pointer to a given offset 742 /// when dealing with unreachable code. 743 fn reset_stack_pointer(&mut self, offset: SPOffset) -> Result<()>; 744 745 /// Get the address of a local slot. 746 fn local_address(&mut self, local: &LocalSlot) -> Result<Self::Address>; 747 748 /// Constructs an address with an offset that is relative to the 749 /// current position of the stack pointer (e.g. [sp + (sp_offset - 750 /// offset)]. 751 fn address_from_sp(&self, offset: SPOffset) -> Result<Self::Address>; 752 753 /// Constructs an address with an offset that is absolute to the 754 /// current position of the stack pointer (e.g. [sp + offset]. 755 fn address_at_sp(&self, offset: SPOffset) -> Result<Self::Address>; 756 757 /// Alias for [`Self::address_at_reg`] using the VMContext register as 758 /// a base. The VMContext register is derived from the ABI type that is 759 /// associated to the MacroAssembler. 760 fn address_at_vmctx(&self, offset: u32) -> Result<Self::Address>; 761 762 /// Construct an address that is absolute to the current position 763 /// of the given register. 764 fn address_at_reg(&self, reg: Reg, offset: u32) -> Result<Self::Address>; 765 766 /// Emit a function call to either a local or external function. 767 fn call( 768 &mut self, 769 stack_args_size: u32, 770 f: impl FnMut(&mut Self) -> Result<(CalleeKind, CallingConvention)>, 771 ) -> Result<u32>; 772 773 /// Get stack pointer offset. 774 fn sp_offset(&self) -> Result<SPOffset>; 775 776 /// Perform a stack store. 777 fn store(&mut self, src: RegImm, dst: Self::Address, size: OperandSize) -> Result<()>; 778 779 /// Alias for `MacroAssembler::store` with the operand size corresponding 780 /// to the pointer size of the target. 781 fn store_ptr(&mut self, src: Reg, dst: Self::Address) -> Result<()>; 782 783 /// Perform a WebAssembly store. 784 /// A WebAssembly store introduces several additional invariants compared to 785 /// [Self::store], more precisely, it can implicitly trap, in certain 786 /// circumstances, even if explicit bounds checks are elided, in that sense, 787 /// we consider this type of load as untrusted. It can also differ with 788 /// regards to the endianness depending on the target ISA. For this reason, 789 /// [Self::wasm_store], should be explicitly used when emitting WebAssembly 790 /// stores. 791 fn wasm_store( 792 &mut self, 793 src: Reg, 794 dst: Self::Address, 795 size: OperandSize, 796 op_kind: MemOpKind, 797 ) -> Result<()>; 798 799 /// Perform a zero-extended stack load. 800 fn load(&mut self, src: Self::Address, dst: WritableReg, size: OperandSize) -> Result<()>; 801 802 /// Perform a WebAssembly load. 803 /// A WebAssembly load introduces several additional invariants compared to 804 /// [Self::load], more precisely, it can implicitly trap, in certain 805 /// circumstances, even if explicit bounds checks are elided, in that sense, 806 /// we consider this type of load as untrusted. It can also differ with 807 /// regards to the endianness depending on the target ISA. For this reason, 808 /// [Self::wasm_load], should be explicitly used when emitting WebAssembly 809 /// loads. 810 fn wasm_load( 811 &mut self, 812 src: Self::Address, 813 dst: WritableReg, 814 kind: LoadKind, 815 op_kind: MemOpKind, 816 ) -> Result<()>; 817 818 /// Alias for `MacroAssembler::load` with the operand size corresponding 819 /// to the pointer size of the target. 820 fn load_ptr(&mut self, src: Self::Address, dst: WritableReg) -> Result<()>; 821 822 /// Loads the effective address into destination. 823 fn load_addr( 824 &mut self, 825 _src: Self::Address, 826 _dst: WritableReg, 827 _size: OperandSize, 828 ) -> Result<()>; 829 830 /// Pop a value from the machine stack into the given register. 831 fn pop(&mut self, dst: WritableReg, size: OperandSize) -> Result<()>; 832 833 /// Perform a move. 834 fn mov(&mut self, dst: WritableReg, src: RegImm, size: OperandSize) -> Result<()>; 835 836 /// Perform a conditional move. 837 fn cmov(&mut self, dst: WritableReg, src: Reg, cc: IntCmpKind, size: OperandSize) 838 -> Result<()>; 839 840 /// Performs a memory move of bytes from src to dest. 841 /// Bytes are moved in blocks of 8 bytes, where possible. 842 fn memmove( 843 &mut self, 844 src: SPOffset, 845 dst: SPOffset, 846 bytes: u32, 847 direction: MemMoveDirection, 848 ) -> Result<()> { 849 match direction { 850 MemMoveDirection::LowToHigh => debug_assert!(dst.as_u32() < src.as_u32()), 851 MemMoveDirection::HighToLow => debug_assert!(dst.as_u32() > src.as_u32()), 852 } 853 // At least 4 byte aligned. 854 debug_assert!(bytes % 4 == 0); 855 let mut remaining = bytes; 856 let word_bytes = <Self::ABI as abi::ABI>::word_bytes(); 857 let scratch = scratch!(Self); 858 859 let mut dst_offs = dst.as_u32() - bytes; 860 let mut src_offs = src.as_u32() - bytes; 861 862 let word_bytes = word_bytes as u32; 863 while remaining >= word_bytes { 864 remaining -= word_bytes; 865 dst_offs += word_bytes; 866 src_offs += word_bytes; 867 868 self.load_ptr( 869 self.address_from_sp(SPOffset::from_u32(src_offs))?, 870 writable!(scratch), 871 )?; 872 self.store_ptr( 873 scratch.into(), 874 self.address_from_sp(SPOffset::from_u32(dst_offs))?, 875 )?; 876 } 877 878 if remaining > 0 { 879 let half_word = word_bytes / 2; 880 let ptr_size = OperandSize::from_bytes(half_word as u8); 881 debug_assert!(remaining == half_word); 882 dst_offs += half_word; 883 src_offs += half_word; 884 885 self.load( 886 self.address_from_sp(SPOffset::from_u32(src_offs))?, 887 writable!(scratch), 888 ptr_size, 889 )?; 890 self.store( 891 scratch.into(), 892 self.address_from_sp(SPOffset::from_u32(dst_offs))?, 893 ptr_size, 894 )?; 895 } 896 Ok(()) 897 } 898 899 /// Perform add operation. 900 fn add(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>; 901 902 /// Perform a checked unsigned integer addition, emitting the provided trap 903 /// if the addition overflows. 904 fn checked_uadd( 905 &mut self, 906 dst: WritableReg, 907 lhs: Reg, 908 rhs: RegImm, 909 size: OperandSize, 910 trap: TrapCode, 911 ) -> Result<()>; 912 913 /// Perform subtraction operation. 914 fn sub(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>; 915 916 /// Perform multiplication operation. 917 fn mul(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>; 918 919 /// Perform a floating point add operation. 920 fn float_add(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>; 921 922 /// Perform a floating point subtraction operation. 923 fn float_sub(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>; 924 925 /// Perform a floating point multiply operation. 926 fn float_mul(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>; 927 928 /// Perform a floating point divide operation. 929 fn float_div(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>; 930 931 /// Perform a floating point minimum operation. In x86, this will emit 932 /// multiple instructions. 933 fn float_min(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>; 934 935 /// Perform a floating point maximum operation. In x86, this will emit 936 /// multiple instructions. 937 fn float_max(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) -> Result<()>; 938 939 /// Perform a floating point copysign operation. In x86, this will emit 940 /// multiple instructions. 941 fn float_copysign( 942 &mut self, 943 dst: WritableReg, 944 lhs: Reg, 945 rhs: Reg, 946 size: OperandSize, 947 ) -> Result<()>; 948 949 /// Perform a floating point abs operation. 950 fn float_abs(&mut self, dst: WritableReg, size: OperandSize) -> Result<()>; 951 952 /// Perform a floating point negation operation. 953 fn float_neg(&mut self, dst: WritableReg, size: OperandSize) -> Result<()>; 954 955 /// Perform a floating point floor operation. 956 fn float_round< 957 F: FnMut(&mut FuncEnv<Self::Ptr>, &mut CodeGenContext<Emission>, &mut Self) -> Result<()>, 958 >( 959 &mut self, 960 mode: RoundingMode, 961 env: &mut FuncEnv<Self::Ptr>, 962 context: &mut CodeGenContext<Emission>, 963 size: OperandSize, 964 fallback: F, 965 ) -> Result<()>; 966 967 /// Perform a floating point square root operation. 968 fn float_sqrt(&mut self, dst: WritableReg, src: Reg, size: OperandSize) -> Result<()>; 969 970 /// Perform logical and operation. 971 fn and(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>; 972 973 /// Perform logical or operation. 974 fn or(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>; 975 976 /// Perform logical exclusive or operation. 977 fn xor(&mut self, dst: WritableReg, lhs: Reg, rhs: RegImm, size: OperandSize) -> Result<()>; 978 979 /// Perform a shift operation between a register and an immediate. 980 fn shift_ir( 981 &mut self, 982 dst: WritableReg, 983 imm: u64, 984 lhs: Reg, 985 kind: ShiftKind, 986 size: OperandSize, 987 ) -> Result<()>; 988 989 /// Perform a shift operation between two registers. 990 /// This case is special in that some architectures have specific expectations 991 /// regarding the location of the instruction arguments. To free the 992 /// caller from having to deal with the architecture specific constraints 993 /// we give this function access to the code generation context, allowing 994 /// each implementation to decide the lowering path. 995 fn shift( 996 &mut self, 997 context: &mut CodeGenContext<Emission>, 998 kind: ShiftKind, 999 size: OperandSize, 1000 ) -> Result<()>; 1001 1002 /// Perform division operation. 1003 /// Division is special in that some architectures have specific 1004 /// expectations regarding the location of the instruction 1005 /// arguments and regarding the location of the quotient / 1006 /// remainder. To free the caller from having to deal with the 1007 /// architecture specific constraints we give this function access 1008 /// to the code generation context, allowing each implementation 1009 /// to decide the lowering path. For cases in which division is a 1010 /// unconstrained binary operation, the caller can decide to use 1011 /// the `CodeGenContext::i32_binop` or `CodeGenContext::i64_binop` 1012 /// functions. 1013 fn div( 1014 &mut self, 1015 context: &mut CodeGenContext<Emission>, 1016 kind: DivKind, 1017 size: OperandSize, 1018 ) -> Result<()>; 1019 1020 /// Calculate remainder. 1021 fn rem( 1022 &mut self, 1023 context: &mut CodeGenContext<Emission>, 1024 kind: RemKind, 1025 size: OperandSize, 1026 ) -> Result<()>; 1027 1028 /// Compares `src1` against `src2` for the side effect of setting processor 1029 /// flags. 1030 /// 1031 /// Note that `src1` is the left-hand-side of the comparison and `src2` is 1032 /// the right-hand-side, so if testing `a < b` then `src1 == a` and 1033 /// `src2 == b` 1034 fn cmp(&mut self, src1: Reg, src2: RegImm, size: OperandSize) -> Result<()>; 1035 1036 /// Compare src and dst and put the result in dst. 1037 /// This function will potentially emit a series of instructions. 1038 /// 1039 /// The initial value in `dst` is the left-hand-side of the comparison and 1040 /// the initial value in `src` is the right-hand-side of the comparison. 1041 /// That means for `a < b` then `dst == a` and `src == b`. 1042 fn cmp_with_set( 1043 &mut self, 1044 dst: WritableReg, 1045 src: RegImm, 1046 kind: IntCmpKind, 1047 size: OperandSize, 1048 ) -> Result<()>; 1049 1050 /// Compare floats in src1 and src2 and put the result in dst. 1051 /// In x86, this will emit multiple instructions. 1052 fn float_cmp_with_set( 1053 &mut self, 1054 dst: WritableReg, 1055 src1: Reg, 1056 src2: Reg, 1057 kind: FloatCmpKind, 1058 size: OperandSize, 1059 ) -> Result<()>; 1060 1061 /// Count the number of leading zeroes in src and put the result in dst. 1062 /// In x64, this will emit multiple instructions if the `has_lzcnt` flag is 1063 /// false. 1064 fn clz(&mut self, dst: WritableReg, src: Reg, size: OperandSize) -> Result<()>; 1065 1066 /// Count the number of trailing zeroes in src and put the result in dst.masm 1067 /// In x64, this will emit multiple instructions if the `has_tzcnt` flag is 1068 /// false. 1069 fn ctz(&mut self, dst: WritableReg, src: Reg, size: OperandSize) -> Result<()>; 1070 1071 /// Push the register to the stack, returning the stack slot metadata. 1072 // NB 1073 // The stack alignment should not be assumed after any call to `push`, 1074 // unless explicitly aligned otherwise. Typically, stack alignment is 1075 // maintained at call sites and during the execution of 1076 // epilogues. 1077 fn push(&mut self, src: Reg, size: OperandSize) -> Result<StackSlot>; 1078 1079 /// Finalize the assembly and return the result. 1080 fn finalize(self, base: Option<SourceLoc>) -> Result<MachBufferFinalized<Final>>; 1081 1082 /// Zero a particular register. 1083 fn zero(&mut self, reg: WritableReg) -> Result<()>; 1084 1085 /// Count the number of 1 bits in src and put the result in dst. In x64, 1086 /// this will emit multiple instructions if the `has_popcnt` flag is false. 1087 fn popcnt(&mut self, context: &mut CodeGenContext<Emission>, size: OperandSize) -> Result<()>; 1088 1089 /// Converts an i64 to an i32 by discarding the high 32 bits. 1090 fn wrap(&mut self, dst: WritableReg, src: Reg) -> Result<()>; 1091 1092 /// Extends an integer of a given size to a larger size. 1093 fn extend(&mut self, dst: WritableReg, src: Reg, kind: ExtendKind) -> Result<()>; 1094 1095 /// Emits one or more instructions to perform a signed truncation of a 1096 /// float into an integer. 1097 fn signed_truncate( 1098 &mut self, 1099 dst: WritableReg, 1100 src: Reg, 1101 src_size: OperandSize, 1102 dst_size: OperandSize, 1103 kind: TruncKind, 1104 ) -> Result<()>; 1105 1106 /// Emits one or more instructions to perform an unsigned truncation of a 1107 /// float into an integer. 1108 fn unsigned_truncate( 1109 &mut self, 1110 context: &mut CodeGenContext<Emission>, 1111 src_size: OperandSize, 1112 dst_size: OperandSize, 1113 kind: TruncKind, 1114 ) -> Result<()>; 1115 1116 /// Emits one or more instructions to perform a signed convert of an 1117 /// integer into a float. 1118 fn signed_convert( 1119 &mut self, 1120 dst: WritableReg, 1121 src: Reg, 1122 src_size: OperandSize, 1123 dst_size: OperandSize, 1124 ) -> Result<()>; 1125 1126 /// Emits one or more instructions to perform an unsigned convert of an 1127 /// integer into a float. 1128 fn unsigned_convert( 1129 &mut self, 1130 dst: WritableReg, 1131 src: Reg, 1132 tmp_gpr: Reg, 1133 src_size: OperandSize, 1134 dst_size: OperandSize, 1135 ) -> Result<()>; 1136 1137 /// Reinterpret a float as an integer. 1138 fn reinterpret_float_as_int( 1139 &mut self, 1140 dst: WritableReg, 1141 src: Reg, 1142 size: OperandSize, 1143 ) -> Result<()>; 1144 1145 /// Reinterpret an integer as a float. 1146 fn reinterpret_int_as_float( 1147 &mut self, 1148 dst: WritableReg, 1149 src: Reg, 1150 size: OperandSize, 1151 ) -> Result<()>; 1152 1153 /// Demote an f64 to an f32. 1154 fn demote(&mut self, dst: WritableReg, src: Reg) -> Result<()>; 1155 1156 /// Promote an f32 to an f64. 1157 fn promote(&mut self, dst: WritableReg, src: Reg) -> Result<()>; 1158 1159 /// Zero a given memory range. 1160 /// 1161 /// The default implementation divides the given memory range 1162 /// into word-sized slots. Then it unrolls a series of store 1163 /// instructions, effectively assigning zero to each slot. 1164 fn zero_mem_range(&mut self, mem: &Range<u32>) -> Result<()> { 1165 let word_size = <Self::ABI as abi::ABI>::word_bytes() as u32; 1166 if mem.is_empty() { 1167 return Ok(()); 1168 } 1169 1170 let start = if mem.start % word_size == 0 { 1171 mem.start 1172 } else { 1173 // Ensure that the start of the range is at least 4-byte aligned. 1174 assert!(mem.start % 4 == 0); 1175 let start = align_to(mem.start, word_size); 1176 let addr: Self::Address = self.local_address(&LocalSlot::i32(start))?; 1177 self.store(RegImm::i32(0), addr, OperandSize::S32)?; 1178 // Ensure that the new start of the range, is word-size aligned. 1179 assert!(start % word_size == 0); 1180 start 1181 }; 1182 1183 let end = align_to(mem.end, word_size); 1184 let slots = (end - start) / word_size; 1185 1186 if slots == 1 { 1187 let slot = LocalSlot::i64(start + word_size); 1188 let addr: Self::Address = self.local_address(&slot)?; 1189 self.store(RegImm::i64(0), addr, OperandSize::S64)?; 1190 } else { 1191 // TODO 1192 // Add an upper bound to this generation; 1193 // given a considerably large amount of slots 1194 // this will be inefficient. 1195 let zero = scratch!(Self); 1196 self.zero(writable!(zero))?; 1197 let zero = RegImm::reg(zero); 1198 1199 for step in (start..end).into_iter().step_by(word_size as usize) { 1200 let slot = LocalSlot::i64(step + word_size); 1201 let addr: Self::Address = self.local_address(&slot)?; 1202 self.store(zero, addr, OperandSize::S64)?; 1203 } 1204 } 1205 1206 Ok(()) 1207 } 1208 1209 /// Generate a label. 1210 fn get_label(&mut self) -> Result<MachLabel>; 1211 1212 /// Bind the given label at the current code offset. 1213 fn bind(&mut self, label: MachLabel) -> Result<()>; 1214 1215 /// Conditional branch. 1216 /// 1217 /// Performs a comparison between the two operands, 1218 /// and immediately after emits a jump to the given 1219 /// label destination if the condition is met. 1220 fn branch( 1221 &mut self, 1222 kind: IntCmpKind, 1223 lhs: Reg, 1224 rhs: RegImm, 1225 taken: MachLabel, 1226 size: OperandSize, 1227 ) -> Result<()>; 1228 1229 /// Emits and unconditional jump to the given label. 1230 fn jmp(&mut self, target: MachLabel) -> Result<()>; 1231 1232 /// Emits a jump table sequence. The default label is specified as 1233 /// the last element of the targets slice. 1234 fn jmp_table(&mut self, targets: &[MachLabel], index: Reg, tmp: Reg) -> Result<()>; 1235 1236 /// Emit an unreachable code trap. 1237 fn unreachable(&mut self) -> Result<()>; 1238 1239 /// Emit an unconditional trap. 1240 fn trap(&mut self, code: TrapCode) -> Result<()>; 1241 1242 /// Traps if the condition code is met. 1243 fn trapif(&mut self, cc: IntCmpKind, code: TrapCode) -> Result<()>; 1244 1245 /// Trap if the source register is zero. 1246 fn trapz(&mut self, src: Reg, code: TrapCode) -> Result<()>; 1247 1248 /// Ensures that the stack pointer is correctly positioned before an unconditional 1249 /// jump according to the requirements of the destination target. 1250 fn ensure_sp_for_jump(&mut self, target: SPOffset) -> Result<()> { 1251 let bytes = self 1252 .sp_offset()? 1253 .as_u32() 1254 .checked_sub(target.as_u32()) 1255 .unwrap_or(0); 1256 1257 if bytes > 0 { 1258 self.free_stack(bytes)?; 1259 } 1260 1261 Ok(()) 1262 } 1263 1264 /// Mark the start of a source location returning the machine code offset 1265 /// and the relative source code location. 1266 fn start_source_loc(&mut self, loc: RelSourceLoc) -> Result<(CodeOffset, RelSourceLoc)>; 1267 1268 /// Mark the end of a source location. 1269 fn end_source_loc(&mut self) -> Result<()>; 1270 1271 /// The current offset, in bytes from the beginning of the function. 1272 fn current_code_offset(&self) -> Result<CodeOffset>; 1273 1274 /// Performs a 128-bit addition 1275 fn add128( 1276 &mut self, 1277 dst_lo: WritableReg, 1278 dst_hi: WritableReg, 1279 lhs_lo: Reg, 1280 lhs_hi: Reg, 1281 rhs_lo: Reg, 1282 rhs_hi: Reg, 1283 ) -> Result<()>; 1284 1285 /// Performs a 128-bit subtraction 1286 fn sub128( 1287 &mut self, 1288 dst_lo: WritableReg, 1289 dst_hi: WritableReg, 1290 lhs_lo: Reg, 1291 lhs_hi: Reg, 1292 rhs_lo: Reg, 1293 rhs_hi: Reg, 1294 ) -> Result<()>; 1295 1296 /// Performs a widening multiplication from two 64-bit operands into a 1297 /// 128-bit result. 1298 /// 1299 /// Note that some platforms require special handling of registers in this 1300 /// instruction (e.g. x64) so full access to `CodeGenContext` is provided. 1301 fn mul_wide(&mut self, context: &mut CodeGenContext<Emission>, kind: MulWideKind) 1302 -> Result<()>; 1303 1304 /// Takes the value in a src operand and replicates it across lanes of 1305 /// `size` in a destination result. 1306 fn splat(&mut self, context: &mut CodeGenContext<Emission>, size: SplatKind) -> Result<()>; 1307 1308 /// Performs a shuffle between two 128-bit vectors into a 128-bit result 1309 /// using lanes as a mask to select which indexes to copy. 1310 fn shuffle(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, lanes: [u8; 16]) -> Result<()>; 1311 1312 /// Performs the RMW `op` operation on the passed `addr`. 1313 /// 1314 /// The value *before* the operation was performed is written back to the `operand` register. 1315 fn atomic_rmw( 1316 &mut self, 1317 addr: Self::Address, 1318 operand: WritableReg, 1319 size: OperandSize, 1320 op: RmwOp, 1321 flags: MemFlags, 1322 extend: Option<ExtendKind>, 1323 ) -> Result<()>; 1324 } 1325