Skip to main content

winch_codegen/isa/x64/
asm.rs

1//! Assembler library implementation for x64.
2
3use crate::{
4    constant_pool::ConstantPool,
5    isa::{CallingConvention, reg::Reg},
6    masm::{
7        DivKind, Extend, ExtendKind, ExtendType, IntCmpKind, MulWideKind, OperandSize, RemKind,
8        RoundingMode, ShiftKind, Signed, V128ExtendKind, V128LoadExtendKind, Zero,
9    },
10    reg::writable,
11};
12use cranelift_codegen::{
13    CallInfo, MachBuffer, MachBufferFinalized, MachInst, MachInstEmit, MachInstEmitState,
14    MachLabel, PatchRegion, Writable,
15    binemit::Reloc,
16    ir::{ExternalName, MemFlagsData, SourceLoc, TrapCode, Type, UserExternalNameRef, types},
17    isa::{
18        unwind::UnwindInst,
19        x64::{
20            AtomicRmwSeqOp, EmitInfo, EmitState, Inst,
21            args::{
22                self, Amode, CC, ExtMode, FromWritableReg, Gpr, GprMem, GprMemImm, RegMem,
23                RegMemImm, SyntheticAmode, WritableGpr, WritableXmm, Xmm, XmmMem, XmmMemImm,
24            },
25            external::{PairedGpr, PairedXmm},
26            settings as x64_settings,
27        },
28    },
29    settings,
30};
31
32use crate::reg::WritableReg;
33use cranelift_assembler_x64 as asm;
34
35use super::address::Address;
36use smallvec::SmallVec;
37
38// Conversions between winch-codegen x64 types and cranelift-codegen x64 types.
39
40impl From<Reg> for RegMemImm {
41    fn from(reg: Reg) -> Self {
42        RegMemImm::reg(reg.into())
43    }
44}
45
46impl From<Reg> for RegMem {
47    fn from(value: Reg) -> Self {
48        RegMem::Reg { reg: value.into() }
49    }
50}
51
52impl From<Reg> for WritableGpr {
53    fn from(reg: Reg) -> Self {
54        let writable = Writable::from_reg(reg.into());
55        WritableGpr::from_writable_reg(writable).expect("valid writable gpr")
56    }
57}
58
59impl From<Reg> for WritableXmm {
60    fn from(reg: Reg) -> Self {
61        let writable = Writable::from_reg(reg.into());
62        WritableXmm::from_writable_reg(writable).expect("valid writable xmm")
63    }
64}
65
66/// Convert a writable GPR register to the read-write pair expected by
67/// `cranelift-codegen`.
68fn pair_gpr(reg: WritableReg) -> PairedGpr {
69    assert!(reg.to_reg().is_int());
70    let read = Gpr::unwrap_new(reg.to_reg().into());
71    let write = WritableGpr::from_reg(reg.to_reg().into());
72    PairedGpr { read, write }
73}
74
75impl From<Reg> for asm::Gpr<Gpr> {
76    fn from(reg: Reg) -> Self {
77        asm::Gpr::new(reg.into())
78    }
79}
80
81impl From<Reg> for asm::GprMem<Gpr, Gpr> {
82    fn from(reg: Reg) -> Self {
83        asm::GprMem::Gpr(reg.into())
84    }
85}
86
87/// Convert a writable XMM register to the read-write pair expected by
88/// `cranelift-codegen`.
89fn pair_xmm(reg: WritableReg) -> PairedXmm {
90    assert!(reg.to_reg().is_float());
91    let read = Xmm::unwrap_new(reg.to_reg().into());
92    let write = WritableXmm::from_reg(reg.to_reg().into());
93    PairedXmm { read, write }
94}
95
96impl From<Reg> for asm::Xmm<Xmm> {
97    fn from(reg: Reg) -> Self {
98        asm::Xmm::new(reg.into())
99    }
100}
101
102impl From<Reg> for asm::XmmMem<Xmm, Gpr> {
103    fn from(reg: Reg) -> Self {
104        asm::XmmMem::Xmm(reg.into())
105    }
106}
107
108impl From<Reg> for Gpr {
109    fn from(reg: Reg) -> Self {
110        Gpr::unwrap_new(reg.into())
111    }
112}
113
114impl From<Reg> for GprMem {
115    fn from(value: Reg) -> Self {
116        GprMem::unwrap_new(value.into())
117    }
118}
119
120impl From<Reg> for GprMemImm {
121    fn from(reg: Reg) -> Self {
122        GprMemImm::unwrap_new(reg.into())
123    }
124}
125
126impl From<Reg> for Xmm {
127    fn from(reg: Reg) -> Self {
128        Xmm::unwrap_new(reg.into())
129    }
130}
131
132impl From<Reg> for XmmMem {
133    fn from(value: Reg) -> Self {
134        XmmMem::unwrap_new(value.into())
135    }
136}
137
138impl From<Reg> for XmmMemImm {
139    fn from(value: Reg) -> Self {
140        XmmMemImm::unwrap_new(value.into())
141    }
142}
143
144impl From<OperandSize> for args::OperandSize {
145    fn from(size: OperandSize) -> Self {
146        match size {
147            OperandSize::S8 => Self::Size8,
148            OperandSize::S16 => Self::Size16,
149            OperandSize::S32 => Self::Size32,
150            OperandSize::S64 => Self::Size64,
151            s => panic!("Invalid operand size {s:?}"),
152        }
153    }
154}
155
156impl From<IntCmpKind> for CC {
157    fn from(value: IntCmpKind) -> Self {
158        match value {
159            IntCmpKind::Eq => CC::Z,
160            IntCmpKind::Ne => CC::NZ,
161            IntCmpKind::LtS => CC::L,
162            IntCmpKind::LtU => CC::B,
163            IntCmpKind::GtS => CC::NLE,
164            IntCmpKind::GtU => CC::NBE,
165            IntCmpKind::LeS => CC::LE,
166            IntCmpKind::LeU => CC::BE,
167            IntCmpKind::GeS => CC::NL,
168            IntCmpKind::GeU => CC::NB,
169        }
170    }
171}
172
173impl<T: ExtendType> From<Extend<T>> for ExtMode {
174    fn from(value: Extend<T>) -> Self {
175        match value {
176            Extend::I32Extend8 => ExtMode::BL,
177            Extend::I32Extend16 => ExtMode::WL,
178            Extend::I64Extend8 => ExtMode::BQ,
179            Extend::I64Extend16 => ExtMode::WQ,
180            Extend::I64Extend32 => ExtMode::LQ,
181            Extend::__Kind(_) => unreachable!(),
182        }
183    }
184}
185
186impl From<ExtendKind> for ExtMode {
187    fn from(value: ExtendKind) -> Self {
188        match value {
189            ExtendKind::Signed(s) => s.into(),
190            ExtendKind::Unsigned(u) => u.into(),
191        }
192    }
193}
194
195/// Kinds of extends supported by `vpmov`.
196pub(super) enum VpmovKind {
197    /// Sign extends 8 lanes of 8-bit integers to 8 lanes of 16-bit integers.
198    E8x8S,
199    /// Zero extends 8 lanes of 8-bit integers to 8 lanes of 16-bit integers.
200    E8x8U,
201    /// Sign extends 4 lanes of 16-bit integers to 4 lanes of 32-bit integers.
202    E16x4S,
203    /// Zero extends 4 lanes of 16-bit integers to 4 lanes of 32-bit integers.
204    E16x4U,
205    /// Sign extends 2 lanes of 32-bit integers to 2 lanes of 64-bit integers.
206    E32x2S,
207    /// Zero extends 2 lanes of 32-bit integers to 2 lanes of 64-bit integers.
208    E32x2U,
209}
210
211impl From<V128LoadExtendKind> for VpmovKind {
212    fn from(value: V128LoadExtendKind) -> Self {
213        match value {
214            V128LoadExtendKind::E8x8S => Self::E8x8S,
215            V128LoadExtendKind::E8x8U => Self::E8x8U,
216            V128LoadExtendKind::E16x4S => Self::E16x4S,
217            V128LoadExtendKind::E16x4U => Self::E16x4U,
218            V128LoadExtendKind::E32x2S => Self::E32x2S,
219            V128LoadExtendKind::E32x2U => Self::E32x2U,
220        }
221    }
222}
223
224impl From<V128ExtendKind> for VpmovKind {
225    fn from(value: V128ExtendKind) -> Self {
226        match value {
227            V128ExtendKind::LowI8x16S | V128ExtendKind::HighI8x16S => Self::E8x8S,
228            V128ExtendKind::LowI8x16U => Self::E8x8U,
229            V128ExtendKind::LowI16x8S | V128ExtendKind::HighI16x8S => Self::E16x4S,
230            V128ExtendKind::LowI16x8U => Self::E16x4U,
231            V128ExtendKind::LowI32x4S | V128ExtendKind::HighI32x4S => Self::E32x2S,
232            V128ExtendKind::LowI32x4U => Self::E32x2U,
233            _ => unimplemented!(),
234        }
235    }
236}
237
238/// Kinds of comparisons supported by `vcmp`.
239pub(super) enum VcmpKind {
240    /// Equal comparison.
241    Eq,
242    /// Not equal comparison.
243    Ne,
244    /// Less than comparison.
245    Lt,
246    /// Less than or equal comparison.
247    Le,
248    /// Unordered comparison. Sets result to all 1s if either source operand is
249    /// NaN.
250    Unord,
251}
252
253/// Kinds of conversions supported by `vcvt`.
254pub(super) enum VcvtKind {
255    /// Converts 32-bit integers to 32-bit floats.
256    I32ToF32,
257    /// Converts doubleword integers to double precision floats.
258    I32ToF64,
259    /// Converts double precision floats to single precision floats.
260    F64ToF32,
261    // Converts double precision floats to 32-bit integers.
262    F64ToI32,
263    /// Converts single precision floats to double precision floats.
264    F32ToF64,
265    /// Converts single precision floats to 32-bit integers.
266    F32ToI32,
267}
268
269/// Modes supported by `vround`.
270pub(crate) enum VroundMode {
271    /// Rounds toward nearest (ties to even).
272    TowardNearest,
273    /// Rounds toward negative infinity.
274    TowardNegativeInfinity,
275    /// Rounds toward positive infinity.
276    TowardPositiveInfinity,
277    /// Rounds toward zero.
278    TowardZero,
279}
280
281/// Low level assembler implementation for x64.
282pub(crate) struct Assembler {
283    /// The machine instruction buffer.
284    buffer: MachBuffer<Inst>,
285    /// Constant emission information.
286    emit_info: EmitInfo,
287    /// Emission state.
288    emit_state: EmitState,
289    /// x64 flags.
290    isa_flags: x64_settings::Flags,
291    /// Constant pool.
292    pool: ConstantPool,
293}
294
295impl Assembler {
296    /// Create a new x64 assembler.
297    pub fn new(shared_flags: settings::Flags, isa_flags: x64_settings::Flags) -> Self {
298        Self {
299            buffer: MachBuffer::<Inst>::new(),
300            emit_state: Default::default(),
301            emit_info: EmitInfo::new(shared_flags, isa_flags.clone()),
302            pool: ConstantPool::new(),
303            isa_flags,
304        }
305    }
306
307    /// Get a mutable reference to underlying
308    /// machine buffer.
309    pub fn buffer_mut(&mut self) -> &mut MachBuffer<Inst> {
310        &mut self.buffer
311    }
312
313    /// Get a reference to the underlying machine buffer.
314    pub fn buffer(&self) -> &MachBuffer<Inst> {
315        &self.buffer
316    }
317
318    /// Adds a constant to the constant pool and returns its address.
319    pub fn add_constant(&mut self, constant: &[u8]) -> Address {
320        let handle = self.pool.register(constant, &mut self.buffer);
321        Address::constant(handle)
322    }
323
324    /// Load a floating point constant, using the constant pool.
325    pub fn load_fp_const(&mut self, dst: WritableReg, constant: &[u8], size: OperandSize) {
326        let addr = self.add_constant(constant);
327        self.xmm_mov_mr(&addr, dst, size, MemFlagsData::trusted());
328    }
329
330    /// Return the emitted code.
331    pub fn finalize(mut self, loc: Option<SourceLoc>) -> MachBufferFinalized {
332        let mut buffer = self
333            .buffer
334            .finish(&self.pool.constants(), self.emit_state.ctrl_plane_mut());
335        buffer.apply_base_srcloc(loc.unwrap_or_default());
336        buffer
337    }
338
339    fn emit(&mut self, inst: Inst) {
340        inst.emit(&mut self.buffer, &self.emit_info, &mut self.emit_state);
341    }
342
343    fn to_synthetic_amode(addr: &Address, memflags: MemFlagsData) -> SyntheticAmode {
344        match *addr {
345            Address::Offset { base, offset } => {
346                let amode = Amode::imm_reg(offset as i32, base.into()).with_flags(memflags);
347                SyntheticAmode::real(amode)
348            }
349            Address::Const(c) => SyntheticAmode::ConstantOffset(c),
350            Address::ImmRegRegShift {
351                simm32,
352                base,
353                index,
354                shift,
355            } => SyntheticAmode::Real(Amode::ImmRegRegShift {
356                simm32,
357                base: base.into(),
358                index: index.into(),
359                shift,
360                flags: memflags,
361            }),
362        }
363    }
364
365    /// Emit an unwind instruction.
366    pub fn unwind_inst(&mut self, inst: UnwindInst) {
367        self.emit(Inst::Unwind { inst })
368    }
369
370    /// Push register.
371    pub fn push_r(&mut self, reg: Reg) {
372        let inst = asm::inst::pushq_o::new(reg).into();
373        self.emit(Inst::External { inst });
374    }
375
376    /// Pop to register.
377    pub fn pop_r(&mut self, dst: WritableReg) {
378        let writable: WritableGpr = dst.map(Into::into);
379        let inst = asm::inst::popq_o::new(writable).into();
380        self.emit(Inst::External { inst });
381    }
382
383    /// Return and pop the given number of stack-argument bytes.
384    pub fn ret(&mut self, stack_args_size: u16) {
385        let inst = if stack_args_size == 0 {
386            asm::inst::retq_zo::new().into()
387        } else {
388            asm::inst::retq_i::new(stack_args_size).into()
389        };
390        self.emit(Inst::External { inst });
391    }
392
393    /// Register-to-register move.
394    pub fn mov_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
395        let dst: WritableGpr = dst.map(|r| r.into());
396        let inst = match size {
397            OperandSize::S8 => asm::inst::movb_mr::new(dst, src).into(),
398            OperandSize::S16 => asm::inst::movw_mr::new(dst, src).into(),
399            OperandSize::S32 => asm::inst::movl_mr::new(dst, src).into(),
400            OperandSize::S64 => asm::inst::movq_mr::new(dst, src).into(),
401            _ => unreachable!(),
402        };
403        self.emit(Inst::External { inst });
404    }
405
406    /// Register-to-memory move.
407    pub fn mov_rm(&mut self, src: Reg, addr: &Address, size: OperandSize, flags: MemFlagsData) {
408        assert!(addr.is_offset());
409        let dst = Self::to_synthetic_amode(addr, flags);
410        let inst = match size {
411            OperandSize::S8 => asm::inst::movb_mr::new(dst, src).into(),
412            OperandSize::S16 => asm::inst::movw_mr::new(dst, src).into(),
413            OperandSize::S32 => asm::inst::movl_mr::new(dst, src).into(),
414            OperandSize::S64 => asm::inst::movq_mr::new(dst, src).into(),
415            _ => unreachable!(),
416        };
417        self.emit(Inst::External { inst });
418    }
419
420    /// Immediate-to-memory move.
421    pub fn mov_im(&mut self, src: i32, addr: &Address, size: OperandSize, flags: MemFlagsData) {
422        assert!(addr.is_offset());
423        let dst = Self::to_synthetic_amode(addr, flags);
424        let inst = match size {
425            OperandSize::S8 => {
426                let src = i8::try_from(src).unwrap();
427                asm::inst::movb_mi::new(dst, src.cast_unsigned()).into()
428            }
429            OperandSize::S16 => {
430                let src = i16::try_from(src).unwrap();
431                asm::inst::movw_mi::new(dst, src.cast_unsigned()).into()
432            }
433            OperandSize::S32 => asm::inst::movl_mi::new(dst, src.cast_unsigned()).into(),
434            OperandSize::S64 => asm::inst::movq_mi_sxl::new(dst, src).into(),
435            _ => unreachable!(),
436        };
437        self.emit(Inst::External { inst });
438    }
439
440    /// Immediate-to-register move.
441    pub fn mov_ir(&mut self, imm: u64, dst: WritableReg, size: OperandSize) {
442        self.emit(Inst::imm(size.into(), imm, dst.map(Into::into)));
443    }
444
445    /// Zero-extend memory-to-register load.
446    pub fn movzx_mr(
447        &mut self,
448        addr: &Address,
449        dst: WritableReg,
450        ext: Option<Extend<Zero>>,
451        memflags: MemFlagsData,
452    ) {
453        let src = Self::to_synthetic_amode(addr, memflags);
454
455        if let Some(ext) = ext {
456            let dst = WritableGpr::from_reg(dst.to_reg().into());
457            let inst = match ext.into() {
458                ExtMode::BL => asm::inst::movzbl_rm::new(dst, src).into(),
459                ExtMode::BQ => asm::inst::movzbq_rm::new(dst, src).into(),
460                ExtMode::WL => asm::inst::movzwl_rm::new(dst, src).into(),
461                ExtMode::WQ => asm::inst::movzwq_rm::new(dst, src).into(),
462                ExtMode::LQ => {
463                    // This instruction selection may seem strange but is
464                    // correct in 64-bit mode: section 3.4.1.1 of the Intel
465                    // manual says that "32-bit operands generate a 32-bit
466                    // result, zero-extended to a 64-bit result in the
467                    // destination general-purpose register." This is applicable
468                    // beyond `mov` but we use this fact to zero-extend `src`
469                    // into `dst`.
470                    asm::inst::movl_rm::new(dst, src).into()
471                }
472            };
473            self.emit(Inst::External { inst });
474        } else {
475            let dst = WritableGpr::from_reg(dst.to_reg().into());
476            let inst = asm::inst::movq_rm::new(dst, src).into();
477            self.emit(Inst::External { inst });
478        }
479    }
480
481    // Sign-extend memory-to-register load.
482    pub fn movsx_mr(
483        &mut self,
484        addr: &Address,
485        dst: WritableReg,
486        ext: Extend<Signed>,
487        memflags: MemFlagsData,
488    ) {
489        let src = Self::to_synthetic_amode(addr, memflags);
490        let dst = WritableGpr::from_reg(dst.to_reg().into());
491        let inst = match ext.into() {
492            ExtMode::BL => asm::inst::movsbl_rm::new(dst, src).into(),
493            ExtMode::BQ => asm::inst::movsbq_rm::new(dst, src).into(),
494            ExtMode::WL => asm::inst::movswl_rm::new(dst, src).into(),
495            ExtMode::WQ => asm::inst::movswq_rm::new(dst, src).into(),
496            ExtMode::LQ => asm::inst::movslq_rm::new(dst, src).into(),
497        };
498        self.emit(Inst::External { inst });
499    }
500
501    /// Register-to-register move with zero extension.
502    pub fn movzx_rr(&mut self, src: Reg, dst: WritableReg, kind: Extend<Zero>) {
503        let dst = WritableGpr::from_reg(dst.to_reg().into());
504        let inst = match kind.into() {
505            ExtMode::BL => asm::inst::movzbl_rm::new(dst, src).into(),
506            ExtMode::BQ => asm::inst::movzbq_rm::new(dst, src).into(),
507            ExtMode::WL => asm::inst::movzwl_rm::new(dst, src).into(),
508            ExtMode::WQ => asm::inst::movzwq_rm::new(dst, src).into(),
509            ExtMode::LQ => {
510                // This instruction selection may seem strange but is correct in
511                // 64-bit mode: section 3.4.1.1 of the Intel manual says that
512                // "32-bit operands generate a 32-bit result, zero-extended to a
513                // 64-bit result in the destination general-purpose register."
514                // This is applicable beyond `mov` but we use this fact to
515                // zero-extend `src` into `dst`.
516                asm::inst::movl_rm::new(dst, src).into()
517            }
518        };
519        self.emit(Inst::External { inst });
520    }
521
522    /// Register-to-register move with sign extension.
523    pub fn movsx_rr(&mut self, src: Reg, dst: WritableReg, kind: Extend<Signed>) {
524        let dst = WritableGpr::from_reg(dst.to_reg().into());
525        let inst = match kind.into() {
526            ExtMode::BL => asm::inst::movsbl_rm::new(dst, src).into(),
527            ExtMode::BQ => asm::inst::movsbq_rm::new(dst, src).into(),
528            ExtMode::WL => asm::inst::movswl_rm::new(dst, src).into(),
529            ExtMode::WQ => asm::inst::movswq_rm::new(dst, src).into(),
530            ExtMode::LQ => asm::inst::movslq_rm::new(dst, src).into(),
531        };
532        self.emit(Inst::External { inst });
533    }
534
535    /// Integer register conditional move.
536    pub fn cmov(&mut self, src: Reg, dst: WritableReg, cc: IntCmpKind, size: OperandSize) {
537        use IntCmpKind::*;
538        use OperandSize::*;
539
540        let dst: WritableGpr = dst.map(Into::into);
541        let inst = match size {
542            S8 | S16 | S32 => match cc {
543                Eq => asm::inst::cmovel_rm::new(dst, src).into(),
544                Ne => asm::inst::cmovnel_rm::new(dst, src).into(),
545                LtS => asm::inst::cmovll_rm::new(dst, src).into(),
546                LtU => asm::inst::cmovbl_rm::new(dst, src).into(),
547                GtS => asm::inst::cmovgl_rm::new(dst, src).into(),
548                GtU => asm::inst::cmoval_rm::new(dst, src).into(),
549                LeS => asm::inst::cmovlel_rm::new(dst, src).into(),
550                LeU => asm::inst::cmovbel_rm::new(dst, src).into(),
551                GeS => asm::inst::cmovgel_rm::new(dst, src).into(),
552                GeU => asm::inst::cmovael_rm::new(dst, src).into(),
553            },
554            S64 => match cc {
555                Eq => asm::inst::cmoveq_rm::new(dst, src).into(),
556                Ne => asm::inst::cmovneq_rm::new(dst, src).into(),
557                LtS => asm::inst::cmovlq_rm::new(dst, src).into(),
558                LtU => asm::inst::cmovbq_rm::new(dst, src).into(),
559                GtS => asm::inst::cmovgq_rm::new(dst, src).into(),
560                GtU => asm::inst::cmovaq_rm::new(dst, src).into(),
561                LeS => asm::inst::cmovleq_rm::new(dst, src).into(),
562                LeU => asm::inst::cmovbeq_rm::new(dst, src).into(),
563                GeS => asm::inst::cmovgeq_rm::new(dst, src).into(),
564                GeU => asm::inst::cmovaeq_rm::new(dst, src).into(),
565            },
566            _ => unreachable!(),
567        };
568        self.emit(Inst::External { inst });
569    }
570
571    /// Single and double precision floating point
572    /// register-to-register move.
573    pub fn xmm_mov_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
574        let ty = match size {
575            OperandSize::S32 => types::F32,
576            OperandSize::S64 => types::F64,
577            OperandSize::S128 => types::I32X4,
578            OperandSize::S8 | OperandSize::S16 => unreachable!(),
579        };
580        self.emit(Inst::gen_move(dst.map(|r| r.into()), src.into(), ty));
581    }
582
583    /// Single and double precision floating point load.
584    pub fn xmm_mov_mr(
585        &mut self,
586        src: &Address,
587        dst: WritableReg,
588        size: OperandSize,
589        flags: MemFlagsData,
590    ) {
591        use OperandSize::*;
592
593        assert!(dst.to_reg().is_float());
594
595        let src = Self::to_synthetic_amode(src, flags);
596        let dst: WritableXmm = dst.map(|r| r.into());
597        let inst = match size {
598            S32 => asm::inst::movss_a_m::new(dst, src).into(),
599            S64 => asm::inst::movsd_a_m::new(dst, src).into(),
600            S128 => asm::inst::movdqu_a::new(dst, src).into(),
601            S8 | S16 => unreachable!(),
602        };
603        self.emit(Inst::External { inst });
604    }
605
606    /// Vector load and extend.
607    pub fn xmm_vpmov_mr(
608        &mut self,
609        src: &Address,
610        dst: WritableReg,
611        kind: VpmovKind,
612        flags: MemFlagsData,
613    ) {
614        assert!(dst.to_reg().is_float());
615        let src = Self::to_synthetic_amode(src, flags);
616        let dst: WritableXmm = dst.map(|r| r.into());
617        let inst = match kind {
618            VpmovKind::E8x8S => asm::inst::vpmovsxbw_a::new(dst, src).into(),
619            VpmovKind::E8x8U => asm::inst::vpmovzxbw_a::new(dst, src).into(),
620            VpmovKind::E16x4S => asm::inst::vpmovsxwd_a::new(dst, src).into(),
621            VpmovKind::E16x4U => asm::inst::vpmovzxwd_a::new(dst, src).into(),
622            VpmovKind::E32x2S => asm::inst::vpmovsxdq_a::new(dst, src).into(),
623            VpmovKind::E32x2U => asm::inst::vpmovzxdq_a::new(dst, src).into(),
624        };
625        self.emit(Inst::External { inst });
626    }
627
628    /// Extends vector of integers in `src` and puts results in `dst`.
629    pub fn xmm_vpmov_rr(&mut self, src: Reg, dst: WritableReg, kind: VpmovKind) {
630        let dst: WritableXmm = dst.map(|r| r.into());
631        let inst = match kind {
632            VpmovKind::E8x8S => asm::inst::vpmovsxbw_a::new(dst, src).into(),
633            VpmovKind::E8x8U => asm::inst::vpmovzxbw_a::new(dst, src).into(),
634            VpmovKind::E16x4S => asm::inst::vpmovsxwd_a::new(dst, src).into(),
635            VpmovKind::E16x4U => asm::inst::vpmovzxwd_a::new(dst, src).into(),
636            VpmovKind::E32x2S => asm::inst::vpmovsxdq_a::new(dst, src).into(),
637            VpmovKind::E32x2U => asm::inst::vpmovzxdq_a::new(dst, src).into(),
638        };
639        self.emit(Inst::External { inst });
640    }
641
642    /// Vector load and broadcast.
643    pub fn xmm_vpbroadcast_mr(
644        &mut self,
645        src: &Address,
646        dst: WritableReg,
647        size: OperandSize,
648        flags: MemFlagsData,
649    ) {
650        assert!(dst.to_reg().is_float());
651        let src = Self::to_synthetic_amode(src, flags);
652        let dst: WritableXmm = dst.map(|r| r.into());
653        let inst = match size {
654            OperandSize::S8 => asm::inst::vpbroadcastb_a::new(dst, src).into(),
655            OperandSize::S16 => asm::inst::vpbroadcastw_a::new(dst, src).into(),
656            OperandSize::S32 => asm::inst::vpbroadcastd_a::new(dst, src).into(),
657            _ => unimplemented!(),
658        };
659        self.emit(Inst::External { inst });
660    }
661
662    /// Value in `src` is broadcast into lanes of `size` in `dst`.
663    pub fn xmm_vpbroadcast_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
664        assert!(src.is_float() && dst.to_reg().is_float());
665        let dst: WritableXmm = dst.map(|r| r.into());
666        let inst = match size {
667            OperandSize::S8 => asm::inst::vpbroadcastb_a::new(dst, src).into(),
668            OperandSize::S16 => asm::inst::vpbroadcastw_a::new(dst, src).into(),
669            OperandSize::S32 => asm::inst::vpbroadcastd_a::new(dst, src).into(),
670            _ => unimplemented!(),
671        };
672        self.emit(Inst::External { inst });
673    }
674
675    /// Memory to register shuffle of bytes in vector.
676    pub fn xmm_vpshuf_mr(
677        &mut self,
678        src: &Address,
679        dst: WritableReg,
680        mask: u8,
681        size: OperandSize,
682        flags: MemFlagsData,
683    ) {
684        let dst: WritableXmm = dst.map(|r| r.into());
685        let src = Self::to_synthetic_amode(src, flags);
686        let inst = match size {
687            OperandSize::S32 => asm::inst::vpshufd_a::new(dst, src, mask).into(),
688            _ => unimplemented!(),
689        };
690        self.emit(Inst::External { inst });
691    }
692
693    /// Register to register shuffle of bytes in vector.
694    pub fn xmm_vpshuf_rr(&mut self, src: Reg, dst: WritableReg, mask: u8, size: OperandSize) {
695        let dst: WritableXmm = dst.map(|r| r.into());
696
697        let inst = match size {
698            OperandSize::S16 => asm::inst::vpshuflw_a::new(dst, src, mask).into(),
699            OperandSize::S32 => asm::inst::vpshufd_a::new(dst, src, mask).into(),
700            _ => unimplemented!(),
701        };
702
703        self.emit(Inst::External { inst });
704    }
705
706    /// Single and double precision floating point store.
707    pub fn xmm_mov_rm(&mut self, src: Reg, dst: &Address, size: OperandSize, flags: MemFlagsData) {
708        use OperandSize::*;
709
710        assert!(src.is_float());
711
712        let dst = Self::to_synthetic_amode(dst, flags);
713        let src: Xmm = src.into();
714        let inst = match size {
715            S32 => asm::inst::movss_c_m::new(dst, src).into(),
716            S64 => asm::inst::movsd_c_m::new(dst, src).into(),
717            S128 => asm::inst::movdqu_b::new(dst, src).into(),
718            S16 | S8 => unreachable!(),
719        };
720        self.emit(Inst::External { inst })
721    }
722
723    /// Floating point register conditional move.
724    pub fn xmm_cmov(&mut self, src: Reg, dst: WritableReg, cc: IntCmpKind, size: OperandSize) {
725        let dst: WritableXmm = dst.map(Into::into);
726        let ty = match size {
727            OperandSize::S32 => types::F32,
728            OperandSize::S64 => types::F64,
729            // Move the entire 128 bits via movdqa.
730            OperandSize::S128 => types::I32X4,
731            OperandSize::S8 | OperandSize::S16 => unreachable!(),
732        };
733
734        self.emit(Inst::XmmCmove {
735            ty,
736            cc: cc.into(),
737            consequent: Xmm::unwrap_new(src.into()),
738            alternative: dst.to_reg(),
739            dst,
740        })
741    }
742
743    /// Subtract register and register
744    pub fn sub_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
745        let dst = pair_gpr(dst);
746        let inst = match size {
747            OperandSize::S8 => asm::inst::subb_rm::new(dst, src).into(),
748            OperandSize::S16 => asm::inst::subw_rm::new(dst, src).into(),
749            OperandSize::S32 => asm::inst::subl_rm::new(dst, src).into(),
750            OperandSize::S64 => asm::inst::subq_rm::new(dst, src).into(),
751            OperandSize::S128 => unimplemented!(),
752        };
753        self.emit(Inst::External { inst });
754    }
755
756    /// Subtract immediate register.
757    pub fn sub_ir(&mut self, imm: i32, dst: WritableReg, size: OperandSize) {
758        let dst = pair_gpr(dst);
759        let inst = match size {
760            OperandSize::S8 => asm::inst::subb_mi::new(dst, u8::try_from(imm).unwrap()).into(),
761            OperandSize::S16 => asm::inst::subw_mi::new(dst, u16::try_from(imm).unwrap()).into(),
762            OperandSize::S32 => asm::inst::subl_mi::new(dst, imm as u32).into(),
763            OperandSize::S64 => asm::inst::subq_mi_sxl::new(dst, imm).into(),
764            OperandSize::S128 => unimplemented!(),
765        };
766        self.emit(Inst::External { inst });
767    }
768
769    /// "and" two registers.
770    pub fn and_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
771        let dst = pair_gpr(dst);
772        let inst = match size {
773            OperandSize::S8 => asm::inst::andb_rm::new(dst, src).into(),
774            OperandSize::S16 => asm::inst::andw_rm::new(dst, src).into(),
775            OperandSize::S32 => asm::inst::andl_rm::new(dst, src).into(),
776            OperandSize::S64 => asm::inst::andq_rm::new(dst, src).into(),
777            OperandSize::S128 => unimplemented!(),
778        };
779        self.emit(Inst::External { inst });
780    }
781
782    pub fn and_ir(&mut self, imm: i32, dst: WritableReg, size: OperandSize) {
783        let dst = pair_gpr(dst);
784        let inst = match size {
785            OperandSize::S8 => asm::inst::andb_mi::new(dst, u8::try_from(imm).unwrap()).into(),
786            OperandSize::S16 => asm::inst::andw_mi::new(dst, u16::try_from(imm).unwrap()).into(),
787            OperandSize::S32 => asm::inst::andl_mi::new(dst, imm as u32).into(),
788            OperandSize::S64 => asm::inst::andq_mi_sxl::new(dst, imm).into(),
789            OperandSize::S128 => unimplemented!(),
790        };
791        self.emit(Inst::External { inst });
792    }
793
794    /// "and" two float registers.
795    pub fn xmm_and_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
796        let dst = pair_xmm(dst);
797        let inst = match size {
798            OperandSize::S32 => asm::inst::andps_a::new(dst, src).into(),
799            OperandSize::S64 => asm::inst::andpd_a::new(dst, src).into(),
800            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
801        };
802        self.emit(Inst::External { inst });
803    }
804
805    /// "and not" two float registers.
806    pub fn xmm_andn_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
807        let dst = pair_xmm(dst);
808        let inst = match size {
809            OperandSize::S32 => asm::inst::andnps_a::new(dst, src).into(),
810            OperandSize::S64 => asm::inst::andnpd_a::new(dst, src).into(),
811            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
812        };
813        self.emit(Inst::External { inst });
814    }
815
816    pub fn gpr_to_xmm(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
817        let dst: WritableXmm = dst.map(|r| r.into());
818        let inst = match size {
819            OperandSize::S32 => asm::inst::movd_a::new(dst, src).into(),
820            OperandSize::S64 => asm::inst::movq_a::new(dst, src).into(),
821            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
822        };
823
824        self.emit(Inst::External { inst });
825    }
826
827    pub fn xmm_to_gpr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
828        let dst: WritableGpr = dst.map(Into::into);
829        let src: Xmm = src.into();
830        let inst = match size {
831            OperandSize::S32 => asm::inst::movd_b::new(dst, src).into(),
832            OperandSize::S64 => asm::inst::movq_b::new(dst, src).into(),
833            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
834        };
835
836        self.emit(Inst::External { inst })
837    }
838
839    /// Convert float to signed int.
840    pub fn cvt_float_to_sint_seq(
841        &mut self,
842        src: Reg,
843        dst: WritableReg,
844        tmp_gpr: Reg,
845        tmp_xmm: Reg,
846        src_size: OperandSize,
847        dst_size: OperandSize,
848        saturating: bool,
849    ) {
850        self.emit(Inst::CvtFloatToSintSeq {
851            dst_size: dst_size.into(),
852            src_size: src_size.into(),
853            is_saturating: saturating,
854            src: src.into(),
855            dst: dst.map(Into::into),
856            tmp_gpr: tmp_gpr.into(),
857            tmp_xmm: tmp_xmm.into(),
858        });
859    }
860
861    /// Convert float to unsigned int.
862    pub fn cvt_float_to_uint_seq(
863        &mut self,
864        src: Reg,
865        dst: WritableReg,
866        tmp_gpr: Reg,
867        tmp_xmm: Reg,
868        tmp_xmm2: Reg,
869        src_size: OperandSize,
870        dst_size: OperandSize,
871        saturating: bool,
872    ) {
873        self.emit(Inst::CvtFloatToUintSeq {
874            dst_size: dst_size.into(),
875            src_size: src_size.into(),
876            is_saturating: saturating,
877            src: src.into(),
878            dst: dst.map(Into::into),
879            tmp_gpr: tmp_gpr.into(),
880            tmp_xmm: tmp_xmm.into(),
881            tmp_xmm2: tmp_xmm2.into(),
882        });
883    }
884
885    /// Convert signed int to float.
886    pub fn cvt_sint_to_float(
887        &mut self,
888        src: Reg,
889        dst: WritableReg,
890        src_size: OperandSize,
891        dst_size: OperandSize,
892    ) {
893        use OperandSize::*;
894        let dst = pair_xmm(dst);
895        let inst = match (src_size, dst_size) {
896            (S32, S32) => asm::inst::cvtsi2ssl_a::new(dst, src).into(),
897            (S32, S64) => asm::inst::cvtsi2sdl_a::new(dst, src).into(),
898            (S64, S32) => asm::inst::cvtsi2ssq_a::new(dst, src).into(),
899            (S64, S64) => asm::inst::cvtsi2sdq_a::new(dst, src).into(),
900            _ => unreachable!(),
901        };
902        self.emit(Inst::External { inst });
903    }
904
905    /// Convert unsigned 64-bit int to float.
906    pub fn cvt_uint64_to_float_seq(
907        &mut self,
908        src: Reg,
909        dst: WritableReg,
910        tmp_gpr1: Reg,
911        tmp_gpr2: Reg,
912        dst_size: OperandSize,
913    ) {
914        self.emit(Inst::CvtUint64ToFloatSeq {
915            dst_size: dst_size.into(),
916            src: src.into(),
917            dst: dst.map(Into::into),
918            tmp_gpr1: tmp_gpr1.into(),
919            tmp_gpr2: tmp_gpr2.into(),
920        });
921    }
922
923    /// Change precision of float.
924    pub fn cvt_float_to_float(
925        &mut self,
926        src: Reg,
927        dst: WritableReg,
928        src_size: OperandSize,
929        dst_size: OperandSize,
930    ) {
931        use OperandSize::*;
932        let dst = pair_xmm(dst);
933        let inst = match (src_size, dst_size) {
934            (S32, S64) => asm::inst::cvtss2sd_a::new(dst, src).into(),
935            (S64, S32) => asm::inst::cvtsd2ss_a::new(dst, src).into(),
936            _ => unimplemented!(),
937        };
938        self.emit(Inst::External { inst });
939    }
940
941    pub fn or_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
942        let dst = pair_gpr(dst);
943        let inst = match size {
944            OperandSize::S8 => asm::inst::orb_rm::new(dst, src).into(),
945            OperandSize::S16 => asm::inst::orw_rm::new(dst, src).into(),
946            OperandSize::S32 => asm::inst::orl_rm::new(dst, src).into(),
947            OperandSize::S64 => asm::inst::orq_rm::new(dst, src).into(),
948            OperandSize::S128 => unimplemented!(),
949        };
950        self.emit(Inst::External { inst });
951    }
952
953    pub fn or_ir(&mut self, imm: i32, dst: WritableReg, size: OperandSize) {
954        let dst = pair_gpr(dst);
955        let inst = match size {
956            OperandSize::S8 => asm::inst::orb_mi::new(dst, u8::try_from(imm).unwrap()).into(),
957            OperandSize::S16 => asm::inst::orw_mi::new(dst, u16::try_from(imm).unwrap()).into(),
958            OperandSize::S32 => asm::inst::orl_mi::new(dst, imm as u32).into(),
959            OperandSize::S64 => asm::inst::orq_mi_sxl::new(dst, imm).into(),
960            OperandSize::S128 => unimplemented!(),
961        };
962        self.emit(Inst::External { inst });
963    }
964
965    pub fn xmm_or_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
966        let dst = pair_xmm(dst);
967        let inst = match size {
968            OperandSize::S32 => asm::inst::orps_a::new(dst, src).into(),
969            OperandSize::S64 => asm::inst::orpd_a::new(dst, src).into(),
970            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
971        };
972        self.emit(Inst::External { inst });
973    }
974
975    /// Logical exclusive or with registers.
976    pub fn xor_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
977        let dst = pair_gpr(dst);
978        let inst = match size {
979            OperandSize::S8 => asm::inst::xorb_rm::new(dst, src).into(),
980            OperandSize::S16 => asm::inst::xorw_rm::new(dst, src).into(),
981            OperandSize::S32 => asm::inst::xorl_rm::new(dst, src).into(),
982            OperandSize::S64 => asm::inst::xorq_rm::new(dst, src).into(),
983            OperandSize::S128 => unimplemented!(),
984        };
985        self.emit(Inst::External { inst });
986    }
987
988    pub fn xor_ir(&mut self, imm: i32, dst: WritableReg, size: OperandSize) {
989        let dst = pair_gpr(dst);
990        let inst = match size {
991            OperandSize::S8 => asm::inst::xorb_mi::new(dst, u8::try_from(imm).unwrap()).into(),
992            OperandSize::S16 => asm::inst::xorw_mi::new(dst, u16::try_from(imm).unwrap()).into(),
993            OperandSize::S32 => asm::inst::xorl_mi::new(dst, imm as u32).into(),
994            OperandSize::S64 => asm::inst::xorq_mi_sxl::new(dst, imm).into(),
995            OperandSize::S128 => unimplemented!(),
996        };
997        self.emit(Inst::External { inst });
998    }
999
1000    /// Logical exclusive or with float registers.
1001    pub fn xmm_xor_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1002        let dst = pair_xmm(dst);
1003        let inst = match size {
1004            OperandSize::S32 => asm::inst::xorps_a::new(dst, src).into(),
1005            OperandSize::S64 => asm::inst::xorpd_a::new(dst, src).into(),
1006            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
1007        };
1008        self.emit(Inst::External { inst });
1009    }
1010
1011    /// Shift with register and register.
1012    pub fn shift_rr(&mut self, src: Reg, dst: WritableReg, kind: ShiftKind, size: OperandSize) {
1013        let dst = pair_gpr(dst);
1014        let src: Gpr = src.into();
1015        let inst = match (kind, size) {
1016            (ShiftKind::Shl, OperandSize::S32) => asm::inst::shll_mc::new(dst, src).into(),
1017            (ShiftKind::Shl, OperandSize::S64) => asm::inst::shlq_mc::new(dst, src).into(),
1018            (ShiftKind::Shl, _) => todo!(),
1019            (ShiftKind::ShrS, OperandSize::S32) => asm::inst::sarl_mc::new(dst, src).into(),
1020            (ShiftKind::ShrS, OperandSize::S64) => asm::inst::sarq_mc::new(dst, src).into(),
1021            (ShiftKind::ShrS, _) => todo!(),
1022            (ShiftKind::ShrU, OperandSize::S32) => asm::inst::shrl_mc::new(dst, src).into(),
1023            (ShiftKind::ShrU, OperandSize::S64) => asm::inst::shrq_mc::new(dst, src).into(),
1024            (ShiftKind::ShrU, _) => todo!(),
1025            (ShiftKind::Rotl, OperandSize::S32) => asm::inst::roll_mc::new(dst, src).into(),
1026            (ShiftKind::Rotl, OperandSize::S64) => asm::inst::rolq_mc::new(dst, src).into(),
1027            (ShiftKind::Rotl, _) => todo!(),
1028            (ShiftKind::Rotr, OperandSize::S32) => asm::inst::rorl_mc::new(dst, src).into(),
1029            (ShiftKind::Rotr, OperandSize::S64) => asm::inst::rorq_mc::new(dst, src).into(),
1030            (ShiftKind::Rotr, _) => todo!(),
1031        };
1032        self.emit(Inst::External { inst });
1033    }
1034
1035    /// Shift with immediate and register.
1036    pub fn shift_ir(&mut self, imm: u8, dst: WritableReg, kind: ShiftKind, size: OperandSize) {
1037        let dst = pair_gpr(dst);
1038        let inst = match (kind, size) {
1039            (ShiftKind::Shl, OperandSize::S32) => asm::inst::shll_mi::new(dst, imm).into(),
1040            (ShiftKind::Shl, OperandSize::S64) => asm::inst::shlq_mi::new(dst, imm).into(),
1041            (ShiftKind::Shl, _) => todo!(),
1042            (ShiftKind::ShrS, OperandSize::S32) => asm::inst::sarl_mi::new(dst, imm).into(),
1043            (ShiftKind::ShrS, OperandSize::S64) => asm::inst::sarq_mi::new(dst, imm).into(),
1044            (ShiftKind::ShrS, _) => todo!(),
1045            (ShiftKind::ShrU, OperandSize::S32) => asm::inst::shrl_mi::new(dst, imm).into(),
1046            (ShiftKind::ShrU, OperandSize::S64) => asm::inst::shrq_mi::new(dst, imm).into(),
1047            (ShiftKind::ShrU, _) => todo!(),
1048            (ShiftKind::Rotl, OperandSize::S32) => asm::inst::roll_mi::new(dst, imm).into(),
1049            (ShiftKind::Rotl, OperandSize::S64) => asm::inst::rolq_mi::new(dst, imm).into(),
1050            (ShiftKind::Rotl, _) => todo!(),
1051            (ShiftKind::Rotr, OperandSize::S32) => asm::inst::rorl_mi::new(dst, imm).into(),
1052            (ShiftKind::Rotr, OperandSize::S64) => asm::inst::rorq_mi::new(dst, imm).into(),
1053            (ShiftKind::Rotr, _) => todo!(),
1054        };
1055        self.emit(Inst::External { inst });
1056    }
1057
1058    /// Signed/unsigned division.
1059    ///
1060    /// Emits a sequence of instructions to ensure the correctness of
1061    /// the division invariants.  This function assumes that the
1062    /// caller has correctly allocated the dividend as `(rdx:rax)` and
1063    /// accounted for the quotient to be stored in `rax`.
1064    pub fn div(&mut self, divisor: Reg, dst: (Reg, Reg), kind: DivKind, size: OperandSize) {
1065        let trap = match kind {
1066            // Signed division has two trapping conditions, integer overflow and
1067            // divide-by-zero. Check for divide-by-zero explicitly and let the
1068            // hardware detect overflow.
1069            DivKind::Signed => {
1070                self.cmp_ir(divisor, 0, size);
1071                self.emit(Inst::TrapIf {
1072                    cc: CC::Z,
1073                    trap_code: TrapCode::INTEGER_DIVISION_BY_ZERO,
1074                });
1075
1076                // Sign-extend the dividend with tailor-made instructoins for
1077                // just this operation.
1078                let ext_dst: WritableGpr = dst.1.into();
1079                let ext_src: Gpr = dst.0.into();
1080                let inst = match size {
1081                    OperandSize::S32 => asm::inst::cltd_zo::new(ext_dst, ext_src).into(),
1082                    OperandSize::S64 => asm::inst::cqto_zo::new(ext_dst, ext_src).into(),
1083                    _ => unimplemented!(),
1084                };
1085                self.emit(Inst::External { inst });
1086                TrapCode::INTEGER_OVERFLOW
1087            }
1088
1089            // Unsigned division only traps in one case, on divide-by-zero, so
1090            // defer that to the trap opcode.
1091            //
1092            // The divisor_hi reg is initialized with zero through an
1093            // xor-against-itself op.
1094            DivKind::Unsigned => {
1095                self.xor_rr(dst.1, writable!(dst.1), size);
1096                TrapCode::INTEGER_DIVISION_BY_ZERO
1097            }
1098        };
1099        let dst0 = pair_gpr(writable!(dst.0));
1100        let dst1 = pair_gpr(writable!(dst.1));
1101        let inst = match (kind, size) {
1102            (DivKind::Signed, OperandSize::S32) => {
1103                asm::inst::idivl_m::new(dst0, dst1, divisor, trap).into()
1104            }
1105            (DivKind::Unsigned, OperandSize::S32) => {
1106                asm::inst::divl_m::new(dst0, dst1, divisor, trap).into()
1107            }
1108            (DivKind::Signed, OperandSize::S64) => {
1109                asm::inst::idivq_m::new(dst0, dst1, divisor, trap).into()
1110            }
1111            (DivKind::Unsigned, OperandSize::S64) => {
1112                asm::inst::divq_m::new(dst0, dst1, divisor, trap).into()
1113            }
1114            _ => todo!(),
1115        };
1116        self.emit(Inst::External { inst });
1117    }
1118
1119    /// Signed/unsigned remainder.
1120    ///
1121    /// Emits a sequence of instructions to ensure the correctness of the
1122    /// division invariants and ultimately calculate the remainder.
1123    /// This function assumes that the
1124    /// caller has correctly allocated the dividend as `(rdx:rax)` and
1125    /// accounted for the remainder to be stored in `rdx`.
1126    pub fn rem(&mut self, divisor: Reg, dst: (Reg, Reg), kind: RemKind, size: OperandSize) {
1127        match kind {
1128            // Signed remainder goes through a pseudo-instruction which has
1129            // some internal branching. The `dividend_hi`, or `rdx`, is
1130            // initialized here with a `SignExtendData` instruction.
1131            RemKind::Signed => {
1132                let ext_dst: WritableGpr = dst.1.into();
1133
1134                // Initialize `dividend_hi`, or `rdx`, with a tailor-made
1135                // instruction for this operation.
1136                let ext_src: Gpr = dst.0.into();
1137                let inst = match size {
1138                    OperandSize::S32 => asm::inst::cltd_zo::new(ext_dst, ext_src).into(),
1139                    OperandSize::S64 => asm::inst::cqto_zo::new(ext_dst, ext_src).into(),
1140                    _ => unimplemented!(),
1141                };
1142                self.emit(Inst::External { inst });
1143                self.emit(Inst::CheckedSRemSeq {
1144                    size: size.into(),
1145                    divisor: divisor.into(),
1146                    dividend_lo: dst.0.into(),
1147                    dividend_hi: dst.1.into(),
1148                    dst_quotient: dst.0.into(),
1149                    dst_remainder: dst.1.into(),
1150                });
1151            }
1152
1153            // Unsigned remainder initializes `dividend_hi` with zero and
1154            // then executes a normal `div` instruction.
1155            RemKind::Unsigned => {
1156                self.xor_rr(dst.1, writable!(dst.1), size);
1157                let dst0 = pair_gpr(writable!(dst.0));
1158                let dst1 = pair_gpr(writable!(dst.1));
1159                let trap = TrapCode::INTEGER_DIVISION_BY_ZERO;
1160                let inst = match size {
1161                    OperandSize::S32 => asm::inst::divl_m::new(dst0, dst1, divisor, trap).into(),
1162                    OperandSize::S64 => asm::inst::divq_m::new(dst0, dst1, divisor, trap).into(),
1163                    _ => todo!(),
1164                };
1165                self.emit(Inst::External { inst });
1166            }
1167        }
1168    }
1169
1170    /// Multiply immediate and register.
1171    pub fn mul_ir(&mut self, imm: i32, dst: WritableReg, size: OperandSize) {
1172        use OperandSize::*;
1173        let src = dst.to_reg();
1174        let dst: WritableGpr = dst.to_reg().into();
1175        let inst = match size {
1176            S16 => asm::inst::imulw_rmi::new(dst, src, u16::try_from(imm).unwrap()).into(),
1177            S32 => asm::inst::imull_rmi::new(dst, src, imm as u32).into(),
1178            S64 => asm::inst::imulq_rmi_sxl::new(dst, src, imm).into(),
1179            S8 | S128 => unimplemented!(),
1180        };
1181        self.emit(Inst::External { inst });
1182    }
1183
1184    /// Multiply register and register.
1185    pub fn mul_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1186        use OperandSize::*;
1187        let dst = pair_gpr(dst);
1188        let inst = match size {
1189            S16 => asm::inst::imulw_rm::new(dst, src).into(),
1190            S32 => asm::inst::imull_rm::new(dst, src).into(),
1191            S64 => asm::inst::imulq_rm::new(dst, src).into(),
1192            S8 | S128 => unimplemented!(),
1193        };
1194        self.emit(Inst::External { inst });
1195    }
1196
1197    /// Add immediate and register.
1198    pub fn add_ir(&mut self, imm: i32, dst: WritableReg, size: OperandSize) {
1199        let dst = pair_gpr(dst);
1200        let inst = match size {
1201            OperandSize::S8 => asm::inst::addb_mi::new(dst, u8::try_from(imm).unwrap()).into(),
1202            OperandSize::S16 => asm::inst::addw_mi::new(dst, u16::try_from(imm).unwrap()).into(),
1203            OperandSize::S32 => asm::inst::addl_mi::new(dst, imm as u32).into(),
1204            OperandSize::S64 => asm::inst::addq_mi_sxl::new(dst, imm).into(),
1205            OperandSize::S128 => unimplemented!(),
1206        };
1207        self.emit(Inst::External { inst });
1208    }
1209
1210    /// Add a sign-extended 8-bit immediate to a 64-bit register.
1211    pub fn add_ir8(&mut self, imm: i8, dst: WritableReg) {
1212        let inst = asm::inst::addq_mi_sxb::new(pair_gpr(dst), imm).into();
1213        self.emit(Inst::External { inst });
1214    }
1215
1216    /// Add register and register.
1217    pub fn add_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1218        let dst = pair_gpr(dst);
1219        let inst = match size {
1220            OperandSize::S8 => asm::inst::addb_rm::new(dst, src).into(),
1221            OperandSize::S16 => asm::inst::addw_rm::new(dst, src).into(),
1222            OperandSize::S32 => asm::inst::addl_rm::new(dst, src).into(),
1223            OperandSize::S64 => asm::inst::addq_rm::new(dst, src).into(),
1224            OperandSize::S128 => unimplemented!(),
1225        };
1226        self.emit(Inst::External { inst });
1227    }
1228
1229    pub fn lock_xadd(
1230        &mut self,
1231        addr: Address,
1232        dst: WritableReg,
1233        size: OperandSize,
1234        flags: MemFlagsData,
1235    ) {
1236        assert!(addr.is_offset());
1237        let mem = Self::to_synthetic_amode(&addr, flags);
1238        let dst = pair_gpr(dst);
1239        let inst = match size {
1240            OperandSize::S8 => asm::inst::lock_xaddb_mr::new(mem, dst).into(),
1241            OperandSize::S16 => asm::inst::lock_xaddw_mr::new(mem, dst).into(),
1242            OperandSize::S32 => asm::inst::lock_xaddl_mr::new(mem, dst).into(),
1243            OperandSize::S64 => asm::inst::lock_xaddq_mr::new(mem, dst).into(),
1244            OperandSize::S128 => unimplemented!(),
1245        };
1246
1247        self.emit(Inst::External { inst });
1248    }
1249
1250    pub fn atomic_rmw_seq(
1251        &mut self,
1252        addr: Address,
1253        operand: Reg,
1254        dst: WritableReg,
1255        temp: WritableReg,
1256        size: OperandSize,
1257        flags: MemFlagsData,
1258        op: AtomicRmwSeqOp,
1259    ) {
1260        assert!(addr.is_offset());
1261        let mem = Self::to_synthetic_amode(&addr, flags);
1262        self.emit(Inst::AtomicRmwSeq {
1263            ty: Type::int_with_byte_size(size.bytes() as _).unwrap(),
1264            mem,
1265            operand: operand.into(),
1266            temp: temp.map(Into::into),
1267            dst_old: dst.map(Into::into),
1268            op,
1269        });
1270    }
1271
1272    pub fn xchg(
1273        &mut self,
1274        addr: Address,
1275        dst: WritableReg,
1276        size: OperandSize,
1277        flags: MemFlagsData,
1278    ) {
1279        assert!(addr.is_offset());
1280        let mem = Self::to_synthetic_amode(&addr, flags);
1281        let dst = pair_gpr(dst);
1282        let inst = match size {
1283            OperandSize::S8 => asm::inst::xchgb_rm::new(dst, mem).into(),
1284            OperandSize::S16 => asm::inst::xchgw_rm::new(dst, mem).into(),
1285            OperandSize::S32 => asm::inst::xchgl_rm::new(dst, mem).into(),
1286            OperandSize::S64 => asm::inst::xchgq_rm::new(dst, mem).into(),
1287            OperandSize::S128 => unimplemented!(),
1288        };
1289
1290        self.emit(Inst::External { inst });
1291    }
1292    pub fn cmpxchg(
1293        &mut self,
1294        addr: Address,
1295        replacement: Reg,
1296        dst: WritableReg,
1297        size: OperandSize,
1298        flags: MemFlagsData,
1299    ) {
1300        assert!(addr.is_offset());
1301        let mem = Self::to_synthetic_amode(&addr, flags);
1302        let dst = pair_gpr(dst);
1303        let inst = match size {
1304            OperandSize::S8 => asm::inst::lock_cmpxchgb_mr::new(mem, replacement, dst).into(),
1305            OperandSize::S16 => asm::inst::lock_cmpxchgw_mr::new(mem, replacement, dst).into(),
1306            OperandSize::S32 => asm::inst::lock_cmpxchgl_mr::new(mem, replacement, dst).into(),
1307            OperandSize::S64 => asm::inst::lock_cmpxchgq_mr::new(mem, replacement, dst).into(),
1308            OperandSize::S128 => unimplemented!(),
1309        };
1310
1311        self.emit(Inst::External { inst });
1312    }
1313
1314    pub fn cmp_ir(&mut self, src1: Reg, imm: i32, size: OperandSize) {
1315        let inst = match size {
1316            OperandSize::S8 => {
1317                let imm = i8::try_from(imm).unwrap();
1318                asm::inst::cmpb_mi::new(src1, imm.cast_unsigned()).into()
1319            }
1320            OperandSize::S16 => match i8::try_from(imm) {
1321                Ok(imm8) => asm::inst::cmpw_mi_sxb::new(src1, imm8).into(),
1322                Err(_) => {
1323                    asm::inst::cmpw_mi::new(src1, i16::try_from(imm).unwrap().cast_unsigned())
1324                        .into()
1325                }
1326            },
1327            OperandSize::S32 => match i8::try_from(imm) {
1328                Ok(imm8) => asm::inst::cmpl_mi_sxb::new(src1, imm8).into(),
1329                Err(_) => asm::inst::cmpl_mi::new(src1, imm.cast_unsigned()).into(),
1330            },
1331            OperandSize::S64 => match i8::try_from(imm) {
1332                Ok(imm8) => asm::inst::cmpq_mi_sxb::new(src1, imm8).into(),
1333                Err(_) => asm::inst::cmpq_mi::new(src1, imm).into(),
1334            },
1335            OperandSize::S128 => unimplemented!(),
1336        };
1337
1338        self.emit(Inst::External { inst });
1339    }
1340
1341    pub fn cmp_rr(&mut self, src1: Reg, src2: Reg, size: OperandSize) {
1342        let inst = match size {
1343            OperandSize::S8 => asm::inst::cmpb_rm::new(src1, src2).into(),
1344            OperandSize::S16 => asm::inst::cmpw_rm::new(src1, src2).into(),
1345            OperandSize::S32 => asm::inst::cmpl_rm::new(src1, src2).into(),
1346            OperandSize::S64 => asm::inst::cmpq_rm::new(src1, src2).into(),
1347            OperandSize::S128 => unimplemented!(),
1348        };
1349
1350        self.emit(Inst::External { inst });
1351    }
1352
1353    /// Compares values in src1 and src2 and sets ZF, PF, and CF flags in EFLAGS
1354    /// register.
1355    pub fn ucomis(&mut self, src1: Reg, src2: Reg, size: OperandSize) {
1356        let inst = match size {
1357            OperandSize::S32 => asm::inst::ucomiss_a::new(src1, src2).into(),
1358            OperandSize::S64 => asm::inst::ucomisd_a::new(src1, src2).into(),
1359            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
1360        };
1361        self.emit(Inst::External { inst });
1362    }
1363
1364    pub fn popcnt(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1365        assert!(
1366            self.isa_flags.has_popcnt() && self.isa_flags.has_sse42(),
1367            "Requires has_popcnt and has_sse42 flags"
1368        );
1369        let dst = WritableGpr::from_reg(dst.to_reg().into());
1370        let inst = match size {
1371            OperandSize::S16 => asm::inst::popcntw_rm::new(dst, src).into(),
1372            OperandSize::S32 => asm::inst::popcntl_rm::new(dst, src).into(),
1373            OperandSize::S64 => asm::inst::popcntq_rm::new(dst, src).into(),
1374            OperandSize::S8 | OperandSize::S128 => unreachable!(),
1375        };
1376        self.emit(Inst::External { inst });
1377    }
1378
1379    /// Emit a test instruction with two register operands.
1380    pub fn test_rr(&mut self, src1: Reg, src2: Reg, size: OperandSize) {
1381        let inst = match size {
1382            OperandSize::S8 => asm::inst::testb_mr::new(src1, src2).into(),
1383            OperandSize::S16 => asm::inst::testw_mr::new(src1, src2).into(),
1384            OperandSize::S32 => asm::inst::testl_mr::new(src1, src2).into(),
1385            OperandSize::S64 => asm::inst::testq_mr::new(src1, src2).into(),
1386            OperandSize::S128 => unimplemented!(),
1387        };
1388
1389        self.emit(Inst::External { inst });
1390    }
1391
1392    /// Set value in dst to `0` or `1` based on flags in status register and
1393    /// [`CmpKind`].
1394    pub fn setcc(&mut self, kind: IntCmpKind, dst: WritableReg) {
1395        self.setcc_impl(kind.into(), dst);
1396    }
1397
1398    /// Set value in dst to `1` if parity flag in status register is set, `0`
1399    /// otherwise.
1400    pub fn setp(&mut self, dst: WritableReg) {
1401        self.setcc_impl(CC::P, dst);
1402    }
1403
1404    /// Set value in dst to `1` if parity flag in status register is not set,
1405    /// `0` otherwise.
1406    pub fn setnp(&mut self, dst: WritableReg) {
1407        self.setcc_impl(CC::NP, dst);
1408    }
1409
1410    fn setcc_impl(&mut self, cc: CC, dst: WritableReg) {
1411        // Clear the dst register or bits 1 to 31 may be incorrectly set.
1412        // Don't use xor since it updates the status register.
1413        let dst: WritableGpr = dst.map(Into::into);
1414        let inst = asm::inst::movl_oi::new(dst, 0).into();
1415        self.emit(Inst::External { inst });
1416
1417        // Copy correct bit from status register into dst register.
1418        //
1419        // Note that some of these mnemonics don't match exactly and that's
1420        // intentional as there are multiple mnemonics for the same encoding in
1421        // some cases and the assembler picked ones that match Capstone rather
1422        // than Cranelift.
1423        let inst = match cc {
1424            CC::O => asm::inst::seto_m::new(dst).into(),
1425            CC::NO => asm::inst::setno_m::new(dst).into(),
1426            CC::B => asm::inst::setb_m::new(dst).into(),
1427            CC::NB => asm::inst::setae_m::new(dst).into(), //  nb == ae
1428            CC::Z => asm::inst::sete_m::new(dst).into(),   //   z ==  e
1429            CC::NZ => asm::inst::setne_m::new(dst).into(), //  nz == ne
1430            CC::BE => asm::inst::setbe_m::new(dst).into(),
1431            CC::NBE => asm::inst::seta_m::new(dst).into(), // nbe ==  a
1432            CC::S => asm::inst::sets_m::new(dst).into(),
1433            CC::NS => asm::inst::setns_m::new(dst).into(),
1434            CC::L => asm::inst::setl_m::new(dst).into(),
1435            CC::NL => asm::inst::setge_m::new(dst).into(), //  nl == ge
1436            CC::LE => asm::inst::setle_m::new(dst).into(),
1437            CC::NLE => asm::inst::setg_m::new(dst).into(), // nle ==  g
1438            CC::P => asm::inst::setp_m::new(dst).into(),
1439            CC::NP => asm::inst::setnp_m::new(dst).into(),
1440        };
1441        self.emit(Inst::External { inst });
1442    }
1443
1444    /// Store the count of leading zeroes in src in dst.
1445    /// Requires `has_lzcnt` flag.
1446    pub fn lzcnt(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1447        assert!(self.isa_flags.has_lzcnt(), "Requires has_lzcnt flag");
1448        let dst = WritableGpr::from_reg(dst.to_reg().into());
1449        let inst = match size {
1450            OperandSize::S16 => asm::inst::lzcntw_rm::new(dst, src).into(),
1451            OperandSize::S32 => asm::inst::lzcntl_rm::new(dst, src).into(),
1452            OperandSize::S64 => asm::inst::lzcntq_rm::new(dst, src).into(),
1453            OperandSize::S8 | OperandSize::S128 => unreachable!(),
1454        };
1455        self.emit(Inst::External { inst });
1456    }
1457
1458    /// Store the count of trailing zeroes in src in dst.
1459    /// Requires `has_bmi1` flag.
1460    pub fn tzcnt(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1461        assert!(self.isa_flags.has_bmi1(), "Requires has_bmi1 flag");
1462        let dst = WritableGpr::from_reg(dst.to_reg().into());
1463        let inst = match size {
1464            OperandSize::S16 => asm::inst::tzcntw_a::new(dst, src).into(),
1465            OperandSize::S32 => asm::inst::tzcntl_a::new(dst, src).into(),
1466            OperandSize::S64 => asm::inst::tzcntq_a::new(dst, src).into(),
1467            OperandSize::S8 | OperandSize::S128 => unreachable!(),
1468        };
1469        self.emit(Inst::External { inst });
1470    }
1471
1472    /// Stores position of the most significant bit set in src in dst.
1473    /// Zero flag is set if src is equal to 0.
1474    pub fn bsr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1475        let dst: WritableGpr = WritableGpr::from_reg(dst.to_reg().into());
1476        let inst = match size {
1477            OperandSize::S16 => asm::inst::bsrw_rm::new(dst, src).into(),
1478            OperandSize::S32 => asm::inst::bsrl_rm::new(dst, src).into(),
1479            OperandSize::S64 => asm::inst::bsrq_rm::new(dst, src).into(),
1480            OperandSize::S8 | OperandSize::S128 => unreachable!(),
1481        };
1482        self.emit(Inst::External { inst });
1483    }
1484
1485    /// Performs integer negation on `src` and places result in `dst`.
1486    pub fn neg(&mut self, read: Reg, write: WritableReg, size: OperandSize) {
1487        let gpr = PairedGpr {
1488            read: read.into(),
1489            write: WritableGpr::from_reg(write.to_reg().into()),
1490        };
1491        let inst = match size {
1492            OperandSize::S8 => asm::inst::negb_m::new(gpr).into(),
1493            OperandSize::S16 => asm::inst::negw_m::new(gpr).into(),
1494            OperandSize::S32 => asm::inst::negl_m::new(gpr).into(),
1495            OperandSize::S64 => asm::inst::negq_m::new(gpr).into(),
1496            OperandSize::S128 => unreachable!(),
1497        };
1498        self.emit(Inst::External { inst });
1499    }
1500
1501    /// Stores position of the least significant bit set in src in dst.
1502    /// Zero flag is set if src is equal to 0.
1503    pub fn bsf(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1504        let dst: WritableGpr = WritableGpr::from_reg(dst.to_reg().into());
1505        let inst = match size {
1506            OperandSize::S16 => asm::inst::bsfw_rm::new(dst, src).into(),
1507            OperandSize::S32 => asm::inst::bsfl_rm::new(dst, src).into(),
1508            OperandSize::S64 => asm::inst::bsfq_rm::new(dst, src).into(),
1509            OperandSize::S8 | OperandSize::S128 => unreachable!(),
1510        };
1511        self.emit(Inst::External { inst });
1512    }
1513
1514    /// Performs float addition on src and dst and places result in dst.
1515    pub fn xmm_add_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1516        let dst = pair_xmm(dst);
1517        let inst = match size {
1518            OperandSize::S32 => asm::inst::addss_a::new(dst, src).into(),
1519            OperandSize::S64 => asm::inst::addsd_a::new(dst, src).into(),
1520            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
1521        };
1522        self.emit(Inst::External { inst });
1523    }
1524
1525    /// Performs float subtraction on src and dst and places result in dst.
1526    pub fn xmm_sub_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1527        let dst = pair_xmm(dst);
1528        let inst = match size {
1529            OperandSize::S32 => asm::inst::subss_a::new(dst, src).into(),
1530            OperandSize::S64 => asm::inst::subsd_a::new(dst, src).into(),
1531            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
1532        };
1533        self.emit(Inst::External { inst });
1534    }
1535
1536    /// Performs float multiplication on src and dst and places result in dst.
1537    pub fn xmm_mul_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1538        use OperandSize::*;
1539        let dst = pair_xmm(dst);
1540        let inst = match size {
1541            S32 => asm::inst::mulss_a::new(dst, src).into(),
1542            S64 => asm::inst::mulsd_a::new(dst, src).into(),
1543            S8 | S16 | S128 => unreachable!(),
1544        };
1545        self.emit(Inst::External { inst });
1546    }
1547
1548    /// Performs float division on src and dst and places result in dst.
1549    pub fn xmm_div_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1550        let dst = pair_xmm(dst);
1551        let inst = match size {
1552            OperandSize::S32 => asm::inst::divss_a::new(dst, src).into(),
1553            OperandSize::S64 => asm::inst::divsd_a::new(dst, src).into(),
1554            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
1555        };
1556        self.emit(Inst::External { inst });
1557    }
1558
1559    /// Minimum for src and dst XMM registers with results put in dst.
1560    pub fn xmm_min_seq(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1561        self.emit(Inst::XmmMinMaxSeq {
1562            size: size.into(),
1563            is_min: true,
1564            lhs: src.into(),
1565            rhs: dst.to_reg().into(),
1566            dst: dst.map(Into::into),
1567        });
1568    }
1569
1570    /// Maximum for src and dst XMM registers with results put in dst.
1571    pub fn xmm_max_seq(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1572        self.emit(Inst::XmmMinMaxSeq {
1573            size: size.into(),
1574            is_min: false,
1575            lhs: src.into(),
1576            rhs: dst.to_reg().into(),
1577            dst: dst.map(Into::into),
1578        });
1579    }
1580
1581    /// Perform rounding operation on float register src and place results in
1582    /// float register dst.
1583    pub fn xmm_rounds_rr(
1584        &mut self,
1585        src: Reg,
1586        dst: WritableReg,
1587        mode: RoundingMode,
1588        size: OperandSize,
1589    ) {
1590        let dst = dst.map(|r| r.into());
1591
1592        let imm: u8 = match mode {
1593            RoundingMode::Nearest => 0x00,
1594            RoundingMode::Down => 0x01,
1595            RoundingMode::Up => 0x02,
1596            RoundingMode::Zero => 0x03,
1597        };
1598
1599        let inst = match size {
1600            OperandSize::S32 => asm::inst::roundss_rmi::new(dst, src, imm).into(),
1601            OperandSize::S64 => asm::inst::roundsd_rmi::new(dst, src, imm).into(),
1602            OperandSize::S8 | OperandSize::S16 | OperandSize::S128 => unreachable!(),
1603        };
1604
1605        self.emit(Inst::External { inst });
1606    }
1607
1608    pub fn sqrt(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1609        use OperandSize::*;
1610        let dst = pair_xmm(dst);
1611        let inst = match size {
1612            S32 => asm::inst::sqrtss_a::new(dst, src).into(),
1613            S64 => asm::inst::sqrtsd_a::new(dst, src).into(),
1614            S8 | S16 | S128 => unimplemented!(),
1615        };
1616        self.emit(Inst::External { inst });
1617    }
1618
1619    /// Emit a call to an unknown location through a register.
1620    pub fn call_with_reg(&mut self, cc: CallingConvention, callee: Reg) {
1621        self.emit(Inst::CallUnknown {
1622            info: Box::new(CallInfo::empty(RegMem::reg(callee.into()), cc.into())),
1623        });
1624    }
1625
1626    /// Emit a call to a locally defined function through an index.
1627    pub fn call_with_name(&mut self, cc: CallingConvention, name: UserExternalNameRef) {
1628        self.emit(Inst::CallKnown {
1629            info: Box::new(CallInfo::empty(ExternalName::user(name), cc.into())),
1630        });
1631    }
1632
1633    /// Emit a tail jump to an unknown location through a register.
1634    pub fn tail_jump_with_reg(&mut self, callee: Reg) {
1635        let callee = Gpr::unwrap_new(callee.into());
1636        let inst = asm::inst::jmpq_m::new(callee).into();
1637        self.emit(Inst::External { inst });
1638        self.buffer.add_call_site();
1639    }
1640
1641    /// Emit a tail jump to a locally defined function through an index.
1642    pub fn tail_jump_with_name(&mut self, name: UserExternalNameRef) {
1643        let inst = asm::inst::jmp_d32::new(0).into();
1644        self.emit(Inst::External { inst });
1645
1646        let offset = self.buffer.cur_offset();
1647        self.buffer.add_reloc_at_offset(
1648            offset - 4,
1649            Reloc::X86CallPCRel4,
1650            &ExternalName::user(name),
1651            -4,
1652        );
1653        self.buffer.add_call_site();
1654    }
1655
1656    /// Emits a conditional jump to the given label.
1657    pub fn jmp_if(&mut self, cc: impl Into<CC>, taken: MachLabel) {
1658        self.emit(Inst::WinchJmpIf {
1659            cc: cc.into(),
1660            taken,
1661        });
1662    }
1663
1664    /// Performs an unconditional jump to the given label.
1665    pub fn jmp(&mut self, target: MachLabel) {
1666        self.emit(Inst::JmpKnown { dst: target });
1667    }
1668
1669    /// Emits a jump table sequence.
1670    pub fn jmp_table(
1671        &mut self,
1672        targets: SmallVec<[MachLabel; 4]>,
1673        default: MachLabel,
1674        index: Reg,
1675        tmp1: Reg,
1676        tmp2: Reg,
1677    ) {
1678        self.emit(Inst::JmpTableSeq {
1679            idx: index.into(),
1680            tmp1: Writable::from_reg(tmp1.into()),
1681            tmp2: Writable::from_reg(tmp2.into()),
1682            default_target: default,
1683            targets: Box::new(targets.to_vec()),
1684        })
1685    }
1686
1687    /// Emit a trap instruction.
1688    pub fn trap(&mut self, code: TrapCode) {
1689        let inst = asm::inst::ud2_zo::new(code).into();
1690        self.emit(Inst::External { inst });
1691    }
1692
1693    /// Conditional trap.
1694    pub fn trapif(&mut self, cc: impl Into<CC>, trap_code: TrapCode) {
1695        self.emit(Inst::TrapIf {
1696            cc: cc.into(),
1697            trap_code,
1698        });
1699    }
1700
1701    /// Load effective address.
1702    pub fn lea(&mut self, addr: &Address, dst: WritableReg, size: OperandSize) {
1703        let addr = Self::to_synthetic_amode(addr, MemFlagsData::trusted());
1704        let dst: WritableGpr = dst.map(Into::into);
1705        let inst = match size {
1706            OperandSize::S16 => asm::inst::leaw_rm::new(dst, addr).into(),
1707            OperandSize::S32 => asm::inst::leal_rm::new(dst, addr).into(),
1708            OperandSize::S64 => asm::inst::leaq_rm::new(dst, addr).into(),
1709            OperandSize::S8 | OperandSize::S128 => unimplemented!(),
1710        };
1711        self.emit(Inst::External { inst });
1712    }
1713
1714    pub fn adc_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1715        let dst = pair_gpr(dst);
1716        let inst = match size {
1717            OperandSize::S8 => asm::inst::adcb_rm::new(dst, src).into(),
1718            OperandSize::S16 => asm::inst::adcw_rm::new(dst, src).into(),
1719            OperandSize::S32 => asm::inst::adcl_rm::new(dst, src).into(),
1720            OperandSize::S64 => asm::inst::adcq_rm::new(dst, src).into(),
1721            OperandSize::S128 => unimplemented!(),
1722        };
1723        self.emit(Inst::External { inst });
1724    }
1725
1726    pub fn sbb_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
1727        let dst = pair_gpr(dst);
1728        let inst = match size {
1729            OperandSize::S8 => asm::inst::sbbb_rm::new(dst, src).into(),
1730            OperandSize::S16 => asm::inst::sbbw_rm::new(dst, src).into(),
1731            OperandSize::S32 => asm::inst::sbbl_rm::new(dst, src).into(),
1732            OperandSize::S64 => asm::inst::sbbq_rm::new(dst, src).into(),
1733            OperandSize::S128 => unimplemented!(),
1734        };
1735        self.emit(Inst::External { inst });
1736    }
1737
1738    pub fn mul_wide(
1739        &mut self,
1740        dst_lo: WritableReg,
1741        dst_hi: WritableReg,
1742        lhs: Reg,
1743        rhs: Reg,
1744        kind: MulWideKind,
1745        size: OperandSize,
1746    ) {
1747        use MulWideKind::*;
1748        use OperandSize::*;
1749        let rax = asm::Fixed(PairedGpr {
1750            read: lhs.into(),
1751            write: WritableGpr::from_reg(dst_lo.to_reg().into()),
1752        });
1753        let rdx = asm::Fixed(dst_hi.to_reg().into());
1754        if size == S8 {
1755            // For `mulb` and `imulb`, both the high and low bits are written to
1756            // RAX.
1757            assert_eq!(dst_lo, dst_hi);
1758        }
1759        let inst = match (size, kind) {
1760            (S8, Unsigned) => asm::inst::mulb_m::new(rax, rhs).into(),
1761            (S8, Signed) => asm::inst::imulb_m::new(rax, rhs).into(),
1762            (S16, Unsigned) => asm::inst::mulw_m::new(rax, rdx, rhs).into(),
1763            (S16, Signed) => asm::inst::imulw_m::new(rax, rdx, rhs).into(),
1764            (S32, Unsigned) => asm::inst::mull_m::new(rax, rdx, rhs).into(),
1765            (S32, Signed) => asm::inst::imull_m::new(rax, rdx, rhs).into(),
1766            (S64, Unsigned) => asm::inst::mulq_m::new(rax, rdx, rhs).into(),
1767            (S64, Signed) => asm::inst::imulq_m::new(rax, rdx, rhs).into(),
1768            (S128, _) => unimplemented!(),
1769        };
1770        self.emit(Inst::External { inst });
1771    }
1772
1773    /// Shuffles bytes in `src` according to contents of `mask` and puts
1774    /// result in `dst`.
1775    pub fn xmm_vpshufb_rrm(&mut self, dst: WritableReg, src: Reg, mask: &Address) {
1776        let dst: WritableXmm = dst.map(|r| r.into());
1777        let mask = Self::to_synthetic_amode(mask, MemFlagsData::trusted());
1778        let inst = asm::inst::vpshufb_b::new(dst, src, mask).into();
1779        self.emit(Inst::External { inst });
1780    }
1781
1782    /// Shuffles bytes in `src` according to contents of `mask` and puts
1783    /// result in `dst`.
1784    pub fn xmm_vpshufb_rrr(&mut self, dst: WritableReg, src: Reg, mask: Reg) {
1785        let dst: WritableXmm = dst.map(|r| r.into());
1786        let inst = asm::inst::vpshufb_b::new(dst, src, mask).into();
1787        self.emit(Inst::External { inst });
1788    }
1789
1790    /// Add unsigned integers with unsigned saturation.
1791    ///
1792    /// Adds the src operands but when an individual byte result is larger than
1793    /// an unsigned byte integer, 0xFF is written instead.
1794    pub fn xmm_vpaddus_rrm(
1795        &mut self,
1796        dst: WritableReg,
1797        src1: Reg,
1798        src2: &Address,
1799        size: OperandSize,
1800    ) {
1801        let dst: WritableXmm = dst.map(|r| r.into());
1802        let src2 = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
1803        let inst = match size {
1804            OperandSize::S8 => asm::inst::vpaddusb_b::new(dst, src1, src2).into(),
1805            OperandSize::S32 => asm::inst::vpaddusw_b::new(dst, src1, src2).into(),
1806            _ => unimplemented!(),
1807        };
1808        self.emit(Inst::External { inst });
1809    }
1810
1811    /// Add unsigned integers with unsigned saturation.
1812    ///
1813    /// Adds the src operands but when an individual byte result is larger than
1814    /// an unsigned byte integer, 0xFF is written instead.
1815    pub fn xmm_vpaddus_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg, size: OperandSize) {
1816        let dst: WritableXmm = dst.map(|r| r.into());
1817        let inst = match size {
1818            OperandSize::S8 => asm::inst::vpaddusb_b::new(dst, src1, src2).into(),
1819            OperandSize::S16 => asm::inst::vpaddusw_b::new(dst, src1, src2).into(),
1820            _ => unimplemented!(),
1821        };
1822        self.emit(Inst::External { inst });
1823    }
1824
1825    /// Add signed integers.
1826    pub fn xmm_vpadds_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg, size: OperandSize) {
1827        let dst: WritableXmm = dst.map(|r| r.into());
1828        let inst = match size {
1829            OperandSize::S8 => asm::inst::vpaddsb_b::new(dst, src1, src2).into(),
1830            OperandSize::S16 => asm::inst::vpaddsw_b::new(dst, src1, src2).into(),
1831            _ => unimplemented!(),
1832        };
1833        self.emit(Inst::External { inst });
1834    }
1835
1836    pub fn xmm_vpadd_rmr(
1837        &mut self,
1838        src1: Reg,
1839        src2: &Address,
1840        dst: WritableReg,
1841        size: OperandSize,
1842    ) {
1843        let dst: WritableXmm = dst.map(|r| r.into());
1844        let address = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
1845        let inst = match size {
1846            OperandSize::S8 => asm::inst::vpaddb_b::new(dst, src1, address).into(),
1847            OperandSize::S16 => asm::inst::vpaddw_b::new(dst, src1, address).into(),
1848            OperandSize::S32 => asm::inst::vpaddd_b::new(dst, src1, address).into(),
1849            _ => unimplemented!(),
1850        };
1851        self.emit(Inst::External { inst });
1852    }
1853
1854    /// Adds vectors of integers in `src1` and `src2` and puts the results in
1855    /// `dst`.
1856    pub fn xmm_vpadd_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
1857        let dst: WritableXmm = dst.map(|r| r.into());
1858        let inst = match size {
1859            OperandSize::S8 => asm::inst::vpaddb_b::new(dst, src1, src2).into(),
1860            OperandSize::S16 => asm::inst::vpaddw_b::new(dst, src1, src2).into(),
1861            OperandSize::S32 => asm::inst::vpaddd_b::new(dst, src1, src2).into(),
1862            OperandSize::S64 => asm::inst::vpaddq_b::new(dst, src1, src2).into(),
1863            _ => unimplemented!(),
1864        };
1865        self.emit(Inst::External { inst });
1866    }
1867
1868    pub fn mfence(&mut self) {
1869        self.emit(Inst::External {
1870            inst: asm::inst::mfence_zo::new().into(),
1871        });
1872    }
1873
1874    /// Extract a value from `src` into `addr` determined by `lane`.
1875    pub(crate) fn xmm_vpextr_rm(
1876        &mut self,
1877        addr: &Address,
1878        src: Reg,
1879        lane: u8,
1880        size: OperandSize,
1881        flags: MemFlagsData,
1882    ) {
1883        assert!(addr.is_offset());
1884        let dst = Self::to_synthetic_amode(addr, flags);
1885        let inst = match size {
1886            OperandSize::S8 => asm::inst::vpextrb_a::new(dst, src, lane).into(),
1887            OperandSize::S16 => asm::inst::vpextrw_b::new(dst, src, lane).into(),
1888            OperandSize::S32 => asm::inst::vpextrd_a::new(dst, src, lane).into(),
1889            OperandSize::S64 => asm::inst::vpextrq_a::new(dst, src, lane).into(),
1890            _ => unimplemented!(),
1891        };
1892        self.emit(Inst::External { inst });
1893    }
1894
1895    /// Extract a value from `src` into `dst` (zero extended) determined by `lane`.
1896    pub fn xmm_vpextr_rr(&mut self, dst: WritableReg, src: Reg, lane: u8, size: OperandSize) {
1897        let dst: WritableGpr = dst.map(|r| r.into());
1898        let inst = match size {
1899            OperandSize::S8 => asm::inst::vpextrb_a::new(dst, src, lane).into(),
1900            OperandSize::S16 => asm::inst::vpextrw_a::new(dst, src, lane).into(),
1901            OperandSize::S32 => asm::inst::vpextrd_a::new(dst, src, lane).into(),
1902            OperandSize::S64 => asm::inst::vpextrq_a::new(dst, src, lane).into(),
1903            _ => unimplemented!(),
1904        };
1905        self.emit(Inst::External { inst });
1906    }
1907
1908    /// Copy value from `src2`, merge into `src1`, and put result in `dst` at
1909    /// the location specified in `count`.
1910    pub fn xmm_vpinsr_rrm(
1911        &mut self,
1912        dst: WritableReg,
1913        src1: Reg,
1914        src2: &Address,
1915        count: u8,
1916        size: OperandSize,
1917    ) {
1918        let src2 = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
1919        let dst: WritableXmm = dst.map(|r| r.into());
1920
1921        let inst = match size {
1922            OperandSize::S8 => asm::inst::vpinsrb_b::new(dst, src1, src2, count).into(),
1923            OperandSize::S16 => asm::inst::vpinsrw_b::new(dst, src1, src2, count).into(),
1924            OperandSize::S32 => asm::inst::vpinsrd_b::new(dst, src1, src2, count).into(),
1925            OperandSize::S64 => asm::inst::vpinsrq_b::new(dst, src1, src2, count).into(),
1926            OperandSize::S128 => unreachable!(),
1927        };
1928        self.emit(Inst::External { inst });
1929    }
1930
1931    /// Copy value from `src2`, merge into `src1`, and put result in `dst` at
1932    /// the location specified in `count`.
1933    pub fn xmm_vpinsr_rrr(
1934        &mut self,
1935        dst: WritableReg,
1936        src1: Reg,
1937        src2: Reg,
1938        count: u8,
1939        size: OperandSize,
1940    ) {
1941        let dst: WritableXmm = dst.map(|r| r.into());
1942        let inst = match size {
1943            OperandSize::S8 => asm::inst::vpinsrb_b::new(dst, src1, src2, count).into(),
1944            OperandSize::S16 => asm::inst::vpinsrw_b::new(dst, src1, src2, count).into(),
1945            OperandSize::S32 => asm::inst::vpinsrd_b::new(dst, src1, src2, count).into(),
1946            OperandSize::S64 => asm::inst::vpinsrq_b::new(dst, src1, src2, count).into(),
1947            OperandSize::S128 => unreachable!(),
1948        };
1949        self.emit(Inst::External { inst });
1950    }
1951
1952    /// Copy a 32-bit float in `src2`, merge into `src1`, and put result in `dst`.
1953    pub fn xmm_vinsertps_rrm(&mut self, dst: WritableReg, src1: Reg, address: &Address, imm: u8) {
1954        let dst: WritableXmm = dst.map(|r| r.into());
1955        let address = Self::to_synthetic_amode(address, MemFlagsData::trusted());
1956        let inst = asm::inst::vinsertps_b::new(dst, src1, address, imm).into();
1957        self.emit(Inst::External { inst });
1958    }
1959
1960    /// Copy a 32-bit float in `src2`, merge into `src1`, and put result in `dst`.
1961    pub fn xmm_vinsertps_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg, imm: u8) {
1962        let dst: WritableXmm = dst.map(|r| r.into());
1963        let inst = asm::inst::vinsertps_b::new(dst, src1, src2, imm).into();
1964        self.emit(Inst::External { inst });
1965    }
1966
1967    /// Moves lower 64-bit float from `src2` into lower 64-bits of `dst` and the
1968    /// upper 64-bits in `src1` into the upper 64-bits of `dst`.
1969    pub fn xmm_vmovsd_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg) {
1970        let dst: WritableXmm = dst.map(|r| r.into());
1971        let inst = asm::inst::vmovsd_b::new(dst, src1, src2).into();
1972        self.emit(Inst::External { inst });
1973    }
1974
1975    /// Moves 64-bit float from `src` into lower 64-bits of `dst`.
1976    /// Zeroes out the upper 64 bits of `dst`.
1977    pub fn xmm_vmovsd_rm(&mut self, dst: WritableReg, src: &Address) {
1978        let src = Self::to_synthetic_amode(src, MemFlagsData::trusted());
1979        let dst: WritableXmm = dst.map(|r| r.into());
1980        let inst = asm::inst::vmovsd_d::new(dst, src).into();
1981        self.emit(Inst::External { inst });
1982    }
1983
1984    /// Moves two 32-bit floats from `src2` to the upper 64-bits of `dst`.
1985    /// Copies two 32-bit floats from the lower 64-bits of `src1` to lower
1986    /// 64-bits of `dst`.
1987    pub fn xmm_vmovlhps_rrm(&mut self, dst: WritableReg, src1: Reg, src2: &Address) {
1988        let src2 = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
1989        let dst: WritableXmm = dst.map(|r| r.into());
1990        let inst = asm::inst::vmovhps_b::new(dst, src1, src2).into();
1991        self.emit(Inst::External { inst });
1992    }
1993
1994    /// Moves two 32-bit floats from the lower 64-bits of `src2` to the upper
1995    /// 64-bits of `dst`. Copies two 32-bit floats from the lower 64-bits of
1996    /// `src1` to lower 64-bits of `dst`.
1997    pub fn xmm_vmovlhps_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg) {
1998        let dst: WritableXmm = dst.map(|r| r.into());
1999        let inst = asm::inst::vmovlhps_rvm::new(dst, src1, src2).into();
2000        self.emit(Inst::External { inst });
2001    }
2002
2003    /// Move unaligned packed integer values from address `src` to `dst`.
2004    pub fn xmm_vmovdqu_mr(&mut self, src: &Address, dst: WritableReg, flags: MemFlagsData) {
2005        let src = Self::to_synthetic_amode(src, flags);
2006        let dst: WritableXmm = dst.map(|r| r.into());
2007        let inst = asm::inst::vmovdqu_a::new(dst, src).into();
2008        self.emit(Inst::External { inst });
2009    }
2010
2011    /// Move integer from `src` to xmm register `dst` using an AVX instruction.
2012    pub fn avx_gpr_to_xmm(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
2013        let dst: WritableXmm = dst.map(|r| r.into());
2014        let inst = match size {
2015            OperandSize::S32 => asm::inst::vmovd_a::new(dst, src).into(),
2016            OperandSize::S64 => asm::inst::vmovq_a::new(dst, src).into(),
2017            _ => unreachable!(),
2018        };
2019
2020        self.emit(Inst::External { inst });
2021    }
2022
2023    pub fn xmm_vptest(&mut self, src1: Reg, src2: Reg) {
2024        let inst = asm::inst::vptest_rm::new(src1, src2).into();
2025        self.emit(Inst::External { inst });
2026    }
2027
2028    /// Converts vector of integers into vector of floating values.
2029    pub fn xmm_vcvt_rr(&mut self, src: Reg, dst: WritableReg, kind: VcvtKind) {
2030        let dst: WritableXmm = dst.map(|x| x.into());
2031        let inst = match kind {
2032            VcvtKind::I32ToF32 => asm::inst::vcvtdq2ps_a::new(dst, src).into(),
2033            VcvtKind::I32ToF64 => asm::inst::vcvtdq2pd_a::new(dst, src).into(),
2034            VcvtKind::F64ToF32 => asm::inst::vcvtpd2ps_a::new(dst, src).into(),
2035            VcvtKind::F64ToI32 => asm::inst::vcvttpd2dq_a::new(dst, src).into(),
2036            VcvtKind::F32ToF64 => asm::inst::vcvtps2pd_a::new(dst, src).into(),
2037            VcvtKind::F32ToI32 => asm::inst::vcvttps2dq_a::new(dst, src).into(),
2038        };
2039        self.emit(Inst::External { inst });
2040    }
2041
2042    /// Subtract floats in vector `src1` to floats in vector `src2`.
2043    pub fn xmm_vsubp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2044        let dst: WritableXmm = dst.map(|r| r.into());
2045        let inst = match size {
2046            OperandSize::S32 => asm::inst::vsubps_b::new(dst, src1, src2).into(),
2047            OperandSize::S64 => asm::inst::vsubpd_b::new(dst, src1, src2).into(),
2048            _ => unimplemented!(),
2049        };
2050        self.emit(Inst::External { inst });
2051    }
2052
2053    /// Subtract integers in vector `src1` from integers in vector `src2`.
2054    pub fn xmm_vpsub_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2055        let dst: WritableXmm = dst.map(|r| r.into());
2056        let inst = match size {
2057            OperandSize::S8 => asm::inst::vpsubb_b::new(dst, src1, src2).into(),
2058            OperandSize::S16 => asm::inst::vpsubw_b::new(dst, src1, src2).into(),
2059            OperandSize::S32 => asm::inst::vpsubd_b::new(dst, src1, src2).into(),
2060            OperandSize::S64 => asm::inst::vpsubq_b::new(dst, src1, src2).into(),
2061            _ => unimplemented!(),
2062        };
2063        self.emit(Inst::External { inst });
2064    }
2065
2066    /// Subtract unsigned integers with unsigned saturation.
2067    pub fn xmm_vpsubus_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg, size: OperandSize) {
2068        let dst: WritableXmm = dst.map(|r| r.into());
2069        let inst = match size {
2070            OperandSize::S8 => asm::inst::vpsubusb_b::new(dst, src1, src2).into(),
2071            OperandSize::S16 => asm::inst::vpsubusw_b::new(dst, src1, src2).into(),
2072            _ => unimplemented!(),
2073        };
2074        self.emit(Inst::External { inst });
2075    }
2076
2077    /// Subtract signed integers with signed saturation.
2078    pub fn xmm_vpsubs_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg, size: OperandSize) {
2079        let dst: WritableXmm = dst.map(|r| r.into());
2080        let inst = match size {
2081            OperandSize::S8 => asm::inst::vpsubsb_b::new(dst, src1, src2).into(),
2082            OperandSize::S16 => asm::inst::vpsubsw_b::new(dst, src1, src2).into(),
2083            _ => unimplemented!(),
2084        };
2085        self.emit(Inst::External { inst });
2086    }
2087
2088    /// Add floats in vector `src1` to floats in vector `src2`.
2089    pub fn xmm_vaddp_rrm(
2090        &mut self,
2091        src1: Reg,
2092        src2: &Address,
2093        dst: WritableReg,
2094        size: OperandSize,
2095    ) {
2096        let dst: WritableXmm = dst.map(|r| r.into());
2097        let address = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
2098        let inst = match size {
2099            OperandSize::S32 => asm::inst::vaddps_b::new(dst, src1, address).into(),
2100            OperandSize::S64 => asm::inst::vaddpd_b::new(dst, src1, address).into(),
2101            _ => unimplemented!(),
2102        };
2103        self.emit(Inst::External { inst });
2104    }
2105
2106    /// Add floats in vector `src1` to floats in vector `src2`.
2107    pub fn xmm_vaddp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2108        let dst: WritableXmm = dst.map(|r| r.into());
2109        let inst = match size {
2110            OperandSize::S32 => asm::inst::vaddps_b::new(dst, src1, src2).into(),
2111            OperandSize::S64 => asm::inst::vaddpd_b::new(dst, src1, src2).into(),
2112            _ => unimplemented!(),
2113        };
2114        self.emit(Inst::External { inst });
2115    }
2116
2117    /// Compare vector register `lhs` with a vector of integers in `rhs` for
2118    /// equality between packed integers and write the resulting vector into
2119    /// `dst`.
2120    pub fn xmm_vpcmpeq_rrm(
2121        &mut self,
2122        dst: WritableReg,
2123        lhs: Reg,
2124        address: &Address,
2125        size: OperandSize,
2126    ) {
2127        let dst: WritableXmm = dst.map(|r| r.into());
2128        let address = Self::to_synthetic_amode(address, MemFlagsData::trusted());
2129        let inst = match size {
2130            OperandSize::S8 => asm::inst::vpcmpeqb_b::new(dst, lhs, address).into(),
2131            OperandSize::S16 => asm::inst::vpcmpeqw_b::new(dst, lhs, address).into(),
2132            OperandSize::S32 => asm::inst::vpcmpeqd_b::new(dst, lhs, address).into(),
2133            OperandSize::S64 => asm::inst::vpcmpeqq_b::new(dst, lhs, address).into(),
2134            _ => unimplemented!(),
2135        };
2136        self.emit(Inst::External { inst });
2137    }
2138
2139    /// Compare vector registers `lhs` and `rhs` for equality between packed
2140    /// integers and write the resulting vector into `dst`.
2141    pub fn xmm_vpcmpeq_rrr(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) {
2142        let dst: WritableXmm = dst.map(|r| r.into());
2143        let inst = match size {
2144            OperandSize::S8 => asm::inst::vpcmpeqb_b::new(dst, lhs, rhs).into(),
2145            OperandSize::S16 => asm::inst::vpcmpeqw_b::new(dst, lhs, rhs).into(),
2146            OperandSize::S32 => asm::inst::vpcmpeqd_b::new(dst, lhs, rhs).into(),
2147            OperandSize::S64 => asm::inst::vpcmpeqq_b::new(dst, lhs, rhs).into(),
2148            _ => unimplemented!(),
2149        };
2150        self.emit(Inst::External { inst });
2151    }
2152
2153    /// Performs a greater than comparison with vectors of signed integers in
2154    /// `lhs` and `rhs` and puts the results in `dst`.
2155    pub fn xmm_vpcmpgt_rrr(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) {
2156        let dst: WritableXmm = dst.map(|r| r.into());
2157        let inst = match size {
2158            OperandSize::S8 => asm::inst::vpcmpgtb_b::new(dst, lhs, rhs).into(),
2159            OperandSize::S16 => asm::inst::vpcmpgtw_b::new(dst, lhs, rhs).into(),
2160            OperandSize::S32 => asm::inst::vpcmpgtd_b::new(dst, lhs, rhs).into(),
2161            OperandSize::S64 => asm::inst::vpcmpgtq_b::new(dst, lhs, rhs).into(),
2162            _ => unimplemented!(),
2163        };
2164        self.emit(Inst::External { inst });
2165    }
2166
2167    /// Performs a max operation with vectors of signed integers in `lhs` and
2168    /// `rhs` and puts the results in `dst`.
2169    pub fn xmm_vpmaxs_rrr(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) {
2170        let dst: WritableXmm = dst.map(|r| r.into());
2171        let inst = match size {
2172            OperandSize::S8 => asm::inst::vpmaxsb_b::new(dst, lhs, rhs).into(),
2173            OperandSize::S16 => asm::inst::vpmaxsw_b::new(dst, lhs, rhs).into(),
2174            OperandSize::S32 => asm::inst::vpmaxsd_b::new(dst, lhs, rhs).into(),
2175            _ => unimplemented!(),
2176        };
2177        self.emit(Inst::External { inst });
2178    }
2179
2180    /// Performs a max operation with vectors of unsigned integers in `lhs` and
2181    /// `rhs` and puts the results in `dst`.
2182    pub fn xmm_vpmaxu_rrr(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) {
2183        let dst: WritableXmm = dst.map(|r| r.into());
2184        let inst = match size {
2185            OperandSize::S8 => asm::inst::vpmaxub_b::new(dst, lhs, rhs).into(),
2186            OperandSize::S16 => asm::inst::vpmaxuw_b::new(dst, lhs, rhs).into(),
2187            OperandSize::S32 => asm::inst::vpmaxud_b::new(dst, lhs, rhs).into(),
2188            _ => unimplemented!(),
2189        };
2190        self.emit(Inst::External { inst });
2191    }
2192
2193    /// Performs a min operation with vectors of signed integers in `lhs` and
2194    /// `rhs` and puts the results in `dst`.
2195    pub fn xmm_vpmins_rrr(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) {
2196        let dst: WritableXmm = dst.map(|r| r.into());
2197        let inst = match size {
2198            OperandSize::S8 => asm::inst::vpminsb_b::new(dst, lhs, rhs).into(),
2199            OperandSize::S16 => asm::inst::vpminsw_b::new(dst, lhs, rhs).into(),
2200            OperandSize::S32 => asm::inst::vpminsd_b::new(dst, lhs, rhs).into(),
2201            _ => unimplemented!(),
2202        };
2203        self.emit(Inst::External { inst });
2204    }
2205
2206    /// Performs a min operation with vectors of unsigned integers in `lhs` and
2207    /// `rhs` and puts the results in `dst`.
2208    pub fn xmm_vpminu_rrr(&mut self, dst: WritableReg, lhs: Reg, rhs: Reg, size: OperandSize) {
2209        let dst: WritableXmm = dst.map(|r| r.into());
2210        let inst = match size {
2211            OperandSize::S8 => asm::inst::vpminub_b::new(dst, lhs, rhs).into(),
2212            OperandSize::S16 => asm::inst::vpminuw_b::new(dst, lhs, rhs).into(),
2213            OperandSize::S32 => asm::inst::vpminud_b::new(dst, lhs, rhs).into(),
2214            _ => unimplemented!(),
2215        };
2216        self.emit(Inst::External { inst });
2217    }
2218
2219    /// Performs a comparison operation between vectors of floats in `lhs` and
2220    /// `rhs` and puts the results in `dst`.
2221    pub fn xmm_vcmpp_rrr(
2222        &mut self,
2223        dst: WritableReg,
2224        lhs: Reg,
2225        rhs: Reg,
2226        size: OperandSize,
2227        kind: VcmpKind,
2228    ) {
2229        let dst: WritableXmm = dst.map(|r| r.into());
2230        let imm = match kind {
2231            VcmpKind::Eq => 0,
2232            VcmpKind::Lt => 1,
2233            VcmpKind::Le => 2,
2234            VcmpKind::Unord => 3,
2235            VcmpKind::Ne => 4,
2236        };
2237        let inst = match size {
2238            OperandSize::S32 => asm::inst::vcmpps_b::new(dst, lhs, rhs, imm).into(),
2239            OperandSize::S64 => asm::inst::vcmppd_b::new(dst, lhs, rhs, imm).into(),
2240            _ => unimplemented!(),
2241        };
2242        self.emit(Inst::External { inst });
2243    }
2244
2245    /// Performs a subtraction on two vectors of floats and puts the results in
2246    /// `dst`.
2247    pub fn xmm_vsub_rrm(&mut self, src1: Reg, src2: &Address, dst: WritableReg, size: OperandSize) {
2248        let dst: WritableXmm = dst.map(|r| r.into());
2249        let address = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
2250        let inst = match size {
2251            OperandSize::S64 => asm::inst::vsubpd_b::new(dst, src1, address).into(),
2252            _ => unimplemented!(),
2253        };
2254        self.emit(Inst::External { inst });
2255    }
2256
2257    /// Performs a subtraction on two vectors of floats and puts the results in
2258    /// `dst`.
2259    pub fn xmm_vsub_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2260        let dst: WritableXmm = dst.map(|r| r.into());
2261        let inst = match size {
2262            OperandSize::S32 => asm::inst::vsubps_b::new(dst, src1, src2).into(),
2263            OperandSize::S64 => asm::inst::vsubpd_b::new(dst, src1, src2).into(),
2264            _ => unimplemented!(),
2265        };
2266        self.emit(Inst::External { inst });
2267    }
2268
2269    /// Converts a vector of signed integers into a vector of narrower integers
2270    /// using saturation to handle overflow.
2271    pub fn xmm_vpackss_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2272        let dst: WritableXmm = dst.map(|r| r.into());
2273        let inst = match size {
2274            OperandSize::S8 => asm::inst::vpacksswb_b::new(dst, src1, src2).into(),
2275            OperandSize::S16 => asm::inst::vpackssdw_b::new(dst, src1, src2).into(),
2276            _ => unimplemented!(),
2277        };
2278        self.emit(Inst::External { inst });
2279    }
2280
2281    /// Converts a vector of unsigned integers into a vector of narrower
2282    /// integers using saturation to handle overflow.
2283    pub fn xmm_vpackus_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2284        let dst: WritableXmm = dst.map(|r| r.into());
2285        let inst = match size {
2286            OperandSize::S8 => asm::inst::vpackuswb_b::new(dst, src1, src2).into(),
2287            OperandSize::S16 => asm::inst::vpackusdw_b::new(dst, src1, src2).into(),
2288            _ => unimplemented!(),
2289        };
2290        self.emit(Inst::External { inst });
2291    }
2292
2293    /// Concatenates `src1` and `src2` and shifts right by `imm` and puts
2294    /// result in `dst`.
2295    pub fn xmm_vpalignr_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, imm: u8) {
2296        let dst: WritableXmm = dst.map(|r| r.into());
2297        let inst = asm::inst::vpalignr_b::new(dst, src1, src2, imm).into();
2298        self.emit(Inst::External { inst });
2299    }
2300
2301    /// Takes the lower lanes of vectors of floats in `src1` and `src2` and
2302    /// interleaves them in `dst`.
2303    pub fn xmm_vunpcklp_rrm(
2304        &mut self,
2305        src1: Reg,
2306        src2: &Address,
2307        dst: WritableReg,
2308        size: OperandSize,
2309    ) {
2310        let dst: WritableXmm = dst.map(|r| r.into());
2311        let address = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
2312        let inst = match size {
2313            OperandSize::S32 => asm::inst::vunpcklps_b::new(dst, src1, address).into(),
2314            _ => unimplemented!(),
2315        };
2316        self.emit(Inst::External { inst });
2317    }
2318
2319    /// Unpacks and interleaves high order data of floats in `src1` and `src2`
2320    /// and puts the results in `dst`.
2321    pub fn xmm_vunpckhp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2322        let dst: WritableXmm = dst.map(|r| r.into());
2323        let inst = match size {
2324            OperandSize::S32 => asm::inst::vunpckhps_b::new(dst, src1, src2).into(),
2325            _ => unimplemented!(),
2326        };
2327        self.emit(Inst::External { inst });
2328    }
2329
2330    /// Unpacks and interleaves the lower lanes of vectors of integers in `src1`
2331    /// and `src2` and puts the results in `dst`.
2332    pub fn xmm_vpunpckl_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2333        let dst: WritableXmm = dst.map(|r| r.into());
2334        let inst = match size {
2335            OperandSize::S8 => asm::inst::vpunpcklbw_b::new(dst, src1, src2).into(),
2336            OperandSize::S16 => asm::inst::vpunpcklwd_b::new(dst, src1, src2).into(),
2337            _ => unimplemented!(),
2338        };
2339        self.emit(Inst::External { inst });
2340    }
2341
2342    /// Unpacks and interleaves the higher lanes of vectors of integers in
2343    /// `src1` and `src2` and puts the results in `dst`.
2344    pub fn xmm_vpunpckh_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2345        let dst: WritableXmm = dst.map(|r| r.into());
2346        let inst = match size {
2347            OperandSize::S8 => asm::inst::vpunpckhbw_b::new(dst, src1, src2).into(),
2348            OperandSize::S16 => asm::inst::vpunpckhwd_b::new(dst, src1, src2).into(),
2349            _ => unimplemented!(),
2350        };
2351        self.emit(Inst::External { inst });
2352    }
2353
2354    pub(crate) fn vpmullq(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2355        let dst: WritableXmm = dst.map(|r| r.into());
2356        let inst = asm::inst::vpmullq_c::new(dst, src1, src2).into();
2357        self.emit(Inst::External { inst });
2358    }
2359
2360    /// Creates a mask made up of the most significant bit of each byte of
2361    /// `src` and stores the result in `dst`.
2362    pub fn xmm_vpmovmsk_rr(
2363        &mut self,
2364        src: Reg,
2365        dst: WritableReg,
2366        src_size: OperandSize,
2367        dst_size: OperandSize,
2368    ) {
2369        assert_eq!(dst_size, OperandSize::S32);
2370        let dst: WritableGpr = dst.map(|r| r.into());
2371        let inst = match src_size {
2372            OperandSize::S8 => asm::inst::vpmovmskb_rm::new(dst, src).into(),
2373            _ => unimplemented!(),
2374        };
2375
2376        self.emit(Inst::External { inst });
2377    }
2378
2379    /// Creates a mask made up of the most significant bit of each byte of
2380    /// in `src` and stores the result in `dst`.
2381    pub fn xmm_vmovskp_rr(
2382        &mut self,
2383        src: Reg,
2384        dst: WritableReg,
2385        src_size: OperandSize,
2386        dst_size: OperandSize,
2387    ) {
2388        assert_eq!(dst_size, OperandSize::S32);
2389        let dst: WritableGpr = dst.map(|r| r.into());
2390        let inst = match src_size {
2391            OperandSize::S32 => asm::inst::vmovmskps_rm::new(dst, src).into(),
2392            OperandSize::S64 => asm::inst::vmovmskpd_rm::new(dst, src).into(),
2393            _ => unimplemented!(),
2394        };
2395
2396        self.emit(Inst::External { inst });
2397    }
2398
2399    /// Compute the absolute value of elements in vector `src` and put the
2400    /// results in `dst`.
2401    pub fn xmm_vpabs_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
2402        let dst: WritableXmm = dst.map(|r| r.into());
2403        let inst = match size {
2404            OperandSize::S8 => asm::inst::vpabsb_a::new(dst, src).into(),
2405            OperandSize::S16 => asm::inst::vpabsw_a::new(dst, src).into(),
2406            OperandSize::S32 => asm::inst::vpabsd_a::new(dst, src).into(),
2407            _ => unimplemented!(),
2408        };
2409        self.emit(Inst::External { inst });
2410    }
2411
2412    /// Arithmetically (sign preserving) right shift on vector in `src` by
2413    /// `amount` with result written to `dst`.
2414    pub fn xmm_vpsra_rrr(&mut self, src: Reg, amount: Reg, dst: WritableReg, size: OperandSize) {
2415        let dst: WritableXmm = dst.map(|r| r.into());
2416        let inst = match size {
2417            OperandSize::S16 => asm::inst::vpsraw_c::new(dst, src, amount).into(),
2418            OperandSize::S32 => asm::inst::vpsrad_c::new(dst, src, amount).into(),
2419            _ => unimplemented!(),
2420        };
2421        self.emit(Inst::External { inst });
2422    }
2423
2424    /// Arithmetically (sign preserving) right shift on vector in `src` by
2425    /// `imm` with result written to `dst`.
2426    pub fn xmm_vpsra_rri(&mut self, src: Reg, dst: WritableReg, imm: u32, size: OperandSize) {
2427        let dst: WritableXmm = dst.map(|r| r.into());
2428        let imm = u8::try_from(imm).expect("immediate must fit in 8 bits");
2429        let inst = match size {
2430            OperandSize::S32 => asm::inst::vpsrad_d::new(dst, src, imm).into(),
2431            _ => unimplemented!(),
2432        };
2433        self.emit(Inst::External { inst });
2434    }
2435
2436    /// Shift vector data left by `imm`.
2437    pub fn xmm_vpsll_rri(&mut self, src: Reg, dst: WritableReg, imm: u32, size: OperandSize) {
2438        let dst: WritableXmm = dst.map(|r| r.into());
2439        let imm = u8::try_from(imm).expect("immediate must fit in 8 bits");
2440        let inst = match size {
2441            OperandSize::S32 => asm::inst::vpslld_d::new(dst, src, imm).into(),
2442            OperandSize::S64 => asm::inst::vpsllq_d::new(dst, src, imm).into(),
2443            _ => unimplemented!(),
2444        };
2445        self.emit(Inst::External { inst });
2446    }
2447
2448    /// Shift vector data left by `amount`.
2449    pub fn xmm_vpsll_rrr(&mut self, src: Reg, amount: Reg, dst: WritableReg, size: OperandSize) {
2450        let dst: WritableXmm = dst.map(|r| r.into());
2451        let inst = match size {
2452            OperandSize::S16 => asm::inst::vpsllw_c::new(dst, src, amount).into(),
2453            OperandSize::S32 => asm::inst::vpslld_c::new(dst, src, amount).into(),
2454            OperandSize::S64 => asm::inst::vpsllq_c::new(dst, src, amount).into(),
2455            _ => unimplemented!(),
2456        };
2457        self.emit(Inst::External { inst });
2458    }
2459
2460    /// Shift vector data right by `imm`.
2461    pub fn xmm_vpsrl_rri(&mut self, src: Reg, dst: WritableReg, imm: u32, size: OperandSize) {
2462        let dst: WritableXmm = dst.map(|r| r.into());
2463        let imm = u8::try_from(imm).expect("immediate must fit in 8 bits");
2464        let inst = match size {
2465            OperandSize::S16 => asm::inst::vpsrlw_d::new(dst, src, imm).into(),
2466            OperandSize::S32 => asm::inst::vpsrld_d::new(dst, src, imm).into(),
2467            OperandSize::S64 => asm::inst::vpsrlq_d::new(dst, src, imm).into(),
2468            _ => unimplemented!(),
2469        };
2470        self.emit(Inst::External { inst });
2471    }
2472
2473    /// Shift vector data right by `amount`.
2474    pub fn xmm_vpsrl_rrr(&mut self, src: Reg, amount: Reg, dst: WritableReg, size: OperandSize) {
2475        let dst: WritableXmm = dst.map(|r| r.into());
2476        let inst = match size {
2477            OperandSize::S16 => asm::inst::vpsrlw_c::new(dst, src, amount).into(),
2478            OperandSize::S32 => asm::inst::vpsrld_c::new(dst, src, amount).into(),
2479            OperandSize::S64 => asm::inst::vpsrlq_c::new(dst, src, amount).into(),
2480            _ => unimplemented!(),
2481        };
2482        self.emit(Inst::External { inst });
2483    }
2484
2485    /// Perform an `and` operation on vectors of floats in `src1` and `src2`
2486    /// and put the results in `dst`.
2487    pub fn xmm_vandp_rrm(
2488        &mut self,
2489        src1: Reg,
2490        src2: &Address,
2491        dst: WritableReg,
2492        size: OperandSize,
2493    ) {
2494        let dst: WritableXmm = dst.map(|r| r.into());
2495        let address = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
2496        let inst = match size {
2497            OperandSize::S32 => asm::inst::vandps_b::new(dst, src1, address).into(),
2498            OperandSize::S64 => asm::inst::vandpd_b::new(dst, src1, address).into(),
2499            _ => unimplemented!(),
2500        };
2501        self.emit(Inst::External { inst });
2502    }
2503
2504    /// Perform an `and` operation on vectors of floats in `src1` and `src2`
2505    /// and put the results in `dst`.
2506    pub fn xmm_vandp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2507        let dst: WritableXmm = dst.map(|r| r.into());
2508        let inst = match size {
2509            OperandSize::S32 => asm::inst::vandps_b::new(dst, src1, src2).into(),
2510            OperandSize::S64 => asm::inst::vandpd_b::new(dst, src1, src2).into(),
2511            _ => unimplemented!(),
2512        };
2513        self.emit(Inst::External { inst });
2514    }
2515
2516    /// Performs a bitwise `and` operation on the vectors in `src1` and `src2`
2517    /// and stores the results in `dst`.
2518    pub fn xmm_vpand_rrm(&mut self, src1: Reg, src2: &Address, dst: WritableReg) {
2519        let dst: WritableXmm = dst.map(|r| r.into());
2520        let address = Self::to_synthetic_amode(&src2, MemFlagsData::trusted());
2521        let inst = asm::inst::vpand_b::new(dst, src1, address).into();
2522        self.emit(Inst::External { inst });
2523    }
2524
2525    /// Performs a bitwise `and` operation on the vectors in `src1` and `src2`
2526    /// and stores the results in `dst`.
2527    pub fn xmm_vpand_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2528        let dst: WritableXmm = dst.map(|r| r.into());
2529        let inst = asm::inst::vpand_b::new(dst, src1, src2).into();
2530        self.emit(Inst::External { inst });
2531    }
2532
2533    /// Perform an `and not` operation on vectors of floats in `src1` and
2534    /// `src2` and put the results in `dst`.
2535    pub fn xmm_vandnp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2536        let dst: WritableXmm = dst.map(|r| r.into());
2537        let inst = match size {
2538            OperandSize::S32 => asm::inst::vandnps_b::new(dst, src1, src2).into(),
2539            OperandSize::S64 => asm::inst::vandnpd_b::new(dst, src1, src2).into(),
2540            _ => unimplemented!(),
2541        };
2542        self.emit(Inst::External { inst });
2543    }
2544
2545    /// Perform an `and not` operation on vectors in `src1` and `src2` and put
2546    /// the results in `dst`.
2547    pub fn xmm_vpandn_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2548        let dst: WritableXmm = dst.map(|r| r.into());
2549        let inst = asm::inst::vpandn_b::new(dst, src1, src2).into();
2550        self.emit(Inst::External { inst });
2551    }
2552
2553    /// Perform an or operation for the vectors of floats in `src1` and `src2`
2554    /// and put the results in `dst`.
2555    pub fn xmm_vorp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2556        let dst: WritableXmm = dst.map(|r| r.into());
2557        let inst = match size {
2558            OperandSize::S32 => asm::inst::vorps_b::new(dst, src1, src2).into(),
2559            OperandSize::S64 => asm::inst::vorpd_b::new(dst, src1, src2).into(),
2560            _ => unimplemented!(),
2561        };
2562        self.emit(Inst::External { inst });
2563    }
2564
2565    /// Bitwise OR of `src1` and `src2`.
2566    pub fn xmm_vpor_rrr(&mut self, dst: WritableReg, src1: Reg, src2: Reg) {
2567        let dst: WritableXmm = dst.map(|r| r.into());
2568        let inst = asm::inst::vpor_b::new(dst, src1, src2).into();
2569        self.emit(Inst::External { inst });
2570    }
2571
2572    /// Bitwise logical xor of vectors of floats in `src1` and `src2` and puts
2573    /// the results in `dst`.
2574    pub fn xmm_vxorp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2575        let dst: WritableXmm = dst.map(|r| r.into());
2576        let inst = match size {
2577            OperandSize::S32 => asm::inst::vxorps_b::new(dst, src1, src2).into(),
2578            OperandSize::S64 => asm::inst::vxorpd_b::new(dst, src1, src2).into(),
2579            _ => unimplemented!(),
2580        };
2581        self.emit(Inst::External { inst });
2582    }
2583
2584    /// Perform a logical on vector in `src` and in `address` and put the
2585    /// results in `dst`.
2586    pub fn xmm_vpxor_rmr(&mut self, src: Reg, address: &Address, dst: WritableReg) {
2587        let dst: WritableXmm = dst.map(|r| r.into());
2588        let address = Self::to_synthetic_amode(address, MemFlagsData::trusted());
2589        let inst = asm::inst::vpxor_b::new(dst, src, address).into();
2590        self.emit(Inst::External { inst });
2591    }
2592
2593    /// Perform a logical on vectors in `src1` and `src2` and put the results in
2594    /// `dst`.
2595    pub fn xmm_vpxor_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2596        let dst: WritableXmm = dst.map(|r| r.into());
2597        let inst = asm::inst::vpxor_b::new(dst, src1, src2).into();
2598        self.emit(Inst::External { inst });
2599    }
2600
2601    /// Perform a max operation across two vectors of floats and put the
2602    /// results in `dst`.
2603    pub fn xmm_vmaxp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2604        let dst: WritableXmm = dst.map(|r| r.into());
2605        let inst = match size {
2606            OperandSize::S32 => asm::inst::vmaxps_b::new(dst, src1, src2).into(),
2607            OperandSize::S64 => asm::inst::vmaxpd_b::new(dst, src1, src2).into(),
2608            _ => unimplemented!(),
2609        };
2610        self.emit(Inst::External { inst });
2611    }
2612
2613    // Perform a min operation across two vectors of floats and put the
2614    // results in `dst`.
2615    pub fn xmm_vminp_rrm(
2616        &mut self,
2617        src1: Reg,
2618        src2: &Address,
2619        dst: WritableReg,
2620        size: OperandSize,
2621    ) {
2622        let dst: WritableXmm = dst.map(|r| r.into());
2623        let address = Self::to_synthetic_amode(src2, MemFlagsData::trusted());
2624        let inst = match size {
2625            OperandSize::S32 => asm::inst::vminps_b::new(dst, src1, address).into(),
2626            OperandSize::S64 => asm::inst::vminpd_b::new(dst, src1, address).into(),
2627            _ => unimplemented!(),
2628        };
2629        self.emit(Inst::External { inst });
2630    }
2631
2632    // Perform a min operation across two vectors of floats and put the
2633    // results in `dst`.
2634    pub fn xmm_vminp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2635        let dst: WritableXmm = dst.map(|r| r.into());
2636        let inst = match size {
2637            OperandSize::S32 => asm::inst::vminps_b::new(dst, src1, src2).into(),
2638            OperandSize::S64 => asm::inst::vminpd_b::new(dst, src1, src2).into(),
2639            _ => unimplemented!(),
2640        };
2641        self.emit(Inst::External { inst });
2642    }
2643
2644    // Round a vector of floats.
2645    pub fn xmm_vroundp_rri(
2646        &mut self,
2647        src: Reg,
2648        dst: WritableReg,
2649        mode: VroundMode,
2650        size: OperandSize,
2651    ) {
2652        let dst: WritableXmm = dst.map(|r| r.into());
2653        let imm = match mode {
2654            VroundMode::TowardNearest => 0,
2655            VroundMode::TowardNegativeInfinity => 1,
2656            VroundMode::TowardPositiveInfinity => 2,
2657            VroundMode::TowardZero => 3,
2658        };
2659
2660        let inst = match size {
2661            OperandSize::S32 => asm::inst::vroundps_rmi::new(dst, src, imm).into(),
2662            OperandSize::S64 => asm::inst::vroundpd_rmi::new(dst, src, imm).into(),
2663            _ => unimplemented!(),
2664        };
2665
2666        self.emit(Inst::External { inst });
2667    }
2668
2669    /// Shuffle of vectors of floats.
2670    pub fn xmm_vshufp_rrri(
2671        &mut self,
2672        src1: Reg,
2673        src2: Reg,
2674        dst: WritableReg,
2675        imm: u8,
2676        size: OperandSize,
2677    ) {
2678        let dst: WritableXmm = dst.map(|r| r.into());
2679        let inst = match size {
2680            OperandSize::S32 => asm::inst::vshufps_b::new(dst, src1, src2, imm).into(),
2681            _ => unimplemented!(),
2682        };
2683        self.emit(Inst::External { inst });
2684    }
2685
2686    /// Each lane in `src1` is multiplied by the corresponding lane in `src2`
2687    /// producing intermediate 32-bit operands. Each intermediate 32-bit
2688    /// operand is truncated to 18 most significant bits. Rounding is performed
2689    /// by adding 1 to the least significant bit of the 18-bit intermediate
2690    /// result. The 16 bits immediately to the right of the most significant
2691    /// bit of each 18-bit intermediate result is placed in each lane of `dst`.
2692    pub fn xmm_vpmulhrs_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2693        let dst: WritableXmm = dst.map(|r| r.into());
2694        let inst = match size {
2695            OperandSize::S16 => asm::inst::vpmulhrsw_b::new(dst, src1, src2).into(),
2696            _ => unimplemented!(),
2697        };
2698        self.emit(Inst::External { inst });
2699    }
2700
2701    pub fn xmm_vpmuldq_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2702        let dst: WritableXmm = dst.map(|r| r.into());
2703        let inst = asm::inst::vpmuldq_b::new(dst, src1, src2).into();
2704        self.emit(Inst::External { inst });
2705    }
2706
2707    pub fn xmm_vpmuludq_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2708        let dst: WritableXmm = dst.map(|r| r.into());
2709        let inst = asm::inst::vpmuludq_b::new(dst, src1, src2).into();
2710        self.emit(Inst::External { inst });
2711    }
2712
2713    pub fn xmm_vpmull_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2714        let dst: WritableXmm = dst.map(|r| r.into());
2715        let inst = match size {
2716            OperandSize::S16 => asm::inst::vpmullw_b::new(dst, src1, src2).into(),
2717            OperandSize::S32 => asm::inst::vpmulld_b::new(dst, src1, src2).into(),
2718            _ => unimplemented!(),
2719        };
2720        self.emit(Inst::External { inst });
2721    }
2722
2723    pub fn xmm_vmulp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2724        let dst: WritableXmm = dst.map(|r| r.into());
2725        let inst = match size {
2726            OperandSize::S32 => asm::inst::vmulps_b::new(dst, src1, src2).into(),
2727            OperandSize::S64 => asm::inst::vmulpd_b::new(dst, src1, src2).into(),
2728            _ => unimplemented!(),
2729        };
2730        self.emit(Inst::External { inst });
2731    }
2732
2733    /// Perform an average operation for the vector of unsigned integers in
2734    /// `src1` and `src2` and put the results in `dst`.
2735    pub fn xmm_vpavg_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2736        let dst: WritableXmm = dst.map(|r| r.into());
2737        let inst = match size {
2738            OperandSize::S8 => asm::inst::vpavgb_b::new(dst, src1, src2).into(),
2739            OperandSize::S16 => asm::inst::vpavgw_b::new(dst, src1, src2).into(),
2740            _ => unimplemented!(),
2741        };
2742        self.emit(Inst::External { inst });
2743    }
2744
2745    /// Divide the vector of floats in `src1` by the vector of floats in `src2`
2746    /// and put the results in `dst`.
2747    pub fn xmm_vdivp_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg, size: OperandSize) {
2748        let dst: WritableXmm = dst.map(|r| r.into());
2749        let inst = match size {
2750            OperandSize::S32 => asm::inst::vdivps_b::new(dst, src1, src2).into(),
2751            OperandSize::S64 => asm::inst::vdivpd_b::new(dst, src1, src2).into(),
2752            _ => unimplemented!(),
2753        };
2754        self.emit(Inst::External { inst });
2755    }
2756
2757    /// Compute square roots of vector of floats in `src` and put the results
2758    /// in `dst`.
2759    pub fn xmm_vsqrtp_rr(&mut self, src: Reg, dst: WritableReg, size: OperandSize) {
2760        let dst: WritableXmm = dst.map(|r| r.into());
2761        let inst = match size {
2762            OperandSize::S32 => asm::inst::vsqrtps_b::new(dst, src).into(),
2763            OperandSize::S64 => asm::inst::vsqrtpd_b::new(dst, src).into(),
2764            _ => unimplemented!(),
2765        };
2766        self.emit(Inst::External { inst });
2767    }
2768
2769    /// Multiply and add packed signed and unsigned bytes.
2770    pub fn xmm_vpmaddubsw_rmr(&mut self, src: Reg, address: &Address, dst: WritableReg) {
2771        let dst: WritableXmm = dst.map(|r| r.into());
2772        let address = Self::to_synthetic_amode(address, MemFlagsData::trusted());
2773        let inst = asm::inst::vpmaddubsw_b::new(dst, src, address).into();
2774        self.emit(Inst::External { inst });
2775    }
2776
2777    /// Multiply and add packed signed and unsigned bytes.
2778    pub fn xmm_vpmaddubsw_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2779        let dst: WritableXmm = dst.map(|r| r.into());
2780        let inst = asm::inst::vpmaddubsw_b::new(dst, src1, src2).into();
2781        self.emit(Inst::External { inst });
2782    }
2783
2784    /// Multiple and add packed integers.
2785    pub fn xmm_vpmaddwd_rmr(&mut self, src: Reg, address: &Address, dst: WritableReg) {
2786        let dst: WritableXmm = dst.map(|r| r.into());
2787        let address = Self::to_synthetic_amode(address, MemFlagsData::trusted());
2788        let inst = asm::inst::vpmaddwd_b::new(dst, src, address).into();
2789        self.emit(Inst::External { inst });
2790    }
2791
2792    /// Multiple and add packed integers.
2793    pub fn xmm_vpmaddwd_rrr(&mut self, src1: Reg, src2: Reg, dst: WritableReg) {
2794        let dst: WritableXmm = dst.map(|r| r.into());
2795        let inst = asm::inst::vpmaddwd_b::new(dst, src1, src2).into();
2796        self.emit(Inst::External { inst });
2797    }
2798}
2799
2800/// Captures the region in a MachBuffer where an add-with-immediate instruction would be emitted,
2801/// but the immediate is not yet known. Currently, this implementation expects a 32-bit immediate,
2802/// so 8 and 16 bit operand sizes are not supported.
2803pub(crate) struct PatchableAddToReg {
2804    /// The region to be patched in the [`MachBuffer`]. It must contain a valid add instruction
2805    /// sequence, accepting a 32-bit immediate.
2806    region: PatchRegion,
2807
2808    /// The offset into the patchable region where the patchable constant begins.
2809    constant_offset: usize,
2810}
2811
2812impl PatchableAddToReg {
2813    /// Create a new [`PatchableAddToReg`] by capturing a region in the output buffer where the
2814    /// add-with-immediate occurs. The [`MachBuffer`] will have and add-with-immediate instruction
2815    /// present in that region, though it will add `0` until the `::finalize` method is called.
2816    ///
2817    /// Currently this implementation expects to be able to patch a 32-bit immediate, which means
2818    /// that 8 and 16-bit addition cannot be supported.
2819    pub(crate) fn new(reg: Reg, size: OperandSize, asm: &mut Assembler) -> Self {
2820        let open = asm.buffer_mut().start_patchable();
2821        let start = asm.buffer().cur_offset();
2822
2823        // Emit the opcode and register use for the add instruction.
2824        let reg = pair_gpr(Writable::from_reg(reg));
2825        let inst = match size {
2826            OperandSize::S32 => asm::inst::addl_mi::new(reg, 0_u32).into(),
2827            OperandSize::S64 => asm::inst::addq_mi_sxl::new(reg, 0_i32).into(),
2828            _ => {
2829                panic!(
2830                    "{}-bit addition is not supported, please see the comment on PatchableAddToReg::new",
2831                    size.num_bits(),
2832                )
2833            }
2834        };
2835        asm.emit(Inst::External { inst });
2836
2837        // The offset to the constant is the width of what was just emitted
2838        // minus 4, the width of the 32-bit immediate.
2839        let constant_offset = usize::try_from(asm.buffer().cur_offset() - start - 4).unwrap();
2840
2841        let region = asm.buffer_mut().end_patchable(open);
2842
2843        Self {
2844            region,
2845            constant_offset,
2846        }
2847    }
2848
2849    /// Patch the [`MachBuffer`] with the known constant to be added to the register. The final
2850    /// value is passed in as an i32, but the instruction encoding is fixed when
2851    /// [`PatchableAddToReg::new`] is called.
2852    pub(crate) fn finalize(self, val: i32, buffer: &mut MachBuffer<Inst>) {
2853        let slice = self.region.patch(buffer);
2854        debug_assert_eq!(slice.len(), self.constant_offset + 4);
2855        slice[self.constant_offset..].copy_from_slice(val.to_le_bytes().as_slice());
2856    }
2857}