Skip to main content

cranelift_codegen/isa/aarch64/inst/
mod.rs

1//! This module defines aarch64-specific machine instruction types.
2
3use crate::binemit::{Addend, CodeOffset, Reloc};
4use crate::ir::types::{F16, F32, F64, F128, I8, I8X16, I16, I32, I64, I128};
5use crate::ir::{MemFlagsData, Type, types};
6use crate::isa::{CallConv, FunctionAlignment};
7use crate::machinst::*;
8use crate::{CodegenError, CodegenResult, settings};
9
10use crate::machinst::{PrettyPrint, Reg, RegClass, Writable};
11
12use alloc::string::{String, ToString};
13use alloc::vec::Vec;
14use core::fmt::Write;
15use core::slice;
16use smallvec::{SmallVec, smallvec};
17
18pub(crate) mod regs;
19pub use self::regs::*;
20pub mod imms;
21pub use self::imms::*;
22pub mod args;
23pub use self::args::*;
24pub mod emit;
25pub(crate) use self::emit::*;
26use crate::isa::aarch64::abi::AArch64MachineDeps;
27
28pub(crate) mod unwind;
29
30#[cfg(test)]
31mod emit_tests;
32
33//=============================================================================
34// Instructions (top level): definition
35
36pub use crate::isa::aarch64::lower::isle::generated_code::{
37    ALUOp, ALUOp3, AMode, APIKey, AtomicRMWLoopOp, AtomicRMWOp, BfmOp, BitOp, BranchTargetType,
38    FPUOp1, FPUOp2, FPUOp3, FpuRoundMode, FpuToIntOp, IntToFpuOp, MInst as Inst, MoveWideOp,
39    VecALUModOp, VecALUOp, VecExtendOp, VecLanesOp, VecMisc2, VecPairOp, VecRRLongOp,
40    VecRRNarrowOp, VecRRPairLongOp, VecRRRLongModOp, VecRRRLongOp, VecShiftImmModOp, VecShiftImmOp,
41};
42
43/// A floating-point unit (FPU) operation with two args, a register and an immediate.
44#[derive(Copy, Clone, Debug)]
45pub enum FPUOpRI {
46    /// Unsigned right shift. Rd = Rn << #imm
47    UShr32(FPURightShiftImm),
48    /// Unsigned right shift. Rd = Rn << #imm
49    UShr64(FPURightShiftImm),
50}
51
52/// A floating-point unit (FPU) operation with two args, a register and
53/// an immediate that modifies its dest (so takes that input value as a
54/// separate virtual register).
55#[derive(Copy, Clone, Debug)]
56pub enum FPUOpRIMod {
57    /// Shift left and insert. Rd |= Rn << #imm
58    Sli32(FPULeftShiftImm),
59    /// Shift left and insert. Rd |= Rn << #imm
60    Sli64(FPULeftShiftImm),
61}
62
63impl BfmOp {
64    /// Get the assembly mnemonic for this opcode.
65    pub fn op_str(&self) -> &'static str {
66        match self {
67            BfmOp::UBfm => "ubfm",
68            BfmOp::SBfm => "sbfm",
69        }
70    }
71}
72
73impl BitOp {
74    /// Get the assembly mnemonic for this opcode.
75    pub fn op_str(&self) -> &'static str {
76        match self {
77            BitOp::RBit => "rbit",
78            BitOp::Clz => "clz",
79            BitOp::Cls => "cls",
80            BitOp::Rev16 => "rev16",
81            BitOp::Rev32 => "rev32",
82            BitOp::Rev64 => "rev64",
83        }
84    }
85}
86
87/// Additional information for `return_call[_ind]` instructions, left out of
88/// line to lower the size of the `Inst` enum.
89#[derive(Clone, Debug)]
90pub struct ReturnCallInfo<T> {
91    /// Where this call is going to
92    pub dest: T,
93    /// Arguments to the call instruction.
94    pub uses: CallArgList,
95    /// The size of the new stack frame's stack arguments. This is necessary
96    /// for copying the frame over our current frame. It must already be
97    /// allocated on the stack.
98    pub new_stack_arg_size: u32,
99    /// API key to use to restore the return address, if any.
100    pub key: Option<APIKey>,
101    /// Whether pointer-auth return addresses are signed even without frame setup.
102    pub sign_return_address_all: bool,
103}
104
105fn count_zero_half_words(mut value: u64, num_half_words: u8) -> usize {
106    let mut count = 0;
107    for _ in 0..num_half_words {
108        if value & 0xffff == 0 {
109            count += 1;
110        }
111        value >>= 16;
112    }
113
114    count
115}
116
117impl Inst {
118    /// Create an instruction that loads a constant, using one of several options (MOVZ, MOVN,
119    /// logical immediate, or constant pool).
120    pub fn load_constant(rd: Writable<Reg>, value: u64) -> SmallVec<[Inst; 4]> {
121        // NB: this is duplicated in `lower/isle.rs` and `inst.isle` right now,
122        // if modifications are made here before this is deleted after moving to
123        // ISLE then those locations should be updated as well.
124
125        if let Some(imm) = MoveWideConst::maybe_from_u64(value) {
126            // 16-bit immediate (shifted by 0, 16, 32 or 48 bits) in MOVZ
127            smallvec![Inst::MovWide {
128                op: MoveWideOp::MovZ,
129                rd,
130                imm,
131                size: OperandSize::Size64
132            }]
133        } else if let Some(imm) = MoveWideConst::maybe_from_u64(!value) {
134            // 16-bit immediate (shifted by 0, 16, 32 or 48 bits) in MOVN
135            smallvec![Inst::MovWide {
136                op: MoveWideOp::MovN,
137                rd,
138                imm,
139                size: OperandSize::Size64
140            }]
141        } else if let Some(imml) = ImmLogic::maybe_from_u64(value, I64) {
142            // Weird logical-instruction immediate in ORI using zero register
143            smallvec![Inst::AluRRImmLogic {
144                alu_op: ALUOp::Orr,
145                size: OperandSize::Size64,
146                rd,
147                rn: zero_reg(),
148                imml,
149            }]
150        } else {
151            let mut insts = smallvec![];
152
153            // If the top 32 bits are zero, use 32-bit `mov` operations.
154            let (num_half_words, size, negated) = if value >> 32 == 0 {
155                (2, OperandSize::Size32, (!value << 32) >> 32)
156            } else {
157                (4, OperandSize::Size64, !value)
158            };
159
160            // If the number of 0xffff half words is greater than the number of 0x0000 half words
161            // it is more efficient to use `movn` for the first instruction.
162            let first_is_inverted = count_zero_half_words(negated, num_half_words)
163                > count_zero_half_words(value, num_half_words);
164
165            // Either 0xffff or 0x0000 half words can be skipped, depending on the first
166            // instruction used.
167            let ignored_halfword = if first_is_inverted { 0xffff } else { 0 };
168
169            let halfwords: SmallVec<[_; 4]> = (0..num_half_words)
170                .filter_map(|i| {
171                    let imm16 = (value >> (16 * i)) & 0xffff;
172                    if imm16 == ignored_halfword {
173                        None
174                    } else {
175                        Some((i, imm16))
176                    }
177                })
178                .collect();
179
180            let mut prev_result = None;
181            for (i, imm16) in halfwords {
182                let shift = i * 16;
183
184                if let Some(rn) = prev_result {
185                    let imm = MoveWideConst::maybe_with_shift(imm16 as u16, shift).unwrap();
186                    insts.push(Inst::MovK { rd, rn, imm, size });
187                } else {
188                    if first_is_inverted {
189                        let imm =
190                            MoveWideConst::maybe_with_shift(((!imm16) & 0xffff) as u16, shift)
191                                .unwrap();
192                        insts.push(Inst::MovWide {
193                            op: MoveWideOp::MovN,
194                            rd,
195                            imm,
196                            size,
197                        });
198                    } else {
199                        let imm = MoveWideConst::maybe_with_shift(imm16 as u16, shift).unwrap();
200                        insts.push(Inst::MovWide {
201                            op: MoveWideOp::MovZ,
202                            rd,
203                            imm,
204                            size,
205                        });
206                    }
207                }
208
209                prev_result = Some(rd.to_reg());
210            }
211
212            assert!(prev_result.is_some());
213
214            insts
215        }
216    }
217
218    /// Generic constructor for a load (zero-extending where appropriate).
219    pub fn gen_load(into_reg: Writable<Reg>, mem: AMode, ty: Type, flags: MemFlagsData) -> Inst {
220        match ty {
221            I8 => Inst::ULoad8 {
222                rd: into_reg,
223                mem,
224                flags,
225            },
226            I16 => Inst::ULoad16 {
227                rd: into_reg,
228                mem,
229                flags,
230            },
231            I32 => Inst::ULoad32 {
232                rd: into_reg,
233                mem,
234                flags,
235            },
236            I64 => Inst::ULoad64 {
237                rd: into_reg,
238                mem,
239                flags,
240            },
241            _ => {
242                if ty.is_vector() || ty.is_float() {
243                    let bits = ty_bits(ty);
244                    let rd = into_reg;
245
246                    match bits {
247                        128 => Inst::FpuLoad128 { rd, mem, flags },
248                        64 => Inst::FpuLoad64 { rd, mem, flags },
249                        32 => Inst::FpuLoad32 { rd, mem, flags },
250                        16 => Inst::FpuLoad16 { rd, mem, flags },
251                        _ => unimplemented!("gen_load({})", ty),
252                    }
253                } else {
254                    unimplemented!("gen_load({})", ty);
255                }
256            }
257        }
258    }
259
260    /// Generic constructor for a store.
261    pub fn gen_store(mem: AMode, from_reg: Reg, ty: Type, flags: MemFlagsData) -> Inst {
262        match ty {
263            I8 => Inst::Store8 {
264                rd: from_reg,
265                mem,
266                flags,
267            },
268            I16 => Inst::Store16 {
269                rd: from_reg,
270                mem,
271                flags,
272            },
273            I32 => Inst::Store32 {
274                rd: from_reg,
275                mem,
276                flags,
277            },
278            I64 => Inst::Store64 {
279                rd: from_reg,
280                mem,
281                flags,
282            },
283            _ => {
284                if ty.is_vector() || ty.is_float() {
285                    let bits = ty_bits(ty);
286                    let rd = from_reg;
287
288                    match bits {
289                        128 => Inst::FpuStore128 { rd, mem, flags },
290                        64 => Inst::FpuStore64 { rd, mem, flags },
291                        32 => Inst::FpuStore32 { rd, mem, flags },
292                        16 => Inst::FpuStore16 { rd, mem, flags },
293                        _ => unimplemented!("gen_store({})", ty),
294                    }
295                } else {
296                    unimplemented!("gen_store({})", ty);
297                }
298            }
299        }
300    }
301
302    /// What type does this load or store instruction access in memory? When
303    /// uimm12 encoding is used, the size of this type is the amount that
304    /// immediate offsets are scaled by.
305    pub fn mem_type(&self) -> Option<Type> {
306        match self {
307            Inst::ULoad8 { .. } => Some(I8),
308            Inst::SLoad8 { .. } => Some(I8),
309            Inst::ULoad16 { .. } => Some(I16),
310            Inst::SLoad16 { .. } => Some(I16),
311            Inst::ULoad32 { .. } => Some(I32),
312            Inst::SLoad32 { .. } => Some(I32),
313            Inst::ULoad64 { .. } => Some(I64),
314            Inst::FpuLoad16 { .. } => Some(F16),
315            Inst::FpuLoad32 { .. } => Some(F32),
316            Inst::FpuLoad64 { .. } => Some(F64),
317            Inst::FpuLoad128 { .. } => Some(I8X16),
318            Inst::Store8 { .. } => Some(I8),
319            Inst::Store16 { .. } => Some(I16),
320            Inst::Store32 { .. } => Some(I32),
321            Inst::Store64 { .. } => Some(I64),
322            Inst::FpuStore16 { .. } => Some(F16),
323            Inst::FpuStore32 { .. } => Some(F32),
324            Inst::FpuStore64 { .. } => Some(F64),
325            Inst::FpuStore128 { .. } => Some(I8X16),
326            _ => None,
327        }
328    }
329}
330
331//=============================================================================
332// Instructions: get_regs
333
334fn memarg_operands(memarg: &mut AMode, collector: &mut impl OperandVisitor) {
335    match memarg {
336        AMode::Unscaled { rn, .. } | AMode::UnsignedOffset { rn, .. } => {
337            collector.reg_use(rn);
338        }
339        AMode::RegReg { rn, rm, .. }
340        | AMode::RegScaled { rn, rm, .. }
341        | AMode::RegScaledExtended { rn, rm, .. }
342        | AMode::RegExtended { rn, rm, .. } => {
343            collector.reg_use(rn);
344            collector.reg_use(rm);
345        }
346        AMode::Label { .. } => {}
347        AMode::SPPreIndexed { .. } | AMode::SPPostIndexed { .. } => {}
348        AMode::FPOffset { .. } | AMode::IncomingArg { .. } => {}
349        AMode::SPOffset { .. } | AMode::SlotOffset { .. } => {}
350        AMode::RegOffset { rn, .. } => {
351            collector.reg_use(rn);
352        }
353        AMode::Const { .. } => {}
354    }
355}
356
357fn pairmemarg_operands(pairmemarg: &mut PairAMode, collector: &mut impl OperandVisitor) {
358    match pairmemarg {
359        PairAMode::SignedOffset { reg, .. } => {
360            collector.reg_use(reg);
361        }
362        PairAMode::SPPreIndexed { .. } | PairAMode::SPPostIndexed { .. } => {}
363    }
364}
365
366fn aarch64_get_operands(inst: &mut Inst, collector: &mut impl OperandVisitor) {
367    match inst {
368        Inst::AluRRR { rd, rn, rm, .. } => {
369            collector.reg_def(rd);
370            collector.reg_use(rn);
371            collector.reg_use(rm);
372        }
373        Inst::AluRRRR { rd, rn, rm, ra, .. } => {
374            collector.reg_def(rd);
375            collector.reg_use(rn);
376            collector.reg_use(rm);
377            collector.reg_use(ra);
378        }
379        Inst::AluRRImm12 { rd, rn, .. } => {
380            collector.reg_def(rd);
381            collector.reg_use(rn);
382        }
383        Inst::AluRRImmLogic { rd, rn, .. } => {
384            collector.reg_def(rd);
385            collector.reg_use(rn);
386        }
387        Inst::AluRRImmShift { rd, rn, .. } => {
388            collector.reg_def(rd);
389            collector.reg_use(rn);
390        }
391        Inst::AluRRRShift { rd, rn, rm, .. } => {
392            collector.reg_def(rd);
393            collector.reg_use(rn);
394            collector.reg_use(rm);
395        }
396        Inst::AluRRRExtend { rd, rn, rm, .. } => {
397            collector.reg_def(rd);
398            collector.reg_use(rn);
399            collector.reg_use(rm);
400        }
401        Inst::BitRR { rd, rn, .. } => {
402            collector.reg_def(rd);
403            collector.reg_use(rn);
404        }
405        Inst::ULoad8 { rd, mem, .. }
406        | Inst::SLoad8 { rd, mem, .. }
407        | Inst::ULoad16 { rd, mem, .. }
408        | Inst::SLoad16 { rd, mem, .. }
409        | Inst::ULoad32 { rd, mem, .. }
410        | Inst::SLoad32 { rd, mem, .. }
411        | Inst::ULoad64 { rd, mem, .. } => {
412            collector.reg_def(rd);
413            memarg_operands(mem, collector);
414        }
415        Inst::Store8 { rd, mem, .. }
416        | Inst::Store16 { rd, mem, .. }
417        | Inst::Store32 { rd, mem, .. }
418        | Inst::Store64 { rd, mem, .. } => {
419            collector.reg_use(rd);
420            memarg_operands(mem, collector);
421        }
422        Inst::StoreP64 { rt, rt2, mem, .. } => {
423            collector.reg_use(rt);
424            collector.reg_use(rt2);
425            pairmemarg_operands(mem, collector);
426        }
427        Inst::LoadP64 { rt, rt2, mem, .. } => {
428            collector.reg_def(rt);
429            collector.reg_def(rt2);
430            pairmemarg_operands(mem, collector);
431        }
432        Inst::Mov { rd, rm, .. } => {
433            collector.reg_def(rd);
434            collector.reg_use(rm);
435        }
436        Inst::MovFromPReg { rd, rm } => {
437            debug_assert!(rd.to_reg().is_virtual());
438            collector.reg_def(rd);
439            collector.reg_fixed_nonallocatable(*rm);
440        }
441        Inst::MovToPReg { rd, rm } => {
442            debug_assert!(rm.is_virtual());
443            collector.reg_fixed_nonallocatable(*rd);
444            collector.reg_use(rm);
445        }
446        Inst::MovK { rd, rn, .. } => {
447            collector.reg_use(rn);
448            collector.reg_reuse_def(rd, 0); // `rn` == `rd`.
449        }
450        Inst::MovWide { rd, .. } => {
451            collector.reg_def(rd);
452        }
453        Inst::CSel { rd, rn, rm, .. } => {
454            collector.reg_def(rd);
455            collector.reg_use(rn);
456            collector.reg_use(rm);
457        }
458        Inst::CSNeg { rd, rn, rm, .. } => {
459            collector.reg_def(rd);
460            collector.reg_use(rn);
461            collector.reg_use(rm);
462        }
463        Inst::CSet { rd, .. } | Inst::CSetm { rd, .. } => {
464            collector.reg_def(rd);
465        }
466        Inst::CCmp { rn, rm, .. } => {
467            collector.reg_use(rn);
468            collector.reg_use(rm);
469        }
470        Inst::CCmpImm { rn, .. } => {
471            collector.reg_use(rn);
472        }
473        Inst::AtomicRMWLoop {
474            op,
475            addr,
476            operand,
477            oldval,
478            scratch1,
479            scratch2,
480            ..
481        } => {
482            collector.reg_fixed_use(addr, xreg(25));
483            collector.reg_fixed_use(operand, xreg(26));
484            collector.reg_fixed_def(oldval, xreg(27));
485            collector.reg_fixed_def(scratch1, xreg(24));
486            if *op != AtomicRMWLoopOp::Xchg {
487                collector.reg_fixed_def(scratch2, xreg(28));
488            }
489        }
490        Inst::AtomicRMW128Loop {
491            op,
492            addr,
493            operand_lo,
494            operand_hi,
495            oldval_lo,
496            oldval_hi,
497            scratch1,
498            scratch2,
499            scratch3,
500            ..
501        } => {
502            collector.reg_fixed_use(addr, xreg(25));
503            collector.reg_fixed_use(operand_lo, xreg(26));
504            collector.reg_fixed_use(operand_hi, xreg(22));
505            collector.reg_fixed_def(oldval_lo, xreg(27));
506            collector.reg_fixed_def(oldval_hi, xreg(23));
507            collector.reg_fixed_def(scratch1, xreg(24));
508            if *op != AtomicRMWLoopOp::Xchg {
509                collector.reg_fixed_def(scratch2, xreg(28));
510                collector.reg_fixed_def(scratch3, xreg(21));
511            }
512        }
513        Inst::AtomicRMW { rs, rt, rn, .. } => {
514            collector.reg_use(rs);
515            collector.reg_def(rt);
516            collector.reg_use(rn);
517        }
518        Inst::AtomicCAS { rd, rs, rt, rn, .. } => {
519            collector.reg_reuse_def(rd, 1); // reuse `rs`.
520            collector.reg_use(rs);
521            collector.reg_use(rt);
522            collector.reg_use(rn);
523        }
524        Inst::AtomicCAS128 { args } => {
525            let AtomicCAS128Args {
526                rd_lo,
527                rd_hi,
528                rs_lo,
529                rs_hi,
530                rt_lo,
531                rt_hi,
532                rn,
533                flags: _,
534            } = &mut **args;
535            // `casp` requires two consecutive even-aligned register pairs,
536            // which regalloc2 cannot express, so pin everything down.
537            collector.reg_fixed_use(rs_lo, xreg(24));
538            collector.reg_fixed_use(rs_hi, xreg(25));
539            collector.reg_fixed_def(rd_lo, xreg(24));
540            collector.reg_fixed_def(rd_hi, xreg(25));
541            collector.reg_fixed_use(rt_lo, xreg(26));
542            collector.reg_fixed_use(rt_hi, xreg(27));
543            collector.reg_fixed_use(rn, xreg(28));
544        }
545        Inst::AtomicCASLoop {
546            addr,
547            expected,
548            replacement,
549            oldval,
550            scratch,
551            ..
552        } => {
553            collector.reg_fixed_use(addr, xreg(25));
554            collector.reg_fixed_use(expected, xreg(26));
555            collector.reg_fixed_use(replacement, xreg(28));
556            collector.reg_fixed_def(oldval, xreg(27));
557            collector.reg_fixed_def(scratch, xreg(24));
558        }
559        Inst::AtomicCAS128Loop {
560            addr,
561            expected_lo,
562            expected_hi,
563            replacement_lo,
564            replacement_hi,
565            oldval_lo,
566            oldval_hi,
567            scratch,
568            ..
569        } => {
570            collector.reg_fixed_use(addr, xreg(25));
571            collector.reg_fixed_use(expected_lo, xreg(26));
572            collector.reg_fixed_use(expected_hi, xreg(23));
573            collector.reg_fixed_use(replacement_lo, xreg(28));
574            collector.reg_fixed_use(replacement_hi, xreg(22));
575            collector.reg_fixed_def(oldval_lo, xreg(27));
576            collector.reg_fixed_def(oldval_hi, xreg(21));
577            collector.reg_fixed_def(scratch, xreg(24));
578        }
579        Inst::LoadAcquire { rt, rn, .. } => {
580            collector.reg_use(rn);
581            collector.reg_def(rt);
582        }
583        Inst::LoadAcquire128 {
584            rt1,
585            rt2,
586            rn,
587            scratch,
588            ..
589        } => {
590            collector.reg_use(rn);
591            collector.reg_def(rt1);
592            collector.reg_def(rt2);
593            collector.reg_def(scratch);
594        }
595        Inst::StoreRelease { rt, rn, .. } => {
596            collector.reg_use(rn);
597            collector.reg_use(rt);
598        }
599        Inst::StoreRelease128 {
600            rt1,
601            rt2,
602            rn,
603            scratch,
604            ..
605        } => {
606            collector.reg_use(rn);
607            collector.reg_use(rt1);
608            collector.reg_use(rt2);
609            collector.reg_early_def(scratch);
610        }
611        Inst::Fence {} | Inst::Csdb {} => {}
612        Inst::FpuMove32 { rd, rn } => {
613            collector.reg_def(rd);
614            collector.reg_use(rn);
615        }
616        Inst::FpuMove64 { rd, rn } => {
617            collector.reg_def(rd);
618            collector.reg_use(rn);
619        }
620        Inst::FpuMove128 { rd, rn } => {
621            collector.reg_def(rd);
622            collector.reg_use(rn);
623        }
624        Inst::FpuMoveFromVec { rd, rn, .. } => {
625            collector.reg_def(rd);
626            collector.reg_use(rn);
627        }
628        Inst::FpuExtend { rd, rn, .. } => {
629            collector.reg_def(rd);
630            collector.reg_use(rn);
631        }
632        Inst::FpuRR { rd, rn, .. } => {
633            collector.reg_def(rd);
634            collector.reg_use(rn);
635        }
636        Inst::FpuRRR { rd, rn, rm, .. } => {
637            collector.reg_def(rd);
638            collector.reg_use(rn);
639            collector.reg_use(rm);
640        }
641        Inst::FpuRRI { rd, rn, .. } => {
642            collector.reg_def(rd);
643            collector.reg_use(rn);
644        }
645        Inst::FpuRRIMod { rd, ri, rn, .. } => {
646            collector.reg_reuse_def(rd, 1); // reuse `ri`.
647            collector.reg_use(ri);
648            collector.reg_use(rn);
649        }
650        Inst::FpuRRRR { rd, rn, rm, ra, .. } => {
651            collector.reg_def(rd);
652            collector.reg_use(rn);
653            collector.reg_use(rm);
654            collector.reg_use(ra);
655        }
656        Inst::VecMisc { rd, rn, .. } => {
657            collector.reg_def(rd);
658            collector.reg_use(rn);
659        }
660
661        Inst::VecLanes { rd, rn, .. } => {
662            collector.reg_def(rd);
663            collector.reg_use(rn);
664        }
665        Inst::VecShiftImm { rd, rn, .. } => {
666            collector.reg_def(rd);
667            collector.reg_use(rn);
668        }
669        Inst::VecShiftImmMod { rd, ri, rn, .. } => {
670            collector.reg_reuse_def(rd, 1); // `rd` == `ri`.
671            collector.reg_use(ri);
672            collector.reg_use(rn);
673        }
674        Inst::VecExtract { rd, rn, rm, .. } => {
675            collector.reg_def(rd);
676            collector.reg_use(rn);
677            collector.reg_use(rm);
678        }
679        Inst::VecTbl { rd, rn, rm } => {
680            collector.reg_use(rn);
681            collector.reg_use(rm);
682            collector.reg_def(rd);
683        }
684        Inst::VecTblExt { rd, ri, rn, rm } => {
685            collector.reg_use(rn);
686            collector.reg_use(rm);
687            collector.reg_reuse_def(rd, 3); // `rd` == `ri`.
688            collector.reg_use(ri);
689        }
690
691        Inst::VecTbl2 { rd, rn, rn2, rm } => {
692            // Constrain to v30 / v31 so that we satisfy the "adjacent
693            // registers" constraint without use of pinned vregs in
694            // lowering.
695            collector.reg_fixed_use(rn, vreg(30));
696            collector.reg_fixed_use(rn2, vreg(31));
697            collector.reg_use(rm);
698            collector.reg_def(rd);
699        }
700        Inst::VecTbl2Ext {
701            rd,
702            ri,
703            rn,
704            rn2,
705            rm,
706        } => {
707            // Constrain to v30 / v31 so that we satisfy the "adjacent
708            // registers" constraint without use of pinned vregs in
709            // lowering.
710            collector.reg_fixed_use(rn, vreg(30));
711            collector.reg_fixed_use(rn2, vreg(31));
712            collector.reg_use(rm);
713            collector.reg_reuse_def(rd, 4); // `rd` == `ri`.
714            collector.reg_use(ri);
715        }
716        Inst::VecLoadReplicate { rd, rn, .. } => {
717            collector.reg_def(rd);
718            collector.reg_use(rn);
719        }
720        Inst::VecCSel { rd, rn, rm, .. } => {
721            collector.reg_def(rd);
722            collector.reg_use(rn);
723            collector.reg_use(rm);
724        }
725        Inst::FpuCmp { rn, rm, .. } => {
726            collector.reg_use(rn);
727            collector.reg_use(rm);
728        }
729        Inst::FpuLoad16 { rd, mem, .. } => {
730            collector.reg_def(rd);
731            memarg_operands(mem, collector);
732        }
733        Inst::FpuLoad32 { rd, mem, .. } => {
734            collector.reg_def(rd);
735            memarg_operands(mem, collector);
736        }
737        Inst::FpuLoad64 { rd, mem, .. } => {
738            collector.reg_def(rd);
739            memarg_operands(mem, collector);
740        }
741        Inst::FpuLoad128 { rd, mem, .. } => {
742            collector.reg_def(rd);
743            memarg_operands(mem, collector);
744        }
745        Inst::FpuStore16 { rd, mem, .. } => {
746            collector.reg_use(rd);
747            memarg_operands(mem, collector);
748        }
749        Inst::FpuStore32 { rd, mem, .. } => {
750            collector.reg_use(rd);
751            memarg_operands(mem, collector);
752        }
753        Inst::FpuStore64 { rd, mem, .. } => {
754            collector.reg_use(rd);
755            memarg_operands(mem, collector);
756        }
757        Inst::FpuStore128 { rd, mem, .. } => {
758            collector.reg_use(rd);
759            memarg_operands(mem, collector);
760        }
761        Inst::FpuLoadP64 { rt, rt2, mem, .. } => {
762            collector.reg_def(rt);
763            collector.reg_def(rt2);
764            pairmemarg_operands(mem, collector);
765        }
766        Inst::FpuStoreP64 { rt, rt2, mem, .. } => {
767            collector.reg_use(rt);
768            collector.reg_use(rt2);
769            pairmemarg_operands(mem, collector);
770        }
771        Inst::FpuLoadP128 { rt, rt2, mem, .. } => {
772            collector.reg_def(rt);
773            collector.reg_def(rt2);
774            pairmemarg_operands(mem, collector);
775        }
776        Inst::FpuStoreP128 { rt, rt2, mem, .. } => {
777            collector.reg_use(rt);
778            collector.reg_use(rt2);
779            pairmemarg_operands(mem, collector);
780        }
781        Inst::FpuToInt { rd, rn, .. } => {
782            collector.reg_def(rd);
783            collector.reg_use(rn);
784        }
785        Inst::IntToFpu { rd, rn, .. } => {
786            collector.reg_def(rd);
787            collector.reg_use(rn);
788        }
789        Inst::FpuCSel16 { rd, rn, rm, .. }
790        | Inst::FpuCSel32 { rd, rn, rm, .. }
791        | Inst::FpuCSel64 { rd, rn, rm, .. } => {
792            collector.reg_def(rd);
793            collector.reg_use(rn);
794            collector.reg_use(rm);
795        }
796        Inst::FpuRound { rd, rn, .. } => {
797            collector.reg_def(rd);
798            collector.reg_use(rn);
799        }
800        Inst::MovToFpu { rd, rn, .. } => {
801            collector.reg_def(rd);
802            collector.reg_use(rn);
803        }
804        Inst::FpuMoveFPImm { rd, .. } => {
805            collector.reg_def(rd);
806        }
807        Inst::MovToVec { rd, ri, rn, .. } => {
808            collector.reg_reuse_def(rd, 1); // `rd` == `ri`.
809            collector.reg_use(ri);
810            collector.reg_use(rn);
811        }
812        Inst::MovFromVec { rd, rn, .. } | Inst::MovFromVecSigned { rd, rn, .. } => {
813            collector.reg_def(rd);
814            collector.reg_use(rn);
815        }
816        Inst::VecDup { rd, rn, .. } => {
817            collector.reg_def(rd);
818            collector.reg_use(rn);
819        }
820        Inst::VecDupFromFpu { rd, rn, .. } => {
821            collector.reg_def(rd);
822            collector.reg_use(rn);
823        }
824        Inst::VecDupFPImm { rd, .. } => {
825            collector.reg_def(rd);
826        }
827        Inst::VecDupImm { rd, .. } => {
828            collector.reg_def(rd);
829        }
830        Inst::VecExtend { rd, rn, .. } => {
831            collector.reg_def(rd);
832            collector.reg_use(rn);
833        }
834        Inst::VecMovElement { rd, ri, rn, .. } => {
835            collector.reg_reuse_def(rd, 1); // `rd` == `ri`.
836            collector.reg_use(ri);
837            collector.reg_use(rn);
838        }
839        Inst::VecRRLong { rd, rn, .. } => {
840            collector.reg_def(rd);
841            collector.reg_use(rn);
842        }
843        Inst::VecRRNarrowLow { rd, rn, .. } => {
844            collector.reg_use(rn);
845            collector.reg_def(rd);
846        }
847        Inst::VecRRNarrowHigh { rd, ri, rn, .. } => {
848            collector.reg_use(rn);
849            collector.reg_reuse_def(rd, 2); // `rd` == `ri`.
850            collector.reg_use(ri);
851        }
852        Inst::VecRRPair { rd, rn, .. } => {
853            collector.reg_def(rd);
854            collector.reg_use(rn);
855        }
856        Inst::VecRRRLong { rd, rn, rm, .. } => {
857            collector.reg_def(rd);
858            collector.reg_use(rn);
859            collector.reg_use(rm);
860        }
861        Inst::VecRRRLongMod { rd, ri, rn, rm, .. } => {
862            collector.reg_reuse_def(rd, 1); // `rd` == `ri`.
863            collector.reg_use(ri);
864            collector.reg_use(rn);
865            collector.reg_use(rm);
866        }
867        Inst::VecRRPairLong { rd, rn, .. } => {
868            collector.reg_def(rd);
869            collector.reg_use(rn);
870        }
871        Inst::VecRRR { rd, rn, rm, .. } => {
872            collector.reg_def(rd);
873            collector.reg_use(rn);
874            collector.reg_use(rm);
875        }
876        Inst::VecRRRMod { rd, ri, rn, rm, .. } | Inst::VecFmlaElem { rd, ri, rn, rm, .. } => {
877            collector.reg_reuse_def(rd, 1); // `rd` == `ri`.
878            collector.reg_use(ri);
879            collector.reg_use(rn);
880            collector.reg_use(rm);
881        }
882        Inst::MovToNZCV { rn } => {
883            collector.reg_use(rn);
884        }
885        Inst::MovFromNZCV { rd } => {
886            collector.reg_def(rd);
887        }
888        Inst::Extend { rd, rn, .. } => {
889            collector.reg_def(rd);
890            collector.reg_use(rn);
891        }
892        Inst::BitfieldMove { rd, rn, .. } => {
893            // The UBFM and SBFM instructions overwrite all bits in `rd`,
894            // unlike BFM which is represented as `BitfieldMoveMod` instead.
895            collector.reg_def(rd);
896            collector.reg_use(rn);
897        }
898        Inst::BitfieldMoveMod { rd, ri, rn, .. } => {
899            collector.reg_reuse_def(rd, 1); // `rd` == `ri`.
900            collector.reg_use(ri);
901            collector.reg_use(rn);
902        }
903        Inst::Args { args } => {
904            for ArgPair { vreg, preg } in args {
905                collector.reg_fixed_def(vreg, *preg);
906            }
907        }
908        Inst::Rets { rets } => {
909            for RetPair { vreg, preg } in rets {
910                collector.reg_fixed_use(vreg, *preg);
911            }
912        }
913        Inst::Ret { .. } | Inst::AuthenticatedRet { .. } => {}
914        Inst::Jump { .. } => {}
915        Inst::Call { info, .. } => {
916            let CallInfo { uses, defs, .. } = &mut **info;
917            for CallArgPair { vreg, preg } in uses {
918                collector.reg_fixed_use(vreg, *preg);
919            }
920            for CallRetPair { vreg, location } in defs {
921                match location {
922                    RetLocation::Reg(preg, ..) => collector.reg_fixed_def(vreg, *preg),
923                    RetLocation::Stack(..) => collector.any_def(vreg),
924                }
925            }
926            collector.reg_clobbers(info.clobbers);
927            if let Some(try_call_info) = &mut info.try_call_info {
928                try_call_info.collect_operands(collector);
929            }
930        }
931        Inst::CallInd { info, .. } => {
932            let CallInfo {
933                dest, uses, defs, ..
934            } = &mut **info;
935            collector.reg_use(dest);
936            for CallArgPair { vreg, preg } in uses {
937                collector.reg_fixed_use(vreg, *preg);
938            }
939            for CallRetPair { vreg, location } in defs {
940                match location {
941                    RetLocation::Reg(preg, ..) => collector.reg_fixed_def(vreg, *preg),
942                    RetLocation::Stack(..) => collector.any_def(vreg),
943                }
944            }
945            collector.reg_clobbers(info.clobbers);
946            if let Some(try_call_info) = &mut info.try_call_info {
947                try_call_info.collect_operands(collector);
948            }
949        }
950        Inst::ReturnCall { info } => {
951            for CallArgPair { vreg, preg } in &mut info.uses {
952                collector.reg_fixed_use(vreg, *preg);
953            }
954        }
955        Inst::ReturnCallInd { info } => {
956            // TODO(https://github.com/bytecodealliance/regalloc2/issues/145):
957            // This shouldn't be a fixed register constraint, but it's not clear how to pick a
958            // register that won't be clobbered by the callee-save restore code emitted with a
959            // return_call_indirect.
960            collector.reg_fixed_use(&mut info.dest, xreg(1));
961            for CallArgPair { vreg, preg } in &mut info.uses {
962                collector.reg_fixed_use(vreg, *preg);
963            }
964        }
965        Inst::CondBr { kind, .. } => match kind {
966            CondBrKind::Zero(rt, _) | CondBrKind::NotZero(rt, _) => collector.reg_use(rt),
967            CondBrKind::Cond(_) => {}
968        },
969        Inst::TestBitAndBranch { rn, .. } => {
970            collector.reg_use(rn);
971        }
972        Inst::IndirectBr { rn, .. } => {
973            collector.reg_use(rn);
974        }
975        Inst::Nop0 | Inst::Nop4 => {}
976        Inst::Brk => {}
977        Inst::Udf { .. } => {}
978        Inst::TrapIf { kind, .. } => match kind {
979            CondBrKind::Zero(rt, _) | CondBrKind::NotZero(rt, _) => collector.reg_use(rt),
980            CondBrKind::Cond(_) => {}
981        },
982        Inst::Adr { rd, .. } | Inst::Adrp { rd, .. } => {
983            collector.reg_def(rd);
984        }
985        Inst::Word4 { .. } | Inst::Word8 { .. } => {}
986        Inst::JTSequence {
987            ridx, rtmp1, rtmp2, ..
988        } => {
989            collector.reg_use(ridx);
990            collector.reg_early_def(rtmp1);
991            collector.reg_early_def(rtmp2);
992        }
993        Inst::LoadExtNameGot { rd, .. }
994        | Inst::LoadExtNameNear { rd, .. }
995        | Inst::LoadExtNameFar { rd, .. } => {
996            collector.reg_def(rd);
997        }
998        Inst::LoadAddr { rd, mem } => {
999            collector.reg_def(rd);
1000            memarg_operands(mem, collector);
1001        }
1002        Inst::Paci { .. } | Inst::Xpaclri => {
1003            // Neither LR nor SP is an allocatable register, so there is no need
1004            // to do anything.
1005        }
1006        Inst::Bti { .. } => {}
1007
1008        Inst::ElfTlsGetAddr { rd, tmp, .. } => {
1009            // TLSDESC has a very neat calling convention. It is required to preserve
1010            // all registers except x0 and x30. X30 is non allocatable in cranelift since
1011            // its the link register.
1012            //
1013            // Additionally we need a second register as a temporary register for the
1014            // TLSDESC sequence. This register can be any register other than x0 (and x30).
1015            collector.reg_fixed_def(rd, regs::xreg(0));
1016            collector.reg_early_def(tmp);
1017        }
1018        Inst::MachOTlsGetAddr { rd, .. } => {
1019            collector.reg_fixed_def(rd, regs::xreg(0));
1020            let mut clobbers =
1021                AArch64MachineDeps::get_regs_clobbered_by_call(CallConv::AppleAarch64, false);
1022            clobbers.remove(regs::xreg_preg(0));
1023            collector.reg_clobbers(clobbers);
1024        }
1025        Inst::Unwind { .. } => {}
1026        Inst::EmitIsland { .. } => {}
1027        Inst::DummyUse { reg } => {
1028            collector.reg_use(reg);
1029        }
1030        Inst::LabelAddress { dst, .. } => {
1031            collector.reg_def(dst);
1032        }
1033        Inst::SequencePoint { .. } => {}
1034        Inst::StackProbeLoop { start, end, .. } => {
1035            collector.reg_early_def(start);
1036            collector.reg_use(end);
1037        }
1038    }
1039}
1040
1041//=============================================================================
1042// Instructions: misc functions and external interface
1043
1044impl MachInst for Inst {
1045    type ABIMachineSpec = AArch64MachineDeps;
1046    type LabelUse = LabelUse;
1047
1048    // "CLIF" in hex, to make the trap recognizable during
1049    // debugging.
1050    const TRAP_OPCODE: &'static [u8] = &0xc11f_u32.to_le_bytes();
1051
1052    fn get_operands(&mut self, collector: &mut impl OperandVisitor) {
1053        aarch64_get_operands(self, collector);
1054    }
1055
1056    fn is_move(&self) -> Option<(Writable<Reg>, Reg)> {
1057        match self {
1058            &Inst::Mov {
1059                size: OperandSize::Size64,
1060                rd,
1061                rm,
1062            } => Some((rd, rm)),
1063            &Inst::FpuMove64 { rd, rn } => Some((rd, rn)),
1064            &Inst::FpuMove128 { rd, rn } => Some((rd, rn)),
1065            _ => None,
1066        }
1067    }
1068
1069    fn is_included_in_clobbers(&self) -> bool {
1070        let (caller, callee, is_exception) = match self {
1071            Inst::Args { .. } => return false,
1072            Inst::Call { info } => (
1073                info.caller_conv,
1074                info.callee_conv,
1075                info.try_call_info.is_some(),
1076            ),
1077            Inst::CallInd { info } => (
1078                info.caller_conv,
1079                info.callee_conv,
1080                info.try_call_info.is_some(),
1081            ),
1082            _ => return true,
1083        };
1084
1085        // We exclude call instructions from the clobber-set when they are calls
1086        // from caller to callee that both clobber the same register (such as
1087        // using the same or similar ABIs). Such calls cannot possibly force any
1088        // new registers to be saved in the prologue, because anything that the
1089        // callee clobbers, the caller is also allowed to clobber. This both
1090        // saves work and enables us to more precisely follow the
1091        // half-caller-save, half-callee-save SysV ABI for some vector
1092        // registers.
1093        //
1094        // See the note in [crate::isa::aarch64::abi::is_caller_save_reg] for
1095        // more information on this ABI-implementation hack.
1096        let caller_clobbers = AArch64MachineDeps::get_regs_clobbered_by_call(caller, false);
1097        let callee_clobbers = AArch64MachineDeps::get_regs_clobbered_by_call(callee, is_exception);
1098
1099        let mut all_clobbers = caller_clobbers;
1100        all_clobbers.union_from(callee_clobbers);
1101        all_clobbers != caller_clobbers
1102    }
1103
1104    fn is_trap(&self) -> bool {
1105        match self {
1106            Self::Udf { .. } => true,
1107            _ => false,
1108        }
1109    }
1110
1111    fn is_args(&self) -> bool {
1112        match self {
1113            Self::Args { .. } => true,
1114            _ => false,
1115        }
1116    }
1117
1118    fn call_type(&self) -> CallType {
1119        match self {
1120            Inst::Call { .. }
1121            | Inst::CallInd { .. }
1122            | Inst::ElfTlsGetAddr { .. }
1123            | Inst::MachOTlsGetAddr { .. } => CallType::Regular,
1124
1125            Inst::ReturnCall { .. } | Inst::ReturnCallInd { .. } => CallType::TailCall,
1126
1127            _ => CallType::None,
1128        }
1129    }
1130
1131    fn is_term(&self) -> MachTerminator {
1132        match self {
1133            &Inst::Rets { .. } => MachTerminator::Ret,
1134            &Inst::ReturnCall { .. } | &Inst::ReturnCallInd { .. } => MachTerminator::RetCall,
1135            &Inst::Jump { .. } => MachTerminator::Branch,
1136            &Inst::CondBr { .. } => MachTerminator::Branch,
1137            &Inst::TestBitAndBranch { .. } => MachTerminator::Branch,
1138            &Inst::IndirectBr { .. } => MachTerminator::Branch,
1139            &Inst::JTSequence { .. } => MachTerminator::Branch,
1140            &Inst::Call { ref info } if info.try_call_info.is_some() => MachTerminator::Branch,
1141            &Inst::CallInd { ref info } if info.try_call_info.is_some() => MachTerminator::Branch,
1142            _ => MachTerminator::None,
1143        }
1144    }
1145
1146    fn is_mem_access(&self) -> bool {
1147        match self {
1148            &Inst::ULoad8 { .. }
1149            | &Inst::SLoad8 { .. }
1150            | &Inst::ULoad16 { .. }
1151            | &Inst::SLoad16 { .. }
1152            | &Inst::ULoad32 { .. }
1153            | &Inst::SLoad32 { .. }
1154            | &Inst::ULoad64 { .. }
1155            | &Inst::LoadP64 { .. }
1156            | &Inst::FpuLoad16 { .. }
1157            | &Inst::FpuLoad32 { .. }
1158            | &Inst::FpuLoad64 { .. }
1159            | &Inst::FpuLoad128 { .. }
1160            | &Inst::FpuLoadP64 { .. }
1161            | &Inst::FpuLoadP128 { .. }
1162            | &Inst::Store8 { .. }
1163            | &Inst::Store16 { .. }
1164            | &Inst::Store32 { .. }
1165            | &Inst::Store64 { .. }
1166            | &Inst::StoreP64 { .. }
1167            | &Inst::FpuStore16 { .. }
1168            | &Inst::FpuStore32 { .. }
1169            | &Inst::FpuStore64 { .. }
1170            | &Inst::FpuStore128 { .. } => true,
1171            // TODO: verify this carefully
1172            _ => false,
1173        }
1174    }
1175
1176    fn gen_move(to_reg: Writable<Reg>, from_reg: Reg, ty: Type) -> Inst {
1177        let bits = ty.bits();
1178
1179        assert!(bits <= 128);
1180        assert!(to_reg.to_reg().class() == from_reg.class());
1181        match from_reg.class() {
1182            RegClass::Int => Inst::Mov {
1183                size: OperandSize::Size64,
1184                rd: to_reg,
1185                rm: from_reg,
1186            },
1187            RegClass::Float => {
1188                if bits > 64 {
1189                    Inst::FpuMove128 {
1190                        rd: to_reg,
1191                        rn: from_reg,
1192                    }
1193                } else {
1194                    Inst::FpuMove64 {
1195                        rd: to_reg,
1196                        rn: from_reg,
1197                    }
1198                }
1199            }
1200            RegClass::Vector => unreachable!(),
1201        }
1202    }
1203
1204    fn is_safepoint(&self) -> bool {
1205        match self {
1206            Inst::Call { .. } | Inst::CallInd { .. } => true,
1207            _ => false,
1208        }
1209    }
1210
1211    fn gen_dummy_use(reg: Reg) -> Inst {
1212        Inst::DummyUse { reg }
1213    }
1214
1215    fn gen_nop(preferred_size: usize) -> Inst {
1216        if preferred_size == 0 {
1217            return Inst::Nop0;
1218        }
1219        // We can't give a NOP (or any insn) < 4 bytes.
1220        assert!(preferred_size >= 4);
1221        Inst::Nop4
1222    }
1223
1224    fn gen_nop_units() -> Vec<Vec<u8>> {
1225        vec![vec![0x1f, 0x20, 0x03, 0xd5]]
1226    }
1227
1228    fn rc_for_type(ty: &Type) -> CodegenResult<(&[RegClass], &[Type])> {
1229        match *ty {
1230            I8 | I16 | I32 | I64 => Ok((&[RegClass::Int], slice::from_ref(ty))),
1231            F16 | F32 | F64 | F128 => Ok((&[RegClass::Float], slice::from_ref(ty))),
1232            I128 => Ok((&[RegClass::Int, RegClass::Int], &[I64, I64])),
1233            _ if ty.is_vector() && ty.bits() <= 128 => {
1234                let types = &[types::I8X2, types::I8X4, types::I8X8, types::I8X16];
1235                Ok((
1236                    &[RegClass::Float],
1237                    slice::from_ref(&types[ty.bytes().ilog2() as usize - 1]),
1238                ))
1239            }
1240            _ if ty.is_dynamic_vector() => Ok((&[RegClass::Float], &[I8X16])),
1241            _ => Err(CodegenError::Unsupported(format!(
1242                "Unexpected SSA-value type: {ty}"
1243            ))),
1244        }
1245    }
1246
1247    fn canonical_type_for_rc(rc: RegClass) -> Type {
1248        match rc {
1249            RegClass::Float => types::I8X16,
1250            RegClass::Int => types::I64,
1251            RegClass::Vector => unreachable!(),
1252        }
1253    }
1254
1255    fn gen_jump(target: MachLabel) -> Inst {
1256        Inst::Jump {
1257            dest: BranchTarget::Label(target),
1258        }
1259    }
1260
1261    fn worst_case_size() -> CodeOffset {
1262        // The maximum size, in bytes, of any `Inst`'s emitted code. We have at least one case of
1263        // an 8-instruction sequence (saturating int-to-float conversions) with three embedded
1264        // 64-bit f64 constants.
1265        //
1266        // Note that inline jump-tables handle island/pool insertion separately, so we do not need
1267        // to account for them here (otherwise the worst case would be 2^31 * 4, clearly not
1268        // feasible for other reasons).
1269        44
1270    }
1271
1272    fn worst_case_island_growth() -> CodeOffset {
1273        // A single `Inst` may add to the buffer's pending-island state:
1274        //
1275        // - Up to three 8-byte constants (the saturating int-to-float sequence
1276        //   noted above); count alignment padding into each.
1277        // - Up to one deferred trap (TrapIf and similar), 4 bytes.
1278        // - Up to one fixup per emitted instruction word, each contributing at
1279        //   most `worst_case_veneer_size()` (= 20) bytes of veneer.
1280        //
1281        // We pick a conservative bound that comfortably covers these.
1282        128
1283    }
1284
1285    fn gen_block_start(
1286        is_indirect_branch_target: bool,
1287        is_forward_edge_cfi_enabled: bool,
1288    ) -> Option<Self> {
1289        if is_indirect_branch_target && is_forward_edge_cfi_enabled {
1290            Some(Inst::Bti {
1291                targets: BranchTargetType::J,
1292            })
1293        } else {
1294            None
1295        }
1296    }
1297
1298    fn function_alignment() -> FunctionAlignment {
1299        // We use 32-byte alignment for performance reasons, but for correctness
1300        // we would only need 4-byte alignment.
1301        FunctionAlignment {
1302            minimum: 4,
1303            preferred: 32,
1304        }
1305    }
1306}
1307
1308//=============================================================================
1309// Pretty-printing of instructions.
1310
1311fn mem_finalize_for_show(mem: &AMode, access_ty: Type, state: &EmitState) -> (String, String) {
1312    let (mem_insts, mem) = mem_finalize(None, mem, access_ty, state);
1313    let mut mem_str = mem_insts
1314        .into_iter()
1315        .map(|inst| inst.print_with_state(&mut EmitState::default()))
1316        .collect::<Vec<_>>()
1317        .join(" ; ");
1318    if !mem_str.is_empty() {
1319        mem_str += " ; ";
1320    }
1321
1322    let mem = mem.pretty_print(access_ty.bytes() as u8);
1323    (mem_str, mem)
1324}
1325
1326fn pretty_print_try_call(info: &TryCallInfo) -> String {
1327    format!(
1328        "; b {:?}; catch [{}]",
1329        info.continuation,
1330        info.pretty_print_dests()
1331    )
1332}
1333
1334impl Inst {
1335    #[expect(
1336        missing_docs,
1337        reason = "exposed for cranelift-isle/veri pretty-printing"
1338    )]
1339    pub fn print_with_state(&self, state: &mut EmitState) -> String {
1340        fn op_name(alu_op: ALUOp) -> &'static str {
1341            match alu_op {
1342                ALUOp::Add => "add",
1343                ALUOp::Sub => "sub",
1344                ALUOp::Orr => "orr",
1345                ALUOp::And => "and",
1346                ALUOp::AndS => "ands",
1347                ALUOp::Eor => "eor",
1348                ALUOp::AddS => "adds",
1349                ALUOp::SubS => "subs",
1350                ALUOp::SMulH => "smulh",
1351                ALUOp::UMulH => "umulh",
1352                ALUOp::SDiv => "sdiv",
1353                ALUOp::UDiv => "udiv",
1354                ALUOp::AndNot => "bic",
1355                ALUOp::OrrNot => "orn",
1356                ALUOp::EorNot => "eon",
1357                ALUOp::Extr => "extr",
1358                ALUOp::Lsr => "lsr",
1359                ALUOp::Asr => "asr",
1360                ALUOp::Lsl => "lsl",
1361                ALUOp::Adc => "adc",
1362                ALUOp::AdcS => "adcs",
1363                ALUOp::Sbc => "sbc",
1364                ALUOp::SbcS => "sbcs",
1365            }
1366        }
1367
1368        match self {
1369            &Inst::Nop0 => "nop-zero-len".to_string(),
1370            &Inst::Nop4 => "nop".to_string(),
1371            &Inst::AluRRR {
1372                alu_op,
1373                size,
1374                rd,
1375                rn,
1376                rm,
1377            } => {
1378                let op = op_name(alu_op);
1379                let rd = pretty_print_ireg(rd.to_reg(), size);
1380                let rn = pretty_print_ireg(rn, size);
1381                let rm = pretty_print_ireg(rm, size);
1382                format!("{op} {rd}, {rn}, {rm}")
1383            }
1384            &Inst::AluRRRR {
1385                alu_op,
1386                size,
1387                rd,
1388                rn,
1389                rm,
1390                ra,
1391            } => {
1392                let (op, da_size) = match alu_op {
1393                    ALUOp3::MAdd => ("madd", size),
1394                    ALUOp3::MSub => ("msub", size),
1395                    ALUOp3::UMAddL => ("umaddl", OperandSize::Size64),
1396                    ALUOp3::SMAddL => ("smaddl", OperandSize::Size64),
1397                };
1398                let rd = pretty_print_ireg(rd.to_reg(), da_size);
1399                let rn = pretty_print_ireg(rn, size);
1400                let rm = pretty_print_ireg(rm, size);
1401                let ra = pretty_print_ireg(ra, da_size);
1402
1403                format!("{op} {rd}, {rn}, {rm}, {ra}")
1404            }
1405            &Inst::AluRRImm12 {
1406                alu_op,
1407                size,
1408                rd,
1409                rn,
1410                ref imm12,
1411            } => {
1412                let op = op_name(alu_op);
1413                let rd = pretty_print_ireg(rd.to_reg(), size);
1414                let rn = pretty_print_ireg(rn, size);
1415
1416                if imm12.bits == 0 && alu_op == ALUOp::Add && size.is64() {
1417                    // special-case MOV (used for moving into SP).
1418                    format!("mov {rd}, {rn}")
1419                } else {
1420                    let imm12 = imm12.pretty_print(0);
1421                    format!("{op} {rd}, {rn}, {imm12}")
1422                }
1423            }
1424            &Inst::AluRRImmLogic {
1425                alu_op,
1426                size,
1427                rd,
1428                rn,
1429                ref imml,
1430            } => {
1431                let op = op_name(alu_op);
1432                let rd = pretty_print_ireg(rd.to_reg(), size);
1433                let rn = pretty_print_ireg(rn, size);
1434                let imml = imml.pretty_print(0);
1435                format!("{op} {rd}, {rn}, {imml}")
1436            }
1437            &Inst::AluRRImmShift {
1438                alu_op,
1439                size,
1440                rd,
1441                rn,
1442                ref immshift,
1443            } => {
1444                let op = op_name(alu_op);
1445                let rd = pretty_print_ireg(rd.to_reg(), size);
1446                let rn = pretty_print_ireg(rn, size);
1447                let immshift = immshift.pretty_print(0);
1448                format!("{op} {rd}, {rn}, {immshift}")
1449            }
1450            &Inst::AluRRRShift {
1451                alu_op,
1452                size,
1453                rd,
1454                rn,
1455                rm,
1456                ref shiftop,
1457            } => {
1458                let op = op_name(alu_op);
1459                let rd = pretty_print_ireg(rd.to_reg(), size);
1460                let rn = pretty_print_ireg(rn, size);
1461                let rm = pretty_print_ireg(rm, size);
1462                let shiftop = shiftop.pretty_print(0);
1463                format!("{op} {rd}, {rn}, {rm}, {shiftop}")
1464            }
1465            &Inst::AluRRRExtend {
1466                alu_op,
1467                size,
1468                rd,
1469                rn,
1470                rm,
1471                ref extendop,
1472            } => {
1473                let op = op_name(alu_op);
1474                let rd = pretty_print_ireg(rd.to_reg(), size);
1475                let rn = pretty_print_ireg(rn, size);
1476                let rm = pretty_print_ireg(rm, size);
1477                let extendop = extendop.pretty_print(0);
1478                format!("{op} {rd}, {rn}, {rm}, {extendop}")
1479            }
1480            &Inst::BitRR { op, size, rd, rn } => {
1481                let op = op.op_str();
1482                let rd = pretty_print_ireg(rd.to_reg(), size);
1483                let rn = pretty_print_ireg(rn, size);
1484                format!("{op} {rd}, {rn}")
1485            }
1486            &Inst::ULoad8 { rd, ref mem, .. }
1487            | &Inst::SLoad8 { rd, ref mem, .. }
1488            | &Inst::ULoad16 { rd, ref mem, .. }
1489            | &Inst::SLoad16 { rd, ref mem, .. }
1490            | &Inst::ULoad32 { rd, ref mem, .. }
1491            | &Inst::SLoad32 { rd, ref mem, .. }
1492            | &Inst::ULoad64 { rd, ref mem, .. } => {
1493                let is_unscaled = match &mem {
1494                    &AMode::Unscaled { .. } => true,
1495                    _ => false,
1496                };
1497                let (op, size) = match (self, is_unscaled) {
1498                    (&Inst::ULoad8 { .. }, false) => ("ldrb", OperandSize::Size32),
1499                    (&Inst::ULoad8 { .. }, true) => ("ldurb", OperandSize::Size32),
1500                    (&Inst::SLoad8 { .. }, false) => ("ldrsb", OperandSize::Size64),
1501                    (&Inst::SLoad8 { .. }, true) => ("ldursb", OperandSize::Size64),
1502                    (&Inst::ULoad16 { .. }, false) => ("ldrh", OperandSize::Size32),
1503                    (&Inst::ULoad16 { .. }, true) => ("ldurh", OperandSize::Size32),
1504                    (&Inst::SLoad16 { .. }, false) => ("ldrsh", OperandSize::Size64),
1505                    (&Inst::SLoad16 { .. }, true) => ("ldursh", OperandSize::Size64),
1506                    (&Inst::ULoad32 { .. }, false) => ("ldr", OperandSize::Size32),
1507                    (&Inst::ULoad32 { .. }, true) => ("ldur", OperandSize::Size32),
1508                    (&Inst::SLoad32 { .. }, false) => ("ldrsw", OperandSize::Size64),
1509                    (&Inst::SLoad32 { .. }, true) => ("ldursw", OperandSize::Size64),
1510                    (&Inst::ULoad64 { .. }, false) => ("ldr", OperandSize::Size64),
1511                    (&Inst::ULoad64 { .. }, true) => ("ldur", OperandSize::Size64),
1512                    _ => unreachable!(),
1513                };
1514
1515                let rd = pretty_print_ireg(rd.to_reg(), size);
1516                let mem = mem.clone();
1517                let access_ty = self.mem_type().unwrap();
1518                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
1519
1520                format!("{mem_str}{op} {rd}, {mem}")
1521            }
1522            &Inst::Store8 { rd, ref mem, .. }
1523            | &Inst::Store16 { rd, ref mem, .. }
1524            | &Inst::Store32 { rd, ref mem, .. }
1525            | &Inst::Store64 { rd, ref mem, .. } => {
1526                let is_unscaled = match &mem {
1527                    &AMode::Unscaled { .. } => true,
1528                    _ => false,
1529                };
1530                let (op, size) = match (self, is_unscaled) {
1531                    (&Inst::Store8 { .. }, false) => ("strb", OperandSize::Size32),
1532                    (&Inst::Store8 { .. }, true) => ("sturb", OperandSize::Size32),
1533                    (&Inst::Store16 { .. }, false) => ("strh", OperandSize::Size32),
1534                    (&Inst::Store16 { .. }, true) => ("sturh", OperandSize::Size32),
1535                    (&Inst::Store32 { .. }, false) => ("str", OperandSize::Size32),
1536                    (&Inst::Store32 { .. }, true) => ("stur", OperandSize::Size32),
1537                    (&Inst::Store64 { .. }, false) => ("str", OperandSize::Size64),
1538                    (&Inst::Store64 { .. }, true) => ("stur", OperandSize::Size64),
1539                    _ => unreachable!(),
1540                };
1541
1542                let rd = pretty_print_ireg(rd, size);
1543                let mem = mem.clone();
1544                let access_ty = self.mem_type().unwrap();
1545                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
1546
1547                format!("{mem_str}{op} {rd}, {mem}")
1548            }
1549            &Inst::StoreP64 {
1550                rt, rt2, ref mem, ..
1551            } => {
1552                let rt = pretty_print_ireg(rt, OperandSize::Size64);
1553                let rt2 = pretty_print_ireg(rt2, OperandSize::Size64);
1554                let mem = mem.clone();
1555                let mem = mem.pretty_print_default();
1556                format!("stp {rt}, {rt2}, {mem}")
1557            }
1558            &Inst::LoadP64 {
1559                rt, rt2, ref mem, ..
1560            } => {
1561                let rt = pretty_print_ireg(rt.to_reg(), OperandSize::Size64);
1562                let rt2 = pretty_print_ireg(rt2.to_reg(), OperandSize::Size64);
1563                let mem = mem.clone();
1564                let mem = mem.pretty_print_default();
1565                format!("ldp {rt}, {rt2}, {mem}")
1566            }
1567            &Inst::Mov { size, rd, rm } => {
1568                let rd = pretty_print_ireg(rd.to_reg(), size);
1569                let rm = pretty_print_ireg(rm, size);
1570                format!("mov {rd}, {rm}")
1571            }
1572            &Inst::MovFromPReg { rd, rm } => {
1573                let rd = pretty_print_ireg(rd.to_reg(), OperandSize::Size64);
1574                let rm = show_ireg_sized(rm.into(), OperandSize::Size64);
1575                format!("mov {rd}, {rm}")
1576            }
1577            &Inst::MovToPReg { rd, rm } => {
1578                let rd = show_ireg_sized(rd.into(), OperandSize::Size64);
1579                let rm = pretty_print_ireg(rm, OperandSize::Size64);
1580                format!("mov {rd}, {rm}")
1581            }
1582            &Inst::MovWide {
1583                op,
1584                rd,
1585                ref imm,
1586                size,
1587            } => {
1588                let op_str = match op {
1589                    MoveWideOp::MovZ => "movz",
1590                    MoveWideOp::MovN => "movn",
1591                };
1592                let rd = pretty_print_ireg(rd.to_reg(), size);
1593                let imm = imm.pretty_print(0);
1594                format!("{op_str} {rd}, {imm}")
1595            }
1596            &Inst::MovK {
1597                rd,
1598                rn,
1599                ref imm,
1600                size,
1601            } => {
1602                let rn = pretty_print_ireg(rn, size);
1603                let rd = pretty_print_ireg(rd.to_reg(), size);
1604                let imm = imm.pretty_print(0);
1605                format!("movk {rd}, {rn}, {imm}")
1606            }
1607            &Inst::CSel { rd, rn, rm, cond } => {
1608                let rd = pretty_print_ireg(rd.to_reg(), OperandSize::Size64);
1609                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1610                let rm = pretty_print_ireg(rm, OperandSize::Size64);
1611                let cond = cond.pretty_print(0);
1612                format!("csel {rd}, {rn}, {rm}, {cond}")
1613            }
1614            &Inst::CSNeg { rd, rn, rm, cond } => {
1615                let rd = pretty_print_ireg(rd.to_reg(), OperandSize::Size64);
1616                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1617                let rm = pretty_print_ireg(rm, OperandSize::Size64);
1618                let cond = cond.pretty_print(0);
1619                format!("csneg {rd}, {rn}, {rm}, {cond}")
1620            }
1621            &Inst::CSet { rd, cond } => {
1622                let rd = pretty_print_ireg(rd.to_reg(), OperandSize::Size64);
1623                let cond = cond.pretty_print(0);
1624                format!("cset {rd}, {cond}")
1625            }
1626            &Inst::CSetm { rd, cond } => {
1627                let rd = pretty_print_ireg(rd.to_reg(), OperandSize::Size64);
1628                let cond = cond.pretty_print(0);
1629                format!("csetm {rd}, {cond}")
1630            }
1631            &Inst::CCmp {
1632                size,
1633                rn,
1634                rm,
1635                nzcv,
1636                cond,
1637            } => {
1638                let rn = pretty_print_ireg(rn, size);
1639                let rm = pretty_print_ireg(rm, size);
1640                let nzcv = nzcv.pretty_print(0);
1641                let cond = cond.pretty_print(0);
1642                format!("ccmp {rn}, {rm}, {nzcv}, {cond}")
1643            }
1644            &Inst::CCmpImm {
1645                size,
1646                rn,
1647                imm,
1648                nzcv,
1649                cond,
1650            } => {
1651                let rn = pretty_print_ireg(rn, size);
1652                let imm = imm.pretty_print(0);
1653                let nzcv = nzcv.pretty_print(0);
1654                let cond = cond.pretty_print(0);
1655                format!("ccmp {rn}, {imm}, {nzcv}, {cond}")
1656            }
1657            &Inst::AtomicRMW {
1658                rs, rt, rn, ty, op, ..
1659            } => {
1660                let op = match op {
1661                    AtomicRMWOp::Add => "ldaddal",
1662                    AtomicRMWOp::Clr => "ldclral",
1663                    AtomicRMWOp::Eor => "ldeoral",
1664                    AtomicRMWOp::Set => "ldsetal",
1665                    AtomicRMWOp::Smax => "ldsmaxal",
1666                    AtomicRMWOp::Umax => "ldumaxal",
1667                    AtomicRMWOp::Smin => "ldsminal",
1668                    AtomicRMWOp::Umin => "lduminal",
1669                    AtomicRMWOp::Swp => "swpal",
1670                };
1671
1672                let size = OperandSize::from_ty(ty);
1673                let rs = pretty_print_ireg(rs, size);
1674                let rt = pretty_print_ireg(rt.to_reg(), size);
1675                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1676
1677                let ty_suffix = match ty {
1678                    I8 => "b",
1679                    I16 => "h",
1680                    _ => "",
1681                };
1682                format!("{op}{ty_suffix} {rs}, {rt}, [{rn}]")
1683            }
1684            &Inst::AtomicRMWLoop {
1685                ty,
1686                op,
1687                addr,
1688                operand,
1689                oldval,
1690                scratch1,
1691                scratch2,
1692                ..
1693            } => {
1694                let op = match op {
1695                    AtomicRMWLoopOp::Add => "add",
1696                    AtomicRMWLoopOp::Sub => "sub",
1697                    AtomicRMWLoopOp::Eor => "eor",
1698                    AtomicRMWLoopOp::Orr => "orr",
1699                    AtomicRMWLoopOp::And => "and",
1700                    AtomicRMWLoopOp::Nand => "nand",
1701                    AtomicRMWLoopOp::Smin => "smin",
1702                    AtomicRMWLoopOp::Smax => "smax",
1703                    AtomicRMWLoopOp::Umin => "umin",
1704                    AtomicRMWLoopOp::Umax => "umax",
1705                    AtomicRMWLoopOp::Xchg => "xchg",
1706                };
1707                let addr = pretty_print_ireg(addr, OperandSize::Size64);
1708                let operand = pretty_print_ireg(operand, OperandSize::Size64);
1709                let oldval = pretty_print_ireg(oldval.to_reg(), OperandSize::Size64);
1710                let scratch1 = pretty_print_ireg(scratch1.to_reg(), OperandSize::Size64);
1711                let scratch2 = pretty_print_ireg(scratch2.to_reg(), OperandSize::Size64);
1712                format!(
1713                    "atomic_rmw_loop_{}_{} addr={} operand={} oldval={} scratch1={} scratch2={}",
1714                    op,
1715                    ty.bits(),
1716                    addr,
1717                    operand,
1718                    oldval,
1719                    scratch1,
1720                    scratch2,
1721                )
1722            }
1723            &Inst::AtomicRMW128Loop {
1724                op,
1725                addr,
1726                operand_lo,
1727                operand_hi,
1728                oldval_lo,
1729                oldval_hi,
1730                scratch1,
1731                scratch2,
1732                scratch3,
1733                ..
1734            } => {
1735                let op = match op {
1736                    AtomicRMWLoopOp::Add => "add",
1737                    AtomicRMWLoopOp::Sub => "sub",
1738                    AtomicRMWLoopOp::Eor => "eor",
1739                    AtomicRMWLoopOp::Orr => "orr",
1740                    AtomicRMWLoopOp::And => "and",
1741                    AtomicRMWLoopOp::Nand => "nand",
1742                    AtomicRMWLoopOp::Smin => "smin",
1743                    AtomicRMWLoopOp::Smax => "smax",
1744                    AtomicRMWLoopOp::Umin => "umin",
1745                    AtomicRMWLoopOp::Umax => "umax",
1746                    AtomicRMWLoopOp::Xchg => "xchg",
1747                };
1748                let addr = pretty_print_ireg(addr, OperandSize::Size64);
1749                let operand_lo = pretty_print_ireg(operand_lo, OperandSize::Size64);
1750                let operand_hi = pretty_print_ireg(operand_hi, OperandSize::Size64);
1751                let oldval_lo = pretty_print_ireg(oldval_lo.to_reg(), OperandSize::Size64);
1752                let oldval_hi = pretty_print_ireg(oldval_hi.to_reg(), OperandSize::Size64);
1753                let scratch1 = pretty_print_ireg(scratch1.to_reg(), OperandSize::Size64);
1754                let scratch2 = pretty_print_ireg(scratch2.to_reg(), OperandSize::Size64);
1755                let scratch3 = pretty_print_ireg(scratch3.to_reg(), OperandSize::Size64);
1756                format!(
1757                    "atomic_rmw_128_loop_{op} addr={addr} operand_lo={operand_lo} operand_hi={operand_hi} oldval_lo={oldval_lo} oldval_hi={oldval_hi} scratch1={scratch1} scratch2={scratch2} scratch3={scratch3}",
1758                )
1759            }
1760            &Inst::AtomicCAS {
1761                rd, rs, rt, rn, ty, ..
1762            } => {
1763                let op = match ty {
1764                    I8 => "casalb",
1765                    I16 => "casalh",
1766                    I32 | I64 => "casal",
1767                    _ => panic!("Unsupported type: {ty}"),
1768                };
1769                let size = OperandSize::from_ty(ty);
1770                let rd = pretty_print_ireg(rd.to_reg(), size);
1771                let rs = pretty_print_ireg(rs, size);
1772                let rt = pretty_print_ireg(rt, size);
1773                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1774
1775                format!("{op} {rd}, {rs}, {rt}, [{rn}]")
1776            }
1777            Inst::AtomicCAS128 { args } => {
1778                let &AtomicCAS128Args {
1779                    rd_lo,
1780                    rd_hi,
1781                    rs_lo,
1782                    rs_hi,
1783                    rt_lo,
1784                    rt_hi,
1785                    rn,
1786                    flags: _,
1787                } = &**args;
1788                let size = OperandSize::Size64;
1789                let rd_lo = pretty_print_ireg(rd_lo.to_reg(), size);
1790                let rd_hi = pretty_print_ireg(rd_hi.to_reg(), size);
1791                let rs_lo = pretty_print_ireg(rs_lo, size);
1792                let rs_hi = pretty_print_ireg(rs_hi, size);
1793                let rt_lo = pretty_print_ireg(rt_lo, size);
1794                let rt_hi = pretty_print_ireg(rt_hi, size);
1795                let rn = pretty_print_ireg(rn, size);
1796
1797                format!("caspal {rd_lo}, {rd_hi}, {rs_lo}, {rs_hi}, {rt_lo}, {rt_hi}, [{rn}]")
1798            }
1799            &Inst::AtomicCASLoop {
1800                ty,
1801                addr,
1802                expected,
1803                replacement,
1804                oldval,
1805                scratch,
1806                ..
1807            } => {
1808                let addr = pretty_print_ireg(addr, OperandSize::Size64);
1809                let expected = pretty_print_ireg(expected, OperandSize::Size64);
1810                let replacement = pretty_print_ireg(replacement, OperandSize::Size64);
1811                let oldval = pretty_print_ireg(oldval.to_reg(), OperandSize::Size64);
1812                let scratch = pretty_print_ireg(scratch.to_reg(), OperandSize::Size64);
1813                format!(
1814                    "atomic_cas_loop_{} addr={}, expect={}, replacement={}, oldval={}, scratch={}",
1815                    ty.bits(),
1816                    addr,
1817                    expected,
1818                    replacement,
1819                    oldval,
1820                    scratch,
1821                )
1822            }
1823            &Inst::AtomicCAS128Loop {
1824                addr,
1825                expected_lo,
1826                expected_hi,
1827                replacement_lo,
1828                replacement_hi,
1829                oldval_lo,
1830                oldval_hi,
1831                scratch,
1832                ..
1833            } => {
1834                let addr = pretty_print_ireg(addr, OperandSize::Size64);
1835                let expected_lo = pretty_print_ireg(expected_lo, OperandSize::Size64);
1836                let expected_hi = pretty_print_ireg(expected_hi, OperandSize::Size64);
1837                let replacement_lo = pretty_print_ireg(replacement_lo, OperandSize::Size64);
1838                let replacement_hi = pretty_print_ireg(replacement_hi, OperandSize::Size64);
1839                let oldval_lo = pretty_print_ireg(oldval_lo.to_reg(), OperandSize::Size64);
1840                let oldval_hi = pretty_print_ireg(oldval_hi.to_reg(), OperandSize::Size64);
1841                let scratch = pretty_print_ireg(scratch.to_reg(), OperandSize::Size64);
1842                format!(
1843                    "atomic_cas_128_loop addr={addr}, expected_lo={expected_lo}, expected_hi={expected_hi}, replacement_lo={replacement_lo}, replacement_hi={replacement_hi}, oldval_lo={oldval_lo}, oldval_hi={oldval_hi}, scratch={scratch}",
1844                )
1845            }
1846            &Inst::LoadAcquire {
1847                access_ty, rt, rn, ..
1848            } => {
1849                let (op, ty) = match access_ty {
1850                    I8 => ("ldarb", I32),
1851                    I16 => ("ldarh", I32),
1852                    I32 => ("ldar", I32),
1853                    I64 => ("ldar", I64),
1854                    _ => panic!("Unsupported type: {access_ty}"),
1855                };
1856                let size = OperandSize::from_ty(ty);
1857                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1858                let rt = pretty_print_ireg(rt.to_reg(), size);
1859                format!("{op} {rt}, [{rn}]")
1860            }
1861            &Inst::LoadAcquire128 {
1862                rt1,
1863                rt2,
1864                rn,
1865                scratch,
1866                ..
1867            } => {
1868                let rt1 = pretty_print_ireg(rt1.to_reg(), OperandSize::Size64);
1869                let rt2 = pretty_print_ireg(rt2.to_reg(), OperandSize::Size64);
1870                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1871                let scratch = pretty_print_ireg(scratch.to_reg(), OperandSize::Size64);
1872                format!("load_acquire_128 {rt1}, {rt2}, [{rn}], scratch={scratch}")
1873            }
1874            &Inst::StoreRelease {
1875                access_ty, rt, rn, ..
1876            } => {
1877                let (op, ty) = match access_ty {
1878                    I8 => ("stlrb", I32),
1879                    I16 => ("stlrh", I32),
1880                    I32 => ("stlr", I32),
1881                    I64 => ("stlr", I64),
1882                    _ => panic!("Unsupported type: {access_ty}"),
1883                };
1884                let size = OperandSize::from_ty(ty);
1885                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1886                let rt = pretty_print_ireg(rt, size);
1887                format!("{op} {rt}, [{rn}]")
1888            }
1889            &Inst::StoreRelease128 {
1890                rt1,
1891                rt2,
1892                rn,
1893                scratch,
1894                ..
1895            } => {
1896                let rt1 = pretty_print_ireg(rt1, OperandSize::Size64);
1897                let rt2 = pretty_print_ireg(rt2, OperandSize::Size64);
1898                let rn = pretty_print_ireg(rn, OperandSize::Size64);
1899                let scratch = pretty_print_ireg(scratch.to_reg(), OperandSize::Size64);
1900                format!("store_release_128 {rt1}, {rt2}, [{rn}], scratch={scratch}")
1901            }
1902            &Inst::Fence {} => {
1903                format!("dmb ish")
1904            }
1905            &Inst::Csdb {} => {
1906                format!("csdb")
1907            }
1908            &Inst::FpuMove32 { rd, rn } => {
1909                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size32);
1910                let rn = pretty_print_vreg_scalar(rn, ScalarSize::Size32);
1911                format!("fmov {rd}, {rn}")
1912            }
1913            &Inst::FpuMove64 { rd, rn } => {
1914                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size64);
1915                let rn = pretty_print_vreg_scalar(rn, ScalarSize::Size64);
1916                format!("fmov {rd}, {rn}")
1917            }
1918            &Inst::FpuMove128 { rd, rn } => {
1919                let rd = pretty_print_reg(rd.to_reg());
1920                let rn = pretty_print_reg(rn);
1921                format!("mov {rd}.16b, {rn}.16b")
1922            }
1923            &Inst::FpuMoveFromVec { rd, rn, idx, size } => {
1924                let rd = pretty_print_vreg_scalar(rd.to_reg(), size.lane_size());
1925                let rn = pretty_print_vreg_element(rn, idx as usize, size.lane_size());
1926                format!("mov {rd}, {rn}")
1927            }
1928            &Inst::FpuExtend { rd, rn, size } => {
1929                let rd = pretty_print_vreg_scalar(rd.to_reg(), size);
1930                let rn = pretty_print_vreg_scalar(rn, size);
1931                format!("fmov {rd}, {rn}")
1932            }
1933            &Inst::FpuRR {
1934                fpu_op,
1935                size,
1936                rd,
1937                rn,
1938            } => {
1939                let op = match fpu_op {
1940                    FPUOp1::Abs => "fabs",
1941                    FPUOp1::Neg => "fneg",
1942                    FPUOp1::Sqrt => "fsqrt",
1943                    FPUOp1::Cvt16To32 | FPUOp1::Cvt32To64 | FPUOp1::Cvt64To32 => "fcvt",
1944                };
1945                let dst_size = match fpu_op {
1946                    FPUOp1::Cvt32To64 => ScalarSize::Size64,
1947                    FPUOp1::Cvt16To32 | FPUOp1::Cvt64To32 => ScalarSize::Size32,
1948                    _ => size,
1949                };
1950                let rd = pretty_print_vreg_scalar(rd.to_reg(), dst_size);
1951                let rn = pretty_print_vreg_scalar(rn, size);
1952                format!("{op} {rd}, {rn}")
1953            }
1954            &Inst::FpuRRR {
1955                fpu_op,
1956                size,
1957                rd,
1958                rn,
1959                rm,
1960            } => {
1961                let op = match fpu_op {
1962                    FPUOp2::Add => "fadd",
1963                    FPUOp2::Sub => "fsub",
1964                    FPUOp2::Mul => "fmul",
1965                    FPUOp2::Div => "fdiv",
1966                    FPUOp2::Max => "fmax",
1967                    FPUOp2::Min => "fmin",
1968                };
1969                let rd = pretty_print_vreg_scalar(rd.to_reg(), size);
1970                let rn = pretty_print_vreg_scalar(rn, size);
1971                let rm = pretty_print_vreg_scalar(rm, size);
1972                format!("{op} {rd}, {rn}, {rm}")
1973            }
1974            &Inst::FpuRRI { fpu_op, rd, rn } => {
1975                let (op, imm, vector) = match fpu_op {
1976                    FPUOpRI::UShr32(imm) => ("ushr", imm.pretty_print(0), true),
1977                    FPUOpRI::UShr64(imm) => ("ushr", imm.pretty_print(0), false),
1978                };
1979
1980                let (rd, rn) = if vector {
1981                    (
1982                        pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size32x2),
1983                        pretty_print_vreg_vector(rn, VectorSize::Size32x2),
1984                    )
1985                } else {
1986                    (
1987                        pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size64),
1988                        pretty_print_vreg_scalar(rn, ScalarSize::Size64),
1989                    )
1990                };
1991                format!("{op} {rd}, {rn}, {imm}")
1992            }
1993            &Inst::FpuRRIMod { fpu_op, rd, ri, rn } => {
1994                let (op, imm, vector) = match fpu_op {
1995                    FPUOpRIMod::Sli32(imm) => ("sli", imm.pretty_print(0), true),
1996                    FPUOpRIMod::Sli64(imm) => ("sli", imm.pretty_print(0), false),
1997                };
1998
1999                let (rd, ri, rn) = if vector {
2000                    (
2001                        pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size32x2),
2002                        pretty_print_vreg_vector(ri, VectorSize::Size32x2),
2003                        pretty_print_vreg_vector(rn, VectorSize::Size32x2),
2004                    )
2005                } else {
2006                    (
2007                        pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size64),
2008                        pretty_print_vreg_scalar(ri, ScalarSize::Size64),
2009                        pretty_print_vreg_scalar(rn, ScalarSize::Size64),
2010                    )
2011                };
2012                format!("{op} {rd}, {ri}, {rn}, {imm}")
2013            }
2014            &Inst::FpuRRRR {
2015                fpu_op,
2016                size,
2017                rd,
2018                rn,
2019                rm,
2020                ra,
2021            } => {
2022                let op = match fpu_op {
2023                    FPUOp3::MAdd => "fmadd",
2024                    FPUOp3::MSub => "fmsub",
2025                    FPUOp3::NMAdd => "fnmadd",
2026                    FPUOp3::NMSub => "fnmsub",
2027                };
2028                let rd = pretty_print_vreg_scalar(rd.to_reg(), size);
2029                let rn = pretty_print_vreg_scalar(rn, size);
2030                let rm = pretty_print_vreg_scalar(rm, size);
2031                let ra = pretty_print_vreg_scalar(ra, size);
2032                format!("{op} {rd}, {rn}, {rm}, {ra}")
2033            }
2034            &Inst::FpuCmp { size, rn, rm } => {
2035                let rn = pretty_print_vreg_scalar(rn, size);
2036                let rm = pretty_print_vreg_scalar(rm, size);
2037                format!("fcmp {rn}, {rm}")
2038            }
2039            &Inst::FpuLoad16 { rd, ref mem, .. } => {
2040                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size16);
2041                let mem = mem.clone();
2042                let access_ty = self.mem_type().unwrap();
2043                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2044                format!("{mem_str}ldr {rd}, {mem}")
2045            }
2046            &Inst::FpuLoad32 { rd, ref mem, .. } => {
2047                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size32);
2048                let mem = mem.clone();
2049                let access_ty = self.mem_type().unwrap();
2050                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2051                format!("{mem_str}ldr {rd}, {mem}")
2052            }
2053            &Inst::FpuLoad64 { rd, ref mem, .. } => {
2054                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size64);
2055                let mem = mem.clone();
2056                let access_ty = self.mem_type().unwrap();
2057                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2058                format!("{mem_str}ldr {rd}, {mem}")
2059            }
2060            &Inst::FpuLoad128 { rd, ref mem, .. } => {
2061                let rd = pretty_print_reg(rd.to_reg());
2062                let rd = "q".to_string() + &rd[1..];
2063                let mem = mem.clone();
2064                let access_ty = self.mem_type().unwrap();
2065                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2066                format!("{mem_str}ldr {rd}, {mem}")
2067            }
2068            &Inst::FpuStore16 { rd, ref mem, .. } => {
2069                let rd = pretty_print_vreg_scalar(rd, ScalarSize::Size16);
2070                let mem = mem.clone();
2071                let access_ty = self.mem_type().unwrap();
2072                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2073                format!("{mem_str}str {rd}, {mem}")
2074            }
2075            &Inst::FpuStore32 { rd, ref mem, .. } => {
2076                let rd = pretty_print_vreg_scalar(rd, ScalarSize::Size32);
2077                let mem = mem.clone();
2078                let access_ty = self.mem_type().unwrap();
2079                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2080                format!("{mem_str}str {rd}, {mem}")
2081            }
2082            &Inst::FpuStore64 { rd, ref mem, .. } => {
2083                let rd = pretty_print_vreg_scalar(rd, ScalarSize::Size64);
2084                let mem = mem.clone();
2085                let access_ty = self.mem_type().unwrap();
2086                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2087                format!("{mem_str}str {rd}, {mem}")
2088            }
2089            &Inst::FpuStore128 { rd, ref mem, .. } => {
2090                let rd = pretty_print_reg(rd);
2091                let rd = "q".to_string() + &rd[1..];
2092                let mem = mem.clone();
2093                let access_ty = self.mem_type().unwrap();
2094                let (mem_str, mem) = mem_finalize_for_show(&mem, access_ty, state);
2095                format!("{mem_str}str {rd}, {mem}")
2096            }
2097            &Inst::FpuLoadP64 {
2098                rt, rt2, ref mem, ..
2099            } => {
2100                let rt = pretty_print_vreg_scalar(rt.to_reg(), ScalarSize::Size64);
2101                let rt2 = pretty_print_vreg_scalar(rt2.to_reg(), ScalarSize::Size64);
2102                let mem = mem.clone();
2103                let mem = mem.pretty_print_default();
2104
2105                format!("ldp {rt}, {rt2}, {mem}")
2106            }
2107            &Inst::FpuStoreP64 {
2108                rt, rt2, ref mem, ..
2109            } => {
2110                let rt = pretty_print_vreg_scalar(rt, ScalarSize::Size64);
2111                let rt2 = pretty_print_vreg_scalar(rt2, ScalarSize::Size64);
2112                let mem = mem.clone();
2113                let mem = mem.pretty_print_default();
2114
2115                format!("stp {rt}, {rt2}, {mem}")
2116            }
2117            &Inst::FpuLoadP128 {
2118                rt, rt2, ref mem, ..
2119            } => {
2120                let rt = pretty_print_vreg_scalar(rt.to_reg(), ScalarSize::Size128);
2121                let rt2 = pretty_print_vreg_scalar(rt2.to_reg(), ScalarSize::Size128);
2122                let mem = mem.clone();
2123                let mem = mem.pretty_print_default();
2124
2125                format!("ldp {rt}, {rt2}, {mem}")
2126            }
2127            &Inst::FpuStoreP128 {
2128                rt, rt2, ref mem, ..
2129            } => {
2130                let rt = pretty_print_vreg_scalar(rt, ScalarSize::Size128);
2131                let rt2 = pretty_print_vreg_scalar(rt2, ScalarSize::Size128);
2132                let mem = mem.clone();
2133                let mem = mem.pretty_print_default();
2134
2135                format!("stp {rt}, {rt2}, {mem}")
2136            }
2137            &Inst::FpuToInt { op, rd, rn } => {
2138                let (op, sizesrc, sizedest) = match op {
2139                    FpuToIntOp::F32ToI32 => ("fcvtzs", ScalarSize::Size32, OperandSize::Size32),
2140                    FpuToIntOp::F32ToU32 => ("fcvtzu", ScalarSize::Size32, OperandSize::Size32),
2141                    FpuToIntOp::F32ToI64 => ("fcvtzs", ScalarSize::Size32, OperandSize::Size64),
2142                    FpuToIntOp::F32ToU64 => ("fcvtzu", ScalarSize::Size32, OperandSize::Size64),
2143                    FpuToIntOp::F64ToI32 => ("fcvtzs", ScalarSize::Size64, OperandSize::Size32),
2144                    FpuToIntOp::F64ToU32 => ("fcvtzu", ScalarSize::Size64, OperandSize::Size32),
2145                    FpuToIntOp::F64ToI64 => ("fcvtzs", ScalarSize::Size64, OperandSize::Size64),
2146                    FpuToIntOp::F64ToU64 => ("fcvtzu", ScalarSize::Size64, OperandSize::Size64),
2147                };
2148                let rd = pretty_print_ireg(rd.to_reg(), sizedest);
2149                let rn = pretty_print_vreg_scalar(rn, sizesrc);
2150                format!("{op} {rd}, {rn}")
2151            }
2152            &Inst::IntToFpu { op, rd, rn } => {
2153                let (op, sizesrc, sizedest) = match op {
2154                    IntToFpuOp::I32ToF32 => ("scvtf", OperandSize::Size32, ScalarSize::Size32),
2155                    IntToFpuOp::U32ToF32 => ("ucvtf", OperandSize::Size32, ScalarSize::Size32),
2156                    IntToFpuOp::I64ToF32 => ("scvtf", OperandSize::Size64, ScalarSize::Size32),
2157                    IntToFpuOp::U64ToF32 => ("ucvtf", OperandSize::Size64, ScalarSize::Size32),
2158                    IntToFpuOp::I32ToF64 => ("scvtf", OperandSize::Size32, ScalarSize::Size64),
2159                    IntToFpuOp::U32ToF64 => ("ucvtf", OperandSize::Size32, ScalarSize::Size64),
2160                    IntToFpuOp::I64ToF64 => ("scvtf", OperandSize::Size64, ScalarSize::Size64),
2161                    IntToFpuOp::U64ToF64 => ("ucvtf", OperandSize::Size64, ScalarSize::Size64),
2162                };
2163                let rd = pretty_print_vreg_scalar(rd.to_reg(), sizedest);
2164                let rn = pretty_print_ireg(rn, sizesrc);
2165                format!("{op} {rd}, {rn}")
2166            }
2167            &Inst::FpuCSel16 { rd, rn, rm, cond } => {
2168                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size16);
2169                let rn = pretty_print_vreg_scalar(rn, ScalarSize::Size16);
2170                let rm = pretty_print_vreg_scalar(rm, ScalarSize::Size16);
2171                let cond = cond.pretty_print(0);
2172                format!("fcsel {rd}, {rn}, {rm}, {cond}")
2173            }
2174            &Inst::FpuCSel32 { rd, rn, rm, cond } => {
2175                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size32);
2176                let rn = pretty_print_vreg_scalar(rn, ScalarSize::Size32);
2177                let rm = pretty_print_vreg_scalar(rm, ScalarSize::Size32);
2178                let cond = cond.pretty_print(0);
2179                format!("fcsel {rd}, {rn}, {rm}, {cond}")
2180            }
2181            &Inst::FpuCSel64 { rd, rn, rm, cond } => {
2182                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size64);
2183                let rn = pretty_print_vreg_scalar(rn, ScalarSize::Size64);
2184                let rm = pretty_print_vreg_scalar(rm, ScalarSize::Size64);
2185                let cond = cond.pretty_print(0);
2186                format!("fcsel {rd}, {rn}, {rm}, {cond}")
2187            }
2188            &Inst::FpuRound { op, rd, rn } => {
2189                let (inst, size) = match op {
2190                    FpuRoundMode::Minus32 => ("frintm", ScalarSize::Size32),
2191                    FpuRoundMode::Minus64 => ("frintm", ScalarSize::Size64),
2192                    FpuRoundMode::Plus32 => ("frintp", ScalarSize::Size32),
2193                    FpuRoundMode::Plus64 => ("frintp", ScalarSize::Size64),
2194                    FpuRoundMode::Zero32 => ("frintz", ScalarSize::Size32),
2195                    FpuRoundMode::Zero64 => ("frintz", ScalarSize::Size64),
2196                    FpuRoundMode::Nearest32 => ("frintn", ScalarSize::Size32),
2197                    FpuRoundMode::Nearest64 => ("frintn", ScalarSize::Size64),
2198                };
2199                let rd = pretty_print_vreg_scalar(rd.to_reg(), size);
2200                let rn = pretty_print_vreg_scalar(rn, size);
2201                format!("{inst} {rd}, {rn}")
2202            }
2203            &Inst::MovToFpu { rd, rn, size } => {
2204                let operand_size = size.operand_size();
2205                let rd = pretty_print_vreg_scalar(rd.to_reg(), size);
2206                let rn = pretty_print_ireg(rn, operand_size);
2207                format!("fmov {rd}, {rn}")
2208            }
2209            &Inst::FpuMoveFPImm { rd, imm, size } => {
2210                let imm = imm.pretty_print(0);
2211                let rd = pretty_print_vreg_scalar(rd.to_reg(), size);
2212
2213                format!("fmov {rd}, {imm}")
2214            }
2215            &Inst::MovToVec {
2216                rd,
2217                ri,
2218                rn,
2219                idx,
2220                size,
2221            } => {
2222                let rd = pretty_print_vreg_element(rd.to_reg(), idx as usize, size.lane_size());
2223                let ri = pretty_print_vreg_element(ri, idx as usize, size.lane_size());
2224                let rn = pretty_print_ireg(rn, size.operand_size());
2225                format!("mov {rd}, {ri}, {rn}")
2226            }
2227            &Inst::MovFromVec { rd, rn, idx, size } => {
2228                let op = match size {
2229                    ScalarSize::Size8 => "umov",
2230                    ScalarSize::Size16 => "umov",
2231                    ScalarSize::Size32 => "mov",
2232                    ScalarSize::Size64 => "mov",
2233                    _ => unimplemented!(),
2234                };
2235                let rd = pretty_print_ireg(rd.to_reg(), size.operand_size());
2236                let rn = pretty_print_vreg_element(rn, idx as usize, size);
2237                format!("{op} {rd}, {rn}")
2238            }
2239            &Inst::MovFromVecSigned {
2240                rd,
2241                rn,
2242                idx,
2243                size,
2244                scalar_size,
2245            } => {
2246                let rd = pretty_print_ireg(rd.to_reg(), scalar_size);
2247                let rn = pretty_print_vreg_element(rn, idx as usize, size.lane_size());
2248                format!("smov {rd}, {rn}")
2249            }
2250            &Inst::VecDup { rd, rn, size } => {
2251                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2252                let rn = pretty_print_ireg(rn, size.operand_size());
2253                format!("dup {rd}, {rn}")
2254            }
2255            &Inst::VecDupFromFpu { rd, rn, size, lane } => {
2256                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2257                let rn = pretty_print_vreg_element(rn, lane.into(), size.lane_size());
2258                format!("dup {rd}, {rn}")
2259            }
2260            &Inst::VecDupFPImm { rd, imm, size } => {
2261                let imm = imm.pretty_print(0);
2262                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2263
2264                format!("fmov {rd}, {imm}")
2265            }
2266            &Inst::VecDupImm {
2267                rd,
2268                imm,
2269                invert,
2270                size,
2271            } => {
2272                let imm = imm.pretty_print(0);
2273                let op = if invert { "mvni" } else { "movi" };
2274                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2275
2276                format!("{op} {rd}, {imm}")
2277            }
2278            &Inst::VecExtend {
2279                t,
2280                rd,
2281                rn,
2282                high_half,
2283                lane_size,
2284            } => {
2285                let vec64 = VectorSize::from_lane_size(lane_size.narrow(), false);
2286                let vec128 = VectorSize::from_lane_size(lane_size.narrow(), true);
2287                let rd_size = VectorSize::from_lane_size(lane_size, true);
2288                let (op, rn_size) = match (t, high_half) {
2289                    (VecExtendOp::Sxtl, false) => ("sxtl", vec64),
2290                    (VecExtendOp::Sxtl, true) => ("sxtl2", vec128),
2291                    (VecExtendOp::Uxtl, false) => ("uxtl", vec64),
2292                    (VecExtendOp::Uxtl, true) => ("uxtl2", vec128),
2293                };
2294                let rd = pretty_print_vreg_vector(rd.to_reg(), rd_size);
2295                let rn = pretty_print_vreg_vector(rn, rn_size);
2296                format!("{op} {rd}, {rn}")
2297            }
2298            &Inst::VecMovElement {
2299                rd,
2300                ri,
2301                rn,
2302                dest_idx,
2303                src_idx,
2304                size,
2305            } => {
2306                let rd =
2307                    pretty_print_vreg_element(rd.to_reg(), dest_idx as usize, size.lane_size());
2308                let ri = pretty_print_vreg_element(ri, dest_idx as usize, size.lane_size());
2309                let rn = pretty_print_vreg_element(rn, src_idx as usize, size.lane_size());
2310                format!("mov {rd}, {ri}, {rn}")
2311            }
2312            &Inst::VecRRLong {
2313                op,
2314                rd,
2315                rn,
2316                high_half,
2317            } => {
2318                let (op, rd_size, size, suffix) = match (op, high_half) {
2319                    (VecRRLongOp::Fcvtl16, false) => {
2320                        ("fcvtl", VectorSize::Size32x4, VectorSize::Size16x4, "")
2321                    }
2322                    (VecRRLongOp::Fcvtl16, true) => {
2323                        ("fcvtl2", VectorSize::Size32x4, VectorSize::Size16x8, "")
2324                    }
2325                    (VecRRLongOp::Fcvtl32, false) => {
2326                        ("fcvtl", VectorSize::Size64x2, VectorSize::Size32x2, "")
2327                    }
2328                    (VecRRLongOp::Fcvtl32, true) => {
2329                        ("fcvtl2", VectorSize::Size64x2, VectorSize::Size32x4, "")
2330                    }
2331                    (VecRRLongOp::Shll8, false) => {
2332                        ("shll", VectorSize::Size16x8, VectorSize::Size8x8, ", #8")
2333                    }
2334                    (VecRRLongOp::Shll8, true) => {
2335                        ("shll2", VectorSize::Size16x8, VectorSize::Size8x16, ", #8")
2336                    }
2337                    (VecRRLongOp::Shll16, false) => {
2338                        ("shll", VectorSize::Size32x4, VectorSize::Size16x4, ", #16")
2339                    }
2340                    (VecRRLongOp::Shll16, true) => {
2341                        ("shll2", VectorSize::Size32x4, VectorSize::Size16x8, ", #16")
2342                    }
2343                    (VecRRLongOp::Shll32, false) => {
2344                        ("shll", VectorSize::Size64x2, VectorSize::Size32x2, ", #32")
2345                    }
2346                    (VecRRLongOp::Shll32, true) => {
2347                        ("shll2", VectorSize::Size64x2, VectorSize::Size32x4, ", #32")
2348                    }
2349                };
2350                let rd = pretty_print_vreg_vector(rd.to_reg(), rd_size);
2351                let rn = pretty_print_vreg_vector(rn, size);
2352
2353                format!("{op} {rd}, {rn}{suffix}")
2354            }
2355            &Inst::VecRRNarrowLow {
2356                op,
2357                rd,
2358                rn,
2359                lane_size,
2360                ..
2361            }
2362            | &Inst::VecRRNarrowHigh {
2363                op,
2364                rd,
2365                rn,
2366                lane_size,
2367                ..
2368            } => {
2369                let vec64 = VectorSize::from_lane_size(lane_size, false);
2370                let vec128 = VectorSize::from_lane_size(lane_size, true);
2371                let rn_size = VectorSize::from_lane_size(lane_size.widen(), true);
2372                let high_half = match self {
2373                    &Inst::VecRRNarrowLow { .. } => false,
2374                    &Inst::VecRRNarrowHigh { .. } => true,
2375                    _ => unreachable!(),
2376                };
2377                let (op, rd_size) = match (op, high_half) {
2378                    (VecRRNarrowOp::Xtn, false) => ("xtn", vec64),
2379                    (VecRRNarrowOp::Xtn, true) => ("xtn2", vec128),
2380                    (VecRRNarrowOp::Sqxtn, false) => ("sqxtn", vec64),
2381                    (VecRRNarrowOp::Sqxtn, true) => ("sqxtn2", vec128),
2382                    (VecRRNarrowOp::Sqxtun, false) => ("sqxtun", vec64),
2383                    (VecRRNarrowOp::Sqxtun, true) => ("sqxtun2", vec128),
2384                    (VecRRNarrowOp::Uqxtn, false) => ("uqxtn", vec64),
2385                    (VecRRNarrowOp::Uqxtn, true) => ("uqxtn2", vec128),
2386                    (VecRRNarrowOp::Fcvtn, false) => ("fcvtn", vec64),
2387                    (VecRRNarrowOp::Fcvtn, true) => ("fcvtn2", vec128),
2388                };
2389                let rn = pretty_print_vreg_vector(rn, rn_size);
2390                let rd = pretty_print_vreg_vector(rd.to_reg(), rd_size);
2391                let ri = match self {
2392                    &Inst::VecRRNarrowLow { .. } => "".to_string(),
2393                    &Inst::VecRRNarrowHigh { ri, .. } => {
2394                        format!("{}, ", pretty_print_vreg_vector(ri, rd_size))
2395                    }
2396                    _ => unreachable!(),
2397                };
2398
2399                format!("{op} {rd}, {ri}{rn}")
2400            }
2401            &Inst::VecRRPair { op, rd, rn } => {
2402                let op = match op {
2403                    VecPairOp::Addp => "addp",
2404                };
2405                let rd = pretty_print_vreg_scalar(rd.to_reg(), ScalarSize::Size64);
2406                let rn = pretty_print_vreg_vector(rn, VectorSize::Size64x2);
2407
2408                format!("{op} {rd}, {rn}")
2409            }
2410            &Inst::VecRRPairLong { op, rd, rn } => {
2411                let (op, dest, src) = match op {
2412                    VecRRPairLongOp::Saddlp8 => {
2413                        ("saddlp", VectorSize::Size16x8, VectorSize::Size8x16)
2414                    }
2415                    VecRRPairLongOp::Saddlp16 => {
2416                        ("saddlp", VectorSize::Size32x4, VectorSize::Size16x8)
2417                    }
2418                    VecRRPairLongOp::Uaddlp8 => {
2419                        ("uaddlp", VectorSize::Size16x8, VectorSize::Size8x16)
2420                    }
2421                    VecRRPairLongOp::Uaddlp16 => {
2422                        ("uaddlp", VectorSize::Size32x4, VectorSize::Size16x8)
2423                    }
2424                };
2425                let rd = pretty_print_vreg_vector(rd.to_reg(), dest);
2426                let rn = pretty_print_vreg_vector(rn, src);
2427
2428                format!("{op} {rd}, {rn}")
2429            }
2430            &Inst::VecRRR {
2431                rd,
2432                rn,
2433                rm,
2434                alu_op,
2435                size,
2436            } => {
2437                let (op, size) = match alu_op {
2438                    VecALUOp::Sqadd => ("sqadd", size),
2439                    VecALUOp::Uqadd => ("uqadd", size),
2440                    VecALUOp::Sqsub => ("sqsub", size),
2441                    VecALUOp::Uqsub => ("uqsub", size),
2442                    VecALUOp::Cmeq => ("cmeq", size),
2443                    VecALUOp::Cmge => ("cmge", size),
2444                    VecALUOp::Cmgt => ("cmgt", size),
2445                    VecALUOp::Cmhs => ("cmhs", size),
2446                    VecALUOp::Cmhi => ("cmhi", size),
2447                    VecALUOp::Fcmeq => ("fcmeq", size),
2448                    VecALUOp::Fcmgt => ("fcmgt", size),
2449                    VecALUOp::Fcmge => ("fcmge", size),
2450                    VecALUOp::Umaxp => ("umaxp", size),
2451                    VecALUOp::Add => ("add", size),
2452                    VecALUOp::Sub => ("sub", size),
2453                    VecALUOp::Mul => ("mul", size),
2454                    VecALUOp::Sshl => ("sshl", size),
2455                    VecALUOp::Ushl => ("ushl", size),
2456                    VecALUOp::Umin => ("umin", size),
2457                    VecALUOp::Smin => ("smin", size),
2458                    VecALUOp::Umax => ("umax", size),
2459                    VecALUOp::Smax => ("smax", size),
2460                    VecALUOp::Urhadd => ("urhadd", size),
2461                    VecALUOp::Fadd => ("fadd", size),
2462                    VecALUOp::Fsub => ("fsub", size),
2463                    VecALUOp::Fdiv => ("fdiv", size),
2464                    VecALUOp::Fmax => ("fmax", size),
2465                    VecALUOp::Fmin => ("fmin", size),
2466                    VecALUOp::Fmul => ("fmul", size),
2467                    VecALUOp::Addp => ("addp", size),
2468                    VecALUOp::Zip1 => ("zip1", size),
2469                    VecALUOp::Zip2 => ("zip2", size),
2470                    VecALUOp::Sqrdmulh => ("sqrdmulh", size),
2471                    VecALUOp::Uzp1 => ("uzp1", size),
2472                    VecALUOp::Uzp2 => ("uzp2", size),
2473                    VecALUOp::Trn1 => ("trn1", size),
2474                    VecALUOp::Trn2 => ("trn2", size),
2475
2476                    // Lane division does not affect bitwise operations.
2477                    // However, when printing, use 8-bit lane division to conform to ARM formatting.
2478                    VecALUOp::And => ("and", size.as_scalar8_vector()),
2479                    VecALUOp::Bic => ("bic", size.as_scalar8_vector()),
2480                    VecALUOp::Orr => ("orr", size.as_scalar8_vector()),
2481                    VecALUOp::Orn => ("orn", size.as_scalar8_vector()),
2482                    VecALUOp::Eor => ("eor", size.as_scalar8_vector()),
2483                };
2484                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2485                let rn = pretty_print_vreg_vector(rn, size);
2486                let rm = pretty_print_vreg_vector(rm, size);
2487                format!("{op} {rd}, {rn}, {rm}")
2488            }
2489            &Inst::VecRRRMod {
2490                rd,
2491                ri,
2492                rn,
2493                rm,
2494                alu_op,
2495                size,
2496            } => {
2497                let (op, size) = match alu_op {
2498                    VecALUModOp::Bsl => ("bsl", VectorSize::Size8x16),
2499                    VecALUModOp::Fmla => ("fmla", size),
2500                    VecALUModOp::Fmls => ("fmls", size),
2501                    // Note: the real operand arrangement is .4s, .16b, .16b;
2502                    // this debug print renders all lanes as .4s.
2503                    VecALUModOp::Sdot => ("sdot", VectorSize::Size32x4),
2504                    VecALUModOp::Usdot => ("usdot", VectorSize::Size32x4),
2505                };
2506                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2507                let ri = pretty_print_vreg_vector(ri, size);
2508                let rn = pretty_print_vreg_vector(rn, size);
2509                let rm = pretty_print_vreg_vector(rm, size);
2510                format!("{op} {rd}, {ri}, {rn}, {rm}")
2511            }
2512            &Inst::VecFmlaElem {
2513                rd,
2514                ri,
2515                rn,
2516                rm,
2517                alu_op,
2518                size,
2519                idx,
2520            } => {
2521                let (op, size) = match alu_op {
2522                    VecALUModOp::Fmla => ("fmla", size),
2523                    VecALUModOp::Fmls => ("fmls", size),
2524                    _ => unreachable!(),
2525                };
2526                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2527                let ri = pretty_print_vreg_vector(ri, size);
2528                let rn = pretty_print_vreg_vector(rn, size);
2529                let rm = pretty_print_vreg_element(rm, idx.into(), size.lane_size());
2530                format!("{op} {rd}, {ri}, {rn}, {rm}")
2531            }
2532            &Inst::VecRRRLong {
2533                rd,
2534                rn,
2535                rm,
2536                alu_op,
2537                high_half,
2538            } => {
2539                let (op, dest_size, src_size) = match (alu_op, high_half) {
2540                    (VecRRRLongOp::Smull8, false) => {
2541                        ("smull", VectorSize::Size16x8, VectorSize::Size8x8)
2542                    }
2543                    (VecRRRLongOp::Smull8, true) => {
2544                        ("smull2", VectorSize::Size16x8, VectorSize::Size8x16)
2545                    }
2546                    (VecRRRLongOp::Smull16, false) => {
2547                        ("smull", VectorSize::Size32x4, VectorSize::Size16x4)
2548                    }
2549                    (VecRRRLongOp::Smull16, true) => {
2550                        ("smull2", VectorSize::Size32x4, VectorSize::Size16x8)
2551                    }
2552                    (VecRRRLongOp::Smull32, false) => {
2553                        ("smull", VectorSize::Size64x2, VectorSize::Size32x2)
2554                    }
2555                    (VecRRRLongOp::Smull32, true) => {
2556                        ("smull2", VectorSize::Size64x2, VectorSize::Size32x4)
2557                    }
2558                    (VecRRRLongOp::Umull8, false) => {
2559                        ("umull", VectorSize::Size16x8, VectorSize::Size8x8)
2560                    }
2561                    (VecRRRLongOp::Umull8, true) => {
2562                        ("umull2", VectorSize::Size16x8, VectorSize::Size8x16)
2563                    }
2564                    (VecRRRLongOp::Umull16, false) => {
2565                        ("umull", VectorSize::Size32x4, VectorSize::Size16x4)
2566                    }
2567                    (VecRRRLongOp::Umull16, true) => {
2568                        ("umull2", VectorSize::Size32x4, VectorSize::Size16x8)
2569                    }
2570                    (VecRRRLongOp::Umull32, false) => {
2571                        ("umull", VectorSize::Size64x2, VectorSize::Size32x2)
2572                    }
2573                    (VecRRRLongOp::Umull32, true) => {
2574                        ("umull2", VectorSize::Size64x2, VectorSize::Size32x4)
2575                    }
2576                };
2577                let rd = pretty_print_vreg_vector(rd.to_reg(), dest_size);
2578                let rn = pretty_print_vreg_vector(rn, src_size);
2579                let rm = pretty_print_vreg_vector(rm, src_size);
2580                format!("{op} {rd}, {rn}, {rm}")
2581            }
2582            &Inst::VecRRRLongMod {
2583                rd,
2584                ri,
2585                rn,
2586                rm,
2587                alu_op,
2588                high_half,
2589            } => {
2590                let (op, dest_size, src_size) = match (alu_op, high_half) {
2591                    (VecRRRLongModOp::Umlal8, false) => {
2592                        ("umlal", VectorSize::Size16x8, VectorSize::Size8x8)
2593                    }
2594                    (VecRRRLongModOp::Umlal8, true) => {
2595                        ("umlal2", VectorSize::Size16x8, VectorSize::Size8x16)
2596                    }
2597                    (VecRRRLongModOp::Umlal16, false) => {
2598                        ("umlal", VectorSize::Size32x4, VectorSize::Size16x4)
2599                    }
2600                    (VecRRRLongModOp::Umlal16, true) => {
2601                        ("umlal2", VectorSize::Size32x4, VectorSize::Size16x8)
2602                    }
2603                    (VecRRRLongModOp::Umlal32, false) => {
2604                        ("umlal", VectorSize::Size64x2, VectorSize::Size32x2)
2605                    }
2606                    (VecRRRLongModOp::Umlal32, true) => {
2607                        ("umlal2", VectorSize::Size64x2, VectorSize::Size32x4)
2608                    }
2609                };
2610                let rd = pretty_print_vreg_vector(rd.to_reg(), dest_size);
2611                let ri = pretty_print_vreg_vector(ri, dest_size);
2612                let rn = pretty_print_vreg_vector(rn, src_size);
2613                let rm = pretty_print_vreg_vector(rm, src_size);
2614                format!("{op} {rd}, {ri}, {rn}, {rm}")
2615            }
2616            &Inst::VecMisc { op, rd, rn, size } => {
2617                let (op, size, suffix) = match op {
2618                    VecMisc2::Neg => ("neg", size, ""),
2619                    VecMisc2::Abs => ("abs", size, ""),
2620                    VecMisc2::Fabs => ("fabs", size, ""),
2621                    VecMisc2::Fneg => ("fneg", size, ""),
2622                    VecMisc2::Fsqrt => ("fsqrt", size, ""),
2623                    VecMisc2::Rev16 => ("rev16", size, ""),
2624                    VecMisc2::Rev32 => ("rev32", size, ""),
2625                    VecMisc2::Rev64 => ("rev64", size, ""),
2626                    VecMisc2::Fcvtzs => ("fcvtzs", size, ""),
2627                    VecMisc2::Fcvtzu => ("fcvtzu", size, ""),
2628                    VecMisc2::Scvtf => ("scvtf", size, ""),
2629                    VecMisc2::Ucvtf => ("ucvtf", size, ""),
2630                    VecMisc2::Frintn => ("frintn", size, ""),
2631                    VecMisc2::Frintz => ("frintz", size, ""),
2632                    VecMisc2::Frintm => ("frintm", size, ""),
2633                    VecMisc2::Frintp => ("frintp", size, ""),
2634                    VecMisc2::Cnt => ("cnt", size, ""),
2635                    VecMisc2::Cmeq0 => ("cmeq", size, ", #0"),
2636                    VecMisc2::Cmge0 => ("cmge", size, ", #0"),
2637                    VecMisc2::Cmgt0 => ("cmgt", size, ", #0"),
2638                    VecMisc2::Cmle0 => ("cmle", size, ", #0"),
2639                    VecMisc2::Cmlt0 => ("cmlt", size, ", #0"),
2640                    VecMisc2::Fcmeq0 => ("fcmeq", size, ", #0.0"),
2641                    VecMisc2::Fcmge0 => ("fcmge", size, ", #0.0"),
2642                    VecMisc2::Fcmgt0 => ("fcmgt", size, ", #0.0"),
2643                    VecMisc2::Fcmle0 => ("fcmle", size, ", #0.0"),
2644                    VecMisc2::Fcmlt0 => ("fcmlt", size, ", #0.0"),
2645
2646                    // Lane division does not affect bitwise operations.
2647                    // However, when printing, use 8-bit lane division to conform to ARM formatting.
2648                    VecMisc2::Not => ("mvn", size.as_scalar8_vector(), ""),
2649                };
2650                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2651                let rn = pretty_print_vreg_vector(rn, size);
2652                format!("{op} {rd}, {rn}{suffix}")
2653            }
2654            &Inst::VecLanes { op, rd, rn, size } => {
2655                let op = match op {
2656                    VecLanesOp::Uminv => "uminv",
2657                    VecLanesOp::Addv => "addv",
2658                };
2659                let rd = pretty_print_vreg_scalar(rd.to_reg(), size.lane_size());
2660                let rn = pretty_print_vreg_vector(rn, size);
2661                format!("{op} {rd}, {rn}")
2662            }
2663            &Inst::VecShiftImm {
2664                op,
2665                rd,
2666                rn,
2667                size,
2668                imm,
2669            } => {
2670                let op = match op {
2671                    VecShiftImmOp::Shl => "shl",
2672                    VecShiftImmOp::Ushr => "ushr",
2673                    VecShiftImmOp::Sshr => "sshr",
2674                };
2675                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2676                let rn = pretty_print_vreg_vector(rn, size);
2677                format!("{op} {rd}, {rn}, #{imm}")
2678            }
2679            &Inst::VecShiftImmMod {
2680                op,
2681                rd,
2682                ri,
2683                rn,
2684                size,
2685                imm,
2686            } => {
2687                let op = match op {
2688                    VecShiftImmModOp::Sli => "sli",
2689                };
2690                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2691                let ri = pretty_print_vreg_vector(ri, size);
2692                let rn = pretty_print_vreg_vector(rn, size);
2693                format!("{op} {rd}, {ri}, {rn}, #{imm}")
2694            }
2695            &Inst::VecExtract { rd, rn, rm, imm4 } => {
2696                let rd = pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size8x16);
2697                let rn = pretty_print_vreg_vector(rn, VectorSize::Size8x16);
2698                let rm = pretty_print_vreg_vector(rm, VectorSize::Size8x16);
2699                format!("ext {rd}, {rn}, {rm}, #{imm4}")
2700            }
2701            &Inst::VecTbl { rd, rn, rm } => {
2702                let rn = pretty_print_vreg_vector(rn, VectorSize::Size8x16);
2703                let rm = pretty_print_vreg_vector(rm, VectorSize::Size8x16);
2704                let rd = pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size8x16);
2705                format!("tbl {rd}, {{ {rn} }}, {rm}")
2706            }
2707            &Inst::VecTblExt { rd, ri, rn, rm } => {
2708                let rn = pretty_print_vreg_vector(rn, VectorSize::Size8x16);
2709                let rm = pretty_print_vreg_vector(rm, VectorSize::Size8x16);
2710                let rd = pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size8x16);
2711                let ri = pretty_print_vreg_vector(ri, VectorSize::Size8x16);
2712                format!("tbx {rd}, {ri}, {{ {rn} }}, {rm}")
2713            }
2714            &Inst::VecTbl2 { rd, rn, rn2, rm } => {
2715                let rn = pretty_print_vreg_vector(rn, VectorSize::Size8x16);
2716                let rn2 = pretty_print_vreg_vector(rn2, VectorSize::Size8x16);
2717                let rm = pretty_print_vreg_vector(rm, VectorSize::Size8x16);
2718                let rd = pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size8x16);
2719                format!("tbl {rd}, {{ {rn}, {rn2} }}, {rm}")
2720            }
2721            &Inst::VecTbl2Ext {
2722                rd,
2723                ri,
2724                rn,
2725                rn2,
2726                rm,
2727            } => {
2728                let rn = pretty_print_vreg_vector(rn, VectorSize::Size8x16);
2729                let rn2 = pretty_print_vreg_vector(rn2, VectorSize::Size8x16);
2730                let rm = pretty_print_vreg_vector(rm, VectorSize::Size8x16);
2731                let rd = pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size8x16);
2732                let ri = pretty_print_vreg_vector(ri, VectorSize::Size8x16);
2733                format!("tbx {rd}, {ri}, {{ {rn}, {rn2} }}, {rm}")
2734            }
2735            &Inst::VecLoadReplicate { rd, rn, size, .. } => {
2736                let rd = pretty_print_vreg_vector(rd.to_reg(), size);
2737                let rn = pretty_print_reg(rn);
2738
2739                format!("ld1r {{ {rd} }}, [{rn}]")
2740            }
2741            &Inst::VecCSel { rd, rn, rm, cond } => {
2742                let rd = pretty_print_vreg_vector(rd.to_reg(), VectorSize::Size8x16);
2743                let rn = pretty_print_vreg_vector(rn, VectorSize::Size8x16);
2744                let rm = pretty_print_vreg_vector(rm, VectorSize::Size8x16);
2745                let cond = cond.pretty_print(0);
2746                format!("vcsel {rd}, {rn}, {rm}, {cond} (if-then-else diamond)")
2747            }
2748            &Inst::MovToNZCV { rn } => {
2749                let rn = pretty_print_reg(rn);
2750                format!("msr nzcv, {rn}")
2751            }
2752            &Inst::MovFromNZCV { rd } => {
2753                let rd = pretty_print_reg(rd.to_reg());
2754                format!("mrs {rd}, nzcv")
2755            }
2756            &Inst::Extend {
2757                rd,
2758                rn,
2759                signed: false,
2760                from_bits: 1,
2761                ..
2762            } => {
2763                let rd = pretty_print_ireg(rd.to_reg(), OperandSize::Size32);
2764                let rn = pretty_print_ireg(rn, OperandSize::Size32);
2765                format!("and {rd}, {rn}, #1")
2766            }
2767            &Inst::Extend {
2768                rd,
2769                rn,
2770                signed: false,
2771                from_bits: 32,
2772                to_bits: 64,
2773            } => {
2774                // The case of a zero extension from 32 to 64 bits, is implemented
2775                // with a "mov" to a 32-bit (W-reg) dest, because this zeroes
2776                // the top 32 bits.
2777                let rd = pretty_print_ireg(rd.to_reg(), OperandSize::Size32);
2778                let rn = pretty_print_ireg(rn, OperandSize::Size32);
2779                format!("mov {rd}, {rn}")
2780            }
2781            &Inst::Extend {
2782                rd,
2783                rn,
2784                signed,
2785                from_bits,
2786                to_bits,
2787            } => {
2788                assert!(from_bits <= to_bits);
2789                let op = match (signed, from_bits) {
2790                    (false, 8) => "uxtb",
2791                    (true, 8) => "sxtb",
2792                    (false, 16) => "uxth",
2793                    (true, 16) => "sxth",
2794                    (true, 32) => "sxtw",
2795                    (true, _) => "sbfx",
2796                    (false, _) => "ubfx",
2797                };
2798                if op == "sbfx" || op == "ubfx" {
2799                    let dest_size = OperandSize::from_bits(to_bits);
2800                    let rd = pretty_print_ireg(rd.to_reg(), dest_size);
2801                    let rn = pretty_print_ireg(rn, dest_size);
2802                    format!("{op} {rd}, {rn}, #0, #{from_bits}")
2803                } else {
2804                    let dest_size = if signed {
2805                        OperandSize::from_bits(to_bits)
2806                    } else {
2807                        OperandSize::Size32
2808                    };
2809                    let rd = pretty_print_ireg(rd.to_reg(), dest_size);
2810                    let rn = pretty_print_ireg(rn, OperandSize::from_bits(from_bits));
2811                    format!("{op} {rd}, {rn}")
2812                }
2813            }
2814            &Inst::BitfieldMove {
2815                size,
2816                bfm_op,
2817                rd,
2818                rn,
2819                immr,
2820                imms,
2821            } => {
2822                let op = bfm_op.op_str();
2823                let rd = pretty_print_ireg(rd.to_reg(), size);
2824                let rn = pretty_print_ireg(rn, size);
2825                let immr = immr.pretty_print(0);
2826                let imms = imms.pretty_print(0);
2827                format!("{op} {rd}, {rn}, {immr}, {imms}")
2828            }
2829            &Inst::BitfieldMoveMod {
2830                size,
2831                rd,
2832                ri,
2833                rn,
2834                immr,
2835                imms,
2836            } => {
2837                let rd = pretty_print_ireg(rd.to_reg(), size);
2838                let ri = pretty_print_ireg(ri, size);
2839                let rn = pretty_print_ireg(rn, size);
2840                let immr = immr.pretty_print(0);
2841                let imms = imms.pretty_print(0);
2842                format!("bfm {rd}, {ri}, {rn}, {immr}, {imms}")
2843            }
2844            &Inst::Call { ref info } => {
2845                let try_call = info
2846                    .try_call_info
2847                    .as_ref()
2848                    .map(|tci| pretty_print_try_call(tci))
2849                    .unwrap_or_default();
2850                format!("bl 0{try_call}")
2851            }
2852            &Inst::CallInd { ref info } => {
2853                let rn = pretty_print_reg(info.dest);
2854                let try_call = info
2855                    .try_call_info
2856                    .as_ref()
2857                    .map(|tci| pretty_print_try_call(tci))
2858                    .unwrap_or_default();
2859                format!("blr {rn}{try_call}")
2860            }
2861            &Inst::ReturnCall { ref info } => {
2862                let mut s = format!(
2863                    "return_call {:?} new_stack_arg_size:{}",
2864                    info.dest, info.new_stack_arg_size
2865                );
2866                for ret in &info.uses {
2867                    let preg = pretty_print_reg(ret.preg);
2868                    let vreg = pretty_print_reg(ret.vreg);
2869                    write!(&mut s, " {vreg}={preg}").unwrap();
2870                }
2871                s
2872            }
2873            &Inst::ReturnCallInd { ref info } => {
2874                let callee = pretty_print_reg(info.dest);
2875                let mut s = format!(
2876                    "return_call_ind {callee} new_stack_arg_size:{}",
2877                    info.new_stack_arg_size
2878                );
2879                for ret in &info.uses {
2880                    let preg = pretty_print_reg(ret.preg);
2881                    let vreg = pretty_print_reg(ret.vreg);
2882                    write!(&mut s, " {vreg}={preg}").unwrap();
2883                }
2884                s
2885            }
2886            &Inst::Args { ref args } => {
2887                let mut s = "args".to_string();
2888                for arg in args {
2889                    let preg = pretty_print_reg(arg.preg);
2890                    let def = pretty_print_reg(arg.vreg.to_reg());
2891                    write!(&mut s, " {def}={preg}").unwrap();
2892                }
2893                s
2894            }
2895            &Inst::Rets { ref rets } => {
2896                let mut s = "rets".to_string();
2897                for ret in rets {
2898                    let preg = pretty_print_reg(ret.preg);
2899                    let vreg = pretty_print_reg(ret.vreg);
2900                    write!(&mut s, " {vreg}={preg}").unwrap();
2901                }
2902                s
2903            }
2904            &Inst::Ret {} => "ret".to_string(),
2905            &Inst::AuthenticatedRet { key, is_hint } => {
2906                let key = match key {
2907                    APIKey::AZ => "az",
2908                    APIKey::BZ => "bz",
2909                    APIKey::ASP => "asp",
2910                    APIKey::BSP => "bsp",
2911                };
2912                match is_hint {
2913                    false => format!("reta{key}"),
2914                    true => format!("auti{key} ; ret"),
2915                }
2916            }
2917            &Inst::Jump { ref dest } => {
2918                let dest = dest.pretty_print(0);
2919                format!("b {dest}")
2920            }
2921            &Inst::CondBr {
2922                ref taken,
2923                ref not_taken,
2924                ref kind,
2925            } => {
2926                let taken = taken.pretty_print(0);
2927                let not_taken = not_taken.pretty_print(0);
2928                match kind {
2929                    &CondBrKind::Zero(reg, size) => {
2930                        let reg = pretty_print_reg_sized(reg, size);
2931                        format!("cbz {reg}, {taken} ; b {not_taken}")
2932                    }
2933                    &CondBrKind::NotZero(reg, size) => {
2934                        let reg = pretty_print_reg_sized(reg, size);
2935                        format!("cbnz {reg}, {taken} ; b {not_taken}")
2936                    }
2937                    &CondBrKind::Cond(c) => {
2938                        let c = c.pretty_print(0);
2939                        format!("b.{c} {taken} ; b {not_taken}")
2940                    }
2941                }
2942            }
2943            &Inst::TestBitAndBranch {
2944                kind,
2945                ref taken,
2946                ref not_taken,
2947                rn,
2948                bit,
2949            } => {
2950                let cond = match kind {
2951                    TestBitAndBranchKind::Z => "z",
2952                    TestBitAndBranchKind::NZ => "nz",
2953                };
2954                let taken = taken.pretty_print(0);
2955                let not_taken = not_taken.pretty_print(0);
2956                let rn = pretty_print_reg(rn);
2957                format!("tb{cond} {rn}, #{bit}, {taken} ; b {not_taken}")
2958            }
2959            &Inst::IndirectBr { rn, .. } => {
2960                let rn = pretty_print_reg(rn);
2961                format!("br {rn}")
2962            }
2963            &Inst::Brk => "brk #0xf000".to_string(),
2964            &Inst::Udf { .. } => "udf #0xc11f".to_string(),
2965            &Inst::TrapIf {
2966                ref kind,
2967                trap_code,
2968            } => match kind {
2969                &CondBrKind::Zero(reg, size) => {
2970                    let reg = pretty_print_reg_sized(reg, size);
2971                    format!("cbz {reg}, #trap={trap_code}")
2972                }
2973                &CondBrKind::NotZero(reg, size) => {
2974                    let reg = pretty_print_reg_sized(reg, size);
2975                    format!("cbnz {reg}, #trap={trap_code}")
2976                }
2977                &CondBrKind::Cond(c) => {
2978                    let c = c.pretty_print(0);
2979                    format!("b.{c} #trap={trap_code}")
2980                }
2981            },
2982            &Inst::Adr { rd, off } => {
2983                let rd = pretty_print_reg(rd.to_reg());
2984                format!("adr {rd}, pc+{off}")
2985            }
2986            &Inst::Adrp { rd, off } => {
2987                let rd = pretty_print_reg(rd.to_reg());
2988                // This instruction addresses 4KiB pages, so multiply it by the page size.
2989                let byte_offset = off * 4096;
2990                format!("adrp {rd}, pc+{byte_offset}")
2991            }
2992            &Inst::Word4 { data } => format!("data.i32 {data}"),
2993            &Inst::Word8 { data } => format!("data.i64 {data}"),
2994            &Inst::JTSequence {
2995                default,
2996                ref targets,
2997                ridx,
2998                rtmp1,
2999                rtmp2,
3000                ..
3001            } => {
3002                let ridx = pretty_print_reg(ridx);
3003                let rtmp1 = pretty_print_reg(rtmp1.to_reg());
3004                let rtmp2 = pretty_print_reg(rtmp2.to_reg());
3005                let default_target = BranchTarget::Label(default).pretty_print(0);
3006                format!(
3007                    concat!(
3008                        "b.hs {} ; ",
3009                        "csel {}, xzr, {}, hs ; ",
3010                        "csdb ; ",
3011                        "adr {}, pc+16 ; ",
3012                        "ldrsw {}, [{}, {}, uxtw #2] ; ",
3013                        "add {}, {}, {} ; ",
3014                        "br {} ; ",
3015                        "jt_entries {:?}"
3016                    ),
3017                    default_target,
3018                    rtmp2,
3019                    ridx,
3020                    rtmp1,
3021                    rtmp2,
3022                    rtmp1,
3023                    rtmp2,
3024                    rtmp1,
3025                    rtmp1,
3026                    rtmp2,
3027                    rtmp1,
3028                    targets
3029                )
3030            }
3031            &Inst::LoadExtNameGot { rd, ref name } => {
3032                let rd = pretty_print_reg(rd.to_reg());
3033                format!("load_ext_name_got {rd}, {name:?}")
3034            }
3035            &Inst::LoadExtNameNear {
3036                rd,
3037                ref name,
3038                offset,
3039            } => {
3040                let rd = pretty_print_reg(rd.to_reg());
3041                format!("load_ext_name_near {rd}, {name:?}+{offset}")
3042            }
3043            &Inst::LoadExtNameFar {
3044                rd,
3045                ref name,
3046                offset,
3047            } => {
3048                let rd = pretty_print_reg(rd.to_reg());
3049                format!("load_ext_name_far {rd}, {name:?}+{offset}")
3050            }
3051            &Inst::LoadAddr { rd, ref mem } => {
3052                // TODO: we really should find a better way to avoid duplication of
3053                // this logic between `emit()` and `show_rru()` -- a separate 1-to-N
3054                // expansion stage (i.e., legalization, but without the slow edit-in-place
3055                // of the existing legalization framework).
3056                let mem = mem.clone();
3057                let (mem_insts, mem) = mem_finalize(None, &mem, I8, state);
3058                let mut ret = String::new();
3059                for inst in mem_insts.into_iter() {
3060                    ret.push_str(&inst.print_with_state(&mut EmitState::default()));
3061                }
3062                let (reg, index_reg, offset) = match mem {
3063                    AMode::RegExtended { rn, rm, extendop } => (rn, Some((rm, extendop)), 0),
3064                    AMode::Unscaled { rn, simm9 } => (rn, None, simm9.value()),
3065                    AMode::UnsignedOffset { rn, uimm12 } => (rn, None, uimm12.value() as i32),
3066                    _ => panic!("Unsupported case for LoadAddr: {mem:?}"),
3067                };
3068                let abs_offset = if offset < 0 {
3069                    -offset as u64
3070                } else {
3071                    offset as u64
3072                };
3073                let alu_op = if offset < 0 { ALUOp::Sub } else { ALUOp::Add };
3074
3075                if let Some((idx, extendop)) = index_reg {
3076                    let add = Inst::AluRRRExtend {
3077                        alu_op: ALUOp::Add,
3078                        size: OperandSize::Size64,
3079                        rd,
3080                        rn: reg,
3081                        rm: idx,
3082                        extendop,
3083                    };
3084
3085                    ret.push_str(&add.print_with_state(&mut EmitState::default()));
3086                } else if offset == 0 {
3087                    let mov = Inst::gen_move(rd, reg, I64);
3088                    ret.push_str(&mov.print_with_state(&mut EmitState::default()));
3089                } else if let Some(imm12) = Imm12::maybe_from_u64(abs_offset) {
3090                    let add = Inst::AluRRImm12 {
3091                        alu_op,
3092                        size: OperandSize::Size64,
3093                        rd,
3094                        rn: reg,
3095                        imm12,
3096                    };
3097                    ret.push_str(&add.print_with_state(&mut EmitState::default()));
3098                } else {
3099                    let tmp = writable_spilltmp_reg();
3100                    for inst in Inst::load_constant(tmp, abs_offset).into_iter() {
3101                        ret.push_str(&inst.print_with_state(&mut EmitState::default()));
3102                    }
3103                    let add = Inst::AluRRR {
3104                        alu_op,
3105                        size: OperandSize::Size64,
3106                        rd,
3107                        rn: reg,
3108                        rm: tmp.to_reg(),
3109                    };
3110                    ret.push_str(&add.print_with_state(&mut EmitState::default()));
3111                }
3112                ret
3113            }
3114            &Inst::Paci { key } => {
3115                let key = match key {
3116                    APIKey::AZ => "az",
3117                    APIKey::BZ => "bz",
3118                    APIKey::ASP => "asp",
3119                    APIKey::BSP => "bsp",
3120                };
3121
3122                "paci".to_string() + key
3123            }
3124            &Inst::Xpaclri => "xpaclri".to_string(),
3125            &Inst::Bti { targets } => {
3126                let targets = match targets {
3127                    BranchTargetType::None => "",
3128                    BranchTargetType::C => " c",
3129                    BranchTargetType::J => " j",
3130                    BranchTargetType::JC => " jc",
3131                };
3132
3133                "bti".to_string() + targets
3134            }
3135            &Inst::EmitIsland { needed_space } => format!("emit_island {needed_space}"),
3136
3137            &Inst::ElfTlsGetAddr {
3138                ref symbol,
3139                rd,
3140                tmp,
3141            } => {
3142                let rd = pretty_print_reg(rd.to_reg());
3143                let tmp = pretty_print_reg(tmp.to_reg());
3144                format!("elf_tls_get_addr {}, {}, {}", rd, tmp, symbol.display(None))
3145            }
3146            &Inst::MachOTlsGetAddr { ref symbol, rd } => {
3147                let rd = pretty_print_reg(rd.to_reg());
3148                format!("macho_tls_get_addr {}, {}", rd, symbol.display(None))
3149            }
3150            &Inst::Unwind { ref inst } => {
3151                format!("unwind {inst:?}")
3152            }
3153            &Inst::DummyUse { reg } => {
3154                let reg = pretty_print_reg(reg);
3155                format!("dummy_use {reg}")
3156            }
3157            &Inst::LabelAddress { dst, label } => {
3158                let dst = pretty_print_reg(dst.to_reg());
3159                format!("label_address {dst}, {label:?}")
3160            }
3161            &Inst::SequencePoint {} => {
3162                format!("sequence_point")
3163            }
3164            &Inst::StackProbeLoop { start, end, step } => {
3165                let start = pretty_print_reg(start.to_reg());
3166                let end = pretty_print_reg(end);
3167                let step = step.pretty_print(0);
3168                format!("stack_probe_loop {start}, {end}, {step}")
3169            }
3170        }
3171    }
3172}
3173
3174//=============================================================================
3175// Label fixups and jump veneers.
3176
3177/// Different forms of label references for different instruction formats.
3178#[derive(Clone, Copy, Debug, PartialEq, Eq)]
3179pub enum LabelUse {
3180    /// 14-bit branch offset (conditional branches). PC-rel, offset is imm <<
3181    /// 2. Immediate is 14 signed bits, in bits 18:5. Used by tbz and tbnz.
3182    Branch14,
3183    /// 19-bit branch offset (conditional branches). PC-rel, offset is imm << 2. Immediate is 19
3184    /// signed bits, in bits 23:5. Used by cbz, cbnz, b.cond.
3185    Branch19,
3186    /// 26-bit branch offset (unconditional branches). PC-rel, offset is imm << 2. Immediate is 26
3187    /// signed bits, in bits 25:0. Used by b, bl.
3188    Branch26,
3189    /// 19-bit offset for LDR (load literal). PC-rel, offset is imm << 2. Immediate is 19 signed bits,
3190    /// in bits 23:5.
3191    Ldr19,
3192    /// 21-bit offset for ADR (get address of label). PC-rel, offset is not shifted. Immediate is
3193    /// 21 signed bits, with high 19 bits in bits 23:5 and low 2 bits in bits 30:29.
3194    Adr21,
3195    /// 32-bit PC relative constant offset (from address of constant itself),
3196    /// signed. Used in jump tables.
3197    PCRel32,
3198}
3199
3200impl MachInstLabelUse for LabelUse {
3201    /// Alignment for veneer code. Every AArch64 instruction must be 4-byte-aligned.
3202    const ALIGN: CodeOffset = 4;
3203
3204    /// Maximum PC-relative range (positive), inclusive.
3205    fn max_pos_range(self) -> CodeOffset {
3206        match self {
3207            // N-bit immediate, left-shifted by 2, for (N+2) bits of total
3208            // range. Signed, so +2^(N+1) from zero. Likewise for two other
3209            // shifted cases below.
3210            LabelUse::Branch14 => (1 << 15) - 1,
3211            LabelUse::Branch19 => (1 << 20) - 1,
3212            LabelUse::Branch26 => (1 << 27) - 1,
3213            LabelUse::Ldr19 => (1 << 20) - 1,
3214            // Adr does not shift its immediate, so the 21-bit immediate gives 21 bits of total
3215            // range.
3216            LabelUse::Adr21 => (1 << 20) - 1,
3217            LabelUse::PCRel32 => 0x7fffffff,
3218        }
3219    }
3220
3221    /// Maximum PC-relative range (negative).
3222    fn max_neg_range(self) -> CodeOffset {
3223        // All forms are twos-complement signed offsets, so negative limit is one more than
3224        // positive limit.
3225        self.max_pos_range() + 1
3226    }
3227
3228    /// Size of window into code needed to do the patch.
3229    fn patch_size(self) -> CodeOffset {
3230        // Patch is on one instruction only for all of these label reference types.
3231        4
3232    }
3233
3234    /// Perform the patch.
3235    fn patch(self, buffer: &mut [u8], use_offset: CodeOffset, label_offset: CodeOffset) {
3236        let pc_rel = (label_offset as i64) - (use_offset as i64);
3237        debug_assert!(pc_rel <= self.max_pos_range() as i64);
3238        debug_assert!(pc_rel >= -(self.max_neg_range() as i64));
3239        let pc_rel = pc_rel as u32;
3240        let insn_word = u32::from_le_bytes([buffer[0], buffer[1], buffer[2], buffer[3]]);
3241        let mask = match self {
3242            LabelUse::Branch14 => 0x0007ffe0, // bits 18..5 inclusive
3243            LabelUse::Branch19 => 0x00ffffe0, // bits 23..5 inclusive
3244            LabelUse::Branch26 => 0x03ffffff, // bits 25..0 inclusive
3245            LabelUse::Ldr19 => 0x00ffffe0,    // bits 23..5 inclusive
3246            LabelUse::Adr21 => 0x60ffffe0,    // bits 30..29, 25..5 inclusive
3247            LabelUse::PCRel32 => 0xffffffff,
3248        };
3249        let pc_rel_shifted = match self {
3250            LabelUse::Adr21 | LabelUse::PCRel32 => pc_rel,
3251            _ => {
3252                debug_assert!(pc_rel & 3 == 0);
3253                pc_rel >> 2
3254            }
3255        };
3256        let pc_rel_inserted = match self {
3257            LabelUse::Branch14 => (pc_rel_shifted & 0x3fff) << 5,
3258            LabelUse::Branch19 | LabelUse::Ldr19 => (pc_rel_shifted & 0x7ffff) << 5,
3259            LabelUse::Branch26 => pc_rel_shifted & 0x3ffffff,
3260            // Note: the *low* two bits of offset are put in the
3261            // *high* bits (30, 29).
3262            LabelUse::Adr21 => (pc_rel_shifted & 0x1ffffc) << 3 | (pc_rel_shifted & 3) << 29,
3263            LabelUse::PCRel32 => pc_rel_shifted,
3264        };
3265        let is_add = match self {
3266            LabelUse::PCRel32 => true,
3267            _ => false,
3268        };
3269        let insn_word = if is_add {
3270            insn_word.wrapping_add(pc_rel_inserted)
3271        } else {
3272            (insn_word & !mask) | pc_rel_inserted
3273        };
3274        buffer[0..4].clone_from_slice(&u32::to_le_bytes(insn_word));
3275    }
3276
3277    /// Is a veneer supported for this label reference type?
3278    fn supports_veneer(self) -> bool {
3279        match self {
3280            LabelUse::Branch14 | LabelUse::Branch19 => true, // veneer is a Branch26
3281            LabelUse::Branch26 => true,                      // veneer is a PCRel32
3282            _ => false,
3283        }
3284    }
3285
3286    /// How large is the veneer, if supported?
3287    fn veneer_size(self) -> CodeOffset {
3288        match self {
3289            LabelUse::Branch14 | LabelUse::Branch19 => 4,
3290            LabelUse::Branch26 => 20,
3291            _ => unreachable!(),
3292        }
3293    }
3294
3295    fn worst_case_veneer_size() -> CodeOffset {
3296        20
3297    }
3298
3299    /// Generate a veneer into the buffer, given that this veneer is at `veneer_offset`, and return
3300    /// an offset and label-use for the veneer's use of the original label.
3301    fn generate_veneer(
3302        self,
3303        buffer: &mut [u8],
3304        veneer_offset: CodeOffset,
3305    ) -> (CodeOffset, LabelUse) {
3306        match self {
3307            LabelUse::Branch14 | LabelUse::Branch19 => {
3308                // veneer is a Branch26 (unconditional branch). Just encode directly here -- don't
3309                // bother with constructing an Inst.
3310                let insn_word = 0b000101 << 26;
3311                buffer[0..4].clone_from_slice(&u32::to_le_bytes(insn_word));
3312                (veneer_offset, LabelUse::Branch26)
3313            }
3314
3315            // This is promoting a 26-bit call/jump to a 32-bit call/jump to
3316            // get a further range. This jump translates to a jump to a
3317            // relative location based on the address of the constant loaded
3318            // from here.
3319            //
3320            // If this path is taken from a call instruction then caller-saved
3321            // registers are available (minus arguments), so x16/x17 are
3322            // available. Otherwise for intra-function jumps we also reserve
3323            // x16/x17 as spill-style registers. In both cases these are
3324            // available for us to use.
3325            LabelUse::Branch26 => {
3326                let tmp1 = regs::spilltmp_reg();
3327                let tmp1_w = regs::writable_spilltmp_reg();
3328                let tmp2 = regs::tmp2_reg();
3329                let tmp2_w = regs::writable_tmp2_reg();
3330                // ldrsw x16, 16
3331                let ldr = emit::enc_ldst_imm19(0b1001_1000, 16 / 4, tmp1);
3332                // adr x17, 12
3333                let adr = emit::enc_adr(12, tmp2_w);
3334                // add x16, x16, x17
3335                let add = emit::enc_arith_rrr(0b10001011_000, 0, tmp1_w, tmp1, tmp2);
3336                // br x16
3337                let br = emit::enc_br(tmp1);
3338                buffer[0..4].clone_from_slice(&u32::to_le_bytes(ldr));
3339                buffer[4..8].clone_from_slice(&u32::to_le_bytes(adr));
3340                buffer[8..12].clone_from_slice(&u32::to_le_bytes(add));
3341                buffer[12..16].clone_from_slice(&u32::to_le_bytes(br));
3342                // the 4-byte signed immediate we'll load is after these
3343                // instructions, 16-bytes in.
3344                (veneer_offset + 16, LabelUse::PCRel32)
3345            }
3346
3347            _ => panic!("Unsupported label-reference type for veneer generation!"),
3348        }
3349    }
3350
3351    fn from_reloc(reloc: Reloc, addend: Addend) -> Option<LabelUse> {
3352        match (reloc, addend) {
3353            (Reloc::Arm64Call, 0) => Some(LabelUse::Branch26),
3354            _ => None,
3355        }
3356    }
3357}
3358
3359#[test]
3360#[cfg(target_pointer_width = "64")]
3361fn inst_size_test() {
3362    // This test will help with unintentionally growing the size
3363    // of the Inst enum.
3364    assert_eq!(48, core::mem::size_of::<Inst>());
3365}