Skip to main content

rustc_codegen_llvm/
va_arg.rs

1use rustc_abi::{
2    Align, BackendRepr, CVariadicStatus, Endian, Float, HasDataLayout, Integer, Primitive, Size,
3};
4use rustc_codegen_ssa::common::IntPredicate;
5use rustc_codegen_ssa::mir::operand::OperandRef;
6use rustc_codegen_ssa::traits::{
7    BaseTypeCodegenMethods, BuilderMethods, ConstCodegenMethods, LayoutTypeCodegenMethods,
8};
9use rustc_middle::ty::Ty;
10use rustc_middle::ty::layout::{HasTyCtxt, LayoutOf, TyAndLayout};
11use rustc_span::bug;
12use rustc_target::spec::{Arch, Env, LlvmAbi, RustcAbi};
13
14use crate::builder::Builder;
15use crate::llvm::Value;
16use crate::type_of::LayoutLlvmExt;
17
18fn round_up_to_alignment<'ll>(
19    bx: &mut Builder<'_, 'll, '_>,
20    mut value: &'ll Value,
21    align: Align,
22) -> &'ll Value {
23    value = bx.add(value, bx.cx().const_i32(align.bytes() as i32 - 1));
24    return bx.and(value, bx.cx().const_i32(-(align.bytes() as i32)));
25}
26
27fn round_pointer_up_to_alignment<'ll>(
28    bx: &mut Builder<'_, 'll, '_>,
29    addr: &'ll Value,
30    align: Align,
31) -> &'ll Value {
32    let ptr = bx.inbounds_ptradd(addr, bx.const_i32(align.bytes() as i32 - 1));
33    let pointer_width = bx.tcx().sess.target.pointer_width;
34    let mask = align.bytes().wrapping_neg() & (u64::MAX >> (64 - pointer_width));
35    bx.call_intrinsic(
36        "llvm.ptrmask",
37        &[bx.type_ptr(), bx.type_isize()],
38        &[ptr, bx.const_usize(mask)],
39    )
40}
41
42fn emit_direct_ptr_va_arg<'ll, 'tcx>(
43    bx: &mut Builder<'_, 'll, 'tcx>,
44    list: OperandRef<'tcx, &'ll Value>,
45    size: Size,
46    align: Align,
47    slot_size: Align,
48    allow_higher_align: bool,
49    force_right_adjust: bool,
50) -> (&'ll Value, Align) {
51    let va_list_ty = bx.type_ptr();
52    let va_list_addr = list.immediate();
53
54    let ptr_align_abi = bx.tcx().data_layout.pointer_align().abi;
55    let ptr = bx.load(va_list_ty, va_list_addr, ptr_align_abi);
56
57    let (addr, addr_align) = if allow_higher_align && align > slot_size {
58        (round_pointer_up_to_alignment(bx, ptr, align), align)
59    } else {
60        (ptr, slot_size)
61    };
62
63    let aligned_size = size.align_to(slot_size).bytes() as i32;
64    let full_direct_size = bx.cx().const_i32(aligned_size);
65    let next = bx.inbounds_ptradd(addr, full_direct_size);
66    bx.store(next, va_list_addr, ptr_align_abi);
67
68    if size.bytes() < slot_size.bytes()
69        && bx.tcx().sess.target.endian == Endian::Big
70        && force_right_adjust
71    {
72        let adjusted_size = bx.cx().const_i32((slot_size.bytes() - size.bytes()) as i32);
73        let adjusted = bx.inbounds_ptradd(addr, adjusted_size);
74        // We're in the middle of a slot now, so use the type's alignment, not the slot's.
75        (adjusted, align)
76    } else {
77        (addr, addr_align)
78    }
79}
80
81/// Some backends apply special alignment rules to c-variadic arguments.
82fn get_param_type_alignment<'ll, 'tcx>(
83    bx: &mut Builder<'_, 'll, 'tcx>,
84    layout: TyAndLayout<'tcx>,
85) -> Align {
86    let BackendRepr::Scalar(scalar) = layout.backend_repr else {
87        ::rustc_span::macros::bug_impl(None,
    format_args!("unexpected backend repr {0:?}", layout.backend_repr),
    Location::caller());bug!("unexpected backend repr {:?}", layout.backend_repr);
88    };
89
90    match bx.cx.tcx.sess.target.arch {
91        Arch::PowerPC64 => match scalar.primitive() {
92            Primitive::Int(integer, _) => match integer {
93                Integer::I8 | Integer::I16 => ::core::panicking::panic("internal error: entered unreachable code")unreachable!(),
94                Integer::I32 | Integer::I64 => { /* fall through */ }
95                Integer::I128 => return Align::EIGHT,
96            },
97            Primitive::Float(float) => match float {
98                Float::F16 | Float::F16B | Float::F32 => ::core::panicking::panic("internal error: entered unreachable code")unreachable!(),
99                Float::F64 => { /* fall through */ }
100                Float::F128 => return Align::from_bytes(16).unwrap(),
101            },
102            Primitive::Pointer(_) => { /* fall through */ }
103        },
104        _ => { /* fall through */ }
105    }
106
107    layout.align.abi
108}
109
110enum PassMode {
111    Direct,
112    Indirect,
113}
114
115enum SlotSize {
116    Bytes8 = 8,
117    Bytes4 = 4,
118    Bytes1 = 1,
119}
120
121/// Whether to respect a value alignment that is higher than the slot alignment.
122///
123/// When `No` the argument is in the next slot, when `Yes` there will be empty slots
124/// until a slot's starting address has the required alignment.
125enum AllowHigherAlign {
126    No,
127    Yes,
128}
129
130/// Determines where in the slot the value is located. Only takes effect on big-endian targets.
131///
132/// with 8-byte slots, a 32-bit integer is either stored right-adjusted:
133///
134/// ```text
135/// [0x0, 0x0, 0x0, 0x0, 0xaa, 0xaa, 0xaa, 0xaa]
136/// ```
137///
138/// or left-adjusted:
139///
140/// ```text
141/// [0xaa, 0xaa, 0xaa, 0xaa, 0x0, 0x0, 0x0, 0x0]
142/// ```
143///
144/// Most big-endian targets store values as right-adjusted.
145enum ForceRightAdjust {
146    No,
147    Yes,
148}
149
150fn emit_ptr_va_arg<'ll, 'tcx>(
151    bx: &mut Builder<'_, 'll, 'tcx>,
152    list: OperandRef<'tcx, &'ll Value>,
153    layout: TyAndLayout<'tcx>,
154    pass_mode: PassMode,
155    slot_size: SlotSize,
156    allow_higher_align: AllowHigherAlign,
157    force_right_adjust: ForceRightAdjust,
158) -> &'ll Value {
159    let indirect = #[allow(non_exhaustive_omitted_patterns)] match pass_mode {
    PassMode::Indirect => true,
    _ => false,
}matches!(pass_mode, PassMode::Indirect);
160    let allow_higher_align = #[allow(non_exhaustive_omitted_patterns)] match allow_higher_align {
    AllowHigherAlign::Yes => true,
    _ => false,
}matches!(allow_higher_align, AllowHigherAlign::Yes);
161    let force_right_adjust = #[allow(non_exhaustive_omitted_patterns)] match force_right_adjust {
    ForceRightAdjust::Yes => true,
    _ => false,
}matches!(force_right_adjust, ForceRightAdjust::Yes);
162    let slot_size = Align::from_bytes(slot_size as u64).unwrap();
163
164    let (llty, size, align) = if indirect {
165        (
166            bx.cx.layout_of(Ty::new_imm_ptr(bx.cx.tcx, layout.ty)).llvm_type(bx.cx),
167            bx.cx.data_layout().pointer_size(),
168            bx.cx.data_layout().pointer_align().abi,
169        )
170    } else {
171        (layout.llvm_type(bx.cx), layout.size, get_param_type_alignment(bx, layout))
172    };
173    let (addr, addr_align) = emit_direct_ptr_va_arg(
174        bx,
175        list,
176        size,
177        align,
178        slot_size,
179        allow_higher_align,
180        force_right_adjust,
181    );
182    if indirect {
183        let tmp_ret = bx.load(llty, addr, addr_align);
184        bx.load(layout.llvm_type(bx.cx), tmp_ret, align)
185    } else {
186        bx.load(llty, addr, addr_align)
187    }
188}
189
190fn emit_aapcs_va_arg<'ll, 'tcx>(
191    bx: &mut Builder<'_, 'll, 'tcx>,
192    list: OperandRef<'tcx, &'ll Value>,
193    layout: TyAndLayout<'tcx>,
194) -> &'ll Value {
195    let dl = bx.cx.data_layout();
196
197    // Implementation of the AAPCS64 calling convention for va_args see
198    // https://github.com/ARM-software/abi-aa/blob/master/aapcs64/aapcs64.rst
199    //
200    // typedef struct  va_list {
201    //     void * stack; // next stack param
202    //     void * gr_top; // end of GP arg reg save area
203    //     void * vr_top; // end of FP/SIMD arg reg save area
204    //     int gr_offs; // offset from  gr_top to next GP register arg
205    //     int vr_offs; // offset from  vr_top to next FP/SIMD register arg
206    // } va_list;
207    let va_list_addr = list.immediate();
208
209    // There is no padding between fields since `void*` is size=8 align=8, `int` is size=4 align=4.
210    // See https://github.com/ARM-software/abi-aa/blob/master/aapcs64/aapcs64.rst
211    // Table 1, Byte size and byte alignment of fundamental data types
212    // Table 3, Mapping of C & C++ built-in data types
213    let ptr_offset = 8;
214    let i32_offset = 4;
215    let gr_top = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(ptr_offset));
216    let vr_top = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(2 * ptr_offset));
217    let gr_offs = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(3 * ptr_offset));
218    let vr_offs = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(3 * ptr_offset + i32_offset));
219
220    let maybe_reg = bx.append_sibling_block("va_arg.maybe_reg");
221    let in_reg = bx.append_sibling_block("va_arg.in_reg");
222    let on_stack = bx.append_sibling_block("va_arg.on_stack");
223    let end = bx.append_sibling_block("va_arg.end");
224    let zero = bx.const_i32(0);
225    let offset_align = Align::from_bytes(4).unwrap();
226
227    let gr_type = layout.ty.is_any_ptr() || layout.ty.is_integral();
228    let (reg_off, reg_top, slot_size) = if gr_type {
229        let nreg = layout.size.bytes().div_ceil(8);
230        (gr_offs, gr_top, nreg * 8)
231    } else {
232        let nreg = layout.size.bytes().div_ceil(16);
233        (vr_offs, vr_top, nreg * 16)
234    };
235
236    // if the offset >= 0 then the value will be on the stack
237    let mut reg_off_v = bx.load(bx.type_i32(), reg_off, offset_align);
238    let use_stack = bx.icmp(IntPredicate::IntSGE, reg_off_v, zero);
239    bx.cond_br(use_stack, on_stack, maybe_reg);
240
241    // The value at this point might be in a register, but there is a chance that
242    // it could be on the stack so we have to update the offset and then check
243    // the offset again.
244
245    bx.switch_to_block(maybe_reg);
246    if gr_type && layout.align.bytes() > 8 {
247        reg_off_v = bx.add(reg_off_v, bx.const_i32(15));
248        reg_off_v = bx.and(reg_off_v, bx.const_i32(-16));
249    }
250    let new_reg_off_v = bx.add(reg_off_v, bx.const_i32(slot_size as i32));
251
252    bx.store(new_reg_off_v, reg_off, offset_align);
253
254    // Check to see if we have overflowed the registers as a result of this.
255    // If we have then we need to use the stack for this value
256    let use_stack = bx.icmp(IntPredicate::IntSGT, new_reg_off_v, zero);
257    bx.cond_br(use_stack, on_stack, in_reg);
258
259    bx.switch_to_block(in_reg);
260    let top_type = bx.type_ptr();
261    let top = bx.load(top_type, reg_top, dl.pointer_align().abi);
262
263    // reg_value = *(@top + reg_off_v);
264    let mut reg_addr = bx.ptradd(top, reg_off_v);
265    if bx.tcx().sess.target.endian == Endian::Big && layout.size.bytes() != slot_size {
266        // On big-endian systems the value is right-aligned in its slot.
267        let offset = bx.const_i32((slot_size - layout.size.bytes()) as i32);
268        reg_addr = bx.ptradd(reg_addr, offset);
269    }
270    let reg_type = layout.llvm_type(bx);
271    let reg_value = bx.load(reg_type, reg_addr, layout.align.abi);
272    bx.br(end);
273
274    // On Stack block
275    bx.switch_to_block(on_stack);
276    let stack_value = emit_ptr_va_arg(
277        bx,
278        list,
279        layout,
280        PassMode::Direct,
281        SlotSize::Bytes8,
282        AllowHigherAlign::Yes,
283        ForceRightAdjust::No,
284    );
285    bx.br(end);
286
287    bx.switch_to_block(end);
288    let val =
289        bx.phi(layout.immediate_llvm_type(bx), &[reg_value, stack_value], &[in_reg, on_stack]);
290
291    val
292}
293
294fn emit_powerpc_va_arg<'ll, 'tcx>(
295    bx: &mut Builder<'_, 'll, 'tcx>,
296    list: OperandRef<'tcx, &'ll Value>,
297    layout: TyAndLayout<'tcx>,
298) -> &'ll Value {
299    let dl = bx.cx.data_layout();
300
301    // struct __va_list_tag {
302    //   unsigned char gpr;
303    //   unsigned char fpr;
304    //   unsigned short reserved;
305    //   void *overflow_arg_area;
306    //   void *reg_save_area;
307    // };
308    let va_list_addr = list.immediate();
309
310    // Peel off any newtype wrappers.
311    let layout = layout.peel_transparent_wrappers(bx.cx);
312
313    // Rust does not currently support any powerpc softfloat targets.
314    let target = &bx.cx.tcx.sess.target;
315    let is_soft_float_abi = target.rustc_abi == Some(RustcAbi::Softfloat);
316    if !!is_soft_float_abi {
    ::core::panicking::panic("assertion failed: !is_soft_float_abi")
};assert!(!is_soft_float_abi);
317
318    // All instances of VaArgSafe are passed directly.
319    let is_indirect = false;
320
321    let (is_i64, is_int, is_f64) = match layout.layout.backend_repr() {
322        BackendRepr::Scalar(scalar) => match scalar.primitive() {
323            rustc_abi::Primitive::Int(integer, _) => (integer.size().bits() == 64, true, false),
324            rustc_abi::Primitive::Float(float) => (false, false, float.size().bits() == 64),
325            rustc_abi::Primitive::Pointer(_) => (false, true, false),
326        },
327        _ => {
    ::core::panicking::panic_fmt(format_args!("internal error: entered unreachable code: {0}",
            format_args!("all instances of VaArgSafe are represented as scalars")));
}unreachable!("all instances of VaArgSafe are represented as scalars"),
328    };
329
330    let num_regs_addr = if is_int || is_soft_float_abi {
331        va_list_addr // gpr
332    } else {
333        bx.inbounds_ptradd(va_list_addr, bx.const_usize(1)) // fpr
334    };
335
336    let mut num_regs = bx.load(bx.type_i8(), num_regs_addr, dl.i8_align);
337
338    // "Align" the register count when the type is passed as `i64`.
339    if is_i64 || (is_f64 && is_soft_float_abi) {
340        num_regs = bx.add(num_regs, bx.const_u8(1));
341        num_regs = bx.and(num_regs, bx.const_u8(0b1111_1110));
342    }
343
344    let max_regs = 8u8;
345    let use_regs = bx.icmp(IntPredicate::IntULT, num_regs, bx.const_u8(max_regs));
346    let ptr_align_abi = bx.tcx().data_layout.pointer_align().abi;
347
348    let in_reg = bx.append_sibling_block("va_arg.in_reg");
349    let in_mem = bx.append_sibling_block("va_arg.in_mem");
350    let end = bx.append_sibling_block("va_arg.end");
351
352    bx.cond_br(use_regs, in_reg, in_mem);
353
354    let reg_addr = {
355        bx.switch_to_block(in_reg);
356
357        let reg_safe_area_ptr = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(1 + 1 + 2 + 4));
358        let mut reg_addr = bx.load(bx.type_ptr(), reg_safe_area_ptr, ptr_align_abi);
359
360        // Floating-point registers start after the general-purpose registers.
361        if !is_int && !is_soft_float_abi {
362            reg_addr = bx.inbounds_ptradd(reg_addr, bx.cx.const_usize(32))
363        }
364
365        // Get the address of the saved value by scaling the number of
366        // registers we've used by the number of.
367        let reg_size = if is_int || is_soft_float_abi { 4 } else { 8 };
368        let reg_offset = bx.mul(num_regs, bx.cx().const_u8(reg_size));
369        let reg_addr = bx.inbounds_ptradd(reg_addr, reg_offset);
370
371        // Increase the used-register count.
372        let reg_incr = if is_i64 || (is_f64 && is_soft_float_abi) { 2 } else { 1 };
373        let new_num_regs = bx.add(num_regs, bx.cx.const_u8(reg_incr));
374        bx.store(new_num_regs, num_regs_addr, dl.i8_align);
375
376        bx.br(end);
377
378        reg_addr
379    };
380
381    let mem_addr = {
382        bx.switch_to_block(in_mem);
383
384        bx.store(bx.const_u8(max_regs), num_regs_addr, dl.i8_align);
385
386        // Everything in the overflow area is rounded up to a size of at least 4.
387        let overflow_area_align = Align::from_bytes(4).unwrap();
388
389        let size = if !is_indirect {
390            layout.layout.size.align_to(overflow_area_align)
391        } else {
392            dl.pointer_size()
393        };
394
395        let overflow_area_ptr = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(1 + 1 + 2));
396        let mut overflow_area = bx.load(bx.type_ptr(), overflow_area_ptr, ptr_align_abi);
397
398        // Round up address of argument to alignment
399        if layout.layout.align.abi > overflow_area_align {
400            overflow_area =
401                round_pointer_up_to_alignment(bx, overflow_area, layout.layout.align.abi);
402        }
403
404        let mem_addr = overflow_area;
405
406        // Increase the overflow area.
407        overflow_area = bx.inbounds_ptradd(overflow_area, bx.const_usize(size.bytes()));
408        bx.store(overflow_area, overflow_area_ptr, ptr_align_abi);
409
410        bx.br(end);
411
412        mem_addr
413    };
414
415    // Return the appropriate result.
416    bx.switch_to_block(end);
417    let val_addr = bx.phi(bx.type_ptr(), &[reg_addr, mem_addr], &[in_reg, in_mem]);
418    let val_type = layout.llvm_type(bx);
419    let val_addr =
420        if is_indirect { bx.load(bx.cx.type_ptr(), val_addr, ptr_align_abi) } else { val_addr };
421    bx.load(val_type, val_addr, layout.align.abi)
422}
423
424fn emit_s390x_va_arg<'ll, 'tcx>(
425    bx: &mut Builder<'_, 'll, 'tcx>,
426    list: OperandRef<'tcx, &'ll Value>,
427    layout: TyAndLayout<'tcx>,
428) -> &'ll Value {
429    let dl = bx.cx.data_layout();
430
431    // Implementation of the s390x ELF ABI calling convention for va_args see
432    // https://github.com/IBM/s390x-abi (chapter 1.2.4)
433    //
434    // typedef struct __va_list_tag {
435    //     long __gpr;
436    //     long __fpr;
437    //     void *__overflow_arg_area;
438    //     void *__reg_save_area;
439    // } va_list[1];
440    let va_list_addr = list.immediate();
441
442    // There is no padding between fields since `long` and `void*` both have size=8 align=8.
443    // https://github.com/IBM/s390x-abi (Table 1.1.: Scalar types)
444    let i64_offset = 8;
445    let ptr_offset = 8;
446    let gpr = va_list_addr;
447    let fpr = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(i64_offset));
448    let overflow_arg_area = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(2 * i64_offset));
449    let reg_save_area =
450        bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(2 * i64_offset + ptr_offset));
451
452    let in_reg = bx.append_sibling_block("va_arg.in_reg");
453    let in_mem = bx.append_sibling_block("va_arg.in_mem");
454    let end = bx.append_sibling_block("va_arg.end");
455    let ptr_align_abi = dl.pointer_align().abi;
456
457    // FIXME: vector ABI not yet supported.
458    let target_ty_size = layout.size.bytes();
459    let indirect: bool = target_ty_size > 8 || !target_ty_size.is_power_of_two();
460    let unpadded_size = if indirect { 8 } else { target_ty_size };
461    let padded_size = 8;
462    let padding = padded_size - unpadded_size;
463
464    // NOTE: if we ever allow aggregate types, this should handle structs with a single fp element.
465    let is_single_fp_element = |layout: TyAndLayout<'_>| -> bool {
466        match layout.layout.backend_repr() {
467            BackendRepr::Scalar(scalar) => match scalar.primitive() {
468                Primitive::Float(Float::F16 | Float::F32 | Float::F64) => true,
469                Primitive::Float(Float::F128) => false,
470                Primitive::Int(_, _) | Primitive::Pointer(_) => false,
471                Primitive::Float(Float::F16B) => {
472                    ::rustc_span::macros::bug_impl(None,
    format_args!("`f16b` use in varadics unsupported on s390x"),
    Location::caller())bug!("`f16b` use in varadics unsupported on s390x")
473                }
474            },
475
476            _ => false,
477        }
478    };
479
480    let gpr_type = indirect || !is_single_fp_element(layout);
481    let (max_regs, reg_count, reg_save_index, reg_padding) =
482        if gpr_type { (5, gpr, 2, padding) } else { (4, fpr, 16, 0) };
483
484    // Check whether the value was passed in a register or in memory.
485    let reg_count_v = bx.load(bx.type_i64(), reg_count, Align::from_bytes(8).unwrap());
486    let use_regs = bx.icmp(IntPredicate::IntULT, reg_count_v, bx.const_u64(max_regs));
487    bx.cond_br(use_regs, in_reg, in_mem);
488
489    // Emit code to load the value if it was passed in a register.
490    bx.switch_to_block(in_reg);
491
492    // Work out the address of the value in the register save area.
493    let reg_ptr_v = bx.load(bx.type_ptr(), reg_save_area, ptr_align_abi);
494    let scaled_reg_count = bx.mul(reg_count_v, bx.const_u64(8));
495    let reg_off = bx.add(scaled_reg_count, bx.const_u64(reg_save_index * 8 + reg_padding));
496    let reg_addr = bx.ptradd(reg_ptr_v, reg_off);
497
498    // Update the register count.
499    let new_reg_count_v = bx.add(reg_count_v, bx.const_u64(1));
500    bx.store(new_reg_count_v, reg_count, Align::from_bytes(8).unwrap());
501    bx.br(end);
502
503    // Emit code to load the value if it was passed in memory.
504    bx.switch_to_block(in_mem);
505
506    // Work out the address of the value in the argument overflow area.
507    let arg_ptr_v = bx.load(bx.type_ptr(), overflow_arg_area, ptr_align_abi);
508    let arg_off = bx.const_u64(padding);
509    let mem_addr = bx.ptradd(arg_ptr_v, arg_off);
510
511    // Update the argument overflow area pointer.
512    let arg_size = bx.cx().const_u64(padded_size);
513    let new_arg_ptr_v = bx.inbounds_ptradd(arg_ptr_v, arg_size);
514    bx.store(new_arg_ptr_v, overflow_arg_area, ptr_align_abi);
515    bx.br(end);
516
517    // Return the appropriate result.
518    bx.switch_to_block(end);
519    let val_addr = bx.phi(bx.type_ptr(), &[reg_addr, mem_addr], &[in_reg, in_mem]);
520    let val_type = layout.llvm_type(bx);
521    let val_addr =
522        if indirect { bx.load(bx.cx.type_ptr(), val_addr, ptr_align_abi) } else { val_addr };
523    bx.load(val_type, val_addr, layout.align.abi)
524}
525
526fn emit_x86_64_sysv64_va_arg<'ll, 'tcx>(
527    bx: &mut Builder<'_, 'll, 'tcx>,
528    list: OperandRef<'tcx, &'ll Value>,
529    layout: TyAndLayout<'tcx>,
530) -> &'ll Value {
531    let dl = bx.cx.data_layout();
532
533    // Implementation of the systemv x86_64 ABI calling convention for va_args, see
534    // https://gitlab.com/x86-psABIs/x86-64-ABI (section 3.5.7). This implementation is heavily
535    // based on the one in clang.
536
537    // We're able to take some shortcuts because the return type of `va_arg` must implement the
538    // `VaArgSafe` trait. Currently, only pointers, f64, i32, u32, i64 and u64 implement this trait.
539
540    // typedef struct __va_list_tag {
541    //     unsigned int gp_offset;
542    //     unsigned int fp_offset;
543    //     void *overflow_arg_area;
544    //     void *reg_save_area;
545    // } va_list[1];
546    let va_list_addr = list.immediate();
547
548    // Peel off any newtype wrappers.
549    //
550    // The "C" ABI does not unwrap newtypes (see `ReprOptions::inhibit_newtype_abi_optimization`).
551    // Here, we do actually want the unwrapped representation, because that is how LLVM/Clang
552    // pass such types to variadic functions.
553    //
554    // An example of a type that must be unwrapped is `Foo` below. Without the unwrapping, it has
555    // `BackendRepr::Memory`, but we need it to be `BackendRepr::Scalar` to generate correct code.
556    //
557    // ```
558    // #[repr(C)]
559    // struct Empty;
560    //
561    // #[repr(C)]
562    // struct Foo([Empty; 8], i32);
563    // ```
564    let layout = layout.peel_transparent_wrappers(bx.cx);
565
566    // AMD64-ABI 3.5.7p5: Step 1. Determine whether type may be passed
567    // in the registers. If not go to step 7.
568
569    // AMD64-ABI 3.5.7p5: Step 2. Compute num_gp to hold the number of
570    // general purpose registers needed to pass type and num_fp to hold
571    // the number of floating point registers needed.
572
573    let mut num_gp_registers = 0;
574    let mut num_fp_registers = 0;
575
576    let mut registers_for_primitive = |p| match p {
577        Primitive::Int(integer, _is_signed) => {
578            num_gp_registers += integer.size().bytes().div_ceil(8) as u32;
579        }
580        Primitive::Float(float) => {
581            num_fp_registers += float.size().bytes().div_ceil(16) as u32;
582        }
583        Primitive::Pointer(_) => {
584            num_gp_registers += 1;
585        }
586    };
587
588    match layout.layout.backend_repr() {
589        BackendRepr::Scalar(scalar) => {
590            registers_for_primitive(scalar.primitive());
591        }
592        BackendRepr::ScalarPair { a: scalar1, b: scalar2, b_offset: _ } => {
593            registers_for_primitive(scalar1.primitive());
594            registers_for_primitive(scalar2.primitive());
595        }
596        BackendRepr::SimdVector { .. } | BackendRepr::SimdScalableVector { .. } => {
597            // Because no instance of VaArgSafe uses a non-scalar `BackendRepr`.
598            {
    ::core::panicking::panic_fmt(format_args!("internal error: entered unreachable code: {0}",
            format_args!("No x86-64 SysV va_arg implementation for {0:?}",
                layout.layout.backend_repr())));
}unreachable!(
599                "No x86-64 SysV va_arg implementation for {:?}",
600                layout.layout.backend_repr()
601            )
602        }
603        BackendRepr::Memory { .. } => {
604            let mem_addr = x86_64_sysv64_va_arg_from_memory(bx, va_list_addr, layout);
605            return bx.load(layout.llvm_type(bx), mem_addr, layout.align.abi);
606        }
607    };
608
609    // AMD64-ABI 3.5.7p5: Step 3. Verify whether arguments fit into
610    // registers. In the case: l->gp_offset > 48 - num_gp * 8 or
611    // l->fp_offset > 176 - num_fp * 16 go to step 7.
612
613    // We support x86_64-unknown-linux-gnux32 which uses 4-byte pointers.
614    let unsigned_int_offset = 4;
615    let ptr_offset = bx.tcx().data_layout.pointer_size().bytes();
616
617    let gp_offset_ptr = va_list_addr;
618    let fp_offset_ptr = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(unsigned_int_offset));
619
620    let gp_offset_v = bx.load(bx.type_i32(), gp_offset_ptr, Align::from_bytes(8).unwrap());
621    let fp_offset_v = bx.load(bx.type_i32(), fp_offset_ptr, Align::from_bytes(4).unwrap());
622
623    let mut use_regs = bx.const_bool(false);
624
625    if num_gp_registers > 0 {
626        let max_offset_val = 48u32 - num_gp_registers * 8;
627        let fits_in_gp = bx.icmp(IntPredicate::IntULE, gp_offset_v, bx.const_u32(max_offset_val));
628        use_regs = fits_in_gp;
629    }
630
631    if num_fp_registers > 0 {
632        let max_offset_val = 176u32 - num_fp_registers * 16;
633        let fits_in_fp = bx.icmp(IntPredicate::IntULE, fp_offset_v, bx.const_u32(max_offset_val));
634        use_regs = if num_gp_registers > 0 { bx.and(use_regs, fits_in_fp) } else { fits_in_fp };
635    }
636
637    let in_reg = bx.append_sibling_block("va_arg.in_reg");
638    let in_mem = bx.append_sibling_block("va_arg.in_mem");
639    let end = bx.append_sibling_block("va_arg.end");
640
641    bx.cond_br(use_regs, in_reg, in_mem);
642
643    // Emit code to load the value if it was passed in a register.
644    bx.switch_to_block(in_reg);
645
646    // AMD64-ABI 3.5.7p5: Step 4. Fetch type from l->reg_save_area with
647    // an offset of l->gp_offset and/or l->fp_offset. This may require
648    // copying to a temporary location in case the parameter is passed
649    // in different register classes or requires an alignment greater
650    // than 8 for general purpose registers and 16 for XMM registers.
651    //
652    // FIXME(llvm): This really results in shameful code when we end up needing to
653    // collect arguments from different places; often what should result in a
654    // simple assembling of a structure from scattered addresses has many more
655    // loads than necessary. Can we clean this up?
656    let reg_save_area_ptr =
657        bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(2 * unsigned_int_offset + ptr_offset));
658    let reg_save_area_v = bx.load(bx.type_ptr(), reg_save_area_ptr, dl.pointer_align().abi);
659
660    let reg_addr = match layout.layout.backend_repr() {
661        BackendRepr::Scalar(scalar) => match scalar.primitive() {
662            Primitive::Int(_, _) | Primitive::Pointer(_) => {
663                let reg_addr = bx.inbounds_ptradd(reg_save_area_v, gp_offset_v);
664
665                // Copy into a temporary if the type is more aligned than the register save area.
666                let gp_align = Align::from_bytes(8).unwrap();
667                copy_to_temporary_if_more_aligned(bx, reg_addr, layout, gp_align)
668            }
669            Primitive::Float(_) => bx.inbounds_ptradd(reg_save_area_v, fp_offset_v),
670        },
671        BackendRepr::ScalarPair { a: scalar1, b: scalar2, b_offset: offset } => {
672            let ty_lo = bx.cx().scalar_pair_element_backend_type(layout, 0, false);
673            let ty_hi = bx.cx().scalar_pair_element_backend_type(layout, 1, false);
674
675            let align_lo = layout.field(bx.cx, 0).layout.align().abi;
676            let align_hi = layout.field(bx.cx, 1).layout.align().abi;
677
678            match (scalar1.primitive(), scalar2.primitive()) {
679                (Primitive::Float(_), Primitive::Float(_)) => {
680                    // SSE registers are spaced 16 bytes apart in the register save
681                    // area, we need to collect the two eightbytes together.
682                    // The ABI isn't explicit about this, but it seems reasonable
683                    // to assume that the slots are 16-byte aligned, since the stack is
684                    // naturally 16-byte aligned and the prologue is expected to store
685                    // all the SSE registers to the RSA.
686                    let reg_lo_addr = bx.inbounds_ptradd(reg_save_area_v, fp_offset_v);
687                    let reg_hi_addr = bx.inbounds_ptradd(reg_lo_addr, bx.const_i32(16));
688
689                    let align = layout.layout.align().abi;
690                    let tmp = bx.alloca(layout.size, layout.align.abi);
691
692                    let reg_lo = bx.load(ty_lo, reg_lo_addr, align_lo);
693                    let reg_hi = bx.load(ty_hi, reg_hi_addr, align_hi);
694
695                    let field0 = tmp;
696                    let field1 = bx.inbounds_ptradd(tmp, bx.const_u32(offset.bytes() as u32));
697
698                    bx.store(reg_lo, field0, align);
699                    bx.store(reg_hi, field1, align);
700
701                    tmp
702                }
703                (Primitive::Float(_), _) | (_, Primitive::Float(_)) => {
704                    let gp_addr = bx.inbounds_ptradd(reg_save_area_v, gp_offset_v);
705                    let fp_addr = bx.inbounds_ptradd(reg_save_area_v, fp_offset_v);
706
707                    let (reg_lo_addr, reg_hi_addr) = match scalar1.primitive() {
708                        Primitive::Float(_) => (fp_addr, gp_addr),
709                        Primitive::Int(_, _) | Primitive::Pointer(_) => (gp_addr, fp_addr),
710                    };
711
712                    let tmp = bx.alloca(layout.size, layout.align.abi);
713
714                    let reg_lo = bx.load(ty_lo, reg_lo_addr, align_lo);
715                    let reg_hi = bx.load(ty_hi, reg_hi_addr, align_hi);
716
717                    let field0 = tmp;
718                    let field1 = bx.inbounds_ptradd(tmp, bx.const_u32(offset.bytes() as u32));
719
720                    bx.store(reg_lo, field0, align_lo);
721                    bx.store(reg_hi, field1, align_hi);
722
723                    tmp
724                }
725                (_, _) => {
726                    // Two integer/pointer values are just contiguous in memory.
727                    let reg_addr = bx.inbounds_ptradd(reg_save_area_v, gp_offset_v);
728
729                    // Copy into a temporary if the type is more aligned than the register save area.
730                    let gp_align = Align::from_bytes(8).unwrap();
731                    copy_to_temporary_if_more_aligned(bx, reg_addr, layout, gp_align)
732                }
733            }
734        }
735        // The Previous match on `BackendRepr` means control flow already escaped.
736        BackendRepr::SimdVector { .. }
737        | BackendRepr::SimdScalableVector { .. }
738        | BackendRepr::Memory { .. } => ::core::panicking::panic("internal error: entered unreachable code")unreachable!(),
739    };
740
741    // AMD64-ABI 3.5.7p5: Step 5. Set:
742    // l->gp_offset = l->gp_offset + num_gp * 8
743    if num_gp_registers > 0 {
744        let offset = bx.const_u32(num_gp_registers * 8);
745        let sum = bx.add(gp_offset_v, offset);
746        // An alignment of 8 because `__va_list_tag` is 8-aligned and this is its first field.
747        bx.store(sum, gp_offset_ptr, Align::from_bytes(8).unwrap());
748    }
749
750    // l->fp_offset = l->fp_offset + num_fp * 16.
751    if num_fp_registers > 0 {
752        let offset = bx.const_u32(num_fp_registers * 16);
753        let sum = bx.add(fp_offset_v, offset);
754        bx.store(sum, fp_offset_ptr, Align::from_bytes(4).unwrap());
755    }
756
757    bx.br(end);
758
759    bx.switch_to_block(in_mem);
760    let mem_addr = x86_64_sysv64_va_arg_from_memory(bx, va_list_addr, layout);
761    bx.br(end);
762
763    bx.switch_to_block(end);
764
765    let val_type = layout.llvm_type(bx);
766    let val_addr = bx.phi(bx.type_ptr(), &[reg_addr, mem_addr], &[in_reg, in_mem]);
767
768    bx.load(val_type, val_addr, layout.align.abi)
769}
770
771/// Copy into a temporary if the type is more aligned than the register save area.
772fn copy_to_temporary_if_more_aligned<'ll, 'tcx>(
773    bx: &mut Builder<'_, 'll, 'tcx>,
774    reg_addr: &'ll Value,
775    layout: TyAndLayout<'tcx>,
776    src_align: Align,
777) -> &'ll Value {
778    if layout.layout.align.abi > src_align {
779        if !layout.ty.is_integral() {
    ::core::panicking::panic("assertion failed: layout.ty.is_integral()")
};assert!(layout.ty.is_integral());
780
781        // A memcpy below optimizes poorly for 128-bit integers.
782        let tmp = bx.alloca(layout.size, layout.align.abi);
783        let val = bx.load(layout.llvm_type(bx), reg_addr, src_align);
784        bx.store(val, tmp, layout.align.abi);
785        tmp
786    } else {
787        reg_addr
788    }
789}
790
791fn x86_64_sysv64_va_arg_from_memory<'ll, 'tcx>(
792    bx: &mut Builder<'_, 'll, 'tcx>,
793    va_list_addr: &'ll Value,
794    layout: TyAndLayout<'tcx>,
795) -> &'ll Value {
796    let dl = bx.cx.data_layout();
797    let ptr_align_abi = dl.data_layout().pointer_align().abi;
798
799    let overflow_arg_area_ptr = bx.inbounds_ptradd(va_list_addr, bx.const_usize(8));
800
801    let overflow_arg_area_v = bx.load(bx.type_ptr(), overflow_arg_area_ptr, ptr_align_abi);
802    // AMD64-ABI 3.5.7p5: Step 7. Align l->overflow_arg_area upwards to a 16
803    // byte boundary if alignment needed by type exceeds 8 byte boundary.
804    // It isn't stated explicitly in the standard, but in practice we use
805    // alignment greater than 16 where necessary.
806    // The AMD64 psABI leaves unspecified what to do for alignments above 16, but
807    // this behavior for 32+ alignment matches clang.
808    // It currently (2026 July) can only occur for 16-byte-aligned types.
809    let overflow_arg_area_v = if layout.layout.align.bytes() > 8 {
810        round_pointer_up_to_alignment(bx, overflow_arg_area_v, layout.layout.align.abi)
811    } else {
812        overflow_arg_area_v
813    };
814
815    // AMD64-ABI 3.5.7p5: Step 8. Fetch type from l->overflow_arg_area.
816    let mem_addr = overflow_arg_area_v;
817
818    // AMD64-ABI 3.5.7p5: Step 9. Set l->overflow_arg_area to:
819    // l->overflow_arg_area + sizeof(type).
820    // AMD64-ABI 3.5.7p5: Step 10. Align l->overflow_arg_area upwards to
821    // an 8 byte boundary.
822    let size_in_bytes = layout.layout.size().bytes();
823    let offset = bx.const_i32(size_in_bytes.next_multiple_of(8) as i32);
824    let overflow_arg_area = bx.inbounds_ptradd(overflow_arg_area_v, offset);
825    bx.store(overflow_arg_area, overflow_arg_area_ptr, ptr_align_abi);
826
827    mem_addr
828}
829
830fn emit_hexagon_va_arg_musl<'ll, 'tcx>(
831    bx: &mut Builder<'_, 'll, 'tcx>,
832    list: OperandRef<'tcx, &'ll Value>,
833    layout: TyAndLayout<'tcx>,
834) -> &'ll Value {
835    // Implementation of va_arg for Hexagon musl target.
836    // Based on LLVM's HexagonBuiltinVaList implementation.
837    //
838    // struct __va_list_tag {
839    //   void *__current_saved_reg_area_pointer;
840    //   void *__saved_reg_area_end_pointer;
841    //   void *__overflow_area_pointer;
842    // };
843    //
844    // All variadic arguments are passed on the stack, but the musl implementation
845    //  uses a register save area for compatibility.
846    let va_list_addr = list.immediate();
847    let ptr_align_abi = bx.tcx().data_layout.pointer_align().abi;
848    let ptr_size = bx.tcx().data_layout.pointer_size().bytes();
849
850    // Check if argument fits in register save area
851    let maybe_reg = bx.append_sibling_block("va_arg.maybe_reg");
852    let from_overflow = bx.append_sibling_block("va_arg.from_overflow");
853    let end = bx.append_sibling_block("va_arg.end");
854
855    // Load the three pointers from va_list
856    let current_ptr_addr = va_list_addr;
857    let end_ptr_addr = bx.inbounds_ptradd(va_list_addr, bx.const_usize(ptr_size));
858    let overflow_ptr_addr = bx.inbounds_ptradd(va_list_addr, bx.const_usize(2 * ptr_size));
859
860    let current_ptr = bx.load(bx.type_ptr(), current_ptr_addr, ptr_align_abi);
861    let end_ptr = bx.load(bx.type_ptr(), end_ptr_addr, ptr_align_abi);
862    let overflow_ptr = bx.load(bx.type_ptr(), overflow_ptr_addr, ptr_align_abi);
863
864    // Align current pointer based on argument type size (following LLVM's implementation)
865    // Arguments <= 32 bits (4 bytes) use 4-byte alignment, > 32 bits use 8-byte alignment
866    let type_size_bits = layout.size.bits();
867    let arg_align = if type_size_bits > 32 {
868        Align::from_bytes(8).unwrap()
869    } else {
870        Align::from_bytes(4).unwrap()
871    };
872    let aligned_current = round_pointer_up_to_alignment(bx, current_ptr, arg_align);
873
874    // Calculate next pointer position (following LLVM's logic)
875    // Arguments <= 32 bits take 4 bytes, > 32 bits take 8 bytes
876    let arg_size = if type_size_bits > 32 { 8 } else { 4 };
877    let next_ptr = bx.inbounds_ptradd(aligned_current, bx.const_usize(arg_size));
878
879    // Check if argument fits in register save area
880    let fits_in_regs = bx.icmp(IntPredicate::IntULE, next_ptr, end_ptr);
881    bx.cond_br(fits_in_regs, maybe_reg, from_overflow);
882
883    // Load from register save area
884    bx.switch_to_block(maybe_reg);
885    let reg_value_addr = aligned_current;
886    // Update current pointer
887    bx.store(next_ptr, current_ptr_addr, ptr_align_abi);
888    bx.br(end);
889
890    // Load from overflow area (stack)
891    bx.switch_to_block(from_overflow);
892
893    // Align overflow pointer using the same alignment rules
894    let aligned_overflow = round_pointer_up_to_alignment(bx, overflow_ptr, arg_align);
895
896    let overflow_value_addr = aligned_overflow;
897    // Update overflow pointer - use the same size calculation
898    let next_overflow = bx.inbounds_ptradd(aligned_overflow, bx.const_usize(arg_size));
899    bx.store(next_overflow, overflow_ptr_addr, ptr_align_abi);
900
901    // IMPORTANT: Also update the current saved register area pointer to match
902    // This synchronizes the pointers when switching to overflow area
903    bx.store(next_overflow, current_ptr_addr, ptr_align_abi);
904    bx.br(end);
905
906    // Return the value
907    bx.switch_to_block(end);
908    let value_addr =
909        bx.phi(bx.type_ptr(), &[reg_value_addr, overflow_value_addr], &[maybe_reg, from_overflow]);
910    bx.load(layout.llvm_type(bx), value_addr, layout.align.abi)
911}
912
913fn emit_hexagon_va_arg_bare_metal<'ll, 'tcx>(
914    bx: &mut Builder<'_, 'll, 'tcx>,
915    list: OperandRef<'tcx, &'ll Value>,
916    layout: TyAndLayout<'tcx>,
917) -> &'ll Value {
918    // Implementation of va_arg for Hexagon bare-metal (non-musl) targets.
919    // Based on LLVM's EmitVAArgForHexagon implementation.
920    //
921    // va_list is a simple pointer (char *)
922    let va_list_addr = list.immediate();
923    let ptr_align_abi = bx.tcx().data_layout.pointer_align().abi;
924
925    // Load current pointer from va_list
926    let current_ptr = bx.load(bx.type_ptr(), va_list_addr, ptr_align_abi);
927
928    // Handle address alignment for types with alignment > 4 bytes
929    let ty_align = layout.align.abi;
930    let aligned_ptr = if ty_align.bytes() > 4 {
931        // Ensure alignment is a power of 2
932        if true {
    if !ty_align.bytes().is_power_of_two() {
        {
            ::core::panicking::panic_fmt(format_args!("Alignment is not power of 2!"));
        }
    };
};debug_assert!(ty_align.bytes().is_power_of_two(), "Alignment is not power of 2!");
933        round_pointer_up_to_alignment(bx, current_ptr, ty_align)
934    } else {
935        current_ptr
936    };
937
938    // Calculate offset: round up type size to 4-byte boundary (minimum stack slot size)
939    let type_size = layout.size.bytes();
940    let offset = type_size.next_multiple_of(4); // align to 4 bytes
941
942    // Update va_list to point to next argument
943    let next_ptr = bx.inbounds_ptradd(aligned_ptr, bx.const_usize(offset));
944    bx.store(next_ptr, va_list_addr, ptr_align_abi);
945
946    // Load and return the argument value
947    bx.load(layout.llvm_type(bx), aligned_ptr, layout.align.abi)
948}
949
950fn emit_xtensa_va_arg<'ll, 'tcx>(
951    bx: &mut Builder<'_, 'll, 'tcx>,
952    list: OperandRef<'tcx, &'ll Value>,
953    layout: TyAndLayout<'tcx>,
954) -> &'ll Value {
955    // Implementation of va_arg for Xtensa. There doesn't seem to be an authoritative source for
956    // this, other than "what GCC does".
957    //
958    // The va_list type has three fields:
959    // struct __va_list_tag {
960    //   int32_t *va_stk; // Arguments passed on the stack
961    //   int32_t *va_reg; // Arguments passed in registers, saved to memory by the prologue.
962    //   int32_t va_ndx; // Offset into the arguments, in bytes
963    // };
964    //
965    // The first 24 bytes (equivalent to 6 registers) come from va_reg, the rest from va_stk.
966    // Thus if va_ndx is less than 24, the next va_arg *may* read from va_reg,
967    // otherwise it must come from va_stk.
968    //
969    // Primitive arguments are never split between registers and the stack. For example, if loading an 8 byte
970    // primitive value and va_ndx = 20, we instead bump the offset and read everything from va_stk.
971    let va_list_addr = list.immediate();
972    // FIXME: handle multi-field structs that split across regsave/stack?
973    let from_stack = bx.append_sibling_block("va_arg.from_stack");
974    let from_regsave = bx.append_sibling_block("va_arg.from_regsave");
975    let end = bx.append_sibling_block("va_arg.end");
976    let ptr_align_abi = bx.tcx().data_layout.pointer_align().abi;
977
978    // (*va).va_ndx
979    let va_reg_offset = 4;
980    let va_ndx_offset = va_reg_offset + 4;
981    let offset_ptr = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(va_ndx_offset));
982
983    let offset = bx.load(bx.type_i32(), offset_ptr, bx.tcx().data_layout.i32_align);
984    let offset = round_up_to_alignment(bx, offset, layout.align.abi);
985
986    let slot_size = layout.size.align_to(Align::from_bytes(4).unwrap()).bytes() as i32;
987
988    // Update the offset in va_list, by adding the slot's size.
989    let offset_next = bx.add(offset, bx.const_i32(slot_size));
990
991    // Figure out where to look for our value. We do that by checking the end of our slot (offset_next).
992    // If that is within the regsave area, then load from there. Otherwise load from the stack area.
993    let regsave_size = bx.const_i32(24);
994    let use_regsave = bx.icmp(IntPredicate::IntULE, offset_next, regsave_size);
995    bx.cond_br(use_regsave, from_regsave, from_stack);
996
997    bx.switch_to_block(from_regsave);
998    // update va_ndx
999    bx.store(offset_next, offset_ptr, ptr_align_abi);
1000
1001    // (*va).va_reg
1002    let regsave_area_ptr = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(va_reg_offset));
1003    let regsave_area = bx.load(bx.type_ptr(), regsave_area_ptr, ptr_align_abi);
1004    let regsave_value_ptr = bx.inbounds_ptradd(regsave_area, offset);
1005    bx.br(end);
1006
1007    bx.switch_to_block(from_stack);
1008
1009    // The first time we switch from regsave to stack we needs to adjust our offsets a bit.
1010    // va_stk is set up such that the first stack argument is always at va_stk + 32.
1011    // The corrected offset is written back into the va_list struct.
1012
1013    // let offset_corrected = cmp::max(offset, 32);
1014    let stack_offset_start = bx.const_i32(32);
1015    let needs_correction = bx.icmp(IntPredicate::IntULE, offset, stack_offset_start);
1016    let offset_corrected = bx.select(needs_correction, stack_offset_start, offset);
1017
1018    // let offset_next_corrected = offset_corrected + slot_size;
1019    // va_ndx = offset_next_corrected;
1020    let offset_next_corrected = bx.add(offset_corrected, bx.const_i32(slot_size));
1021    // update va_ndx
1022    bx.store(offset_next_corrected, offset_ptr, ptr_align_abi);
1023
1024    // let stack_value_ptr = unsafe { (*va).va_stk.byte_add(offset_corrected) };
1025    let stack_area_ptr = bx.inbounds_ptradd(va_list_addr, bx.cx.const_usize(0));
1026    let stack_area = bx.load(bx.type_ptr(), stack_area_ptr, ptr_align_abi);
1027    let stack_value_ptr = bx.inbounds_ptradd(stack_area, offset_corrected);
1028    bx.br(end);
1029
1030    bx.switch_to_block(end);
1031
1032    // On big-endian, for values smaller than the slot size we'd have to align the read to the end
1033    // of the slot rather than the start. While the ISA and GCC support big-endian, all the Xtensa
1034    // targets supported by rustc are little-endian so don't worry about it.
1035
1036    // if from_regsave {
1037    //     unsafe { *regsave_value_ptr }
1038    // } else {
1039    //     unsafe { *stack_value_ptr }
1040    // }
1041    if !(bx.tcx().sess.target.endian == Endian::Little) {
    ::core::panicking::panic("assertion failed: bx.tcx().sess.target.endian == Endian::Little")
};assert!(bx.tcx().sess.target.endian == Endian::Little);
1042    let value_ptr =
1043        bx.phi(bx.type_ptr(), &[regsave_value_ptr, stack_value_ptr], &[from_regsave, from_stack]);
1044    return bx.load(layout.llvm_type(bx), value_ptr, layout.align.abi);
1045}
1046
1047/// Determine the va_arg implementation to use. The LLVM va_arg instruction
1048/// is lacking in some instances, so we should only use it as a fallback.
1049///
1050/// <https://llvm.org/docs/LangRef.html#va-arg-instruction>
1051pub(super) fn emit_va_arg<'ll, 'tcx>(
1052    bx: &mut Builder<'_, 'll, 'tcx>,
1053    addr: OperandRef<'tcx, &'ll Value>,
1054    layout: TyAndLayout<'tcx>,
1055) -> &'ll Value {
1056    // Some ABIs have special behavior for zero-sized types. currently `VaArgSafe` is not
1057    // implemented for any zero-sized types, so this assert should always hold.
1058    if !!layout.is_zst() {
    ::core::panicking::panic("assertion failed: !layout.is_zst()")
};assert!(!layout.is_zst());
1059
1060    let target = &bx.cx.tcx.sess.target;
1061    let stability = target.supports_c_variadic_definitions();
1062
1063    match target.arch {
1064        Arch::X86 => {
1065            // A small deviation from clang to get the right behavior for f128.
1066            //
1067            // Note that i64 and f64 have an alignment of only 4 on this architecture.
1068            // We need to be careful when adding future types with an alignment bigger
1069            // than 4 (e.g. i128), clang has a bunch of custom logic for them.
1070            let allow_higher_align = if layout.ty == bx.tcx().types.f128 {
1071                AllowHigherAlign::Yes
1072            } else {
1073                AllowHigherAlign::No
1074            };
1075
1076            emit_ptr_va_arg(
1077                bx,
1078                addr,
1079                layout,
1080                PassMode::Direct,
1081                SlotSize::Bytes4,
1082                allow_higher_align,
1083                ForceRightAdjust::No,
1084            )
1085        }
1086        Arch::Arm64EC => emit_ptr_va_arg(
1087            bx,
1088            addr,
1089            layout,
1090            // MS x64 ABI requirement: "Any argument that doesn't fit in 8 bytes, or is
1091            // not 1, 2, 4, or 8 bytes, must be passed by reference."
1092            if layout.size.bytes() > 8 || !layout.size.bytes().is_power_of_two() {
1093                PassMode::Indirect
1094            } else {
1095                PassMode::Direct
1096            },
1097            SlotSize::Bytes8,
1098            AllowHigherAlign::No,
1099            ForceRightAdjust::No,
1100        ),
1101        Arch::AArch64 if target.is_like_windows || target.is_like_darwin => emit_ptr_va_arg(
1102            bx,
1103            addr,
1104            layout,
1105            PassMode::Direct,
1106            SlotSize::Bytes8,
1107            AllowHigherAlign::Yes,
1108            ForceRightAdjust::No,
1109        ),
1110        Arch::AArch64 => emit_aapcs_va_arg(bx, addr, layout),
1111        Arch::Arm => {
1112            // Types wider than 16 bytes are not currently supported. Clang has special logic for
1113            // such types, but `VaArgSafe` is not implemented for any type that is this large on
1114            // arm (i.e. 32-bit) targets.
1115            if !(layout.size.bytes() <= 16) {
    ::core::panicking::panic("assertion failed: layout.size.bytes() <= 16")
};assert!(layout.size.bytes() <= 16);
1116
1117            emit_ptr_va_arg(
1118                bx,
1119                addr,
1120                layout,
1121                PassMode::Direct,
1122                SlotSize::Bytes4,
1123                AllowHigherAlign::Yes,
1124                ForceRightAdjust::No,
1125            )
1126        }
1127        Arch::S390x => emit_s390x_va_arg(bx, addr, layout),
1128        Arch::PowerPC => emit_powerpc_va_arg(bx, addr, layout),
1129        Arch::PowerPC64 => emit_ptr_va_arg(
1130            bx,
1131            addr,
1132            layout,
1133            PassMode::Direct,
1134            SlotSize::Bytes8,
1135            AllowHigherAlign::Yes,
1136            // ForceRightAdjust only takes effect on big-endian architectures.
1137            ForceRightAdjust::Yes,
1138        ),
1139        Arch::RiscV32 if target.llvm_abiname == LlvmAbi::Ilp32e => {
1140            {
    match stability {
        CVariadicStatus::Unstable { .. } => {}
        ref left_val => {
            ::core::panicking::assert_matches_failed(left_val,
                "CVariadicStatus::Unstable { .. }",
                ::core::option::Option::None);
        }
    }
};std::assert_matches!(stability, CVariadicStatus::Unstable { .. });
1141            // FIXME: clang manually adjusts the alignment for this ABI. It notes:
1142            //
1143            // > To be compatible with GCC's behaviors, we force arguments with
1144            // > 2×XLEN-bit alignment and size at most 2×XLEN bits like `long long`,
1145            // > `unsigned long long` and `double` to have 4-byte alignment. This
1146            // > behavior may be changed when RV32E/ILP32E is ratified.
1147            ::rustc_span::macros::bug_impl(None,
    format_args!("c-variadic calls with ilp32e use a custom ABI and are not currently implemented"),
    Location::caller());bug!("c-variadic calls with ilp32e use a custom ABI and are not currently implemented");
1148        }
1149        Arch::RiscV32 | Arch::LoongArch32 => emit_ptr_va_arg(
1150            bx,
1151            addr,
1152            layout,
1153            if layout.size.bytes() > 2 * 4 { PassMode::Indirect } else { PassMode::Direct },
1154            SlotSize::Bytes4,
1155            AllowHigherAlign::Yes,
1156            ForceRightAdjust::No,
1157        ),
1158        Arch::RiscV64 | Arch::LoongArch64 => emit_ptr_va_arg(
1159            bx,
1160            addr,
1161            layout,
1162            if layout.size.bytes() > 2 * 8 { PassMode::Indirect } else { PassMode::Direct },
1163            SlotSize::Bytes8,
1164            AllowHigherAlign::Yes,
1165            ForceRightAdjust::No,
1166        ),
1167        Arch::AmdGpu => emit_ptr_va_arg(
1168            bx,
1169            addr,
1170            layout,
1171            PassMode::Direct,
1172            SlotSize::Bytes4,
1173            AllowHigherAlign::No,
1174            ForceRightAdjust::No,
1175        ),
1176        Arch::Nvptx64 => emit_ptr_va_arg(
1177            bx,
1178            addr,
1179            layout,
1180            PassMode::Direct,
1181            SlotSize::Bytes1,
1182            AllowHigherAlign::Yes,
1183            ForceRightAdjust::No,
1184        ),
1185        Arch::Wasm32 | Arch::Wasm64 => emit_ptr_va_arg(
1186            bx,
1187            addr,
1188            layout,
1189            if layout.is_aggregate() || layout.is_zst() || layout.is_1zst() {
1190                PassMode::Indirect
1191            } else {
1192                PassMode::Direct
1193            },
1194            SlotSize::Bytes4,
1195            AllowHigherAlign::Yes,
1196            ForceRightAdjust::No,
1197        ),
1198        Arch::CSky => emit_ptr_va_arg(
1199            bx,
1200            addr,
1201            layout,
1202            PassMode::Direct,
1203            SlotSize::Bytes4,
1204            AllowHigherAlign::Yes,
1205            ForceRightAdjust::No,
1206        ),
1207        // Windows x86_64
1208        Arch::X86_64 if target.is_like_windows => emit_ptr_va_arg(
1209            bx,
1210            addr,
1211            layout,
1212            if layout.size.bytes() > 8 || !layout.size.bytes().is_power_of_two() {
1213                PassMode::Indirect
1214            } else {
1215                PassMode::Direct
1216            },
1217            SlotSize::Bytes8,
1218            AllowHigherAlign::No,
1219            ForceRightAdjust::No,
1220        ),
1221        // This includes `target.is_like_darwin`, which on x86_64 targets is like sysv64.
1222        Arch::X86_64 => emit_x86_64_sysv64_va_arg(bx, addr, layout),
1223        Arch::Xtensa => emit_xtensa_va_arg(bx, addr, layout),
1224        Arch::Hexagon => match target.env {
1225            Env::Musl => emit_hexagon_va_arg_musl(bx, addr, layout),
1226            _ => emit_hexagon_va_arg_bare_metal(bx, addr, layout),
1227        },
1228        Arch::Sparc64 => emit_ptr_va_arg(
1229            bx,
1230            addr,
1231            layout,
1232            if layout.size.bytes() > 2 * 8 { PassMode::Indirect } else { PassMode::Direct },
1233            SlotSize::Bytes8,
1234            AllowHigherAlign::Yes,
1235            // sparc64 is a big-endian target and stores variable arguments right-adjusted.
1236            ForceRightAdjust::Yes,
1237        ),
1238        Arch::Sparc => {
1239            {
    match stability {
        CVariadicStatus::Unstable { .. } => {}
        ref left_val => {
            ::core::panicking::assert_matches_failed(left_val,
                "CVariadicStatus::Unstable { .. }",
                ::core::option::Option::None);
        }
    }
};std::assert_matches!(stability, CVariadicStatus::Unstable { .. });
1240
1241            // f128 is passed indirectly.
1242            let pass_mode = match layout.layout.backend_repr() {
1243                BackendRepr::Scalar(scalar) => match scalar.primitive() {
1244                    Primitive::Float(Float::F128) => PassMode::Indirect,
1245                    _ => PassMode::Direct,
1246                },
1247                _ => PassMode::Direct,
1248            };
1249
1250            emit_ptr_va_arg(
1251                bx,
1252                addr,
1253                layout,
1254                pass_mode,
1255                SlotSize::Bytes4,
1256                AllowHigherAlign::No,
1257                ForceRightAdjust::Yes,
1258            )
1259        }
1260        Arch::Mips | Arch::Mips32r6 | Arch::Mips64 | Arch::Mips64r6 => emit_ptr_va_arg(
1261            bx,
1262            addr,
1263            layout,
1264            PassMode::Direct,
1265            match &target.llvm_abiname {
1266                LlvmAbi::N32 | LlvmAbi::N64 => SlotSize::Bytes8,
1267                LlvmAbi::O32 => SlotSize::Bytes4,
1268                other => ::rustc_span::macros::bug_impl(None,
    format_args!("unexpected LLVM ABI {0}", other), Location::caller())bug!("unexpected LLVM ABI {other}"),
1269            },
1270            AllowHigherAlign::Yes,
1271            // In big-endian mode the actual value is stored in the right side of the slot, meaning
1272            // that when the value is smaller than a slot, we need to adjust the pointer we read
1273            // to somewhere in the middle of the slot.
1274            match bx.tcx().sess.target.endian {
1275                Endian::Big => ForceRightAdjust::Yes,
1276                Endian::Little => ForceRightAdjust::No,
1277            },
1278        ),
1279
1280        Arch::Bpf => ::rustc_span::macros::bug_impl(None,
    format_args!("bpf does not support c-variadic functions"),
    Location::caller())bug!("bpf does not support c-variadic functions"),
1281        Arch::SpirV => ::rustc_span::macros::bug_impl(None,
    format_args!("spirv does not support c-variadic functions"),
    Location::caller())bug!("spirv does not support c-variadic functions"),
1282
1283        Arch::Avr | Arch::M68k | Arch::Msp430 => {
1284            {
    match stability {
        CVariadicStatus::Unstable { .. } => {}
        ref left_val => {
            ::core::panicking::assert_matches_failed(left_val,
                "CVariadicStatus::Unstable { .. }",
                ::core::option::Option::None);
        }
    }
};std::assert_matches!(stability, CVariadicStatus::Unstable { .. });
1285
1286            // Clang uses the LLVM implementation for these architectures.
1287            bx.va_arg(addr.immediate(), layout.llvm_type(bx.cx))
1288        }
1289
1290        Arch::Other(ref arch) => {
1291            {
    match stability {
        CVariadicStatus::Unstable { .. } => {}
        ref left_val => {
            ::core::panicking::assert_matches_failed(left_val,
                "CVariadicStatus::Unstable { .. }",
                ::core::option::Option::None);
        }
    }
};std::assert_matches!(stability, CVariadicStatus::Unstable { .. });
1292
1293            // Just to be safe we error out explicitly here, instead of crossing our fingers that
1294            // the default LLVM implementation has the correct behavior for this target.
1295            ::rustc_span::macros::bug_impl(None,
    format_args!("c-variadic functions are not currently implemented for custom target {0}",
        arch), Location::caller())bug!("c-variadic functions are not currently implemented for custom target {arch}")
1296        }
1297    }
1298}