Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
17 commits
Select commit Hold shift + click to select a range
08f3fa0
cc: build c17 against the library instead of recompiling every module
jgarzik Sep 29, 2026
d910ef4
cc: link a for loop's back edge from the block the branch lands in
jgarzik Sep 30, 2026
3caab02
cc: discard array initializers past the last element instead of stori…
jgarzik Sep 30, 2026
4b99687
cc: mark volatile accesses on the instruction so DCE keeps them
jgarzik Sep 30, 2026
0738a07
cc: qualify a member access by the object that holds it
jgarzik Sep 30, 2026
98e4b03
cc: build every two-armed conditional through one forking helper
jgarzik Sep 30, 2026
36f6318
cc: move every block of bytes through the one helper that bounds it
jgarzik Sep 30, 2026
e285d58
cc: move a parameter's bytes in its own widths, and bound the bulk ones
jgarzik Sep 30, 2026
cf75b55
cc: make entering a declaration scope enter its VLA scope
jgarzik Sep 30, 2026
33ff4e0
cc: override the subobject a designator names, not everything it over…
jgarzik Sep 30, 2026
2eb9cbc
cc: compute an atomic compound assignment the way the ordinary one is
jgarzik Sep 30, 2026
c301bfb
cc: convert a case label to the type the switch compares at
jgarzik Sep 30, 2026
c54f27e
cc: ask one question about whether a division traps
jgarzik Sep 30, 2026
18ee7ed
cc: give the caller the bytes of an inlined aggregate return, not its…
jgarzik Sep 30, 2026
0a33e60
cc: override a bit-field, and a member of the union already held
jgarzik Sep 30, 2026
87489eb
cc: widen a store over a slot only where the slot is one scalar
jgarzik Sep 30, 2026
ce123a4
cc: ask both back ends' store lowerings the same question about a slot
jgarzik Sep 30, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 8 additions & 8 deletions cc/arch/aarch64/call.rs
Original file line number Diff line number Diff line change
Expand Up @@ -750,14 +750,15 @@ impl Aarch64CodeGen {
if let StackKind::Composite { bytes } = stack_arg.kind {
// The pseudo locates the aggregate; its bytes go into the
// slot.
// AAPCS64 B.4 replaces a composite above sixteen bytes with a
// pointer to the caller's copy, so the run here is bounded by
// the ABI at two eightbytes; the widths are `block_chunks`'s,
// so a composite that is not a multiple of eight reads and
// writes its tail as wide as the tail is.
let src = self.aggregate_arg_address(stack_arg.pseudo);
let mut done = 0;
while done < bytes {
let chunk = [8, 4, 2, 1]
.into_iter()
.find(|c| *c <= bytes - done)
.unwrap_or(1);
let size = OperandSize::from_bits(chunk as u32 * 8);
for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) {
let size = OperandSize::from_bits(chunk.bits());
let done = done as i32;
self.push_lir(Aarch64Inst::Ldr {
size,
addr: MemAddr::BaseOffset {
Expand All @@ -774,7 +775,6 @@ impl Aarch64CodeGen {
offset: offset + done,
},
});
done += chunk;
}
continue;
}
Expand Down
4 changes: 2 additions & 2 deletions cc/arch/aarch64/codegen.rs
Original file line number Diff line number Diff line change
Expand Up @@ -78,7 +78,7 @@ pub struct Aarch64CodeGen {
/// Stack allocation size for locals (for zero_stack_frame)
pub(super) stack_alloc_size: i32,
/// Sym pseudo ID → type size in bits (for distinguishing scalar vs struct stores)
pub(super) sym_type_sizes: HashMap<PseudoId, u32>,
pub(super) sym_slots: HashMap<PseudoId, crate::arch::codegen::SymSlot>,
/// Which register this function's locals are addressed through
pub(super) frame_base: FrameBase,
}
Expand All @@ -103,7 +103,7 @@ impl Aarch64CodeGen {
pic_mode: false,
unique_label_counter: 0,
stack_alloc_size: 0,
sym_type_sizes: HashMap::new(),
sym_slots: HashMap::new(),
frame_base: FrameBase::Fp,
}
}
Expand Down
58 changes: 48 additions & 10 deletions cc/arch/aarch64/features.rs
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@

use super::call::HfaElem;
use super::codegen::Aarch64CodeGen;
use super::frame::UNROLL_LIMIT_BYTES;
use super::lir::{Aarch64Inst, GpOperand, MemAddr};
use super::regalloc::{Loc, Reg, VReg};
use crate::arch::codegen::BswapSize;
Expand Down Expand Up @@ -636,8 +637,15 @@ impl Aarch64CodeGen {
}

/// Copy `bytes` of an aggregate from `[addr + off]` into the destination,
/// in descending power-of-two chunks so nothing past the object is
/// written. X16 is the shuttle -- linker scratch, never allocated.
/// in the descending power-of-two chunks `block_chunks` gives, so nothing
/// past the object is written. X16 is the shuttle -- linker scratch, never
/// allocated.
///
/// Past [`UNROLL_LIMIT_BYTES`] the copy becomes a counted loop, which
/// **advances `addr`**: only the two `VaAggKind::Indirect` paths can reach
/// that size, and both pass a register they are finished with. Every other
/// kind is at most sixteen bytes (a composite in general registers) or
/// sixty-four (an HFA of four binary128s), so it never gets there.
fn emit_va_arg_bytes(
&mut self,
dst_loc: &Loc,
Expand All @@ -663,13 +671,44 @@ impl Aarch64CodeGen {
return;
}
}
let mut done = 0;
while done < bytes {
let chunk = [8, 4, 2, 1]
.into_iter()
.find(|c| *c <= bytes - done)
.unwrap_or(1);
let size = OperandSize::from_bits(chunk as u32 * 8);
if i64::from(bytes) > UNROLL_LIMIT_BYTES {
// X16 becomes the destination cursor; the source cursor is `addr`
// itself, advanced past the object.
match dst_loc {
Loc::Stack(_) | Loc::IncomingArg(_) => {
let (base, disp) = self
.loc_addr_parts(dst_loc)
.expect("a frame location has a base and a displacement");
self.push_lir(Aarch64Inst::Add {
size: OperandSize::B64,
src1: base,
src2: GpOperand::Imm(disp.into()),
dst: Reg::X16,
});
}
// The register holds the aggregate's address, and it is the
// result, so the cursor is a copy of it.
Loc::Reg(r) if !holds_value => self.push_lir(Aarch64Inst::Mov {
size: OperandSize::B64,
src: GpOperand::Reg(*r),
dst: Reg::X16,
}),
_ => return,
}
if off != 0 {
self.push_lir(Aarch64Inst::Add {
size: OperandSize::B64,
src1: addr,
src2: GpOperand::Imm(off.into()),
dst: addr,
});
}
self.emit_block_copy_loop(Reg::X16, addr, bytes.into());
return;
}
for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) {
let size = OperandSize::from_bits(chunk.bits());
let done = done as i32;
self.push_lir(Aarch64Inst::Ldr {
size,
addr: MemAddr::BaseOffset {
Expand Down Expand Up @@ -700,7 +739,6 @@ impl Aarch64CodeGen {
}),
_ => return,
}
done += chunk;
}
}

Expand Down
108 changes: 90 additions & 18 deletions cc/arch/aarch64/frame.rs
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,12 @@ use crate::ir::{Function, Instruction, PseudoId, PseudoKind};
use crate::types::{TypeId, TypeKind, TypeTable};
use std::collections::HashSet;

/// The most bytes this back end moves with unrolled loads and stores; past it,
/// [`Aarch64CodeGen::emit_block_copy_loop`]. It is the bound the IR puts on an
/// expanded `memcpy`, for the same reason: the instruction count of an unrolled
/// copy is linear in the object.
pub(super) const UNROLL_LIMIT_BYTES: i64 = crate::ir::memexpand::INLINE_LIMIT_BYTES;

impl Aarch64CodeGen {
pub(super) fn emit_function(&mut self, func: &Function, types: &TypeTable) {
self.base.func_pos = crate::arch::func_pos(func);
Expand All @@ -45,16 +51,7 @@ impl Aarch64CodeGen {
self.locations = alloc.allocate(func, types);
self.pseudos = crate::arch::codegen::PseudoTable::new(&func.pseudos);

// Build sym type size map for emit_store to distinguish struct fields from scalars
self.sym_type_sizes.clear();
for pseudo in &func.pseudos {
// By identity: a global whose name collides with a parameter's
// would otherwise be recorded with the parameter's type size.
if let Some(local_var) = func.local_of(pseudo.id) {
self.sym_type_sizes
.insert(pseudo.id, types.size_bits(local_var.typ));
}
}
self.sym_slots = crate::arch::codegen::sym_slots(func, types);

let stack_size = alloc.stack_size();
self.frame_base = alloc.frame_base();
Expand Down Expand Up @@ -538,6 +535,81 @@ impl Aarch64CodeGen {
}
}

/// Copy `bytes` bytes from `[src]` to `[dst]`, in a counted loop over the
/// whole sixteen-byte pairs and `block_chunks` for what is left.
///
/// This is the back end's bulk block move. It cannot synthesize a call to
/// `memcpy` -- it is past the point where a call can be built -- and one
/// load/store pair per chunk is linear in the object: 4 KB of `va_arg`
/// aggregate cost about 1100 instructions, and 256 KB would be the
/// compile-time explosion the IR's own
/// [`crate::ir::memexpand::INLINE_LIMIT_BYTES`] exists to prevent. A
/// counted loop is the answer [`Self::emit_zero_loop`] gives the frame.
///
/// Sixteen bytes an iteration through V16, which is reserved codegen
/// scratch: it keeps the loop to three general registers, which is all the
/// `va_arg` sequence has free. Both `src` and `dst` are cursors and are
/// left past the object, so the caller must be finished with them; X17
/// holds the end of the paired part and then shuttles the tail.
pub(super) fn emit_block_copy_loop(&mut self, dst: Reg, src: Reg, bytes: i64) {
let pairs = bytes & !15;
if pairs > 0 {
self.push_lir(Aarch64Inst::Add {
size: OperandSize::B64,
src1: src,
src2: GpOperand::Imm(pairs),
dst: Reg::X17,
});
let top = self.next_unique_label("block_copy");
self.push_lir(Aarch64Inst::Directive(Directive::BlockLabel(top.clone())));
self.push_lir(Aarch64Inst::LdrFp {
size: FpSize::Quad,
addr: MemAddr::PostIndex {
base: src,
offset: 16,
},
dst: VReg::V16,
});
self.push_lir(Aarch64Inst::StrFp {
size: FpSize::Quad,
src: VReg::V16,
addr: MemAddr::PostIndex {
base: dst,
offset: 16,
},
});
self.push_lir(Aarch64Inst::Cmp {
size: OperandSize::B64,
src1: src,
src2: GpOperand::Reg(Reg::X17),
});
self.push_lir(Aarch64Inst::BCond {
cond: CondCode::Ult,
target: top,
});
}
// The cursors point at the tail, so its pieces are offsets from them.
for (off, chunk) in crate::ir::memexpand::block_chunks(bytes - pairs) {
let size = OperandSize::from_bits(chunk.bits());
self.push_lir(Aarch64Inst::Ldr {
size,
addr: MemAddr::BaseOffset {
base: src,
offset: off as i32,
},
dst: Reg::X17,
});
self.push_lir(Aarch64Inst::Str {
size,
src: Reg::X17,
addr: MemAddr::BaseOffset {
base: dst,
offset: off as i32,
},
});
}
}

/// Save callee-saved GP registers in pairs (or single if odd count)
fn save_callee_saved_gp_regs(&mut self, total_frame: i32, callee_saved: &[Reg]) {
let mut offset = 16; // Start after fp/lr
Expand Down Expand Up @@ -960,13 +1032,14 @@ impl Aarch64CodeGen {
crate::arch::func_pos(func),
"a stacked parameter",
);
let mut done = 0;
while done < bytes {
let chunk = [8, 4, 2, 1]
.into_iter()
.find(|c| *c <= bytes - done)
.unwrap_or(1);
let size = OperandSize::from_bits(chunk as u32 * 8);
// A composite that reaches here is at most two
// eightbytes, so the run is bounded by the ABI;
// the widths are `block_chunks`'s so the tail of
// one that is not a multiple of eight is moved as
// wide as it is and no wider.
for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) {
let size = OperandSize::from_bits(chunk.bits());
let done = done as i32;
self.push_lir(Aarch64Inst::Ldr {
size,
addr: self.incoming_mem_plus(incoming, done),
Expand All @@ -977,7 +1050,6 @@ impl Aarch64CodeGen {
src: Reg::X16,
addr: self.stack_mem_plus(local_off, done),
});
done += chunk;
}
}
}
Expand Down
28 changes: 9 additions & 19 deletions cc/arch/aarch64/memory.rs
Original file line number Diff line number Diff line change
Expand Up @@ -601,26 +601,16 @@ impl Aarch64CodeGen {
return;
}

// Widen 32-bit stores at offset 0 to 64-bit to prevent stale
// upper bits when a 32-bit result is stored into a 64-bit
// local (e.g., int-to-long, int-to-pointer assignments).
// Only widen for known local variables (in sym_type_sizes).
// Do NOT widen stores to globals/statics (not in sym_type_sizes)
// or stores through pointers — widening could clobber adjacent data.
// Exception: struct/union fields at offset 0 must use exact
// size to avoid clobbering the adjacent field at offset 4.
// Widen a 32-bit store at offset 0 to 64 bits, so a narrow value going
// into a wider slot leaves no stale upper bits behind it (an
// int-to-long or int-to-pointer assignment). Only where the slot holds
// one scalar: see `SymSlot`. Only for a known local, too -- a global or
// a store through a pointer keeps its exact width, since nothing here
// knows what adjoins it.
let store_size = if mem_size == 32 && insn.offset == 0 {
if let Some(&sym_bits) = self.sym_type_sizes.get(&addr) {
// Known local variable — safe to widen if scalar and > 32 bits
if sym_bits > 64 {
OperandSize::from_bits(mem_size) // struct field: exact size
} else if sym_bits > 32 {
OperandSize::B64 // scalar/pointer local: safe to widen
} else {
OperandSize::from_bits(mem_size)
}
} else {
OperandSize::from_bits(mem_size) // global/static/pointer: exact size
match self.sym_slots.get(&addr) {
Some(slot) if slot.widenable() && slot.bits > 32 => OperandSize::B64,
_ => OperandSize::from_bits(mem_size),
}
} else {
OperandSize::from_bits(mem_size)
Expand Down
50 changes: 50 additions & 0 deletions cc/arch/codegen.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1222,6 +1222,56 @@ pub fn check_tls_reached_only_by_address(
}
}

/// What a store lowering needs to know about a local's stack slot.
///
/// Both back ends widen a narrow store at offset 0 of a local so that a value
/// going into a wider slot leaves no stale upper bits behind it. That is only
/// sound where the slot holds a single scalar: an aggregate or a complex has
/// another member at offset 4 or 8, which the widened store would write over.
///
/// The size alone cannot answer it -- a `long` and a `struct { int x, y; }`
/// are both sixty-four bits -- and asking for *more* than sixty-four spares
/// only the aggregates too large to be mistaken for a scalar in the first
/// place. Both back ends had that test and both got an eight-byte aggregate
/// wrong, so the question is asked once, here.
pub struct SymSlot {
/// Width of the declared type, in bits.
pub bits: u32,
/// The slot holds a single scalar value, so anything above a narrow store
/// at offset 0 is stale bits of that same object.
pub one_scalar: bool,
}

impl SymSlot {
/// Whether a 32-bit store at offset 0 of this slot may be widened to 64.
pub fn widenable(&self) -> bool {
self.one_scalar && self.bits <= 64
}
}

/// Record, for each of `func`'s locals, what its stack slot holds.
pub fn sym_slots(
func: &crate::ir::Function,
types: &crate::types::TypeTable,
) -> std::collections::HashMap<crate::ir::PseudoId, SymSlot> {
let mut slots = std::collections::HashMap::new();
for pseudo in &func.pseudos {
// By identity: a global whose name collides with a parameter's would
// otherwise be recorded with the parameter's type.
if let Some(local_var) = func.local_of(pseudo.id) {
let typ = local_var.typ;
slots.insert(
pseudo.id,
SymSlot {
bits: types.size_bits(typ),
one_scalar: types.is_scalar(typ) && !types.is_complex(typ),
},
);
}
}
slots
}

/// The current function's pseudos, looked up by id.
///
/// A pseudo's id is not its position in `Function::pseudos`, so a lookup
Expand Down
Loading
Loading