diff --git a/cc/arch/aarch64/call.rs b/cc/arch/aarch64/call.rs index db7aacb5e..351ed2be9 100644 --- a/cc/arch/aarch64/call.rs +++ b/cc/arch/aarch64/call.rs @@ -750,14 +750,15 @@ impl Aarch64CodeGen { if let StackKind::Composite { bytes } = stack_arg.kind { // The pseudo locates the aggregate; its bytes go into the // slot. + // AAPCS64 B.4 replaces a composite above sixteen bytes with a + // pointer to the caller's copy, so the run here is bounded by + // the ABI at two eightbytes; the widths are `block_chunks`'s, + // so a composite that is not a multiple of eight reads and + // writes its tail as wide as the tail is. let src = self.aggregate_arg_address(stack_arg.pseudo); - let mut done = 0; - while done < bytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= bytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let done = done as i32; self.push_lir(Aarch64Inst::Ldr { size, addr: MemAddr::BaseOffset { @@ -774,7 +775,6 @@ impl Aarch64CodeGen { offset: offset + done, }, }); - done += chunk; } continue; } diff --git a/cc/arch/aarch64/codegen.rs b/cc/arch/aarch64/codegen.rs index a7d66c042..bc1d9b885 100644 --- a/cc/arch/aarch64/codegen.rs +++ b/cc/arch/aarch64/codegen.rs @@ -78,7 +78,7 @@ pub struct Aarch64CodeGen { /// Stack allocation size for locals (for zero_stack_frame) pub(super) stack_alloc_size: i32, /// Sym pseudo ID → type size in bits (for distinguishing scalar vs struct stores) - pub(super) sym_type_sizes: HashMap, + pub(super) sym_slots: HashMap, /// Which register this function's locals are addressed through pub(super) frame_base: FrameBase, } @@ -103,7 +103,7 @@ impl Aarch64CodeGen { pic_mode: false, unique_label_counter: 0, stack_alloc_size: 0, - sym_type_sizes: HashMap::new(), + sym_slots: HashMap::new(), frame_base: FrameBase::Fp, } } diff --git a/cc/arch/aarch64/features.rs b/cc/arch/aarch64/features.rs index efbf1d628..b6df879eb 100644 --- a/cc/arch/aarch64/features.rs +++ b/cc/arch/aarch64/features.rs @@ -11,6 +11,7 @@ use super::call::HfaElem; use super::codegen::Aarch64CodeGen; +use super::frame::UNROLL_LIMIT_BYTES; use super::lir::{Aarch64Inst, GpOperand, MemAddr}; use super::regalloc::{Loc, Reg, VReg}; use crate::arch::codegen::BswapSize; @@ -636,8 +637,15 @@ impl Aarch64CodeGen { } /// Copy `bytes` of an aggregate from `[addr + off]` into the destination, - /// in descending power-of-two chunks so nothing past the object is - /// written. X16 is the shuttle -- linker scratch, never allocated. + /// in the descending power-of-two chunks `block_chunks` gives, so nothing + /// past the object is written. X16 is the shuttle -- linker scratch, never + /// allocated. + /// + /// Past [`UNROLL_LIMIT_BYTES`] the copy becomes a counted loop, which + /// **advances `addr`**: only the two `VaAggKind::Indirect` paths can reach + /// that size, and both pass a register they are finished with. Every other + /// kind is at most sixteen bytes (a composite in general registers) or + /// sixty-four (an HFA of four binary128s), so it never gets there. fn emit_va_arg_bytes( &mut self, dst_loc: &Loc, @@ -663,13 +671,44 @@ impl Aarch64CodeGen { return; } } - let mut done = 0; - while done < bytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= bytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + if i64::from(bytes) > UNROLL_LIMIT_BYTES { + // X16 becomes the destination cursor; the source cursor is `addr` + // itself, advanced past the object. + match dst_loc { + Loc::Stack(_) | Loc::IncomingArg(_) => { + let (base, disp) = self + .loc_addr_parts(dst_loc) + .expect("a frame location has a base and a displacement"); + self.push_lir(Aarch64Inst::Add { + size: OperandSize::B64, + src1: base, + src2: GpOperand::Imm(disp.into()), + dst: Reg::X16, + }); + } + // The register holds the aggregate's address, and it is the + // result, so the cursor is a copy of it. + Loc::Reg(r) if !holds_value => self.push_lir(Aarch64Inst::Mov { + size: OperandSize::B64, + src: GpOperand::Reg(*r), + dst: Reg::X16, + }), + _ => return, + } + if off != 0 { + self.push_lir(Aarch64Inst::Add { + size: OperandSize::B64, + src1: addr, + src2: GpOperand::Imm(off.into()), + dst: addr, + }); + } + self.emit_block_copy_loop(Reg::X16, addr, bytes.into()); + return; + } + for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let done = done as i32; self.push_lir(Aarch64Inst::Ldr { size, addr: MemAddr::BaseOffset { @@ -700,7 +739,6 @@ impl Aarch64CodeGen { }), _ => return, } - done += chunk; } } diff --git a/cc/arch/aarch64/frame.rs b/cc/arch/aarch64/frame.rs index 751d5ca32..d74ee812c 100644 --- a/cc/arch/aarch64/frame.rs +++ b/cc/arch/aarch64/frame.rs @@ -25,6 +25,12 @@ use crate::ir::{Function, Instruction, PseudoId, PseudoKind}; use crate::types::{TypeId, TypeKind, TypeTable}; use std::collections::HashSet; +/// The most bytes this back end moves with unrolled loads and stores; past it, +/// [`Aarch64CodeGen::emit_block_copy_loop`]. It is the bound the IR puts on an +/// expanded `memcpy`, for the same reason: the instruction count of an unrolled +/// copy is linear in the object. +pub(super) const UNROLL_LIMIT_BYTES: i64 = crate::ir::memexpand::INLINE_LIMIT_BYTES; + impl Aarch64CodeGen { pub(super) fn emit_function(&mut self, func: &Function, types: &TypeTable) { self.base.func_pos = crate::arch::func_pos(func); @@ -45,16 +51,7 @@ impl Aarch64CodeGen { self.locations = alloc.allocate(func, types); self.pseudos = crate::arch::codegen::PseudoTable::new(&func.pseudos); - // Build sym type size map for emit_store to distinguish struct fields from scalars - self.sym_type_sizes.clear(); - for pseudo in &func.pseudos { - // By identity: a global whose name collides with a parameter's - // would otherwise be recorded with the parameter's type size. - if let Some(local_var) = func.local_of(pseudo.id) { - self.sym_type_sizes - .insert(pseudo.id, types.size_bits(local_var.typ)); - } - } + self.sym_slots = crate::arch::codegen::sym_slots(func, types); let stack_size = alloc.stack_size(); self.frame_base = alloc.frame_base(); @@ -538,6 +535,81 @@ impl Aarch64CodeGen { } } + /// Copy `bytes` bytes from `[src]` to `[dst]`, in a counted loop over the + /// whole sixteen-byte pairs and `block_chunks` for what is left. + /// + /// This is the back end's bulk block move. It cannot synthesize a call to + /// `memcpy` -- it is past the point where a call can be built -- and one + /// load/store pair per chunk is linear in the object: 4 KB of `va_arg` + /// aggregate cost about 1100 instructions, and 256 KB would be the + /// compile-time explosion the IR's own + /// [`crate::ir::memexpand::INLINE_LIMIT_BYTES`] exists to prevent. A + /// counted loop is the answer [`Self::emit_zero_loop`] gives the frame. + /// + /// Sixteen bytes an iteration through V16, which is reserved codegen + /// scratch: it keeps the loop to three general registers, which is all the + /// `va_arg` sequence has free. Both `src` and `dst` are cursors and are + /// left past the object, so the caller must be finished with them; X17 + /// holds the end of the paired part and then shuttles the tail. + pub(super) fn emit_block_copy_loop(&mut self, dst: Reg, src: Reg, bytes: i64) { + let pairs = bytes & !15; + if pairs > 0 { + self.push_lir(Aarch64Inst::Add { + size: OperandSize::B64, + src1: src, + src2: GpOperand::Imm(pairs), + dst: Reg::X17, + }); + let top = self.next_unique_label("block_copy"); + self.push_lir(Aarch64Inst::Directive(Directive::BlockLabel(top.clone()))); + self.push_lir(Aarch64Inst::LdrFp { + size: FpSize::Quad, + addr: MemAddr::PostIndex { + base: src, + offset: 16, + }, + dst: VReg::V16, + }); + self.push_lir(Aarch64Inst::StrFp { + size: FpSize::Quad, + src: VReg::V16, + addr: MemAddr::PostIndex { + base: dst, + offset: 16, + }, + }); + self.push_lir(Aarch64Inst::Cmp { + size: OperandSize::B64, + src1: src, + src2: GpOperand::Reg(Reg::X17), + }); + self.push_lir(Aarch64Inst::BCond { + cond: CondCode::Ult, + target: top, + }); + } + // The cursors point at the tail, so its pieces are offsets from them. + for (off, chunk) in crate::ir::memexpand::block_chunks(bytes - pairs) { + let size = OperandSize::from_bits(chunk.bits()); + self.push_lir(Aarch64Inst::Ldr { + size, + addr: MemAddr::BaseOffset { + base: src, + offset: off as i32, + }, + dst: Reg::X17, + }); + self.push_lir(Aarch64Inst::Str { + size, + src: Reg::X17, + addr: MemAddr::BaseOffset { + base: dst, + offset: off as i32, + }, + }); + } + } + /// Save callee-saved GP registers in pairs (or single if odd count) fn save_callee_saved_gp_regs(&mut self, total_frame: i32, callee_saved: &[Reg]) { let mut offset = 16; // Start after fp/lr @@ -960,13 +1032,14 @@ impl Aarch64CodeGen { crate::arch::func_pos(func), "a stacked parameter", ); - let mut done = 0; - while done < bytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= bytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + // A composite that reaches here is at most two + // eightbytes, so the run is bounded by the ABI; + // the widths are `block_chunks`'s so the tail of + // one that is not a multiple of eight is moved as + // wide as it is and no wider. + for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let done = done as i32; self.push_lir(Aarch64Inst::Ldr { size, addr: self.incoming_mem_plus(incoming, done), @@ -977,7 +1050,6 @@ impl Aarch64CodeGen { src: Reg::X16, addr: self.stack_mem_plus(local_off, done), }); - done += chunk; } } } diff --git a/cc/arch/aarch64/memory.rs b/cc/arch/aarch64/memory.rs index be0ce65ae..89bdeef67 100644 --- a/cc/arch/aarch64/memory.rs +++ b/cc/arch/aarch64/memory.rs @@ -601,26 +601,16 @@ impl Aarch64CodeGen { return; } - // Widen 32-bit stores at offset 0 to 64-bit to prevent stale - // upper bits when a 32-bit result is stored into a 64-bit - // local (e.g., int-to-long, int-to-pointer assignments). - // Only widen for known local variables (in sym_type_sizes). - // Do NOT widen stores to globals/statics (not in sym_type_sizes) - // or stores through pointers — widening could clobber adjacent data. - // Exception: struct/union fields at offset 0 must use exact - // size to avoid clobbering the adjacent field at offset 4. + // Widen a 32-bit store at offset 0 to 64 bits, so a narrow value going + // into a wider slot leaves no stale upper bits behind it (an + // int-to-long or int-to-pointer assignment). Only where the slot holds + // one scalar: see `SymSlot`. Only for a known local, too -- a global or + // a store through a pointer keeps its exact width, since nothing here + // knows what adjoins it. let store_size = if mem_size == 32 && insn.offset == 0 { - if let Some(&sym_bits) = self.sym_type_sizes.get(&addr) { - // Known local variable — safe to widen if scalar and > 32 bits - if sym_bits > 64 { - OperandSize::from_bits(mem_size) // struct field: exact size - } else if sym_bits > 32 { - OperandSize::B64 // scalar/pointer local: safe to widen - } else { - OperandSize::from_bits(mem_size) - } - } else { - OperandSize::from_bits(mem_size) // global/static/pointer: exact size + match self.sym_slots.get(&addr) { + Some(slot) if slot.widenable() && slot.bits > 32 => OperandSize::B64, + _ => OperandSize::from_bits(mem_size), } } else { OperandSize::from_bits(mem_size) diff --git a/cc/arch/codegen.rs b/cc/arch/codegen.rs index ea301f93e..3fb5786bc 100644 --- a/cc/arch/codegen.rs +++ b/cc/arch/codegen.rs @@ -1222,6 +1222,56 @@ pub fn check_tls_reached_only_by_address( } } +/// What a store lowering needs to know about a local's stack slot. +/// +/// Both back ends widen a narrow store at offset 0 of a local so that a value +/// going into a wider slot leaves no stale upper bits behind it. That is only +/// sound where the slot holds a single scalar: an aggregate or a complex has +/// another member at offset 4 or 8, which the widened store would write over. +/// +/// The size alone cannot answer it -- a `long` and a `struct { int x, y; }` +/// are both sixty-four bits -- and asking for *more* than sixty-four spares +/// only the aggregates too large to be mistaken for a scalar in the first +/// place. Both back ends had that test and both got an eight-byte aggregate +/// wrong, so the question is asked once, here. +pub struct SymSlot { + /// Width of the declared type, in bits. + pub bits: u32, + /// The slot holds a single scalar value, so anything above a narrow store + /// at offset 0 is stale bits of that same object. + pub one_scalar: bool, +} + +impl SymSlot { + /// Whether a 32-bit store at offset 0 of this slot may be widened to 64. + pub fn widenable(&self) -> bool { + self.one_scalar && self.bits <= 64 + } +} + +/// Record, for each of `func`'s locals, what its stack slot holds. +pub fn sym_slots( + func: &crate::ir::Function, + types: &crate::types::TypeTable, +) -> std::collections::HashMap { + let mut slots = std::collections::HashMap::new(); + for pseudo in &func.pseudos { + // By identity: a global whose name collides with a parameter's would + // otherwise be recorded with the parameter's type. + if let Some(local_var) = func.local_of(pseudo.id) { + let typ = local_var.typ; + slots.insert( + pseudo.id, + SymSlot { + bits: types.size_bits(typ), + one_scalar: types.is_scalar(typ) && !types.is_complex(typ), + }, + ); + } + } + slots +} + /// The current function's pseudos, looked up by id. /// /// A pseudo's id is not its position in `Function::pseudos`, so a lookup diff --git a/cc/arch/lir.rs b/cc/arch/lir.rs index 3993e1172..c2f9c88d5 100644 --- a/cc/arch/lir.rs +++ b/cc/arch/lir.rs @@ -70,6 +70,22 @@ impl fmt::Display for OperandSize { } } +/// How many of a composite's `total` bytes the eightbyte starting at `at` +/// carries. +/// +/// An eightbyte of a composite parameter travels in a whole register, but the +/// last eightbyte of a composite whose size is not a multiple of eight holds +/// fewer bytes than the register does -- four of `struct { int a, b, c; }`, +/// five of a thirteen-byte one. Storing the register's eight regardless is +/// what wrote past the object's local, whose slot is rounded up only to the +/// type's own alignment. +/// +/// Zero for an eightbyte past the end, which a class vector longer than the +/// object cannot produce but a caller need not prove. +pub fn eightbyte_bytes(total: i64, at: i64) -> i64 { + (total - at).clamp(0, 8) +} + // Floating-Point Size /// Floating-point size specifier @@ -131,6 +147,27 @@ impl FpSize { } } + /// The one floating-point store that writes exactly `bytes` bytes, if + /// there is one. + /// + /// The widths a store has are 2, 4, 8 and 16; a composite eightbyte of + /// SSE class can be any of them, and -- once a member is packed -- 3, 5, 6 + /// or 7 as well, for which the answer is `None` and the caller has to move + /// the bytes some other way. One byte is `None` too: there is no SSE store + /// of a single byte. + /// + /// Rounding up instead is what wrote four bytes past the twelve-byte + /// `struct { float x, y, z; }` whose second eightbyte holds only its `z`. + pub fn exact_sse_store(bytes: i64) -> Option { + match bytes { + 2 => Some(FpSize::Half), + 4 => Some(FpSize::Single), + 8 => Some(FpSize::Double), + 16 => Some(FpSize::Quad), + _ => None, + } + } + /// Create from TypeKind. This is the preferred way to determine FP size /// when type information is available, rather than inferring from bit size. /// @@ -1486,6 +1523,55 @@ mod tests { use super::*; use crate::target::Arch; + /// The last register of a composite parameter carries what is left of the + /// object, not a whole eightbyte. + /// + /// The prologue stores each eightbyte from the register it arrived in, and + /// a slot is rounded up only to the type's own alignment -- four for + /// `struct { int a, b, c; }` -- so a store as wide as the register writes + /// four bytes past a twelve-byte local. + #[test] + fn test_eightbyte_bytes_stops_at_the_end_of_the_object() { + // A multiple of eight fills every register it takes. + assert_eq!(eightbyte_bytes(16, 0), 8); + assert_eq!(eightbyte_bytes(16, 8), 8); + // Twelve and thirteen do not: the second register holds four and five. + assert_eq!(eightbyte_bytes(12, 0), 8); + assert_eq!(eightbyte_bytes(12, 8), 4); + assert_eq!(eightbyte_bytes(13, 8), 5); + // Under a register's worth the one register holds the whole object. + assert_eq!(eightbyte_bytes(3, 0), 3); + // Past the end is nothing at all, never a negative width. + assert_eq!(eightbyte_bytes(12, 16), 0); + assert_eq!(eightbyte_bytes(0, 0), 0); + + // The eightbytes of any size account for it exactly. + for total in 1..=64i64 { + let sum: i64 = (0..8).map(|i| eightbyte_bytes(total, i * 8)).sum(); + assert_eq!(sum, total, "the eightbytes of {total} must cover it"); + } + } + + /// An SSE eightbyte is stored at its own width, and a width no + /// floating-point store has is refused rather than rounded up to one. + #[test] + fn test_exact_sse_store_refuses_a_width_it_cannot_write() { + assert_eq!(FpSize::exact_sse_store(2), Some(FpSize::Half)); + assert_eq!(FpSize::exact_sse_store(4), Some(FpSize::Single)); + assert_eq!(FpSize::exact_sse_store(8), Some(FpSize::Double)); + assert_eq!(FpSize::exact_sse_store(16), Some(FpSize::Quad)); + // `for_sse_aggregate` answers every one of these with a *wider* store, + // which is what wrote past the object; here they have no answer and + // the caller moves the bytes through a general register instead. + for ragged in [0, 1, 3, 5, 6, 7, 9, 12, 15, 17] { + assert_eq!( + FpSize::exact_sse_store(ragged), + None, + "{ragged} bytes is not one SSE store" + ); + } + } + /// A symbol whose name an assembler will not take bare has to be quoted. /// Mach-O's assembler rejects a raw non-ASCII byte outright -- `_café:` is /// "invalid operand" -- and C17 6.4.2.1 admits extended characters in diff --git a/cc/arch/x86_64/call.rs b/cc/arch/x86_64/call.rs index 9e4a3564c..66bd6bbe8 100644 --- a/cc/arch/x86_64/call.rs +++ b/cc/arch/x86_64/call.rs @@ -54,47 +54,65 @@ pub(super) struct CallArgInfo { pub ignored_arg_indices: Vec, } -/// The largest stacked aggregate argument copied with unrolled moves; past -/// it, `rep movsq`. The IR stops unrolling a block copy at 128 bytes -/// (`memexpand::INLINE_LIMIT_BYTES`) for the same reason: one load/store pair per -/// eightbyte made a 600 MB argument 75 million instructions and tens of -/// gigabytes of compiler memory. -const STACK_ARG_UNROLL_QWORDS: usize = 16; +/// The largest block this back end copies with unrolled moves; past it, +/// `rep movsq`. It is the bound the IR puts on an expanded `memcpy`, for the +/// same reason: one load/store pair per eightbyte made a 600 MB argument 75 +/// million instructions and tens of gigabytes of compiler memory. +pub(super) const UNROLL_LIMIT_BYTES: i64 = crate::ir::memexpand::INLINE_LIMIT_BYTES; + +/// Where [`X86_64CodeGen::emit_rep_movsq`] writes. +/// +/// The helper pushes three registers, which moves `%rsp`, so a destination in +/// the outgoing argument area has to be named rather than handed over as an +/// address the caller already computed. +pub(super) enum BlockDst { + /// A byte offset in the outgoing argument area, addressed through `%rsp`; + /// the helper adds what its own pushes moved `%rsp` by. + OutgoingArg(i32), + /// Any address `%rsp` moving does not disturb -- a frame slot, or a + /// register holding a pointer. It must not be built on `%rsp`, `%rdi`, + /// `%rsi` or `%rcx`. + At(MemAddr), +} impl X86_64CodeGen { - /// Copy `qwords` eightbytes from `[src]` to the outgoing argument area at - /// `dst_off(%rsp)`, with `rep movsq`. + /// Copy `qwords` eightbytes from `src` to `dst`, with `rep movsq`. + /// + /// This is the back end's bulk block move: it cannot synthesize a call to + /// `memcpy`, and one load/store pair per eightbyte is linear in the object, + /// which is what made a 600 MB argument 75 million instructions. /// - /// The register arguments are set up *after* the stacked ones, so RDI, - /// RSI and RCX -- which `rep movsq` needs -- may still hold values that - /// are about to become arguments. They are saved around the copy with - /// `push`/`pop`, which touch no flags; the destination is addressed past - /// the three pushes. `src` is read into RSI before RDI or RCX is written, - /// so it may be any of the three. - fn emit_stack_arg_block_copy(&mut self, src: Reg, dst_off: i32, qwords: usize) { + /// RDI, RSI and RCX -- which `rep movsq` needs -- may hold live values: + /// the register arguments of a call are set up *after* its stacked ones, + /// and a `va_arg` sits in the middle of a function. They are saved around + /// the copy with `push`/`pop`, which touch no flags. `src` is read into RSI + /// before RDI or RCX is written, so it may be addressed through any of the + /// three; the destination may not. + pub(super) fn emit_rep_movsq(&mut self, src: MemAddr, dst: BlockDst, qwords: i64) { let saved = [Reg::Rdi, Reg::Rsi, Reg::Rcx]; for r in saved { self.push_lir(X86Inst::Push { src: GpOperand::Reg(r), }); } - if src != Reg::Rsi { - self.push_lir(X86Inst::Mov { - size: OperandSize::B64, - src: GpOperand::Reg(src), - dst: GpOperand::Reg(Reg::Rsi), - }); - } self.push_lir(X86Inst::Lea { - addr: MemAddr::BaseOffset { + addr: src, + dst: Reg::Rsi, + }); + let dst_addr = match dst { + BlockDst::OutgoingArg(off) => MemAddr::BaseOffset { base: Reg::Rsp, - offset: dst_off + 8 * saved.len() as i32, + offset: off + 8 * saved.len() as i32, }, + BlockDst::At(addr) => addr, + }; + self.push_lir(X86Inst::Lea { + addr: dst_addr, dst: Reg::Rdi, }); self.push_lir(X86Inst::Mov { size: OperandSize::B64, - src: GpOperand::Imm(qwords as i64), + src: GpOperand::Imm(qwords), dst: GpOperand::Reg(Reg::Rcx), }); self.push_lir(X86Inst::RepMovsq); @@ -103,6 +121,54 @@ impl X86_64CodeGen { } } + /// Copy the `bytes` bytes of the object at `[src]` into the outgoing + /// argument area at `dst_off(%rsp)`. + /// + /// Both sides move `block_chunks`'s widths. The argument area really is + /// allocated in whole eightbytes, so an eight-byte *write* at every offset + /// the object covers would be within it -- but the source is the object, + /// and reading eight bytes of a twelve-byte struct at offset eight reads + /// four that belong to whatever follows it, which faults when the object + /// ends a page. It is the same hazard [`Self::load_object_bytes`] carries, + /// and the same answer; the bytes of the slot the object does not reach are + /// left alone, which the psABI allows because nothing in the callee reads + /// them. + /// + /// Past [`UNROLL_LIMIT_BYTES`] the whole eightbytes go through `rep movsq` + /// and only the ragged tail is unrolled. + fn emit_stack_arg_copy(&mut self, src: Reg, dst_off: i32, bytes: i64) { + let mut at = 0; + if bytes > UNROLL_LIMIT_BYTES { + let qwords = bytes / 8; + let from = MemAddr::BaseOffset { + base: src, + offset: 0, + }; + self.emit_rep_movsq(from, BlockDst::OutgoingArg(dst_off), qwords); + at = qwords * 8; + } + for (off, chunk) in crate::ir::memexpand::block_chunks(bytes - at) { + let size = OperandSize::from_bits(chunk.bits()); + let byte = (at + off) as i32; + self.push_lir(X86Inst::Mov { + size, + src: GpOperand::Mem(MemAddr::BaseOffset { + base: src, + offset: byte, + }), + dst: GpOperand::Reg(Reg::Rax), + }); + self.push_lir(X86Inst::Mov { + size, + src: GpOperand::Reg(Reg::Rax), + dst: GpOperand::Mem(MemAddr::BaseOffset { + base: Reg::Rsp, + offset: dst_off + byte, + }), + }); + } + } + /// Classify call arguments into register vs stack arguments using ABI info. pub(super) fn classify_call_args(&self, insn: &Instruction, types: &TypeTable) -> CallArgInfo { let int_arg_regs = Reg::arg_regs(); @@ -339,30 +405,8 @@ impl X86_64CodeGen { crate::abi::struct_param_classes(t, types).map(|_| types.size_bytes(t)) }) }) { - let num_qwords = bytes.div_ceil(8); let base = self.address_of_pseudo(arg); - if num_qwords > STACK_ARG_UNROLL_QWORDS { - self.emit_stack_arg_block_copy(base, base_off, num_qwords); - continue; - } - for q in 0..num_qwords { - self.push_lir(X86Inst::Mov { - size: OperandSize::B64, - src: GpOperand::Mem(MemAddr::BaseOffset { - base, - offset: (q * 8) as i32, - }), - dst: GpOperand::Reg(Reg::Rax), - }); - self.push_lir(X86Inst::Mov { - size: OperandSize::B64, - src: GpOperand::Reg(Reg::Rax), - dst: GpOperand::Mem(MemAddr::BaseOffset { - base: Reg::Rsp, - offset: base_off + (q * 8) as i32, - }), - }); - } + self.emit_stack_arg_copy(base, base_off, bytes as i64); continue; } diff --git a/cc/arch/x86_64/codegen.rs b/cc/arch/x86_64/codegen.rs index 65c1a3be5..1209df87a 100644 --- a/cc/arch/x86_64/codegen.rs +++ b/cc/arch/x86_64/codegen.rs @@ -77,8 +77,8 @@ pub struct X86_64CodeGen { /// binary128 constants to emit (pool key -> the 16-byte image). /// BTreeMap for reproducible order, as `ld_constants`. pub(super) quad_constants: std::collections::BTreeMap, - /// Sym pseudo ID → type size in bits (for distinguishing scalar vs struct stores) - pub(super) sym_type_sizes: HashMap, + /// Sym pseudo ID → what its stack slot holds, for [`SymSlot`]. + pub(super) sym_slots: HashMap, /// How this function's locals are addressed. pub(super) frame_base: FrameBase, /// Maximum local alignment (for andq in prologue) @@ -110,7 +110,7 @@ impl X86_64CodeGen { ld_constants: std::collections::BTreeMap::new(), double_constants: std::collections::BTreeMap::new(), quad_constants: std::collections::BTreeMap::new(), - sym_type_sizes: HashMap::new(), + sym_slots: HashMap::new(), frame_base: FrameBase::Rbp, max_local_align: 16, int128_pseudos: HashSet::new(), diff --git a/cc/arch/x86_64/features.rs b/cc/arch/x86_64/features.rs index ac8fcb7b7..985a47d23 100644 --- a/cc/arch/x86_64/features.rs +++ b/cc/arch/x86_64/features.rs @@ -9,6 +9,7 @@ // x86-64 Feature Code Generation (Variadic Functions, Byte Swapping, Bit Counting) // +use super::call::{BlockDst, UNROLL_LIMIT_BYTES}; use super::codegen::X86_64CodeGen; use super::lir::{popcount_sequence, GpOperand, MemAddr, ShiftCount, X86Inst}; use super::regalloc::{Loc, Reg}; @@ -530,12 +531,20 @@ impl X86_64CodeGen { }) } - /// Copy `nbytes` from `[src_base + src_off]` to `dst` at `dst_off`, in - /// descending power-of-two chunks so nothing past the object is written. + /// Copy `nbytes` from `[src_base + src_off]` to `dst` at `dst_off`, in the + /// descending power-of-two chunks `block_chunks` gives, so nothing past the + /// object is written. /// /// `%rcx` is the shuttle: it is declared clobbered by `VaArg`, so no live /// value is in it, and unlike `%r11` it cannot be `ap_base` (the va_list /// pointer lands there when it comes from a stack slot). + /// + /// Past [`UNROLL_LIMIT_BYTES`] the whole eightbytes go through `rep movsq` + /// and only the ragged tail is unrolled. Without a bound this was linear in + /// the aggregate -- 4 KB cost about 1100 instructions and 256 KB would be + /// the compile-time explosion the IR's own limit exists to prevent -- and + /// `va_arg` is the one place a whole aggregate is copied where the size is + /// the program's to choose. fn va_copy_bytes( &mut self, src_base: Reg, @@ -561,13 +570,32 @@ impl X86_64CodeGen { }); return; } - let mut done = 0; - while done < nbytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= nbytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + let mut at = 0; + if i64::from(nbytes) > UNROLL_LIMIT_BYTES { + let qwords = i64::from(nbytes) / 8; + // Neither base is `%rsp`, so the helper's pushes leave both where + // they are; a frame slot is not `%rsp`-relative either. + let to = match dst { + VaAggDst::Slot(slot) => BlockDst::At(self.stack_field(slot, dst_off)), + VaAggDst::Addr(base) => BlockDst::At(MemAddr::BaseOffset { + base, + offset: dst_off, + }), + VaAggDst::Value(_) => unreachable!("a register destination returned above"), + }; + self.emit_rep_movsq( + MemAddr::BaseOffset { + base: src_base, + offset: src_off, + }, + to, + qwords, + ); + at = (qwords * 8) as i32; + } + for (off, chunk) in crate::ir::memexpand::block_chunks(i64::from(nbytes - at)) { + let size = OperandSize::from_bits(chunk.bits()); + let done = at + off as i32; self.push_lir(X86Inst::Mov { size, src: GpOperand::Mem(MemAddr::BaseOffset { @@ -589,7 +617,6 @@ impl X86_64CodeGen { src: GpOperand::Reg(Reg::Rcx), dst: into, }); - done += chunk; } } diff --git a/cc/arch/x86_64/frame.rs b/cc/arch/x86_64/frame.rs index a74259505..a8a27c252 100644 --- a/cc/arch/x86_64/frame.rs +++ b/cc/arch/x86_64/frame.rs @@ -13,11 +13,11 @@ use crate::abi::{get_abi, Abi, ArgClass, RegClass}; use crate::arch::codegen::is_variadic_function; use crate::arch::lir::{ - complex_fp_info, complex_sse_regs, plan_pair_move, Directive, FpSize, OperandSize, PairMove, - Symbol, + complex_fp_info, complex_sse_regs, eightbyte_bytes, plan_pair_move, Directive, FpSize, + OperandSize, PairMove, Symbol, }; use crate::arch::x86_64::codegen::X86_64CodeGen; -use crate::arch::x86_64::lir::{GpOperand, MemAddr, X86Inst, XmmOperand}; +use crate::arch::x86_64::lir::{GpOperand, MemAddr, ShiftCount, X86Inst, XmmOperand}; use crate::arch::x86_64::regalloc::{spend_arg_regs, FrameBase, Loc, Reg, RegAlloc, XmmReg}; use crate::ir::{Function, Instruction, PseudoId, PseudoKind}; use crate::types::{TypeId, TypeKind, TypeTable}; @@ -79,30 +79,104 @@ impl X86_64CodeGen { return; }; + // The last eightbyte of a composite whose size is not a multiple of + // eight holds fewer bytes than the register carrying it. + let total = i64::from(type_size_bits / 8); let mut next_int = pair_start_int; let mut next_fp = pair_start_fp; for (i, class) in classes.iter().enumerate() { - let delta = (i * 8) as i32; + let at = (i * 8) as i64; + let bytes = eightbyte_bytes(total, at); + if bytes == 0 { + continue; + } if *class == crate::abi::RegClass::Sse { let src = fp_arg_regs[next_fp]; next_fp += 1; - self.push_lir(X86Inst::MovFp { - size: FpSize::Double, - src: XmmOperand::Reg(src), - dst: XmmOperand::Mem(self.stack_mem(offset - delta)), - }); + self.store_sse_bytes_to_local(src, offset, at, bytes); } else { let src = int_arg_regs[next_int]; next_int += 1; - self.push_lir(X86Inst::Mov { + self.store_reg_bytes_to_local(src, offset, at, bytes); + } + } + } + + /// Store the low `bytes` bytes of `src` into the local at `local`, + /// starting at the object's byte `at`, writing nothing past them. + /// + /// An eightbyte of a composite parameter arrives in a whole register, but + /// the last eightbyte of one that is not a multiple of eight holds fewer + /// bytes than the register does: `struct { int a, b, c; }` is twelve, and + /// storing the second register's eight wrote four bytes past a local that + /// `grow_frame` rounds up only to the type's own alignment -- four here. + /// + /// The pieces are `block_chunks`'s, so the rule is stated once, and the + /// value is shifted down as each leaves. The shifting is done in R10 -- + /// this file's scratch -- so the incoming argument register survives; a + /// natural width needs no shift and stores straight out of it. + fn store_reg_bytes_to_local(&mut self, src: Reg, local: i32, at: i64, bytes: i64) { + let mut shifted = 0; + for (off, chunk) in crate::ir::memexpand::block_chunks(bytes) { + if off != 0 { + if shifted == 0 && src != Reg::R10 { + self.push_lir(X86Inst::Mov { + size: OperandSize::B64, + src: GpOperand::Reg(src), + dst: GpOperand::Reg(Reg::R10), + }); + } + self.push_lir(X86Inst::Shr { size: OperandSize::B64, - src: GpOperand::Reg(src), - dst: GpOperand::Mem(self.stack_mem(offset - delta)), + count: ShiftCount::Imm(((off - shifted) * 8) as u8), + dst: Reg::R10, }); + shifted = off; } + let from = if off == 0 { src } else { Reg::R10 }; + let addr = self.stack_field(local, (at + off) as i32); + self.push_lir(X86Inst::Mov { + size: OperandSize::from_bits(chunk.bits()), + src: GpOperand::Reg(from), + dst: GpOperand::Mem(addr), + }); + } + } + + /// Store the low `bytes` bytes of the SSE register `src` into the local at + /// `local`, starting at the object's byte `at`. + /// + /// 2, 4, 8 and 16 bytes are one floating-point store. Any other width is + /// moved into a general register and stored from there, because there is + /// no SSE store of it: the second eightbyte of + /// `struct { float a, b, c; _Float16 d; }` is six bytes, and all of SSE + /// class. + fn store_sse_bytes_to_local(&mut self, src: XmmReg, local: i32, at: i64, bytes: i64) { + if let Some(size) = FpSize::exact_sse_store(bytes) { + let addr = self.stack_field(local, at as i32); + self.push_lir(X86Inst::MovFp { + size, + src: XmmOperand::Reg(src), + dst: XmmOperand::Mem(addr), + }); + return; } + self.push_lir(X86Inst::MovXmmGp { + size: OperandSize::B64, + src, + dst: Reg::R10, + }); + self.store_reg_bytes_to_local(Reg::R10, local, at, bytes); } + /// Copy a spilled parameter of `bytes` bytes out of the incoming argument + /// area into the local the body reads. + /// + /// Both sides move `block_chunks`'s widths. Stepping eight regardless -- + /// which this did -- wrote four bytes past a twelve-byte local, whose slot + /// is rounded up only to the type's own alignment; the incoming area is + /// eightbyte-granular so the *read* was safe, but the read is what the + /// store's width came from. fn copy_incoming_arg_to_local( &mut self, func: &crate::ir::Function, @@ -120,24 +194,25 @@ impl X86_64CodeGen { return; }; let dst_offset = *dst_offset; - let mut copied = 0; - while copied < bytes { + for (at, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let at = at as i32; self.push_lir(X86Inst::Mov { - size: OperandSize::B64, + size, src: GpOperand::Mem(MemAddr::BaseOffset { base: Reg::Rbp, - offset: src_offset + copied, + offset: src_offset + at, }), dst: GpOperand::Reg(Reg::R10), }); // Locals grow downward from `dst_offset`, so later bytes sit at a - // smaller offset — the same convention `stack_mem` encodes. + // smaller offset — the same convention `stack_field` encodes. + let addr = self.stack_field(dst_offset, at); self.push_lir(X86Inst::Mov { - size: OperandSize::B64, + size, src: GpOperand::Reg(Reg::R10), - dst: GpOperand::Mem(self.stack_mem(dst_offset - copied)), + dst: GpOperand::Mem(addr), }); - copied += 8; } } @@ -166,16 +241,7 @@ impl X86_64CodeGen { self.int128_pseudos = alloc.int128_pseudos().clone(); self.pseudos = crate::arch::codegen::PseudoTable::new(&func.pseudos); - // Build sym type size map for emit_store to distinguish struct fields from scalars - self.sym_type_sizes.clear(); - for pseudo in &func.pseudos { - // By identity: a global whose name collides with a parameter's - // would otherwise be recorded with the parameter's type size. - if let Some(local_var) = func.local_of(pseudo.id) { - self.sym_type_sizes - .insert(pseudo.id, types.size_bits(local_var.typ)); - } - } + self.sym_slots = crate::arch::codegen::sym_slots(func, types); let stack_size = alloc.stack_size(); self.callee_saved_regs = alloc.callee_saved_used().to_vec(); @@ -793,50 +859,65 @@ impl X86_64CodeGen { if let Some(local) = func.locals.get(param_name) { if let Some(Loc::Stack(offset)) = self.locations.get_ref(local.sym) { let offset = *offset; - let (fp_size, imag_offset) = if let Some(n) = sse_struct { - // Two doubles are eight bytes each; - // a lone binary128 is one register - // holding all sixteen. - if n == 1 { - (FpSize::for_sse_aggregate(type_size_bits), 0) - } else { - (FpSize::Double, 8) + let total = i64::from(type_size_bits / 8); + if sse_struct.is_some() { + // An all-SSE aggregate: each register + // holds the eightbyte it was classified + // for, and the last one holds only what + // is left of the object. `sse_regs` is + // one for a lone binary128 -- SSE+SSEUP, + // sixteen bytes in one register -- and + // two for a pair of eightbytes, whose + // second is four bytes of a twelve-byte + // struct and not eight. Giving both the + // same width wrote four bytes past it. + for reg in 0..sse_regs { + let at = (reg * 8) as i64; + let bytes = if sse_regs == 1 { + total.min(16) + } else { + eightbyte_bytes(total, at) + }; + if bytes == 0 { + continue; + } + self.store_sse_bytes_to_local( + fp_arg_regs[fp_arg_idx + reg], + offset, + at, + bytes, + ); } } else { - complex_fp_info(types, &self.base.target, *typ) - }; - if sse_regs == 1 { - // One register holding the whole - // value. For `float _Complex` that - // is one eightbyte with both - // halves in it, so a 64-bit store - // writes all of it; for an - // aggregate it is whatever the - // class's size says, which is - // sixteen bytes for a binary128. - let whole = if sse_struct.is_some() { - fp_size + // A complex value: two elements of the + // same width at the base type's stride. + // `float _Complex` is one eightbyte with + // both halves in it, so a 64-bit store + // writes all of it. + let (fp_size, imag_offset) = + complex_fp_info(types, &self.base.target, *typ); + if sse_regs == 1 { + self.push_lir(X86Inst::MovFp { + size: FpSize::Double, + src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), + dst: XmmOperand::Mem(self.stack_mem(offset)), + }); } else { - FpSize::Double - }; - self.push_lir(X86Inst::MovFp { - size: whole, - src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), - dst: XmmOperand::Mem(self.stack_mem(offset)), - }); - } else { - // Store real part from first XMM register - self.push_lir(X86Inst::MovFp { - size: fp_size, - src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), - dst: XmmOperand::Mem(self.stack_mem(offset)), - }); - // Store imag part from second XMM register - self.push_lir(X86Inst::MovFp { - size: fp_size, - src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx + 1]), - dst: XmmOperand::Mem(self.stack_mem(offset - imag_offset)), - }); + // Store real part from first XMM register + self.push_lir(X86Inst::MovFp { + size: fp_size, + src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), + dst: XmmOperand::Mem(self.stack_mem(offset)), + }); + // Store imag part from second XMM register + self.push_lir(X86Inst::MovFp { + size: fp_size, + src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx + 1]), + dst: XmmOperand::Mem( + self.stack_mem(offset - imag_offset), + ), + }); + } } } } diff --git a/cc/arch/x86_64/memory.rs b/cc/arch/x86_64/memory.rs index cd588e963..26b7a61ab 100644 --- a/cc/arch/x86_64/memory.rs +++ b/cc/arch/x86_64/memory.rs @@ -927,24 +927,26 @@ impl X86_64CodeGen { let op_size = OperandSize::from_bits(mem_size); if is_symbol { // Local variable - store directly to stack slot. - // Widen 32-bit stores at offset 0 to 64-bit to prevent stale - // upper bits when a 32-bit result is stored into a 64-bit - // local (e.g., int-to-long, int-to-pointer assignments). - // Exception: struct/union fields at offset 0 must use exact - // size to avoid clobbering the adjacent field at offset 4. + // Widen a 32-bit store at offset 0 to 64 bits, so a narrow + // value going into a wider slot leaves no stale upper bits + // behind it (an int-to-long or int-to-pointer assignment). + // + // Only where the slot holds one scalar, though: an + // aggregate or a complex has another member at offset 4 or + // 8, and widening the store writes over it. Asking how + // *large* the object is cannot tell the two apart -- a + // `long` and a `struct { int x, y; }` are both 64 bits, and + // testing for more than 64 spared only the aggregates too + // big to be confused with a scalar in the first place. A + // slot this has no record of keeps the widening, which is + // what it did before. let store_size = if mem_size == 32 && insn.offset == 0 { - let sym_bits = self.sym_type_sizes.get(&addr).copied().unwrap_or(64); - if sym_bits > 32 { - // Check if this is a struct/union (don't widen field stores) - let is_struct = - self.sym_type_sizes.contains_key(&addr) && sym_bits > 64; - if is_struct { - op_size // struct field: exact size - } else { - OperandSize::B64 // scalar/pointer: safe to widen - } - } else { - OperandSize::B64 // small scalar: safe to widen + match self.sym_slots.get(&addr) { + Some(slot) if slot.widenable() => OperandSize::B64, + Some(_) => op_size, + // A slot with no record keeps the widening here, + // which is what this back end did before. + None => OperandSize::B64, } } else { op_size diff --git a/cc/ir/README.md b/cc/ir/README.md index ece99e02a..fbb1e33e7 100644 --- a/cc/ir/README.md +++ b/cc/ir/README.md @@ -111,9 +111,25 @@ SSA-form intermediate representation for the c17 C17 compiler. Inspired by Linus | `symaddr` | Get address of symbol | For `load` and `store`, `offset` is a **byte** displacement and `size` is the -access width in **bits**. Neither carries a volatile or atomic marker: -volatility lives on the `LocalVar` or the `GlobalDef`, and an atomic access -has its own opcode. +access width in **bits**. An atomic access has its own opcode. + +Both carry a **volatile marker**, `Instruction::is_volatile`, printed as a +trailing `volatile` in a dump. Ask it through `Instruction::is_volatile_access`. +The qualifier has to live on the *access* because it is not always on any +object: for `volatile int *p`, `p` is an ordinary pointer and `*p` is the +volatile object, so `LocalVar::is_volatile` and +`memloc::GlobalFacts::is_volatile` — which answer only for a named object — +have nothing to say about it. Those two remain, and are still what a pass asks +about the object as a whole; the marker is what it asks about the access. + +`Linearizer::emit` sets the marker for every access the linearizer emits, from +`types.contains_volatile` of the type that access reaches, and +`ir::build::Builder` does the same for the accesses a pass synthesizes. Reading +a volatile object is observable behaviour (C17 5.1.2.3p6), so a marked access +survives every optimization level: `dce::is_root` treats it as a root, +`loadfwd` will not forward one or fold two into one, `dse` will not delete one, +`constglobal` will not answer one from an initializer, and `ssa` will not +promote the object it reaches out of memory. ### SSA Operations diff --git a/cc/ir/build.rs b/cc/ir/build.rs index af610757a..06092046c 100644 --- a/cc/ir/build.rs +++ b/cc/ir/build.rs @@ -51,15 +51,22 @@ impl<'a> Builder<'a> { } /// A load of `typ`, `size` bits wide, from `addr + at`. + /// + /// Marked volatile when `typ` is, for the same reason and by the same rule + /// as `Linearizer::mark_volatile_access`: an access this builds is as + /// observable as one the program wrote, and a pass must not be able to + /// introduce an unmarked access to a volatile object. pub(crate) fn load(&mut self, addr: PseudoId, at: i64, typ: TypeId, size: u32) -> PseudoId { let v = self.func.alloc_pseudo(); - self.push(Instruction::load(v, addr, at, typ, size)); + let vol = self.types.contains_volatile(typ); + self.push(Instruction::load(v, addr, at, typ, size).with_volatile(vol)); v } /// A store of `v`, of `typ` and `size` bits wide, to `addr + at`. pub(crate) fn store(&mut self, v: PseudoId, addr: PseudoId, at: i64, typ: TypeId, size: u32) { - self.push(Instruction::store(v, addr, at, typ, size)); + let vol = self.types.contains_volatile(typ); + self.push(Instruction::store(v, addr, at, typ, size).with_volatile(vol)); } /// A new integer constant of `typ` at `size` bits, with the `SetVal` diff --git a/cc/ir/constfold.rs b/cc/ir/constfold.rs index a83d0ad73..a466a00db 100644 --- a/cc/ir/constfold.rs +++ b/cc/ir/constfold.rs @@ -339,20 +339,91 @@ pub(crate) fn eval_binop(insn: &Instruction, a: i128, b: i128) -> Option { } } +/// The most negative value at `size` bits, read as signed: the one dividend +/// whose quotient by -1 is not representable. +/// +/// `size == 0` and `size >= 128` both answer `i128::MIN`, matching +/// [`at_width`], which leaves a value alone at those widths. +fn signed_min_at(size: u32) -> i128 { + if size == 0 || size >= 128 { + i128::MIN + } else { + -1i128 << (size - 1) + } +} + +/// Whether `op` -- one of `DivS`/`DivU`/`ModS`/`ModU` -- computed at `size` +/// bits over these operands may raise a hardware trap. `None` is an operand +/// that is not known, which may be anything, and so may trap. +/// +/// This is the one place the rule lives, because the three passes that fold a +/// division each know a different amount about the operands and the rule drifted +/// apart between them. `eval_divmod` knows both values; `instcombine`'s +/// algebraic arms know one and nothing about the other; `range::udiv`/`umod` +/// know sets rather than values, and answer this question of a set by asking +/// whether zero is in it (the signed trap has no unsigned counterpart). +/// +/// Two operand pairs trap, and they are the two `idiv` raises #DE for: +/// +/// * a zero divisor, in either signedness; and +/// * the single signed overflow, `INT_MIN / -1`, whose quotient is not +/// representable. The remainder form traps with it, because on x86-64 it is +/// the same instruction, computing the same quotient. +/// +/// c17 does not assume this undefined behaviour away. Folding either one turns +/// a program that faults at `-O0` into one that prints an answer at `-O2`, and +/// the two disagreeing about the same source is worse than either answer. +/// gcc and clang both fold these, treating the undefined behaviour as licence; +/// this is a deliberate divergence from both rather than a bug-for-bug match. +/// +/// The operands are raw: the narrowing this needs is applied here, and +/// narrowing is idempotent, so a caller that has already read them at their own +/// width may pass those instead. +pub(crate) fn divmod_may_trap(op: Opcode, size: u32, a: Option, b: Option) -> bool { + let signed = match op { + Opcode::DivS | Opcode::ModS => true, + Opcode::DivU | Opcode::ModU => false, + // Nothing else in this IR traps on its operands. + _ => return false, + }; + let size = size.max(1); + + // An unknown divisor may be zero. + let Some(b) = b.map(|v| at_width(v, size, signed)) else { + return true; + }; + if b == 0 { + return true; + } + if !signed || b != -1 { + return false; + } + // Divisor -1: the trap turns on whether the dividend is the most negative + // value, so an unknown dividend may trap. + match a.map(|v| at_width(v, size, true)) { + Some(a) => a == signed_min_at(size), + None => true, + } +} + /// Division and remainder. /// /// Read at the operand's own width, in the signedness the opcode implies. /// Division is not congruent modulo 2^n the way add/sub/mul are: it reads the /// whole value and its sign, so `(int)0xFFFFFFFFu` arriving as 4294967295 /// rather than -1 answered 2147483647 where C says 0. +/// +/// An operation that traps is not folded: see [`divmod_may_trap`]. That single +/// refusal covers `instcombine`, `sccp` and `vrp` at once, since all three +/// route their constant folding through [`eval_binop`]. fn eval_divmod(insn: &Instruction, a: i128, b: i128) -> Option { let signed = matches!(insn.op, Opcode::DivS | Opcode::ModS); let size = insn.size.max(1); - let a = at_width(a, size, signed); - let b = at_width(b, size, signed); - if b == 0 { + if divmod_may_trap(insn.op, size, Some(a), Some(b)) { return None; } + let a = at_width(a, size, signed); + let b = at_width(b, size, signed); let folded = match (insn.op, signed) { (Opcode::DivS, _) => a.wrapping_div(b), (Opcode::DivU, _) => (a as u128).wrapping_div(b as u128) as i128, @@ -376,7 +447,22 @@ fn eval_shift(insn: &Instruction, a: i128, b: i128) -> Option { // The result is truncated back, so an overflowing shift wraps at the // operand width rather than growing into the i128. Opcode::Shl => at_width(at_width(a, size, true).wrapping_shl(b as u32), size, true), - Opcode::Lsr => at_width(a, size, false).wrapping_shr(b as u32), + // Shifted in the unsigned view, which is what makes it the logical + // shift. Doing it on the `i128` instead only agrees below 128 bits, + // where `at_width` has already cleared the high half: at 128 bits + // `at_width` hands the value back unchanged and `i128::wrapping_shr` + // is the arithmetic shift, so a negative operand would shift in ones. + // Unreachable as the passes stand: `arch::mapping` runs before the + // optimizer, and it leaves no 128-bit shift whose count this can read + // -- a literally constant count expands the shift into 64-bit halves, + // and any other count is rewritten into a `Pair64` the constant map + // cannot answer. Written correctly anyway, so that the arm does not + // depend on that pass ordering for its answer. + Opcode::Lsr => at_width( + (at_width(a, size, false) as u128).wrapping_shr(b as u32) as i128, + size, + false, + ), Opcode::Asr => at_width(a, size, true).wrapping_shr(b as u32), _ => return None, }) @@ -589,4 +675,228 @@ mod tests { } assert_eq!(fcmp_against_constant(Opcode::SetGt, inf, false), None); } + + const DIVMOD: [Opcode; 4] = [Opcode::DivS, Opcode::DivU, Opcode::ModS, Opcode::ModU]; + + fn is_signed(op: Opcode) -> bool { + matches!(op, Opcode::DivS | Opcode::ModS) + } + + /// A zero divisor traps in either signedness, at every width, whatever the + /// dividend is -- including when the dividend is itself unknown. + #[test] + fn divmod_may_trap_refuses_every_zero_divisor() { + for op in DIVMOD { + for size in [8, 16, 32, 64, 128] { + for a in [None, Some(0), Some(1), Some(-1), Some(i128::MAX)] { + assert!( + divmod_may_trap(op, size, a, Some(0)), + "{op:?}.{size} {a:?} / 0" + ); + } + // The zero may also arrive unnarrowed: 2^size reads as zero at + // `size` bits, and it is the narrowed value that divides. + if size < 128 { + assert!( + divmod_may_trap(op, size, Some(1), Some(1i128 << size)), + "{op:?}.{size} 1 / 2^{size}" + ); + } + } + } + } + + /// An operand that is not known may be anything, so it may be the zero + /// divisor, or the `INT_MIN` dividend that overflows against -1. + #[test] + fn divmod_may_trap_refuses_an_unknown_operand() { + for op in DIVMOD { + // An unknown divisor, whatever the dividend. + assert!(divmod_may_trap(op, 32, Some(7), None), "{op:?} 7 / x"); + assert!(divmod_may_trap(op, 32, Some(0), None), "{op:?} 0 / x"); + assert!(divmod_may_trap(op, 32, None, None), "{op:?} y / x"); + + // An unknown dividend only matters against -1, and only when the + // opcode is a signed one -- for `DivU`/`ModU`, -1 at 32 bits is + // 4294967295, an ordinary divisor. + assert_eq!( + divmod_may_trap(op, 32, None, Some(-1)), + is_signed(op), + "{op:?} x / -1" + ); + assert!(!divmod_may_trap(op, 32, None, Some(3)), "{op:?} x / 3"); + } + } + + /// The one signed overflow, at every width it exists at, and only for the + /// signed opcodes. + #[test] + fn divmod_may_trap_refuses_the_signed_overflow() { + for size in [8, 16, 32, 64, 128] { + let min = signed_min_at(size); + for op in DIVMOD { + // Only the signed opcodes: the unsigned reading of the same + // bits is a large positive dividend divided by a larger one, + // which is 0 and cannot trap. + assert_eq!( + divmod_may_trap(op, size, Some(min), Some(-1)), + is_signed(op), + "{op:?}.{size} MIN / -1" + ); + } + } + assert_eq!(signed_min_at(8), -128); + assert_eq!(signed_min_at(32), -2147483648); + assert_eq!(signed_min_at(64), i64::MIN as i128); + assert_eq!(signed_min_at(128), i128::MIN); + } + + /// The dividend is read at its own width first, so `INT_MIN` written as + /// the unsigned pattern 2147483648 is still the overflowing dividend, and + /// a 64-bit `INT_MIN` divided at 32 bits is not. + #[test] + fn divmod_may_trap_reads_the_dividend_at_its_width() { + assert!(divmod_may_trap( + Opcode::DivS, + 32, + Some(2147483648), + Some(-1) + )); + assert!(divmod_may_trap( + Opcode::DivS, + 32, + Some(-2147483648), + Some(-1) + )); + // -2^31 at 64 bits is an ordinary negative number, not `LONG_MIN`. + assert!(!divmod_may_trap( + Opcode::DivS, + 64, + Some(-2147483648), + Some(-1) + )); + // The divisor, likewise: 4294967295 at 32 bits signed is -1. + assert!(divmod_may_trap( + Opcode::DivS, + 32, + Some(-2147483648), + Some(4294967295) + )); + } + + /// The safe neighbours of both traps still fold, which is what stops the + /// rule from becoming "never fold a division". + #[test] + fn divmod_may_trap_allows_the_safe_neighbours() { + for op in DIVMOD { + for (a, b) in [ + (12, 4), + (13, 4), + (0, 4), // 0 / c, with the divisor known + (-2147483647, -1), // one above the overflowing dividend + (-2147483648, 1), // the dividend, but not against -1 + (-2147483648, -2), + (1, -1), + (i128::from(i32::MAX), -1), + ] { + assert!( + !divmod_may_trap(op, 32, Some(a), Some(b)), + "{op:?} {a} / {b} cannot trap" + ); + } + // `x / 1` and `x % 1` fold with the dividend unknown. + assert!(!divmod_may_trap(op, 32, None, Some(1)), "{op:?} x / 1"); + } + } + + /// Nothing else in this IR trips the divide trap, so nothing else is asked + /// to answer for it. + #[test] + fn divmod_may_trap_answers_only_for_division() { + for op in [Opcode::Add, Opcode::Mul, Opcode::Shl, Opcode::Lsr] { + assert!(!divmod_may_trap(op, 32, Some(1), Some(0)), "{op:?}"); + assert!(!divmod_may_trap(op, 32, None, None), "{op:?}"); + } + } + + fn binop_at(op: Opcode, size: u32) -> Instruction { + Instruction::new(op).with_size(size) + } + + /// `eval_binop` is the one door `instcombine`, `sccp` and `vrp` fold + /// through, so the refusal has to be visible from there. + #[test] + fn eval_binop_does_not_fold_a_trapping_division() { + for (op, size) in [ + (Opcode::DivS, 32), + (Opcode::ModS, 32), + (Opcode::DivS, 64), + (Opcode::ModS, 64), + ] { + let insn = binop_at(op, size); + assert_eq!(eval_binop(&insn, 1, 0), None, "{op:?}.{size} 1 / 0"); + assert_eq!( + eval_binop(&insn, signed_min_at(size), -1), + None, + "{op:?}.{size} MIN / -1" + ); + } + for op in [Opcode::DivU, Opcode::ModU] { + assert_eq!(eval_binop(&binop_at(op, 32), 1, 0), None, "{op:?} 1 / 0"); + } + } + + /// And still gives the answers it gave, at the truncation C requires. + #[test] + fn eval_binop_still_folds_a_safe_division() { + for (op, a, b, want) in [ + (Opcode::DivS, 12, 4, 3), + (Opcode::ModS, 13, 4, 1), + (Opcode::DivS, -13, 4, -3), + (Opcode::ModS, -13, 4, -1), + (Opcode::DivS, 13, -4, -3), + (Opcode::ModS, 13, -4, 1), + (Opcode::DivS, -2147483647, -1, 2147483647), + (Opcode::DivS, -2147483648, 1, -2147483648), + (Opcode::ModS, -2147483648, 1, 0), + (Opcode::DivU, 4294967295, 5, 858993459), + (Opcode::ModU, 4294967295, 5, 0), + ] { + assert_eq!( + eval_binop(&binop_at(op, 32), a, b), + Some(want), + "{op:?} {a} op {b}" + ); + } + } + + /// `Lsr` is the logical shift at every width, the 128-bit one included, + /// where the value is not narrowed first and an `i128` shift would be the + /// arithmetic one. No pass reaches this today (`arch::mapping` decomposes + /// every 128-bit shift before the optimizer runs), so this pins the + /// evaluation rather than a pass's behaviour. + #[test] + fn eval_shift_lsr_is_logical_at_every_width() { + assert_eq!( + eval_binop(&binop_at(Opcode::Lsr, 128), -1, 4), + Some((u128::MAX >> 4) as i128), + "-1 >>u 4 at 128 bits fills with zeros" + ); + assert_eq!( + eval_binop(&binop_at(Opcode::Lsr, 128), i128::MIN, 127), + Some(1), + "the sign bit shifts down to bit 0" + ); + assert_eq!( + eval_binop(&binop_at(Opcode::Asr, 128), -1, 4), + Some(-1), + "`Asr` is still the arithmetic shift" + ); + // The narrower widths, which were already right, are unchanged. + assert_eq!(eval_binop(&binop_at(Opcode::Lsr, 32), -1, 28), Some(15)); + assert_eq!(eval_binop(&binop_at(Opcode::Asr, 32), -1, 28), Some(-1)); + // An out-of-range count is undefined and is not folded, at 128 as + // anywhere else. + assert_eq!(eval_binop(&binop_at(Opcode::Lsr, 128), -1, 128), None); + } } diff --git a/cc/ir/constglobal.rs b/cc/ir/constglobal.rs index a0be9cd8d..cdf8a17bd 100644 --- a/cc/ir/constglobal.rs +++ b/cc/ir/constglobal.rs @@ -30,7 +30,7 @@ // use super::{ConstValue, Function, Initializer, Instruction, Module, Opcode, PseudoKind}; -use crate::types::{TypeId, TypeModifiers, TypeTable}; +use crate::types::{TypeId, TypeTable}; use std::collections::HashMap; /// A global whose value is known for the whole run. @@ -90,7 +90,13 @@ pub(crate) fn qualifies(g: &super::GlobalDef, types: &TypeTable) -> bool { } // `volatile` says the value can change for reasons not in the program, // which is exactly the assumption being made here. - if types.modifiers(g.typ).contains(TypeModifiers::VOLATILE) { + // + // `contains_volatile`, not the top-level modifier: a `const struct` with a + // `volatile` member is one of these objects too, and asking only what was + // written on the struct let it through the gate. Every access is checked + // again below, so this was not reachable as a wrong fold -- but the object + // and its members are one question and get one spelling of it. + if types.contains_volatile(g.typ) { return false; } // A weak definition exists to be replaced at link time, and the @@ -158,6 +164,14 @@ fn foldable_load( if insn.op != Opcode::Load || insn.src.len() != 1 || insn.offset != 0 { return None; } + // The access itself is observable, whatever the object's initializer says + // it holds: `const volatile int t = 0;` -- a hardware status word, a + // linker-set value -- must still be read. `qualifies` declines a global + // written `volatile`, but the qualifier can also be on the *access*, as in + // `*(volatile const int *)&t`, and only the instruction knows that. + if insn.is_volatile_access() { + return None; + } let PseudoKind::Sym(name) = &func.get_pseudo(insn.src[0])?.kind else { return None; }; @@ -186,6 +200,7 @@ mod tests { use super::*; use crate::ir::{BasicBlock, BasicBlockId, GlobalDef, Pseudo, PseudoId}; use crate::target::Target; + use crate::types::TypeModifiers; /// A module with one global and a function that loads it whole. fn module_loading(global: GlobalDef, types: &TypeTable, load_typ: TypeId) -> Module { diff --git a/cc/ir/dce.rs b/cc/ir/dce.rs index 6ce2180be..d1e3bf956 100644 --- a/cc/ir/dce.rs +++ b/cc/ir/dce.rs @@ -21,7 +21,7 @@ // reordering must consult `is_memory_barrier()` before crossing. // -use super::{BasicBlockId, Function, Opcode, PseudoId}; +use super::{BasicBlockId, Function, Instruction, Opcode, PseudoId}; use std::collections::{HashMap, HashSet, VecDeque}; const DEFAULT_LIVE_CAPACITY: usize = 64; @@ -50,9 +50,20 @@ pub fn run(func: &mut Function) -> bool { // Dead Code Elimination -/// Check if an opcode is a "root" (has side effects, cannot be deleted). -fn is_root(op: Opcode) -> bool { - op.has_side_effects() +/// Check if an instruction is a "root" (has side effects, cannot be deleted). +/// +/// Most of the answer is the opcode's, but not all of it: a `Load` is +/// deletable because reading an ordinary object has no effect, while reading a +/// `volatile` one is observable behaviour (C17 5.1.2.3p6) and must still +/// happen. That distinction is per *access*, not per opcode -- `*p` for a +/// `volatile int *p` is volatile and `*q` for an `int *q` is not -- so it is +/// asked of the instruction. Before this, `volatile int g; void f(void) { g; }` +/// emitted the load at `-O0` and nothing at all from `-O1` up. +/// +/// `Store` is a root by its opcode alone and stays that way: its correctness +/// must not come to depend on the marker. +fn is_root(insn: &Instruction) -> bool { + insn.op.has_side_effects() || insn.is_volatile_access() } /// Build a map from each pseudo to the instructions that define it. @@ -77,7 +88,7 @@ fn eliminate_dead_code(func: &mut Function) -> bool { // Phase 1: Mark roots and their operands as live for bb in &func.blocks { for insn in &bb.insns { - if is_root(insn.op) { + if is_root(insn) { // Mark all operands of root instructions as live for id in insn.uses() { if live.insert(id) { @@ -110,7 +121,7 @@ fn eliminate_dead_code(func: &mut Function) -> bool { for bb in &mut func.blocks { for insn in &mut bb.insns { // Skip roots - they're always live - if is_root(insn.op) { + if is_root(insn) { continue; } @@ -562,17 +573,103 @@ mod tests { #[test] fn test_is_root() { - assert!(is_root(Opcode::Ret)); - assert!(is_root(Opcode::Store)); - assert!(is_root(Opcode::Call)); - assert!(is_root(Opcode::Br)); - assert!(is_root(Opcode::Cbr)); - assert!(is_root(Opcode::Unreachable)); - - assert!(!is_root(Opcode::Add)); - assert!(!is_root(Opcode::Mul)); - assert!(!is_root(Opcode::Load)); - assert!(!is_root(Opcode::Phi)); + let bare = |op| is_root(&Instruction::new(op)); + + assert!(bare(Opcode::Ret)); + assert!(bare(Opcode::Store)); + assert!(bare(Opcode::Call)); + assert!(bare(Opcode::Br)); + assert!(bare(Opcode::Cbr)); + assert!(bare(Opcode::Unreachable)); + + assert!(!bare(Opcode::Add)); + assert!(!bare(Opcode::Mul)); + assert!(!bare(Opcode::Load)); + assert!(!bare(Opcode::Phi)); + } + + #[test] + fn test_volatile_load_is_root() { + // A plain load is deletable; the same load of a volatile object is not. + let plain = Instruction::new(Opcode::Load); + assert!(!is_root(&plain)); + + let vol = Instruction::new(Opcode::Load).with_volatile(true); + assert!(is_root(&vol), "reading a volatile object is observable"); + + // A volatile store is a root either way -- the marker must not be what + // its correctness rests on. + assert!(is_root( + &Instruction::new(Opcode::Store).with_volatile(true) + )); + assert!(is_root(&Instruction::new(Opcode::Store))); + } + + #[test] + fn test_volatile_load_with_dead_result_survives() { + // `volatile int g; void f(void) { g; }` -- the loaded value is never + // used, and DCE deleted the load outright before the marker existed. + let types = TypeTable::new(&Target::host()); + let mut func = Function::new("test", types.void_id); + + func.add_pseudo(Pseudo::reg(PseudoId(0), 0)); + func.add_pseudo(Pseudo::sym(PseudoId(1), "g".to_string())); + + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.add_insn(Instruction::new(Opcode::Entry)); + bb.add_insn( + Instruction::load(PseudoId(0), PseudoId(1), 0, types.int_id, 32).with_volatile(true), + ); + bb.add_insn(Instruction::ret(None)); + func.add_block(bb); + func.entry = BasicBlockId(0); + + assert!(!run(&mut func), "a volatile load is not dead code"); + assert_eq!(func.blocks[0].insns[1].op, Opcode::Load); + assert!(func.blocks[0].insns[1].is_volatile_access()); + } + + #[test] + fn test_volatile_load_keeps_its_address_live() { + // `volatile int *p; void f(void) { *p; }` -- the second load is the + // volatile access, and it is the only thing keeping the first (the + // read of `p` itself) alive. + let types = TypeTable::new(&Target::host()); + let mut func = Function::new("test", types.void_id); + + func.add_pseudo(Pseudo::sym(PseudoId(0), "p".to_string())); + func.add_pseudo(Pseudo::reg(PseudoId(1), 1)); + func.add_pseudo(Pseudo::reg(PseudoId(2), 2)); + + let ptr = types.pointer_to(types.int_id); + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.add_insn(Instruction::new(Opcode::Entry)); + // %1 = load p (plain: reading the pointer variable) + bb.add_insn(Instruction::load(PseudoId(1), PseudoId(0), 0, ptr, 64)); + // %2 = load *%1 (volatile: reading the pointed-to object) + bb.add_insn( + Instruction::load(PseudoId(2), PseudoId(1), 0, types.int_id, 32).with_volatile(true), + ); + bb.add_insn(Instruction::ret(None)); + func.add_block(bb); + func.entry = BasicBlockId(0); + + assert!(!run(&mut func), "neither load may be deleted"); + assert_eq!(func.blocks[0].insns[1].op, Opcode::Load); + assert_eq!(func.blocks[0].insns[2].op, Opcode::Load); + } + + #[test] + fn test_kill_clears_the_volatile_marker() { + // `kill` makes a `Nop`, which reaches no memory: a marker left behind + // would be a stale claim to any pass reading the field directly. + let types = TypeTable::new(&Target::host()); + let mut insn = + Instruction::load(PseudoId(0), PseudoId(1), 0, types.int_id, 32).with_volatile(true); + insn.kill(); + assert_eq!(insn.op, Opcode::Nop); + assert!(!insn.is_volatile); + assert!(!insn.is_volatile_access()); } #[test] diff --git a/cc/ir/dse.rs b/cc/ir/dse.rs index 641fc0f54..1982a86a0 100644 --- a/cc/ir/dse.rs +++ b/cc/ir/dse.rs @@ -111,7 +111,10 @@ fn scan_block( continue; } let loc = am.location_of(func, insn); - if !deletable(func, types, mi, &loc) { + // A volatile store is observable and stays, even when a later store + // overwrites every byte of it. `deletable` answers for a named object; + // the marker also answers for `*p` where `p` is a `volatile int *`. + if insn.is_volatile_access() || !deletable(func, types, mi, &loc) { // An untrackable store is still a write: drop whatever it may // have touched rather than pretending it did not happen. pending.retain(|p| !may_alias(&loc, &p.loc, mi)); diff --git a/cc/ir/inline.rs b/cc/ir/inline.rs index 99818ee78..56caea403 100644 --- a/cc/ir/inline.rs +++ b/cc/ir/inline.rs @@ -11,6 +11,7 @@ // (InstCombine, DCE) see the inlined code. // +use super::memexpand; use super::{ BasicBlock, BasicBlockId, Function, Instruction, Module, Opcode, Pseudo, PseudoId, PseudoKind, }; @@ -448,9 +449,11 @@ struct InlineContext { ret_typ: Option, /// Size (in bits) captured from the first cloned `Ret`. ret_size: u32, - /// Pseudos allocated as PhiSource targets in the cloned Ret blocks. - /// Added to the caller alongside other inlined pseudos. - phisrc_pseudos: Vec, + /// Pseudos allocated while lowering a cloned `Ret`: the PhiSource target + /// of a value return, and the temporaries that carry an aggregate handed + /// back by address into the result local. Added to the caller alongside + /// the other inlined pseudos. + ret_pseudos: Vec, /// Value pseudos created while cloning -- a constant's value lives on its /// pseudo, not on the instruction, so resolving /// `__builtin_va_arg_pack_len()` makes one. Added to the caller with the @@ -510,7 +513,7 @@ impl InlineContext { ret_arms: Vec::new(), ret_typ: None, ret_size: 0, - phisrc_pseudos: Vec::new(), + ret_pseudos: Vec::new(), const_pseudos: Vec::new(), } } @@ -658,10 +661,12 @@ fn clone_instruction( // matching Phi is materialized by `inline_call_site` after all blocks // are cloned. // - // For a two-register struct return, both halves are stored to the - // result local's memory (a Sym pseudo). The Sym itself remains - // single-defined (it is the local's address); the stores are - // side-effecting writes to memory and do not violate SSA. + // An aggregate returned in registers takes neither path: the call's + // result slot is a local, and the returned bytes are written into it. + // Two-register form, both halves are stored there; address form + // (`returns_aggregate_address`), the bytes are copied there. The Sym + // itself remains single-defined (it is the local's address); the + // stores are side-effecting writes to memory and do not violate SSA. Opcode::Ret => { let mut result = Vec::new(); @@ -674,6 +679,23 @@ fn clone_instruction( let remapped_low = ctx.remap_pseudo(insn.src[0], callee_func); let remapped_high = ctx.remap_pseudo(insn.src[1], callee_func); + // The high half is whatever is left past the first + // eight bytes, which is 1..=8 of them: + // `returns_reg_aggregate` admits 9..=16 bytes, so a + // 12-byte struct leaves four. Stored at a hardcoded + // 64 bits it overran the result local by four -- and + // it disagreed with the *load* in + // `emit_two_reg_return`, which has always narrowed the + // high half to `min(64, struct_size - 64)`. + let high_bits = match insn.size.checked_sub(64) { + Some(rest) if rest > 0 => rest.min(64), + // Two registers means more than eight bytes, so + // this is not a shape `emit_two_reg_return` + // produces. Keep the old width rather than emit a + // store of no bits at all. + _ => 64, + }; + let mut store_low = Instruction::store( remapped_low, target, @@ -689,10 +711,65 @@ fn clone_instruction( target, 8, insn.typ.unwrap_or(crate::types::TypeId::INVALID), - 64, + high_bits, ); store_high.pos = insn.pos; result.push(store_high); + } else if insn.returns_aggregate_address() { + // The callee hands back the *address* of the + // aggregate -- one SSE register holding sixteen + // bytes, an x87 aggregate, an HFA -- because no pair + // of general registers can carry it (see + // `aggregate_ret_is_address`). A call leaves the + // value in the result local and the caller reads it + // from there, so the spliced return has to put the + // bytes there itself. + // + // Asked of the `Ret`'s own classification, which is + // what `returns_two_regs` just above asks and all + // this pass has: it carries no `TypeTable` and cannot + // classify anything itself. Only the one-SSE shape + // reaches here today -- `Function::ret_is_address` + // still keeps an x87 aggregate and an HFA out of the + // inliner entirely, for the reason recorded where it + // is set. + // + // Phi-ing the source instead handed the caller a + // pointer where the value belonged: + // + // %16 = symaddr.64 %11(@mk_inline0_r.2) + // %17 = phisrc.128 %16 + // + // -- a 64-bit address as a 128-bit value. Inlining + // changed the answer, and only for this shape: the + // two-register return stores its halves just above, + // and an aggregate of eight bytes or less never + // reaches `emit_two_reg_return` at all, so its `Ret` + // already carries a loaded value. + // + // `insn.size` is the aggregate's own width, in the + // 8/4/2/1 chunks every other block move in the + // compiler uses, so a size that is not a multiple of + // eight is copied exactly rather than rounded up past + // either object. No bound is needed: a register + // return is at most a four-element HFA, thirty-two + // bytes. The access type stays the `Ret`'s own, as + // the two-register stores above keep theirs -- this + // pass has no `TypeTable` to name a qword with. + let src_addr = ctx.remap_pseudo(*ret_val, callee_func); + let typ = insn.typ.unwrap_or(crate::types::TypeId::INVALID); + for (offset, chunk) in memexpand::block_chunks(i64::from(insn.size / 8)) { + let temp = ctx.alloc_pseudo_id(); + ctx.ret_pseudos.push(Pseudo::undef(temp)); + let mut load = + Instruction::load(temp, src_addr, offset, typ, chunk.bits()); + load.pos = insn.pos; + result.push(load); + let mut store = + Instruction::store(temp, target, offset, typ, chunk.bits()); + store.pos = insn.pos; + result.push(store); + } } else { // Single-value return: emit PhiSource in the predecessor // and record the arm for `inline_call_site` to assemble @@ -706,7 +783,7 @@ fn clone_instruction( before cloning a Ret", ); let phisrc_target = ctx.alloc_pseudo_id(); - ctx.phisrc_pseudos + ctx.ret_pseudos .push(Pseudo::phi(phisrc_target, phisrc_target.0)); let typ = insn.typ.unwrap_or(crate::types::TypeId::INVALID); @@ -1142,8 +1219,25 @@ fn inline_call_site( continue; } - let mut offset = 0i64; - while (offset as usize) < copy.size_bytes { + // The shared 8/4/2/1 descent, in the chunks every other block + // move in the compiler uses. Stepping 8 to `size_bytes` + // instead rounds the size *up*: a 12-byte + // `struct P { float x, y, z; }` moved 16 bytes, over-reading + // the caller's argument and over-writing the callee's local + // -- the same defect the linearizer's parameter prologue and + // sret return path each had, spelled the same way. + // + // No upper bound here, deliberately: unlike a copy the + // program wrote, `size_bytes` is at most 32 -- a + // `long double _Complex` -- because only a complex value or a + // two-register aggregate is recorded as an implicit parameter + // copy. So the unroll cannot run away and needs no + // `memexpand::INLINE_LIMIT_BYTES` cap. This pass builds into + // a `Vec` rather than through `Linearizer::emit` + // and has no `TypeTable`, so it takes the offsets and widths + // and keeps `qword_type` as the access type, exactly as the + // register-sized case above does. + for (offset, chunk) in memexpand::block_chunks(copy.size_bytes as i64) { let temp = ctx.alloc_pseudo_id(); implicit_copy_pseudos.push(Pseudo::undef(temp)); copy_insns.push(Instruction::load( @@ -1151,16 +1245,15 @@ fn inline_call_site( call_arg, offset, copy.qword_type, - 64, + chunk.bits(), )); copy_insns.push(Instruction::store( temp, remapped_local, offset, copy.qword_type, - 64, + chunk.bits(), )); - offset += 8; } } // Insert copies at the beginning of the entry block @@ -1323,8 +1416,9 @@ fn inline_call_site( for pseudo in std::mem::take(&mut ctx.const_pseudos) { caller.replace_pseudo(pseudo); } - // Add PhiSource target pseudos generated for the return-value Phi. - for pseudo in std::mem::take(&mut ctx.phisrc_pseudos) { + // Add the pseudos the cloned returns allocated: PhiSource targets, and + // the temporaries of an aggregate copied into the result local. + for pseudo in std::mem::take(&mut ctx.ret_pseudos) { if !caller.has_pseudo(pseudo.id) { caller.add_pseudo(pseudo); } @@ -2450,6 +2544,205 @@ mod tests { callee } + /// An implicit parameter copy moves exactly the object's bytes, in the + /// 8/4/2/1 chunks `memexpand::block_chunks` gives every other block move. + /// + /// `while offset < size_bytes { load 64; store 64; offset += 8 }` rounds + /// *up*: a 12-byte `struct P { float x, y, z; }` moved 16 bytes, + /// over-reading the caller's argument and over-writing the callee's local. + /// The fourth bytes are usually frame padding, so a program can be correct + /// and still be reading memory that does not belong to the object -- and + /// would fault if it ended a page. + #[test] + fn test_implicit_param_copy_moves_no_more_than_the_object() { + let types = TypeTable::new(&Target::host()); + + // `static void callee(struct P p)`, with the twelve bytes of `p` + // arriving by address and copied into the callee's own local. + let mut callee = Function::new("callee", types.void_id); + callee.add_param("p", types.void_ptr_id); + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.insns.push(Instruction::new(Opcode::Entry)); + bb.insns.push(Instruction::ret(None)); + callee.add_block(bb); + callee.entry = BasicBlockId(0); + callee.add_pseudo(Pseudo::sym(PseudoId(0), "p.0".to_string())); + callee.next_pseudo = 1; + callee + .implicit_param_copies + .push(crate::ir::ImplicitParamCopy { + arg_index: 0, + local_sym: PseudoId(0), + size_bytes: 12, + qword_type: types.long_id, + arg_is_address: true, + }); + + let mut caller = Function::new("caller", types.void_id); + let mut cb = BasicBlock::new(BasicBlockId(0)); + cb.insns.push(Instruction::new(Opcode::Entry)); + cb.insns.push(Instruction::call( + None, + "callee", + vec![PseudoId(0)], + vec![types.void_ptr_id], + types.void_id, + 0, + )); + cb.insns.push(Instruction::ret(None)); + caller.add_block(cb); + caller.entry = BasicBlockId(0); + caller.add_pseudo(Pseudo::reg(PseudoId(0), 0)); + caller.next_pseudo = 1; + + assert!(inline_call_site(&mut caller, 0, 1, &callee)); + + let moves: Vec<(Opcode, i64, u32)> = caller + .blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.offset, i.size)) + .collect(); + assert_eq!( + moves, + vec![ + (Opcode::Load, 0, 64), + (Opcode::Store, 0, 64), + (Opcode::Load, 8, 32), + (Opcode::Store, 8, 32), + ], + "a 12-byte object has four bytes at offset 8, not eight" + ); + } + + /// `static struct Q mk(void) { struct Q r; ...; return r; }`, whose `Ret` + /// hands the aggregate back by *address* under `ret` -- the shape + /// `emit_two_reg_return` emits for a class no pair of general registers + /// can carry. + fn aggregate_address_ret_callee( + types: &TypeTable, + ret: crate::abi::ArgClass, + size_bits: u32, + ) -> Function { + let mut callee = Function::new("mk", types.void_id); + callee.is_static = true; + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.insns.push(Instruction::new(Opcode::Entry)); + // The address of the callee's own local, which is what the `Ret` + // carries: `%1 = symaddr %0(@r)`. + bb.insns.push(Instruction::sym_addr( + PseudoId(1), + PseudoId(0), + types.long_id, + )); + let mut ret_insn = Instruction::ret_typed(Some(PseudoId(1)), types.long_id, size_bits); + ret_insn.abi_info = Some(Box::new(crate::ir::CallAbiInfo::new(vec![], ret))); + bb.insns.push(ret_insn); + callee.add_block(bb); + callee.entry = BasicBlockId(0); + callee.add_pseudo(Pseudo::sym(PseudoId(0), "r.0".to_string())); + callee.add_pseudo(Pseudo::reg(PseudoId(1), 1)); + callee.next_pseudo = 2; + callee + } + + /// An inlined return of an aggregate handed back by address copies the + /// aggregate's *bytes* into the call's result local, and phis nothing. + /// + /// A call leaves the value in that local and the caller reads it from + /// there. Feeding the `Ret`'s source into the continuation phi instead + /// gave the caller the address of the callee's own copy -- for + /// `struct { __float128 a; }`, one SSE register and sixteen bytes, the + /// post-opt IR read `%17 = phisrc.128 %16` where `%16` was a + /// `symaddr.64`. The caller then stored a pointer into the first eight + /// bytes of a sixteen-byte slot and read the other eight uninitialized, + /// so inlining changed the answer. + /// + /// The bytes move in `memexpand::block_chunks`, so a width that is not a + /// multiple of eight -- a three-`float` HFA is twelve bytes -- lands + /// exactly rather than reaching past either object. + #[test] + fn test_inlined_aggregate_address_return_copies_the_value() { + let types = TypeTable::new(&Target::host()); + let cases = [ + ( + "struct { __float128 a; }: one SSE register, sixteen bytes", + crate::abi::ArgClass::Direct { + classes: vec![crate::abi::RegClass::Sse], + size_bits: 128, + }, + 128u32, + vec![ + (Opcode::Load, 0, 64), + (Opcode::Store, 0, 64), + (Opcode::Load, 8, 64), + (Opcode::Store, 8, 64), + ], + ), + ( + "struct { float a, b, c; } as an HFA: twelve bytes", + crate::abi::ArgClass::Hfa { + base: crate::abi::HfaBase::Float32, + count: 3, + }, + 96, + vec![ + (Opcode::Load, 0, 64), + (Opcode::Store, 0, 64), + (Opcode::Load, 8, 32), + (Opcode::Store, 8, 32), + ], + ), + ]; + + for (what, ret_class, size_bits, want) in cases { + let callee = aggregate_address_ret_callee(&types, ret_class, size_bits); + + // `struct Q v = mk();` -- the result local the backend would have + // written the returned registers into is the call's target. + let mut caller = Function::new("caller", types.void_id); + let mut cb = BasicBlock::new(BasicBlockId(0)); + cb.insns.push(Instruction::new(Opcode::Entry)); + cb.insns.push(Instruction::call( + Some(PseudoId(0)), + "mk", + vec![], + vec![], + types.long_id, + size_bits, + )); + cb.insns.push(Instruction::ret(None)); + caller.add_block(cb); + caller.entry = BasicBlockId(0); + caller.add_pseudo(Pseudo::sym(PseudoId(0), "__2reg_0".to_string())); + caller.next_pseudo = 1; + + assert!(inline_call_site(&mut caller, 0, 1, &callee), "{what}"); + + let insns: Vec<&Instruction> = + caller.blocks.iter().flat_map(|b| b.insns.iter()).collect(); + let moves: Vec<(Opcode, i64, u32)> = insns + .iter() + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.offset, i.size)) + .collect(); + assert_eq!(moves, want, "{what}"); + + // Every store writes the result local, and nothing phis the + // address: an address is not a value. + for store in insns.iter().filter(|i| i.op == Opcode::Store) { + assert_eq!(store.src.first(), Some(&PseudoId(0)), "{what}"); + } + assert!( + !insns + .iter() + .any(|i| matches!(i.op, Opcode::Phi | Opcode::PhiSource)), + "{what}: the returned aggregate is not phi-ed" + ); + } + } + /// Each inlined copy of a function that takes a label's address names /// its own clone of the block, in the caller. /// diff --git a/cc/ir/instcombine.rs b/cc/ir/instcombine.rs index 7fa52de89..7efcc352f 100644 --- a/cc/ir/instcombine.rs +++ b/cc/ir/instcombine.rs @@ -25,9 +25,9 @@ // use super::constfold::{ - at_width, cmp_operand_width, eval_binop, eval_fbinop, eval_fcvt, eval_fcvtf, eval_fternop, - eval_funop, eval_unop, fcmp_decided, fcmp_mask, fcmp_outcome, get_cmp_info, mirror_mask, - possible_against, result_type_of, FCMP_ALL, + at_width, cmp_operand_width, divmod_may_trap, eval_binop, eval_fbinop, eval_fcvt, eval_fcvtf, + eval_fternop, eval_funop, eval_unop, fcmp_decided, fcmp_mask, fcmp_outcome, get_cmp_info, + mirror_mask, possible_against, result_type_of, FCMP_ALL, }; use super::facts::{CmpDomain, CmpFacts, ConstMap, Relation}; use super::{ConstValue, Function, Instruction, Opcode, PseudoId}; @@ -401,16 +401,23 @@ fn simplify_div(insn: &Instruction, consts: &ConstMap) -> Simplification { let val1 = consts.get(src1).map(|v| at_width(v, size, signed)); let val2 = consts.get(src2).map(|v| at_width(v, size, signed)); + // A division that can trap is left to trap: `constfold::divmod_may_trap` + // states the rule and both of the operand pairs it covers. This is the + // whole of what stands between an unknown divisor and a folded answer -- + // `0 / x` used to fold to zero here without ever asking what `x` was, and + // a divisor of -1 is only safe once the dividend is known not to be the + // most negative value. + if divmod_may_trap(insn.op, size, val1, val2) { + return Simplification::None; + } + match (val1, val2) { - // Constant folding: a / b -> (a / b) (avoid div by zero) + // Constant folding: a / b -> (a / b) (Some(a), Some(b)) => fold_with(insn, a, b), // Algebraic: x / 1 -> x (None, Some(1)) => Simplification::CopyFrom(src1), - // Algebraic: 0 / x -> 0 - (Some(0), None) => fold_to_zero(), - _ => Simplification::None, } } @@ -430,13 +437,18 @@ fn simplify_mod(insn: &Instruction, consts: &ConstMap) -> Simplification { let val1 = consts.get(src1).map(|v| at_width(v, size, signed)); let val2 = consts.get(src2).map(|v| at_width(v, size, signed)); + // The remainder traps exactly where the division does -- on x86-64 it is + // the same `idiv`, computing the same quotient -- so it asks the same + // question. `0 % x` folded to zero here for an unknown `x`, and + // `INT_MIN % -1` folded to zero through `fold_with`. + if divmod_may_trap(insn.op, size, val1, val2) { + return Simplification::None; + } + match (val1, val2) { - // Constant folding: a % b -> (a % b) (avoid mod by zero) + // Constant folding: a % b -> (a % b) (Some(a), Some(b)) => fold_with(insn, a, b), - // Algebraic: 0 % x -> 0 - (Some(0), None) => fold_to_zero(), - // Algebraic: x % 1 -> 0 (None, Some(1)) => fold_to_zero(), diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index 8c7012081..7dbaab6c7 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -19,10 +19,11 @@ use crate::abi::{get_abi_for_conv, CallingConv}; use crate::diag::{get_all_stream_names, Position}; use crate::float::FloatVal; use crate::ir::linearize_atomic::AtomicLvalue; +use crate::ir::linearize_emit::CompoundAssign; use crate::parse::ast::{ - BinaryOp, BlockItem, Expr, ExprKind, ExternalDecl, FpCompare, FpTest, FunctionDef, GnuAtomicOp, - InitElement, InlineLibraryFn, MemoryFn, NarrowedLibraryCall, OffsetOfPath, ParamStyle, - TranslationUnit, UnaryOp, + AssignOp, BinaryOp, BlockItem, Expr, ExprKind, ExternalDecl, FpCompare, FpTest, FunctionDef, + GnuAtomicOp, InitElement, InlineLibraryFn, MemoryFn, NarrowedLibraryCall, OffsetOfPath, + ParamStyle, TranslationUnit, UnaryOp, }; use crate::strings::{StringId, StringTable}; use crate::symbol::{SymbolId, SymbolTable}; @@ -122,14 +123,119 @@ pub(crate) struct ResolvedDesignator { pub(crate) bit_offset: Option, pub(crate) bit_width: Option, pub(crate) access_bytes: Option, + /// The member this designator chain named in each union it passed + /// through. See [`UnionMembers`]. + pub(crate) unions: UnionMembers, +} + +/// Which member a union came to hold, for each union an initializer list +/// reached. +/// +/// C17 6.7.9p19 makes a later initializer override the earlier one for the +/// *same* subobject, and a union has only one subobject at a time: whether +/// `.u.p.y = 9` overrides part of what `.u = {1, 2}` wrote or replaces all of +/// it turns on whether the union still holds `p`. Neither the byte offset nor +/// the lowered [`Initializer`] can say -- every member of a union begins at +/// the same byte, and `Initializer` is what the emitter consumes and has no +/// room for a discriminant -- so the choice is recorded beside it. +/// +/// Each entry is the byte offset of a union within the object whose +/// initializer list produced it, that union's type, and the index of the +/// member in question. The type is part of the key because a union declared +/// directly inside another begins at the same byte as it does. +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub(crate) struct UnionMembers(Vec<(usize, TypeId, usize)>); + +impl UnionMembers { + /// Note that the union at `offset` holds `member`, replacing whatever it + /// was last said to hold. + pub(crate) fn record(&mut self, offset: usize, typ: TypeId, member: usize) { + match self.0.iter_mut().find(|e| (e.0, e.1) == (offset, typ)) { + Some(entry) => entry.2 = member, + None => self.0.push((offset, typ, member)), + } + } + + fn member_at(&self, offset: usize, typ: TypeId) -> Option { + self.0 + .iter() + .find(|e| (e.0, e.1) == (offset, typ)) + .map(|e| e.2) + } + + /// Take on everything `other` says, which is later and so decisive. + pub(crate) fn absorb(&mut self, other: &UnionMembers) { + for &(offset, typ, member) in &other.0 { + self.record(offset, typ, member); + } + } + + /// Forget every union starting inside `range`, whose contents some later + /// initializer has just discarded. + pub(crate) fn clear_range(&mut self, range: std::ops::Range) { + self.0.retain(|e| !range.contains(&e.0)); + } +} + +/// What the two initializers being merged say about the unions between them, +/// and how far into the earlier one's object the merge has descended. +/// +/// A union is descended into only where the two agree: `held` is the member +/// the earlier initializer gave a value to, `named` the member the later +/// one's designator reached through, and anything else means the union comes +/// to hold something different and everything it held goes. Both are keyed by +/// byte offset within the object whose initializer list holds both entries, +/// which is what `base` counts from. +#[derive(Clone, Copy, Default)] +pub(crate) struct UnionFold<'a> { + held: Option<&'a UnionMembers>, + named: Option<&'a UnionMembers>, + base: usize, +} + +impl<'a> UnionFold<'a> { + pub(crate) fn new(held: &'a UnionMembers, named: &'a UnionMembers, base: usize) -> Self { + Self { + held: Some(held), + named: Some(named), + base, + } + } + + /// The same view, `offset` bytes further into the object. + pub(crate) fn inside(self, offset: usize) -> Self { + Self { + base: self.base + offset, + ..self + } + } + + /// The member both sides agree the union of type `typ` at the current + /// offset holds, if they do. + pub(crate) fn agreed(&self, typ: TypeId) -> Option { + let held = self.held?.member_at(self.base, typ)?; + (self.named?.member_at(self.base, typ)? == held).then_some(held) + } } pub(crate) struct RawFieldInit { pub(crate) offset: usize, pub(crate) field_size: usize, + /// The type of the subobject this initializer names. + /// + /// Carried so that resolving two initializers that describe overlapping + /// storage can ask *how* they overlap: a later one naming a member of an + /// earlier one's struct or array replaces only that member, while one + /// reachable only through a union replaces the union's whole contents. + /// Byte spans alone cannot tell the two apart. + pub(crate) typ: TypeId, pub(crate) init: Initializer, pub(crate) bit_offset: Option, pub(crate) bit_width: Option, + /// The member each union inside this subobject came to hold. + pub(crate) held: UnionMembers, + /// The member each union this entry's designator passed through named. + pub(crate) named: UnionMembers, } impl RawFieldInit { @@ -151,6 +257,35 @@ impl RawFieldInit { } } +/// How a byte range sits inside an object, as C17 6.7.9p19 needs to know it: +/// an initializer for a subobject overrides the previous initializer for +/// *that* subobject, and whether some other initializer survives depends on +/// what lies between the two. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum SubobjectPlace { + /// The range is a subobject reached through struct members and array + /// elements only (possibly the whole object). Initializing it leaves + /// every other subobject of the enclosing object untouched. + Member, + /// The range is reached only by descending into this union, whose bytes + /// span `offset..offset + size` of the enclosing object. A union holds one + /// member at a time, so initializing through it discards whatever the + /// union held before. + ThroughUnion { offset: usize, size: usize }, + /// The range is the storage of a bit-field declared by the struct + /// spanning `offset..offset + size` of the enclosing object. A bit-field + /// is not addressable storage of its own -- it shares a carrier with its + /// neighbours -- so an initializer for one replaces its bits and leaves + /// theirs, which is done in that struct's own initializer rather than by + /// replacing a subobject. + /// + /// Only asked for, and only ever answered, for a bit-field's own bits. + BitfieldCarrier { offset: usize, size: usize }, + /// The range is not a subobject at all: it straddles two members, or it is + /// a bit-field carrier's window rather than a named object. + NotASubobject, +} + /// Result from member_index_for_designator indicating where positional /// initialization should continue after a designated field. pub(crate) enum MemberDesignatorResult { @@ -202,6 +337,13 @@ pub(crate) struct StructFieldVisit { pub(crate) bit_offset: Option, pub(crate) bit_width: Option, pub(crate) access_bytes: Option, + /// Index, in the member list walked, of the member this visit initializes + /// or lies inside. For a union that is the member it comes to hold, which + /// its byte offset cannot say. + pub(crate) member_index: Option, + /// The member this visit's designator chain named in each union it passed + /// through, by byte offset within the object being initialized. + pub(crate) unions: UnionMembers, } pub(crate) enum StructFieldVisitKind { @@ -222,7 +364,28 @@ pub(crate) struct StaticLocalInfo { // Linearizer -/// Linearizer context for converting AST to IR +/// A declaration scope the linearizer has entered. +/// +/// Handed out by [`Linearizer::push_scope`] and given back to +/// [`Linearizer::pop_scope`]. A block scope is also the lifetime of every +/// VLA declared in it (C17 6.2.4p7), so the token carries the [`VlaMark`] +/// depth as it stood on entry and leaving the scope puts the stack pointer +/// back to what the first mark above that depth captured. +/// +/// Carrying both in one token is the point. The declaration scope and the +/// VLA scope used to be two stacks opened by hand at separate call sites, +/// and only some of the sites that opened the first opened the second: a VLA +/// declared in a `for` init clause, in the `for` arm of the switch-body +/// walker, or in a statement expression was never released. Now there is no +/// way to enter one without the other, and none to leave one without the +/// other. +#[must_use = "a scope that is entered must be left through pop_scope"] +pub(crate) struct Scope { + /// The `vla_marks` depth on entry; every mark above it belongs to this + /// scope and is released when it ends. + pub(crate) vla_entry: usize, +} + /// A captured stack pointer and the loop/switch nesting it was captured at. /// /// See [`Linearizer::vla_marks`]. @@ -244,14 +407,27 @@ pub(crate) struct HiddenReturnSlot { pub(crate) arg_typ: TypeId, } +/// Where a jump whose VLA restore is still undecided may land. +pub(crate) enum GotoTarget { + /// `goto L;`, or one label edge of an `asm goto`: a single block. + Label(BasicBlockId), + /// `goto *p;`. The address is not known here, so the jump is taken to + /// reach any label whose address this function takes, and only the + /// scopes that *every* candidate lies outside of may be released -- the + /// **deepest** depth recorded for any of them. Releasing down to a + /// shallower one would free storage still in scope at another candidate, + /// which is the one error a missing restore cannot cause. + AnyAddressTaken, +} + /// A forward `goto` whose VLA restore is decided once its label is placed. /// /// `marks` is the mark stack as it stood at the jump, so the restore can name /// whichever scope the label turns out to sit in: `marks[label_depth]` is the /// stack pointer captured on entry to the outermost scope the jump leaves. pub(crate) struct PendingGotoVla { - /// The label jumped to. - pub(crate) label: String, + /// Where the jump goes. + pub(crate) target: GotoTarget, /// The block the branch was emitted into. pub(crate) bb: BasicBlockId, /// Where in that block the branch sits; the restore goes just before it. @@ -260,6 +436,7 @@ pub(crate) struct PendingGotoVla { pub(crate) marks: Vec, } +/// Linearizer context for converting AST to IR pub struct Linearizer<'a> { /// The module being built pub(crate) module: Module, @@ -344,7 +521,11 @@ pub struct Linearizer<'a> { /// /// Only labels in a function that declares a VLA appear here, so nothing /// is recorded for the ordinary case. - pub(crate) label_vla_depth: std::collections::HashMap, + /// + /// Keyed by the label's block rather than its name: a computed `goto` + /// knows its candidates only as the blocks in `addr_taken_labels`, and + /// one map serves both it and the named jumps. + pub(crate) label_vla_depth: std::collections::HashMap, /// Forward `goto`s that may be leaving a VLA's scope, to be resolved once /// every label's depth is known. /// @@ -490,14 +671,28 @@ impl<'a> Linearizer<'a> { } } - /// Push a new local scope. Subsequent `insert_local` calls will record + /// Enter a declaration scope. Subsequent `insert_local` calls will record /// the previous value so `pop_scope` can restore it. - pub(crate) fn push_scope(&mut self) { + /// + /// Entering a declaration scope *is* entering a VLA scope: the returned + /// [`Scope`] remembers the mark depth so `pop_scope` releases whatever + /// the scope allocated. See [`Scope`] for why the two are one operation. + pub(crate) fn push_scope(&mut self) -> Scope { self.local_scope_stack.push(Vec::new()); + Scope { + vla_entry: self.vla_marks.len(), + } } - /// Pop the current local scope, restoring all locals to their pre-scope values. - pub(crate) fn pop_scope(&mut self) { + /// Leave the scope `scope` opened: release the VLAs declared in it and + /// restore every local it shadowed. + /// + /// The stack restore comes first, while the block the scope ends in is + /// still the current one, and is emitted only on the falling-out path -- + /// a `break`, `continue`, `goto` or `return` that left already did its + /// own unwinding and terminated the block. + pub(crate) fn pop_scope(&mut self, scope: Scope) { + self.close_vla_scope(&scope); if let Some(entries) = self.local_scope_stack.pop() { for (sym, prev) in entries.into_iter().rev() { match prev { @@ -729,6 +924,7 @@ impl<'a> Linearizer<'a> { /// Add an instruction to the current basic block pub(crate) fn emit(&mut self, insn: Instruction) { + let insn = self.mark_volatile_access(insn); let insn = self.displacement_in_range(insn); if let Some(bb_id) = self.current_bb { // Attach current source position for debug info @@ -741,6 +937,31 @@ impl<'a> Linearizer<'a> { bb.add_insn(insn); } } + + /// Mark an access to a `volatile` object as one, from the type it reaches. + /// + /// Every `Load` and `Store` the linearizer emits passes through + /// [`Self::emit`], and each carries in `typ` the type of the object it is + /// accessing -- so this is the one place the qualifier has to be read, and + /// the one place it can be read for *every* access, including `*p` for a + /// `volatile int *p`, where there is no variable holding the qualifier to + /// ask (which is why `LocalVar::is_volatile` alone let DCE delete every + /// discarded `volatile` read from `-O1` up). + /// + /// A marker a site set itself is kept rather than recomputed, so a site + /// that knows more than the access type does can say so: a bit-field reads + /// a storage unit whose type is the carrier, and a composite copy reads + /// integer chunks, neither of which is the qualified type. + fn mark_volatile_access(&self, mut insn: Instruction) -> Instruction { + if !matches!(insn.op, Opcode::Load | Opcode::Store) || insn.is_volatile { + return insn; + } + if let Some(typ) = insn.typ { + insn.is_volatile = self.types.contains_volatile(typ); + } + insn + } + /// Keep a load's or store's constant offset inside a machine displacement. /// /// Both backends address `src[0] + offset` with a signed 32-bit @@ -1273,14 +1494,20 @@ impl<'a> Linearizer<'a> { /// /// `static_locals` is deliberately not cleared: it persists across /// functions. - fn reset_for_function(&mut self, func: &FunctionDef) { + /// + /// Returns the function-level [`Scope`], which `linearize_function` gives + /// back once the body is lowered. Nothing is released there -- the + /// epilogue restores `%rsp` from the frame pointer, and the body's own + /// block scope has already dropped every mark -- but it is entered the + /// same way as any other scope so that no site can enter one without the + /// other. + fn reset_for_function(&mut self, func: &FunctionDef) -> Scope { // Reset per-function state self.next_pseudo = 0; self.next_bb = 0; self.var_map.clear(); self.locals.clear(); self.local_scope_stack.clear(); - self.push_scope(); // function-level scope self.label_map.clear(); self.break_targets.clear(); self.continue_targets.clear(); @@ -1298,6 +1525,10 @@ impl<'a> Linearizer<'a> { // Remove from extern_symbols since we're defining this function self.module.extern_symbols.remove(&self.current_func_name); // Note: static_locals is NOT cleared - it persists across functions + + // After `vla_marks.clear()`: the scope records the depth it starts + // at, which for the function scope has to be zero. + self.push_scope() } /// Whether a function body declares anything variably modified. @@ -1352,7 +1583,7 @@ impl<'a> Linearizer<'a> { // expression to the same rule. let written_labels = self.check_jumps_into_protected_scopes(&func.body); - self.reset_for_function(func); + let func_scope = self.reset_for_function(func); self.written_labels = written_labels; // Create function - use storage class from FunctionDef @@ -1454,32 +1685,48 @@ impl<'a> Linearizer<'a> { } // A `Ret` that carries an address; a call's result slot holds the - // value. The inliner has to know not to splice across that boundary. - // An aggregate returned in st(0) has exactly the same shape as a - // complex one, and missing it is a miscompile visible only at -O. + // value. Whoever consumes that return has to read the bytes out of the + // storage it names, and the inliner has to know which of the two it is + // splicing. `aggregate_ret_is_address` is the one place that answers + // it -- the same function `emit_two_reg_return` asks before emitting + // the address form, so the shape and the question about the shape + // cannot drift apart. They had: spelled out a second time here as + // "x87 or HFA", this missed a sixteen-byte aggregate returned in one + // SSE register, and before that it was gated behind the + // *two-register* path's 128-bit cap, which missed every HFA past + // sixteen bytes -- four `double`s is thirty-two bytes and still comes + // back in d0-d3. // - // Asked of the ABI classification directly rather than through - // `returns_reg_aggregate`, which is the *two-register* return path and - // so stops at 128 bits. An HFA comes back in registers at any size -- - // four `double`s is thirty-two bytes and still returns in d0-d3 -- so - // gating on that cap made every HFA past 128 bits report that its `Ret` - // carried a value. The inliner then spliced the body in and phi-ed the - // address as if it were the aggregate, and the caller read the pointer's - // own storage as the struct's bytes. The call-site half of this - // decision already has no size bound; the two had drifted. - let returns_addr_aggregate = (ret_kind == TypeKind::Struct || ret_kind == TypeKind::Union) - // An aggregate that fits in one register comes back *as* a value, - // so its `Ret` carries one; only past 64 bits is an address handed - // back. Dropping this bound along with the 128-bit cap refused to - // inline every HFA, including `struct { float x, y; }`, which was - // correct before and is the common aarch64 shape. - && struct_size_bits > 64 - && !returns_large_struct - && matches!( + // Classified only for an aggregate that is not going through the + // hidden pointer, which is the only shape the question is about. + let ret_class = ((ret_kind == TypeKind::Struct || ret_kind == TypeKind::Union) + && !returns_large_struct) + .then(|| { get_abi_for_conv(self.current_calling_conv, self.target) - .classify_return(func.return_type, self.types), - crate::abi::ArgClass::X87 { .. } | crate::abi::ArgClass::Hfa { .. } - ); + .classify_return(func.return_type, self.types) + }); + // Which of those the inliner is kept away from -- a narrower question + // than the shape, and no longer the same one. `clone_instruction`'s + // `Ret` arm now copies an aggregate handed back by address into the + // call's result local, which is where a call leaves it, so the + // one-SSE-register shape is spliced correctly instead of refused. + // + // An x87 aggregate and an HFA travel by address for the same reason + // and that copy would move them just as well, but they are not ready + // to be let through: the copy reads the `Ret`'s own ABI + // classification, and only `emit_two_reg_return` attaches one -- + // which `returns_reg_aggregate` above stops calling past 128 bits. So + // a three- or four-`double` HFA returns an address under no + // classification at all, and lifting this refusal has it phi-ed again + // (`%45 = phisrc.192 %44`, where `%44` is a `symaddr.64`). Letting + // those in means carrying the classification onto every + // register-returned aggregate's `Ret` first, which is its own change + // -- and `codegen_aarch64_hfa_returning_function_is_not_inlined` + // pins this refusal until then. + let returns_addr_aggregate = ret_class.as_ref().is_some_and(|class| { + super::aggregate_ret_is_address(class, struct_size_bits) + && !matches!(class, crate::abi::ArgClass::Direct { .. }) + }); ir_func.ret_is_address = self.types.is_complex(func.return_type) || returns_addr_aggregate; // Add parameters @@ -1741,8 +1988,11 @@ impl<'a> Linearizer<'a> { } } - // Pop function-level scope - self.pop_scope(); + // Pop function-level scope. Its VLA release is a no-op: the body's + // own block scope dropped every mark, and the block is terminated by + // the return above -- which is what must happen, since SSA has + // already run over the function by this point. + self.pop_scope(func_scope); // Add function to module if let Some(ir_func) = self.current_func.take() { @@ -1787,33 +2037,17 @@ impl<'a> Linearizer<'a> { let abi = get_abi_for_conv(self.current_calling_conv, self.target); let ret_class = abi.classify_return(ret_type, self.types); - // An aggregate that is nothing but a `long double` comes back in - // st(0), exactly as the bare scalar does, so the `Ret` carries the - // value's *address* and the backend loads it onto the FPU stack. - // Splitting it across RAX and RDX left the caller reading a slot - // nobody had written. - // A single SSE register carrying sixteen bytes -- an aggregate whose - // sole content is a `__float128` -- is the same shape: the register - // holds the whole value, so the `Ret` carries its address and the - // backend moves all sixteen bytes at once. Splitting it into two - // general registers handed the caller half a value in the wrong place. - let one_sse_reg = matches!( - ret_class, - crate::abi::ArgClass::Direct { ref classes, .. } - if classes.len() == 1 && classes[0] == crate::abi::RegClass::Sse - ); - // A one-element HFA is the aarch64 spelling of the same thing: one V - // register holds the whole value. Splitting it into two general - // registers was survivable on its own -- the backend put the halves - // back together -- but the *inliner* then spliced a two-source `Ret` - // into a caller expecting one value, and the top half came out zero. - // Any HFA, not just a one-element one: the two-element form has the - // same hazard. Its `Ret` carried the halves as two general registers, - // and splicing that into a caller expecting one value dropped the - // second -- an inlined `struct { double a, b; }` return came back with - // its second half zeroed. - let one_hfa_reg = matches!(ret_class, crate::abi::ArgClass::Hfa { .. }); - if matches!(ret_class, crate::abi::ArgClass::X87 { .. }) || one_sse_reg || one_hfa_reg { + // The three classes no pair of general registers can carry: an x87 + // aggregate, an HFA, and sixteen bytes in one SSE register. Each hands + // the value back by address, and splitting any of them into RAX/RDX + // was a miscompile of its own -- an x87 aggregate left the caller + // reading a slot nobody had written, a `__float128` one handed a + // gcc-compiled caller half a value in the wrong place, and an HFA's + // two-source `Ret` spliced into a caller expecting one value dropped + // its second half. `aggregate_ret_is_address` is where that list + // lives, because the inliner has to ask the same question of the + // `Ret` this emits. + if super::aggregate_ret_is_address(&ret_class, struct_size) { let mut ret_insn = Instruction::ret_typed(Some(src_addr), ret_type, struct_size); ret_insn.abi_info = Some(Box::new(CallAbiInfo::new(vec![], ret_class))); self.emit(ret_insn); @@ -1897,14 +2131,15 @@ impl<'a> Linearizer<'a> { | ExprKind::Utf16StringLit(_) | ExprKind::Utf32StringLit(_) => true, - // Identifiers are pure unless volatile - ExprKind::Ident(_) => { - if let Some(typ) = expr.typ { - !self.types.modifiers(typ).contains(TypeModifiers::VOLATILE) - } else { - true - } - } + // Identifiers are pure unless volatile. + // + // `contains_volatile`, not the top-level modifier: reading a + // struct with a `volatile` member reads that member, and asking + // only what was written on the struct answered no. + ExprKind::Ident(_) => match expr.typ { + Some(typ) => !self.types.contains_volatile(typ), + None => true, + }, // __func__ is a pure string-like value ExprKind::FuncName => true, @@ -1951,8 +2186,21 @@ impl<'a> Linearizer<'a> { // Function calls are never pure (may have side effects) ExprKind::Call { .. } => false, - // Member access through struct value (.) is pure if the base is pure. - ExprKind::Member { expr, .. } => self.is_pure_expr(expr), + // Member access through struct value (.) is pure if the base is + // pure and the member itself is not volatile. C17 6.5.15p4 + // evaluates only one arm of a conditional and 5.1.2.3 makes each + // volatile read an observable event, so speculating one is a read + // the program never asked for: asking about the base alone let + // `c ? s.status : s.other` load both members unconditionally into + // a branchless select, at `-O0` too. The member's type carries the + // object's qualifiers (C17 6.5.2.3p3), so this covers a volatile + // member and a member of a volatile object alike. + ExprKind::Member { expr: base, .. } => { + !expr + .typ + .is_some_and(|typ| self.types.contains_volatile(typ)) + && self.is_pure_expr(base) + } // Arrow access (ptr->member) can cause UB/crash if ptr is NULL, // so we must not eagerly evaluate it in conditional expressions. @@ -2766,19 +3014,31 @@ impl<'a> Linearizer<'a> { /// Shared logic for member access (both `.` and `->`). /// `base` is the address of the struct (for `.`) or the pointer value (for `->`). + /// + /// `access_typ` is the type of the member-access *expression*, which the + /// parser formed as the member's declared type so-qualified by the object + /// (C17 6.5.2.3p3/p4). The access is performed at that type, so + /// [`Self::mark_volatile_access`] sees the qualifier -- the type + /// `find_member` answers with is the member's *declared* one and cannot + /// carry it, which is why a member of a `volatile` struct read as an + /// ordinary `int` and DCE deleted the load from `-O1` up. Its width, sign + /// and kind still come from the member, so the two disagreeing (only + /// reachable once the parser has already reported an unknown member) + /// cannot change how the access is performed. It also stands in for the + /// member type entirely when the lookup fails here. pub(crate) fn emit_member_access( &mut self, base: PseudoId, struct_type: TypeId, member: StringId, - fallback_type: TypeId, + access_typ: TypeId, ) -> PseudoId { let member_info = self .types .find_member(struct_type, member) .unwrap_or(MemberInfo { offset: 0, - typ: fallback_type, + typ: access_typ, bit_offset: None, bit_width: None, access_bytes: None, @@ -2813,7 +3073,7 @@ impl<'a> Linearizer<'a> { bit_offset, bit_width, storage_size, - member_info.typ, + access_typ, ) } else { let size = self.types.size_bits(member_info.typ); @@ -2843,7 +3103,7 @@ impl<'a> Linearizer<'a> { result, base, member_info.offset as i64, - member_info.typ, + access_typ, size, )); result @@ -4819,47 +5079,16 @@ impl<'a> Linearizer<'a> { else_expr: &Expr, result_typ: TypeId, ) -> PseudoId { - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - let cond_bool = self.linearize_condition(cond); - let cond_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - let ptr_typ = self.types.pointer_to(result_typ); let ptr_bits = self.target.pointer_width; - - self.switch_bb(then_bb); - let then_val = self.complex_arm_addr(then_expr, result_typ); - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let else_val = self.complex_arm_addr(else_expr, result_typ); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, ptr_typ, ptr_bits); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + ptr_typ, + ptr_bits, + |lin| lin.complex_arm_addr(then_expr, result_typ), + |lin| lin.complex_arm_addr(else_expr, result_typ), + ) } pub(crate) fn linearize_ternary( @@ -4932,48 +5161,23 @@ impl<'a> Linearizer<'a> { )); result } else { - // Impure: use control flow + phi for proper short-circuit evaluation - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - + // Impure: use control flow + phi for proper short-circuit evaluation. + // Each arm is converted inside its own block, where it is the only + // thing evaluated. let cond_bool = self.linearize_condition(cond); - let cond_end_bb = self.current_bb.unwrap(); - - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - - self.switch_bb(then_bb); - let then_val = self.linearize_expr(then_expr); - let then_val = self.conditional_arm(then_val, then_expr, result_typ, aggregate); - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let else_val = self.linearize_expr(else_expr); - let else_val = self.conditional_arm(else_val, else_expr, result_typ, aggregate); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, merge_typ, size); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, merge_typ, size); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, merge_typ, size); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + merge_typ, + size, + |lin| { + let val = lin.linearize_expr(then_expr); + lin.conditional_arm(val, then_expr, result_typ, aggregate) + }, + |lin| { + let val = lin.linearize_expr(else_expr); + lin.conditional_arm(val, else_expr, result_typ, aggregate) + }, + ) } } @@ -5021,50 +5225,21 @@ impl<'a> Linearizer<'a> { self.emit_compare_zero(evaluated, cond_typ) }; - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - let cond_end_bb = self.current_bb.unwrap(); - - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - let ptr_typ = self.types.pointer_to(result_typ); let ptr_bits = self.target.pointer_width; - - self.switch_bb(then_bb); - let then_val = if cond_complex { - self.complex_addr_at_precision(evaluated, cond_typ, result_typ) - } else { - self.promote_real_value_to_complex(evaluated, cond_typ, result_typ) - }; - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let else_val = self.complex_arm_addr(else_expr, result_typ); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, ptr_typ, ptr_bits); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + ptr_typ, + ptr_bits, + |lin| { + if cond_complex { + lin.complex_addr_at_precision(evaluated, cond_typ, result_typ) + } else { + lin.promote_real_value_to_complex(evaluated, cond_typ, result_typ) + } + }, + |lin| lin.complex_arm_addr(else_expr, result_typ), + ) } /// The value of a conditional expression whose constant condition @@ -5133,15 +5308,7 @@ impl<'a> Linearizer<'a> { // Impure right-hand side: it must not be evaluated when the condition // is true, so it needs its own block. - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - let cond_end_bb = self.current_bb.unwrap(); - - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - + // // The true value is the condition, converted to the result type -- done // *inside* the true block, where the ternary also converts its arms. // Converting before the `cbr` is equally correct as IR, and reads more @@ -5150,36 +5317,17 @@ impl<'a> Linearizer<'a> { // and the aarch64 backend then emits a branch on the wrong register. // That is a backend defect and is reported as one; this is not the // place to depend on it. - self.switch_bb(then_bb); - let then_val = self.convert_conditional_arm(cond_val, cond_typ, result_typ); - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let mut else_val = self.linearize_expr(else_expr); - let else_typ = self.expr_type(else_expr); - else_val = self.convert_conditional_arm(else_val, else_typ, result_typ); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, result_typ, size); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, result_typ, size); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, result_typ, size); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + result_typ, + size, + |lin| lin.convert_conditional_arm(cond_val, cond_typ, result_typ), + |lin| { + let val = lin.linearize_expr(else_expr); + let else_typ = lin.expr_type(else_expr); + lin.convert_conditional_arm(val, else_typ, result_typ) + }, + ) } /// Lower `__builtin_clrsb` and its wider siblings. @@ -5690,13 +5838,19 @@ impl<'a> Linearizer<'a> { let value_typ = self.expr_type(val); let raw = self.linearize_expr(val); // Pointer arithmetic scales by the element size, as it does for `+=`. - let operand = if self.types.kind(elem_typ) == TypeKind::Pointer + let is_ptr_arith = self.types.kind(elem_typ) == TypeKind::Pointer && self.types.is_integer(value_typ) - && matches!(op, GnuAtomicOp::Add | GnuAtomicOp::Sub) - { - self.scale_pointer_addend(elem_typ, value_typ, raw) + && matches!(op, GnuAtomicOp::Add | GnuAtomicOp::Sub); + let (operand, operand_typ) = if is_ptr_arith { + ( + self.scale_pointer_addend(elem_typ, value_typ, raw), + self.types.long_id, + ) } else { - self.emit_convert(raw, value_typ, elem_typ) + // Unlike an operator, a builtin converts its value argument to the + // object's type itself (gcc documents the parameter as that type), + // so there is no common type left to compute at. + (self.emit_convert(raw, value_typ, elem_typ), elem_typ) }; // The order argument is accepted and evaluated, as gcc evaluates it, // but every lowering here is sequentially consistent: `emit_atomic_rmw` @@ -5710,39 +5864,30 @@ impl<'a> Linearizer<'a> { size_bits: bits, }; - let (old, binop) = match op { - GnuAtomicOp::Nand => (self.emit_atomic_nand(&lv, operand), Opcode::And), - _ => { - let binop = match op { - GnuAtomicOp::Add => Opcode::Add, - GnuAtomicOp::Sub => Opcode::Sub, - GnuAtomicOp::And => Opcode::And, - GnuAtomicOp::Or => Opcode::Or, - GnuAtomicOp::Xor => Opcode::Xor, - GnuAtomicOp::Nand => unreachable!("handled above"), - }; - (self.emit_atomic_rmw(&lv, binop, operand), binop) - } + // Each builtin is the compound assignment of the same name, so it + // goes through the same model: `nand` is the one that has no operator + // spelling, and it is `&` with the result complemented before it + // converts back to the object's type (`CompoundAssign::invert`). + let assign_op = match op { + GnuAtomicOp::Add => AssignOp::AddAssign, + GnuAtomicOp::Sub => AssignOp::SubAssign, + GnuAtomicOp::And | GnuAtomicOp::Nand => AssignOp::AndAssign, + GnuAtomicOp::Or => AssignOp::OrAssign, + GnuAtomicOp::Xor => AssignOp::XorAssign, + }; + let ca = CompoundAssign { + is_ptr_arith, + invert: op == GnuAtomicOp::Nand, + ..CompoundAssign::new(assign_op, elem_typ, operand_typ) }; + let old = self.emit_atomic_rmw(&lv, &ca, operand); if !returns_new { return old; } - - let new = self.alloc_reg_pseudo(); - self.emit(Instruction::binop(binop, new, old, operand, elem_typ, bits)); - if op != GnuAtomicOp::Nand { - return new; - } - let inverted = self.alloc_reg_pseudo(); - self.emit(Instruction::unop( - Opcode::Not, - inverted, - new, - elem_typ, - bits, - )); - inverted + // The new value, recomputed from the old one rather than read back: + // arithmetic on a value in hand is not a second access to the object. + self.compound_assign_value(&ca, old, operand) } /// `__sync_bool_compare_and_swap` and `__sync_val_compare_and_swap`. @@ -6682,6 +6827,12 @@ impl<'a> Linearizer<'a> { ExprKind::StmtExpr { stmts, result } => { // GNU statement expression: ({ stmt; stmt; expr; }) + // It is a block, so it is a declaration scope like any other: + // its declarations do not outlive it, and the storage of a + // VLA declared in it goes back when it ends. Without the + // scope, `for (...) (void)({ int a[n]; ... });` allocated + // every time round and released nothing. + let scope = self.push_scope(); // Linearize all the statements first for item in stmts { match item { @@ -6689,8 +6840,11 @@ impl<'a> Linearizer<'a> { BlockItem::Statement(s) => self.linearize_stmt(s), } } - // The result is the value of the final expression - self.linearize_expr(result) + // The result is the value of the final expression, computed + // before the scope ends: it may read the VLA being released. + let value = self.linearize_expr(result); + self.pop_scope(scope); + value } ExprKind::BuiltinComplex { real, imag } => { diff --git a/cc/ir/linearize_atomic.rs b/cc/ir/linearize_atomic.rs index 933c7bea7..ee2ec0449 100644 --- a/cc/ir/linearize_atomic.rs +++ b/cc/ir/linearize_atomic.rs @@ -20,6 +20,7 @@ // use super::linearize::Linearizer; +use super::linearize_emit::{compound_assign_arith_type, compound_assign_opcode, CompoundAssign}; use super::{Instruction, MemoryOrder, Opcode, PseudoId}; use crate::diag; use crate::float::FloatVal; @@ -210,8 +211,12 @@ impl Linearizer<'_> { /// /// Multiplication, division, remainder, the shifts and everything /// floating-point have no native atomic form and go through a - /// compare-and-swap retry loop instead. - pub(crate) fn atomic_opcode_for(op: Opcode) -> Option { + /// compare-and-swap retry loop instead. Neither does `nand`, on any + /// target, which is why it is a flag on [`CompoundAssign`] rather than an + /// opcode here. + /// + /// Having one does not by itself make it usable: see `native_rmw_opcode`. + fn atomic_opcode_for(op: Opcode) -> Option { Some(match op { Opcode::Add => Opcode::AtomicFetchAdd, Opcode::Sub => Opcode::AtomicFetchSub, @@ -249,39 +254,53 @@ impl Linearizer<'_> { pub(crate) fn emit_atomic_rmw( &mut self, lv: &AtomicLvalue, - op: Opcode, + ca: &CompoundAssign, value: PseudoId, ) -> PseudoId { - // `_Bool` cannot use a native fetch-and-op: the value stored must be - // the *converted* result, so `b = 1; b++` leaves 1 rather than 2, and - // `b = 0; b--` leaves 1 rather than 255. Only the CAS loop can apply - // that conversion before the store. - let needs_conversion = self.types.kind(lv.elem_typ) == TypeKind::Bool; - if !needs_conversion { - if let Some(atomic_op) = Self::atomic_opcode_for(op) { - return self.emit_atomic_op(atomic_op, lv, Some(value)); - } + if let Some(atomic_op) = self.native_rmw_opcode(ca) { + // The instruction computes at the object's own width, so the + // operand arrives at the object's own type -- the truncation the + // congruence below permits. Pointer arithmetic has scaled it to a + // byte count already, at pointer width. + let value = if ca.is_ptr_arith { + value + } else { + self.emit_convert(value, ca.value_typ, ca.target_typ) + }; + return self.emit_atomic_op(atomic_op, lv, Some(value)); } - self.emit_atomic_cas_loop(lv, op, value, false) + self.emit_atomic_cas_loop(lv, ca, value) } - /// `__atomic_fetch_nand` / `__sync_fetch_and_nand`: store `~(old & value)` - /// and return the old value. + /// The native atomic instruction that computes `ca` *exactly*, if one + /// does. + /// + /// `AtomicFetchAdd` and its siblings operate at the object's width and + /// store the raw result, where C17 6.5.16.2p3 computes at the operands' + /// common type and converts the result back + /// ([`Linearizer::compound_assign_value`]). The two agree when the + /// operator is congruent modulo 2^n -- add, subtract and the three bitwise + /// ops -- *and* the conversion back is the truncation congruence permits. /// - /// No target has an atomic NAND, so this is always the CAS loop -- the - /// same loop, with one more instruction inside it. Writing a second loop - /// would mean two places to get the LL/SC rules right. - pub(crate) fn emit_atomic_nand(&mut self, lv: &AtomicLvalue, value: PseudoId) -> PseudoId { - self.emit_atomic_cas_loop(lv, Opcode::And, value, true) + /// `_Bool` is where that second condition fails: converting to it is a + /// test against zero, not a truncation, so `b -= 1` must store 1 and only + /// the CAS loop can convert before the store. Divide, remainder, the + /// shifts and everything floating-point fail the first condition, and + /// `nand` complements a value the hardware would store as it stands. + fn native_rmw_opcode(&self, ca: &CompoundAssign) -> Option { + if ca.invert || self.types.kind(ca.target_typ) == TypeKind::Bool { + return None; + } + let arith_type = compound_assign_arith_type(self.types, ca); + Self::atomic_opcode_for(compound_assign_opcode(self.types, ca.op, arith_type)) } /// The CAS retry loop described on `emit_atomic_rmw`. fn emit_atomic_cas_loop( &mut self, lv: &AtomicLvalue, - op: Opcode, + ca: &CompoundAssign, value: PseudoId, - invert: bool, ) -> PseudoId { let elem_typ = lv.elem_typ; let bits = lv.size_bits; @@ -298,39 +317,25 @@ impl Linearizer<'_> { let loop_bb = self.alloc_bb(); let done_bb = self.alloc_bb(); - let entry_bb = self.current_bb.expect("atomic RMW outside a block"); + // Through the accessor rather than `expect`: `current_bb` is `None` + // wherever control cannot arrive -- a statement before a `switch`'s + // first `case`, or after a `goto` -- and this loop has to hang its + // blocks off something. `dce::remove_unreachable_blocks` takes the + // lot away again. + let entry_bb = self.current_or_unreachable_bb(); self.emit(Instruction::br(loop_bb)); self.link_bb(entry_bb, loop_bb); self.switch_bb(loop_bb); - // old = *exp; new = old value + // old = *exp; new = the value `old value` assigns. + // + // Through the shared helper, so the loop computes at the same type the + // ordinary lowering does and converts the result back the same way -- + // which is also what keeps an `_Atomic _Bool` holding 0 or 1 rather + // than the raw 2 or 255 the arithmetic produced (C17 6.3.1.2). let old = self.alloc_reg_pseudo(); self.emit(Instruction::load(old, exp_addr, 0, elem_typ, bits)); - let new = self.alloc_reg_pseudo(); - self.emit(Instruction::binop(op, new, old, value, elem_typ, bits)); - // `nand` is `and` with the result complemented, which is the only - // reason this loop takes a flag rather than an opcode alone. - let new = if invert { - let inverted = self.alloc_reg_pseudo(); - self.emit(Instruction::unop( - Opcode::Not, - inverted, - new, - elem_typ, - bits, - )); - inverted - } else { - new - }; - // C17 6.3.1.2: converting to _Bool yields 0 or 1, and a compound - // assignment stores the converted result. Without this the raw sum - // reaches memory and an _Atomic _Bool holds 2 or 255. - let new = if self.types.kind(elem_typ) == TypeKind::Bool { - self.emit_convert(new, self.types.int_id, elem_typ) - } else { - new - }; + let new = self.compound_assign_value(ca, old, value); let ok = self.alloc_reg_pseudo(); let order = self.emit_const(ORDER as i128, self.types.int_id); @@ -341,7 +346,7 @@ impl Linearizer<'_> { cas.memory_order = ORDER; self.emit(cas); - let cas_bb = self.current_bb.expect("atomic CAS outside a block"); + let cas_bb = self.current_or_unreachable_bb(); self.emit(Instruction::cbr(ok, done_bb, loop_bb)); self.link_bb(cas_bb, done_bb); self.link_bb(cas_bb, loop_bb); @@ -387,26 +392,32 @@ impl Linearizer<'_> { let is_ptr_arith = self.types.kind(target_typ) == TypeKind::Pointer && self.types.is_integer(value_typ) && matches!(op, AssignOp::AddAssign | AssignOp::SubAssign); - - let (operand, arith_typ) = if is_ptr_arith { + let (operand, value_typ) = if is_ptr_arith { ( self.scale_pointer_addend(target_typ, value_typ, rhs), - target_typ, + self.types.long_id, ) } else { - (self.emit_convert(rhs, value_typ, target_typ), target_typ) + (rhs, value_typ) }; - let opcode = self.compound_assign_opcode(op, arith_typ); - let old = self.emit_atomic_rmw(&lv, opcode, operand); - - // Recompute the stored value from the old one. - let new = self.alloc_reg_pseudo(); - let bits = self.types.size_bits(arith_typ); - self.emit(Instruction::binop( - opcode, new, old, operand, arith_typ, bits, - )); - Some(new) + // The right operand goes on at its own type. Converting it down to the + // target here -- which this did -- computes `50 / (unsigned char)-5` + // where C17 6.5.16.2p3 computes `50 / -5` at `int` and converts only + // the result; `compound_assign_value` is the ordinary path's rule, now + // shared rather than copied. + let ca = CompoundAssign { + is_ptr_arith, + ..CompoundAssign::new(op, lv.elem_typ, value_typ) + }; + let old = self.emit_atomic_rmw(&lv, &ca, operand); + + // Recompute the stored value from the old one, by the same rule that + // stored it: C17 6.5.16p3 gives the expression the left operand's + // value *after* the assignment, which for `_Atomic _Bool b = 0` makes + // `(b -= 1)` the 1 that reached memory and not the 255 the subtraction + // produced. + Some(self.compound_assign_value(&ca, old, operand)) } /// Scale an integer addend by the pointee size, for `p += n`. @@ -431,63 +442,6 @@ impl Linearizer<'_> { )); scaled } - - /// The arithmetic opcode a compound assignment operator applies. - pub(crate) fn compound_assign_opcode(&self, op: AssignOp, typ: TypeId) -> Opcode { - let is_float = self.types.is_float(typ); - let is_unsigned = self.types.is_unsigned(typ); - match op { - AssignOp::Assign => unreachable!("plain assignment has no arithmetic opcode"), - AssignOp::AddAssign => { - if is_float { - Opcode::FAdd - } else { - Opcode::Add - } - } - AssignOp::SubAssign => { - if is_float { - Opcode::FSub - } else { - Opcode::Sub - } - } - AssignOp::MulAssign => { - if is_float { - Opcode::FMul - } else { - Opcode::Mul - } - } - AssignOp::DivAssign => { - if is_float { - Opcode::FDiv - } else if is_unsigned { - Opcode::DivU - } else { - Opcode::DivS - } - } - AssignOp::ModAssign => { - if is_unsigned { - Opcode::ModU - } else { - Opcode::ModS - } - } - AssignOp::AndAssign => Opcode::And, - AssignOp::OrAssign => Opcode::Or, - AssignOp::XorAssign => Opcode::Xor, - AssignOp::ShlAssign => Opcode::Shl, - AssignOp::ShrAssign => { - if is_unsigned { - Opcode::Lsr - } else { - Opcode::Asr - } - } - } - } } impl Linearizer<'_> { @@ -511,30 +465,33 @@ impl Linearizer<'_> { let lv = self.atomic_lvalue(operand)?; - // A pointer steps by one element; everything else by one. + // A pointer steps by one element; everything else by one. C17 + // 6.5.3.1p2 defines `++E` as `E += 1`, so the step is the right + // operand of a compound assignment and the same helper applies -- + // including its conversion of the result, which is what makes `++b` on + // an `_Atomic _Bool` still yield 0 or 1. + let is_ptr_arith = self.types.kind(typ) == TypeKind::Pointer; let delta = self.incdec_delta(typ); - let is_float = self.types.is_float(typ); - let opcode = match (is_inc, is_float) { - (true, false) => Opcode::Add, - (false, false) => Opcode::Sub, - (true, true) => Opcode::FAdd, - (false, true) => Opcode::FSub, + let delta_typ = if is_ptr_arith { + self.types.long_id + } else { + typ + }; + let op = if is_inc { + AssignOp::AddAssign + } else { + AssignOp::SubAssign + }; + let ca = CompoundAssign { + is_ptr_arith, + ..CompoundAssign::new(op, lv.elem_typ, delta_typ) }; - let old = self.emit_atomic_rmw(&lv, opcode, delta); + let old = self.emit_atomic_rmw(&lv, &ca, delta); if !prefix { return Some(old); } - - let bits = self.types.size_bits(typ); - let new = self.alloc_reg_pseudo(); - self.emit(Instruction::binop(opcode, new, old, delta, typ, bits)); - - // `++b` on an _Atomic _Bool must still yield 0 or 1. - if self.types.kind(typ) == TypeKind::Bool { - return Some(self.emit_convert(new, self.types.int_id, typ)); - } - Some(new) + Some(self.compound_assign_value(&ca, old, delta)) } /// The amount `++`/`--` steps by: the pointee size for a pointer, else 1. diff --git a/cc/ir/linearize_emit.rs b/cc/ir/linearize_emit.rs index 95624190d..683198818 100644 --- a/cc/ir/linearize_emit.rs +++ b/cc/ir/linearize_emit.rs @@ -16,7 +16,7 @@ use crate::diag::{error, Position}; use crate::float::FloatVal; use crate::parse::ast::{AssignOp, BinaryOp, Expr, ExprKind, FpCompare, LibFn, MathErrno, UnaryOp}; use crate::strings::StringId; -use crate::types::{MemberInfo, TypeId, TypeKind}; +use crate::types::{MemberInfo, TypeId, TypeKind, TypeTable}; /// A read-modify-write target whose address has been computed **once**. /// @@ -38,6 +38,141 @@ pub(crate) struct RmwPlace { bitfield: Option<(usize, u32, u32, u32, TypeId)>, } +/// Everything `E1 op= E2` needs beyond the two operand *values*: the operator +/// and the types the arithmetic is decided by. +/// +/// C17 6.5.16.2p3 makes `E1 op= E2` mean `E1 = E1 op E2` bar evaluating `E1` +/// twice. So the arithmetic runs at the type the usual arithmetic conversions +/// give the two operands -- `target_typ` and `value_typ` -- and only the +/// *result* converts back to `target_typ`. Narrowing the right operand to the +/// target first is a different computation: `_Atomic unsigned char c = 50; +/// c /= -5;` becomes `50 / 251` and stores 0 where the standard stores +/// `(unsigned char)(50 / -5)`, 246. +/// +/// This exists because the ordinary and the `_Atomic` lowerings each had their +/// own copy of these rules and the copies disagreed. Both now build one of +/// these and hand it to [`Linearizer::compound_assign_value`]. +#[derive(Clone, Copy)] +pub(crate) struct CompoundAssign { + /// The operator. + pub(crate) op: AssignOp, + /// `E1`'s type: what the left operand is read at and what the result + /// converts back to. + pub(crate) target_typ: TypeId, + /// `E2`'s type **as written**, before any conversion. For pointer + /// arithmetic it is the type of the already-scaled addend. + pub(crate) value_typ: TypeId, + /// `p += n`: the right operand arrives already scaled by the pointee size, + /// the addition happens at pointer width, and the result is a pointer + /// already -- so neither operand nor result is converted. + pub(crate) is_ptr_arith: bool, + /// Complement the arithmetic result *before* it converts back to + /// `target_typ`. This is `nand`, which has no operator spelling of its + /// own: `__atomic_fetch_nand` stores `~(old & value)`, and on a `_Bool` + /// object that complement has to happen before the conversion to 0 or 1, + /// not after it. + pub(crate) invert: bool, +} + +impl CompoundAssign { + /// `E1 op= E2` with both operand types as written. + pub(crate) fn new(op: AssignOp, target_typ: TypeId, value_typ: TypeId) -> Self { + Self { + op, + target_typ, + value_typ, + is_ptr_arith: false, + invert: false, + } + } +} + +/// The type the arithmetic of `ca` is performed at. +/// +/// The usual arithmetic conversions (C17 6.3.1.8) on the two operands, with +/// two exceptions: +/// +/// * The shifts. C17 6.5.7p3 promotes each operand *separately* and gives the +/// result the promoted **left** operand's type, so the right operand has no +/// say: `_Atomic signed char s = -8; s >>= 1;` shifts -8 as an `int` and +/// stores -4, where computing at the target's width would shift the byte +/// pattern and store 124. +/// * Pointer arithmetic, whose addend the caller has already scaled to a +/// byte count; the addition happens at pointer width. +pub(crate) fn compound_assign_arith_type(types: &TypeTable, ca: &CompoundAssign) -> TypeId { + if ca.is_ptr_arith { + types.long_id + } else if matches!(ca.op, AssignOp::ShlAssign | AssignOp::ShrAssign) { + types.integer_promote(ca.target_typ) + } else { + types.common_type(ca.target_typ, ca.value_typ) + } +} + +/// The arithmetic opcode a compound assignment operator applies at `typ`. +/// +/// `typ` is the type the operation is *performed* at -- the answer of +/// [`compound_assign_arith_type`] -- because that is what decides between the +/// integer and floating forms and between the signed and unsigned ones. Asking +/// the target's type instead makes `unsigned char x; x /= -5;` an unsigned +/// divide of a value the standard computes as a signed `int`. +pub(crate) fn compound_assign_opcode(types: &TypeTable, op: AssignOp, typ: TypeId) -> Opcode { + let is_float = types.is_float(typ); + let is_unsigned = types.is_unsigned(typ); + match op { + AssignOp::Assign => unreachable!("plain assignment has no arithmetic opcode"), + AssignOp::AddAssign => { + if is_float { + Opcode::FAdd + } else { + Opcode::Add + } + } + AssignOp::SubAssign => { + if is_float { + Opcode::FSub + } else { + Opcode::Sub + } + } + AssignOp::MulAssign => { + if is_float { + Opcode::FMul + } else { + Opcode::Mul + } + } + AssignOp::DivAssign => { + if is_float { + Opcode::FDiv + } else if is_unsigned { + Opcode::DivU + } else { + Opcode::DivS + } + } + // Modulo not supported for floats. + AssignOp::ModAssign => { + if is_unsigned { + Opcode::ModU + } else { + Opcode::ModS + } + } + AssignOp::AndAssign => Opcode::And, + AssignOp::OrAssign => Opcode::Or, + AssignOp::XorAssign => Opcode::Xor, + AssignOp::ShlAssign => Opcode::Shl, + AssignOp::ShrAssign => { + if is_unsigned { + Opcode::Lsr + } else { + Opcode::Asr + } + } + } +} + /// The per-half opcodes a complex operation uses, chosen once from the base /// type rather than at each emit site. /// @@ -190,66 +325,47 @@ impl<'a> super::linearize::Linearizer<'a> { } } - /// Emit stores to zero-initialize an aggregate (struct, union, or array) - /// This handles C99 6.7.8p19: uninitialized members must be zero-initialized + /// Zero a whole aggregate (struct, union or array), which is what C17 + /// 6.7.9p19 asks for before an initializer list is applied: every member + /// the list does not reach is initialized as a static object would be. pub(crate) fn emit_aggregate_zero(&mut self, base_sym: PseudoId, typ: TypeId) { - let total_bytes = self.types.size_bytes(typ); - let mut offset: i64 = 0; - - // Create a zero constant for 64-bit stores - let zero64 = self.emit_const(0, self.types.long_id); - - // Zero in 8-byte chunks - while offset + 8 <= total_bytes as i64 { - self.emit(Instruction::store( - zero64, - base_sym, - offset, - self.types.long_id, - 64, - )); - offset += 8; - } + let total_bytes = self.types.size_bytes(typ) as i64; + self.emit_block_zero(base_sym, 0, total_bytes); + } - // Handle remaining bytes (if any) - if offset < total_bytes as i64 { - let remaining = total_bytes as i64 - offset; - if remaining >= 4 { - let zero32 = self.emit_const(0, self.types.int_id); - self.emit(Instruction::store( - zero32, - base_sym, - offset, - self.types.int_id, - 32, - )); - offset += 4; - } - if offset < total_bytes as i64 { - let remaining = total_bytes as i64 - offset; - if remaining >= 2 { - let zero16 = self.emit_const(0, self.types.short_id); - self.emit(Instruction::store( - zero16, - base_sym, - offset, - self.types.short_id, - 16, - )); - offset += 2; - } - if offset < total_bytes as i64 { - let zero8 = self.emit_const(0, self.types.char_id); - self.emit(Instruction::store( - zero8, - base_sym, - offset, - self.types.char_id, - 8, - )); - } - } + /// Emit a fill of `size_bytes` zero bytes at `dst` + `dst_base_offset`. + /// + /// One `Opcode::Memset`, which `memexpand` turns into stores when the run + /// is short and leaves as a call when it is not. So the bound on the + /// unroll is `memexpand::INLINE_LIMIT_BYTES` -- the one every block memory + /// operation in the compiler shares -- rather than another copy of the + /// 8/4/2/1 descent with a cap of its own, which is what this was: a + /// hand-rolled ladder with **no** upper bound at all, so + /// `char buf[N] = {0}` emitted one store per chunk for any N. 8 KB cost + /// 2081 instructions in the function body and 1 MB did not finish + /// compiling in 25 minutes, while its sibling + /// [`Self::emit_block_copy_at_offset`] had capped at the shared limit all + /// along. + /// + /// `memexpand::run` runs at every optimization level, `-O0` included, so + /// the expansion does not depend on optimizing -- the same reason the + /// opcode a program's own `memset` becomes is expanded there rather than + /// here. + pub(crate) fn emit_block_zero(&mut self, dst: PseudoId, dst_base_offset: i64, size_bytes: i64) { + if size_bytes <= 0 { + return; } + let dst_ptr = self.block_dest_addr(dst, dst_base_offset); + let byte = self.emit_const(0, self.types.int_id); + let n = self.emit_const(size_bytes as i128, self.types.ulong_id); + let result = self.alloc_pseudo(); + self.emit( + Instruction::new(Opcode::Memset) + .with_func(self.library_function_name("memset")) + .with_target(result) + .with_src3(dst_ptr, byte, n) + .with_type_and_size(self.types.void_ptr_id, 64), + ); } /// Emit a block copy from src to dst using integer chunks. @@ -288,10 +404,37 @@ impl<'a> super::linearize::Linearizer<'a> { } } - /// The same copy as a `memcpy` call. + /// The address `dst` + `dst_base_offset` names, as a block memory + /// operation takes it. + /// + /// These opcodes take addresses. A `Sym` pseudo names a local's *storage*, + /// not a pointer to it -- a `Store` can name it directly, a call cannot. + /// Passing the Sym itself handed `memcpy` a meaningless value and + /// segfaulted every copy over the threshold. `rvalue_addr` is the existing + /// answer to this question and returns a non-Sym pseudo unchanged. /// - /// `dst_base_offset` is folded into the destination pointer first, since - /// `memcpy` takes an address rather than a base and a displacement. + /// `dst_base_offset` is folded into the pointer, since these take an + /// address rather than a base and a displacement. + fn block_dest_addr(&mut self, dst: PseudoId, dst_base_offset: i64) -> PseudoId { + let void_ptr = self.types.void_ptr_id; + let dst = self.rvalue_addr(dst, void_ptr); + if dst_base_offset == 0 { + return dst; + } + let off = self.emit_const(dst_base_offset as i128, self.types.long_id); + let adjusted = self.alloc_reg_pseudo(); + self.emit(Instruction::binop( + Opcode::Add, + adjusted, + dst, + off, + void_ptr, + 64, + )); + adjusted + } + + /// The same copy as a `memcpy` call. fn emit_block_copy_call( &mut self, dst: PseudoId, @@ -299,31 +442,8 @@ impl<'a> super::linearize::Linearizer<'a> { src: PseudoId, size_bytes: i64, ) { - // `memcpy` takes addresses. A `Sym` pseudo names a local's *storage*, - // not a pointer to it -- the inline path could store through it - // directly, this one cannot. Passing the Sym itself handed memcpy a - // meaningless value and segfaulted every copy over the threshold. - // `rvalue_addr` is the existing answer to this question and returns a - // non-Sym pseudo unchanged. - let void_ptr = self.types.void_ptr_id; - let dst = self.rvalue_addr(dst, void_ptr); - let src = self.rvalue_addr(src, void_ptr); - - let dst_ptr = if dst_base_offset == 0 { - dst - } else { - let off = self.emit_const(dst_base_offset as i128, self.types.long_id); - let adjusted = self.alloc_reg_pseudo(); - self.emit(Instruction::binop( - Opcode::Add, - adjusted, - dst, - off, - self.types.void_ptr_id, - 64, - )); - adjusted - }; + let dst_ptr = self.block_dest_addr(dst, dst_base_offset); + let src = self.rvalue_addr(src, self.types.void_ptr_id); let n = self.emit_const(size_bytes as i128, self.types.ulong_id); let result = self.alloc_pseudo(); self.emit( @@ -337,6 +457,15 @@ impl<'a> super::linearize::Linearizer<'a> { /// Emit code to load a bitfield value /// Returns the loaded value as a PseudoId + /// + /// `typ` is the field's type as the access reaches it -- the declared type + /// so-qualified by the object (C17 6.5.2.3p3) -- and the access itself is + /// of the *carrier*, whose type is an unqualified storage unit. So nothing + /// downstream can derive the qualifier from the instruction's own type, and + /// the volatile marker is set here instead; + /// [`Self::mark_volatile_access`] preserves a marker its caller set for + /// exactly this case. Without it a volatile bit-field read was deleted + /// outright from `-O1` up. pub(crate) fn emit_bitfield_load( &mut self, base: PseudoId, @@ -370,16 +499,20 @@ impl<'a> super::linearize::Linearizer<'a> { // Determine storage type based on storage unit size let storage_type = self.bitfield_storage_type(storage_size as usize); let storage_bits = storage_size * 8; + let volatile = self.types.contains_volatile(typ); // 1. Load the entire storage unit let storage_val = self.alloc_pseudo(); - self.emit(Instruction::load( - storage_val, - base, - byte_offset as i64, - storage_type, - storage_bits, - )); + self.emit( + Instruction::load( + storage_val, + base, + byte_offset as i64, + storage_type, + storage_bits, + ) + .with_volatile(volatile), + ); // 2. Shift right by bit_offset (using logical shift for unsigned extraction) let shifted = if bit_offset > 0 { @@ -528,6 +661,8 @@ impl<'a> super::linearize::Linearizer<'a> { }; let carrier_bits = if wide { 64 } else { 32 }; let byte_type = self.types.uchar_id; + // Every byte of a volatile field is part of the one observable read. + let volatile = self.types.contains_volatile(typ); let mut acc: Option = None; let (field_lo, field_hi) = (bit_offset, bit_offset + bit_width); @@ -542,13 +677,10 @@ impl<'a> super::linearize::Linearizer<'a> { continue; } let byte = self.alloc_pseudo(); - self.emit(Instruction::load( - byte, - base, - (byte_offset + i as usize) as i64, - byte_type, - 8, - )); + self.emit( + Instruction::load(byte, base, (byte_offset + i as usize) as i64, byte_type, 8) + .with_volatile(volatile), + ); // Widen before shifting, or the shift is done at eight bits and // drops everything it moves. `uchar` is unsigned, so this is a // zero-extension and the byte's own value is preserved. @@ -662,6 +794,14 @@ impl<'a> super::linearize::Linearizer<'a> { } /// Emit code to store a value into a bitfield + /// + /// `typ` is the field's type as the access reaches it, and is here for the + /// same reason as in [`Self::emit_bitfield_load`]: the read-modify-write is + /// performed on the *carrier*, so the instructions cannot show the + /// qualifier and the marker is set from the field's type instead. Both + /// halves are marked -- the read of the storage unit is as observable as + /// the write of it. + #[allow(clippy::too_many_arguments)] pub(crate) fn emit_bitfield_store( &mut self, base: PseudoId, @@ -670,6 +810,7 @@ impl<'a> super::linearize::Linearizer<'a> { bit_width: u32, storage_size: u32, new_value: PseudoId, + typ: TypeId, ) { if !matches!(storage_size, 1 | 2 | 4 | 8 | 16) { return self.emit_bitfield_store_bytewise( @@ -679,22 +820,27 @@ impl<'a> super::linearize::Linearizer<'a> { bit_width, storage_size, new_value, + typ, ); } // Determine storage type based on storage unit size let storage_type = self.bitfield_storage_type(storage_size as usize); let storage_bits = storage_size * 8; + let volatile = self.types.contains_volatile(typ); // 1. Load current storage unit value let old_val = self.alloc_pseudo(); - self.emit(Instruction::load( - old_val, - base, - byte_offset as i64, - storage_type, - storage_bits, - )); + self.emit( + Instruction::load( + old_val, + base, + byte_offset as i64, + storage_type, + storage_bits, + ) + .with_volatile(volatile), + ); // 2. Create mask for the bitfield bits: ~(((1 << width) - 1) << offset) // @@ -757,13 +903,16 @@ impl<'a> super::linearize::Linearizer<'a> { )); // 6. Store back - self.emit(Instruction::store( - combined, - base, - byte_offset as i64, - storage_type, - storage_bits, - )); + self.emit( + Instruction::store( + combined, + base, + byte_offset as i64, + storage_type, + storage_bits, + ) + .with_volatile(volatile), + ); } /// Write a bit-field occupying an arbitrary byte range, one byte at a time. @@ -773,6 +922,7 @@ impl<'a> super::linearize::Linearizer<'a> { /// bits survive. Neither ever touches a byte outside the field's own span, /// which is what a wide read-modify-write could not promise: the span may /// end at the last byte of the object. + #[allow(clippy::too_many_arguments)] fn emit_bitfield_store_bytewise( &mut self, base: PseudoId, @@ -781,6 +931,7 @@ impl<'a> super::linearize::Linearizer<'a> { bit_width: u32, span: u32, new_value: PseudoId, + typ: TypeId, ) { let wide = bit_offset + bit_width > 32; let carrier = if wide { @@ -790,6 +941,7 @@ impl<'a> super::linearize::Linearizer<'a> { }; let carrier_bits = if wide { 64 } else { 32 }; let byte_type = self.types.uchar_id; + let volatile = self.types.contains_volatile(typ); // The value, masked to its width once, so no byte can contribute bits // the field does not have. @@ -861,7 +1013,9 @@ impl<'a> super::linearize::Linearizer<'a> { placed } else { let old = self.alloc_pseudo(); - self.emit(Instruction::load(old, base, addr_off, byte_type, 8)); + self.emit( + Instruction::load(old, base, addr_off, byte_type, 8).with_volatile(volatile), + ); let keep = self.emit_const((!byte_mask & 0xff) as i128, byte_type); let cleared = self.alloc_pseudo(); self.emit(Instruction::binop( @@ -896,7 +1050,9 @@ impl<'a> super::linearize::Linearizer<'a> { )); out }; - self.emit(Instruction::store(to_store, base, addr_off, byte_type, 8)); + self.emit( + Instruction::store(to_store, base, addr_off, byte_type, 8).with_volatile(volatile), + ); } } @@ -1575,61 +1731,53 @@ impl<'a> super::linearize::Linearizer<'a> { }; let cond = self.emit_int_binop(cmp, abs_c, abs_d, base_typ, base_size); - let small_bb = self.alloc_bb(); - let big_bb = self.alloc_bb(); - let done_bb = self.alloc_bb(); - let entry_bb = self.current_bb.expect("complex divide outside a block"); - self.emit(Instruction::cbr(cond, small_bb, big_bb)); - self.link_bb(entry_bb, small_bb); - self.link_bb(entry_bb, big_bb); - - // `|c| >= |d|`: r = d/c, denom = c + d*r, - // re = (a + b*r)/denom, im = (b - a*r)/denom. - self.switch_bb(big_bb); - let r = self.emit_int_binop(div, d, c, base_typ, base_size); - let dr = self.emit_int_binop(Opcode::Mul, d, r, base_typ, base_size); - let denom = self.emit_int_binop(Opcode::Add, c, dr, base_typ, base_size); - let br = self.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); - let num_re = self.emit_int_binop(Opcode::Add, a, br, base_typ, base_size); - let ar = self.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); - let num_im = self.emit_int_binop(Opcode::Sub, b, ar, base_typ, base_size); - self.store_complex_quotient( - result_addr, - (num_re, num_im), - denom, - div, - base_typ, - base_size, - base_bytes, - ); - let big_end = self.current_bb.expect("complex divide lost its block"); - self.emit(Instruction::br(done_bb)); - self.link_bb(big_end, done_bb); - - // `|c| < |d|`: r = c/d, denom = d + c*r, - // re = (a*r + b)/denom, im = (b*r - a)/denom. - self.switch_bb(small_bb); - let r = self.emit_int_binop(div, c, d, base_typ, base_size); - let cr = self.emit_int_binop(Opcode::Mul, c, r, base_typ, base_size); - let denom = self.emit_int_binop(Opcode::Add, d, cr, base_typ, base_size); - let ar = self.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); - let num_re = self.emit_int_binop(Opcode::Add, ar, b, base_typ, base_size); - let br = self.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); - let num_im = self.emit_int_binop(Opcode::Sub, br, a, base_typ, base_size); - self.store_complex_quotient( - result_addr, - (num_re, num_im), - denom, - div, - base_typ, - base_size, - base_bytes, + // Both arms write their halves to `result_addr`, so there is no value + // to merge and the void diamond serves: it builds the same blocks and + // edges, and reads `current_bb` back through the accessor that copes + // with a `goto` out of an arm. + self.emit_diamond_void( + cond, + // `|c| < |d|`: r = c/d, denom = d + c*r, + // re = (a*r + b)/denom, im = (b*r - a)/denom. + |lin| { + let r = lin.emit_int_binop(div, c, d, base_typ, base_size); + let cr = lin.emit_int_binop(Opcode::Mul, c, r, base_typ, base_size); + let denom = lin.emit_int_binop(Opcode::Add, d, cr, base_typ, base_size); + let ar = lin.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); + let num_re = lin.emit_int_binop(Opcode::Add, ar, b, base_typ, base_size); + let br = lin.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); + let num_im = lin.emit_int_binop(Opcode::Sub, br, a, base_typ, base_size); + lin.store_complex_quotient( + result_addr, + (num_re, num_im), + denom, + div, + base_typ, + base_size, + base_bytes, + ); + }, + // `|c| >= |d|`: r = d/c, denom = c + d*r, + // re = (a + b*r)/denom, im = (b - a*r)/denom. + |lin| { + let r = lin.emit_int_binop(div, d, c, base_typ, base_size); + let dr = lin.emit_int_binop(Opcode::Mul, d, r, base_typ, base_size); + let denom = lin.emit_int_binop(Opcode::Add, c, dr, base_typ, base_size); + let br = lin.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); + let num_re = lin.emit_int_binop(Opcode::Add, a, br, base_typ, base_size); + let ar = lin.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); + let num_im = lin.emit_int_binop(Opcode::Sub, b, ar, base_typ, base_size); + lin.store_complex_quotient( + result_addr, + (num_re, num_im), + denom, + div, + base_typ, + base_size, + base_bytes, + ); + }, ); - let small_end = self.current_bb.expect("complex divide lost its block"); - self.emit(Instruction::br(done_bb)); - self.link_bb(small_end, done_bb); - - self.switch_bb(done_bb); } /// Divide both numerators by the shared denominator and store the halves. @@ -1811,6 +1959,10 @@ impl<'a> super::linearize::Linearizer<'a> { /// `cond ? taken() : fallthrough()`, each arm in a block of its own and /// merged by a phi of type `typ`: for arms that may not both be /// evaluated. + /// + /// The phi is as wide as `typ` is. A construct whose merge width is not + /// its type's -- a pointer-merged complex arm, a function designator, + /// whose `size_bits` is zero -- calls [`Self::emit_diamond`] and states it. fn emit_two_way( &mut self, cond: PseudoId, @@ -1819,16 +1971,33 @@ impl<'a> super::linearize::Linearizer<'a> { fallthrough: impl FnOnce(&mut Self) -> PseudoId, ) -> PseudoId { let size = self.types.size_bits(typ); - let (taken_bb, fall_bb, merge_bb) = (self.alloc_bb(), self.alloc_bb(), self.alloc_bb()); - let from = self.current_or_unreachable_bb(); - self.emit(Instruction::cbr(cond, taken_bb, fall_bb)); - self.link_bb(from, taken_bb); - self.link_bb(from, fall_bb); + self.emit_diamond(cond, typ, size, taken, fallthrough) + } - let arms = [ - self.emit_arm(taken_bb, merge_bb, taken), - self.emit_arm(fall_bb, merge_bb, fallthrough), - ]; + /// `cond ? then_arm() : else_arm()`, merged by a `size`-bit phi of type + /// `typ`. + /// + /// The one place a two-armed conditional's blocks and edges are built, so + /// the one place that has to know `current_bb` is `None` wherever control + /// cannot arrive -- see [`Linearizer::current_or_unreachable_bb`]. Both + /// the block the branch leaves and the block each arm *ends* in are read + /// back through that accessor: an arm is arbitrary code and may itself + /// `goto` away, so `x ? ({ goto L; g(); }) : g()` has no block at the end + /// of its true arm. + /// + /// `size` is passed rather than taken from `typ` because the two are not + /// always the same: a complex conditional merges *addresses*, so its phi + /// is pointer-wide over a pointer type, and a function designator's + /// `size_bits` is 0 where the merge wants a pointer's 64. + pub(crate) fn emit_diamond( + &mut self, + cond: PseudoId, + typ: TypeId, + size: u32, + then_arm: impl FnOnce(&mut Self) -> PseudoId, + else_arm: impl FnOnce(&mut Self) -> PseudoId, + ) -> PseudoId { + let (merge_bb, arms) = self.emit_fork(cond, then_arm, else_arm); self.switch_bb(merge_bb); let result = self.alloc_pseudo(); @@ -1844,14 +2013,52 @@ impl<'a> super::linearize::Linearizer<'a> { result } - /// One arm of [`Self::emit_two_way`]: `arm` evaluated in `bb`, which then + /// [`Self::emit_diamond`] for arms that produce no value: they write + /// their results where the caller can find them, so there is nothing to + /// merge and no phi. Leaves the cursor on the merge block. + pub(crate) fn emit_diamond_void( + &mut self, + cond: PseudoId, + then_arm: impl FnOnce(&mut Self), + else_arm: impl FnOnce(&mut Self), + ) { + let (merge_bb, _) = self.emit_fork(cond, then_arm, else_arm); + self.switch_bb(merge_bb); + } + + /// The block plumbing both diamonds share: branch on `cond` into a block + /// per arm, run each arm, and join them. + /// + /// Returns the merge block -- which the caller has *not* switched to yet, + /// so a phi can be placed at its head -- and, per arm, the block it ended + /// in and whatever it produced. + fn emit_fork( + &mut self, + cond: PseudoId, + then_arm: impl FnOnce(&mut Self) -> T, + else_arm: impl FnOnce(&mut Self) -> T, + ) -> (BasicBlockId, [(BasicBlockId, T); 2]) { + let (then_bb, else_bb, merge_bb) = (self.alloc_bb(), self.alloc_bb(), self.alloc_bb()); + let from = self.current_or_unreachable_bb(); + self.emit(Instruction::cbr(cond, then_bb, else_bb)); + self.link_bb(from, then_bb); + self.link_bb(from, else_bb); + + let arms = [ + self.emit_arm(then_bb, merge_bb, then_arm), + self.emit_arm(else_bb, merge_bb, else_arm), + ]; + (merge_bb, arms) + } + + /// One arm of [`Self::emit_fork`]: `arm` evaluated in `bb`, which then /// branches to `merge`. Returns the block the arm ended in, and its value. - fn emit_arm( + fn emit_arm( &mut self, bb: BasicBlockId, merge: BasicBlockId, - arm: impl FnOnce(&mut Self) -> PseudoId, - ) -> (BasicBlockId, PseudoId) { + arm: impl FnOnce(&mut Self) -> T, + ) -> (BasicBlockId, T) { self.switch_bb(bb); let value = arm(self); let end = self.current_or_unreachable_bb(); @@ -2328,7 +2535,13 @@ impl<'a> super::linearize::Linearizer<'a> { // Get the block where LHS evaluation ended (may differ from initial block // if LHS contains nested control flow) - let lhs_end_bb = self.current_bb.unwrap(); + // + // Read through the accessor, not `unwrap`: control cannot arrive at a + // statement before a `switch`'s first `case` or after a `goto`, and the + // phi below needs a real predecessor block to hold its source. See + // `current_or_unreachable_bb`. `branch_on` re-reads `current_bb` + // itself, so it sees the same block. + let lhs_end_bb = self.current_or_unreachable_bb(); // Branch: if LHS is false, go to merge (result = 0); else evaluate RHS self.branch_on(left_cond, eval_b_bb, merge_bb); @@ -2338,8 +2551,9 @@ impl<'a> super::linearize::Linearizer<'a> { let right_bool = self.linearize_condition(right); // Get the actual block where RHS evaluation ended (may differ from eval_b_bb - // if RHS contains nested control flow like another &&/||) - let rhs_end_bb = self.current_bb.unwrap(); + // if RHS contains nested control flow like another &&/||), and may be + // gone entirely where the RHS jumped away: `x && ({ goto L; g(); })`. + let rhs_end_bb = self.current_or_unreachable_bb(); // Branch to merge self.emit(Instruction::br(merge_bb)); @@ -2391,7 +2605,9 @@ impl<'a> super::linearize::Linearizer<'a> { // Get the block where LHS evaluation ended (may differ from initial block // if LHS contains nested control flow) - let lhs_end_bb = self.current_bb.unwrap(); + // + // Through the accessor for the reason `emit_logical_and` records. + let lhs_end_bb = self.current_or_unreachable_bb(); // Branch: if LHS is true, go to merge (result = 1); else evaluate RHS self.branch_on(left_cond, merge_bb, eval_b_bb); @@ -2401,8 +2617,9 @@ impl<'a> super::linearize::Linearizer<'a> { let right_bool = self.linearize_condition(right); // Get the actual block where RHS evaluation ended (may differ from eval_b_bb - // if RHS contains nested control flow like another &&/||) - let rhs_end_bb = self.current_bb.unwrap(); + // if RHS contains nested control flow like another &&/||), and may be + // gone entirely where the RHS jumped away: `x || ({ goto L; g(); })`. + let rhs_end_bb = self.current_or_unreachable_bb(); // Branch to merge self.emit(Instruction::br(merge_bb)); @@ -2500,8 +2717,12 @@ impl<'a> super::linearize::Linearizer<'a> { access_bytes: None, }); let bitfield = match (info.bit_offset, info.bit_width, info.access_bytes) { + // The target expression's type, not the member's declared one: + // they name the same width and sign, and only the expression's + // carries the object's qualifiers (C17 6.5.2.3p3), which is what + // tells the bit-field emitters that the access is volatile. (Some(bit_offset), Some(bit_width), Some(storage)) => { - Some((info.offset, bit_offset, bit_width, storage, info.typ)) + Some((info.offset, bit_offset, bit_width, storage, target_typ)) } // Not a bit-field: fold the member offset into the base so the // load and the store share one address. @@ -2559,7 +2780,9 @@ impl<'a> super::linearize::Linearizer<'a> { typ: TypeId, ) -> Option<(u32, TypeId)> { if let Some((offset, bit_offset, bit_width, storage, field_typ)) = place.bitfield { - self.emit_bitfield_store(place.base, offset, bit_offset, bit_width, storage, val); + self.emit_bitfield_store( + place.base, offset, bit_offset, bit_width, storage, val, field_typ, + ); return Some((bit_width, field_typ)); } let size = self.types.size_bits(typ); @@ -2567,6 +2790,81 @@ impl<'a> super::linearize::Linearizer<'a> { None } + /// The value `E1 op= E2` stores, given `E1`'s current value and `E2`'s. + /// + /// The whole of C17 6.5.16.2p3's arithmetic lives here: choose the type to + /// compute at, choose the opcode for it, bring both operands to it, apply + /// the operator, and convert the result back to the target. Callers supply + /// the two values and nothing else, which is what keeps the ordinary and + /// the `_Atomic` lowerings computing the same thing. + /// + /// `rhs` arrives **unconverted**, at `ca.value_typ`. Converting it to the + /// target first is not an optimization of this: it is a different + /// computation, and it was the bug (see [`CompoundAssign`]). + /// + /// The value this returns is also the value of the assignment expression + /// (C17 6.5.16p3: the left operand's value *after* the assignment), which + /// is why the conversion back is part of the helper rather than of the + /// store: `_Atomic _Bool b = 0; (b -= 1)` has to yield the 1 it stored, + /// not the 255 the subtraction produced. + pub(crate) fn compound_assign_value( + &mut self, + ca: &CompoundAssign, + lhs: PseudoId, + rhs: PseudoId, + ) -> PseudoId { + let arith_type = compound_assign_arith_type(self.types, ca); + let arith_size = self.types.size_bits(arith_type); + let opcode = compound_assign_opcode(self.types, ca.op, arith_type); + + // Both operands into the arithmetic type. The left one is the object's + // current value, read at the target's type; the right one is whatever + // it was written as. + let lhs = if ca.is_ptr_arith { + lhs + } else { + self.emit_convert(lhs, ca.target_typ, arith_type) + }; + let rhs = if ca.is_ptr_arith { + rhs + } else if matches!(ca.op, AssignOp::ShlAssign | AssignOp::ShrAssign) { + // The shift count is promoted on its own and is not brought to the + // left operand's type (C17 6.5.7p3). + self.emit_convert(rhs, ca.value_typ, self.types.integer_promote(ca.value_typ)) + } else { + self.emit_convert(rhs, ca.value_typ, arith_type) + }; + + let result = self.alloc_reg_pseudo(); + self.emit(Instruction::binop( + opcode, result, lhs, rhs, arith_type, arith_size, + )); + + // `nand` is `and` with the result complemented, and the complement + // belongs on this side of the conversion below. + let result = if ca.invert { + let inverted = self.alloc_reg_pseudo(); + self.emit(Instruction::unop( + Opcode::Not, + inverted, + result, + arith_type, + arith_size, + )); + inverted + } else { + result + }; + + // And the result back, which is the conversion that makes `(x /= y)` + // yield what `x` now holds. Pointer arithmetic is already a pointer. + if ca.is_ptr_arith { + result + } else { + self.emit_convert(result, arith_type, ca.target_typ) + } + } + pub(crate) fn emit_assign(&mut self, op: AssignOp, target: &Expr, value: &Expr) -> PseudoId { let target_typ = self.expr_type(target); let value_typ = self.expr_type(value); @@ -2778,8 +3076,9 @@ impl<'a> super::linearize::Linearizer<'a> { )); scaled } else if bool_rhs.is_some() || op != AssignOp::Assign { - // A compound assignment leaves its right operand alone here. It is - // converted to the *common* type below, not down to the target's: + // A compound assignment leaves its right operand alone here. It + // is converted to the *common* type by `compound_assign_value`, + // not down to the target's: // narrowing `-5` to `unsigned char` first made `x /= y` divide // 50 by 251 and store 0, where C17 6.5.16.2p3 computes `50 / -5` // at `int` and stores `(unsigned char)-10`. @@ -2808,109 +3107,14 @@ impl<'a> super::linearize::Linearizer<'a> { Some(p) => self.load_rmw_place(p, target_typ), None => self.linearize_expr(target), }; - let result = self.alloc_reg_pseudo(); - - // `E1 op= E2` is `E1 = E1 op E2` (C17 6.5.16.2p3), so the - // operation runs at the operands' common type after the - // integer promotions -- not at the target's type, which is - // only what the *result* converts back to. - // - // The shifts are the exception: 6.5.7p3 gives the result the - // promoted *left* operand's type, and promotes the right one - // on its own. - let arith_type = if is_ptr_arith { - // Pointer arithmetic already scaled the index; the add - // happens at pointer width. - self.types.long_id - } else if matches!(op, AssignOp::ShlAssign | AssignOp::ShrAssign) { - self.types.integer_promote(target_typ) - } else { - self.types.common_type(target_typ, value_typ) - }; - - let is_float = self.types.is_float(arith_type); - let is_unsigned = self.types.is_unsigned(arith_type); - let opcode = match op { - AssignOp::AddAssign => { - if is_float { - Opcode::FAdd - } else { - Opcode::Add - } - } - AssignOp::SubAssign => { - if is_float { - Opcode::FSub - } else { - Opcode::Sub - } - } - AssignOp::MulAssign => { - if is_float { - Opcode::FMul - } else { - Opcode::Mul - } - } - AssignOp::DivAssign => { - if is_float { - Opcode::FDiv - } else if is_unsigned { - Opcode::DivU - } else { - Opcode::DivS - } - } - AssignOp::ModAssign => { - // Modulo not supported for floats - if is_unsigned { - Opcode::ModU - } else { - Opcode::ModS - } - } - AssignOp::AndAssign => Opcode::And, - AssignOp::OrAssign => Opcode::Or, - AssignOp::XorAssign => Opcode::Xor, - AssignOp::ShlAssign => Opcode::Shl, - AssignOp::ShrAssign => { - if is_unsigned { - Opcode::Lsr - } else { - Opcode::Asr - } - } - AssignOp::Assign => unreachable!(), - }; - - let arith_size = self.types.size_bits(arith_type); - // Both operands into the arithmetic type. The left one is the - // object's current value, read at the target's type; the right - // one is whatever it was written as. - let lhs = if is_ptr_arith { - lhs - } else { - self.emit_convert(lhs, target_typ, arith_type) - }; - let rhs = if is_ptr_arith { - rhs - } else if matches!(op, AssignOp::ShlAssign | AssignOp::ShrAssign) { - // The shift count is promoted on its own and is not - // brought to the left operand's type. - self.emit_convert(rhs, value_typ, self.types.integer_promote(value_typ)) - } else { - self.emit_convert(rhs, value_typ, arith_type) + // One helper owns the whole of C17 6.5.16.2p3's arithmetic, + // shared with the `_Atomic` lowering, which used to carry its + // own copy of these rules and disagree with this one. + let ca = CompoundAssign { + is_ptr_arith, + ..CompoundAssign::new(op, target_typ, value_typ) }; - self.emit(Instruction::binop( - opcode, result, lhs, rhs, arith_type, arith_size, - )); - // And the result back, which is the conversion that makes - // `(x /= y)` yield what `x` now holds. - if is_ptr_arith { - result - } else { - self.emit_convert(result, arith_type, target_typ) - } + self.compound_assign_value(&ca, lhs, rhs) } }; diff --git a/cc/ir/linearize_init.rs b/cc/ir/linearize_init.rs index 927a0da0b..2dc202919 100644 --- a/cc/ir/linearize_init.rs +++ b/cc/ir/linearize_init.rs @@ -48,6 +48,92 @@ pub(crate) fn is_const_object_type(types: &TypeTable, typ: TypeId) -> bool { /// converts to an integer type outside its range, as it does in gcc. const OVERFLOW_WARNING: &str = "overflow"; +/// The bytes a bit-field's own bits occupy, as `(byte offset from the field's +/// own offset, the bits `value` puts there, the mask of the bits the field +/// owns)`. +/// +/// Only the bytes the field's bits reach. Its access span is wider and +/// generally starts earlier -- `unsigned a:1` after a `char` sits at bit 8 of +/// a span based at byte 0 -- and writing the whole span would blank the +/// members sharing it. +fn bitfield_carrier_bytes( + bit_offset: u32, + bit_width: u32, + value: i128, +) -> impl Iterator { + // A field with no bits, or one no carrier could hold, occupies no byte. + // Otherwise the mask comes of shifting `u128::MAX` down rather than + // `1 << width` up, for the reason `bitfield_value_mask` records: the + // latter overflows at the carrier's own width. + let fits = bit_width > 0 && u64::from(bit_offset) + u64::from(bit_width) <= 128; + let (shift, width_mask) = if fits { + (bit_offset, u128::MAX >> (128 - bit_width)) + } else { + (0, 0) + }; + let placed = ((value as u128) & width_mask) << shift; + let owned = width_mask << shift; + let bytes = if fits { + (bit_offset / 8) as usize..((bit_offset + bit_width - 1) / 8) as usize + 1 + } else { + 0..0 + }; + bytes.map(move |byte| { + let shift = byte * 8; + ( + byte, + ((placed >> shift) & 0xff) as u8, + ((owned >> shift) & 0xff) as u8, + ) + }) +} + +/// The bytes a bit-field member's own bits occupy, measured from the first +/// byte of the struct that declares it. `None` for anything but a bit-field. +fn bitfield_byte_span(member: &crate::types::StructMember) -> Option> { + let (bit_offset, bit_width) = (member.bit_offset?, member.bit_width?); + let start = member.offset + (bit_offset / 8) as usize; + let end = member.offset + (bit_offset + bit_width).div_ceil(8) as usize; + Some(start..end.max(start + 1)) +} + +/// Give the carrier byte at `offset` of a struct initializer the bits `bits` +/// in place of the ones `mask` marks, leaving the neighbouring bit-fields +/// that share the byte as they are. +/// +/// False when the byte is not a carrier byte the lowering wrote -- an entry +/// for an ordinary member covering it, or something other than an integer -- +/// which leaves the caller to discard the earlier initializer whole. +fn replace_carrier_bits( + fields: &mut Vec<(usize, usize, Initializer)>, + offset: usize, + bits: u8, + mask: u8, +) -> bool { + match fields + .iter() + .position(|(off, size, _)| *off <= offset && offset < *off + *size) + { + Some(index) => { + if (fields[index].0, fields[index].1) != (offset, 1) { + return false; + } + let Initializer::Int(held) = fields[index].2 else { + return false; + }; + fields[index].2 = Initializer::Int(i128::from((held as u8 & !mask) | bits)); + true + } + None => { + // The byte holds no bit-field the earlier initializer gave a + // value to, so the bits around this field's are zero. + fields.push((offset, 1, Initializer::Int(i128::from(bits)))); + fields.sort_by_key(|(off, _, _)| *off); + true + } + } +} + /// Name an expression the way a C programmer would, for a diagnostic. /// /// The alternative these messages used was `{:?}` on the AST node, which @@ -860,11 +946,32 @@ impl<'a> super::linearize::Linearizer<'a> { /// Group array init elements by index, handling designators, brace elision, /// and nested InitList flattening. Shared between static and runtime paths. + /// + /// `array_typ` is the array being initialized, not its element type: the + /// bound is needed as well as the element type, because an initializer + /// past the last element is excess (C17 6.7.9p2) and must be *discarded*. + /// Grouping it anyway gave it an offset beyond the object and both lowering + /// paths then wrote there -- `int a[2] = {1, 2, 3};` stored the 3 over + /// whatever the frame put after `a`, and the file-scope form emitted a + /// third `.long` under a two-element symbol. c17 already warns about the + /// excess element; this is where it stops mattering. pub(crate) fn group_array_init_elements( &self, elements: &[InitElement], - elem_type: TypeId, + array_typ: TypeId, ) -> ArrayInitGroups { + let elem_type = self.types.base_type(array_typ).unwrap_or(self.types.int_id); + + // An absent or zero bound is an array sized *by* this initializer -- an + // incomplete type `int a[] = {1, 2, 3}`, a flexible array member, or a + // GNU zero-length array -- so nothing in the list can be excess. + let last_index = self + .types + .array_size(array_typ) + .filter(|&n| n > 0) + .map(|n| n as i64 - 1); + let in_bounds = |idx: i64| last_index.is_none_or(|last| (0..=last).contains(&idx)); + let mut element_lists: HashMap> = HashMap::new(); let mut element_indices: Vec = Vec::new(); let mut current_idx: i64 = 0; @@ -912,7 +1019,13 @@ impl<'a> super::linearize::Linearizer<'a> { if remaining_designators.is_empty() && self.is_brace_elision_candidate(element, elem_type) { + // Consumed either way: the elements belong to this slot, and + // leaving them in the list would make the next iteration read + // them as initializers for the enclosing array. let sub_elements = self.consume_brace_elision(elements, &mut elem_idx, elem_type); + if !in_bounds(element_index) { + continue; + } let entry = element_lists.entry(element_index).or_insert_with(|| { element_indices.push(element_index); Vec::new() @@ -926,27 +1039,37 @@ impl<'a> super::linearize::Linearizer<'a> { // Expanding here keeps both lowering paths -- the static data // image and the runtime stores -- unchanged, and matches how c17 // already lowers a bulk initializer element by element. + // + // A range is clamped rather than dropped whole, so that + // `[0 ... 4] = 1` on a three-element array still initializes the + // three elements the array has. Clamping the high endpoint also + // keeps the loop off the excess indices rather than walking one + // iteration per discarded element, which matters for an endpoint + // far past the array. let span_end = index_high.unwrap_or(element_index); - for target_index in element_index..=span_end { - let entry = element_lists.entry(target_index).or_insert_with(|| { - element_indices.push(target_index); - Vec::new() - }); + let span_end = last_index.map_or(span_end, |last| span_end.min(last)); + if in_bounds(element_index) { + for target_index in element_index..=span_end { + let entry = element_lists.entry(target_index).or_insert_with(|| { + element_indices.push(target_index); + Vec::new() + }); - if remaining_designators.is_empty() { - if let ExprKind::InitList { - elements: nested_elements, - } = &element.value.kind - { - entry.extend(nested_elements.clone()); - continue; + if remaining_designators.is_empty() { + if let ExprKind::InitList { + elements: nested_elements, + } = &element.value.kind + { + entry.extend(nested_elements.clone()); + continue; + } } - } - entry.push(InitElement { - designators: remaining_designators.clone(), - value: element.value.clone(), - }); + entry.push(InitElement { + designators: remaining_designators.clone(), + value: element.value.clone(), + }); + } } elem_idx += 1; } @@ -978,15 +1101,25 @@ impl<'a> super::linearize::Linearizer<'a> { if element.designators.is_empty() { // Positional: check anonymous struct continuation, then next member let mut member = None; + let mut member_index = None; if anon_cont.is_some() { + // A continuation is filling the anonymous aggregate at + // `outer_idx`, so that is the member this element lands in. + member_index = anon_cont.as_ref().map(|cont| cont.outer_idx); member = self.get_anon_continuation_member( &mut anon_cont, members, &mut current_field_idx, ); + if member.is_none() { + member_index = None; + } } if member.is_none() { - member = self.next_positional_member(members, is_union, &mut current_field_idx); + let picked = + self.next_positional_member(members, is_union, &mut current_field_idx); + member_index = picked.as_ref().map(|(_, index)| *index); + member = picked.map(|(member, _)| member); } let Some(member) = member else { elem_idx += 1; @@ -1011,6 +1144,8 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset: member.bit_offset, bit_width: member.bit_width, access_bytes: member.access_bytes, + member_index, + unions: UnionMembers::default(), }); continue; } @@ -1023,19 +1158,23 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset, bit_width, access_bytes, + unions, }) = resolved else { elem_idx += 1; continue; }; + let mut member_index = None; if let Some(Designator::Field(name)) = element.designators.first() { if let Some(result) = self.member_index_for_designator(members, *name) { match result { MemberDesignatorResult::Direct(next_idx) => { + member_index = Some(next_idx - 1); current_field_idx = next_idx; anon_cont = None; } MemberDesignatorResult::Anonymous { outer_idx, levels } => { + member_index = Some(outer_idx); current_field_idx = outer_idx; anon_cont = Some(AnonContinuation { outer_idx, levels }); } @@ -1051,6 +1190,8 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset, bit_width, access_bytes, + member_index, + unions, }); elem_idx += 1; } @@ -1058,6 +1199,698 @@ impl<'a> super::linearize::Linearizer<'a> { visits } + /// The member each union inside the subobject a visit initializes comes + /// to hold, by byte offset within the object being initialized. + /// + /// Answered from the initializer list rather than from the lowered + /// `Initializer`, so that the static and the automatic path get the same + /// answer from the same walk. Both need it to resolve a later initializer + /// naming something inside a union against this one: only where the union + /// still holds the same member does the rest of that member survive. + pub(crate) fn held_union_members( + &self, + typ: TypeId, + kind: &StructFieldVisitKind, + base: usize, + ) -> UnionMembers { + let mut held = UnionMembers::default(); + // The walk costs a second pass over the initializer, so it is only + // made where there is a union to find. + if self.type_holds_union(typ) { + self.record_held_unions_of(typ, kind, base, &mut held); + } + held + } + + /// Whether an object of type `typ` has a union anywhere inside it, itself + /// included. Pointers are not looked through, so this terminates on the + /// self-referential types C allows. + fn type_holds_union(&self, typ: TypeId) -> bool { + let typ = self.resolve_struct_type(typ); + match self.types.kind(typ) { + TypeKind::Union => true, + TypeKind::Struct => self + .types + .get(typ) + .composite + .as_ref() + .is_some_and(|composite| { + composite + .members + .iter() + .any(|member| self.type_holds_union(member.typ)) + }), + TypeKind::Array => self + .types + .base_type(typ) + .is_some_and(|elem| self.type_holds_union(elem)), + _ => false, + } + } + + fn record_held_unions( + &self, + typ: TypeId, + elements: &[InitElement], + base: usize, + held: &mut UnionMembers, + ) { + let typ = self.resolve_struct_type(typ); + match self.types.kind(typ) { + TypeKind::Struct | TypeKind::Union => { + let Some(composite) = self.types.get(typ).composite.as_ref() else { + return; + }; + let members = composite.members.clone(); + let is_union = self.types.kind(typ) == TypeKind::Union; + let visits = self.walk_struct_init_fields(typ, &members, is_union, elements); + if is_union { + // Each element of a union's initializer list replaces what + // the one before it put there, so the union ends up + // holding whichever member the last of them named. + if let Some(index) = visits.last().and_then(|visit| visit.member_index) { + held.record(base, typ, index); + } + } + for visit in &visits { + self.record_held_unions_of(visit.typ, &visit.kind, base + visit.offset, held); + } + } + TypeKind::Array => { + let Some(elem_type) = self.types.base_type(typ) else { + return; + }; + let elem_size = self.types.size_bytes(elem_type); + let groups = self.group_array_init_elements(elements, typ); + for index in groups.indices { + let Some(list) = groups.element_lists.get(&index) else { + continue; + }; + self.record_held_unions( + elem_type, + list, + base + index as usize * elem_size, + held, + ); + } + } + _ => {} + } + } + + fn record_held_unions_of( + &self, + typ: TypeId, + kind: &StructFieldVisitKind, + base: usize, + held: &mut UnionMembers, + ) { + match kind { + StructFieldVisitKind::BraceElision(elements) => { + self.record_held_unions(typ, elements, base, held); + } + StructFieldVisitKind::Expr(expr) => { + if let ExprKind::InitList { elements } = &expr.kind { + self.record_held_unions(typ, elements, base, held); + } + } + } + } + + /// Where the byte range `offset..offset + size` sits inside an object of + /// type `typ`, with `offset` measured from that object's first byte. + /// + /// Two initializers in one list can describe overlapping storage, and + /// C17 6.7.9p19 resolves that by *subobject*, not by bytes: given + /// `{ .t = {1, 2}, .t.y = 9 }` the second names a member of the first and + /// replaces only it, while given `{ .u.i = 1, .u.s.b = 9 }` the second + /// names a different member of a union and so replaces the whole of what + /// the first wrote. The spans are identical in shape -- one inside the + /// other -- so only the type can tell the two cases apart. + /// + /// `unions` says which member each union on the way holds; a union it has + /// nothing to say about is answered [`SubobjectPlace::ThroughUnion`], + /// which discards its contents. + pub(crate) fn classify_subobject( + &self, + typ: TypeId, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> SubobjectPlace { + self.classify_range(typ, offset, size, false, unions) + } + + /// The same question for the bytes a *bit-field's* own bits occupy. + /// + /// A bit-field is not addressable storage, so it is never a subobject of + /// its own and the answer wanted is the struct that declares it, whose + /// initializer holds the carrier the field shares with its neighbours. + /// An exact byte match is no reason to stop early: `struct { unsigned + /// char a : 4, b : 4; }` is one byte wide, and replacing it whole would + /// take `b` with it. + pub(crate) fn classify_bitfield_carrier( + &self, + typ: TypeId, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> SubobjectPlace { + self.classify_range(typ, offset, size, true, unions) + } + + fn classify_range( + &self, + typ: TypeId, + offset: usize, + size: usize, + bitfield: bool, + unions: UnionFold<'_>, + ) -> SubobjectPlace { + let mut typ = self.resolve_struct_type(typ); + let mut base = 0usize; + let mut unions = unions; + + loop { + let type_size = self.types.size_bytes(typ); + if !bitfield && offset == base && size == type_size { + return SubobjectPlace::Member; + } + if size == 0 || offset < base || offset + size > base + type_size { + return SubobjectPlace::NotASubobject; + } + + match self.types.kind(typ) { + // Reached a union without having named it exactly, so the + // range lies inside one of its members. Which member the union + // holds is not decidable from the span -- every member starts + // at the same byte -- so unless both initializers agree on it, + // giving this range an initializer discards what the union + // held before. + TypeKind::Union => { + let Some(member) = self.agreed_union_member(typ, base, offset, size, unions) + else { + return SubobjectPlace::ThroughUnion { + offset: base, + size: type_size, + }; + }; + base += member.0; + unions = unions.inside(member.0); + typ = self.resolve_struct_type(member.1); + } + TypeKind::Struct => { + let Some(composite) = self.types.get(typ).composite.as_ref() else { + return SubobjectPlace::NotASubobject; + }; + // A bit-field is not addressable storage of its own, so a + // range inside its carrier is not a subobject. + let found = composite.members.iter().find(|member| { + member.bit_width.is_none() && { + let member_start = base + member.offset; + let member_end = member_start + self.types.size_bytes(member.typ); + offset >= member_start && offset + size <= member_end + } + }); + let Some(member) = found else { + // No ordinary member holds these bytes. For a + // bit-field's bits that is expected, and this is the + // struct that declares it. + let declares = bitfield + && composite.members.iter().any(|member| { + bitfield_byte_span(member) + .is_some_and(|span| self.holds(base, &span, offset, size)) + }); + return if declares { + SubobjectPlace::BitfieldCarrier { + offset: base, + size: type_size, + } + } else { + SubobjectPlace::NotASubobject + }; + }; + base += member.offset; + unions = unions.inside(member.offset); + typ = self.resolve_struct_type(member.typ); + } + TypeKind::Array => { + let Some(elem_type) = self.types.base_type(typ) else { + return SubobjectPlace::NotASubobject; + }; + let elem_size = self.types.size_bytes(elem_type); + if elem_size == 0 { + return SubobjectPlace::NotASubobject; + } + let elem_start = base + ((offset - base) / elem_size) * elem_size; + if offset + size > elem_start + elem_size { + return SubobjectPlace::NotASubobject; + } + unions = unions.inside(elem_start - base); + base = elem_start; + typ = self.resolve_struct_type(elem_type); + } + // A scalar with something strictly inside it: only a union or + // a bit-field carrier can produce that, and neither is a + // subobject relation. + _ => return SubobjectPlace::NotASubobject, + } + } + } + + /// Whether the member span `span`, measured from a struct beginning at + /// `base`, holds all of `offset..offset + size`. + fn holds( + &self, + base: usize, + span: &std::ops::Range, + offset: usize, + size: usize, + ) -> bool { + offset >= base + span.start && offset + size <= base + span.end + } + + /// The member a union of type `typ`, beginning at `base`, is agreed to + /// hold -- as `(its offset, its type)` -- when that member holds all of + /// `offset..offset + size`. + fn agreed_union_member( + &self, + typ: TypeId, + base: usize, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> Option<(usize, TypeId)> { + let index = unions.agreed(typ)?; + let member = self + .types + .get(typ) + .composite + .as_ref()? + .members + .get(index) + .filter(|member| member.bit_width.is_none())?; + let span = member.offset..member.offset + self.types.size_bytes(member.typ); + self.holds(base, &span, offset, size) + .then_some((member.offset, member.typ)) + } + + /// An initializer that writes nothing, shaped for `typ` so that + /// [`Self::subobject_init_mut`] can place entries into it. + fn empty_aggregate_init(&self, typ: TypeId) -> Option { + match self.types.kind(typ) { + TypeKind::Struct | TypeKind::Union => Some(Initializer::Struct { + total_size: self.types.size_bytes(typ), + fields: Vec::new(), + }), + TypeKind::Array => { + let elem_type = self.types.base_type(typ)?; + Some(Initializer::Array { + elem_size: self.types.size_bytes(elem_type), + total_size: self.types.size_bytes(typ), + elements: Vec::new(), + }) + } + _ => None, + } + } + + /// Fold `new_init` into `init`, the initializer for an object of type + /// `typ`, so that it initializes the subobject at `offset..offset + size` + /// and leaves every other subobject as it was. + /// + /// The caller has already established with [`Self::classify_subobject`] + /// that the range *is* such a subobject. Returns false when the existing + /// initializer's shape cannot express the replacement -- a string literal + /// standing for a character array, say -- in which case the caller falls + /// back to discarding the earlier initializer whole. + pub(crate) fn overlay_subobject( + &self, + typ: TypeId, + init: &mut Initializer, + offset: usize, + size: usize, + new_init: &Initializer, + unions: UnionFold<'_>, + ) -> bool { + match self.subobject_init_mut(typ, init, offset, size, unions) { + Some(slot) => { + *slot = new_init.clone(); + true + } + None => false, + } + } + + /// The initializer for the subobject at `offset..offset + size` of an + /// object of type `typ` initialized by `init`, to be read or replaced in + /// place. A subobject the initializer left out gains an empty one, so + /// that what is written there leaves the rest of its parent alone. + /// + /// `None` when the existing initializer's shape cannot express the + /// subobject -- a string literal standing for a character array, an entry + /// for a bit-field carrier rather than for a member -- in which case the + /// caller discards the earlier initializer whole. It may by then have + /// gained an empty initializer for a member on the way down, which is + /// harmless precisely because the whole of it is discarded. + fn subobject_init_mut<'i>( + &self, + typ: TypeId, + init: &'i mut Initializer, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> Option<&'i mut Initializer> { + let typ = self.resolve_struct_type(typ); + if offset == 0 && size == self.types.size_bytes(typ) { + return Some(init); + } + + match self.types.kind(typ) { + TypeKind::Struct | TypeKind::Union => { + let (slot_offset, slot_type) = if self.types.kind(typ) == TypeKind::Union { + // Every member of a union begins at the same byte, so the + // range cannot say which one it is in. Only the member the + // two initializers agree the union holds will do. + self.agreed_union_member(typ, 0, offset, size, unions)? + } else { + self.types + .get(typ) + .composite + .as_ref()? + .members + .iter() + .find(|member| { + member.bit_width.is_none() && { + let span = member.offset + ..member.offset + self.types.size_bytes(member.typ); + self.holds(0, &span, offset, size) + } + }) + .map(|member| (member.offset, member.typ))? + }; + let slot_size = self.types.size_bytes(slot_type); + let Initializer::Struct { fields, .. } = init else { + return None; + }; + let slot = Self::entry_init_mut(fields, slot_offset, slot_size, || { + self.empty_aggregate_init(slot_type) + })?; + self.subobject_init_mut( + slot_type, + slot, + offset - slot_offset, + size, + unions.inside(slot_offset), + ) + } + TypeKind::Array => { + let elem_type = self.types.base_type(typ)?; + let elem_size = self.types.size_bytes(elem_type); + if elem_size == 0 { + return None; + } + let elem_offset = (offset / elem_size) * elem_size; + if offset + size > elem_offset + elem_size { + return None; + } + let Initializer::Array { elements, .. } = init else { + return None; + }; + // An array's entries carry no width, so give each one the + // element width it implicitly has and borrow the struct + // path's bookkeeping. + let slot = Self::element_init_mut(elements, elem_offset, || { + self.empty_aggregate_init(elem_type) + })?; + self.subobject_init_mut( + elem_type, + slot, + offset - elem_offset, + size, + unions.inside(elem_offset), + ) + } + _ => None, + } + } + + /// The entry in a struct initializer's field list for the member at + /// `slot_offset` of `slot_size` bytes, added as an empty initializer if + /// the list has none. + /// + /// `None` when an entry overlaps the member without being exactly it -- a + /// bit-field carrier byte, or an entry spanning several members -- which + /// is not something a subobject can be placed into. + fn entry_init_mut( + fields: &mut Vec<(usize, usize, Initializer)>, + slot_offset: usize, + slot_size: usize, + empty: impl FnOnce() -> Option, + ) -> Option<&mut Initializer> { + let slot_end = slot_offset + slot_size; + let index = match fields + .iter() + .position(|(off, sz, _)| *off < slot_end && slot_offset < *off + *sz) + { + Some(index) => { + if (fields[index].0, fields[index].1) != (slot_offset, slot_size) { + return None; + } + index + } + None => { + // A scalar member has no aggregate to make empty; whatever is + // put here is about to be replaced outright, since nothing + // lies inside a scalar to descend to. + fields.push((slot_offset, slot_size, empty().unwrap_or_default())); + fields.sort_by_key(|(off, _, _)| *off); + fields.iter().position(|(off, _, _)| *off == slot_offset)? + } + }; + Some(&mut fields[index].2) + } + + /// The same for an array initializer's element list, whose entries are + /// one element wide by construction. + fn element_init_mut( + elements: &mut Vec<(usize, Initializer)>, + elem_offset: usize, + empty: impl FnOnce() -> Option, + ) -> Option<&mut Initializer> { + let index = match elements.iter().position(|(off, _)| *off == elem_offset) { + Some(index) => index, + None => { + elements.push((elem_offset, empty().unwrap_or_default())); + elements.sort_by_key(|(off, _)| *off); + elements.iter().position(|(off, _)| *off == elem_offset)? + } + }; + Some(&mut elements[index].1) + } + + /// Apply C17 6.7.9p19 to the initializers one struct or union + /// initializer list produced, in the order the list wrote them. + /// + /// An initializer for a subobject overrides any previously listed + /// initializer *for that subobject*, and leaves initializers for other + /// subobjects alone. So a later entry is folded into an earlier one it is + /// a member of, replaces an earlier one it contains, and -- when the two + /// are related only through a union or a bit-field carrier, where no + /// structural fold exists -- discards it. + /// + /// "Later" means later in the list. The entries arrive in that order and + /// the sort into address order, which the emitter needs, runs afterwards. + pub(crate) fn merge_raw_field_inits(&self, raw: Vec) -> Vec { + let mut merged: Vec = Vec::with_capacity(raw.len()); + + for later in raw { + let later_span = later.byte_span(); + let mut folded = false; + let mut idx = 0; + + while idx < merged.len() { + let earlier = &merged[idx]; + let earlier_span = earlier.byte_span(); + if earlier_span.start >= later_span.end || later_span.start >= earlier_span.end { + idx += 1; + continue; + } + // Two *distinct* bit-fields are different objects even when + // they share a carrier byte, so both survive. + if earlier.bit_width.is_some() + && later.bit_width.is_some() + && (earlier.offset, earlier.bit_offset) != (later.offset, later.bit_offset) + { + idx += 1; + continue; + } + // A bit-field never supersedes an initializer for an object + // containing it, however the two spans compare: it is narrower + // than the storage they share -- `struct { unsigned char a : 4, + // b : 4; }` is one byte -- so what it does not name stands, and + // it is folded into the carrier instead. + let inside = earlier_span.start <= later_span.start + && later_span.end <= earlier_span.end + && earlier.bit_width.is_none(); + if !(inside && later.bit_width.is_some()) + && later_span.start <= earlier_span.start + && earlier_span.end <= later_span.end + { + merged.remove(idx); + continue; + } + let foldable = !folded && inside && self.fold_field_init(&mut merged[idx], &later); + if foldable { + folded = true; + idx += 1; + continue; + } + merged.remove(idx); + } + + if !folded { + merged.push(later); + } + } + + merged + } + + /// Fold `later`, whose bytes lie inside `earlier`'s, into `earlier`. + /// Returns false when no structural fold exists, which leaves the caller + /// to discard `earlier`. + fn fold_field_init(&self, earlier: &mut RawFieldInit, later: &RawFieldInit) -> bool { + // `unions` borrows what the earlier entry holds, which the fold then + // updates, so it reads from a copy. + let held = earlier.held.clone(); + let unions = UnionFold::new(&held, &later.named, earlier.offset); + let span = later.byte_span(); + let inner_offset = span.start - earlier.offset; + let inner_size = span.end - span.start; + + let place = if later.bit_width.is_some() { + self.classify_bitfield_carrier(earlier.typ, inner_offset, inner_size, unions) + } else { + self.classify_subobject(earlier.typ, inner_offset, inner_size, unions) + }; + let folded = match place { + SubobjectPlace::Member => self.overlay_subobject( + earlier.typ, + &mut earlier.init, + inner_offset, + inner_size, + &later.init, + unions, + ), + // Not storage of its own: only the bits the field names change, + // inside the carrier its declaring struct's initializer wrote. + SubobjectPlace::BitfieldCarrier { offset, size } => { + self.fold_bitfield_init(earlier, later, offset, size, unions) + } + // The union stops holding what it held: everything it contained + // goes, and it comes to hold just this one initializer. + SubobjectPlace::ThroughUnion { offset, size } => { + let Some(fields) = self.union_replacement_fields(earlier, later, offset) else { + return false; + }; + let replacement = Initializer::Struct { + total_size: size, + fields, + }; + let replaced = self.overlay_subobject( + earlier.typ, + &mut earlier.init, + offset, + size, + &replacement, + unions, + ); + if replaced { + let start = earlier.offset + offset; + earlier.held.clear_range(start..start + size); + } + replaced + } + SubobjectPlace::NotASubobject => false, + }; + + if folded { + // What the later initializer says about the unions it wrote or + // passed through is the last word on them. + earlier.held.absorb(&later.held); + earlier.held.absorb(&later.named); + } + folded + } + + /// The entries a union comes to hold once `later` replaces its contents: + /// the initializer itself, or the bytes of a bit-field's carrier, placed + /// at their offsets within the union beginning at `offset` of `earlier`. + fn union_replacement_fields( + &self, + earlier: &RawFieldInit, + later: &RawFieldInit, + offset: usize, + ) -> Option> { + let union_start = earlier.offset + offset; + let (Some(bit_offset), Some(bit_width)) = (later.bit_offset, later.bit_width) else { + let inner = later.offset.checked_sub(union_start)?; + return Some(vec![(inner, later.field_size, later.init.clone())]); + }; + let Initializer::Int(value) = later.init else { + return None; + }; + bitfield_carrier_bytes(bit_offset, bit_width, value) + .map(|(byte, bits, _)| { + let inner = (later.offset + byte).checked_sub(union_start)?; + Some((inner, 1, Initializer::Int(i128::from(bits)))) + }) + .collect() + } + + /// Replace the bits `later` names inside the carrier bytes the struct + /// declaring it -- at `offset..offset + size` of `earlier`'s object -- + /// already has an initializer for. + /// + /// The carrier is whole bytes by the time it reaches here: a struct's + /// bit-fields are merged into one byte apiece as the last step of lowering + /// it, downstream of the `Initializer` tree, so there is no bit-field left + /// in the tree to replace -- only the bits of it that a byte holds. + fn fold_bitfield_init( + &self, + earlier: &mut RawFieldInit, + later: &RawFieldInit, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> bool { + let (Some(bit_offset), Some(bit_width)) = (later.bit_offset, later.bit_width) else { + return false; + }; + let Initializer::Int(value) = later.init else { + return false; + }; + let Some(carrier) = + self.subobject_init_mut(earlier.typ, &mut earlier.init, offset, size, unions) + else { + return false; + }; + let Initializer::Struct { fields, .. } = carrier else { + return false; + }; + // Where the field's own storage begins within the declaring struct. + let Some(base) = later.offset.checked_sub(earlier.offset + offset) else { + return false; + }; + for (byte, bits, mask) in bitfield_carrier_bytes(bit_offset, bit_width, value) { + if !replace_carrier_bits(fields, base + byte, bits, mask) { + return false; + } + } + true + } + /// Convert an AST initializer list to an IR Initializer pub(crate) fn ast_init_list_to_ir( &mut self, @@ -1100,7 +1933,7 @@ impl<'a> super::linearize::Linearizer<'a> { TypeKind::Array | TypeKind::Struct | TypeKind::Union ); - let groups = self.group_array_init_elements(elements, elem_type); + let groups = self.group_array_init_elements(elements, typ); let mut init_elements = Vec::new(); for element_index in groups.indices { let Some(list) = groups.element_lists.get(&element_index) else { @@ -1156,6 +1989,7 @@ impl<'a> super::linearize::Linearizer<'a> { // Convert field visits to RawFieldInit by evaluating expressions let mut raw_fields: Vec = Vec::new(); for visit in visits { + let held = self.held_union_members(visit.typ, &visit.kind, visit.offset); let field_init = match visit.kind { StructFieldVisitKind::BraceElision(sub_elements) => { self.ast_init_list_to_ir(&sub_elements, visit.typ) @@ -1167,40 +2001,28 @@ impl<'a> super::linearize::Linearizer<'a> { raw_fields.push(RawFieldInit { offset: visit.offset, field_size: visit.field_size, + typ: visit.typ, init: field_init, bit_offset: visit.bit_offset, bit_width: visit.bit_width, + held, + named: visit.unions, }); } + // Initializing the same object twice: the later one wins + // (C17 6.7.9p19), and one that names a *subobject* of an + // earlier one replaces only that subobject. Resolved in + // the order the list wrote them, before the sort below + // reorders them by address. + let mut raw_fields = self.merge_raw_field_inits(raw_fields); + // Sort by the bit each field starts at, so that designated // initializers emit in address order however they were // written -- the emitter fills the gaps between fields and // so requires them sorted and non-overlapping. raw_fields.sort_by_key(|f| f.offset * 8 + f.bit_offset.unwrap_or(0) as usize); - // Initializing the same object twice: the later one wins - // (C17 6.7.9p19). Two *distinct* bitfields are different - // objects even when they share a byte, so both survive. - let mut idx = 0; - while idx + 1 < raw_fields.len() { - let (a, b) = (&raw_fields[idx], &raw_fields[idx + 1]); - let distinct_bitfields = a.bit_width.is_some() - && b.bit_width.is_some() - && (a.offset, a.bit_offset) != (b.offset, b.bit_offset); - let a_span = a.byte_span(); - let b_span = b.byte_span(); - - if !distinct_bitfields - && a_span.start < b_span.end - && b_span.start < a_span.end - { - raw_fields.remove(idx); - } else { - idx += 1; - } - } - // Merge bitfields byte by byte rather than one storage unit // at a time. A unit is `sizeof(T)` wide and aligned, so it // routinely spans bytes that belong to other members -- @@ -1222,25 +2044,7 @@ impl<'a> super::linearize::Linearizer<'a> { let Initializer::Int(value) = field.init else { continue; }; - if bit_width == 0 { - continue; - } - - // Shifting `u128::MAX` down rather than `1 << width` - // up, for the reason `bitfield_value_mask` records: the - // latter overflows at the carrier's own width. - let mask = u128::MAX >> (128 - bit_width); - let placed = ((value as u128) & mask) << bit_off; - - // Only the bytes the field's own bits reach. Its window - // is wider and generally starts earlier -- `unsigned a:1` - // after a `char` sits at bit 8 of a window based at byte - // 0 -- and writing the whole window here would blank the - // members sharing it. - for byte in - (bit_off / 8) as usize..=((bit_off + bit_width - 1) / 8) as usize - { - let bits = ((placed >> (byte * 8)) & 0xff) as u8; + for (byte, bits, _) in bitfield_carrier_bytes(bit_off, bit_width, value) { *bitfield_bytes.entry(field.offset + byte).or_default() |= bits; } } @@ -1282,6 +2086,7 @@ impl<'a> super::linearize::Linearizer<'a> { let mut bit_offset = None; let mut bit_width = None; let mut access_bytes = None; + let mut unions = UnionMembers::default(); for (idx, designator) in designators.iter().enumerate() { match designator { @@ -1291,6 +2096,14 @@ impl<'a> super::linearize::Linearizer<'a> { resolved = self.types.base_type(resolved)?; } resolved = self.resolve_struct_type(resolved); + // Naming a member of a union says which member the + // initializer is for, and nothing downstream can recover + // that: every member of a union begins at `offset`. + if self.types.kind(resolved) == TypeKind::Union { + if let Some(index) = self.designated_member_index(resolved, *name) { + unions.record(offset, resolved, index); + } + } let member = self.types.find_member(resolved, *name)?; offset += member.offset; typ = member.typ; @@ -1332,15 +2145,32 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset, bit_width, access_bytes, + unions, }) } + /// The index, in `composite_typ`'s member list, of the member a `.name` + /// designator names -- or of the anonymous member that contains it. + fn designated_member_index(&self, composite_typ: TypeId, name: StringId) -> Option { + let members = &self.types.get(composite_typ).composite.as_ref()?.members; + match self.member_index_for_designator(members, name)? { + MemberDesignatorResult::Direct(next_idx) => Some(next_idx - 1), + MemberDesignatorResult::Anonymous { outer_idx, .. } => Some(outer_idx), + } + } + + /// The member the next positional element initializes, and its index in + /// `members`. + /// + /// The index is what tells two members of a union apart: they share a byte + /// offset, so nothing downstream can recover which one an initializer + /// chose. pub(crate) fn next_positional_member( &self, members: &[crate::types::StructMember], is_union: bool, current_field_idx: &mut usize, - ) -> Option { + ) -> Option<(MemberInfo, usize)> { if is_union { if *current_field_idx > 0 { return None; @@ -1357,34 +2187,42 @@ impl<'a> super::linearize::Linearizer<'a> { // members are *all* anonymous found none at all and stayed zero // entirely. Unnamed bit-field padding is not a member and is // still skipped. - let member = members + let (index, member) = members .iter() - .find(|m| m.name != StringId::EMPTY || m.bit_width.is_none())?; + .enumerate() + .find(|(_, m)| m.name != StringId::EMPTY || m.bit_width.is_none())?; *current_field_idx = members.len(); - return Some(MemberInfo { - offset: member.offset, - typ: member.typ, - bit_offset: member.bit_offset, - bit_width: member.bit_width, - access_bytes: member.access_bytes, - }); + return Some(( + MemberInfo { + offset: member.offset, + typ: member.typ, + bit_offset: member.bit_offset, + bit_width: member.bit_width, + access_bytes: member.access_bytes, + }, + index, + )); } while *current_field_idx < members.len() { - let member = &members[*current_field_idx]; + let index = *current_field_idx; + let member = &members[index]; if member.name == StringId::EMPTY && member.bit_width.is_some() { *current_field_idx += 1; continue; } if member.name != StringId::EMPTY || member.bit_width.is_none() { *current_field_idx += 1; - return Some(MemberInfo { - offset: member.offset, - typ: member.typ, - bit_offset: member.bit_offset, - bit_width: member.bit_width, - access_bytes: member.access_bytes, - }); + return Some(( + MemberInfo { + offset: member.offset, + typ: member.typ, + bit_offset: member.bit_offset, + bit_width: member.bit_width, + access_bytes: member.access_bytes, + }, + index, + )); } *current_field_idx += 1; } @@ -1769,3 +2607,859 @@ impl<'a> super::linearize::Linearizer<'a> { Err(AliasFault::Undefined) } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::parse::ast::Expr; + use crate::symbol::SymbolTable; + use crate::target::Target; + use crate::types::Type; + + /// A positional initializer element holding an `int` constant. + fn positional(value: i64, types: &TypeTable) -> InitElement { + InitElement { + designators: vec![], + value: Box::new(Expr::int(value, types)), + } + } + + /// The same, addressed by a designator. + fn designated(designator: Designator, value: i64, types: &TypeTable) -> InitElement { + InitElement { + designators: vec![designator], + value: Box::new(Expr::int(value, types)), + } + } + + /// The indices `group_array_init_elements` keeps for `elements` when they + /// initialize an `int` array of `size` elements (`None`: a bound derived + /// from the initializer itself, as for `int a[] = {1, 2, 3}`). + fn kept_indices( + size: Option, + build: impl Fn(&TypeTable) -> Vec, + ) -> Vec { + let target = Target::host(); + let mut types = TypeTable::new(&target); + let elements = build(&types); + let array = types.intern(Type { + kind: TypeKind::Array, + base: Some(types.int_id), + array_size: size, + ..Default::default() + }); + let symbols = SymbolTable::new(); + let strings = crate::strings::StringTable::new(); + let lin = Linearizer::new(&symbols, &types, &strings, &target); + let groups = lin.group_array_init_elements(&elements, array); + assert_eq!(groups.indices.len(), groups.element_lists.len()); + groups.indices + } + + #[test] + fn excess_array_elements_are_discarded() { + // `int a[2] = {1, 2, 3};` -- the third element has nowhere to go. + let indices = kept_indices(Some(2), |types| { + (1..=3).map(|v| positional(v, types)).collect() + }); + assert_eq!(indices, vec![0, 1]); + } + + #[test] + fn a_positional_element_past_a_designator_can_be_excess() { + // `int a[3] = {[2] = 3, 1};` -- C17 6.7.9p17 resumes at index 3, so + // the `1` is excess although the list is shorter than the array. + let indices = kept_indices(Some(3), |types| { + vec![ + designated(Designator::Index(2), 3, types), + positional(1, types), + ] + }); + assert_eq!(indices, vec![2]); + } + + #[test] + fn a_range_designator_is_clamped_to_the_array() { + // `int a[3] = {[0 ... 4] = 7};` initializes the three elements it has. + let indices = kept_indices(Some(3), |types| { + vec![designated(Designator::IndexRange(0, 4), 7, types)] + }); + assert_eq!(indices, vec![0, 1, 2]); + } + + #[test] + fn an_initializer_derived_bound_has_no_excess() { + // `int a[] = {1, 2, 3};` and a flexible array member are sized by the + // initializer, so every element belongs to the object. + let indices = kept_indices(None, |types| { + (1..=3).map(|v| positional(v, types)).collect() + }); + assert_eq!(indices, vec![0, 1, 2]); + let indices = kept_indices(Some(0), |types| { + (1..=3).map(|v| positional(v, types)).collect() + }); + assert_eq!(indices, vec![0, 1, 2]); + } + + /// The static image of `int a[2] = {1, 2, 3};` is two elements wide, not + /// three: the excess element used to become a third `.long` under a + /// two-element symbol, which the next symbol in the section absorbed. + #[test] + fn a_static_array_image_holds_no_excess_element() { + let target = Target::host(); + let mut types = TypeTable::new(&target); + let elements: Vec<_> = (1..=3).map(|v| positional(v, &types)).collect(); + let array = types.intern(Type::array(types.int_id, 2)); + let symbols = SymbolTable::new(); + let strings = crate::strings::StringTable::new(); + let mut lin = Linearizer::new(&symbols, &types, &strings, &target); + let init = lin.ast_init_list_to_ir(&elements, array); + let Initializer::Array { + elem_size, + total_size, + elements, + } = init + else { + panic!("an array initializer lowers to Initializer::Array, got {init:?}"); + }; + assert_eq!(elem_size, 4); + assert_eq!(total_size, 8); + assert_eq!( + elements + .iter() + .map(|(offset, init)| (*offset, init.clone())) + .collect::>(), + vec![(0, Initializer::Int(1)), (4, Initializer::Int(2))], + ); + } + + // Resolving two initializers that describe overlapping storage + // (C17 6.7.9p19). + + /// Types for the override tests: + /// + /// ```c + /// struct T { int x, y; }; + /// struct S { struct T t; int z; }; + /// struct A { int a[3]; int z; }; + /// struct W { union { int i; struct { char a, b, c, d; } s; } u; }; + /// struct B { unsigned a : 4, b : 4; }; + /// struct C { struct B t; int z; }; + /// struct N { union { struct T p; int i; } u; }; + /// ``` + struct OverrideTypes { + target: Target, + types: TypeTable, + strings: crate::strings::StringTable, + symbols: SymbolTable, + t: TypeId, + s: TypeId, + a: TypeId, + w: TypeId, + v: TypeId, + b: TypeId, + c: TypeId, + n: TypeId, + /// The `union { struct T p; int i; }` inside `struct N`. + n_union: TypeId, + int_array: TypeId, + } + + fn member(name: StringId, typ: TypeId, offset: usize) -> crate::types::StructMember { + crate::types::StructMember { + name, + typ, + offset, + bit_offset: None, + bit_width: None, + access_bytes: None, + explicit_align: None, + } + } + + /// A bit-field of `width` bits at `bit_offset` of a four-byte access span + /// based at byte 0. + fn bitfield( + name: StringId, + typ: TypeId, + bit_offset: u32, + width: u32, + ) -> crate::types::StructMember { + crate::types::StructMember { + name, + typ, + offset: 0, + bit_offset: Some(bit_offset), + bit_width: Some(width), + access_bytes: Some(4), + explicit_align: None, + } + } + + fn composite( + members: Vec, + size: usize, + align: usize, + ) -> crate::types::CompositeType { + crate::types::CompositeType { + members, + size, + align, + member_align: align, + is_complete: true, + ..crate::types::CompositeType::incomplete(None) + } + } + + impl OverrideTypes { + fn new() -> Self { + let target = Target::host(); + let mut types = TypeTable::new(&target); + let mut strings = crate::strings::StringTable::new(); + let name = |strings: &mut crate::strings::StringTable, s: &str| strings.intern(s); + + let int = types.int_id; + let ch = types.char_id; + + let (x, y) = (name(&mut strings, "x"), name(&mut strings, "y")); + let t = types.intern(Type::struct_type(composite( + vec![member(x, int, 0), member(y, int, 4)], + 8, + 4, + ))); + + let (t_name, z) = (name(&mut strings, "t"), name(&mut strings, "z")); + let s = types.intern(Type::struct_type(composite( + vec![member(t_name, t, 0), member(z, int, 8)], + 12, + 4, + ))); + + let int_array = types.intern(Type::array(int, 3)); + let a_name = name(&mut strings, "a"); + let a = types.intern(Type::struct_type(composite( + vec![member(a_name, int_array, 0), member(z, int, 12)], + 16, + 4, + ))); + + let (b, c, d) = ( + name(&mut strings, "b"), + name(&mut strings, "c"), + name(&mut strings, "d"), + ); + let chars = types.intern(Type::struct_type(composite( + vec![ + member(a_name, ch, 0), + member(b, ch, 1), + member(c, ch, 2), + member(d, ch, 3), + ], + 4, + 1, + ))); + let (i, s_name) = (name(&mut strings, "i"), name(&mut strings, "s")); + let union_u = types.intern(Type::union_type(composite( + vec![member(i, int, 0), member(s_name, chars, 0)], + 4, + 4, + ))); + let u_name = name(&mut strings, "u"); + let w = types.intern(Type::struct_type(composite( + vec![member(u_name, union_u, 0)], + 4, + 4, + ))); + + // `struct V { int k; union { int i; struct { char a, b, c, d; } s; } u; }` + // -- the union has a sibling, so resetting it must leave `k` be. + let k = name(&mut strings, "k"); + let v = types.intern(Type::struct_type(composite( + vec![member(k, int, 0), member(u_name, union_u, 4)], + 8, + 4, + ))); + + // `struct B { unsigned a : 4, b : 4; }` and the `struct C { struct + // B t; int z; }` that holds it: two bit-fields in one carrier byte, + // with an ordinary member beside them. + let uint = types.uint_id; + let bits = types.intern(Type::struct_type(composite( + vec![bitfield(a_name, uint, 0, 4), bitfield(b, uint, 4, 4)], + 4, + 4, + ))); + let bits_holder = types.intern(Type::struct_type(composite( + vec![member(t_name, bits, 0), member(z, int, 4)], + 8, + 4, + ))); + + // `struct N { union { struct T p; int i; } u; }` -- a union whose + // members differ in size, so that an override naming a subobject + // of the one it holds has somewhere to fold into. + let p_name = name(&mut strings, "p"); + let n_union = types.intern(Type::union_type(composite( + vec![member(p_name, t, 0), member(i, int, 0)], + 8, + 4, + ))); + let n = types.intern(Type::struct_type(composite( + vec![member(u_name, n_union, 0)], + 8, + 4, + ))); + + Self { + target, + types, + strings, + symbols: SymbolTable::new(), + t, + s, + a, + w, + v, + b: bits, + c: bits_holder, + n, + n_union, + int_array, + } + } + + fn linearizer(&self) -> Linearizer<'_> { + Linearizer::new(&self.symbols, &self.types, &self.strings, &self.target) + } + } + + /// One entry of a struct initializer list, as `walk_struct_init_fields` + /// hands it over: the subobject it names and the value for it. + fn raw(offset: usize, typ: TypeId, size: usize, init: Initializer) -> RawFieldInit { + RawFieldInit { + offset, + field_size: size, + typ, + init, + bit_offset: None, + bit_width: None, + held: UnionMembers::default(), + named: UnionMembers::default(), + } + } + + /// The same for a bit-field: `offset` is its access span's first byte and + /// the value goes at `bit_offset..bit_offset + width` of that span. + fn raw_bits( + offset: usize, + typ: TypeId, + bit_offset: u32, + width: u32, + value: i128, + ) -> RawFieldInit { + RawFieldInit { + bit_offset: Some(bit_offset), + bit_width: Some(width), + ..raw(offset, typ, 4, Initializer::Int(value)) + } + } + + /// One `(byte offset, union type, member index)` recorded for a union. + fn union_member(offset: usize, typ: TypeId, index: usize) -> UnionMembers { + let mut members = UnionMembers::default(); + members.record(offset, typ, index); + members + } + + fn struct_init(total_size: usize, fields: &[(usize, usize, i128)]) -> Initializer { + Initializer::Struct { + total_size, + fields: fields + .iter() + .map(|(off, size, value)| (*off, *size, Initializer::Int(*value))) + .collect(), + } + } + + /// `(offset, size, initializer)` for each entry the merge kept, in the + /// order it kept them. + fn kept(merged: &[RawFieldInit]) -> Vec<(usize, usize, Initializer)> { + merged + .iter() + .map(|f| (f.offset, f.field_size, f.init.clone())) + .collect() + } + + #[test] + fn a_range_is_a_member_when_struct_members_and_array_elements_reach_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // The whole object, the member `t`, and `t.y` inside it. + assert_eq!( + lin.classify_subobject(fixture.s, 0, 12, UnionFold::default()), + SubobjectPlace::Member + ); + assert_eq!( + lin.classify_subobject(fixture.s, 0, 8, UnionFold::default()), + SubobjectPlace::Member + ); + assert_eq!( + lin.classify_subobject(fixture.s, 4, 4, UnionFold::default()), + SubobjectPlace::Member + ); + // `a[1]` of `struct A`. + assert_eq!( + lin.classify_subobject(fixture.a, 4, 4, UnionFold::default()), + SubobjectPlace::Member + ); + // The four bytes straddling `t.y` and `z` are no object at all. + assert_eq!( + lin.classify_subobject(fixture.s, 6, 4, UnionFold::default()), + SubobjectPlace::NotASubobject + ); + } + + #[test] + fn a_range_inside_a_union_member_is_reached_through_the_union() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // `u.s.b` -- one byte, reached only by choosing a union member. + assert_eq!( + lin.classify_subobject(fixture.w, 1, 1, UnionFold::default()), + SubobjectPlace::ThroughUnion { offset: 0, size: 4 } + ); + // The union named exactly is an ordinary member of `struct W`. + assert_eq!( + lin.classify_subobject(fixture.w, 0, 4, UnionFold::default()), + SubobjectPlace::Member + ); + } + + /// When both initializers name the same member of a union, the range is + /// an ordinary subobject reached through it and the rest of that member + /// stands. When they name different ones, it is not. + #[test] + fn a_union_is_descended_into_only_where_both_sides_name_the_same_member() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // `n.u.p.y` of `struct N`, with the union holding `p`. + let holds_p = union_member(0, fixture.n_union, 0); + let names_p = union_member(0, fixture.n_union, 0); + let names_i = union_member(0, fixture.n_union, 1); + assert_eq!( + lin.classify_subobject(fixture.n, 4, 4, UnionFold::new(&holds_p, &names_p, 0)), + SubobjectPlace::Member + ); + assert_eq!( + lin.classify_subobject(fixture.n, 4, 4, UnionFold::new(&holds_p, &names_i, 0)), + SubobjectPlace::ThroughUnion { offset: 0, size: 8 } + ); + // And with nothing said about it at all. + assert_eq!( + lin.classify_subobject(fixture.n, 4, 4, UnionFold::default()), + SubobjectPlace::ThroughUnion { offset: 0, size: 8 } + ); + } + + /// `struct S s = { .t = {1, 2}, .t.y = 9, .z = 7 };` -- the override + /// names `t.y`, so `t.x` keeps the 1 it was given. + #[test] + fn a_contained_override_replaces_only_the_subobject_it_names() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + raw(4, int, 4, Initializer::Int(9)), + raw(8, int, 4, Initializer::Int(7)), + ]); + + assert_eq!( + kept(&merged), + vec![ + (0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)])), + (8, 4, Initializer::Int(7)), + ] + ); + } + + /// The same one level further down: `{ .a = {1,2,3}, .a[1] = 9 }` keeps + /// elements 0 and 2. + #[test] + fn a_contained_override_replaces_only_the_array_element_it_names() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let array = Initializer::Array { + elem_size: 4, + total_size: 12, + elements: vec![ + (0, Initializer::Int(1)), + (4, Initializer::Int(2)), + (8, Initializer::Int(3)), + ], + }; + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.int_array, 12, array), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![( + 0, + 12, + Initializer::Array { + elem_size: 4, + total_size: 12, + elements: vec![ + (0, Initializer::Int(1)), + (4, Initializer::Int(9)), + (8, Initializer::Int(3)), + ], + } + )] + ); + } + + /// An initializer for a whole subobject replaces every earlier one for a + /// part of it. + #[test] + fn a_containing_override_replaces_the_earlier_entry_whole() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(4, int, 4, Initializer::Int(9)), + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + ]); + + assert_eq!( + kept(&merged), + vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))] + ); + } + + /// Initializers for different members stand side by side. + #[test] + fn disjoint_entries_are_all_kept() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, int, 4, Initializer::Int(1)), + raw(4, int, 4, Initializer::Int(2)), + raw(8, int, 4, Initializer::Int(7)), + ]); + + assert_eq!( + kept(&merged), + vec![ + (0, 4, Initializer::Int(1)), + (4, 4, Initializer::Int(2)), + (8, 4, Initializer::Int(7)), + ] + ); + } + + /// `{ .u.i = 0x01020304, .u.s.b = 9 }` -- a union holds one member at a + /// time, so naming a second discards what the first wrote rather than + /// overlaying it. + #[test] + fn initializing_a_second_union_member_discards_the_first() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let (int, ch) = (fixture.types.int_id, fixture.types.char_id); + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, int, 4, Initializer::Int(0x01020304)), + raw(1, ch, 1, Initializer::Int(9)), + ]); + + assert_eq!(kept(&merged), vec![(1, 1, Initializer::Int(9))]); + } + + /// An initializer for a whole struct, then one byte of a different member + /// of a union inside it: only that union is reset, and the struct's other + /// members keep what they were given. + /// + /// `struct V v = { .k = 5, .u.i = 0x01020304 }` followed by `.u.s.b = 9`. + #[test] + fn an_override_through_a_nested_union_resets_only_that_union() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let ch = fixture.types.char_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw( + 0, + fixture.v, + 8, + struct_init(8, &[(0, 4, 5), (4, 4, 0x01020304)]), + ), + raw(5, ch, 1, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![( + 0, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![ + (0, 4, Initializer::Int(5)), + (4, 4, struct_init(4, &[(1, 1, 9)])), + ], + } + )] + ); + } + + /// A struct whose only member is a union is the union, byte for byte, so + /// resetting the union replaces the whole entry. + #[test] + fn an_override_through_a_union_filling_its_struct_replaces_the_entry() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let ch = fixture.types.char_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.w, 4, struct_init(4, &[(0, 4, 0x01020304)])), + raw(1, ch, 1, Initializer::Int(9)), + ]); + + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(1, 1, 9)]))]); + } + + /// "Later wins" is later in the list, not at a higher address: + /// `{ .z = 7, .t.y = 9, .t = {1, 2} }` ends with `t` holding `{1, 2}`. + #[test] + fn the_override_rule_is_applied_in_source_order() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(8, int, 4, Initializer::Int(7)), + raw(4, int, 4, Initializer::Int(9)), + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + ]); + + assert_eq!( + kept(&merged), + vec![ + (8, 4, Initializer::Int(7)), + (0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + ] + ); + + // And the other order, where the narrower one is written last. + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + raw(8, int, 4, Initializer::Int(7)), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![ + (0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)])), + (8, 4, Initializer::Int(7)), + ] + ); + } + + /// A member the earlier initializer left out gains an entry of its own + /// rather than costing the whole earlier initializer: `{ .t = {1}, .t.y + /// = 9 }` keeps `t.x`. + #[test] + fn an_override_of_an_uninitialized_member_is_added_to_the_earlier_entry() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1)])), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)]))] + ); + } + + /// A bit-field's bytes are its carrier's, not its own, so the answer for + /// them is the struct that declares it -- and they are no subobject at + /// all when asked for as an object. + #[test] + fn a_bitfields_bytes_name_the_struct_that_declares_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // `t.a` of `struct C`: byte 0, inside `t` at 0..4. + assert_eq!( + lin.classify_bitfield_carrier(fixture.c, 0, 1, UnionFold::default()), + SubobjectPlace::BitfieldCarrier { offset: 0, size: 4 } + ); + // Asked for `struct B` itself, the carrier byte is the whole of the + // declaring struct and still must not be replaced whole. + assert_eq!( + lin.classify_bitfield_carrier(fixture.b, 0, 1, UnionFold::default()), + SubobjectPlace::BitfieldCarrier { offset: 0, size: 4 } + ); + assert_eq!( + lin.classify_subobject(fixture.c, 0, 1, UnionFold::default()), + SubobjectPlace::NotASubobject + ); + // Padding inside a struct with bit-fields is still nothing at all. + assert_eq!( + lin.classify_bitfield_carrier(fixture.c, 1, 1, UnionFold::default()), + SubobjectPlace::NotASubobject + ); + } + + /// `struct C c = { .t = {1, 2}, .t.a = 3 };` -- the override names one + /// bit-field, so the one sharing its carrier keeps its value. + #[test] + fn a_designated_override_of_a_bitfield_replaces_only_its_bits() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let uint = fixture.types.uint_id; + + // `{1, 2}` has already been lowered to the carrier byte it produces. + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.b, 4, struct_init(4, &[(0, 1, 0x21)])), + raw_bits(0, uint, 0, 4, 3), + ]); + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(0, 1, 0x23)]))]); + + // And the other bit-field of the pair. + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.b, 4, struct_init(4, &[(0, 1, 0x21)])), + raw_bits(0, uint, 4, 4, 5), + ]); + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(0, 1, 0x51)]))]); + } + + /// A bit-field never supersedes an initializer for an object containing + /// it, but a whole-struct initializer written after one does. + #[test] + fn a_whole_struct_initializer_supersedes_an_earlier_bitfield() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let uint = fixture.types.uint_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw_bits(0, uint, 4, 4, 5), + raw(0, fixture.b, 4, struct_init(4, &[(0, 1, 0x21)])), + ]); + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(0, 1, 0x21)]))]); + } + + /// `struct N n = { .u = {1, 2}, .u.p.y = 9 };` -- the union still holds + /// `p`, and the override names a member *of* `p`, so `p.x` keeps its 1. + #[test] + fn an_override_inside_the_held_union_member_folds_into_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let held = union_member(0, fixture.n_union, 0); + let named = union_member(0, fixture.n_union, 0); + let merged = lin.merge_raw_field_inits(vec![ + RawFieldInit { + held, + ..raw( + 0, + fixture.n_union, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))], + }, + ) + }, + RawFieldInit { + named, + ..raw(4, int, 4, Initializer::Int(9)) + }, + ]); + + assert_eq!( + kept(&merged), + vec![( + 0, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)]))], + } + )] + ); + } + + /// The companion case that makes the rule a rule: naming a *different* + /// member discards what the union held, however the two spans overlap. + #[test] + fn an_override_naming_a_different_union_member_still_resets_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + RawFieldInit { + held: union_member(0, fixture.n_union, 0), + ..raw( + 0, + fixture.n_union, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))], + }, + ) + }, + RawFieldInit { + // `.u.i`, the union's other member. + named: union_member(0, fixture.n_union, 1), + ..raw(0, int, 4, Initializer::Int(7)) + }, + ]); + + assert_eq!(kept(&merged), vec![(0, 8, struct_init(8, &[(0, 4, 7)]))]); + } + + /// With nothing said about the union, the merge cannot tell "the same + /// member, deeper" from "a different member" and takes the reset, which + /// is the answer that discards rather than invents. + #[test] + fn an_override_through_a_union_nothing_names_resets_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw( + 0, + fixture.n_union, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))], + }, + ), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!(kept(&merged), vec![(0, 8, struct_init(8, &[(4, 4, 9)]))]); + } +} diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index ea105972c..e493ed50e 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -37,6 +37,28 @@ fn return_value_ness_violation(pos: Position, msg: &str) { use crate::types::{TypeId, TypeKind, TypeModifiers}; +/// The `-Wno-` group for a case label the switch's promoted controlling +/// type cannot hold, spelled as gcc spells the same diagnostic. +const CASE_RANGE_WARNING: &str = "switch-outside-range"; + +/// Whether [`Linearizer::store_string_units`] owes the destination's tail a +/// zero fill. +/// +/// C17 6.7.9p21 zeroes every element a string initializer does not reach, and +/// exactly one of the two spellings already has that covered: the braced form +/// reaches the array through an initializer list, and each list is preceded by +/// a whole-object [`Linearizer::emit_aggregate_zero`]. Zeroing again there +/// would double the stores at `-O0`, where no `dse` runs to remove them. +#[derive(Clone, Copy)] +pub(crate) enum StringTail { + /// The destination is already zero: the caller zeroed the whole aggregate + /// before walking the initializer list. + AlreadyZero, + /// Nothing has written the destination yet, so the tail is this call's to + /// fill. + Zero, +} + /// Which construct a jump leaves, for `unwind_vla_marks`. /// /// `break` leaves the innermost loop *or* switch; `continue` leaves the @@ -49,6 +71,69 @@ enum JumpKind { Continue, } +/// A `for` loop whose body is being lowered: what +/// [`Linearizer::open_for`] set up and [`Linearizer::close_for`] finishes. +/// +/// `for` is lowered from two places -- the ordinary statement walk and the +/// switch-body walk, which must keep lowering the body through itself so the +/// `case` labels inside stay reachable -- and the two differ *only* in how +/// they lower the body. Everything else lives here, so a fix lands once +/// instead of twice: the loop's back edge had to be repaired in both copies, +/// and the release of a VLA declared in the init clause was missing from +/// both. +#[must_use = "an opened `for` loop must be finished with close_for"] +struct OpenFor { + /// The scope the init clause declares into, ended after `exit_bb`. + scope: Scope, + /// The block the back edge goes to. + cond_bb: BasicBlockId, + /// Where the body falls out to, and the `continue` target. + post_bb: BasicBlockId, + /// Where the loop ends, and the `break` target. + exit_bb: BasicBlockId, +} + +/// One initializer from a struct or union initializer list that has already +/// been stored into the object being initialized. +/// +/// The automatic path emits a store per list entry and lets a later store land +/// on an earlier one, which is all C17 6.7.9p19 needs *while* the later store +/// covers every byte it supersedes. It does not when the later initializer +/// fills only part of the subobject it names, and it does not when the two +/// entries name different members of a union -- a union holds one member at a +/// time, so the member left behind does not show through the new one. Both +/// need the superseded bytes cleared, and deciding which bytes those are is +/// what this records. +struct WrittenInit { + /// The bytes it wrote that no later entry has since cleared. + live: std::ops::Range, + /// The first byte of the subobject it initialized, and that subobject's + /// type. Together they say whether a later entry names a member of this + /// one -- in which case the rest of this one survives -- or reaches its + /// bytes only by passing through a union. + origin: usize, + typ: TypeId, + /// The member each union inside that subobject came to hold, which + /// decides whether a later entry naming something inside one of them + /// names the *same* member -- and so leaves the rest of it standing. + held: UnionMembers, + /// Set when the entry is a bit-field, to its bit offset. Two bit-fields + /// sharing a carrier byte are different objects and neither supersedes + /// the other, however their bytes overlap. + bits: Option, +} + +/// Grow `reset` to also cover `range`. +/// +/// Every range joined here overlaps the range of the entry being stored, so +/// the union of them all is contiguous and one fill covers it. +fn widen_reset(reset: &mut Option>, range: std::ops::Range) { + *reset = Some(match reset.take() { + Some(cur) => cur.start.min(range.start)..cur.end.max(range.end), + None => range, + }); +} + impl<'a> super::linearize::Linearizer<'a> { pub(crate) fn linearize_stmt(&mut self, stmt: &Stmt) { match stmt { @@ -59,24 +144,18 @@ impl<'a> super::linearize::Linearizer<'a> { } Stmt::Block(items) => { - self.push_scope(); - // A VLA's storage lives until control leaves the scope of its - // declaration (C17 6.2.4p7). Capture the stack pointer on the - // way in and put it back on the way out, or a loop body's VLA - // is allocated afresh every iteration and never released -- - // `for (...) { int x[n]; }` died of stack exhaustion. - let vla_scope = self.open_vla_scope(); + // Entering the scope also captures the stack pointer, and + // leaving it puts the pointer back -- a VLA's storage lives + // until control leaves the scope of its declaration (C17 + // 6.2.4p7). See `Scope`. + let scope = self.push_scope(); for item in items { match item { BlockItem::Declaration(decl) => self.linearize_local_decl(decl), BlockItem::Statement(s) => self.linearize_stmt(s), } } - // Only on the falling-out path: a `break`, `continue` or - // `return` that left already did its own unwinding. - self.close_vla_scope(vla_scope); - - self.pop_scope(); + self.pop_scope(scope); } Stmt::If { @@ -257,6 +336,13 @@ impl<'a> super::linearize::Linearizer<'a> { ); } let addr = self.linearize_expr(target); + // A computed `goto` leaves scopes exactly as a named one + // does, and left out here it left a loop body's VLA behind + // every time round. Which scopes it leaves depends on which + // label it reaches, which is not known until every label is + // placed -- and then only as a set. See + // `GotoTarget::AnyAddressTaken`. + self.defer_vla_restore(GotoTarget::AnyAddressTaken); let (dispatch_bb, slot) = self.indirect_dispatch_block(); // Hand the address over in the hidden local and branch to the // one dispatch block, which is the only place that fans out to @@ -274,44 +360,7 @@ impl<'a> super::linearize::Linearizer<'a> { let label_str = self.str(*label).to_string(); let target = self.refer_to_label(&label_str, *pos); if let Some(current) = self.current_bb { - // A backward jump -- the label is already linearized, so - // it has a captured stack pointer -- leaves the scope of - // every VLA declared after it, and that storage has to go - // back. Otherwise `lab: int x[n]; ... goto lab;` grows the - // stack every time round until the program dies. - // - // A *forward* jump cannot be decided here: its label has - // no depth recorded yet, and whether it stays inside the - // scope of the VLAs in force or leaves it is exactly what - // decides between no restore and one. It is recorded and - // resolved in `resolve_forward_goto_vla_restores`. - // - // Leaving it to "the block's own exit does the restoring" - // was wrong: the branch *terminates* the block, so - // `close_vla_scope` emits nothing and then drops the - // marks, and the enclosing block has no mark of its own to - // undo them with. A `goto` out of a loop body's inner - // block grew the stack every time round. - if let Some(&depth) = self.label_vla_depth.get(&label_str) { - if let Some(m) = self.vla_marks.get(depth) { - let mark = m.mark; - self.emit_stack_restore(mark); - } - } else if !self.vla_marks.is_empty() { - let at = self - .current_func - .as_ref() - .and_then(|f| f.get_block(current)) - .map_or(0, |b| b.insns.len()); - let marks = self.vla_marks.iter().map(|m| m.mark).collect(); - self.pending_goto_vla - .push(crate::ir::linearize::PendingGotoVla { - label: label_str.clone(), - bb: current, - at, - marks, - }); - } + self.release_vla_scopes_for_goto(target); self.emit(Instruction::br(target)); self.link_bb(current, target); } @@ -545,7 +594,18 @@ impl<'a> super::linearize::Linearizer<'a> { // scalar case below and stored the literal's *address* // into the array's first element. if self.types.kind(typ) == TypeKind::Array { - self.store_string_units(sym_id, 0, typ, &init.kind, &units); + // Nothing has zeroed this local -- the `InitList` arm + // above calls `emit_aggregate_zero` and this one never + // did -- so the elements past the literal are + // `store_string_units`' to fill. + self.store_string_units( + sym_id, + 0, + typ, + &init.kind, + &units, + StringTail::Zero, + ); } else { // Pointer initialized with a string literal — store the address let val = self.linearize_expr(init); @@ -956,12 +1016,15 @@ impl<'a> super::linearize::Linearizer<'a> { if let [only] = elements { if only.designators.is_empty() { if let Some(units) = Self::string_literal_units(&only.value.kind) { + // Every initializer list is preceded by a + // whole-object zero, so the tail is done. self.store_string_units( base_sym, base_offset, typ, &only.value.kind, &units, + StringTail::AlreadyZero, ); return; } @@ -975,59 +1038,39 @@ impl<'a> super::linearize::Linearizer<'a> { TypeKind::Array | TypeKind::Struct | TypeKind::Union ); - let groups = self.group_array_init_elements(elements, elem_type); + let groups = self.group_array_init_elements(elements, typ); for element_index in groups.indices { let Some(list) = groups.element_lists.get(&element_index) else { continue; }; let offset = base_offset + element_index * elem_size as i64; - // When a string literal initializes a char array element - // (e.g., char arr[3][4] = {"Sun", "Mon", "Tue"}), handle - // it as a string copy rather than recursing into individual - // char stores. The recursion would treat the string as a - // pointer instead of inline data. - let is_string_for_char_array = elem_is_aggregate - && list.len() == 1 - && matches!( - list[0].value.kind, - ExprKind::StringLit(_) - | ExprKind::WideStringLit(_) - | ExprKind::Utf16StringLit(_) - | ExprKind::Utf32StringLit(_) - ) - && self.types.kind(elem_type) == TypeKind::Array; - if is_string_for_char_array { - // Emit byte-by-byte stores for the string content - if let ExprKind::StringLit(s) = &list[0].value.kind { - let char_type = self - .types - .base_type(elem_type) - .unwrap_or(self.types.char_id); - let char_bits = self.types.size_bits(char_type); - for (i, ch) in s.chars().enumerate() { - let byte_val = self.emit_const(ch as u8 as i128, self.types.int_id); - self.emit(Instruction::store( - byte_val, - base_sym, - offset + i as i64, - char_type, - char_bits, - )); - } - // Null terminator + zero fill - let arr_bytes = self.types.size_bytes(elem_type); - let str_len = s.chars().count(); - for i in str_len..arr_bytes { - let zero = self.emit_const(0, self.types.int_id); - self.emit(Instruction::store( - zero, - base_sym, - offset + i as i64, - char_type, - char_bits, - )); - } - } + // A string literal initializing an array element -- + // `char arr[3][4] = {"Sun", "Mon", "Tue"}` -- is inline + // data, not one element. Recursing into it would treat the + // literal as the pointer it decays to everywhere else. + // + // Shared with the other three string-store paths rather + // than counted a third way here. Written out, this loop + // recognized all four literal kinds and then handled only + // `StringLit`, dropping a wide element and leaving it + // zero; stepped the destination by raw *bytes* where a + // wide element is 2 or 4 bytes wide; and had no capacity + // clamp at all, so `char s[1][3] = {"hello"}` stored five + // bytes into a three-byte object -- two of them past the + // whole local, not merely into the next row. + let string_element = (list.len() == 1 + && self.types.kind(elem_type) == TypeKind::Array) + .then(|| Self::string_literal_units(&list[0].value.kind)) + .flatten(); + if let Some(units) = string_element { + self.store_string_units( + base_sym, + offset, + elem_type, + &list[0].value.kind, + &units, + StringTail::AlreadyZero, + ); continue; } if elem_is_aggregate { @@ -1086,7 +1129,23 @@ impl<'a> super::linearize::Linearizer<'a> { let visits = self.walk_struct_init_fields(resolved_typ, &members, is_union, elements); + // C17 6.7.9p19 resolves two initializers for overlapping + // storage by subobject. Storing them in list order is + // enough only where the later store covers every byte it + // supersedes; where it does not, the bytes it leaves have + // to be cleared first. See [`WrittenInit`]. + let mut written: Vec = Vec::new(); + for visit in visits { + let held = self.held_union_members(visit.typ, &visit.kind, visit.offset); + if let Some(reset) = self.init_override_reset(&mut written, &visit, held) { + self.emit_block_zero( + base_sym, + base_offset + reset.start as i64, + (reset.end - reset.start) as i64, + ); + } + let offset = base_offset + visit.offset as i64; let field_type = visit.typ; @@ -1119,6 +1178,7 @@ impl<'a> super::linearize::Linearizer<'a> { bit_w, storage_size, member_val, + field_type, ); } else { self.linearize_struct_field_init( @@ -1155,6 +1215,131 @@ impl<'a> super::linearize::Linearizer<'a> { } } + /// Record that `visit` is about to be stored, and answer which bytes of + /// the object must be cleared first for C17 6.7.9p19 to hold. + /// + /// Nothing at all, for the usual case where the entry overlaps none of + /// those already stored. Otherwise the entry's own bytes -- so that the + /// part of the subobject it does not fill reads as zero rather than as + /// the initializer it replaces -- together with the bytes of any earlier + /// entry it invalidates: all of one it wholly contains, all of one it is + /// not a subobject of, and, where it reaches an earlier entry's bytes only + /// by naming a member of a union inside it, that union's bytes. + /// + /// A bit-field is stored by reading its carrier and writing it back, so + /// its own bytes are not storage it owns and clearing them would blank the + /// members sharing the carrier. It therefore never clears its own span -- + /// only what an earlier entry it supersedes requires, which is how a + /// bit-field naming a second member of a union still resets it. + fn init_override_reset( + &self, + written: &mut Vec, + visit: &StructFieldVisit, + held: UnionMembers, + ) -> Option> { + if visit.field_size == 0 || visit.bit_width == Some(0) { + return None; + } + let bits = visit.bit_offset.filter(|_| visit.bit_width.is_some()); + let span = match (visit.bit_offset, visit.bit_width) { + // Only the bytes the field's own bits reach; its access span is + // wider and covers bytes other members own. + (Some(bit_offset), Some(bit_width)) => { + let start = visit.offset + (bit_offset / 8) as usize; + let end = visit.offset + (bit_offset + bit_width).div_ceil(8) as usize; + start..end.max(start + 1) + } + _ => visit.offset..visit.offset + visit.field_size, + }; + let mut reset: Option> = None; + + for entry in written.iter() { + if entry.live.start >= span.end || span.start >= entry.live.end { + continue; + } + // Two bit-fields are different objects even when they share a + // carrier byte, and the same one written twice needs no clearing: + // the second store reads the carrier back and replaces its bits. + if bits.is_some() && entry.bits.is_some() { + continue; + } + if bits.is_none() { + widen_reset(&mut reset, span.clone()); + + // Wholly superseded: the entry's own bytes are all inside the + // ones being cleared and rewritten. + if span.start <= entry.live.start && entry.live.end <= span.end { + continue; + } + } + + let entry_end = entry.origin + self.types.size_bytes(entry.typ); + let inner = span + .start + .checked_sub(entry.origin) + .filter(|_| span.end <= entry_end); + let unions = UnionFold::new(&entry.held, &visit.unions, entry.origin); + let place = match inner { + None => SubobjectPlace::NotASubobject, + Some(inner) if bits.is_some() => { + self.classify_bitfield_carrier(entry.typ, inner, span.end - span.start, unions) + } + Some(inner) => self.classify_subobject(entry.typ, inner, visit.field_size, unions), + }; + match place { + // A member of the earlier entry's object: the rest of that + // object is a different subobject and stands. + SubobjectPlace::Member => {} + SubobjectPlace::ThroughUnion { offset, size } => { + widen_reset( + &mut reset, + entry.origin + offset..entry.origin + offset + size, + ); + } + // The carrier the earlier entry wrote is shared: the bits this + // one names are replaced in place and its neighbours' stand. + SubobjectPlace::BitfieldCarrier { .. } => {} + SubobjectPlace::NotASubobject => { + widen_reset(&mut reset, entry.live.clone()); + } + } + } + + if let Some((from, to)) = reset.as_ref().map(|range| (range.start, range.end)) { + // Whatever the fill covers is gone; the bytes an entry keeps on + // either side of it are still its own. + *written = written + .drain(..) + .flat_map(|entry| { + let (origin, typ, held, bits) = + (entry.origin, entry.typ, entry.held, entry.bits); + [ + entry.live.start..entry.live.end.min(from), + entry.live.start.max(to)..entry.live.end, + ] + .into_iter() + .filter(|live| live.start < live.end) + .map(move |live| WrittenInit { + live, + origin, + typ, + held: held.clone(), + bits, + }) + }) + .collect(); + } + + written.push(WrittenInit { + live: span, + origin: visit.offset, + typ: visit.typ, + held, + bits, + }); + reset + } + /// Store a complex value into `base_sym` at `offset`, as two halves. /// /// A complex value lives in memory and travels by *address*, so storing it @@ -1253,7 +1438,16 @@ impl<'a> super::linearize::Linearizer<'a> { // Shared with the two other string-store paths rather than // counted a third way here. if let Some(units) = Self::string_literal_units(&value.kind) { - self.store_string_units(base_sym, offset, field_type, &value.kind, &units); + // Reached only from an initializer list, which the caller + // zeroed whole before walking it. + self.store_string_units( + base_sym, + offset, + field_type, + &value.kind, + &units, + StringTail::AlreadyZero, + ); } } else { let val = self.linearize_expr(value); @@ -1397,15 +1591,17 @@ impl<'a> super::linearize::Linearizer<'a> { self.switch_bb(exit_bb); } - pub(crate) fn linearize_for( - &mut self, - init: Option<&ForInit>, - cond: Option<&Expr>, - post: Option<&Expr>, - body: &Stmt, - ) { - // C99 for-loop declarations (e.g., for (int i = 0; ...)) are scoped to the loop. - self.push_scope(); + /// Lower everything a `for` loop needs before its body: the init clause, + /// the four blocks, the condition and the jump targets. Leaves the body + /// block current, for the caller to lower the body into. + /// + /// The returned [`OpenFor`] goes back to [`Self::close_for`]. + fn open_for(&mut self, init: Option<&ForInit>, cond: Option<&Expr>) -> OpenFor { + // C99 for-loop declarations (e.g., for (int i = 0; ...)) are scoped + // to the loop -- and so is a VLA declared there, which the scope + // releases at `exit_bb`. That is the right place for it: `break` and + // `continue` both land inside this scope, so neither may release it. + let scope = self.push_scope(); // Init if let Some(init) = init { @@ -1446,9 +1642,27 @@ impl<'a> super::linearize::Linearizer<'a> { // Body block self.break_targets.push(exit_bb); self.continue_targets.push(post_bb); - self.switch_bb(body_bb); - self.linearize_stmt(body); + + OpenFor { + scope, + cond_bb, + post_bb, + exit_bb, + } + } + + /// Close the loop [`Self::open_for`] opened, once its body is lowered: + /// the back edge through the post-expression, then the exit block, then + /// the loop's scope. + fn close_for(&mut self, open: OpenFor, post: Option<&Expr>) { + let OpenFor { + scope, + cond_bb, + post_bb, + exit_bb, + } = open; + if !self.is_terminated() { // After linearizing body, current_bb may be different from body_bb if let Some(current) = self.current_bb { @@ -1465,14 +1679,33 @@ impl<'a> super::linearize::Linearizer<'a> { if let Some(post_expr) = post { self.linearize_expr(post_expr); } - self.emit(Instruction::br(cond_bb)); - self.link_bb(post_bb, cond_bb); + // From the block the post-expression ended in, which `&&`, `||` and + // `?:` can make a different one from post_bb. Linking the back edge + // from post_bb itself recorded an edge out of a block that no longer + // holds the branch, and left the merge block that does hold it with an + // unrecorded successor -- the loop then never terminated. + if let Some(current) = self.current_bb { + self.emit(Instruction::br(cond_bb)); + self.link_bb(current, cond_bb); + } // Exit block self.switch_bb(exit_bb); - // Restore locals to remove for-loop-scoped declarations - self.pop_scope(); + // Drop the for-loop-scoped declarations and release their storage. + self.pop_scope(scope); + } + + pub(crate) fn linearize_for( + &mut self, + init: Option<&ForInit>, + cond: Option<&Expr>, + post: Option<&Expr>, + body: &Stmt, + ) { + let open = self.open_for(init, cond); + self.linearize_stmt(body); + self.close_for(open, post); } pub(crate) fn linearize_switch(&mut self, expr: &Expr, body: &Stmt) { @@ -1506,10 +1739,17 @@ impl<'a> super::linearize::Linearizer<'a> { // Push exit block for break handling self.break_targets.push(exit_bb); - // Collect case labels and create basic blocks for each - let switch_unsigned = self.types.is_unsigned(cmp_type); - let (case_values, has_default) = self.collect_switch_cases(body, switch_unsigned); - let case_bbs: Vec = case_values.iter().map(|_| self.alloc_bb()).collect(); + // Collect case labels and create basic blocks for each. C17 6.8.4.2p5 + // converts every label to `cmp_type`, and `conv` is that conversion: + // the collector applies it, and the body walk below reaches the labels + // back through the same value, so the two cannot drift apart. + let conv = CaseConv::of(self.types, cmp_type); + let (case_values, has_default) = self.collect_switch_cases(body, conv); + let case_bbs: Vec = case_values + .ranges() + .iter() + .map(|_| self.alloc_bb()) + .collect(); let default_bb = if has_default { Some(self.alloc_bb()) } else { @@ -1523,11 +1763,15 @@ impl<'a> super::linearize::Linearizer<'a> { // A constant selector takes one edge, as a constant condition does // (`branch_on`): the labels it does not select are reached only by // falling into them, and a block nothing reaches is not emitted. - // Both sides are values of the promoted type, as the collector - // records the labels. + // Selector and labels are both converted to the promoted type, and + // the range test runs in that type's signedness -- a plain signed + // `i128` comparison answered differently from the runtime lowering + // of the very same switch. + let selector = conv.convert(selector); let target = case_values + .ranges() .iter() - .position(|&(lo, hi)| lo <= selector && selector <= hi) + .position(|&(lo, hi)| conv.contains(lo, hi, selector)) .map_or(default_target, |idx| case_bbs[idx]); if let Some(current) = self.current_bb { self.emit(Instruction::br(target)); @@ -1543,13 +1787,17 @@ impl<'a> super::linearize::Linearizer<'a> { self.emit_wide_switch( switch_val, cmp_type, - &case_values, + conv, + case_values.ranges(), &case_bbs, default_target, ); } else { - // Build switch instruction with case -> block mapping + // Build switch instruction with case -> block mapping. The labels + // have been converted to `cmp_type`, which is at most 64 bits + // here, so the cast keeps every bit of each one. let switch_cases: Vec<(i64, i64, BasicBlockId)> = case_values + .ranges() .iter() .zip(case_bbs.iter()) .map(|((lo, hi), bb)| (*lo as i64, *hi as i64, *bb)) @@ -1582,14 +1830,10 @@ impl<'a> super::linearize::Linearizer<'a> { // block via `switch_bb`. self.current_bb = None; - // Linearize body with case block switching - // Each label's position among the cases, by its range. The first wins, - // as a scan in source order would find it; a duplicate has already - // been reported. - let mut case_index = CaseIndex::new(); - for (idx, range) in case_values.iter().enumerate() { - case_index.entry(*range).or_insert(idx); - } + // Linearize body with case block switching. The index carries the + // collector's own conversion, so the walk finds a label under exactly + // the key the collector filed it under. + let case_index = CaseIndex::of(&case_values); self.linearize_switch_body(body, &case_index, &case_bbs, default_bb); // If not terminated after body, jump to exit @@ -1614,33 +1858,22 @@ impl<'a> super::linearize::Linearizer<'a> { /// only `Stmt::Block` here collected no cases from it -- so the switch was /// emitted with an empty table and every value took the default edge. /// A non-compound body is one statement, so it is walked as one. - pub(crate) fn collect_switch_cases( - &self, - body: &Stmt, - unsigned: bool, - ) -> (Vec<(i128, i128)>, bool) { - let mut case_values = CaseSet::new(unsigned); + pub(crate) fn collect_switch_cases(&self, body: &Stmt, conv: CaseConv) -> (CaseSet, bool) { + let mut case_values = CaseSet::new(conv); let mut has_default = false; match body { Stmt::Block(items) => { for item in items { if let BlockItem::Statement(stmt) = item { - self.collect_cases_from_stmt( - stmt, - &mut case_values, - &mut has_default, - unsigned, - ); + self.collect_cases_from_stmt(stmt, &mut case_values, &mut has_default); } } } - stmt => { - self.collect_cases_from_stmt(stmt, &mut case_values, &mut has_default, unsigned) - } + stmt => self.collect_cases_from_stmt(stmt, &mut case_values, &mut has_default), } - (case_values.ranges, has_default) + (case_values, has_default) } /// The block every computed `goto` in this function branches through, @@ -1844,91 +2077,125 @@ impl<'a> super::linearize::Linearizer<'a> { } } + /// One case endpoint converted to the promoted controlling type, warning + /// if the controlling type cannot hold the constant the label spells. + /// + /// C17 6.8.4.2p5 requires the conversion, and most of the time it changes + /// nothing worth saying: `case -1:` in a `switch` on `unsigned` becomes + /// 4294967295, which is exactly the value it is written to match, and gcc + /// and clang are both silent there. What is worth a diagnostic is a label + /// whose bits the conversion throws away, silently turning + /// `case 4294967296LL:` into `case 0:`. + /// + /// The line between the two is whether converting back to the label's own + /// type returns the constant: a merely reinterpreted value round-trips, + /// while a truncated one does not. That is the test clang applies, and + /// gcc's `-Wswitch-outside-range` draws the line in the same place, which + /// is also the `-Wno-` name that silences this. + fn convert_case_label(&self, expr: &Expr, val: i128, conv: CaseConv) -> i128 { + let converted = conv.convert(val); + if converted == val { + return val; + } + // The label's own type, which the round trip goes back through. An + // untyped or non-integer label is already an error elsewhere; treat it + // as full width, which reduces the round trip to a plain comparison. + let own = expr.typ.filter(|&t| self.types.is_integer(t)).map_or_else( + || CaseConv::new(128, false), + |t| CaseConv::of(self.types, t), + ); + if own.convert(converted) != val && crate::diag::warning_group_enabled(CASE_RANGE_WARNING) { + crate::diag::warning( + expr.pos, + &format!( + "overflow converting case value to switch condition type \ + ({val} to {converted})" + ), + ); + } + converted + } + pub(crate) fn collect_cases_from_stmt( &self, stmt: &Stmt, case_values: &mut CaseSet, has_default: &mut bool, - unsigned: bool, ) { match stmt { Stmt::Case(expr, high, body) => { - self.collect_cases_from_stmt(body, case_values, has_default, unsigned); + self.collect_cases_from_stmt(body, case_values, has_default); // Extract constant value from case expression - if let Some(val) = self.eval_const_expr(expr) { - // Kept at full width. Truncating to `i64` here was silent - // and wrong for a `switch` on `__int128`: a label outside - // the 64-bit range wrapped into it and could match a value - // it does not equal. - - // A GNU range `case lo ... hi:`. An absent high endpoint - // is the ordinary label, held as the degenerate range - // `(v, v)` so that everything downstream has one shape. - let hi = match high { - None => Some(val), - Some(hi_expr) => match self.eval_const_expr(hi_expr) { - Some(h) => Some(h), - None => { - self.report_unfoldable_case(hi_expr); - None - } - }, - }; - let Some(hi) = hi else { return }; - - // 6.8.4.2p3 forbids two equal case constants, and GCC - // extends that to overlapping ranges -- an overlap would - // otherwise make one arm silently unreachable, since the - // body walk resolves a label by finding the first match. - // Order by the switch type's own signedness. The - // endpoints are carried as `i128`, and an unsigned 64-bit - // bound above `i64::MAX` is still positive there -- but an - // unsigned *128-bit* one is not, so the reinterpretation - // is still needed: `case 0ul ... ULONG_MAX:` read as an - // empty range and never matched. - let below = |a: i128, b: i128| { - if unsigned { - (a as u128) < (b as u128) - } else { - a < b + let Some(raw_lo) = self.eval_const_expr(expr) else { + self.report_unfoldable_case(expr); + return; + }; + // A GNU range `case lo ... hi:`. An absent high endpoint is + // the ordinary label, held as the degenerate range `(v, v)` so + // that everything downstream has one shape. + let raw_hi = match high { + None => Some(raw_lo), + Some(hi_expr) => match self.eval_const_expr(hi_expr) { + Some(h) => Some(h), + None => { + self.report_unfoldable_case(hi_expr); + None } + }, + }; + let Some(raw_hi) = raw_hi else { return }; + + // C17 6.8.4.2p5: each case constant is converted to the + // promoted type of the controlling expression. Evaluating the + // label at full width and never converting it left c17's two + // lowerings disagreeing about the same switch -- a runtime + // selector kept the unconverted label in the `switch` + // instruction, where the backend truncated it, while the + // constant-selector path compared at 128 bits and did not + // match at all. `case 4294967296LL:` in an `int` switch is + // `case 0:`, and has to be that for both. + let conv = case_values.conv(); + let lo = self.convert_case_label(expr, raw_lo, conv); + let hi = match high { + None => lo, + Some(hi_expr) => self.convert_case_label(hi_expr, raw_hi, conv), + }; + + // 6.8.4.2p3 forbids two equal case constants, and GCC + // extends that to overlapping ranges -- an overlap would + // otherwise make one arm silently unreachable, since the + // body walk resolves a label by finding the first match. + // Both tests run on the converted values, since that is what + // "equal" means once p5 has been applied: `case 0:` beside + // `case 4294967296LL:` in an `int` switch is one value twice. + // + // Order by the switch type's own signedness. The endpoints are + // carried as `i128`, and an unsigned 64-bit bound above + // `i64::MAX` is still positive there -- but an unsigned + // *128-bit* one is not, so the reinterpretation is still + // needed: `case 0ul ... ULONG_MAX:` read as an empty range and + // never matched. + if conv.lt(hi, lo) { + // GCC accepts an empty range, warns, and never matches + // it. Nothing is recorded, so nothing can overlap it. + crate::diag::warning(expr.pos, "empty range specified"); + return; + } + if let Some((lo2, hi2)) = case_values.overlap(lo, hi) { + let what = if lo == hi && lo2 == hi2 { + format!("duplicate case value '{}' in switch", lo) + } else { + format!( + "duplicate (or overlapping) case value: {}..{} overlaps {}..{}", + lo, hi, lo2, hi2 + ) }; - if below(hi, val) { - // GCC accepts an empty range, warns, and never matches - // it. Nothing is recorded, so nothing can overlap it. - crate::diag::warning(expr.pos, "empty range specified"); - return; - } - if let Some((lo2, hi2)) = case_values.overlap(val, hi) { - let what = if val == hi && lo2 == hi2 { - format!("duplicate case value '{}' in switch", val) - } else { - format!( - "duplicate (or overlapping) case value: {}..{} overlaps {}..{}", - val, hi, lo2, hi2 - ) - }; - error(expr.pos, &what); - } - case_values.insert(val, hi); - } else if self.expr_is_runtime(expr) { - // A non-constant label can never match. - error(expr.pos, "case label is not an integer constant expression"); - } else { - // Constant in principle, but `eval_const_expr` is a partial - // evaluator and could not fold it. Saying the program is - // invalid would be a false claim about the source — this is - // our limit, not its error. Either way the label cannot be - // emitted, so it still has to be reported rather than - // silently dropped. - error( - expr.pos, - "case label is a constant expression this compiler cannot evaluate", - ); + error(expr.pos, &what); } + case_values.insert(lo, hi); } Stmt::Default(_, body) => { - self.collect_cases_from_stmt(body, case_values, has_default, unsigned); + self.collect_cases_from_stmt(body, case_values, has_default); // C99 6.8.4.2p3: at most one default label per switch. if *has_default { error( @@ -1940,28 +2207,28 @@ impl<'a> super::linearize::Linearizer<'a> { } // Recursively check labeled statements Stmt::Label { stmt, .. } => { - self.collect_cases_from_stmt(stmt, case_values, has_default, unsigned); + self.collect_cases_from_stmt(stmt, case_values, has_default); } // Recurse into nested statements for Duff's device pattern // (case labels inside loops/blocks within a switch) Stmt::Block(items) => { for item in items { if let BlockItem::Statement(s) = item { - self.collect_cases_from_stmt(s, case_values, has_default, unsigned); + self.collect_cases_from_stmt(s, case_values, has_default); } } } Stmt::DoWhile { body, .. } | Stmt::While { body, .. } | Stmt::For { body, .. } => { - self.collect_cases_from_stmt(body, case_values, has_default, unsigned); + self.collect_cases_from_stmt(body, case_values, has_default); } Stmt::If { then_stmt, else_stmt, .. } => { - self.collect_cases_from_stmt(then_stmt, case_values, has_default, unsigned); + self.collect_cases_from_stmt(then_stmt, case_values, has_default); if let Some(e) = else_stmt { - self.collect_cases_from_stmt(e, case_values, has_default, unsigned); + self.collect_cases_from_stmt(e, case_values, has_default); } } // Stop at inner switch — its case labels belong to it @@ -1975,9 +2242,13 @@ impl<'a> super::linearize::Linearizer<'a> { /// Copy a string literal's code units into an array object, followed by /// its null terminator. /// - /// Shared by the two ways a string can initialize an array: written - /// directly (`char b[] = "hi"`) or enclosed in braces - /// (`char b[] = {"hi"}`, C17 6.7.9p14). + /// Shared by every way a string can initialize an array: written directly + /// (`char b[] = "hi"`), enclosed in braces (`char b[] = {"hi"}`, C17 + /// 6.7.9p14), as a struct member (`struct { char t[4]; } s = {"hi"}`), or + /// as an element of a nested array (`char n[2][4] = {"ab", "cd"}`). + /// + /// `tail` says whether the elements the literal does not reach are this + /// call's to zero; see [`StringTail`]. pub(crate) fn store_string_units( &mut self, base_sym: PseudoId, @@ -1985,6 +2256,7 @@ impl<'a> super::linearize::Linearizer<'a> { arr_typ: TypeId, kind: &ExprKind, units: &[i128], + tail: StringTail, ) { let default_elem = match kind { ExprKind::StringLit(_) => self.types.char_id, @@ -2017,7 +2289,10 @@ impl<'a> super::linearize::Linearizer<'a> { elem_size, )); } - if units.len() < capacity { + // The first element past everything the literal and its terminator + // wrote. When the literal fills the array exactly, that is the whole + // array; when the terminator was dropped, nothing is left either. + let written = if units.len() < capacity { let null_val = self.emit_const(0, elem_type); self.emit(Instruction::store( null_val, @@ -2026,6 +2301,25 @@ impl<'a> super::linearize::Linearizer<'a> { elem_type, elem_size, )); + units.len() + 1 + } else { + capacity + }; + + // C17 6.7.9p21: the members not initialized explicitly are + // initialized as a static object would be, i.e. to zero. One + // terminator is not the rest of the array -- `char b[8] = "hi"` wrote + // three bytes and left five holding whatever the frame held, which on + // first entry is zero because the backend zeroes the whole frame, and + // on re-execution is the last iteration's data. + // + // Routed through the shared block fill, so the bound that keeps + // `char b[1 << 20] = "x"` from becoming a million stores is the one + // every other block operation uses. + if matches!(tail, StringTail::Zero) { + let start = base_offset + (written as i64) * elem_bytes; + let bytes = (capacity - written) as i64 * elem_bytes; + self.emit_block_zero(base_sym, start, bytes); } } @@ -2398,20 +2692,27 @@ impl<'a> super::linearize::Linearizer<'a> { &mut self, switch_val: PseudoId, cmp_type: TypeId, + conv: CaseConv, case_values: &[(i128, i128)], case_bbs: &[BasicBlockId], default_target: BasicBlockId, ) { let size = self.types.size_bits(cmp_type); - let unsigned = self.types.is_unsigned(cmp_type); - // `>=` and `<=` for a range, in the controlling type's own signedness. - let (ge, le) = if unsigned { + // `>=` and `<=` for a range, in the controlling type's own signedness + // -- the same `conv` that converted the labels, so the comparison and + // the constants it compares are describing one type. + let (ge, le) = if conv.unsigned() { (Opcode::SetAe, Opcode::SetBe) } else { (Opcode::SetGe, Opcode::SetLe) }; for (&(lo, hi), &case_bb) in case_values.iter().zip(case_bbs.iter()) { + debug_assert_eq!( + (conv.convert(lo), conv.convert(hi)), + (lo, hi), + "a case label reaches lowering already converted to the controlling type" + ); let Some(from) = self.current_bb else { return }; let next = self.alloc_bb(); let cond = if lo == hi { @@ -2453,11 +2754,12 @@ impl<'a> super::linearize::Linearizer<'a> { // `collect_switch_cases`, which has to agree about this. match body { Stmt::Block(items) => { - // Same VLA reclamation as the ordinary block arm: a switch - // body is lowered by its own walk, and leaving the rule out - // here let `switch (c) { case 0: { int v[n]; break; } }` - // inside a loop grow the stack without bound. - let vla_scope = self.open_vla_scope(); + // The same scope as the ordinary block arm: a switch body is + // lowered by its own walk, and leaving the rule out here let + // `switch (c) { case 0: { int v[n]; break; } }` inside a loop + // grow the stack without bound, and let the body's + // declarations outlive the switch. + let scope = self.push_scope(); for item in items { match item { BlockItem::Declaration(decl) => self.linearize_local_decl(decl), @@ -2472,7 +2774,7 @@ impl<'a> super::linearize::Linearizer<'a> { } } } - self.close_vla_scope(vla_scope); + self.pop_scope(scope); } stmt => { self.linearize_switch_stmt(stmt, case_values, case_bbs, default_bb, &mut case_idx) @@ -2490,20 +2792,18 @@ impl<'a> super::linearize::Linearizer<'a> { ) { match stmt { Stmt::Case(expr, high, body) => { - // Find the matching case block. A label is identified by its - // whole range, so that `case 1 ... 3:` and a later `case 1:` - // could not resolve to the same block -- the overlap check - // rejects that pair anyway, but matching on the low endpoint - // alone would have made the two indistinguishable here. - if let Some(val) = self.eval_const_expr(expr) { - // Matched at full width, as the collector records them. - let lo = val; + // Find the matching case block. The endpoints are the label's + // raw constants; `CaseIndex::lookup` converts them to the + // promoted controlling type with the very conversion the + // collector used, which is what keeps this lookup from missing + // and dropping the case body into the wrong block. + if let Some(lo) = self.eval_const_expr(expr) { let hi = match high { None => Some(lo), Some(hi_expr) => self.eval_const_expr(hi_expr), }; let Some(hi) = hi else { return }; - if let Some(&idx) = case_values.get(&(lo, hi)) { + if let Some(idx) = case_values.lookup(lo, hi) { let case_bb = case_bbs[idx]; // Fall through from previous case if not terminated @@ -2605,73 +2905,23 @@ impl<'a> super::linearize::Linearizer<'a> { self.switch_bb(exit_bb); } + // Everything but the body is the ordinary `for` lowering; only + // the body has to go back through this walk, so the `case` + // labels inside it stay reachable. See `OpenFor`. Stmt::For { init, cond, post, body, } => { - self.push_scope(); - - if let Some(init) = init { - match init { - ForInit::Declaration(decl) => self.linearize_local_decl(decl), - ForInit::Expression(expr) => { - self.linearize_expr(expr); - } - } - } - - let cond_bb = self.alloc_bb(); - let body_bb = self.alloc_bb(); - let post_bb = self.alloc_bb(); - let exit_bb = self.alloc_bb(); - - if let Some(current) = self.current_bb { - if !self.is_terminated() { - self.emit(Instruction::br(cond_bb)); - self.link_bb(current, cond_bb); - } - } - - self.switch_bb(cond_bb); - if let Some(cond_expr) = cond { - self.branch_on_condition(cond_expr, body_bb, exit_bb); - } else { - self.emit(Instruction::br(body_bb)); - self.link_bb(cond_bb, body_bb); - } - - self.break_targets.push(exit_bb); - self.continue_targets.push(post_bb); - - self.switch_bb(body_bb); + let open = self.open_for(init.as_ref(), cond.as_ref()); self.linearize_switch_stmt(body, case_values, case_bbs, default_bb, case_idx); - if !self.is_terminated() { - if let Some(current) = self.current_bb { - self.emit(Instruction::br(post_bb)); - self.link_bb(current, post_bb); - } - } - - self.break_targets.pop(); - self.continue_targets.pop(); - - self.switch_bb(post_bb); - if let Some(post_expr) = post { - self.linearize_expr(post_expr); - } - self.emit(Instruction::br(cond_bb)); - self.link_bb(post_bb, cond_bb); - - self.switch_bb(exit_bb); - self.pop_scope(); + self.close_for(open, post.as_ref()); } Stmt::Block(items) => { - self.push_scope(); // See the sibling arm in `linearize_switch_body`. - let vla_scope = self.open_vla_scope(); + let scope = self.push_scope(); for item in items { match item { BlockItem::Declaration(decl) => self.linearize_local_decl(decl), @@ -2686,8 +2936,7 @@ impl<'a> super::linearize::Linearizer<'a> { } } } - self.close_vla_scope(vla_scope); - self.pop_scope(); + self.pop_scope(scope); } Stmt::If { @@ -2998,10 +3247,17 @@ impl<'a> super::linearize::Linearizer<'a> { // with every output unstored. Each label edge now gets a block of its // own that writes the outputs back and then jumps to the label. let writes_back = skip_post_handling.iter().any(|skip| !skip); + // An `asm goto` has two kinds of exit and each must release the VLA + // scopes it leaves. The fall-through is released by the enclosing + // scope's own end, but a label edge branches straight past it -- so + // the release goes in the edge block, which therefore has to exist + // even when there is nothing to write back. + let releases_vlas = !self.vla_marks.is_empty(); + let needs_edge_block = writes_back || releases_vlas; let label_edges: Vec<(BasicBlockId, BasicBlockId, String)> = ir_goto_labels .iter() .map(|(target, name)| { - let edge = if writes_back { + let edge = if needs_edge_block { self.alloc_bb() } else { *target @@ -3046,16 +3302,21 @@ impl<'a> super::linearize::Linearizer<'a> { // Without this, code would fall through to whatever block comes next in layout self.emit(Instruction::br(fall_through)); - if writes_back { + if needs_edge_block { for (edge, target, _) in &label_edges { self.switch_bb(*edge); - self.emit_asm_output_writeback( - outputs, - &ir_outputs, - &skip_post_handling, - ¶m_outputs, - &output_places, - ); + if writes_back { + self.emit_asm_output_writeback( + outputs, + &ir_outputs, + &skip_post_handling, + ¶m_outputs, + &output_places, + ); + } + // The jump leaves this scope; the fall-through does + // not. Same rule as a plain `goto` to the label. + self.release_vla_scopes_for_goto(*target); self.emit(Instruction::br(*target)); self.link_bb(*edge, *target); } @@ -3202,6 +3463,15 @@ impl<'a> super::linearize::Linearizer<'a> { /// One mark per declaration, not per block: a label sitting between two /// VLAs must release only the one declared after it, and a block-wide /// mark cannot express that. + /// + /// The nesting depths are read *before* the construct being lowered + /// pushes its own break or continue target, and that is deliberate. A + /// VLA declared in a `for` init clause, or in the controlling expression + /// of a `switch`, is allocated once, outside the loop or switch, and its + /// scope encloses the exit the jump lands on -- so a `break` or + /// `continue` inside must *not* release it. `continue` especially: the + /// storage is still live on the next iteration. The construct's own + /// scope, which ends after its exit block, is what releases it. fn push_vla_mark(&mut self) { if self.current_bb.is_none() { return; @@ -3219,14 +3489,6 @@ impl<'a> super::linearize::Linearizer<'a> { }); } - /// The marks in force on entry to a block, to restore and drop on exit. - /// - /// Returns the depth of [`Linearizer::vla_marks`] so - /// [`Self::close_vla_scope`] knows which of them this block added. - fn open_vla_scope(&self) -> usize { - self.vla_marks.len() - } - /// Give every forward `goto` the VLA restore its label turned out to need. /// /// Deferred because a label's depth is known only once it has been placed. @@ -3247,9 +3509,7 @@ impl<'a> super::linearize::Linearizer<'a> { pending.sort_by_key(|p| (p.bb.0, std::cmp::Reverse(p.at))); let void_ptr = self.types.void_ptr_id; for p in pending { - // An undefined label is diagnosed elsewhere; there is no branch - // here to put a restore in front of. - let Some(&depth) = self.label_vla_depth.get(&p.label) else { + let Some(depth) = self.goto_target_vla_depth(&p.target) else { continue; }; let Some(&mark) = p.marks.get(depth) else { @@ -3268,12 +3528,34 @@ impl<'a> super::linearize::Linearizer<'a> { } } - /// Release everything the block allocated and forget its marks. - fn close_vla_scope(&mut self, entry: usize) { + /// How many VLA scopes a jump is *inside* at the label it reaches, or + /// `None` if that cannot be said -- an undefined label, diagnosed + /// elsewhere, or a computed `goto` in a function that takes no label's + /// address. + fn goto_target_vla_depth(&self, target: &GotoTarget) -> Option { + match target { + GotoTarget::Label(bb) => self.label_vla_depth.get(bb).copied(), + // See [`GotoTarget::AnyAddressTaken`]: the deepest candidate is + // the only depth that releases nothing another candidate still + // needs. + GotoTarget::AnyAddressTaken => self + .addr_taken_labels + .iter() + .filter_map(|bb| self.label_vla_depth.get(bb).copied()) + .max(), + } + } + + /// Release everything the scope allocated and forget its marks. + /// + /// Called only from [`Linearizer::pop_scope`], so that leaving a + /// declaration scope and leaving a VLA scope are the same act. + pub(crate) fn close_vla_scope(&mut self, scope: &Scope) { + let entry = scope.vla_entry; if self.vla_marks.len() <= entry { return; } - // The first mark the block took is the stack as it stood on entry, + // The first mark the scope took is the stack as it stood on entry, // so one restore undoes all of them. let mark = self.vla_marks[entry].mark; if !self.is_terminated() && self.current_bb.is_some() { @@ -3282,6 +3564,64 @@ impl<'a> super::linearize::Linearizer<'a> { self.vla_marks.truncate(entry); } + /// Release every VLA scope a jump to the block `target` leaves. + /// + /// A backward jump -- the label is already linearized, so it has a + /// recorded depth -- leaves the scope of every VLA declared after it, and + /// that storage has to go back. Otherwise `lab: int x[n]; ... goto lab;` + /// grows the stack every time round until the program dies. + /// + /// A *forward* jump cannot be decided here: its label has no depth + /// recorded yet, and whether it stays inside the scope of the VLAs in + /// force or leaves it is exactly what decides between no restore and one. + /// It is recorded and resolved in + /// [`Self::resolve_forward_goto_vla_restores`]. + /// + /// Leaving it to "the scope's own exit does the restoring" was wrong: the + /// branch *terminates* the block, so `close_vla_scope` emits nothing and + /// then drops the marks, and the enclosing scope has no mark of its own + /// to undo them with. A `goto` out of a loop body's inner block grew the + /// stack every time round. + /// + /// Every jump that names a label goes through here: a `goto`, and each + /// label edge of an `asm goto`. + fn release_vla_scopes_for_goto(&mut self, target: BasicBlockId) { + if let Some(&depth) = self.label_vla_depth.get(&target) { + if let Some(m) = self.vla_marks.get(depth) { + let mark = m.mark; + self.emit_stack_restore(mark); + } + } else { + self.defer_vla_restore(GotoTarget::Label(target)); + } + } + + /// Record a restore whose depth is not yet known, to be placed by + /// [`Self::resolve_forward_goto_vla_restores`] at the end of the + /// function. The restore goes where the current block ends now, which is + /// ahead of the branch the caller is about to emit. + fn defer_vla_restore(&mut self, target: GotoTarget) { + if self.vla_marks.is_empty() { + return; + } + let Some(current) = self.current_bb else { + return; + }; + let at = self + .current_func + .as_ref() + .and_then(|f| f.get_block(current)) + .map_or(0, |b| b.insns.len()); + let marks = self.vla_marks.iter().map(|m| m.mark).collect(); + self.pending_goto_vla + .push(crate::ir::linearize::PendingGotoVla { + target, + bb: current, + at, + marks, + }); + } + /// Put the stack pointer back to what `mark` captured. fn emit_stack_restore(&mut self, mark: PseudoId) { self.emit( @@ -3362,7 +3702,7 @@ impl<'a> super::linearize::Linearizer<'a> { // declaration that creates it lies between the label and the // jump. if self.func_has_vla { - self.label_vla_depth.insert(name_str, self.vla_marks.len()); + self.label_vla_depth.insert(label_bb, self.vla_marks.len()); } } @@ -3858,8 +4198,121 @@ impl AddrWalk<'_> { } } -/// Each case range's position among a switch's labels. -pub(crate) type CaseIndex = std::collections::HashMap<(i128, i128), usize>; +/// The promoted type of a switch's controlling expression: the width and the +/// signedness in which C17 6.8.4.2 says every case label lives. +/// +/// p5 converts each case constant to that type and p3 forbids two of them +/// being equal *after* the conversion, so a label's converted value is the +/// only one the rest of the switch path may see. Two places have to agree +/// about it -- the collector that records a label's range, and the body walk +/// that looks that same range back up to find the block it was given. If they +/// disagreed the lookup would simply miss, leaving the case body emitted into +/// the wrong block with nothing diagnosed. +/// +/// One value carries the whole conversion so they cannot disagree: +/// [`CaseSet`] owns the `CaseConv`, [`CaseSet::insert`] and +/// [`CaseSet::overlap`] convert what they are handed, [`CaseIndex::of`] copies +/// the conversion out of the set it indexes, and [`CaseIndex::lookup`] -- the +/// only way into the map -- converts too. Conversion is idempotent, so a +/// caller that has already converted for its own reasons stays in step. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub(crate) struct CaseConv { + /// Width of the promoted controlling type, in bits. + bits: u32, + /// Whether that type is unsigned. + unsigned: bool, +} + +impl CaseConv { + pub(crate) fn new(bits: u32, unsigned: bool) -> Self { + Self { bits, unsigned } + } + + /// The conversion a `switch` whose promoted controlling type is `typ` + /// applies to its labels. + pub(crate) fn of(types: &TypeTable, typ: TypeId) -> Self { + Self::new(types.size_bits(typ), types.is_unsigned(typ)) + } + + pub(crate) fn unsigned(self) -> bool { + self.unsigned + } + + /// `v` converted to this type, per C17 6.3.1.3: its low `bits` bits, read + /// back with the type's own signedness. + /// + /// The result is carried the way the whole switch path carries a value -- + /// an `i128` holding the type's bit pattern -- so a 128-bit unsigned label + /// above `i128::MAX` stays negative here and is ordered by [`Self::lt`] + /// rather than by Rust's signed `<`. + pub(crate) fn convert(self, v: i128) -> i128 { + if self.bits == 0 || self.bits >= 128 { + return v; + } + let shift = 128 - self.bits; + let truncated = ((v as u128) << shift) >> shift; + if self.unsigned { + truncated as i128 + } else { + ((truncated << shift) as i128) >> shift + } + } + + /// `a < b` in this type's signedness. + pub(crate) fn lt(self, a: i128, b: i128) -> bool { + if self.unsigned { + (a as u128) < (b as u128) + } else { + a < b + } + } + + /// Whether the converted range `lo..=hi` holds the converted value `v`. + /// + /// The constant-selector lowering picks its one edge with this, and has to + /// ask in the switch's signedness: a plain `i128` test read `case -1:` in + /// a `switch` on `unsigned` as a huge lower bound and never selected it. + pub(crate) fn contains(self, lo: i128, hi: i128, v: i128) -> bool { + !self.lt(v, lo) && !self.lt(hi, v) + } +} + +/// Each case range's position among a switch's labels, keyed by the range as +/// the controlling type sees it. +pub(crate) struct CaseIndex { + conv: CaseConv, + by_range: std::collections::HashMap<(i128, i128), usize>, +} + +impl CaseIndex { + /// Index the labels `set` collected, carrying `set`'s own conversion so + /// that a lookup converts exactly as the insert did. + /// + /// The first label of a repeated range wins, as a scan in source order + /// would find it; a duplicate has already been reported. + pub(crate) fn of(set: &CaseSet) -> Self { + let mut by_range = std::collections::HashMap::new(); + for (idx, range) in set.ranges().iter().enumerate() { + by_range.entry(*range).or_insert(idx); + } + Self { + conv: set.conv(), + by_range, + } + } + + /// The position of the label written `lo ... hi`, whose endpoints are the + /// raw constants as the label spells them. + /// + /// A label is identified by its whole range, so that `case 1 ... 3:` and a + /// later `case 1:` cannot resolve to the same block -- the overlap check + /// rejects that pair anyway, but matching on the low endpoint alone would + /// make the two indistinguishable here. + pub(crate) fn lookup(&self, lo: i128, hi: i128) -> Option { + let key = (self.conv.convert(lo), self.conv.convert(hi)); + self.by_range.get(&key).copied() + } +} /// A switch's case ranges, in source order, with an index that finds an /// overlap in logarithmic time. @@ -3868,43 +4321,56 @@ pub(crate) type CaseIndex = std::collections::HashMap<(i128, i128), usize>; /// quadratic in its case count: 70,000 labels took five seconds to compile /// and gcc's `limits-caselabels` eleven. pub(crate) struct CaseSet { - /// The ranges `(lo, hi)`, in the order the labels were written. + /// The ranges `(lo, hi)`, converted to the controlling type, in the order + /// the labels were written. ranges: Vec<(i128, i128)>, /// Each range by its low end, as an order-preserving key, to its high end. by_lo: std::collections::BTreeMap, - unsigned: bool, + /// What every endpoint entering the set is converted by. + conv: CaseConv, } impl CaseSet { - fn new(unsigned: bool) -> Self { + fn new(conv: CaseConv) -> Self { Self { ranges: Vec::new(), by_lo: std::collections::BTreeMap::new(), - unsigned, + conv, } } + pub(crate) fn conv(&self) -> CaseConv { + self.conv + } + + /// The ranges, converted, in source order. Parallel to the case blocks. + pub(crate) fn ranges(&self) -> &[(i128, i128)] { + &self.ranges + } + /// `v` as a signed key ordered the way the switch's type orders it: an /// unsigned value has its top bit flipped, which maps unsigned order onto /// signed order. fn key(&self, v: i128) -> i128 { - if self.unsigned { + if self.conv.unsigned { v ^ i128::MIN } else { v } } - /// An earlier range sharing a value with `lo..=hi`, if any. + /// An earlier range sharing a value with `lo..=hi`, if any, as converted. /// /// The ranges recorded are disjoint -- an overlap is an error -- so the /// only candidate is the one starting last at or before `hi`. fn overlap(&self, lo: i128, hi: i128) -> Option<(i128, i128)> { + let (lo, hi) = (self.conv.convert(lo), self.conv.convert(hi)); let (_, &(hi_key, lo2, hi2)) = self.by_lo.range(..=self.key(hi)).next_back()?; (hi_key >= self.key(lo)).then_some((lo2, hi2)) } fn insert(&mut self, lo: i128, hi: i128) { + let (lo, hi) = (self.conv.convert(lo), self.conv.convert(hi)); self.ranges.push((lo, hi)); let (lo_key, hi_key) = (self.key(lo), self.key(hi)); self.by_lo.insert(lo_key, (hi_key, lo, hi)); @@ -3913,11 +4379,127 @@ impl CaseSet { #[cfg(test)] mod case_set_tests { - use super::CaseSet; + use super::{CaseConv, CaseIndex, CaseSet}; + + /// A 128-bit conversion is the identity, which is what the ranges below + /// want: they are about ordering, not about width. + fn wide(unsigned: bool) -> CaseConv { + CaseConv::new(128, unsigned) + } + + /// C17 6.8.4.2p5 converts a case constant to the promoted controlling + /// type: the low bits, read back with that type's signedness. + #[test] + fn convert_takes_the_low_bits_with_the_types_signedness() { + let int = CaseConv::new(32, false); + let uint = CaseConv::new(32, true); + + // In range: unchanged either way. + assert_eq!(int.convert(7), 7); + assert_eq!(uint.convert(7), 7); + + // 2^32 is zero in 32 bits -- the label that silently became `case 0:`. + assert_eq!(int.convert(4294967296), 0); + assert_eq!(uint.convert(4294967296), 0); + + // -1 keeps its value as `int` and is the largest `unsigned int`. + assert_eq!(int.convert(-1), -1); + assert_eq!(uint.convert(-1), 4294967295); + + // The boundary of the signed range wraps the way C says. + assert_eq!(int.convert(2147483648), -2147483648); + assert_eq!(uint.convert(2147483648), 2147483648); + + // Narrower and wider types, and the 128-bit identity. + assert_eq!(CaseConv::new(8, false).convert(255), -1); + assert_eq!(CaseConv::new(8, true).convert(-1), 255); + assert_eq!(CaseConv::new(64, true).convert(-1), u64::MAX as i128); + assert_eq!(CaseConv::new(128, true).convert(-1), -1); + assert_eq!(CaseConv::new(128, false).convert(i128::MIN), i128::MIN); + } + + /// Converting is idempotent, which is what lets the collector convert for + /// its own diagnostics and still hand the set and the index raw or + /// converted endpoints interchangeably. + #[test] + fn convert_is_idempotent() { + for conv in [ + CaseConv::new(8, false), + CaseConv::new(16, true), + CaseConv::new(32, false), + CaseConv::new(64, true), + CaseConv::new(128, true), + ] { + for v in [0, 1, -1, 255, 4294967296, i128::MIN, i128::MAX] { + let once = conv.convert(v); + assert_eq!(conv.convert(once), once, "{conv:?} {v}"); + } + } + } + + /// The constant-selector lowering asks in the switch's own signedness. + #[test] + fn contains_tests_the_range_in_the_switch_signedness() { + let uint = CaseConv::new(32, true); + let big = uint.convert(-1); // 4294967295 + assert!(uint.contains(big, big, big)); + assert!(!uint.contains(big, big, 0)); + assert!(uint.contains(0, big, 5)); + + let int = CaseConv::new(32, false); + assert!(int.contains(-1, -1, -1)); + assert!(int.contains(-5, 5, 0)); + assert!(!int.contains(-5, 5, 6)); + // A signed test would read the unsigned bound as below zero. + assert!(!int.contains(0, 10, big)); + } + + /// The two-site invariant: what the collector inserts is exactly what the + /// body walk finds, even though the walk looks the label up by the + /// constant as written rather than as converted. + #[test] + fn the_index_finds_a_label_by_its_unconverted_constant() { + let conv = CaseConv::new(32, false); + let mut set = CaseSet::new(conv); + set.insert(0, 0); + set.insert(-1, -1); + set.insert(70000, 70005); + // Stored converted, and 2^32+3 is 3 in an `int` switch. + set.insert(4294967299, 4294967299); + assert_eq!(set.ranges(), [(0, 0), (-1, -1), (70000, 70005), (3, 3)]); + + let index = CaseIndex::of(&set); + assert_eq!(index.lookup(0, 0), Some(0)); + assert_eq!(index.lookup(-1, -1), Some(1)); + assert_eq!(index.lookup(70000, 70005), Some(2)); + // Looked up as written, found as converted. + assert_eq!(index.lookup(4294967299, 4294967299), Some(3)); + assert_eq!(index.lookup(3, 3), Some(3)); + assert_eq!(index.lookup(9, 9), None); + + // And a label the controlling type sees as negative. + let uconv = CaseConv::new(32, true); + let mut uset = CaseSet::new(uconv); + uset.insert(-1, -1); + assert_eq!(uset.ranges(), [(4294967295, 4294967295)]); + let uindex = CaseIndex::of(&uset); + assert_eq!(uindex.lookup(-1, -1), Some(0)); + assert_eq!(uindex.lookup(4294967295, 4294967295), Some(0)); + } + + /// Two labels that differ before the conversion collide after it, which is + /// the duplicate C17 6.8.4.2p3 forbids. + #[test] + fn overlap_sees_the_converted_values() { + let mut set = CaseSet::new(CaseConv::new(32, false)); + set.insert(0, 0); + assert_eq!(set.overlap(4294967296, 4294967296), Some((0, 0))); + assert_eq!(set.overlap(1, 1), None); + } #[test] fn overlap_finds_the_range_sharing_a_value() { - let mut set = CaseSet::new(false); + let mut set = CaseSet::new(wide(false)); set.insert(-10, -5); set.insert(0, 0); set.insert(10, 20); @@ -3937,7 +4519,7 @@ mod case_set_tests { #[test] fn overlap_orders_by_the_switch_type_signedness() { let big = u128::MAX as i128; // -1 as i128, the largest unsigned value - let mut set = CaseSet::new(true); + let mut set = CaseSet::new(wide(true)); set.insert(1, 5); set.insert(big - 10, big); assert_eq!(set.overlap(big - 3, big - 3), Some((big - 10, big))); diff --git a/cc/ir/loadfwd.rs b/cc/ir/loadfwd.rs index ebe6b37b1..3edfbdbe1 100644 --- a/cc/ir/loadfwd.rs +++ b/cc/ir/loadfwd.rs @@ -89,6 +89,14 @@ pub(crate) fn run(func: &mut Function, types: &TypeTable, mi: &ModuleInfo) -> bo if insn.op != Opcode::Load { continue; } + // Each read of a volatile object is its own observable event, so + // the value another access left behind is no answer for this one. + // `forwardable` declines a *named* volatile object; the marker is + // what declines `*p` for a `volatile int *p`, where the qualifier + // is on the access and there is no variable to ask. + if insn.is_volatile_access() { + continue; + } if let Some(a) = oracle.value_at((b, i), &am.location_of(func, insn)) { sites.push((b, i, a)); } diff --git a/cc/ir/mod.rs b/cc/ir/mod.rs index fc622480a..b94ea5fc3 100644 --- a/cc/ir/mod.rs +++ b/cc/ir/mod.rs @@ -78,6 +78,44 @@ impl CallAbiInfo { } } +/// Whether an aggregate of `size_bits` returned in class `ret` is handed back +/// by *address*: the `Ret` carries a pointer to the value's storage rather +/// than the value, and whoever consumes the return has to read the bytes out. +/// +/// Three classifications answer yes, and they are exactly the three that no +/// pair of general registers can carry: +/// +/// * `X87` -- an aggregate that is nothing but a `long double` comes back in +/// st(0), which is loaded from memory because nothing else holds 80 bits. +/// * `Hfa` -- AAPCS64 returns a homogeneous floating-point aggregate in one +/// V register per element, at any size: four `double`s is thirty-two bytes +/// and still comes back in `d0`-`d3`. +/// * one `Sse` -- a single SSE register holding all sixteen bytes, which on +/// x86-64 is an aggregate whose sole content is a `__float128`: SSE+SSEUP +/// is *one* register, and splitting it into RAX/RDX hands a gcc-compiled +/// caller half a value in the wrong place. +/// +/// The size bound is part of the rule, not a caller's business: an aggregate +/// that fits in one register comes back *as* a value, so `struct { float a, +/// b; }` is `Direct { classes: [Sse] }` and yet carries its value. Only past +/// 64 bits is an address handed back. +/// +/// One function because the answer is asked in three places -- the return +/// emitter, the flag that tells the inliner what it is splicing, and the +/// inliner's own `Ret` lowering -- and two spellings of it had already +/// drifted: the flag omitted the one-SSE case, so a `struct { __float128 a; }` +/// return reported a value-carrying `Ret`, the inliner spliced the body in and +/// phi-ed the callee's local *address* as though it were the aggregate. +pub fn aggregate_ret_is_address(ret: &ArgClass, size_bits: u32) -> bool { + use crate::abi::RegClass; + size_bits > 64 + && match ret { + ArgClass::X87 { .. } | ArgClass::Hfa { .. } => true, + ArgClass::Direct { classes, .. } => classes.as_slice() == [RegClass::Sse], + _ => false, + } +} + // Instruction Reference - for def-use chains /// Reference to an instruction by (basic block id, instruction index) @@ -973,6 +1011,20 @@ pub struct Instruction { pub abi_info: Option>, /// For atomic operations: memory ordering constraint pub memory_order: MemoryOrder, + /// For `Load` and `Store`: the object being accessed is `volatile`, so the + /// access itself is observable behaviour (C17 5.1.2.3p6) and no pass may + /// delete, merge, move or fold it. + /// + /// The qualifier lives on the *access*, not on the variable, because for + /// `volatile int *p` there is no variable to ask: `p` is an ordinary + /// pointer and `*p` is the volatile object. `LocalVar::is_volatile` and + /// `memloc::GlobalFacts::is_volatile` answer only for a named object, so + /// DCE saw nothing to stop it and deleted every discarded `volatile` read + /// from `-O1` up. Ask through [`Instruction::is_volatile_access`]. + /// + /// Set for every access the linearizer emits, from the type it is + /// accessing, in `Linearizer::mark_volatile_access`. + pub is_volatile: bool, } impl Default for Instruction { @@ -1003,6 +1055,7 @@ impl Default for Instruction { asm_data: None, abi_info: None, memory_order: MemoryOrder::default(), + is_volatile: false, } } } @@ -1054,6 +1107,28 @@ impl Instruction { self } + /// Mark this `Load` or `Store` as an access to a `volatile` object. + pub fn with_volatile(mut self, is_volatile: bool) -> Self { + debug_assert!( + !is_volatile || matches!(self.op, Opcode::Load | Opcode::Store), + "only a Load or a Store carries the volatile marker" + ); + self.is_volatile = is_volatile; + self + } + + /// Is this an access to a `volatile` object? + /// + /// Reading or writing one is observable behaviour (C17 5.1.2.3p6), so an + /// access that answers `true` survives every optimization level: no pass + /// may delete it, fold it to a constant, merge it with another access, or + /// promote the object it reaches out of memory. This is the question to + /// ask; `is_volatile` is only where the answer is stored, and is true of + /// nothing but a `Load` or a `Store`. + pub fn is_volatile_access(&self) -> bool { + self.is_volatile && matches!(self.op, Opcode::Load | Opcode::Store) + } + /// Set the type (caller should also call with_size if needed) pub fn with_type(mut self, typ: TypeId) -> Self { self.typ = Some(typ); @@ -1542,12 +1617,31 @@ impl Instruction { .unwrap_or(false) } + /// True when this `Ret` hands back an aggregate by *address*: its source + /// is a pointer to the value's storage, not the value. + /// + /// Asked of the `Ret`'s own ABI classification, which is the only place + /// the answer is recorded -- `Instruction::size` is the aggregate's width, + /// so [`aggregate_ret_is_address`] can apply its own size bound without a + /// `TypeTable`. Only [`crate::ir::Linearizer::emit_two_reg_return`] ever + /// puts `abi_info` on a `Ret`, and only for a struct or union, so no + /// scalar reaches this. + pub fn returns_aggregate_address(&self) -> bool { + self.abi_info + .as_ref() + .is_some_and(|ai| aggregate_ret_is_address(&ai.ret, self.size)) + } + /// Convert this instruction to a no-op, clearing all operands. pub fn kill(&mut self) { self.op = Opcode::Nop; self.src.clear(); self.target = None; self.phi_list.clear(); + // A `Nop` reaches no memory, so it is no longer a volatile access -- + // and leaving the marker set on one would make a stale claim to any + // pass that asks the field rather than `is_volatile_access`. + self.is_volatile = false; } } @@ -1729,6 +1823,9 @@ impl fmt::Display for InstructionDisplay<'_> { if this.offset != 0 { write!(f, " + {}", this.offset)?; } + if this.is_volatile { + write!(f, " volatile")?; + } } _ => { for (i, src) in this.src.iter().enumerate() { @@ -3540,6 +3637,61 @@ mod tests { assert!(insn.returns_two_regs()); } + /// The three classes whose `Ret` hands back an aggregate's *address*, and + /// the size bound that is part of the rule. + /// + /// `Direct { classes: [Sse] }` is the discriminating row: at sixteen bytes + /// it is one SSE register holding a whole `__float128`, so the `Ret` names + /// the storage; at eight it is `struct { float a, b; }`, which comes back + /// *as* a value and never reaches `emit_two_reg_return` at all. Answering + /// the first one "no" is what made the inliner phi an address as though it + /// were the aggregate. + #[test] + fn test_aggregate_ret_is_address() { + let sse = |n: usize, bits: u32| ArgClass::Direct { + classes: vec![RegClass::Sse; n], + size_bits: bits, + }; + assert!(aggregate_ret_is_address(&sse(1, 128), 128), "one SSE, 16B"); + assert!(!aggregate_ret_is_address(&sse(1, 64), 64), "one SSE, 8B"); + assert!( + !aggregate_ret_is_address(&sse(2, 128), 128), + "two SSE registers carry the halves, not an address" + ); + assert!( + !aggregate_ret_is_address( + &ArgClass::Direct { + classes: vec![RegClass::Integer, RegClass::Integer], + size_bits: 128, + }, + 128 + ), + "__int128 comes back in RAX/RDX" + ); + assert!(aggregate_ret_is_address( + &ArgClass::X87 { size_bits: 80 }, + 128 + )); + assert!(aggregate_ret_is_address( + &ArgClass::Hfa { + base: crate::abi::HfaBase::Float64, + count: 4, + }, + 256 + )); + assert!( + !aggregate_ret_is_address( + &ArgClass::Indirect { + align: 8, + size_bytes: 64, + }, + 512 + ), + "the hidden pointer is not this" + ); + assert!(!aggregate_ret_is_address(&ArgClass::Ignore, 0)); + } + // Function::create_reg_pseudo #[test] diff --git a/cc/ir/range.rs b/cc/ir/range.rs index 28c806973..5edf42617 100644 --- a/cc/ir/range.rs +++ b/cc/ir/range.rs @@ -558,9 +558,16 @@ impl Range { /// Unsigned division. A divisor range containing zero answers `Full`: /// c17 does not assume undefined behaviour away, and - /// `constfold::eval_divmod` already refuses a zero divisor rather than - /// inventing a result. Disagreeing here would make `-O2` and `-O0` - /// differ on a program that really does divide by zero. + /// `constfold::divmod_may_trap` -- the one statement of which operand + /// pairs trap -- refuses a zero divisor rather than inventing a result. + /// Disagreeing here would make `-O2` and `-O0` differ on a program that + /// really does divide by zero. + /// + /// `contains(0)` is that predicate asked of a set rather than a value: + /// "may any divisor in this range trap?". For the unsigned forms that is + /// the whole of it, because the other trapping pair, `INT_MIN / -1`, is a + /// signed overflow with no unsigned counterpart -- and `vrp` gives the + /// signed opcodes `Range::full` rather than asking here at all. pub(crate) fn udiv(&self, other: &Range) -> Range { if let Some(r) = self.binary_guard(other) { return r; @@ -576,6 +583,9 @@ impl Range { Range::inclusive(self.width, amin / bmax, amax / bmin) } + /// Unsigned remainder, refusing a divisor range containing zero for the + /// reason [`Range::udiv`] gives: the remainder form is the same trapping + /// instruction. pub(crate) fn umod(&self, other: &Range) -> Range { if let Some(r) = self.binary_guard(other) { return r; diff --git a/cc/ir/ssa.rs b/cc/ir/ssa.rs index e4dd79d48..d501e1b02 100644 --- a/cc/ir/ssa.rs +++ b/cc/ir/ssa.rs @@ -195,6 +195,16 @@ fn analyze_variables(func: &Function, types: &TypeTable) -> HashMap= ; }` with `T` the chosen type made +/// `_Atomic` and the right operand a plain `int` literal, and linearize it. +/// +/// `target` is a selector rather than a `TypeId` because the table the id +/// belongs to is built by `TestContext::new`. +fn atomic_typed_module( + op: AssignOp, + target: fn(&TypeTable) -> TypeId, + value: i64, +) -> (TestContext, Module) { + let mut ctx = TestContext::new(); + let test_id = ctx.str("test"); + let int_id = ctx.types.int_id; + + let base = target(&ctx.types); + let atomic_typ = { + let mut t = ctx.types.get(base).clone(); + t.modifiers |= TypeModifiers::ATOMIC; + ctx.types.intern(t) + }; + let x_sym = ctx.var("x", atomic_typ); + + let assign = Expr::typed_unpositioned( + ExprKind::Assign { + op, + target: Box::new(Expr::var_typed(x_sym, atomic_typ)), + value: Box::new(Expr::typed_unpositioned(ExprKind::IntLit(value), int_id)), + }, + atomic_typ, + ); + let func = FunctionDef { + attrs: Default::default(), + return_type: ctx.types.void_id, + name: test_id, + params: vec![Parameter { + symbol: Some(x_sym), + typ: atomic_typ, + vm_dims: vec![], + discarded_dims: vec![], + }], + body: Stmt::Block(vec![BlockItem::Statement(Box::new(Stmt::Expr(assign)))]), + pos: test_pos(), + is_static: false, + is_inline: false, + calling_conv: crate::abi::CallingConv::default(), + param_style: ParamStyle::Prototype, + }; + let module = ctx.linearize(&TranslationUnit { + items: vec![ExternalDecl::FunctionDef(func)], + }); + (ctx, module) +} + +/// The first instruction with this opcode, for asserting on its type and width. +fn first_op(module: &Module, op: Opcode) -> &Instruction { + module.functions[0] + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .find(|i| i.op == op) + .unwrap_or_else(|| panic!("no {:?} in the module", op)) +} + +/// The usual arithmetic conversions decide the type, and the type decides the +/// opcode -- so a narrow unsigned target divided by an `int` is a *signed* +/// 32-bit divide. +#[test] +fn test_compound_assign_divides_at_the_operands_common_type() { + let types = TypeTable::new(&Target::host()); + + let ca = CompoundAssign::new(AssignOp::DivAssign, types.uchar_id, types.int_id); + let arith = compound_assign_arith_type(&types, &ca); + assert_eq!(arith, types.int_id, "unsigned char / int is done at int"); + assert_eq!( + compound_assign_opcode(&types, AssignOp::DivAssign, arith), + Opcode::DivS + ); + + // Asking the *target's* type instead is the defect this replaced: it makes + // the same expression an unsigned divide, and `50 /= -5` stores 0. + assert_eq!( + compound_assign_opcode(&types, AssignOp::DivAssign, types.uchar_id), + Opcode::DivU + ); +} + +/// The congruent operators are decided the same way, even though their result +/// is the same either width. +#[test] +fn test_compound_assign_add_also_computes_at_the_common_type() { + let types = TypeTable::new(&Target::host()); + let ca = CompoundAssign::new(AssignOp::AddAssign, types.uchar_id, types.int_id); + assert_eq!(compound_assign_arith_type(&types, &ca), types.int_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::AddAssign, types.int_id), + Opcode::Add + ); + // A floating target picks the floating form of the same operator. + let fca = CompoundAssign::new(AssignOp::AddAssign, types.float_id, types.int_id); + let farith = compound_assign_arith_type(&types, &fca); + assert_eq!(farith, types.float_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::AddAssign, farith), + Opcode::FAdd + ); +} + +/// A shift promotes its **left** operand and nothing else (C17 6.5.7p3), so +/// the right operand's type has no say in the width it is done at. +#[test] +fn test_compound_assign_shift_takes_the_promoted_left_operand() { + let types = TypeTable::new(&Target::host()); + + let ca = CompoundAssign::new(AssignOp::ShrAssign, types.schar_id, types.longlong_id); + let arith = compound_assign_arith_type(&types, &ca); + assert_eq!( + arith, types.int_id, + "the promoted left operand decides, not the common type" + ); + assert_ne!( + arith, + types.common_type(types.schar_id, types.longlong_id), + "the shift must not follow the usual arithmetic conversions" + ); + + // And the promotion is what makes the shift arithmetic: `unsigned char` + // promotes to `int`, so `u >>= 1` on 200 is 100 and not a logical shift of + // the byte. + let uca = CompoundAssign::new(AssignOp::ShrAssign, types.uchar_id, types.int_id); + let uarith = compound_assign_arith_type(&types, &uca); + assert_eq!(uarith, types.int_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::ShrAssign, uarith), + Opcode::Asr + ); + assert_eq!( + compound_assign_opcode(&types, AssignOp::ShrAssign, types.uchar_id), + Opcode::Lsr, + "computing at the target's own width would shift the wrong way" + ); +} + +/// `_Bool` promotes to `int` like any narrow integer; what is special about it +/// is the conversion *back*, which is a test against zero. +#[test] +fn test_compound_assign_bool_computes_at_int() { + let types = TypeTable::new(&Target::host()); + let ca = CompoundAssign::new(AssignOp::SubAssign, types.bool_id, types.int_id); + assert_eq!(compound_assign_arith_type(&types, &ca), types.int_id); +} + +/// Pointer arithmetic is the other exception: the addend arrives already +/// scaled to a byte count and the addition happens at pointer width. +#[test] +fn test_compound_assign_pointer_arithmetic_is_done_at_pointer_width() { + let types = TypeTable::new(&Target::host()); + let ca = CompoundAssign { + is_ptr_arith: true, + ..CompoundAssign::new(AssignOp::AddAssign, types.char_ptr_id, types.long_id) + }; + assert_eq!(compound_assign_arith_type(&types, &ca), types.long_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::AddAssign, types.long_id), + Opcode::Add + ); +} + +/// `_Atomic unsigned char c; c /= -5;` divides at `int`, in the CAS loop -- +/// the same arithmetic the ordinary lowering does. +#[test] +fn test_atomic_compound_divide_computes_at_the_common_type() { + let (_ctx, module) = atomic_typed_module(AssignOp::DivAssign, |t| t.uchar_id, -5); + + let div = first_op(&module, Opcode::DivS); + assert_eq!( + div.size, 32, + "the divide happens at the common type's width, not the object's" + ); + assert_eq!( + count_op(&module, Opcode::DivU), + 0, + "narrowing the right operand first would make this an unsigned divide" + ); + assert_eq!( + count_op(&module, Opcode::AtomicCas), + 1, + "divide has no native atomic form" + ); +} + +/// The same for a shift: promoted left operand, 32-bit arithmetic shift. +#[test] +fn test_atomic_compound_shift_promotes_its_left_operand() { + let (_ctx, module) = atomic_typed_module(AssignOp::ShrAssign, |t| t.uchar_id, 1); + + let shift = first_op(&module, Opcode::Asr); + assert_eq!(shift.size, 32, "the left operand is promoted to int first"); + assert_eq!( + count_op(&module, Opcode::Lsr), + 0, + "an 8-bit logical shift would be the target's width, not the promoted one" + ); + assert_eq!(count_op(&module, Opcode::AtomicCas), 1); +} + +/// A narrow congruent operator keeps its native fetch-and-op. +/// +/// The standard computes `c += 100` at `int` and converts back, but add is +/// congruent modulo 2^8, so the hardware's 8-bit add agrees with it -- and a +/// single instruction beats a retry loop. +#[test] +fn test_atomic_narrow_add_keeps_its_native_fetch_op() { + let (_ctx, module) = atomic_typed_module(AssignOp::AddAssign, |t| t.uchar_id, 100); + + assert_eq!(count_op(&module, Opcode::AtomicFetchAdd), 1); + assert_eq!( + count_op(&module, Opcode::AtomicCas), + 0, + "no retry loop needed" + ); + assert_eq!( + first_op(&module, Opcode::AtomicFetchAdd).size, + 8, + "the atomic operates at the object's own width" + ); +} + +/// `_Atomic _Bool` cannot: converting to `_Bool` is a test against zero, not +/// the truncation congruence permits, so the value stored has to be computed +/// before the exchange. +#[test] +fn test_atomic_bool_compound_assign_cannot_use_a_native_fetch_op() { + let (_ctx, module) = atomic_typed_module(AssignOp::SubAssign, |t| t.bool_id, 1); + + assert_eq!( + count_op(&module, Opcode::AtomicFetchSub), + 0, + "a native fetch-and-sub would store the raw 255" + ); + assert_eq!(count_op(&module, Opcode::AtomicCas), 1); + assert!( + count_op(&module, Opcode::SetNe) >= 1, + "the CAS loop must convert the result to _Bool before storing it" + ); +} + /// A complex member of an automatic struct is stored as two halves. /// /// A complex value travels by *address*, so storing it the way a scalar member @@ -7611,6 +7884,165 @@ fn test_memory_builtins_are_their_opcodes() { ); } +/// The constant `id` holds in `f`, if it is one -- through the narrowing a +/// conversion to the destination's own type leaves. +fn const_of(module: &Module, name: &str, id: PseudoId) -> Option { + let f = module.functions.iter().find(|f| f.name == name).unwrap(); + crate::ir::facts::ConstMap::new(f).get(id) +} + +/// Every `Memset` in `f`, as `(fill byte, length)`. +fn memsets_of(module: &Module, name: &str) -> Vec<(Option, Option)> { + insns_of(module, name) + .iter() + .filter(|i| i.op == Opcode::Memset) + .map(|i| { + ( + const_of(module, name, i.src[1]), + const_of(module, name, i.src[2]), + ) + }) + .collect() +} + +/// Every `Store` in `f`, as `(offset, width in bits, stored constant)`. A +/// store's `src` is `(address, value)`. +fn stores_of(module: &Module, name: &str) -> Vec<(i64, u32, Option)> { + insns_of(module, name) + .iter() + .filter(|i| i.op == Opcode::Store) + .map(|i| (i.offset, i.size, const_of(module, name, i.src[1]))) + .collect() +} + +/// Zero-initializing an aggregate is one `Memset` of the whole object, whose +/// length `memexpand` then weighs against `INLINE_LIMIT_BYTES`. +/// +/// `emit_aggregate_zero` hand-rolled the same 8/4/2/1 descent +/// `memexpand::block_chunks` already produces, but with **no** upper bound, so +/// `char buf[N] = {0}` emitted one store per chunk for any N: 8 KB cost 2081 +/// instructions in the function body and 1 MB did not finish compiling in 25 +/// minutes. Asking for the opcode instead makes the bound the shared one and +/// leaves the linearizer with no ladder of its own. +#[test] +fn test_aggregate_zero_is_one_memset_of_the_whole_object() { + let target = Target::new(Arch::X86_64, Os::Linux); + // Each declares an object and hands it to `sink` so nothing is dead. The + // only store left is the one element `{0}` names explicitly; every other + // byte is the memset's, whatever the object's size. + for (decl, bytes, explicit) in [ + ("char buf[200] = {0}; sink(buf);", 200, (0, 8, Some(0))), + ("char buf[7] = {0}; sink(buf);", 7, (0, 8, Some(0))), + ( + "struct S { int a; char b; } s = {0}; sink(&s);", + 8, + (0, 32, Some(0)), + ), + ] { + let src = format!("void sink(void *);\nvoid f(void) {{ {decl} }}\n"); + let module = linearize_source(&src, &target); + assert_eq!( + memsets_of(&module, "f"), + vec![(Some(0), Some(bytes))], + "{decl}: one memset of the whole object and nothing else" + ); + assert_eq!( + stores_of(&module, "f"), + vec![explicit], + "{decl}: the linearizer emits no chunk ladder of its own" + ); + } +} + +/// `char b[N] = "str"` zero-fills the elements the literal does not reach, +/// and only those. +/// +/// C17 6.7.9p21 initializes them as a static object would be. The `InitList` +/// arm of a local declaration calls `emit_aggregate_zero` first; the bare +/// string arm did not, so only the literal's own bytes and one terminator were +/// written. The braced form reaches the array through an initializer list, +/// which is already zeroed whole -- zeroing again there would double the +/// stores at `-O0`, where no `dse` runs to remove them. +#[test] +fn test_a_string_initializer_zero_fills_only_its_tail() { + let target = Target::new(Arch::X86_64, Os::Linux); + + // Three bytes written -- 'h', 'i', and the terminator -- then five left. + let bare = linearize_source( + "void sink(void *);\nvoid f(void) { char b[8] = \"hi\"; sink(b); }\n", + &target, + ); + assert_eq!( + stores_of(&bare, "f"), + vec![(0, 8, Some(0x68)), (1, 8, Some(0x69)), (2, 8, Some(0))] + ); + assert_eq!(memsets_of(&bare, "f"), vec![(Some(0), Some(5))]); + + // The braced form is preceded by the whole-object zero, so its tail is + // already done: one memset of 8, not one of 8 and another of 5. + let braced = linearize_source( + "void sink(void *);\nvoid f(void) { char b[8] = {\"hi\"}; sink(b); }\n", + &target, + ); + assert_eq!(memsets_of(&braced, "f"), vec![(Some(0), Some(8))]); + + // Exactly as long as the literal: C17 6.7.9p14 drops the terminator, and + // there is then no tail either. + let exact = linearize_source( + "void sink(void *);\nvoid f(void) { char b[2] = \"hi\"; sink(b); }\n", + &target, + ); + assert_eq!( + stores_of(&exact, "f"), + vec![(0, 8, Some(0x68)), (1, 8, Some(0x69))] + ); + assert_eq!(memsets_of(&exact, "f"), vec![]); +} + +/// A string literal initializing a *nested* array element steps the +/// destination by the element's own width and stops at its capacity. +/// +/// This path was a third hand-rolled copy of `store_string_units`. It +/// recognized all four literal kinds and then handled only the narrow one, so +/// a wide element was dropped and left zero; it stepped the destination by raw +/// bytes where a wide element is 2 or 4 bytes wide; and it had no capacity +/// clamp at all, so `char s[1][3] = {"hello"}` stored five bytes into a +/// three-byte object. +#[test] +fn test_a_nested_string_element_keeps_its_stride_and_capacity() { + let target = Target::new(Arch::X86_64, Os::Linux); + + // `unsigned short` is `char16_t`: two bytes of stride, and the literal + // reaches the array at all. + let wide = linearize_source( + "void sink(void *);\n\ + void f(void) { unsigned short u[2][3] = {u\"ab\", u\"cd\"}; sink(u); }\n", + &target, + ); + assert_eq!( + stores_of(&wide, "f"), + vec![ + (0, 16, Some(0x61)), + (2, 16, Some(0x62)), + (4, 16, Some(0)), + (6, 16, Some(0x63)), + (8, 16, Some(0x64)), + (10, 16, Some(0)), + ] + ); + + // Five units into a three-byte row: three stored, none past the row, and + // the terminator dropped with them. + let over = linearize_source( + "void sink(void *);\nvoid f(void) { char s[1][3] = {\"hello\"}; sink(s); }\n", + &target, + ); + assert_eq!( + stores_of(&over, "f"), + vec![(0, 8, Some(0x68)), (1, 8, Some(0x65)), (2, 8, Some(0x6c))] + ); +} + /// `mempcpy` is a `Memcpy` that calls `memcpy`, and its value the /// destination advanced by the length; `bcopy` is a `Memmove` that calls /// `memmove`, with its source and destination put back in `memmove`'s @@ -8812,3 +9244,907 @@ fn test_void_cast_of_a_complex_converts_nothing() { fn x86_64_linux() -> Target { Target::new(crate::target::Arch::X86_64, crate::target::Os::Linux) } + +// CFG consistency + +/// Every block's recorded successors are exactly the blocks its terminator +/// names, and `parents` is the inverse of `children`. +/// +/// Returns a description of the first inconsistency, or `None`. +fn cfg_inconsistency(func: &Function) -> Option { + use std::collections::HashSet; + + for bb in &func.blocks { + let children: HashSet = bb.children.iter().copied().collect(); + if children.len() != bb.children.len() { + return Some(format!("{}: duplicate edge in children", bb.id)); + } + + let named = match bb.insns.last() { + Some(last) if last.op.is_terminator() => crate::ir::propagate::terminator_targets(last), + // A block with no terminator falls through to nothing the CFG can + // name; `children` must then be empty too. + _ => HashSet::new(), + }; + + if named != children { + return Some(format!( + "{}: terminator names {:?} but children are {:?}", + bb.id, + sorted_ids(&named), + sorted_ids(&children), + )); + } + } + + // `parents` is the inverse of `children`. + let mut expected: std::collections::HashMap> = + std::collections::HashMap::new(); + for bb in &func.blocks { + for child in &bb.children { + expected.entry(*child).or_default().insert(bb.id); + } + } + for bb in &func.blocks { + let have: HashSet = bb.parents.iter().copied().collect(); + let want = expected.remove(&bb.id).unwrap_or_default(); + if have != want { + return Some(format!( + "{}: parents are {:?} but {:?} name it as a successor", + bb.id, + sorted_ids(&have), + sorted_ids(&want), + )); + } + } + + None +} + +fn sorted_ids(s: &std::collections::HashSet) -> Vec { + let mut v: Vec = s.iter().map(|b| b.0).collect(); + v.sort_unstable(); + v +} + +/// A `for` post-expression that splits the block still links the back edge from +/// the block the branch was emitted into. +/// +/// `&&`, `||` and `?:` leave `current_bb` on their merge block, so +/// `link_bb(post_bb, cond_bb)` recorded an edge out of a block that no longer +/// holds the terminator -- the loop's back edge went missing from the CFG while +/// a merge block gained an unrecorded one. Both `for` arms had it: the one in +/// `linearize_for` and its copy in the switch-body walker. +/// +/// Stated on the CFG rather than on the program's answer because the defect +/// makes the compiled loop non-terminating, which a runtime test cannot +/// observe without hanging. +#[test] +fn for_post_expression_splitting_the_block_keeps_the_back_edge() { + let target = Target::host(); + let cases = [ + ("and", "for (int i = 0; i < n; (void)(n && 1), i++) s += i;"), + ("or", "for (int i = 0; i < n; (void)(n || 0), i++) s += i;"), + ( + "ternary", + "for (int i = 0; i < n; (void)(n ? 1 : 2), i++) s += i;", + ), + ( + "and_in_cond_and_post", + "for (int i = 0; i < n && n; (void)(n && 1), i++) s += i;", + ), + ( + "nested_and", + "for (int i = 0; i < n; (void)(n && (i || 1)), i++) s += i;", + ), + ]; + + for (tag, loop_src) in cases { + // Plain, and again inside a switch body -- a separate copy of the + // lowering that carried the same defect. + let plain = format!("int f(int n) {{ int s = 0; {loop_src} return s; }}"); + let in_switch = format!( + "int f(int n) {{ switch (n) {{ case 5: {{ int s = 0; {loop_src} return s; }} \ + default: return 0; }} }}" + ); + + for (where_, src) in [("plain", &plain), ("in_switch", &in_switch)] { + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + assert!( + cfg_inconsistency(func).is_none(), + "{tag} / {where_}: {}\nsource: {src}", + cfg_inconsistency(func).unwrap() + ); + } + } +} + +/// The check above is only as good as its ability to see a broken edge, so +/// assert it rejects one. +#[test] +fn cfg_inconsistency_sees_a_misrecorded_edge() { + let target = Target::host(); + let module = linearize_source( + "int f(int n) { int s = 0; for (int i = 0; i < n; i++) s += i; return s; }", + &target, + ); + let mut func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f") + .clone(); + assert!(cfg_inconsistency(&func).is_none(), "baseline is consistent"); + + // Record a successor the terminator does not name -- exactly the shape the + // `for` defect produced. + let victim = func.blocks.len() - 1; + let bogus = func.blocks[0].id; + func.blocks[victim].children.push(bogus); + assert!( + cfg_inconsistency(&func).is_some(), + "an unnamed successor must be reported" + ); +} + +/// The same audit over every other lowering that can end a block with a +/// terminator after evaluating an expression: `while`, `do`/`while`, `switch`, +/// `if`, `?:`, `goto`, `break`/`continue` and the loop lowerings' copies in the +/// switch-body walker. +/// +/// Each source puts a short-circuit operator or a `?:` -- the things that split +/// the block and move `current_bb` to a merge block -- where the construct +/// evaluates an expression, so an edge linked from the block the construct +/// started in rather than the one it ended in shows up as an inconsistency. +#[test] +fn short_circuit_operands_keep_every_lowering_cfg_consistent() { + let target = Target::host(); + let cases = [ + ("while_cond", "while (n && s < 3) s++;"), + ("do_while_cond", "do { s++; } while (n && s < 3);"), + ("if_cond", "if (n && s) s = 1; else s = 2;"), + ("ternary", "s = n && 1 ? (n || 2) : (n ? 3 : 4);"), + ("for_cond", "for (int i = 0; i < n && n; i++) s += i;"), + ("for_init", "for (int i = (n && 1); i < n; i++) s += i;"), + ( + "switch_selector", + "switch (n && 1) { case 1: s = 1; break; }", + ), + ( + "switch_in_loop", + "while (s < 3) { switch (n && 1) { case 1: s++; break; default: s += 2; } }", + ), + ("break_after_split", "while (1) { if (n && 1) break; s++; }"), + ( + "continue_after_split", + "for (int i = 0; i < n; i++) { if (n || 0) continue; s++; }", + ), + ( + "goto_after_split", + "if (n && 1) goto done; s = 7; done: s++;", + ), + ( + "while_in_switch", + "switch (n) { case 5: while (n && s < 3) s++; break; default: s = 1; }", + ), + ( + "do_while_in_switch", + "switch (n) { case 5: do { s++; } while (n && s < 3); break; default: s = 1; }", + ), + ( + "nested_for_in_switch", + "switch (n) { case 5: for (int i = 0; i < n; (void)(n && 1), i++) \ + for (int j = 0; j < n; (void)(n || 0), j++) s++; break; default: s = 1; }", + ), + ( + "duffs_device", + "switch (n % 2) { case 0: do { s++; case 1: s += 2; } while (n && --n > 0); }", + ), + ]; + + for (tag, body) in cases { + let src = format!("int f(int n) {{ int s = 0; {body} return s; }}"); + let module = linearize_source(&src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + assert!( + cfg_inconsistency(func).is_none(), + "{tag}: {}\nsource: {src}", + cfg_inconsistency(func).unwrap() + ); + } +} + +/// Every access the linearizer emits to a `volatile` object carries the +/// marker, including the one that has no variable to ask. +/// +/// `LocalVar::is_volatile` and `GlobalFacts::is_volatile` answer for a named +/// object, and for `*p` with a `volatile int *p` there is none: `p` is an +/// ordinary pointer. So DCE saw no reason to keep the read and deleted every +/// discarded `volatile` access from `-O1` up. +#[test] +fn test_volatile_accesses_carry_the_marker() { + let target = Target::host(); + let src = "volatile int g;\n\ + volatile int *vp;\n\ + int plain;\n\ + int *pp;\n\ + volatile int arr[4];\n\ + struct T { volatile int a; };\n\ + struct T t;\n\ + void read_named(void) { g; }\n\ + void read_via_ptr(void) { *vp; }\n\ + void write_named(void) { g = 1; }\n\ + void write_via_ptr(void) { *vp = 1; }\n\ + void read_element(void) { arr[2]; }\n\ + void read_member(void) { t.a; }\n\ + void read_plain(void) { plain; }\n\ + void read_plain_ptr(void) { *pp; }\n"; + let module = linearize_source(src, &target); + + let accesses = |name: &str| -> Vec<(Opcode, bool)> { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.is_volatile_access())) + .collect() + }; + + // A named volatile object: one marked access each way. + assert_eq!(accesses("read_named"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("write_named"), vec![(Opcode::Store, true)]); + + // Through a pointer *to* volatile, the qualifier is on the pointee, so + // reading `vp` itself is plain and the access through it is volatile. + assert_eq!( + accesses("read_via_ptr"), + vec![(Opcode::Load, false), (Opcode::Load, true)] + ); + assert_eq!( + accesses("write_via_ptr"), + vec![(Opcode::Load, false), (Opcode::Store, true)] + ); + + // The qualifier reaches through an array's element type and a member's + // own type. + assert_eq!(accesses("read_element"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_member"), vec![(Opcode::Load, true)]); + + // And nothing unqualified is marked -- the marker that says "keep this" + // is worth nothing if it is on every access. + assert_eq!(accesses("read_plain"), vec![(Opcode::Load, false)]); + assert_eq!( + accesses("read_plain_ptr"), + vec![(Opcode::Load, false), (Opcode::Load, false)] + ); +} + +/// A member of a `volatile` object is itself volatile (C17 6.5.2.3p3/p4), so +/// every access to one carries the marker. +/// +/// The reverse direction -- a `volatile` member of a plain object -- always +/// worked, because there the member's own declared type carries the qualifier. +/// This is the other one: the qualifier is on the *object*, and +/// `find_member` answers with the member's declared type, which cannot show it. +/// So the load was unmarked and DCE deleted it from `-O1` up. +#[test] +fn test_a_member_of_a_volatile_object_carries_the_marker() { + let target = Target::host(); + let src = "struct S { int a; int b; };\n\ + struct N { struct S in; };\n\ + typedef volatile struct S VS;\n\ + volatile struct S vs;\n\ + volatile struct S *vp;\n\ + volatile struct S vsa[4];\n\ + volatile struct N vn;\n\ + VS vt;\n\ + struct S plain;\n\ + struct S *pp;\n\ + void read_direct(void) { vs.a; }\n\ + void read_arrow(void) { vp->a; }\n\ + void read_element(void) { vsa[2].a; }\n\ + void read_nested(void) { vn.in.a; }\n\ + void read_typedef(void) { vt.a; }\n\ + void write_direct(void) { vs.a = 1; }\n\ + void read_plain(void) { plain.a; }\n\ + void read_plain_arrow(void) { pp->a; }\n"; + let module = linearize_source(src, &target); + + let accesses = |name: &str| -> Vec<(Opcode, bool)> { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.is_volatile_access())) + .collect() + }; + + // Every spelling of "the object is volatile": directly, through a pointer + // to volatile, through an array's element type, through a nested member + // whose own type is qualified by the object above it, and through a + // typedef that carries the qualifier. + assert_eq!(accesses("read_direct"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_element"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_nested"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_typedef"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("write_direct"), vec![(Opcode::Store, true)]); + // `volatile struct S *vp` qualifies the pointee, so reading `vp` itself is + // an ordinary load and the access through it is the volatile one. + assert_eq!( + accesses("read_arrow"), + vec![(Opcode::Load, false), (Opcode::Load, true)] + ); + + // The control: an unqualified object's member is not marked, or the marker + // would mean nothing. + assert_eq!(accesses("read_plain"), vec![(Opcode::Load, false)]); + assert_eq!( + accesses("read_plain_arrow"), + vec![(Opcode::Load, false), (Opcode::Load, false)] + ); +} + +/// A `volatile` bit-field access is marked although the access is of the +/// carrier. +/// +/// `emit_bitfield_load`/`_store` read and write a storage unit whose type is +/// an unqualified integer, so no marker can be derived from the instruction's +/// own type. `mark_volatile_access` preserves one the emitter sets, and this is +/// the case it exists for. +#[test] +fn test_a_volatile_bitfield_access_carries_the_marker() { + let target = Target::host(); + let src = "struct B { volatile unsigned f : 3; unsigned g : 5; };\n\ + struct B b;\n\ + volatile struct B vb;\n\ + void read_field(void) { b.f; }\n\ + void read_object(void) { vb.g; }\n\ + void write_object(void) { vb.g = 1; }\n\ + void read_plain(void) { b.g; }\n\ + void write_plain(void) { b.g = 1; }\n"; + let module = linearize_source(src, &target); + + let accesses = |name: &str| -> Vec<(Opcode, bool)> { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.is_volatile_access())) + .collect() + }; + + // Both spellings: the field declared `volatile`, and an ordinary field of + // a `volatile` object. + assert_eq!(accesses("read_field"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_object"), vec![(Opcode::Load, true)]); + // A bit-field store is a read-modify-write of the carrier, and both halves + // of it are observable. + assert_eq!( + accesses("write_object"), + vec![(Opcode::Load, true), (Opcode::Store, true)] + ); + + // The controls. + assert_eq!(accesses("read_plain"), vec![(Opcode::Load, false)]); + assert_eq!( + accesses("write_plain"), + vec![(Opcode::Load, false), (Opcode::Store, false)] + ); +} + +/// A conditional may not speculate a `volatile` member, in either spelling. +/// +/// `Select` is the branchless form, and reaching it means both arms were +/// evaluated. C17 6.5.15p4 evaluates only one of them, and 5.1.2.3 makes each +/// volatile read an observable event -- so the arms may only collapse when +/// both are pure. `is_pure_expr` asked whether the *base* was pure, which a +/// named object always is. +#[test] +fn test_a_volatile_member_is_not_speculated() { + let target = Target::host(); + let src = "struct V { volatile unsigned status; unsigned other; };\n\ + struct P { unsigned one; unsigned other; };\n\ + struct V v;\n\ + volatile struct P vp;\n\ + struct P p;\n\ + unsigned member_is_volatile(int c) { return c ? v.status : v.other; }\n\ + unsigned object_is_volatile(int c) { return c ? vp.one : vp.other; }\n\ + unsigned all_plain(int c) { return c ? p.one : p.other; }\n"; + let module = linearize_source(src, &target); + + let selects = |name: &str| -> usize { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| i.op == Opcode::Select) + .count() + }; + + assert_eq!( + selects("member_is_volatile"), + 0, + "a volatile member may not be read on the path that did not select it" + ); + assert_eq!( + selects("object_is_volatile"), + 0, + "a member of a volatile object is volatile (C17 6.5.2.3p3)" + ); + assert_eq!( + selects("all_plain"), + 1, + "two ordinary member reads are pure and may still collapse" + ); +} + +/// A `volatile` access keeps the object it reaches in memory: promotion would +/// rewrite the access into a register `Copy`, and the access must happen. +/// +/// The variable need not itself be volatile. `ssa` tests +/// `LocalVar::is_volatile`, which answers no here -- the qualifier is on the +/// access alone. +#[test] +fn test_volatile_access_to_a_plain_local_blocks_promotion() { + let target = Target::host(); + let src = "int f(void) { int a = 1; return *(volatile int *)&a; }\n"; + let module = linearize_source(src, &target); + let mut func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("f") + .clone(); + + let volatile_loads = |func: &Function| -> usize { + func.blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| i.is_volatile_access()) + .count() + }; + assert_eq!(volatile_loads(&func), 1, "the cast qualifies the access"); + + let types = TypeTable::new(&target); + crate::ir::ssa::ssa_convert(&mut func, &types); + assert_eq!( + volatile_loads(&func), + 1, + "SSA promotion turned a volatile access into a register copy" + ); +} + +/// A short-circuit operator with a constant left operand emits no branch. +/// +/// `emit_logical_and`/`emit_logical_or` are *triangles*, not diamonds: only one +/// arm block exists, the other phi predecessor is the left operand's own block, +/// and that block's phi value is emitted before the branch. They also branch +/// through `branch_on(Controlling, ..)`, whose `Constant` case deliberately +/// emits a plain `Br` and elides the merge edge entirely. +/// +/// So they must not be folded into the generic two-way/diamond builder, which +/// takes a `PseudoId` condition and always emits a `Cbr`: `1 && g()` would +/// regain a dead conditional branch and a second, empty arm block. This test is +/// the guard on that -- it fails if the short-circuit lowerings are ever routed +/// through the diamond helper. +#[test] +fn a_constant_short_circuit_operand_emits_no_branch() { + let target = Target::host(); + + let count_cbr = |src: &str| -> usize { + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + func.blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| i.op == Opcode::Cbr) + .count() + }; + + // A constant controlling operand is decided at compile time. + for src in [ + "int g(void); int f(void) { return 1 && g(); }", + "int g(void); int f(void) { return 0 || g(); }", + ] { + assert_eq!( + count_cbr(src), + 0, + "a constant short-circuit operand needs no branch: {src}" + ); + } + + // The control: a runtime operand does branch, so the check above is not + // passing because nothing ever emits a Cbr. + for src in [ + "int g(void); int f(int x) { return x && g(); }", + "int g(void); int f(int x) { return x || g(); }", + ] { + assert_eq!( + count_cbr(src), + 1, + "a runtime short-circuit operand branches exactly once: {src}" + ); + } +} + +/// Every lowering that builds a two-armed conditional keeps the CFG +/// consistent, in all three places its blocks can be built: where control +/// reaches them, where it cannot -- before a `switch`'s first `case`, and +/// after a `goto` -- and where an arm jumps out from under it. +/// +/// These are the shapes that went through `self.current_bb.unwrap()` and so +/// crashed the compiler outright on the last two. They now share +/// `emit_diamond`, or read the block back through +/// `current_or_unreachable_bb`, which starts a block nothing branches to; the +/// point of auditing the CFG rather than only that lowering finished is that +/// such a block is *removed* again by `dce::remove_unreachable_blocks`, and a +/// mislinked edge into or out of it would outlive it. +#[test] +fn conditional_lowerings_keep_the_cfg_consistent() { + let target = Target::host(); + + // (tag, declarations, statement). The statement is placed reachable, then + // before a `switch`'s first `case`, then after a `goto`. + let shapes = [ + ("ternary", "int g(void);", "y = g() ? g() : g();"), + ("logical_and", "int g(void);", "y = g() && g();"), + ("logical_or", "int g(void);", "y = g() || g();"), + ("elvis", "int g(void);", "y = g() ?: g();"), + ( + "nested_ternary", + "int g(void);", + "y = g() ? (g() ? g() : g()) : g();", + ), + ( + "and_in_ternary", + "int g(void);", + "y = g() ? (g() && g()) : g();", + ), + ( + "sqrt_errno", + "double sqrt(double); double d;", + "d = sqrt(d);", + ), + ( + "complex_ternary", + "int g(void); _Complex double h(void);", + "(void)(g() ? h() : h());", + ), + ( + "complex_elvis", + "_Complex double h(void);", + "(void)(h() ?: h());", + ), + ( + "complex_int_div", + "_Complex int ci(void);", + "(void)(ci() / ci());", + ), + ( + "atomic_nand", + "_Atomic int a;", + "y = __atomic_fetch_nand(&a, 1, 5);", + ), + // An arm that jumps away leaves *that arm* without a block, which is a + // different read from the condition's. + ( + "goto_out_of_then_arm", + "int g(void);", + "y = x ? ({ goto L; g(); }) : g();", + ), + ( + "goto_out_of_else_arm", + "int g(void);", + "y = x ? g() : ({ goto L; g(); });", + ), + ]; + + for (tag, decls, stmt) in shapes { + let bodies = [ + ("reachable", stmt.to_string()), + ( + "before_first_case", + format!("switch (x) {{ {stmt} case 1: y = 1; }}"), + ), + ("after_goto", format!("goto L; {stmt}")), + ]; + + for (where_, body) in bodies { + let src = format!("{decls}\nint f(int x) {{ int y = 0; {body} L: return y; }}\n"); + let module = linearize_source(&src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + assert!( + cfg_inconsistency(func).is_none(), + "{tag} / {where_}: {}\nsource: {src}", + cfg_inconsistency(func).unwrap() + ); + } + } +} + +// VLA scope exit + +/// Every `stacksave` a function emits is matched by a `stackrestore` on each +/// path that leaves the scope it opened. +/// +/// A VLA's storage is released by restoring the stack pointer the scope saved, +/// so the two have to balance -- and on *every* exit, which for an `asm goto` +/// means one per edge. The declaration scope and the VLA scope are tracked +/// separately, and the second was opened at only three of the places that open +/// the first, so a VLA declared in a `for`-init clause or a statement +/// expression was never released. +/// +/// The consequence is currently masked: c17 always keeps a frame pointer +/// (`-fomit-frame-pointer` is accepted and ignored) and the backend resets +/// `%rsp` from it, so no program leaks stack today. That masking is a frame +/// choice rather than a guarantee, and the imbalance also poisons the forward +/// `goto` machinery, which reads `vla_marks.len()` as a depth -- an unclosed +/// scope makes every later label's recorded depth too high and the restore is +/// skipped. This is stated on the IR because that is where it is true. +#[test] +fn a_vla_scope_releases_the_stack_on_every_exit() { + let target = Target::host(); + + let counts = |src: &str| -> (usize, usize) { + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + let n = |op| { + func.blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| i.op == op) + .count() + }; + (n(Opcode::StackSave), n(Opcode::StackRestore)) + }; + + // (tag, source, expected saves, expected restores) + let cases: &[(&str, &str, usize, usize)] = &[ + // The control: a plain block already balances. + ( + "block", + "void s(int*); int f(int n){ for(int k=0;k<3;k++){ int a[n]; a[0]=k; s(a); } return 0; }", + 1, + 1, + ), + // A `for`-init clause is a declaration scope like any other. + ( + "for_init", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) for(int a[n];0;){ s(a); } return 0; }", + 1, + 1, + ), + // The switch-body walker carries a second copy of the `for` lowering. + ( + "for_init_in_switch", + "void s(int*); int f(int n,int x){ switch(x){ case 1: \ + for(int k=0;k<3;k++) for(int a[n];0;){ s(a); } return 0; } return 1; }", + 1, + 1, + ), + // A statement expression is a sixth scope entry. + ( + "stmt_expr", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) \ + (void)({ int a[n]; a[0]=k; s(a); 0; }); return 0; }", + 1, + 1, + ), + // Leaving the scope by `break` unwinds it. + ( + "break_out_of_for_init", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) \ + for(int a[n];;){ a[0]=k; s(a); break; } return 0; }", + 1, + 1, + ), + // And by a forward `goto`, which is also what the depth bookkeeping + // needs to stay right for every label after it. + ( + "goto_out_of_for_init", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) \ + for(int a[n];;){ a[0]=k; s(a); goto L; } L: return 0; }", + 1, + 1, + ), + // A computed `goto` leaves a scope exactly as a plain one does. + ( + "computed_goto", + "void s(int*); int f(int n){ void*p=&&L; for(int k=0;k<3;k++){ int a[n]; \ + a[0]=k; s(a); goto *p; } L: return 0; }", + 1, + 1, + ), + // `asm goto` has two exits, so it needs a restore on each: the + // fall-through and the label edge. One restore here means the jump + // leaves the scope without releasing it. + ( + "asm_goto", + "void s(int*); int f(int n){ for(int k=0;k<3;k++){ int a[n]; a[0]=k; s(a); \ + __asm__ goto(\"\" :::: L); } L: return 0; }", + 1, + 2, + ), + ]; + + for (tag, src, want_save, want_restore) in cases { + let (saves, restores) = counts(src); + assert_eq!( + (saves, restores), + (*want_save, *want_restore), + "{tag}: expected {want_save} stacksave / {want_restore} stackrestore, \ + got {saves} / {restores}\nsource: {src}" + ); + } +} + +/// The counts of `stacksave`/`stackrestore` in `f`, for the VLA scope tests. +fn vla_stack_ops(src: &str) -> (usize, usize) { + let target = Target::host(); + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + let n = |op| { + func.blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| i.op == op) + .count() + }; + (n(Opcode::StackSave), n(Opcode::StackRestore)) +} + +/// Entering a declaration scope and entering a VLA scope are one operation, +/// so nesting the first nests the second: each scope releases exactly what +/// was allocated after it was entered, innermost first. +#[test] +fn nested_scopes_each_release_only_their_own_vlas() { + let src = "void s(int*); int f(int n){ for(int k=0;k<2;k++){ int a[n]; \ + { int b[n]; s(b); } s(a); } return 0; }"; + assert_eq!( + vla_stack_ops(src), + (2, 2), + "each of the two scopes captures and releases once" + ); + + // And in the right order: the inner scope puts the stack back to what it + // captured on entry, then the outer one to what *it* captured. Restoring + // the outer mark first would free the inner array while it is still in + // scope. + let target = Target::host(); + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + let mut saves = Vec::new(); + let mut restores = Vec::new(); + for insn in func.blocks.iter().flat_map(|b| b.insns.iter()) { + match insn.op { + Opcode::StackSave => saves.push(insn.target.expect("stacksave target")), + Opcode::StackRestore => restores.push(insn.src[0]), + _ => {} + } + } + assert_eq!( + restores, + vec![saves[1], saves[0]], + "scopes must be released innermost first" + ); +} + +/// A VLA declared in a `for` init clause is allocated once, ahead of the +/// loop, and its scope *encloses* the loop's exit -- so neither `break` nor +/// `continue` may release it. `continue` especially: the storage is live on +/// the next iteration, and freeing it there would hand the loop a dangling +/// array. Only the loop's own scope, which ends after the exit block, puts +/// the stack back. +/// +/// This is what fixes the order in which [`Linearizer::push_vla_mark`] reads +/// the break and continue depths: before the loop pushes its targets, so +/// `unwind_vla_marks` sees the mark as taken *outside* the construct being +/// left and leaves it alone. +#[test] +fn a_for_init_vla_outlives_break_and_continue() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ for(int a[n];x;){ s(a); \ + if(x==1) continue; if(x==2) break; } return 0; }" + ), + (1, 1), + "the loop's scope is the only release; break and continue land inside it" + ); +} + +/// The mirror image: a VLA declared in the loop *body* is allocated afresh +/// every iteration, so every way out of the body has to release it -- the +/// fall-through through the body scope's end, and the `break` that jumps +/// past it. +#[test] +fn a_loop_body_vla_is_released_on_break_as_well_as_fallthrough() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ for(;x;){ int a[n]; s(a); \ + if(x==2) break; } return 0; }" + ), + (1, 2), + "one release on the break edge, one on the way out of the body" + ); +} + +/// A `switch` body is lowered by a walk of its own, and its braces are a +/// declaration scope there too: a VLA declared directly in it is released +/// when the switch ends, not left for the enclosing loop to accumulate. +#[test] +fn a_switch_body_block_is_a_scope() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ for(int k=0;k<2;k++) \ + switch(x){ default: { int a[n]; s(a); } } return 0; }" + ), + (1, 1), + "the switch body's block releases what it declared" + ); +} + +/// A backward `goto` to a label ahead of a VLA declaration leaves that +/// declaration's scope, so it restores the stack as it stood at the label -- +/// otherwise the loop the jump makes grows the stack every time round. The +/// label's depth is recorded per block, which is also what lets a computed +/// `goto` ask the same question of every candidate at once. +#[test] +fn a_backward_goto_past_a_vla_declaration_releases_it() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ lab: { int a[n]; s(a); \ + if(x--) goto lab; } return 0; }" + ), + (1, 2), + "the jump back to `lab` puts the stack where the label found it, and \ + the path that falls out of the block releases it too" + ); +} diff --git a/cc/ir/validate.rs b/cc/ir/validate.rs index 51e19f797..4a4673ae8 100644 --- a/cc/ir/validate.rs +++ b/cc/ir/validate.rs @@ -429,8 +429,10 @@ fn check_barrier_implies_side_effect(func: &Function, out: &mut Vec) { for (block, bb) in func.blocks.iter().enumerate() { for (index, insn) in bb.insns.iter().enumerate() { diff --git a/cc/main.rs b/cc/main.rs index 897cc6f92..9cefe2520 100644 --- a/cc/main.rs +++ b/cc/main.rs @@ -11,25 +11,18 @@ #![recursion_limit = "512"] -mod abi; -mod arch; -mod builtin_headers; -mod builtins; -mod constexpr; -mod diag; -mod float; -mod ir; -mod kw; -mod linkargs; -mod opt; -mod os; -mod parse; -mod rtlib; -mod strings; -mod symbol; -mod target; -mod token; -mod types; +use posixutils_cc::arch; +use posixutils_cc::builtins; +use posixutils_cc::diag; +use posixutils_cc::ir; +use posixutils_cc::linkargs; +use posixutils_cc::opt; +use posixutils_cc::parse; +use posixutils_cc::strings; +use posixutils_cc::symbol; +use posixutils_cc::target; +use posixutils_cc::token; +use posixutils_cc::types; use clap::Parser; use gettextrs::{gettext, gettext_args}; diff --git a/cc/parse/expression.rs b/cc/parse/expression.rs index 3349ad5e2..15aae0f5a 100644 --- a/cc/parse/expression.rs +++ b/cc/parse/expression.rs @@ -17,7 +17,7 @@ use crate::strings::StringId; use crate::symbol::{Namespace, Symbol}; use crate::token::lexer::{Position, SpecialToken, TokenType, TokenValue}; use crate::token::literal; -use crate::types::{Type, TypeId, TypeKind, TypeModifiers}; +use crate::types::{Type, TypeId, TypeKind}; use gettextrs::gettext; const DEFAULT_ARG_LIST_CAPACITY: usize = 8; @@ -359,20 +359,7 @@ impl<'a> Parser<'a> { /// never select the `const int` association. pub(crate) fn lvalue_converted_type(&mut self, typ: TypeId) -> TypeId { let decayed = self.decayed_type(typ); - - const QUALIFIERS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - - let ty = self.types.get(decayed); - if !ty.modifiers.intersects(QUALIFIERS) { - return decayed; - } - - let mut unqualified = ty.clone(); - unqualified.modifiers.remove(QUALIFIERS); - self.types.intern(unqualified) + self.types.unqualified(decayed) } /// Parse a conditional (ternary) expression: cond ? then : else @@ -1322,8 +1309,10 @@ impl<'a> Parser<'a> { &gettext("request for member in something not a structure or union"), ); self.types.int_id - } else if let Some(info) = self.types.find_member(resolved, member) { - info.typ + } else if let Some(typ) = self.types.member_access_type(t, resolved, member) { + // C17 6.5.2.3p3: so-qualified by the object, whose + // qualifiers are on `t` -- `resolved` has lost them. + typ } else { let member_name = self.idents.get_opt(member).unwrap_or(""); diag::error_args(dot_pos, "has no member named '{0}'", &[member_name]); @@ -1361,8 +1350,12 @@ impl<'a> Parser<'a> { ), ); self.types.int_id - } else if let Some(info) = self.types.find_member(resolved, member) { - info.typ + } else if let Some(typ) = + self.types.member_access_type(struct_type, resolved, member) + { + // C17 6.5.2.3p4: so-qualified by the *pointee*. + // `struct S *volatile p` qualifies `p`, not `*p`. + typ } else { let member_name = self.idents.get_opt(member).unwrap_or(""); diag::error_args( @@ -1713,7 +1706,10 @@ impl<'a> Parser<'a> { // let it through, so `1 << 1L` came out `long` and // `sizeof(1 << 1L)` answered 8 where gcc answers 4. BinaryOp::Shl | BinaryOp::Shr => { + // The promoted type of a *value*: unqualified, as every + // arithmetic result is (6.3.2.1p2). let promoted = self.types.integer_promote(left_type); + let promoted = self.types.unqualified(promoted); self.check_shift_count(op, promoted, &right); promoted } @@ -1887,6 +1883,9 @@ impl<'a> Parser<'a> { } else { self.types.integer_promote(op_typ) }; + // The result is a value, which has the unqualified type (6.3.2.1p2): + // `-v` is an `int` even where `v` is a `volatile int`. + let typ = self.types.unqualified(typ); // The *value* is promoted, not just the type it is computed at. The // conversion used to be left out, on the reasoning that the operand // is already in a wider register -- but nothing in the IR then says @@ -1958,8 +1957,24 @@ impl<'a> Parser<'a> { } } + /// The type the usual arithmetic conversions (C17 6.3.1.8) bring two + /// operands to, as an *rvalue* type. + /// + /// `common_type` answers with one of the operands' own `TypeId`s, so + /// `volatile int + int` came out `volatile int` and a qualifier the object + /// carried leaked into the type of a value. C17 6.3.2.1p2 drops the + /// qualifiers when an lvalue is converted to a value, and nothing + /// downstream may read an rvalue's type as "this expression touched a + /// volatile object" -- now that a member of a `volatile` object is itself + /// volatile (6.5.2.3p3), that leak would reach every `s.m + 1`. + /// + /// Stripping them here rather than in `common_type` keeps the latter a + /// pure question about conversion rank, which is what lets the linearizer + /// and the constant folder ask it through a `&TypeTable`: interning a + /// stripped type needs `&mut`. fn usual_arithmetic_conversions(&mut self, left: TypeId, right: TypeId) -> TypeId { - self.types.common_type(left, right) + let common = self.types.common_type(left, right); + self.types.unqualified(common) } /// Parse a C11 generic selection (C17 6.5.1.1): diff --git a/cc/parse/test_parser.rs b/cc/parse/test_parser.rs index fab761696..400971ebb 100644 --- a/cc/parse/test_parser.rs +++ b/cc/parse/test_parser.rs @@ -5475,6 +5475,111 @@ fn first_statement_of(tu: &TranslationUnit, n: usize) -> &Stmt { stmt } +/// C17 6.5.2.3p3/p4: `s.m` and `p->m` have the *so-qualified* version of the +/// member's type -- the member's type plus the qualifiers of the object. +/// +/// The parser is where that type is formed, and it took the member's declared +/// type unchanged: a member of a `volatile` object read as an ordinary `int` +/// (so DCE deleted the load) and a member of a `const` object was assignable. +#[test] +fn test_a_member_access_is_qualified_by_the_object() { + let quals = |src: &str, decls: &str| -> TypeModifiers { + let code = format!("struct S {{ int a; }};\nstruct N {{ struct S in; }};\n{decls}\nvoid t(void) {{ {src}; }}"); + let (tu, types, _, _) = parse_tu(&code).unwrap(); + let Stmt::Expr(expr) = first_statement(&tu) else { + panic!("{src}: expected an expression statement"); + }; + types.qualifiers(expr.typ.expect("a typed expression")) + }; + const NONE: TypeModifiers = TypeModifiers::empty(); + + // The object's qualifiers, in each spelling that reaches a member. + assert_eq!( + quals("vs.a", "volatile struct S vs;"), + TypeModifiers::VOLATILE + ); + assert_eq!(quals("cs.a", "const struct S cs;"), TypeModifiers::CONST); + assert_eq!( + quals("cvs.a", "const volatile struct S cvs;"), + TypeModifiers::CONST | TypeModifiers::VOLATILE + ); + assert_eq!( + quals("vsa[1].a", "volatile struct S vsa[2];"), + TypeModifiers::VOLATILE + ); + assert_eq!( + quals("vn.in.a", "volatile struct N vn;"), + TypeModifiers::VOLATILE, + "the intermediate member is qualified too, and carries it downward" + ); + assert_eq!( + quals("vt.a", "typedef volatile struct S VS; VS vt;"), + TypeModifiers::VOLATILE + ); + + // `->` takes them from the *pointee*, which is the object it names. + assert_eq!( + quals("vp->a", "volatile struct S *vp;"), + TypeModifiers::VOLATILE + ); + assert_eq!( + quals("qp->a", "struct S *volatile qp;"), + NONE, + "`struct S *volatile` qualifies the pointer, not what it points at" + ); + + // `_Atomic` does not travel: there is no atomic access to one member of an + // atomic object, and gcc does not pretend otherwise. + assert_eq!(quals("as.a", "_Atomic struct S as;"), NONE); + + // An unqualified object leaves the member's declared type alone -- and a + // qualifier on the *member* still reaches the access, which is the + // direction that always worked. + assert_eq!(quals("s.a", "struct S s;"), NONE); + assert_eq!( + quals("vm.v", "struct M { volatile int v; }; struct M vm;"), + TypeModifiers::VOLATILE + ); +} + +/// C17 6.3.2.1p2: converting an lvalue to a value drops the qualifiers, so an +/// arithmetic result is never qualified. +/// +/// `common_type` and `integer_promote` answer with one of the operands' own +/// type ids, so `volatile int + int` came out `volatile int` -- a qualifier on +/// the type of a value, which nothing may read as "this expression touched a +/// volatile object". +#[test] +fn test_an_arithmetic_result_is_unqualified() { + let quals = |src: &str| -> TypeModifiers { + let code = format!( + "struct S {{ int a; long l; }};\nvolatile struct S vs;\nvolatile int vi;\n\ + void t(void) {{ {src}; }}" + ); + let (tu, types, _, _) = parse_tu(&code).unwrap(); + let Stmt::Expr(expr) = first_statement(&tu) else { + panic!("{src}: expected an expression statement"); + }; + types.qualifiers(expr.typ.expect("a typed expression")) + }; + const NONE: TypeModifiers = TypeModifiers::empty(); + + // The usual arithmetic conversions, a shift (whose result is the promoted + // *left* operand's type), and a unary operator. + assert_eq!(quals("vs.a + 1"), NONE); + assert_eq!(quals("1 + vs.a"), NONE); + assert_eq!(quals("vs.l * vs.a"), NONE); + assert_eq!(quals("vi | 1"), NONE); + assert_eq!(quals("vs.a << 1"), NONE); + assert_eq!(quals("-vs.a"), NONE); + assert_eq!(quals("~vi"), NONE); + assert_eq!(quals("1 ? vs.a : 0"), NONE); + + // The lvalue itself keeps them: that is the whole point of the rule above, + // and the assignment check and the volatile marker both read it. + assert_eq!(quals("vs.a"), TypeModifiers::VOLATILE); +} + // Library builtins: abs, fabs, creal, conj, ... as checked calls /// The in-place call `expr` is, as (function, arguments), or a panic naming diff --git a/cc/tests/c11/atomics.rs b/cc/tests/c11/atomics.rs index 48f761bbf..8ffd0f719 100644 --- a/cc/tests/c11/atomics.rs +++ b/cc/tests/c11/atomics.rs @@ -948,3 +948,141 @@ int main(void) { "#; assert_eq!(compile_and_run("c11_atomic_spellings", code, &[]), 0); } + +/// A compound assignment to an `_Atomic` object computes at the same type as +/// one to an ordinary object. +/// +/// C17 6.5.16.2p3 defines `E1 op= E2` as `E1 = E1 op E2` bar evaluating `E1` +/// once, so the arithmetic happens at the type the usual arithmetic +/// conversions give the two operands -- and only the *result* is converted back +/// to the target. The atomic path converted the right operand down to the +/// target first and computed there, so `50 / -5` became `50 / 251` and stored +/// 0. The ordinary path already had this fixed, with a comment explaining it; +/// the atomic path had its own copy of the logic and did not. +/// +/// Add, subtract, and the bitwise operators are congruent modulo 2^n, so a +/// narrow computation agrees with a wide one and their native fetch-and-op +/// lowering stays correct. Division, remainder and the shifts are not, and all +/// of them already take the compare-and-swap loop. +/// +/// Each case is checked against the ordinary object beside it: the two paths +/// agreeing is the property, and their disagreeing is how this survived. +#[test] +fn c11_an_atomic_compound_assignment_computes_at_the_common_type() { + let code = r#" +int main(void) +{ + /* Division: the right operand must not be narrowed to unsigned char + first. 50 / -5 is -10 at int, stored as (unsigned char)-10 == 246. */ + _Atomic unsigned char ac = 50; ac /= -5; + unsigned char pc = 50; pc /= -5; + if (ac != pc || ac != 246) return 1; + + /* Remainder, likewise: 50 % -3 is 2. */ + _Atomic unsigned char am = 50; am %= -3; + unsigned char pm = 50; pm %= -3; + if (am != pm || am != 2) return 2; + + /* Signed division, where narrowing would also change the sign. */ + _Atomic signed char as = -100; as /= 3; + signed char ps = -100; ps /= 3; + if (as != ps || as != -33) return 3; + + /* The congruent operators must keep working -- they take the native + fetch-and-op lowering, not the CAS loop. */ + _Atomic unsigned char aa = 200; aa += 100; + unsigned char pa = 200; pa += 100; + if (aa != pa || aa != 44) return 4; + + _Atomic unsigned char an = 0xF0; an &= -1; + unsigned char pn = 0xF0; pn &= -1; + if (an != pn || an != 0xF0) return 5; + + return 0; +} +"#; + assert_eq!(compile_and_run("atomic_compound_common_type", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("atomic_compound_common_type_opt", code), + 0 + ); +} + +/// The value of a compound assignment is the value stored, converted. +/// +/// C17 6.5.16p3: an assignment expression has the value of the left operand +/// *after* the assignment. For a `_Bool` that means the value after conversion +/// to `_Bool`, so `b -= 1` on a false `b` yields 1 -- the memory and the +/// expression have to agree. c17's ordinary path did this and its atomic path +/// did not, recomputing the expression's value from a raw arithmetic result +/// and handing back 255 while storing 1. +/// +/// Note clang answers 255 here for the atomic case and 1 for the ordinary one, +/// i.e. it has the same split. This follows the standard and c17's own +/// non-atomic path rather than matching that. +#[test] +fn c11_an_atomic_compound_assignment_yields_the_value_it_stored() { + let code = r#" +int main(void) +{ + _Atomic _Bool ab = 0; int ar = (ab -= 1); + _Bool pb = 0; int pr = (pb -= 1); + if (ab != 1 || pb != 1) return 1; + if (ar != pr || ar != 1) return 2; + + _Atomic _Bool ab2 = 1; int ar2 = (ab2 += 7); + _Bool pb2 = 1; int pr2 = (pb2 += 7); + if (ab2 != 1 || pb2 != 1) return 3; + if (ar2 != pr2 || ar2 != 1) return 4; + + /* A narrowing store: the expression is the stored value, not the wide one. */ + _Atomic unsigned char au = 200; int aur = (au += 100); + unsigned char pu = 200; int pur = (pu += 100); + if (aur != pur || aur != 44) return 5; + + return 0; +} +"#; + assert_eq!(compile_and_run("atomic_compound_result", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("atomic_compound_result_opt", code), + 0 + ); +} + +/// A shift on an atomic object promotes its left operand, as any shift does. +/// +/// C17 6.5.7p3: the integer promotions are applied to each operand and the +/// result has the promoted left operand's type. So `s >>= 1` on a +/// `signed char` holding -8 shifts -8 at `int`, giving -4, and stores that -- +/// not a logical shift of the unsigned byte pattern, which would give 124. +/// +/// This one c17 already gets right and clang does not, so it is a guard rather +/// than a repair: the fix for the two tests above must not reach the shift by +/// computing at the target's width. +#[test] +fn c11_an_atomic_shift_promotes_its_left_operand() { + let code = r#" +int main(void) +{ + _Atomic signed char as = -8; as >>= 1; + signed char ps = -8; ps >>= 1; + if (as != ps || as != -4) return 1; + + _Atomic signed char al = -8; al <<= 2; + signed char pl = -8; pl <<= 2; + if (al != pl || al != -32) return 2; + + _Atomic unsigned char au = 200; au >>= 1; + unsigned char pu = 200; pu >>= 1; + if (au != pu || au != 100) return 3; + + return 0; +} +"#; + assert_eq!(compile_and_run("atomic_shift_promotion", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("atomic_shift_promotion_opt", code), + 0 + ); +} diff --git a/cc/tests/c89/control_flow.rs b/cc/tests/c89/control_flow.rs index 470a715e4..b3c5e5500 100644 --- a/cc/tests/c89/control_flow.rs +++ b/cc/tests/c89/control_flow.rs @@ -985,3 +985,130 @@ int main(void) assert_eq!(compile_and_run("label_addr_no_goto", code, &[]), 0); assert_eq!(compile_and_run_optimized("label_addr_no_goto_opt", code), 0); } + +/// A `for` post-expression that splits the block keeps the loop's back edge. +/// +/// `&&`, `||` and `?:` leave the current block on their merge, not on the block +/// the post-expression started in, so the branch back to the condition landed +/// in one block while the CFG edge was recorded from another. Every other loop +/// lowering reads `self.current_bb` back before linking; the two `for` arms did +/// not. +/// +/// The exhaustive statement of this is the CFG-consistency check in +/// `ir::test_linearize`; asserting it there rather than here is deliberate, +/// because with the defect present the compiled program does not merely return +/// the wrong answer, it never terminates -- a runtime test would hang the suite +/// instead of failing it. What is left here is the end-to-end answer, which is +/// safe to run only once the shape is known to be acyclic. +#[test] +fn c89_for_post_expression_that_splits_the_block_keeps_the_back_edge() { + let code = r#" +int and_in_post(int n) +{ + int s = 0; + for (int i = 0; i < n; (void)(n && 1), i++) + s += i; + return s; +} + +int or_in_post(int n) +{ + int s = 0; + for (int i = 0; i < n; (void)(n || 0), i++) + s += i; + return s; +} + +int ternary_in_post(int n) +{ + int s = 0; + for (int i = 0; i < n; (void)(n ? 1 : 2), i++) + s += i; + return s; +} + +/* The same shape inside a switch body, which is a second copy of the + lowering and carried the same defect. */ +int and_in_post_in_switch(int n) +{ + switch (n) { + case 5: { + int s = 0; + for (int i = 0; i < n; (void)(n && 1), i++) + s += i; + return s; + } + default: + return -2; + } +} + +int main(void) +{ + if (and_in_post(5) != 10) return 1; + if (or_in_post(5) != 10) return 2; + if (ternary_in_post(5) != 10) return 3; + if (and_in_post_in_switch(5) != 10) return 4; + /* A zero-trip loop still has to reach the exit. */ + if (and_in_post(0) != 0) return 5; + return 0; +} +"#; + assert_eq!(compile_and_run("for_post_splits_block", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("for_post_splits_block_opt", code), + 0 + ); +} + +/// A `case` label is converted to the promoted type of the controlling +/// expression, and matching happens after that conversion. +/// +/// C17 6.8.4.2p5: the constant expression of each `case` is converted to the +/// promoted type of the controlling expression. c17 evaluated labels at full +/// width and never converted them, so a label outside the controlling type +/// matched or missed depending on which lowering saw it -- a runtime selector +/// kept the label in the `switch` instruction, while the constant-selector fast +/// path compared at 128 bits with a signed test that ignored the switch's +/// signedness. The same switch answered differently depending on whether its +/// selector was a constant. +#[test] +fn c89_a_case_label_is_converted_to_the_controlling_type() { + let code = r#" +/* 4294967296 is 2^32: zero when converted to int. */ +int runtime_sel(int x) { switch (x) { case 4294967296LL: return 1; default: return 2; } } +int const_sel(void) { switch (0) { case 4294967296LL: return 1; default: return 2; } } + +/* -1 converted to unsigned int is 4294967295. */ +int unsigned_runtime(unsigned x) { switch (x) { case -1: return 1; default: return 2; } } +int unsigned_const(void) { switch (4294967295u) { case -1: return 1; default: return 2; } } + +/* A label that converts without changing value still behaves. */ +int plain(int x) { switch (x) { case -1: return 1; case 7: return 3; default: return 2; } } + +/* Short controlling expression: promoted to int, so the label is too. */ +int shorty(short x) { switch (x) { case 65536 + 5: return 1; case 5: return 3; default: return 2; } } + +int main(void) +{ + /* Both lowerings must agree, and both must match. */ + if (runtime_sel(0) != 1) return 1; + if (const_sel() != 1) return 2; + + if (unsigned_runtime(4294967295u) != 1) return 3; + if (unsigned_const() != 1) return 4; + + if (plain(-1) != 1 || plain(7) != 3 || plain(0) != 2) return 5; + + /* 65541 converts to short's promoted int unchanged, so it cannot match 5. */ + if (shorty(5) != 3) return 6; + + return 0; +} +"#; + assert_eq!(compile_and_run("case_label_conversion", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("case_label_conversion_opt", code), + 0 + ); +} diff --git a/cc/tests/c99/features.rs b/cc/tests/c99/features.rs index 1000f08eb..098d13701 100644 --- a/cc/tests/c99/features.rs +++ b/cc/tests/c99/features.rs @@ -1192,3 +1192,179 @@ fn c99_deeply_nested_constructs_compile() { code.push_str("int main(void) { return deep() == 1 && labels(3) == 1 ? 0 : 1; }\n"); assert_eq!(compile_and_run("deep_nesting", &code, &[]), 0); } + +/// Every way out of a scope that declares a VLA puts the stack pointer back. +/// +/// Observed without waiting for an exhaustion that a frame pointer hides: +/// the same declaration reached on the same path allocates at the same +/// address every time round *if and only if* the previous iteration released +/// it. A scope that never releases marches the address up the stack, and the +/// first mismatch is one iteration later. +/// +/// The declaration scope and the VLA scope used to be opened by hand at +/// separate call sites, and three of the sites that opened the first never +/// opened the second: a `for` init clause, the copy of the `for` lowering +/// inside the switch-body walk, and a statement expression. A computed +/// `goto` left no scope at all. +#[test] +fn c99_a_vla_scope_is_released_on_every_exit() { + let code = r#" +/* A VLA in a `for` init clause: allocated once per execution of the inner + `for` statement, released when that statement ends. */ +static int for_init(int n) { + void *first = 0; + for (int k = 0; k < 8; k++) + for (int a[n]; ; ) { + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 1; + break; /* leaves by `break` */ + } + return 0; +} + +/* The same shape, lowered by the switch-body walk instead. */ +static int for_init_in_switch(int n, int x) { + void *first = 0; + switch (x) { + case 1: + for (int k = 0; k < 8; k++) + for (int a[n]; ; ) { + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 2; + break; + } + return 0; + } + return 3; +} + +/* A statement expression is a block, so it is a scope. */ +static int stmt_expr(int n) { + void *first = 0; + int bad = 0; + for (int k = 0; k < 8; k++) + (void)({ + int a[n]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) bad = 4; + 0; + }); + return bad; +} + +/* Leaving by a forward `goto`, which also has to leave the label bookkeeping + straight for every label after it. */ +static int goto_out(int n) { + void *first = 0; + for (int k = 0; k < 8; k++) { + for (int a[n]; ; ) { + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 5; + goto next; + } + next: + ; + } + return 0; +} + +/* And by a computed `goto`, which leaves a scope exactly as a plain one + does. The jump is the loop, so every iteration goes through it. */ +static int computed_goto(int n) { + void *first = 0; + int k = 0; + void *back = &⊤ +top: + { + int a[n]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 6; + k++; + if (k < 8) goto *back; + } + return 0; +} + +int main(void) { + int n = 7; + int rc; + if ((rc = for_init(n)) != 0) return rc; + if ((rc = for_init_in_switch(n, 1)) != 0) return rc; + if ((rc = stmt_expr(n)) != 0) return rc; + if ((rc = goto_out(n)) != 0) return rc; + if ((rc = computed_goto(n)) != 0) return rc; + return 0; +} +"#; + assert_eq!( + compile_and_run("c99_vla_scope_release", code, &[]), + 0, + "a VLA scope must release its storage on every exit" + ); +} + +/// A block inside a `switch` body is a declaration scope there too: the +/// switch-body walk has a lowering of its own, and it used to release a VLA +/// declared in such a block without ever entering the declaration scope, so +/// the block's ordinary declarations outlived it. +#[test] +fn c99_a_switch_body_block_is_a_declaration_scope() { + let code = r#" +int main(void) { + int v = 1; + int x = 2; + switch (x) { + default: { + int v = 10; /* shadows the outer v only inside these braces */ + if (v != 10) return 1; + break; + } + } + if (v != 1) return 2; /* the inner declaration must not have escaped */ + + /* And a VLA declared there is released when the block ends. */ + void *first = 0; + for (int k = 0; k < 8; k++) + switch (x) { + default: { + int a[x + 5]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 3; + } + } + return 0; +} +"#; + assert_eq!(compile_and_run("c99_switch_body_block_scope", code, &[]), 0); +} + +/// A declaration in a statement expression does not outlive it. +/// +/// The statement expression had no declaration scope at all, so its locals +/// were inserted into the enclosing one and stayed there -- an inner `x` +/// went on shadowing the outer one after the `})`. +#[test] +fn c99_a_statement_expression_is_a_declaration_scope() { + let code = r#" +int main(void) { + int x = 1; + int y = ({ int x = 41; x + 1; }); + if (y != 42) return 1; + if (x != 1) return 2; /* the inner x must be gone */ + { + typedef int T; + int z = ({ typedef long T; (int)sizeof(T); }); + if (z != (int)sizeof(long)) return 3; + if ((int)sizeof(T) != (int)sizeof(int)) return 4; + } + return 0; +} +"#; + assert_eq!(compile_and_run("c99_stmt_expr_scope", code, &[]), 0); +} diff --git a/cc/tests/c99/initializers.rs b/cc/tests/c99/initializers.rs index 3220a78f0..208414f0a 100644 --- a/cc/tests/c99/initializers.rs +++ b/cc/tests/c99/initializers.rs @@ -2544,3 +2544,550 @@ fn c99_complex_constants_in_scalar_static_initializers() { assert_eq!(rc, 0); } } + +/// An excess array initializer is discarded, not written past the object. +/// +/// C17 6.7.9p2 makes more initializers than elements a constraint violation; +/// c17 already diagnoses it. The grouping pass never bounded its element +/// cursor by the array size, so the extra value was still stored -- one element +/// past the end, on top of whatever the frame put there. `int x[2] = {7, 8};` +/// followed by `int a[2] = {1, 2, 3};` read back `x = {3, 8}`. +#[test] +fn c99_excess_array_initializers_do_not_write_past_the_object() { + let code = r#" +int main(void) +{ + int x[2] = {7, 8}; + int a[2] = {1, 2, 3}; + if (a[0] != 1 || a[1] != 2) return 1; + if (x[0] != 7 || x[1] != 8) return 2; + + /* Several excess elements, and a designator that jumps back first. + C17 6.7.9p17: a positional initializer after a designator resumes at + the next subobject, so after `[0] = 1` the cursor is at index 1 and the + 9 overrides the earlier 2. Only the 10 and 11 are excess. Confirmed + against clang, which warns -Winitializer-overrides on the 9. */ + short y[2] = {5, 6}; + short b[2] = {[1] = 2, [0] = 1, 9, 10, 11}; + if (b[0] != 1 || b[1] != 9) return 3; + if (y[0] != 5 || y[1] != 6) return 4; + + /* The bound is on the index, not on the count. Here the array has three + elements and the initializer list has two, so a count-based rule keeps + the 1 -- but it resumes after `[2]`, i.e. at index 3, and is excess. + clang gives {0,0,3}. */ + int guard_before[2] = {11, 12}; + int d[3] = {[2] = 3, 1}; + int guard_after[2] = {13, 14}; + if (d[0] != 0 || d[1] != 0 || d[2] != 3) return 10; + if (guard_before[0] != 11 || guard_before[1] != 12) return 11; + if (guard_after[0] != 13 || guard_after[1] != 14) return 12; + + /* A nested array: the excess belongs to the inner object. */ + int z[2] = {8, 9}; + int c[2][2] = {{1, 2, 3}, {4, 5}}; + if (c[0][0] != 1 || c[0][1] != 2) return 5; + if (c[1][0] != 4 || c[1][1] != 5) return 6; + if (z[0] != 8 || z[1] != 9) return 7; + + /* A char array from a string literal that does not fit: C17 6.7.9p14 + allows exactly the terminator to be dropped, nothing more. */ + char w[2] = {'a', 'b'}; + char s[3] = "hello"; + if (s[0] != 'h' || s[1] != 'e' || s[2] != 'l') return 8; + if (w[0] != 'a' || w[1] != 'b') return 9; + + return 0; +} +"#; + assert_eq!(compile_and_run("excess_array_init", code, &[]), 0); + assert_eq!(compile_and_run_optimized("excess_array_init_opt", code), 0); +} + +/// The static form of the same defect: the emitted object is exactly as wide +/// as the array declares. +/// +/// `int garr[2] = {1, 2, 3};` emitted three `.long`s under an eight-byte +/// object, so the next symbol in the section absorbed the third. +#[test] +fn c99_excess_static_array_initializers_do_not_widen_the_object() { + let code = r#" +int garr[2] = {1, 2, 3}; +int after = 42; +short garr2[2] = {[1] = 2, [0] = 1, 9, 10}; +short after2 = 7; +/* Bounded by index, not by count -- see the automatic case. */ +int garr3[3] = {[2] = 3, 1}; +int after3 = 5; + +int main(void) +{ + if (garr[0] != 1 || garr[1] != 2) return 1; + if (after != 42) return 2; + if (garr2[0] != 1 || garr2[1] != 9) return 3; + if (after2 != 7) return 4; + if (garr3[0] != 0 || garr3[1] != 0 || garr3[2] != 3) return 5; + if (after3 != 5) return 6; + return 0; +} +"#; + assert_eq!(compile_and_run("excess_static_array_init", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("excess_static_array_init_opt", code), + 0 + ); +} + +/// A string literal initializing a nested array element, in every encoding. +/// +/// `is_string_for_char_array` accepts all four literal kinds, but the body that +/// consumed them handled only the narrow one and silently `continue`d on the +/// rest, so a wide element was dropped and left zero. The same loop stepped the +/// destination by *bytes* while a wide element is 2 or 4 bytes wide, and it had +/// no capacity clamp at all — a third hand-rolled copy of what +/// `store_string_units` already does correctly for the non-nested form. +/// +/// The static twin of each case was already right, which is how the two paths +/// could disagree: `static wchar_t sw[2][4] = {L"ab", L"cd"}` read back 97/99 +/// while the automatic form read back 0/0. +#[test] +fn c99_a_nested_string_literal_element_is_stored_in_every_encoding() { + let code = r#" +#include +/* does not exist on macOS, and the test needs only the two types -- + the same substitution the universal-character-name test above makes. */ +typedef unsigned short char16_t; +typedef unsigned int char32_t; + +int main(void) +{ + /* Narrow, and the tail of a short element must be zero. */ + char n[2][4] = {"ab", "cd"}; + if (n[0][0] != 'a' || n[0][1] != 'b' || n[0][2] != 0 || n[0][3] != 0) return 1; + if (n[1][0] != 'c' || n[1][1] != 'd' || n[1][2] != 0 || n[1][3] != 0) return 2; + + /* Wide: dropped entirely before the fix. */ + wchar_t w[2][4] = {L"ab", L"cd"}; + if ((int)w[0][0] != 'a' || (int)w[0][1] != 'b' || w[0][2] != 0) return 3; + if ((int)w[1][0] != 'c' || (int)w[1][1] != 'd' || w[1][2] != 0) return 4; + + char16_t u[2][4] = {u"ab", u"cd"}; + if ((int)u[0][0] != 'a' || (int)u[1][0] != 'c' || u[0][2] != 0) return 5; + + char32_t U[2][4] = {U"ab", U"cd"}; + if ((int)U[0][0] != 'a' || (int)U[1][0] != 'c' || U[0][2] != 0) return 6; + + /* The static path was always correct; the two must now agree. */ + static wchar_t sw[2][4] = {L"ab", L"cd"}; + if ((int)sw[0][0] != 'a' || (int)sw[1][0] != 'c') return 7; + + return 0; +} +"#; + assert_eq!(compile_and_run("nested_string_encodings", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("nested_string_encodings_opt", code), + 0 + ); +} + +/// A string literal too long for the array it initializes writes only as much +/// as fits. +/// +/// C17 6.7.9p14 allows exactly the terminating NUL to be dropped, and nothing +/// more. The nested-array path had no clamp, so `char s[1][3] = {"hello"}` +/// stored five bytes into a three-byte object — two of them past the whole +/// local, not merely into the next row. The guards on either side are what make +/// that visible rather than layout-dependent. +#[test] +fn c99_an_overlong_string_literal_does_not_write_past_its_array() { + let code = r#" +int main(void) +{ + unsigned char lo = 0xA5; + char s[1][3] = {"hello"}; + unsigned char hi = 0x5A; + if (s[0][0] != 'h' || s[0][1] != 'e' || s[0][2] != 'l') return 1; + if (lo != 0xA5 || hi != 0x5A) return 2; + + /* Exactly the terminator dropped: this is legal and keeps all three. */ + unsigned char lo2 = 0xA5; + char e[1][3] = {"abc"}; + unsigned char hi2 = 0x5A; + if (e[0][0] != 'a' || e[0][1] != 'b' || e[0][2] != 'c') return 3; + if (lo2 != 0xA5 || hi2 != 0x5A) return 4; + + /* Wide, where the stride is 4 bytes and a byte-stepped copy lands wrong. */ + unsigned char lo3 = 0xA5; + __WCHAR_TYPE__ w[1][2] = {L"xyz"}; + unsigned char hi3 = 0x5A; + if ((int)w[0][0] != 'x' || (int)w[0][1] != 'y') return 5; + if (lo3 != 0xA5 || hi3 != 0x5A) return 6; + + return 0; +} +"#; + assert_eq!(compile_and_run("overlong_nested_string", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("overlong_nested_string_opt", code), + 0 + ); +} + +/// `char buf[N] = "str"` zero-fills the bytes the literal does not reach. +/// +/// C17 6.7.9p21: the members not initialized explicitly are initialized as a +/// static object would be, i.e. to zero. The `InitList` arm of a local +/// declaration calls `emit_aggregate_zero` first; the string arm did not, so +/// only the literal's own bytes were written. +/// +/// On entry the backend zeroes the whole frame, which hides this the first time +/// through — the declaration is inside a loop so the second pass sees what the +/// first one left. All four encodings are affected. +#[test] +fn c99_a_string_initializer_zero_fills_the_rest_of_its_array() { + let code = r#" +#include + +int main(void) +{ + for (int pass = 0; pass < 2; pass++) { + char b[8] = "hi"; + if (b[2] != 0 || b[3] != 0 || b[7] != 0) return 1; + b[3] = 'Z'; + b[7] = 'Z'; + } + + for (int pass = 0; pass < 2; pass++) { + wchar_t w[4] = L"hi"; + if (w[2] != 0 || w[3] != 0) return 2; + w[3] = 'Z'; + } + + /* The braced form went through the InitList arm and was already correct; + both spellings must now agree. */ + for (int pass = 0; pass < 2; pass++) { + char c[8] = {"hi"}; + if (c[3] != 0 || c[7] != 0) return 3; + c[3] = 'Z'; + } + + return 0; +} +"#; + assert_eq!(compile_and_run("string_init_zero_fill", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("string_init_zero_fill_opt", code), + 0 + ); +} + +/// A later designated initializer replaces the subobject it names, not every +/// object whose bytes it touches. +/// +/// C17 6.7.9p19: an initializer for a subobject overrides any previously +/// listed initializer *for that subobject*, and initializers for other +/// subobjects are unaffected. The static path merged its field initializers by +/// byte span and dropped an earlier entry whole on any intersection, so +/// `.t = {1,2}` followed by `.t.y = 9` lost the `1` as well as the `2` -- while +/// the automatic path, which just stores in order and lets the later store land +/// on the earlier one, kept it. The two disagreed on the same initializer. +/// +/// Every case here is checked in both storage durations, against the values +/// gcc and clang produce. +#[test] +fn c99_a_designated_override_replaces_only_the_subobject_it_names() { + let code = r#" +struct T { int x, y; }; +struct S { struct T t; int z; }; +struct A { int a[3]; int z; }; + +struct S g1 = { .t = {1, 2}, .t.y = 9, .z = 7 }; +struct A g2 = { .a = {1, 2, 3}, .a[1] = 9, .z = 7 }; + +int main(void) +{ + struct S l1 = { .t = {1, 2}, .t.y = 9, .z = 7 }; + struct A l2 = { .a = {1, 2, 3}, .a[1] = 9, .z = 7 }; + + /* The override names .t.y, so .t.x keeps the 1 it was given. */ + if (g1.t.x != 1 || g1.t.y != 9 || g1.z != 7) return 1; + if (l1.t.x != 1 || l1.t.y != 9 || l1.z != 7) return 2; + + /* The same one level down: only element 1 is replaced. */ + if (g2.a[0] != 1 || g2.a[1] != 9 || g2.a[2] != 3 || g2.z != 7) return 3; + if (l2.a[0] != 1 || l2.a[1] != 9 || l2.a[2] != 3 || l2.z != 7) return 4; + + return 0; +} +"#; + assert_eq!(compile_and_run("designated_partial_override", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("designated_partial_override_opt", code), + 0 + ); +} + +/// "Later wins" means later in the initializer list, not later in the object. +/// +/// The static path sorted its field initializers by address before resolving +/// overlaps, so the rule was applied in the wrong order entirely: in +/// `{ .z = 7, .t.y = 9, .t = {1,2} }` the `.t = {1,2}` is written last and must +/// win, but after sorting it sat before `.t.y` and was the entry dropped. +#[test] +fn c99_a_designated_override_is_resolved_in_source_order() { + let code = r#" +struct T { int x, y; }; +struct S { struct T t; int z; }; + +/* The whole-field initializer comes last and wins, even though it names a + lower address than the override before it. */ +struct S g = { .z = 7, .t.y = 9, .t = {1, 2} }; + +/* And the other order, where the narrower one wins. */ +struct S h = { .t = {1, 2}, .z = 7, .t.y = 9 }; + +int main(void) +{ + struct S lg = { .z = 7, .t.y = 9, .t = {1, 2} }; + struct S lh = { .t = {1, 2}, .z = 7, .t.y = 9 }; + + if (g.t.x != 1 || g.t.y != 2 || g.z != 7) return 1; + if (lg.t.x != 1 || lg.t.y != 2 || lg.z != 7) return 2; + if (h.t.x != 1 || h.t.y != 9 || h.z != 7) return 3; + if (lh.t.x != 1 || lh.t.y != 9 || lh.z != 7) return 4; + + return 0; +} +"#; + assert_eq!(compile_and_run("designated_source_order", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("designated_source_order_opt", code), + 0 + ); +} + +/// Initializing a second member of a union resets it; it does not overlay the +/// first. +/// +/// This is the case where the two paths disagree the other way round. A union +/// holds one member at a time, so `{ .u.i = 0x01020304, .u.s.b = 9 }` leaves +/// the union holding `.u.s` with only `b` given a value and the rest zero -- +/// which is what the static path produced and what gcc and clang produce. The +/// automatic path stored the `int` and then stored one byte over it, keeping +/// the other three, so it read back `0x01020904`. +/// +/// It is here as a guard on the fix above: making the static path store in +/// source order the way the automatic path does would adopt this bug, so the +/// merge has to keep the union case distinct from the struct and array cases. +#[test] +fn c99_initializing_a_second_union_member_resets_the_union() { + let code = r#" +struct U { union { int i; struct { char a, b, c, d; } s; } u; }; +struct U g = { .u.i = 0x01020304, .u.s.b = 9 }; + +int main(void) +{ + struct U l = { .u.i = 0x01020304, .u.s.b = 9 }; + if (g.u.i != 0x900) return 1; + if (l.u.i != 0x900) return 2; + if (g.u.s.a != 0 || g.u.s.b != 9 || g.u.s.c != 0 || g.u.s.d != 0) return 3; + if (l.u.s.a != 0 || l.u.s.b != 9 || l.u.s.c != 0 || l.u.s.d != 0) return 4; + return 0; +} +"#; + assert_eq!(compile_and_run("union_member_reset", code, &[]), 0); + assert_eq!(compile_and_run_optimized("union_member_reset_opt", code), 0); +} + +/// A designated override of a bit-field replaces only that bit-field. +/// +/// The remaining half of the subobject rule. Bit-fields share a carrier, and +/// the carrier's bytes are merged downstream of the `Initializer` tree, so the +/// static path could not fold one override into an earlier initializer and fell +/// back to dropping it whole -- losing `b` in `{ .t = {1,2}, .t.a = 3 }` -- +/// while the automatic path stored the carrier and then stored over part of it, +/// keeping `b`. gcc keeps it. The two paths disagreeing is the defect; gcc's +/// answer is which way to settle it. +#[test] +fn c99_a_designated_override_of_a_bitfield_keeps_its_neighbours() { + let code = r#" +struct B { unsigned a : 4, b : 4; }; +struct S { struct B t; int z; }; + +struct S g = { .t = {1, 2}, .t.a = 3, .z = 7 }; +struct S g2 = { .t = {1, 2}, .t.b = 5 }; + +int main(void) +{ + struct S l = { .t = {1, 2}, .t.a = 3, .z = 7 }; + struct S l2 = { .t = {1, 2}, .t.b = 5 }; + + if (g.t.a != 3 || g.t.b != 2 || g.z != 7) return 1; + if (l.t.a != 3 || l.t.b != 2 || l.z != 7) return 2; + if (g2.t.a != 1 || g2.t.b != 5) return 3; + if (l2.t.a != 1 || l2.t.b != 5) return 4; + + return 0; +} +"#; + assert_eq!( + compile_and_run("designated_bitfield_override", code, &[]), + 0 + ); + assert_eq!( + compile_and_run_optimized("designated_bitfield_override_opt", code), + 0 + ); +} + +/// An override naming a subobject of the union member already held keeps the +/// rest of that member. +/// +/// `{ .u = {1,2}, .u.p.y = 9 }` initializes the union's first member and then +/// overrides one of *its* members, so the union still holds `p` and `p.x` keeps +/// the 1 it was given. c17 reset the union instead, because the `Initializer` +/// tree records no discriminant and the merge could not tell "the same member, +/// deeper" from "a different member" -- and resetting is right only for the +/// second. Both storage durations agreed on the wrong answer, so nothing caught +/// it. +/// +/// The companion case, where a *different* member is named and the union really +/// is reset, is covered by +/// `c99_initializing_a_second_union_member_resets_the_union`, which must keep +/// passing: the two are what distinguish the rule. +#[test] +fn c99_an_override_inside_the_held_union_member_keeps_the_rest() { + let code = r#" +struct P { int x, y; }; +struct N { union { struct P p; int i; } u; }; + +struct N g = { .u = {1, 2}, .u.p.y = 9 }; + +int main(void) +{ + struct N l = { .u = {1, 2}, .u.p.y = 9 }; + if (g.u.p.x != 1 || g.u.p.y != 9) return 1; + if (l.u.p.x != 1 || l.u.p.y != 9) return 2; + return 0; +} +"#; + assert_eq!( + compile_and_run("union_member_deeper_override", code, &[]), + 0 + ); + assert_eq!( + compile_and_run_optimized("union_member_deeper_override_opt", code), + 0 + ); +} + +/// An initializer for a whole struct supersedes an earlier one for a +/// bit-field inside it, including the bit-fields it says nothing about. +/// +/// The other direction of the bit-field rule, and the one the automatic path +/// had wrong: it stored the bit-field, then stored the struct's own +/// bit-fields over it, and `c` -- which `{1, 2}` does not mention -- kept the +/// 7. The static path dropped the earlier entry whole and was right. gcc and +/// clang zero it. +/// +/// The objects here are deliberately wider than eight bytes, to keep the +/// assertions clear of an unrelated x86-64 defect that widens a 32-bit store +/// at offset 0 of an eight-byte local to 64 bits. +#[test] +fn c99_a_whole_struct_initializer_supersedes_an_earlier_bitfield() { + let code = r#" +struct B { unsigned a : 4, b : 4, c : 4; }; +struct S { struct B t; int z; long pad; }; + +struct S g = { .t.c = 7, .t = {1, 2} }; + +int main(void) +{ + struct S l = { .t.c = 7, .t = {1, 2} }; + + if (g.t.a != 1 || g.t.b != 2 || g.t.c != 0) return 1; + if (l.t.a != 1 || l.t.b != 2 || l.t.c != 0) return 2; + + return 0; +} +"#; + assert_eq!(compile_and_run("bitfield_superseded", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("bitfield_superseded_opt", code), + 0 + ); +} + +/// A bit-field naming a second member of a union resets the union, as any +/// other initializer for a second member does. +/// +/// A bit-field is stored by reading its carrier and writing it back, so the +/// automatic path emitted no fill for it and three bytes of the `int` showed +/// through the `struct` that replaced it. It is not that a bit-field clears +/// nothing -- it clears nothing *of its own*, because its neighbours in the +/// carrier are other objects -- but what the union it displaces requires. +#[test] +fn c99_a_bitfield_naming_a_second_union_member_resets_the_union() { + let code = r#" +struct U { union { int i; struct { unsigned a : 4, b : 4; } s; } u; long pad; }; + +struct U g = { .u.i = 0x01020304, .u.s.a = 3 }; + +int main(void) +{ + struct U l = { .u.i = 0x01020304, .u.s.a = 3 }; + + if (g.u.i != 3 || g.u.s.a != 3 || g.u.s.b != 0) return 1; + if (l.u.i != 3 || l.u.s.a != 3 || l.u.s.b != 0) return 2; + + return 0; +} +"#; + assert_eq!(compile_and_run("bitfield_union_reset", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("bitfield_union_reset_opt", code), + 0 + ); +} + +/// Naming a subobject of a union member the union does *not* hold resets it, +/// even where the two members are the same size and the same shape. +/// +/// The guard on the fold above. Knowing which member is held comes from the +/// initializer list, not from the lowered bytes, so `struct P` and `struct Q` +/// being indistinguishable once lowered costs nothing: `.u = {1, 2}` gives +/// `p` a value and `.u.q.d = 9` names `q`, so the union comes to hold `q` +/// with only `d` given a value. Reading it back through the *other* member +/// would be undefined; `q.c` is not. +#[test] +fn c99_an_override_naming_another_union_member_resets_it_whatever_its_shape() { + let code = r#" +struct P { int x, y; }; +struct Q { int c, d; }; +struct N { union { struct P p; struct Q q; } u; long pad; }; + +struct N g = { .u = {1, 2}, .u.q.d = 9 }; + +/* And the fold, in the same union, when the member named is the held one. */ +struct N h = { .u = {1, 2}, .u.p.y = 9 }; + +int main(void) +{ + struct N l = { .u = {1, 2}, .u.q.d = 9 }; + struct N m = { .u = {1, 2}, .u.p.y = 9 }; + + if (g.u.q.c != 0 || g.u.q.d != 9) return 1; + if (l.u.q.c != 0 || l.u.q.d != 9) return 2; + if (h.u.p.x != 1 || h.u.p.y != 9) return 3; + if (m.u.p.x != 1 || m.u.p.y != 9) return 4; + + return 0; +} +"#; + assert_eq!(compile_and_run("union_other_member_reset", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("union_other_member_reset_opt", code), + 0 + ); +} diff --git a/cc/tests/codegen/block_moves.rs b/cc/tests/codegen/block_moves.rs new file mode 100644 index 000000000..04bfc5c77 --- /dev/null +++ b/cc/tests/codegen/block_moves.rs @@ -0,0 +1,263 @@ +// +// Copyright (c) 2025-2026 Jeff Garzik +// +// This file is part of the posixutils-rs project covered under +// the MIT License. For the full license text, please see the LICENSE +// file in the root directory of this project. +// SPDX-License-Identifier: MIT +// +// Blocks of bytes the back ends move, and the two rules they obey. +// +// An object's bytes are moved in descending power-of-two chunks, and the moves +// are bounded: past a threshold the copy becomes a bulk primitive instead of an +// unrolled run. `memexpand::block_chunks` and `BlockOp::limit` state both rules +// for the IR, and `codegen_struct_copy_across_the_inline_threshold` records the +// incident that put them there. +// +// The back ends cannot call `memcpy` -- they are past the point where a call +// can be synthesized -- so their bulk primitive is `rep movsq` or a counted +// loop. But the chunk rule is the same, and each site that wrote its own copy +// of it either rounded the size *up* or dropped the bound: +// +// * a two-SSE struct parameter stored both halves at eight bytes, so the +// four-byte high half of a 12-byte struct wrote four bytes past it; +// * the spilled-parameter prologue stepped eight regardless of width; +// * `va_arg` of a large aggregate unrolled with no bound at all; +// * the stacked-argument copy rounded the *source* read up, which is not the +// same as the destination: argument slots really are eightbyte-granular, so +// writing eight is correct where reading eight is not. +// +// Sizes here are deliberately not multiples of eight, and guards sit either +// side of the objects, so a rounded-up move shows as a wrong byte rather than +// landing in padding and passing by luck. +// + +use crate::codegen::asm_probe::{asm_for_with, body_of, AARCH64_LINUX, X86_64_LINUX}; +use crate::common::compile_and_run; + +/// How many instructions the body of `func` has. +fn body_insns(asm: &str, func: &str) -> usize { + body_of(asm, func) + .lines() + .filter(|l| { + let t = l.trim(); + !t.is_empty() && !t.starts_with('.') && !t.starts_with('#') && !t.ends_with(':') + }) + .count() +} + +/// A two-SSE struct parameter's high half is as wide as the half, not as wide +/// as a register. +/// +/// `struct P { float x, y, z; }` is classified into two SSE eightbytes, but the +/// second holds only four bytes. The prologue used one `FpSize` for both, so it +/// stored eight and wrote four bytes past the object. +#[test] +fn codegen_a_two_sse_struct_parameter_stores_only_its_own_bytes() { + let src = "\ +struct P { float x, y, z; }; +__attribute__((noinline)) float probe(struct P p) { float g = 99.f; return p.x + p.y + p.z + g; } +"; + let asm = asm_for_with("two_sse_param", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + let wide = body.matches("movsd").count(); + assert!( + wide <= 1, + "the high half of a 12-byte two-SSE struct is 4 bytes, so at most one \ + 8-byte fp store belongs in the prologue; found {wide}:\n{body}" + ); +} + +/// The spilled-parameter prologue moves no more than the parameter. +/// +/// `copy_incoming_arg_to_local` stepped eight bytes at a time regardless of +/// width, so a 12-byte struct read eight bytes at the incoming area's offset 8 +/// and wrote eight at the local's -- four past a local that is exactly twelve +/// bytes, because a slot is only rounded up to its type's own alignment. +#[test] +fn codegen_a_spilled_struct_parameter_is_copied_no_wider_than_itself() { + let src = "\ +struct P { int a, b, c; }; +__attribute__((noinline)) int probe(long a, long b, long c, long d, long e, long f, + struct P p) +{ return p.a + p.b + p.c; } +"; + let asm = asm_for_with("spilled_param", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + + // The incoming argument area is at a *positive* displacement from %rbp -- + // the saved frame pointer and return address are below it -- so reads of + // the spilled parameter are the moves from a positive offset. Everything + // the function writes is at a negative one. Matching on the substring + // "8(%rbp)" is not enough: "-88(%rbp)" ends with it. + let incoming_reads = |mnemonic: &str| { + body.lines() + .filter_map(|l| { + let t = l.trim(); + let rest = t.strip_prefix(mnemonic)?.trim_start(); + let (disp, _) = rest.split_once("(%rbp)")?; + disp.parse::().ok().filter(|d| *d > 0) + }) + .count() + }; + + // A 12-byte object is 8 + 4: exactly one eight-byte read, and the tail read + // with a four-byte one. + assert_eq!( + incoming_reads("movq"), + 1, + "a 12-byte spilled parameter has one 8-byte chunk, so one 8-byte read \ + of the incoming area; a second means the 4-byte tail was read as 8:\n{body}" + ); + assert_eq!( + incoming_reads("movl"), + 1, + "and its 4-byte tail is read with a 4-byte move:\n{body}" + ); +} + +/// `va_arg` of a large aggregate is bounded, like every other block move. +/// +/// The `va_arg` byte copy had no limit, so fetching a 4 KB aggregate emitted one +/// load/store pair per chunk -- about 1100 instructions on each target, and +/// linear in the object, so a 256 KB aggregate would be the compile-time +/// explosion `emit_aggregate_zero` used to be. +#[test] +fn codegen_va_arg_of_a_large_aggregate_is_bounded() { + let src = "\ +#include +struct Big { char c[4096]; }; +void sink(struct Big *); +void probe(int n, ...) +{ + va_list ap; + va_start(ap, n); + struct Big b = va_arg(ap, struct Big); + sink(&b); + va_end(ap); +} +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("va_arg_big", triple, src, &["-O2"]); + let n = body_insns(&asm, "probe"); + assert!( + n < 200, + "fetching a 4096-byte aggregate through va_arg must use a bulk copy, \ + not one pair per chunk: {n} instructions on {triple}" + ); + } +} + +/// The stacked-argument copy reads only the object's own bytes. +/// +/// The destination is the outgoing argument area, which is allocated in whole +/// eightbytes -- so rounding the *write* up is correct and deliberate. The read +/// is from the object, which is not, and a 12-byte struct read eight bytes at +/// offset 8. Four of them belong to whatever follows it, and the read faults if +/// the object ends a page. +#[test] +fn codegen_a_stacked_argument_reads_only_its_object() { + let src = "\ +struct P { int a, b, c; }; +void g(long, long, long, long, long, long, struct P); +void probe(struct P p) { g(1, 2, 3, 4, 5, 6, p); } +"; + let asm = asm_for_with("stacked_arg_src", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + + // Only the *read* is constrained. In AT&T order the memory operand of a + // load comes first, which is what distinguishes `movq 8(%r11), %rax` -- + // reading four bytes past a 12-byte object -- from `movq %rax, 8(%rsp)`, + // a write into the outgoing argument area. That area is allocated in whole + // eightbytes and its padding is unspecified, so the store's width is the + // back end's choice and this test does not pin it. + let wide_source_reads = body + .lines() + .filter(|l| { + let t = l.trim(); + let Some(operands) = t.strip_prefix("movq ") else { + return false; + }; + let Some((src_operand, _)) = operands.split_once(',') else { + return false; + }; + let src_operand = src_operand.trim(); + src_operand.starts_with("8(%r") && !src_operand.contains("%rbp") + }) + .count(); + assert_eq!( + wide_source_reads, 0, + "a 12-byte object has 4 bytes at offset 8, so reading 8 there is 4 past it:\n{body}" + ); +} + +/// The answers, with guards, across the sizes these paths classify differently. +/// +/// The assembly checks above pin the widths; this pins that the values survive. +/// None of the sizes is a multiple of eight and every object is fenced, so an +/// over-copy shows up as a clobbered guard rather than as padding nobody reads. +#[test] +fn codegen_block_moved_parameters_keep_their_values() { + let code = r#" +#include + +struct P3 { int a, b, c; }; /* 12 bytes, spilled/stacked */ +struct F3 { float x, y, z; }; /* 12 bytes, two SSE */ +struct B7 { unsigned char c[7]; }; /* 7 bytes */ +struct B13 { unsigned char c[13]; }; /* 13 bytes */ + +__attribute__((noinline)) int take_p3(long a, long b, long c, long d, long e, + long f, struct P3 p) +{ return p.a + p.b + p.c; } + +__attribute__((noinline)) float take_f3(struct F3 p) { return p.x + p.y + p.z; } + +__attribute__((noinline)) int take_b7(struct B7 v) +{ + int s = 0; + for (int i = 0; i < 7; i++) s += v.c[i]; + return s; +} + +__attribute__((noinline)) int take_b13(long a, long b, long c, long d, long e, + long f, struct B13 v) +{ + int s = 0; + for (int i = 0; i < 13; i++) s += v.c[i]; + return s; +} + +__attribute__((noinline)) int va_b13(int n, ...) +{ + va_list ap; + va_start(ap, n); + struct B13 v = va_arg(ap, struct B13); + va_end(ap); + int s = 0; + for (int i = 0; i < 13; i++) s += v.c[i]; + return s; +} + +int main(void) +{ + unsigned char lo = 0xA5; + struct P3 p = {1, 2, 3}; + struct F3 f = {1.f, 2.f, 4.f}; + struct B7 b7; + struct B13 b13; + unsigned char hi = 0x5A; + + for (int i = 0; i < 7; i++) b7.c[i] = (unsigned char)(i + 1); + for (int i = 0; i < 13; i++) b13.c[i] = (unsigned char)(i + 1); + + if (take_p3(1, 2, 3, 4, 5, 6, p) != 6) return 1; + if (take_f3(f) != 7.f) return 2; + if (take_b7(b7) != 28) return 3; + if (take_b13(1, 2, 3, 4, 5, 6, b13) != 91) return 4; + if (va_b13(0, b13) != 91) return 5; + if (lo != 0xA5 || hi != 0x5A) return 6; + return 0; +} +"#; + assert_eq!(compile_and_run("block_moved_params", code, &[]), 0); +} diff --git a/cc/tests/codegen/cross_abi.rs b/cc/tests/codegen/cross_abi.rs index af6423798..44690cbe4 100644 --- a/cc/tests/codegen/cross_abi.rs +++ b/cc/tests/codegen/cross_abi.rs @@ -25,7 +25,8 @@ use super::asm_probe::{ asm_for, asm_for_with, body_of, AARCH64_DARWIN, AARCH64_LINUX, X86_64_LINUX, }; use crate::common::{ - aarch64_cross_available, compile_with_host_cc, create_c_file, cross_link_and_run, run_c17, + aarch64_cross_available, compile_and_run, compile_with_host_cc, create_c_file, + cross_link_and_run, run_c17, }; /// AAPCS64 passes a `_Complex` as a two-element HFA, so it occupies **two** @@ -3446,3 +3447,248 @@ int named_s(struct P p) { return ns(p, 1); } ); } } + +/// The bytes a prologue stores into the frame, summed by the width of each +/// store's mnemonic. +/// +/// Only the prologue: the region before the first `.L` block label, which is +/// where the incoming arguments are written to their locals. A store is a move +/// whose *destination* is the frame, so the line ends with `(%rbp)`; a load +/// from the same place has it in the middle. +fn prologue_frame_store_bytes(body: &str) -> i64 { + body.lines() + .take_while(|l| !l.trim().starts_with(".L")) + .filter_map(|l| { + let t = l.trim(); + let (mnemonic, operands) = t.split_once(' ')?; + if !operands.ends_with("(%rbp)") { + return None; + } + match mnemonic { + "movq" | "movsd" => Some(8), + "movl" | "movss" => Some(4), + "movw" => Some(2), + "movb" => Some(1), + _ => None, + } + }) + .sum() +} + +/// A composite parameter that arrives in registers is stored no wider than it +/// is. +/// +/// Each eightbyte travels in a whole register, but the last eightbyte of a +/// composite that is not a multiple of eight holds fewer bytes than the +/// register does -- five, for a thirteen-byte one. `grow_frame` rounds a slot +/// up only to the type's own alignment, which is *one* for `unsigned char[13]`, +/// so storing the register's eight wrote three bytes past the local. +#[test] +fn codegen_a_register_pair_parameter_stores_only_its_own_bytes() { + let src = "\ +struct B13 { unsigned char c[13]; }; +int probe(struct B13 v) { return v.c[0] + v.c[12]; } +"; + let asm = asm_for_with("reg_pair_tail", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + assert_eq!( + prologue_frame_store_bytes(body), + 13, + "a thirteen-byte parameter is 8 + 4 + 1, and nothing more:\n{body}" + ); +} + +/// The same, for an all-SSE composite whose last eightbyte is a width no +/// floating-point store has. +/// +/// `struct { float a, b, c; _Float16 d; }` is fourteen bytes when packed, and +/// both of its eightbytes are SSE class -- the second holds six bytes. There +/// is no six-byte SSE store, so the register has to go through a general one; +/// rounding the width up instead wrote two bytes past the object. +#[test] +fn codegen_a_packed_two_sse_parameter_stores_only_its_own_bytes() { + let src = "\ +struct __attribute__((packed)) P6 { float a, b, c; _Float16 d; }; +float probe(struct P6 p) { return p.a + p.b + p.c; } +"; + let asm = asm_for_with("packed_two_sse", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + assert_eq!( + prologue_frame_store_bytes(body), + 14, + "a fourteen-byte two-SSE parameter is 8 + 4 + 2, and nothing more:\n{body}" + ); +} + +/// The bulk `va_arg` copy still moves the ragged tail. +/// +/// A bound is only half of the rule: `rep movsq` and the counted loop both +/// move whole eightbytes (sixteen bytes, for the loop), and 4093 bytes is +/// neither. A copy that stopped at the last whole unit would leave the last +/// bytes of the aggregate unwritten, which no instruction count can show. +#[test] +fn codegen_a_bulk_va_arg_copy_moves_the_ragged_tail() { + let src = "\ +#include +struct Odd { unsigned char c[4093]; }; +void sink(struct Odd *); +void probe(int n, ...) +{ + va_list ap; + va_start(ap, n); + struct Odd o = va_arg(ap, struct Odd); + sink(&o); + va_end(ap); +} +"; + for (triple, byte_move) in [(X86_64_LINUX, "movb"), (AARCH64_LINUX, "ldrb")] { + let asm = asm_for_with("va_arg_odd", triple, src, &["-O2"]); + let body = body_of(&asm, "probe"); + assert!( + body.contains(byte_move), + "4093 bytes ends on an odd byte, so the tail needs a byte move on \ + {triple}:\n{body}" + ); + } +} + +/// The values survive the paths above, with guards either side. +/// +/// Every size here is one the register-pair and all-SSE prologues classify +/// into two eightbytes whose second is short, and every object is fenced, so a +/// store that is wider than its object shows up as a clobbered guard. +#[test] +fn codegen_register_composite_parameters_keep_their_values() { + let code = r#" +struct B13 { unsigned char c[13]; }; +struct MX { int a, b; float c; }; +struct __attribute__((packed)) P6 { float a, b, c; _Float16 d; }; +struct __attribute__((packed)) G13 { long x; unsigned char c[5]; }; + +__attribute__((noinline)) int take_b13(struct B13 v) +{ int s = 0; for (int i = 0; i < 13; i++) s += v.c[i]; return s; } +__attribute__((noinline)) float take_mx(struct MX p) { return (float)(p.a + p.b) + p.c; } +__attribute__((noinline)) float take_p6(struct P6 p) { return p.a + p.b + p.c + (float)p.d; } +__attribute__((noinline)) long take_g13(struct G13 p) { return p.x + p.c[0] + p.c[4]; } + +int main(void) +{ + volatile unsigned char lo = 0xA5; + struct B13 b13; + struct MX mx = {1, 2, 4.f}; + struct P6 p6 = {1.f, 2.f, 4.f, (_Float16)8.f}; + struct G13 g13 = {7, {1, 2, 3, 4, 5}}; + volatile unsigned char hi = 0x5A; + + for (int i = 0; i < 13; i++) b13.c[i] = (unsigned char)(i + 1); + + if (take_b13(b13) != 91) return 1; + if (take_mx(mx) != 7.f) return 2; + if (take_p6(p6) != 15.f) return 3; + if (take_g13(g13) != 13) return 4; + if (lo != 0xA5 || hi != 0x5A) return 5; + return 0; +} +"#; + assert_eq!(compile_and_run("register_composite_params", code, &[]), 0); +} + +/// The optimized IR of `src` for `target`, with inlining left on. +fn post_opt_ir_inlined(prefix: &str, src: &str, target: &str, func: &str) -> String { + let dir = plib::tmp::Builder::new() + .prefix(prefix) + .tempdir() + .expect("tempdir"); + let c = dir.path().join("t.c"); + std::fs::write(&c, src).expect("write source"); + let r = run_c17(&[ + "--target", + target, + "-O2", + "--dump-ir", + "post-opt", + "--dump-ir-func", + func, + "-S", + "-o", + "/dev/null", + c.to_str().unwrap(), + ]); + assert!(r.success, "compile failed: {}", r.stderr); + format!("{}{}", r.stdout, r.stderr) +} + +/// An aggregate returned in registers is spliced into its caller as its value, +/// not as the address of the callee's copy. +/// +/// The inliner replaces a `Ret` with a phi of the returned value. For a +/// register-returned aggregate it has to read that value out of the callee's +/// result local first. It did for the two-register case and for a one-register +/// aggregate of eight bytes, but a *sixteen*-byte aggregate returned in one SSE +/// register -- `struct { __float128 a; }` -- had its `symaddr` fed straight +/// into the phi, so the caller received the address where the value belonged: +/// +/// leaq -96(%rbp), %rax ; the callee's result local +/// movq %r10, -64(%rbp) ; stored into eight bytes of a sixteen-byte slot +/// movq -56(%rbp), %rax ; the other eight read uninitialized +/// +/// Inlining therefore changed the answer. Compiling for an explicit target is +/// what makes this testable at all: `__float128` is rejected on Darwin, so the +/// shape cannot be built for the host, and no test covered it. +#[test] +fn codegen_an_inlined_register_aggregate_return_is_a_value() { + let src = "\ +struct Q { __float128 a; }; +static struct Q mk(__float128 x) { struct Q r = {x}; return r; } +__float128 probe(__float128 x) { struct Q v = mk(x); return v.a; } +"; + let ir = post_opt_ir_inlined("inl_sse_ret", src, X86_64_LINUX, "probe"); + + // Every pseudo that holds an address rather than a value. + let addresses: Vec<&str> = ir + .lines() + .filter_map(|l| { + let t = l.trim(); + let (target, rest) = t.split_once(" = ")?; + rest.starts_with("symaddr").then_some(target) + }) + .collect(); + + // A phi source carries the returned value, so none of them may be one. + for line in ir.lines().map(str::trim).filter(|l| l.contains("phisrc")) { + for addr in &addresses { + assert!( + !line + .split_whitespace() + .any(|w| w.trim_end_matches(',') == *addr), + "the inlined return hands the caller {addr}, which is an address, \ + where the aggregate's value belongs:\n {line}\n\nfull IR:\n{ir}" + ); + } + } +} + +/// The control: the shapes that already worked must keep working, so the check +/// above cannot pass by the inliner declining to inline. +#[test] +fn codegen_inlined_aggregate_returns_still_inline() { + let src = "\ +struct F2 { float a, b; }; +struct F4 { float a, b, c, d; }; +static struct F2 mk2(float x) { struct F2 r = {x, x + 1}; return r; } +static struct F4 mk4(float x) { struct F4 r = {x, x+1, x+2, x+3}; return r; } +float probe2(float x) { struct F2 v = mk2(x); return v.a + v.b; } +float probe4(float x) { struct F4 v = mk4(x); return v.a + v.d; } +"; + for func in ["probe2", "probe4"] { + let ir = post_opt_ir_inlined("inl_agg_ok", src, X86_64_LINUX, func); + assert!( + ir.contains("_inline"), + "{func}'s callee must still be inlined, or the check above is vacuous:\n{ir}" + ); + assert!( + ir.lines().any(|l| l.contains("load")), + "{func} must read the returned aggregate's value:\n{ir}" + ); + } +} diff --git a/cc/tests/codegen/inline_asm.rs b/cc/tests/codegen/inline_asm.rs index 8654fb99a..0d7c12df5 100644 --- a/cc/tests/codegen/inline_asm.rs +++ b/cc/tests/codegen/inline_asm.rs @@ -2685,3 +2685,54 @@ fn codegen_inline_asm_operand_address_used_elsewhere() { } } } + +/// An `asm goto` releases the VLA scopes its label edge leaves. +/// +/// It has two exits and needs a release on each. The enclosing scope's +/// release sits on the fall-through, and the label edge branches straight +/// past it -- so a loop whose back edge runs through the jump allocated +/// every time round and freed nothing. Each edge now has a block of its own +/// and the release goes there. +/// +/// Observed by address rather than by exhaustion: the same declaration +/// reached on the same path allocates at the same address every iteration if +/// and only if the previous one was released. +#[test] +#[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))] +fn codegen_asm_goto_releases_a_vla_scope_on_its_label_edge() { + #[cfg(target_arch = "x86_64")] + let jump = "jmp %l[again]"; + #[cfg(target_arch = "aarch64")] + let jump = "b %l[again]"; + + let code = format!( + r#" +static int loop_through_asm_goto(int n) {{ + void *first = 0; + int k = 0; +top: + {{ + int a[n]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 1; + k++; + __asm__ goto ("{jump}" : : : : again); + return 2; /* the asm always branches */ + }} +again: + if (k < 8) goto top; + return 0; +}} + +int main(void) {{ + return loop_through_asm_goto(7); +}} +"# + ); + assert_eq!(compile_and_run("asm_goto_vla_scope", &code, &[]), 0); + assert_eq!( + compile_and_run_optimized("asm_goto_vla_scope_opt", &code), + 0 + ); +} diff --git a/cc/tests/codegen/memopt.rs b/cc/tests/codegen/memopt.rs index 165a0efc7..f87a2e0c8 100644 --- a/cc/tests/codegen/memopt.rs +++ b/cc/tests/codegen/memopt.rs @@ -15,7 +15,10 @@ // if the pass forwards one byte it should not have. // -use crate::codegen::asm_probe::{asm_for_with, body_of, AARCH64_LINUX, X86_64_LINUX}; +use crate::codegen::asm_probe::{ + asm_for_with, assert_body_contains, assert_body_lacks, body_of, count_in_body, AARCH64_LINUX, + X86_64_LINUX, +}; use crate::common::{compile_and_run, compile_and_run_aarch64, compile_and_run_two_units, run_c17}; fn at_o2(name: &str, code: &str) -> i32 { @@ -1264,3 +1267,418 @@ fn codegen_no_composite_is_read_wider_than_itself() { } } } + +/// Reading a `volatile` object is an observable side effect, so the access +/// survives every optimization level -- including a read whose value is +/// discarded, which no data-flow fact keeps alive (C17 5.1.2.3). +/// +/// The property is on the access, not on the result, so a discarded read has +/// nothing an exit status can see. The check is on the emitted instruction, +/// against both targets, because the rule is architecture-independent. +#[test] +fn memopt_a_discarded_volatile_read_is_still_performed() { + // The object names are deliberately unmistakable. A single letter is not a + // sound needle here: every x86-64 body contains `pushq`/`popq` and every + // aarch64 body contains `stp`/`sp`, and `.cfi_startproc` is inside the + // range `body_of` returns -- so searching for "p" passes against a body + // that was emptied, which is exactly the defect. (No empty body on either + // target contains a "g", which is why the other cases were sound.) + let cases = [ + ( + "assign", + "volatile int volobj;\nvoid probe(void) { int a = volobj; (void)a; }\n", + "volobj", + ), + ( + "discard", + "volatile int volobj;\nvoid probe(void) { volobj; }\n", + "volobj", + ), + ( + "via_ptr", + "volatile int *volptr;\nvoid probe(void) { *volptr; }\n", + "volptr", + ), + ( + "cast_void", + "volatile int volobj;\nvoid probe(void) { (void)volobj; }\n", + "volobj", + ), + ]; + + for (tag, src, object) in cases { + for level in ["-O0", "-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("vol_{tag}"), triple, src, &[level]); + assert_body_contains( + &asm, + "probe", + object, + &format!( + "a volatile read is observable: `{tag}` at {level} on {triple} \ + must still access `{object}`" + ), + ); + + // Naming the pointer is not the same as dereferencing it, and + // the qualifier here is on the pointee, so the load through it + // is the access under test. + if tag == "via_ptr" { + let indirect = if triple == X86_64_LINUX { "(%r" } else { "[x" }; + assert_body_contains( + &asm, + "probe", + indirect, + &format!( + "the volatile pointee is read, not just the pointer: \ + {level} on {triple}" + ), + ); + } + } + } + } +} + +/// The counterpart that keeps the fix above honest: an *ordinary* discarded +/// read is still dead code, and DCE still deletes it. +/// +/// Without this, marking every load a root would pass the volatile test. +#[test] +fn memopt_a_discarded_plain_read_is_still_removed() { + // Object names chosen to occur in no mnemonic, register or label the + // body can otherwise contain -- `popq` alone contains both `p` and `pq`, + // and the body a negative assertion searches includes the function's own + // label and prologue. + let cases = [ + ( + "assign", + "int objx;\nvoid probe(void) { int a = objx; (void)a; }\n", + "objx", + ), + ("discard", "int objx;\nvoid probe(void) { objx; }\n", "objx"), + ( + "via_ptr", + "int *ptrx;\nvoid probe(void) { *ptrx; }\n", + "ptrx", + ), + ]; + + for (tag, src, object) in cases { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("plain_{tag}"), triple, src, &["-O2"]); + assert_body_lacks( + &asm, + "probe", + object, + &format!( + "reading a non-volatile object has no effect: `{tag}` on {triple} \ + must not access `{object}`" + ), + ); + } + } +} + +/// Each read of a `volatile` object is its own observable event, so two of +/// them are two accesses -- neither load-forwarding nor DCE may fold the pair +/// into one. +#[test] +fn memopt_two_volatile_reads_are_both_performed() { + // A named object only: two reads through one `volatile int *p` show up as + // a *single* reference to `p` -- reading the pointer itself is not + // volatile and is rightly done once -- so the count says nothing there. + // The through-pointer case is pinned at the IR level instead, by + // `test_volatile_accesses_carry_the_marker` and the `dce` unit tests. + // + // The name occurs in no mnemonic, register or label the body can + // otherwise contain: `popq` alone contains both `p` and `pq`. + let cases = [( + "named", + "volatile int objx;\nint sink(int, int);\n\ + int probe(void) { int a = objx; int b = objx; return sink(a, b); }\n", + "objx", + )]; + + for (tag, src, object) in cases { + for level in ["-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("vol_two_reads_{tag}"), triple, src, &[level]); + let n = count_in_body(&asm, "probe", object); + assert!( + n >= 2, + "both volatile reads are observable: `{tag}` at {level} on {triple} \ + kept {n} reference(s) to `{object}`:\n{}", + body_of(&asm, "probe") + ); + } + } + } +} + +/// An `_Atomic` read is observable for the same reason, and reaches DCE by a +/// different route: `AtomicLoad` is a side-effecting opcode outright, so this +/// cross-checks that the two spellings of "this read must happen" agree. +#[test] +fn memopt_a_discarded_atomic_read_is_still_performed() { + let cases = [ + ("discard", "_Atomic int g;\nvoid probe(void) { g; }\n", "g"), + ( + "assign", + "_Atomic int g;\nvoid probe(void) { int a = g; (void)a; }\n", + "g", + ), + ]; + + for (tag, src, object) in cases { + for level in ["-O0", "-O2"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("atomic_{tag}"), triple, src, &[level]); + assert_body_contains( + &asm, + "probe", + object, + &format!( + "an atomic read is observable: `{tag}` at {level} on {triple} \ + must still access `{object}`" + ), + ); + } + } + } +} + +/// A `volatile` read inside a loop happens once per iteration: the value is +/// not a loop invariant, whatever the compiler can see written to the object. +/// +/// Nothing in c17 hoists memory out of a loop today (see the ordering contract +/// in `cc/ir/dce.rs`), so this passes by construction -- it exists to fail the +/// day something does, because the exit status is where that would show up. +#[test] +fn memopt_a_volatile_read_in_a_loop_is_repeated() { + // `g` changes between iterations through a pointer the loop writes, so a + // read hoisted to the top would sum 7 three times instead of 7 + 10 + 11. + let code = r#" +volatile int g; +static int *alias(void) { return (int *)&g; } +int main(void) { + int sum = 0; + *alias() = 7; + for (int i = 0; i < 3; i++) { + sum += g; + *alias() = 10 + i; + } + return sum == 28 ? 0 : 1; +} +"#; + assert_eq!(at_o2("memopt_volatile_in_loop", code), 0); + if let Some(rc) = compile_and_run_aarch64("memopt_volatile_in_loop_a64", code, "-O2") { + assert_eq!(rc, 0, "aarch64 at -O2"); + } +} + +/// A `volatile` store is observable for the same reason, and DSE must not drop +/// the earlier of two writes to one. +/// +/// The companion to the read case above: a test that only checked loads would +/// pass against an `has_side_effects` that named `Store` and not `Load`. +#[test] +fn memopt_two_volatile_stores_are_both_performed() { + let src = "volatile int g;\nvoid probe(void) { g = 1; g = 2; }\n"; + for level in ["-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_two_stores", triple, src, &[level]); + let n = count_in_body(&asm, "probe", "g"); + assert!( + n >= 2, + "both volatile stores are observable: {level} on {triple} kept {n} \ + reference(s) to `g`:\n{}", + body_of(&asm, "probe") + ); + } + } +} + +/// A member of a `volatile` object is itself volatile, so reading it is an +/// observable event that survives every optimization level. +/// +/// C17 6.5.2.3p3/p4: the result of `s.m` has the *so-qualified* version of the +/// member's type — it inherits the qualifiers of the object. c17 took the +/// member's declared type unchanged, so a member of a `volatile` struct read as +/// an ordinary `int` and DCE deleted it from `-O1` up. The reverse direction +/// (`struct T { volatile int a; }`) always worked, because there the member's +/// own type carries the qualifier; that case is the control below. +#[test] +fn memopt_a_member_of_a_volatile_object_is_volatile() { + // Distinctive names: a single letter matches `pushq`/`stp`/`.cfi_startproc` + // inside the body range and would pass against an emptied function. + let src = "\ +struct S { int a; int b; }; +volatile struct S vqobj; +volatile struct S *vqptr; +void probe_direct(void) { vqobj.a; } +void probe_arrow(void) { vqptr->a; } +void probe_assign(void) { int t = vqobj.a; (void)t; } +void probe_second(void) { vqobj.b; } +"; + for level in ["-O0", "-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_member", triple, src, &[level]); + for (func, object) in [ + ("probe_direct", "vqobj"), + ("probe_arrow", "vqptr"), + ("probe_assign", "vqobj"), + ("probe_second", "vqobj"), + ] { + assert_body_contains( + &asm, + func, + object, + &format!( + "a member of a volatile object is volatile (C17 6.5.2.3p3): \ + {func} at {level} on {triple} must still access `{object}`" + ), + ); + } + } + } +} + +/// The control for the test above: an ordinary aggregate's member read is still +/// deleted, so that test cannot pass by marking every member access volatile. +#[test] +fn memopt_a_member_of_a_plain_object_is_still_removed() { + let src = "\ +struct S { int a; }; +struct S pqobj; +void probe(void) { pqobj.a; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("plain_member", triple, src, &["-O2"]); + assert_body_lacks( + &asm, + "probe", + "pqobj", + "reading an ordinary member has no effect and is dead code", + ); + } +} + +/// A `volatile` member is not speculatable, so a conditional must not read the +/// arm it did not take. +/// +/// C17 6.5.15p4 evaluates only one of the second and third operands, and +/// 5.1.2.3 makes each volatile read an observable event. `is_pure_expr`'s +/// `Member` arm asked only whether the *base* was pure, so the read was +/// hoisted and both members were loaded unconditionally into a branchless +/// select — at `-O0` too. +#[test] +fn memopt_a_volatile_member_is_not_speculated_by_a_conditional() { + let src = "\ +struct S { volatile unsigned status; unsigned other; }; +struct S sqobj; +unsigned probe(int c) { return c ? sqobj.status : sqobj.other; } +"; + for level in ["-O0", "-O2"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_member_select", triple, src, &[level]); + let select = if triple == X86_64_LINUX { + "cmov" + } else { + "csel" + }; + assert_body_lacks( + &asm, + "probe", + select, + &format!( + "a volatile member read cannot be speculated, so the arms may not \ + collapse into a conditional move: {level} on {triple}" + ), + ); + } + } +} + +/// The control for the test above: with no volatile member, the branchless +/// select is still allowed, so that test is asserting the qualifier and not +/// merely that c17 stopped emitting conditional moves. +#[test] +fn memopt_a_plain_member_may_still_be_speculated() { + let src = "\ +struct S { unsigned one; unsigned other; }; +struct S pqsel; +unsigned probe(int c) { return c ? pqsel.one : pqsel.other; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("plain_member_select", triple, src, &["-O2"]); + let select = if triple == X86_64_LINUX { + "cmov" + } else { + "csel" + }; + assert_body_contains( + &asm, + "probe", + select, + "two ordinary member reads are pure and may still collapse to a select", + ); + } +} + +/// A `volatile` bit-field read is observable, even though the access is of the +/// carrier and the carrier can never carry the qualifier. +/// +/// The bit-field emitters build their load and store at +/// `bitfield_storage_type`, which is the unqualified storage unit, so nothing +/// derived the marker from the access type. `mark_volatile_access` anticipates +/// exactly this ("a bit-field reads a storage unit whose type is the carrier") +/// and preserves a marker the site sets itself — neither emitter set one, and +/// the read was deleted outright from `-O1` up. Both spellings are covered: the +/// field declared `volatile`, and an ordinary field of a `volatile` object. +#[test] +fn memopt_a_volatile_bitfield_read_is_performed() { + let src = "\ +struct B { volatile unsigned f : 3; unsigned g : 5; }; +struct B bfqobj; +volatile struct B vbfqobj; +void probe_field(void) { bfqobj.f; } +void probe_object(void) { vbfqobj.g; } +"; + for level in ["-O0", "-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_bitfield", triple, src, &[level]); + for (func, object) in [("probe_field", "bfqobj"), ("probe_object", "vbfqobj")] { + assert_body_contains( + &asm, + func, + object, + &format!( + "a volatile bit-field read is observable: {func} at {level} \ + on {triple} must still access `{object}`" + ), + ); + } + } + } +} + +/// The control: an ordinary bit-field read is still dead code, so the test +/// above cannot pass by marking every bit-field access volatile. +#[test] +fn memopt_a_plain_bitfield_read_is_still_removed() { + let src = "\ +struct B { unsigned f : 3; }; +struct B pbfqobj; +void probe(void) { pbfqobj.f; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("plain_bitfield", triple, src, &["-O2"]); + assert_body_lacks( + &asm, + "probe", + "pbfqobj", + "reading an ordinary bit-field has no effect and is dead code", + ); + } +} diff --git a/cc/tests/codegen/misc.rs b/cc/tests/codegen/misc.rs index 1c3aa387d..093d22d04 100644 --- a/cc/tests/codegen/misc.rs +++ b/cc/tests/codegen/misc.rs @@ -16438,3 +16438,288 @@ int main(void) { ); } } + +/// Zero-initializing an aggregate, across the size at which unrolling stops. +/// +/// The companion to `codegen_struct_copy_across_the_inline_threshold`, for the +/// other half of the same family. `emit_aggregate_zero` hand-rolled the same +/// 8/4/2/1 descent that `memexpand::block_chunks` already produces, but with +/// **no upper bound** — so `char buf[N] = {0}` emitted one store per chunk for +/// any N. Measured before the fix: 8 KB cost 2081 instructions in the function +/// body and 1 MB did not finish compiling in 25 minutes, while its sibling +/// `emit_block_copy_at_offset` had capped at `INLINE_LIMIT_BYTES` all along. +/// +/// The declaration is inside a loop on purpose. On entry the backend zeroes the +/// whole frame, which masks a missing zero-fill the first time through; only +/// re-execution shows it. +/// +/// Sizes straddle 128 and none is a multiple of 8, so a rounded-up or +/// short-by-a-tail fill shows as a wrong byte rather than passing by luck. +#[test] +fn codegen_aggregate_zero_across_the_inline_threshold() { + let code = r#" +void sink(char *p); + +#define MKZ(N) \ + static int zero##N(void) { \ + for (int pass = 0; pass < 2; pass++) { \ + unsigned char lo = 0xA5; \ + char buf[N] = {0}; \ + unsigned char hi = 0x5A; \ + for (int i = 0; i < N; i++) \ + if (buf[i] != 0) return 1; \ + if (lo != 0xA5 || hi != 0x5A) return 2; \ + for (int i = 0; i < N; i++) buf[i] = (char)(i + 1); \ + sink(buf); \ + } \ + return 0; \ + } + +MKZ(7) +MKZ(12) +MKZ(13) +MKZ(127) +MKZ(129) +MKZ(200) +MKZ(1000) + +void sink(char *p) { (void)p; } + +int main(void) +{ + if (zero7()) return 1; + if (zero12()) return 2; + if (zero13()) return 3; + if (zero127()) return 4; + if (zero129()) return 5; + if (zero200()) return 6; + if (zero1000()) return 7; + return 0; +} +"#; + assert_eq!(compile_and_run("aggregate_zero_threshold", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("aggregate_zero_threshold_opt", code), + 0 + ); +} + +/// The bound itself, not just the answer: above the threshold the zero-fill is +/// a `memset` call, below it is still stores. +/// +/// The behavioural test above passes either way — a million unrolled stores +/// produce a correctly zeroed object, just not in a time anyone will wait for. +/// This is the test that the *bound* exists, and the negative half keeps it +/// from being satisfied by calling `memset` for every size, which would cost +/// more than the stores it replaced for a small object. +#[test] +fn codegen_a_large_aggregate_zero_is_a_memset_call() { + use crate::codegen::asm_probe::{asm_for_with, AARCH64_LINUX, X86_64_LINUX}; + + let src = |n: usize| { + format!("void sink(char *);\nvoid probe(void) {{ char buf[{n}] = {{0}}; sink(buf); }}\n") + }; + + for triple in [X86_64_LINUX, AARCH64_LINUX] { + // Comfortably over `INLINE_LIMIT_BYTES` (128). + let big = asm_for_with("aggzero_big", triple, &src(4096), &["-O2"]); + assert!( + big.contains("memset"), + "a 4096-byte zero-fill belongs in a memset call, not 512 stores, on {triple}:\n{big}" + ); + + // And the unrolled form is still used where it is cheaper than a call. + let small = asm_for_with("aggzero_small", triple, &src(16), &["-O2"]); + assert!( + !small.contains("memset"), + "a 16-byte zero-fill is cheaper unrolled than called, on {triple}:\n{small}" + ); + } +} + +/// The inliner moves an implicit parameter copy in whole eight-byte chunks, +/// which reads and writes past a object whose size is not a multiple of eight. +/// +/// The third copy of the unrolled block move, after the two named in +/// `codegen_struct_copy_across_the_inline_threshold`. `while offset < size_bytes +/// { load 64; store 64; offset += 8 }` rounds *up*: a 12-byte `struct P` moved +/// 16 bytes, over-reading the argument and over-writing the callee's local. +/// `memexpand::block_chunks` descends 8/4/2/1 and is what the already-fixed +/// twin in the linearizer uses. +/// +/// Stated on the IR because the overrun is layout-dependent: the four extra +/// bytes usually land in frame padding, so a program can be correct and still +/// be reading memory that does not belong to the object — and would fault if it +/// ended a page. +#[test] +fn codegen_an_inlined_parameter_copy_moves_no_more_than_the_object() { + let src = r#" +struct P { float x, y, z; }; +static float sum(struct P p) { return p.x + p.y + p.z; } +float probe(void) { struct P q = {1, 2, 3}; return sum(q); } +"#; + let dir = plib::tmp::Builder::new() + .prefix("inline_param_copy") + .tempdir() + .expect("tempdir"); + let c = dir.path().join("t.c"); + std::fs::write(&c, src).expect("write source"); + let r = crate::common::run_c17(&[ + "-O2", + "--dump-ir", + "post-opt", + "--dump-ir-func", + "probe", + "-S", + "-o", + "/dev/null", + c.to_str().unwrap(), + ]); + assert!(r.success, "compile failed: {}", r.stderr); + let ir = format!("{}{}", r.stdout, r.stderr); + + // A 12-byte object has four bytes at offset 8, so a 64-bit access there is + // four bytes past the end -- on the load side and again on the store side. + // A correct copy reaches it with a 32-bit access. + let overruns: Vec<&str> = ir + .lines() + .map(str::trim) + .filter(|l| l.contains("+ 8") && (l.contains("load.64") || l.contains("store.64"))) + .collect(); + assert!( + overruns.is_empty(), + "a 12-byte object has 4 bytes at offset 8, so these access 4 past it:\n {}\n\nfull IR:\n{ir}", + overruns.join("\n ") + ); + + // The control: the copy has to still be there. If the inliner stopped + // inlining, or the parameter stopped being copied, the check above would + // pass while testing nothing. + assert!( + ir.contains("+ 8"), + "expected the inlined parameter copy to reach offset 8 at all:\n{ir}" + ); +} + +/// Storing one field of an eight-byte aggregate leaves the other alone. +/// +/// The x86-64 store lowering widens a 32-bit store at offset 0 of a local to +/// 64 bits, to clear stale upper bits when a narrow value goes into a wider +/// slot. Its own comment records the exception that needs: "struct/union +/// fields at offset 0 must use exact size to avoid clobbering the adjacent +/// field at offset 4". The exception asked whether the object was *larger than* +/// 64 bits, which an eight-byte aggregate is not -- so exactly the case the +/// comment describes was the one that fell through. +/// +/// Only at `-O0`: with the optimizer on, the field is promoted out of memory +/// before the store lowering sees it. +#[test] +fn codegen_a_field_store_does_not_widen_over_its_neighbour() { + let code = r#" +struct P { int x, y; }; +struct S { struct P t; }; +union U { struct P p; double d; }; + +int main(void) +{ + /* The reported shape: a designated override inside an eight-byte struct. */ + struct S a = { .t = {1, 2}, .t.x = 3 }; + if (a.t.x != 3 || a.t.y != 2) return 1; + + /* The same store reached other ways. */ + struct P b = {1, 2}; + b.x = 3; + if (b.x != 3 || b.y != 2) return 2; + + struct P c; + c.y = 2; + c.x = 3; + if (c.x != 3 || c.y != 2) return 3; + + struct P *p = &b; + p->x = 9; + if (b.x != 9 || b.y != 2) return 4; + + union U u; + u.p.y = 7; + u.p.x = 5; + if (u.p.x != 5 || u.p.y != 7) return 5; + + /* Arrays are the same shape at the same size. */ + int arr[2] = {1, 2}; + arr[0] = 3; + if (arr[0] != 3 || arr[1] != 2) return 6; + + /* Exactly eight bytes made of narrower fields. */ + struct Q { short a, b, c, d; } q = {1, 2, 3, 4}; + q.a = 9; + if (q.a != 9 || q.b != 2 || q.c != 3 || q.d != 4) return 7; + + return 0; +} +"#; + // `-O0` explicitly: the default matrix compiles at `-O`, where the field is + // promoted out of memory before the store lowering ever sees it, so the + // defect is invisible there. + assert_eq!( + compile_and_run("field_store_no_widen", code, &["-O0".to_string()]), + 0 + ); + assert_eq!(compile_and_run("field_store_no_widen_matrix", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("field_store_no_widen_opt", code), + 0 + ); +} + +/// The control: a narrow value stored into a wider scalar slot still leaves no +/// stale upper bits. +/// +/// This is what the widening is for, and it is why the fix has to ask whether +/// the object is an aggregate rather than simply stop widening. Each case +/// writes a wide value into the slot first, so a store that failed to clear the +/// upper half would read it back. +#[test] +fn codegen_a_narrow_store_into_a_wide_slot_clears_it() { + let code = r#" +int wide(void) { return -1; } + +int main(void) +{ + /* Put a known wide pattern in the slot, then overwrite it narrowly. */ + long l = 0x7fffffff7fffffffL; + int i = 5; + l = i; + if (l != 5) return 1; + + unsigned long ul = 0xffffffffffffffffUL; + unsigned ui = 7; + ul = ui; + if (ul != 7UL) return 2; + + void *vp = (void *)0x7fffffffffffL; + unsigned addr = 0; + vp = (void *)(unsigned long)addr; + if (vp != (void *)0) return 3; + + /* Through a call, so the value is not a constant the optimizer can see. */ + long l2 = 0x7fffffff7fffffffL; + l2 = wide(); + if (l2 != -1L) return 4; + + return 0; +} +"#; + assert_eq!( + compile_and_run("narrow_store_clears_slot", code, &["-O0".to_string()]), + 0 + ); + assert_eq!( + compile_and_run("narrow_store_clears_slot_matrix", code, &[]), + 0 + ); + assert_eq!( + compile_and_run_optimized("narrow_store_clears_slot_opt", code), + 0 + ); +} diff --git a/cc/tests/codegen/mod.rs b/cc/tests/codegen/mod.rs index 79f628fe5..e727a6169 100644 --- a/cc/tests/codegen/mod.rs +++ b/cc/tests/codegen/mod.rs @@ -16,6 +16,7 @@ mod aarch64_runtime; pub mod asm_probe; mod atomics_asm; mod binary128; +mod block_moves; mod complex_fold; mod constant_branch; mod cross_abi; @@ -31,3 +32,4 @@ mod scaling; mod sections; mod stacked_args; mod tls_models; +mod trapping_folds; diff --git a/cc/tests/codegen/trapping_folds.rs b/cc/tests/codegen/trapping_folds.rs new file mode 100644 index 000000000..d4f0b700e --- /dev/null +++ b/cc/tests/codegen/trapping_folds.rs @@ -0,0 +1,134 @@ +// +// Copyright (c) 2025-2026 Jeff Garzik +// +// This file is part of the posixutils-rs project covered under +// the MIT License. For the full license text, please see the LICENSE +// file in the root directory of this project. +// SPDX-License-Identifier: MIT +// +// Operations that trap are not folded away. +// +// `cc/ir/range.rs` states the rule this file guards: c17 does not assume +// undefined behaviour away, because doing so makes `-O0` and `-O2` disagree +// about a program that really does divide by zero. `eval_divmod` already +// refuses a zero divisor for exactly that reason -- and then folded the other +// trap the same instruction raises. +// +// On x86-64 `idiv` raises #DE for a zero divisor *and* for the one signed +// overflow, `INT_MIN / -1`, whose quotient is not representable; the remainder +// form traps identically, because it is the same instruction. So folding +// `INT_MIN / -1` to `INT_MIN`, or `0 / x` to `0` without knowing `x`, turns a +// program that faults at `-O0` into one that prints an answer at `-O2`. +// +// gcc and clang both fold these -- they treat the undefined behaviour as +// licence. c17's stated policy is the opposite, so these tests are written +// against the policy rather than against another compiler. +// +// They assert on the emitted instruction rather than on a fault: the point is +// that the division survives to run, and a test that asserts a crash is worse +// at saying so. +// + +use crate::codegen::asm_probe::{asm_for_with, body_of, AARCH64_LINUX, X86_64_LINUX}; +use crate::common::compile_and_run; + +/// How many divide instructions the body of `func` contains. +fn divisions(asm: &str, func: &str) -> usize { + body_of(asm, func) + .lines() + .filter(|l| { + let t = l.trim(); + t.starts_with("idiv") + || t.starts_with("div") + || t.starts_with("sdiv") + || t.starts_with("udiv") + }) + .count() +} + +/// The two traps `idiv` raises are both left alone. +#[test] +fn codegen_a_trapping_division_is_not_folded() { + let src = "\ +int ovf_div(void) { int a = -2147483647 - 1, b = -1; return a / b; } +int ovf_mod(void) { int a = -2147483647 - 1, b = -1; return a % b; } +long ovf_div_long(void) { long a = -9223372036854775807L - 1, b = -1; return a / b; } +int zero_div(int x) { return 0 / x; } +int zero_mod(int x) { return 0 % x; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("trap_fold", triple, src, &["-O2"]); + for func in ["ovf_div", "ovf_mod", "ovf_div_long", "zero_div", "zero_mod"] { + assert!( + divisions(&asm, func) >= 1, + "{func} traps at -O0 and must still divide at -O2 on {triple}, \ + not be folded to an answer:\n{}", + body_of(&asm, func) + ); + } + } +} + +/// The control: a division that cannot trap is still folded. +/// +/// Without this the test above is satisfied by never folding a division at +/// all, which would cost every program that divides by a constant. +/// +/// `x / 8` is deliberately not here: with `x` unknown there is nothing to fold, +/// and c17 has no divide-by-constant strength reduction, so it emits a division +/// for reasons that have nothing to do with trapping. +#[test] +fn codegen_a_safe_division_is_still_folded() { + let src = "\ +int by_one(int x) { return x / 1; } +int mod_one(int x) { return x % 1; } +int both_known(void) { return 12 / 4; } +int both_known_mod(void) { return 13 % 4; } +int neg_but_safe(void) { int a = -2147483647, b = -1; return a / b; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("safe_fold", triple, src, &["-O2"]); + for func in [ + "by_one", + "mod_one", + "both_known", + "both_known_mod", + "neg_but_safe", + ] { + assert_eq!( + divisions(&asm, func), + 0, + "{func} cannot trap, so it is still folded on {triple}:\n{}", + body_of(&asm, func) + ); + } + } +} + +/// The answers the folder does give are unchanged. +#[test] +fn codegen_constant_division_still_computes_the_right_answer() { + let code = r#" +int main(void) +{ + if (12 / 4 != 3) return 1; + if (13 % 4 != 1) return 2; + if (-13 / 4 != -3) return 3; + if (-13 % 4 != -1) return 4; + if (13 / -4 != -3) return 5; + if (13 % -4 != 1) return 6; + + /* The largest magnitudes that do not overflow. */ + if ((-2147483647 - 1) / 1 != -2147483647 - 1) return 7; + if (-2147483647 / -1 != 2147483647) return 8; + if ((-2147483647 - 1) % -1 != 0 && 0) return 9; + + unsigned u = 4294967295u; + if (u / 5u != 858993459u) return 10; + if (u % 5u != 0u) return 11; + + return 0; +} +"#; + assert_eq!(compile_and_run("const_division_answers", code, &[]), 0); +} diff --git a/cc/tests/diagnostics/mod.rs b/cc/tests/diagnostics/mod.rs index 8ae50ad8a..25dff57a5 100644 --- a/cc/tests/diagnostics/mod.rs +++ b/cc/tests/diagnostics/mod.rs @@ -7347,3 +7347,139 @@ fn diagnostics_escape_out_of_range_names_the_literals_line() { ); } } + +/// A member of a `const` object is itself `const`, so writing it is a +/// constraint violation. +/// +/// C17 6.5.2.3p3/p4 gives `s.m` the *so-qualified* version of the member's +/// type, and 6.5.16p2 requires a modifiable lvalue on the left of an +/// assignment. c17 took the member's declared type unqualified, so +/// `check_const_assignment` saw an ordinary `int` and every one of these was +/// accepted silently. gcc and clang reject them. +#[test] +fn diagnostics_writing_a_member_of_a_const_object_is_rejected() { + compile_expect_error( + "const_aggregate_member", + "struct S { int a; };\nconst struct S cs = {1};\nvoid f(void){ cs.a = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_member_arrow", + "struct S { int a; };\nvoid f(const struct S *p){ p->a = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_member_nested", + "struct T { int x; };\nstruct S { struct T t; };\n\ + const struct S cs;\nvoid f(void){ cs.t.x = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_member_increment", + "struct S { int a; };\nconst struct S cs;\nvoid f(void){ cs.a++; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_element", + "struct S { int a; };\nconst struct S cs[2];\nvoid f(void){ cs[1].a = 2; }\n", + "read-only", + ); +} + +/// The other direction: the same shapes without the `const` must still compile, +/// so the checks above cannot pass by rejecting every member assignment. +#[test] +fn diagnostics_writing_a_member_of_a_plain_object_is_accepted() { + compile_expect_ok( + "plain_aggregate_member", + "struct S { int a; };\nstruct S s;\nvoid f(void){ s.a = 2; }\n\ + void g(struct S *p){ p->a = 2; }\n\ + struct T { struct S in; };\nstruct T t;\nvoid h(void){ t.in.a = 2; }\n\ + struct S arr[2];\nvoid i(void){ arr[1].a = 2; }\n\ + void j(void){ s.a++; }\n", + ); + // A `const` *pointer* to a non-const object leaves the pointee writable: + // the qualifier is on the pointer, not on what it points at. + compile_expect_ok( + "const_pointer_not_pointee", + "struct S { int a; };\nvoid f(struct S *const p){ p->a = 2; }\n", + ); +} + +/// An *array* member of a `const` object is an array of `const`, so writing an +/// element of it is a constraint violation too. +/// +/// C17 6.7.3p10: where an array type is qualified, the element type is +/// so-qualified and the array is not -- which is what a subscript reads. So the +/// so-qualified version of `int [4]` is an array of `const int`, and +/// `cs.arr[0] = 1` is a write to a `const int`. Qualifying the array itself +/// instead would leave the element an ordinary `int` and accept the write. +#[test] +fn diagnostics_writing_an_array_member_of_a_const_object_is_rejected() { + compile_expect_error( + "const_aggregate_array_member", + "struct S { int arr[4]; };\nconst struct S cs;\nvoid f(void){ cs.arr[0] = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_array_member_2d", + "struct S { int grid[2][2]; };\nvoid f(const struct S *p){ p->grid[1][1] = 2; }\n", + "read-only", + ); + // The control: the same writes through an unqualified object. + compile_expect_ok( + "plain_aggregate_array_member", + "struct S { int arr[4]; int grid[2][2]; };\nstruct S s;\n\ + void f(void){ s.arr[0] = 2; s.grid[1][1] = 3; }\n\ + void g(struct S *p){ p->arr[0] = 2; }\n", + ); +} + +/// A `case` label whose conversion to the controlling type changes its value +/// is diagnosed, and two labels that become equal are a constraint violation. +/// +/// C17 6.8.4.2p5 converts each label to the promoted type of the controlling +/// expression, and p3 forbids two labels in one switch having the same value +/// *after* that conversion. c17 kept labels at full width, so it diagnosed +/// neither: `case 4294967296LL` in an `int` switch silently became `case 0`, +/// and sitting beside a real `case 0` it was silently accepted. +/// +/// gcc and clang both warn on the value-changing conversion and reject the +/// collision. +#[test] +fn diagnostics_a_case_label_outside_the_controlling_type_is_diagnosed() { + compile_expect_warning( + "case_label_overflow", + "int f(int x){ switch(x){ case 4294967296LL: return 1; default: return 2; } }\n", + "case", + ); + compile_expect_error( + "case_label_duplicate_after_conversion", + "int f(int x){ switch(x){ case 0: return 1; case 4294967296LL: return 2; } return 0; }\n", + "duplicate", + ); +} + +/// The other direction: a conversion that preserves the value is silent, so +/// the check above cannot pass by warning about every label. +/// +/// `case -1` in a `switch` on `unsigned` converts to 4294967295 and genuinely +/// matches it -- the conversion is value-changing in representation but well +/// defined and intended, which is why gcc and clang say nothing here either. +#[test] +fn diagnostics_a_case_label_inside_the_controlling_type_is_silent() { + compile_expect_no_diagnostic( + "case_label_in_range", + "int f(int x){ switch(x){ case -1: return 1; case 7: return 3; default: return 2; } }\n", + "case", + ); + compile_expect_no_diagnostic( + "case_label_negative_in_unsigned", + "int f(unsigned x){ switch(x){ case -1: return 1; default: return 2; } }\n", + "case", + ); + compile_expect_ok( + "case_labels_distinct_after_conversion", + "int f(int x){ switch(x){ case 0: return 1; case 1: return 2; default: return 3; } }\n", + ); +} diff --git a/cc/tests/misc/mod.rs b/cc/tests/misc/mod.rs index 68ce708c0..ee82ffcae 100644 --- a/cc/tests/misc/mod.rs +++ b/cc/tests/misc/mod.rs @@ -11,5 +11,6 @@ // Tests for setjmp/longjmp, statement expressions, and other misc features. // +mod no_current_block; mod setjmp; mod stmt_expr; diff --git a/cc/tests/misc/no_current_block.rs b/cc/tests/misc/no_current_block.rs new file mode 100644 index 000000000..27c84168f --- /dev/null +++ b/cc/tests/misc/no_current_block.rs @@ -0,0 +1,145 @@ +// +// Copyright (c) 2025-2026 Jeff Garzik +// +// This file is part of the posixutils-rs project covered under +// the MIT License. For the full license text, please see the LICENSE +// file in the root directory of this project. +// SPDX-License-Identifier: MIT +// +// Valid C for which the linearizer holds no current basic block. +// +// `Linearizer::current_bb` is legitimately `None` in two situations the +// language allows: after a `goto`, and inside a `switch` body before the first +// `case` label. Code in either place is unreachable but well-formed, and the +// standard requires it to be translated, not rejected -- and certainly not to +// crash the compiler. +// +// Every lowering that ends a block has to read `current_bb` back afterwards. +// The ones that reached for `.unwrap()` instead turned each of these programs +// into an internal compiler error. `current_or_unreachable_bb()` is the +// accessor that starts a fresh unreachable block rather than panicking. +// +// These are compile-only where the construct is genuinely unreachable, because +// there is no answer to assert; the two that are reachable check the answer as +// well. +// + +use crate::common::{compile_and_run, compile_expect_ok}; + +/// A statement before a `switch`'s first `case` has no enclosing block. +/// +/// Each of these ended in a panic, one per lowering: the ternary, the two +/// short-circuit operators, the complex ternary, both spellings of the GNU +/// elvis operator, complex integer division, and the atomic CAS loop. +#[test] +fn misc_a_statement_before_the_first_case_does_not_ice() { + for (tag, body) in [ + ("ternary", "g() ? g() : g();"), + ("logical_and", "(void)(g() && g());"), + ("logical_or", "(void)(g() || g());"), + ("elvis", "(void)(g() ?: g());"), + ("nested_ternary", "(void)(g() ? (g() ? g() : g()) : g());"), + ("and_in_ternary", "(void)(g() ? (g() && g()) : g());"), + ] { + compile_expect_ok( + &format!("nocur_switch_{tag}"), + &format!( + "int g(void);\n\ + int f(int x) {{ switch (x) {{ {body} case 1: return 1; }} return 0; }}\n" + ), + ); + } + + compile_expect_ok( + "nocur_switch_complex_ternary", + "int g2(void);\n_Complex double h(void);\n\ + int f(int x) { switch (x) { (void)(g2() ? h() : h()); case 1: return 1; } return 0; }\n", + ); + compile_expect_ok( + "nocur_switch_complex_elvis", + "_Complex double h(void);\n\ + int f(int x) { switch (x) { (void)(h() ?: h()); case 1: return 1; } return 0; }\n", + ); + compile_expect_ok( + "nocur_switch_complex_int_div", + "_Complex int ci(void);\n\ + int f(int x) { switch (x) { ci() / ci(); case 1: return 1; } return 0; }\n", + ); + compile_expect_ok( + "nocur_switch_atomic_nand", + "_Atomic int a;\n\ + int f(int x) { switch (x) { __atomic_fetch_nand(&a, 1, 5); case 1: return 1; } \ + return 0; }\n", + ); +} + +/// The same shapes after a `goto`, which is the other way to have no block. +#[test] +fn misc_a_statement_after_a_goto_does_not_ice() { + for (tag, body) in [ + ("ternary", "x = g() ? g() : g();"), + ("logical_and", "x = g() && g();"), + ("logical_or", "x = g() || g();"), + ("elvis", "x = g() ?: g();"), + ] { + compile_expect_ok( + &format!("nocur_goto_{tag}"), + &format!("int g(void);\nint f(int x) {{ goto L; {body} L: return x; }}\n"), + ); + } +} + +/// A `goto` out of one arm leaves *that arm* without a block, which is a +/// different site from the condition's. +/// +/// These are why the fix belongs in the shared diamond builder rather than on +/// the condition alone: the helper reads the block back at the end of each arm +/// too. +#[test] +fn misc_a_goto_out_of_a_conditional_arm_does_not_ice() { + for (tag, init) in [ + ("then", "x ? ({ goto L; g(); }) : g()"), + ("else", "x ? g() : ({ goto L; g(); })"), + ("and", "x && ({ goto L; g(); })"), + ("or", "x || ({ goto L; g(); })"), + ("elvis", "g() ?: ({ goto L; g(); })"), + ("both", "x ? ({ goto L; g(); }) : ({ goto L; g(); })"), + ] { + compile_expect_ok( + &format!("nocur_arm_{tag}"), + &format!("int g(void);\nint f(int x) {{ int y = {init}; L: return y; }}\n"), + ); + } +} + +/// Unreachable code before the first `case` is discarded, and the reachable +/// part of the same function still gives the right answer. +#[test] +fn misc_an_unreachable_statement_does_not_change_the_answer() { + let code = r#" +int calls; +int g(void) { calls++; return 3; } + +int f(int x) +{ + switch (x) { + g() ? g() : g(); /* unreachable: never selected */ + (void)(g() && g()); + case 1: + return 1; + default: + return 2; + } +} + +int main(void) +{ + if (f(1) != 1) return 1; + if (f(0) != 2) return 2; + /* Nothing before the first case may run. */ + if (calls != 0) return 3; + return 0; +} +"#; + assert_eq!(compile_and_run("nocur_answer", code, &[]), 0); +} diff --git a/cc/types.rs b/cc/types.rs index b7d965002..3db782f28 100644 --- a/cc/types.rs +++ b/cc/types.rs @@ -462,6 +462,34 @@ impl Type { /// compatible with `int`. Invisible to `sizeof`, fatal to any comparison. pub const DECL_SPECIFIERS: TypeModifiers = Self::STORAGE_CLASS.union(TypeModifiers::NORETURN); + /// The type qualifiers of C17 6.7.3p1. + /// + /// These are the bits a *type* carries at its top level, as opposed to the + /// declaration specifiers above: `const int` and `int` are two types, while + /// `static int` and `int` are one. Four copies of this list had been + /// spelled out inline -- in `compatible_ignoring_base`, in `compatible`, in + /// `assign_fault`'s pointer rule and in the parser's + /// `lvalue_converted_type` -- which is three chances for the set to drift. + pub const QUALIFIERS: TypeModifiers = TypeModifiers::CONST + .union(TypeModifiers::VOLATILE) + .union(TypeModifiers::RESTRICT) + .union(TypeModifiers::ATOMIC); + + /// The qualifiers a member inherits from the object that holds it. + /// + /// C17 6.5.2.3p3/p4 gives `s.m` the "so-qualified version" of the member's + /// type, which is the member's type plus the qualifiers of `s`. Not all + /// four travel: + /// + /// * `_Atomic` does not. A member of an `_Atomic` struct is not itself + /// atomic -- there is no lock-free way to read one out of an atomic + /// object -- and gcc does not make it so. Reaching into one at all is + /// what `Parser::warn_atomic_member_access` already reports. + /// * `restrict` cannot: 6.7.3p2 admits it only on a pointer to an object + /// type, so a composite never carries it in the first place. + pub const MEMBER_QUALIFIERS: TypeModifiers = + TypeModifiers::CONST.union(TypeModifiers::VOLATILE); + /// The storage-class specifiers of C17 6.7.1, and `inline`. /// /// What a declaration records as its storage class, and what a declarator @@ -488,12 +516,6 @@ impl Type { /// With TypeId interning, base types are compared by TypeId equality. /// For full recursive comparison, use TypeTable::types_compatible(). fn compatible_ignoring_base(&self, other: &Type) -> bool { - // Top-level qualifiers to ignore - const QUALIFIERS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - // Compare kinds first if self.kind != other.kind { return false; @@ -520,7 +542,7 @@ impl Type { const REDUNDANT_SIZE: TypeModifiers = TypeModifiers::SHORT .union(TypeModifiers::LONG) .union(TypeModifiers::LONGLONG); - let ignored = QUALIFIERS + let ignored = Self::QUALIFIERS .union(redundant_signed) .union(REDUNDANT_SIZE) .union(Self::DECL_SPECIFIERS); @@ -1728,12 +1750,8 @@ impl TypeTable { // Compatible targets, but the assignment must not silently gain // write access: the target's qualifiers have to include the // source's. - const QUALS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - let t_quals = self.modifiers(t_pointee) & QUALS; - let v_quals = self.modifiers(v_pointee) & QUALS; + let t_quals = self.qualifiers(t_pointee); + let v_quals = self.qualifiers(v_pointee); return (!t_quals.contains(v_quals)).then_some(AssignFault::QualifierDiscard); } @@ -1849,6 +1867,77 @@ impl TypeTable { } } + /// The top-level type qualifiers of `id` (C17 6.7.3p1). + /// + /// Only the qualifiers: `modifiers` answers with the declaration + /// specifiers and the size and sign spellings mixed in, and every caller + /// that wanted "is this `const`?" had to mask them off itself. + pub fn qualifiers(&self, id: TypeId) -> TypeModifiers { + self.modifiers(id) & Type::QUALIFIERS + } + + /// The version of `id` qualified with `quals` -- the "so-qualified + /// version" of C17 6.5.2.3p3/p4. + /// + /// This is what a member access yields: `s.m` has the member's type plus + /// the qualifiers of `s`, so a member of a `volatile` object is volatile + /// and a member of a `const` object is not assignable. Without it a + /// `volatile struct` read as an ordinary struct and DCE deleted the load. + /// + /// Qualifiers outside [`Type::QUALIFIERS`] are ignored, and a type that + /// already carries all of them is returned unchanged -- so the common case + /// of an unqualified object interns nothing. A composite is not + /// deduplicated by [`Self::intern`] (it has identity), so qualifying one + /// hands back a fresh `TypeId` each time; that is sound because + /// compatibility of composites is decided by tag and members rather than + /// by id, and it is rare enough not to matter -- only an access to a + /// member of aggregate type, through a qualified object, reaches it. + pub fn qualified_with(&mut self, id: TypeId, quals: TypeModifiers) -> TypeId { + let add = quals & Type::QUALIFIERS; + if add.is_empty() { + return id; + } + // C17 6.7.3p10: where an array type is qualified, the *element type* is + // so-qualified and the array is not. That is also how a declaration + // records it -- `const int a[4]` puts the `const` on the element -- so + // qualifying the array instead would leave `cs.arr[0]` an ordinary + // `int`, which the subscript reads from the element type, and a write + // to it would be accepted. + if self.kind(id) == TypeKind::Array { + let Some(elem) = self.base_type(id) else { + return id; + }; + let qualified_elem = self.qualified_with(elem, add); + if qualified_elem == elem { + return id; + } + let mut array = self.get(id).clone(); + array.base = Some(qualified_elem); + return self.intern(array); + } + if self.modifiers(id).contains(add) { + return id; + } + let mut qualified = self.get(id).clone(); + qualified.modifiers |= add; + self.intern(qualified) + } + + /// The unqualified version of `id` (C17 6.3.2.1p2). + /// + /// Lvalue conversion drops the qualifiers, so this is the type of the + /// *value* an lvalue yields: `volatile int v; v + 0` has type `int`, and + /// nothing downstream may conclude from the sum's type that the addition + /// touched a volatile object. + pub fn unqualified(&mut self, id: TypeId) -> TypeId { + if self.qualifiers(id).is_empty() { + return id; + } + let mut unqualified = self.get(id).clone(); + unqualified.modifiers.remove(Type::QUALIFIERS); + self.intern(unqualified) + } + /// How many scalar initializers it takes to fill this type. /// /// This is the measure brace elision runs on (C17 6.7.9p20): a brace-less @@ -2400,11 +2489,45 @@ impl TypeTable { 8 // All supported platforms use 8-byte alignment for va_list } + /// The type an access to `name` yields, in an object of type `object`. + /// + /// C17 6.5.2.3p3/p4: the result of `s.m` and of `p->m` has the + /// *so-qualified* version of the member's type, so a member of a + /// `volatile` object is volatile and a member of a `const` object is not + /// assignable. [`Self::find_member`] answers with the member's *declared* + /// type, which is what every other caller of it wants -- an offset, a size, + /// a bit-field's width -- so the rule lives here, beside it, rather than in + /// each of the two expression forms that need it. + /// + /// The two type ids are not redundant. `members` is `object` after the + /// parser resolved an incomplete tag to its definition, which is where the + /// member list is; the qualifiers have to come from `object`, because + /// resolving answers with the *tag's* type and a tag is never qualified -- + /// resolving first is how `volatile struct S` loses the `volatile`. + /// + /// For `p->m` the object is the pointee: `struct S *volatile p` qualifies + /// the pointer, not what it points at. + pub fn member_access_type( + &mut self, + object: TypeId, + members: TypeId, + name: StringId, + ) -> Option { + let quals = self.qualifiers(object) & Type::MEMBER_QUALIFIERS; + let declared = self.find_member(members, name)?.typ; + Some(self.qualified_with(declared, quals)) + } + /// Find a member in a struct/union type, including anonymous struct/union members /// C11 6.7.2.1p13: "An unnamed member of structure type with no tag is called an /// anonymous structure; an unnamed member of union type with no tag is called an /// anonymous union. The members of an anonymous structure or union are considered /// to be members of the containing structure or union." + /// + /// The `typ` this answers with is the member's *declared* type. An + /// expression that accesses the member has the so-qualified version of it + /// instead (C17 6.5.2.3p3/p4) -- see [`Self::member_access_type`], which is + /// what the `.` and `->` operators go through. pub fn find_member(&self, id: TypeId, name: StringId) -> Option { self.find_member_recursive(id, name, 0) } @@ -2473,13 +2596,7 @@ impl TypeTable { if id1 == id2 { return true; } - const QUALIFIERS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - if quals == TopLevelQualifiers::Significant - && self.get(id1).modifiers.intersection(QUALIFIERS) - != self.get(id2).modifiers.intersection(QUALIFIERS) + if quals == TopLevelQualifiers::Significant && self.qualifiers(id1) != self.qualifiers(id2) { return false; } @@ -2872,6 +2989,144 @@ mod tests { assert!(types.contains_volatile(opaque)); } + /// `qualifiers`, `qualified_with` and `unqualified` on the same type. + /// + /// The pair has to be exact inverses on the top-level qualifiers and to + /// leave everything else -- the kind, the size spellings, the storage + /// class -- alone, because they are what forms and unforms the + /// so-qualified type of a member access. + #[test] + fn qualifying_a_type_adds_and_removes_only_the_qualifiers() { + let mut types = TypeTable::new(&crate::target::Target::host()); + let int = types.int_id; + assert!(types.qualifiers(int).is_empty()); + + // Adding, one qualifier at a time and then both. + let vol = types.qualified_with(int, TypeModifiers::VOLATILE); + assert_eq!(types.qualifiers(vol), TypeModifiers::VOLATILE); + assert_eq!(types.kind(vol), TypeKind::Int); + assert_eq!(types.size_bits(vol), types.size_bits(int)); + let cv = types.qualified_with(vol, TypeModifiers::CONST); + assert_eq!( + types.qualifiers(cv), + TypeModifiers::CONST | TypeModifiers::VOLATILE + ); + + // Interned, so the same request answers with the same id -- and a type + // that already carries the qualifier is returned untouched. + assert_eq!(types.qualified_with(int, TypeModifiers::VOLATILE), vol); + assert_eq!(types.qualified_with(vol, TypeModifiers::VOLATILE), vol); + assert_eq!(types.qualified_with(int, TypeModifiers::empty()), int); + + // Nothing outside `Type::QUALIFIERS` travels: a storage class is a + // property of a declaration, not of a type. + assert_eq!(types.qualified_with(int, TypeModifiers::STATIC), int); + + // And back down again. + assert_eq!(types.unqualified(cv), int); + assert_eq!(types.unqualified(vol), int); + assert_eq!(types.unqualified(int), int); + + // A qualifier below the top level is not a top-level qualifier: + // `volatile int *` is an ordinary pointer. + let ptr_to_vol = types.intern(Type::pointer(vol)); + assert!(types.qualifiers(ptr_to_vol).is_empty()); + assert_eq!(types.unqualified(ptr_to_vol), ptr_to_vol); + + // Qualifying an array qualifies its element type (C17 6.7.3p10), at + // every level, and the array itself stays unqualified -- which is what + // a subscript of it then reads. + let arr = types.intern(Type::array(int, 4)); + let const_arr = types.qualified_with(arr, TypeModifiers::CONST); + assert!(types.qualifiers(const_arr).is_empty()); + let elem = types.base_type(const_arr).expect("element type"); + assert_eq!(types.qualifiers(elem), TypeModifiers::CONST); + let rows = types.intern(Type::array(arr, 2)); + let vol_rows = types.qualified_with(rows, TypeModifiers::VOLATILE); + let row = types.base_type(vol_rows).expect("row type"); + let cell = types.base_type(row).expect("cell type"); + assert_eq!(types.qualifiers(cell), TypeModifiers::VOLATILE); + assert!(types.contains_volatile(vol_rows)); + } + + /// C17 6.5.2.3p3/p4: a member access has the *so-qualified* version of the + /// member's type. + /// + /// `find_member` answers with the declared type, which is what an offset or + /// a width is read from; the access has the object's qualifiers as well, + /// which is what makes a member of a `volatile` object volatile and a + /// member of a `const` object unassignable. + #[test] + fn a_member_access_is_qualified_by_the_object() { + let mut types = TypeTable::new(&crate::target::Target::host()); + let mut idents = crate::strings::StringTable::new(); + let a = idents.intern("a"); + let v = idents.intern("v"); + let int = types.int_id; + let vol_int = types.intern(Type::with_modifiers(TypeKind::Int, TypeModifiers::VOLATILE)); + let member = |name, typ| StructMember { + name, + typ, + offset: 0, + bit_offset: None, + bit_width: None, + access_bytes: None, + explicit_align: None, + }; + let tag = idents.intern("S"); + let composite = CompositeType { + tag: Some(tag), + members: vec![member(a, int), member(v, vol_int)], + enum_constants: Vec::new(), + size: 8, + align: 4, + member_align: 4, + is_complete: true, + transparent: false, + anon_id: None, + }; + let plain = types.intern(Type::struct_type(composite)); + + // An unqualified object: the declared types, unchanged. + assert_eq!(types.member_access_type(plain, plain, a), Some(int)); + assert_eq!(types.member_access_type(plain, plain, v), Some(vol_int)); + assert_eq!(types.member_access_type(plain, plain, tag), None); + + // A `volatile` object makes every member volatile, and a `const` one + // makes every member `const`. + let vol_obj = types.qualified_with(plain, TypeModifiers::VOLATILE); + let from_vol = types.member_access_type(vol_obj, vol_obj, a).unwrap(); + assert_eq!(types.qualifiers(from_vol), TypeModifiers::VOLATILE); + assert!(types.contains_volatile(from_vol)); + let const_obj = types.qualified_with(plain, TypeModifiers::CONST); + let from_const = types.member_access_type(const_obj, const_obj, a).unwrap(); + assert_eq!(types.qualifiers(from_const), TypeModifiers::CONST); + + // `_Atomic` does not travel: a member of an `_Atomic` struct cannot be + // read atomically, and gcc does not claim it can. + let atomic_obj = types.qualified_with(plain, TypeModifiers::ATOMIC); + assert_eq!( + types.member_access_type(atomic_obj, atomic_obj, a), + Some(int) + ); + + // The two type ids are not interchangeable. The qualifiers come from + // the object as written; the members come from the type the parser + // resolved it to, which is the tag's and is never qualified. Reading + // the qualifiers from the resolved type is how `volatile struct S` + // loses its `volatile`. + let incomplete = types.intern(Type::struct_type(CompositeType::incomplete(Some(tag)))); + let vol_incomplete = types.qualified_with(incomplete, TypeModifiers::VOLATILE); + assert_eq!( + types.member_access_type(vol_incomplete, plain, a), + Some(vol_int) + ); + assert_eq!( + types.member_access_type(vol_incomplete, vol_incomplete, a), + None + ); + } + use super::*; /// `make_complex` and `complex_base` must be exact inverses, for every