From 08f3fa044a614023df03bbffb94a5a1e3ae1d854 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Tue, 29 Sep 2026 17:15:16 -0400 Subject: [PATCH 01/17] cc: build c17 against the library instead of recompiling every module main.rs declared its own `mod abi; mod arch; mod ir; ...` for 19 of the 21 modules lib.rs already exports, so rustc compiled the whole ~160k-line compiler twice per build, once for the lib and once for the c17 binary. The other three binaries (cflow, ctags, cxref) already went through the library. Import the modules from posixutils_cc instead. Nothing in main.rs reached past the library's public surface, and seven of the nineteen modules were unused there, so they are dropped rather than imported. Cold `cargo build -p posixutils-cc`, dev profile, CARGO_INCREMENTAL=0: 42.0s -> 22.1s. The two ~22s units were serialized, because cargo makes a bin depend on its package's lib, and each held ~1GB RSS. `cargo test` also stops building and running the in-module test files twice. Co-Authored-By: Claude Opus 5 (1M context) --- cc/main.rs | 31 ++++++++++++------------------- 1 file changed, 12 insertions(+), 19 deletions(-) diff --git a/cc/main.rs b/cc/main.rs index 897cc6f92..9cefe2520 100644 --- a/cc/main.rs +++ b/cc/main.rs @@ -11,25 +11,18 @@ #![recursion_limit = "512"] -mod abi; -mod arch; -mod builtin_headers; -mod builtins; -mod constexpr; -mod diag; -mod float; -mod ir; -mod kw; -mod linkargs; -mod opt; -mod os; -mod parse; -mod rtlib; -mod strings; -mod symbol; -mod target; -mod token; -mod types; +use posixutils_cc::arch; +use posixutils_cc::builtins; +use posixutils_cc::diag; +use posixutils_cc::ir; +use posixutils_cc::linkargs; +use posixutils_cc::opt; +use posixutils_cc::parse; +use posixutils_cc::strings; +use posixutils_cc::symbol; +use posixutils_cc::target; +use posixutils_cc::token; +use posixutils_cc::types; use clap::Parser; use gettextrs::{gettext, gettext_args}; From d910ef40822ab67b576db2f3eb24e226a4b01eb2 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Tue, 29 Sep 2026 22:02:59 -0400 Subject: [PATCH 02/17] cc: link a for loop's back edge from the block the branch lands in `linearize_for` emitted `br cond_bb` into the current block and then linked the edge from `post_bb`. When the post-expression contains `&&`, `||` or `?:`, lowering it splits the block and leaves the current block on the merge, so the terminator and the recorded edge ended up in different blocks: `post_bb` claimed a successor it no longer branched to, and the merge block had one nothing recorded. The compiled loop then never terminated -- `for (int i = 0; i < n; (void)(n && 1), i++)` hangs. Link from the block the branch was emitted into, which is what every other loop lowering in the file already does, and emit nothing when there is no current block (the documented post-`goto` state). The switch-body walker carries its own copy of the `for` lowering and had the same defect. Tested on the CFG rather than on the program's answer, because with the defect present the program does not return a wrong value, it fails to return: a runtime test would hang the suite instead of failing it. The new `cfg_inconsistency` helper states the invariant -- a block's `children` are exactly its terminator's targets, and `parents` is their inverse -- over every position a lowering evaluates an expression before ending a block, and a meta-test asserts the helper rejects a misrecorded edge. The end-to-end test runs only now that the shape is known to be acyclic. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize_stmt.rs | 22 +++- cc/ir/test_linearize.rs | 218 +++++++++++++++++++++++++++++++++++ cc/tests/c89/control_flow.rs | 75 ++++++++++++ 3 files changed, 311 insertions(+), 4 deletions(-) diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index ea105972c..19519d6a8 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -1465,8 +1465,15 @@ impl<'a> super::linearize::Linearizer<'a> { if let Some(post_expr) = post { self.linearize_expr(post_expr); } - self.emit(Instruction::br(cond_bb)); - self.link_bb(post_bb, cond_bb); + // From the block the post-expression ended in, which `&&`, `||` and + // `?:` can make a different one from post_bb. Linking the back edge + // from post_bb itself recorded an edge out of a block that no longer + // holds the branch, and left the merge block that does hold it with an + // unrecorded successor -- the loop then never terminated. + if let Some(current) = self.current_bb { + self.emit(Instruction::br(cond_bb)); + self.link_bb(current, cond_bb); + } // Exit block self.switch_bb(exit_bb); @@ -2661,8 +2668,15 @@ impl<'a> super::linearize::Linearizer<'a> { if let Some(post_expr) = post { self.linearize_expr(post_expr); } - self.emit(Instruction::br(cond_bb)); - self.link_bb(post_bb, cond_bb); + // From the block the post-expression ended in, which `&&`, `||` and + // `?:` can make a different one from post_bb. Linking the back edge + // from post_bb itself recorded an edge out of a block that no longer + // holds the branch, and left the merge block that does hold it with an + // unrecorded successor -- the loop then never terminated. + if let Some(current) = self.current_bb { + self.emit(Instruction::br(cond_bb)); + self.link_bb(current, cond_bb); + } self.switch_bb(exit_bb); self.pop_scope(); diff --git a/cc/ir/test_linearize.rs b/cc/ir/test_linearize.rs index 523381e0a..347b7a1fa 100644 --- a/cc/ir/test_linearize.rs +++ b/cc/ir/test_linearize.rs @@ -8812,3 +8812,221 @@ fn test_void_cast_of_a_complex_converts_nothing() { fn x86_64_linux() -> Target { Target::new(crate::target::Arch::X86_64, crate::target::Os::Linux) } + +// CFG consistency + +/// Every block's recorded successors are exactly the blocks its terminator +/// names, and `parents` is the inverse of `children`. +/// +/// Returns a description of the first inconsistency, or `None`. +fn cfg_inconsistency(func: &Function) -> Option { + use std::collections::HashSet; + + for bb in &func.blocks { + let children: HashSet = bb.children.iter().copied().collect(); + if children.len() != bb.children.len() { + return Some(format!("{}: duplicate edge in children", bb.id)); + } + + let named = match bb.insns.last() { + Some(last) if last.op.is_terminator() => crate::ir::propagate::terminator_targets(last), + // A block with no terminator falls through to nothing the CFG can + // name; `children` must then be empty too. + _ => HashSet::new(), + }; + + if named != children { + return Some(format!( + "{}: terminator names {:?} but children are {:?}", + bb.id, + sorted_ids(&named), + sorted_ids(&children), + )); + } + } + + // `parents` is the inverse of `children`. + let mut expected: std::collections::HashMap> = + std::collections::HashMap::new(); + for bb in &func.blocks { + for child in &bb.children { + expected.entry(*child).or_default().insert(bb.id); + } + } + for bb in &func.blocks { + let have: HashSet = bb.parents.iter().copied().collect(); + let want = expected.remove(&bb.id).unwrap_or_default(); + if have != want { + return Some(format!( + "{}: parents are {:?} but {:?} name it as a successor", + bb.id, + sorted_ids(&have), + sorted_ids(&want), + )); + } + } + + None +} + +fn sorted_ids(s: &std::collections::HashSet) -> Vec { + let mut v: Vec = s.iter().map(|b| b.0).collect(); + v.sort_unstable(); + v +} + +/// A `for` post-expression that splits the block still links the back edge from +/// the block the branch was emitted into. +/// +/// `&&`, `||` and `?:` leave `current_bb` on their merge block, so +/// `link_bb(post_bb, cond_bb)` recorded an edge out of a block that no longer +/// holds the terminator -- the loop's back edge went missing from the CFG while +/// a merge block gained an unrecorded one. Both `for` arms had it: the one in +/// `linearize_for` and its copy in the switch-body walker. +/// +/// Stated on the CFG rather than on the program's answer because the defect +/// makes the compiled loop non-terminating, which a runtime test cannot +/// observe without hanging. +#[test] +fn for_post_expression_splitting_the_block_keeps_the_back_edge() { + let target = Target::host(); + let cases = [ + ("and", "for (int i = 0; i < n; (void)(n && 1), i++) s += i;"), + ("or", "for (int i = 0; i < n; (void)(n || 0), i++) s += i;"), + ( + "ternary", + "for (int i = 0; i < n; (void)(n ? 1 : 2), i++) s += i;", + ), + ( + "and_in_cond_and_post", + "for (int i = 0; i < n && n; (void)(n && 1), i++) s += i;", + ), + ( + "nested_and", + "for (int i = 0; i < n; (void)(n && (i || 1)), i++) s += i;", + ), + ]; + + for (tag, loop_src) in cases { + // Plain, and again inside a switch body -- a separate copy of the + // lowering that carried the same defect. + let plain = format!("int f(int n) {{ int s = 0; {loop_src} return s; }}"); + let in_switch = format!( + "int f(int n) {{ switch (n) {{ case 5: {{ int s = 0; {loop_src} return s; }} \ + default: return 0; }} }}" + ); + + for (where_, src) in [("plain", &plain), ("in_switch", &in_switch)] { + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + assert!( + cfg_inconsistency(func).is_none(), + "{tag} / {where_}: {}\nsource: {src}", + cfg_inconsistency(func).unwrap() + ); + } + } +} + +/// The check above is only as good as its ability to see a broken edge, so +/// assert it rejects one. +#[test] +fn cfg_inconsistency_sees_a_misrecorded_edge() { + let target = Target::host(); + let module = linearize_source( + "int f(int n) { int s = 0; for (int i = 0; i < n; i++) s += i; return s; }", + &target, + ); + let mut func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f") + .clone(); + assert!(cfg_inconsistency(&func).is_none(), "baseline is consistent"); + + // Record a successor the terminator does not name -- exactly the shape the + // `for` defect produced. + let victim = func.blocks.len() - 1; + let bogus = func.blocks[0].id; + func.blocks[victim].children.push(bogus); + assert!( + cfg_inconsistency(&func).is_some(), + "an unnamed successor must be reported" + ); +} + +/// The same audit over every other lowering that can end a block with a +/// terminator after evaluating an expression: `while`, `do`/`while`, `switch`, +/// `if`, `?:`, `goto`, `break`/`continue` and the loop lowerings' copies in the +/// switch-body walker. +/// +/// Each source puts a short-circuit operator or a `?:` -- the things that split +/// the block and move `current_bb` to a merge block -- where the construct +/// evaluates an expression, so an edge linked from the block the construct +/// started in rather than the one it ended in shows up as an inconsistency. +#[test] +fn short_circuit_operands_keep_every_lowering_cfg_consistent() { + let target = Target::host(); + let cases = [ + ("while_cond", "while (n && s < 3) s++;"), + ("do_while_cond", "do { s++; } while (n && s < 3);"), + ("if_cond", "if (n && s) s = 1; else s = 2;"), + ("ternary", "s = n && 1 ? (n || 2) : (n ? 3 : 4);"), + ("for_cond", "for (int i = 0; i < n && n; i++) s += i;"), + ("for_init", "for (int i = (n && 1); i < n; i++) s += i;"), + ( + "switch_selector", + "switch (n && 1) { case 1: s = 1; break; }", + ), + ( + "switch_in_loop", + "while (s < 3) { switch (n && 1) { case 1: s++; break; default: s += 2; } }", + ), + ("break_after_split", "while (1) { if (n && 1) break; s++; }"), + ( + "continue_after_split", + "for (int i = 0; i < n; i++) { if (n || 0) continue; s++; }", + ), + ( + "goto_after_split", + "if (n && 1) goto done; s = 7; done: s++;", + ), + ( + "while_in_switch", + "switch (n) { case 5: while (n && s < 3) s++; break; default: s = 1; }", + ), + ( + "do_while_in_switch", + "switch (n) { case 5: do { s++; } while (n && s < 3); break; default: s = 1; }", + ), + ( + "nested_for_in_switch", + "switch (n) { case 5: for (int i = 0; i < n; (void)(n && 1), i++) \ + for (int j = 0; j < n; (void)(n || 0), j++) s++; break; default: s = 1; }", + ), + ( + "duffs_device", + "switch (n % 2) { case 0: do { s++; case 1: s += 2; } while (n && --n > 0); }", + ), + ]; + + for (tag, body) in cases { + let src = format!("int f(int n) {{ int s = 0; {body} return s; }}"); + let module = linearize_source(&src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + assert!( + cfg_inconsistency(func).is_none(), + "{tag}: {}\nsource: {src}", + cfg_inconsistency(func).unwrap() + ); + } +} diff --git a/cc/tests/c89/control_flow.rs b/cc/tests/c89/control_flow.rs index 470a715e4..b310d96d2 100644 --- a/cc/tests/c89/control_flow.rs +++ b/cc/tests/c89/control_flow.rs @@ -985,3 +985,78 @@ int main(void) assert_eq!(compile_and_run("label_addr_no_goto", code, &[]), 0); assert_eq!(compile_and_run_optimized("label_addr_no_goto_opt", code), 0); } + +/// A `for` post-expression that splits the block keeps the loop's back edge. +/// +/// `&&`, `||` and `?:` leave the current block on their merge, not on the block +/// the post-expression started in, so the branch back to the condition landed +/// in one block while the CFG edge was recorded from another. Every other loop +/// lowering reads `self.current_bb` back before linking; the two `for` arms did +/// not. +/// +/// The exhaustive statement of this is the CFG-consistency check in +/// `ir::test_linearize`; asserting it there rather than here is deliberate, +/// because with the defect present the compiled program does not merely return +/// the wrong answer, it never terminates -- a runtime test would hang the suite +/// instead of failing it. What is left here is the end-to-end answer, which is +/// safe to run only once the shape is known to be acyclic. +#[test] +fn c89_for_post_expression_that_splits_the_block_keeps_the_back_edge() { + let code = r#" +int and_in_post(int n) +{ + int s = 0; + for (int i = 0; i < n; (void)(n && 1), i++) + s += i; + return s; +} + +int or_in_post(int n) +{ + int s = 0; + for (int i = 0; i < n; (void)(n || 0), i++) + s += i; + return s; +} + +int ternary_in_post(int n) +{ + int s = 0; + for (int i = 0; i < n; (void)(n ? 1 : 2), i++) + s += i; + return s; +} + +/* The same shape inside a switch body, which is a second copy of the + lowering and carried the same defect. */ +int and_in_post_in_switch(int n) +{ + switch (n) { + case 5: { + int s = 0; + for (int i = 0; i < n; (void)(n && 1), i++) + s += i; + return s; + } + default: + return -2; + } +} + +int main(void) +{ + if (and_in_post(5) != 10) return 1; + if (or_in_post(5) != 10) return 2; + if (ternary_in_post(5) != 10) return 3; + if (and_in_post_in_switch(5) != 10) return 4; + /* A zero-trip loop still has to reach the exit. */ + if (and_in_post(0) != 0) return 5; + return 0; +} +"#; + assert_eq!(compile_and_run("for_post_splits_block", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("for_post_splits_block_opt", code), + 0 + ); +} From 3caab0290f1121c8f946f793dbecec58af11b643 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Tue, 29 Sep 2026 22:03:15 -0400 Subject: [PATCH 03/17] cc: discard array initializers past the last element instead of storing them `group_array_init_elements` tracked an element cursor but never bounded it by the array, and did not even receive the array's type -- only its element type. An initializer past the last element therefore got a group, an offset beyond the object, and a store: `int x[2] = {7, 8}; int a[2] = {1, 2, 3};` read back `x = {3, 8}`, the excess 3 having landed on the neighbouring local, and the file-scope form emitted a third `.long` under a two-element symbol. C17 6.7.9p2 makes the excess element a constraint violation, which c17 already diagnoses; this is where the diagnosis stops being advisory. Take the array type, derive the element type from it, and drop any group whose index is out of range. Both lowering paths and every caller -- static globals, static locals, automatic locals, compound literals, nested levels -- share this function, so one bound covers them all. An absent or zero size means an array sized *by* its initializer (an incomplete type, a flexible array member, a GNU zero-length array), so nothing in the list can be excess and the bound does not apply. A range designator is clamped rather than dropped, so `[0 ... 4] = 1` on an `int[3]` still fills the three elements that exist. The bound is on the index, not on the count: `int d[3] = {[2] = 3, 1};` has three elements and two initializers, but the 1 resumes after `[2]` at index 3 (C17 6.7.9p17) and is excess. Counting would have kept it and written past the array -- the very defect -- so the tests state that case explicitly, with guard objects either side. Struct and union excess members were already dropped correctly and needed no change. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize_init.rs | 201 +++++++++++++++++++++++++++++++---- cc/ir/linearize_stmt.rs | 2 +- cc/tests/c99/initializers.rs | 93 ++++++++++++++++ 3 files changed, 276 insertions(+), 20 deletions(-) diff --git a/cc/ir/linearize_init.rs b/cc/ir/linearize_init.rs index 927a0da0b..0128c57fa 100644 --- a/cc/ir/linearize_init.rs +++ b/cc/ir/linearize_init.rs @@ -860,11 +860,32 @@ impl<'a> super::linearize::Linearizer<'a> { /// Group array init elements by index, handling designators, brace elision, /// and nested InitList flattening. Shared between static and runtime paths. + /// + /// `array_typ` is the array being initialized, not its element type: the + /// bound is needed as well as the element type, because an initializer + /// past the last element is excess (C17 6.7.9p2) and must be *discarded*. + /// Grouping it anyway gave it an offset beyond the object and both lowering + /// paths then wrote there -- `int a[2] = {1, 2, 3};` stored the 3 over + /// whatever the frame put after `a`, and the file-scope form emitted a + /// third `.long` under a two-element symbol. c17 already warns about the + /// excess element; this is where it stops mattering. pub(crate) fn group_array_init_elements( &self, elements: &[InitElement], - elem_type: TypeId, + array_typ: TypeId, ) -> ArrayInitGroups { + let elem_type = self.types.base_type(array_typ).unwrap_or(self.types.int_id); + + // An absent or zero bound is an array sized *by* this initializer -- an + // incomplete type `int a[] = {1, 2, 3}`, a flexible array member, or a + // GNU zero-length array -- so nothing in the list can be excess. + let last_index = self + .types + .array_size(array_typ) + .filter(|&n| n > 0) + .map(|n| n as i64 - 1); + let in_bounds = |idx: i64| last_index.is_none_or(|last| (0..=last).contains(&idx)); + let mut element_lists: HashMap> = HashMap::new(); let mut element_indices: Vec = Vec::new(); let mut current_idx: i64 = 0; @@ -912,7 +933,13 @@ impl<'a> super::linearize::Linearizer<'a> { if remaining_designators.is_empty() && self.is_brace_elision_candidate(element, elem_type) { + // Consumed either way: the elements belong to this slot, and + // leaving them in the list would make the next iteration read + // them as initializers for the enclosing array. let sub_elements = self.consume_brace_elision(elements, &mut elem_idx, elem_type); + if !in_bounds(element_index) { + continue; + } let entry = element_lists.entry(element_index).or_insert_with(|| { element_indices.push(element_index); Vec::new() @@ -926,27 +953,37 @@ impl<'a> super::linearize::Linearizer<'a> { // Expanding here keeps both lowering paths -- the static data // image and the runtime stores -- unchanged, and matches how c17 // already lowers a bulk initializer element by element. + // + // A range is clamped rather than dropped whole, so that + // `[0 ... 4] = 1` on a three-element array still initializes the + // three elements the array has. Clamping the high endpoint also + // keeps the loop off the excess indices rather than walking one + // iteration per discarded element, which matters for an endpoint + // far past the array. let span_end = index_high.unwrap_or(element_index); - for target_index in element_index..=span_end { - let entry = element_lists.entry(target_index).or_insert_with(|| { - element_indices.push(target_index); - Vec::new() - }); + let span_end = last_index.map_or(span_end, |last| span_end.min(last)); + if in_bounds(element_index) { + for target_index in element_index..=span_end { + let entry = element_lists.entry(target_index).or_insert_with(|| { + element_indices.push(target_index); + Vec::new() + }); - if remaining_designators.is_empty() { - if let ExprKind::InitList { - elements: nested_elements, - } = &element.value.kind - { - entry.extend(nested_elements.clone()); - continue; + if remaining_designators.is_empty() { + if let ExprKind::InitList { + elements: nested_elements, + } = &element.value.kind + { + entry.extend(nested_elements.clone()); + continue; + } } - } - entry.push(InitElement { - designators: remaining_designators.clone(), - value: element.value.clone(), - }); + entry.push(InitElement { + designators: remaining_designators.clone(), + value: element.value.clone(), + }); + } } elem_idx += 1; } @@ -1100,7 +1137,7 @@ impl<'a> super::linearize::Linearizer<'a> { TypeKind::Array | TypeKind::Struct | TypeKind::Union ); - let groups = self.group_array_init_elements(elements, elem_type); + let groups = self.group_array_init_elements(elements, typ); let mut init_elements = Vec::new(); for element_index in groups.indices { let Some(list) = groups.element_lists.get(&element_index) else { @@ -1769,3 +1806,129 @@ impl<'a> super::linearize::Linearizer<'a> { Err(AliasFault::Undefined) } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::parse::ast::Expr; + use crate::symbol::SymbolTable; + use crate::target::Target; + use crate::types::Type; + + /// A positional initializer element holding an `int` constant. + fn positional(value: i64, types: &TypeTable) -> InitElement { + InitElement { + designators: vec![], + value: Box::new(Expr::int(value, types)), + } + } + + /// The same, addressed by a designator. + fn designated(designator: Designator, value: i64, types: &TypeTable) -> InitElement { + InitElement { + designators: vec![designator], + value: Box::new(Expr::int(value, types)), + } + } + + /// The indices `group_array_init_elements` keeps for `elements` when they + /// initialize an `int` array of `size` elements (`None`: a bound derived + /// from the initializer itself, as for `int a[] = {1, 2, 3}`). + fn kept_indices( + size: Option, + build: impl Fn(&TypeTable) -> Vec, + ) -> Vec { + let target = Target::host(); + let mut types = TypeTable::new(&target); + let elements = build(&types); + let array = types.intern(Type { + kind: TypeKind::Array, + base: Some(types.int_id), + array_size: size, + ..Default::default() + }); + let symbols = SymbolTable::new(); + let strings = crate::strings::StringTable::new(); + let lin = Linearizer::new(&symbols, &types, &strings, &target); + let groups = lin.group_array_init_elements(&elements, array); + assert_eq!(groups.indices.len(), groups.element_lists.len()); + groups.indices + } + + #[test] + fn excess_array_elements_are_discarded() { + // `int a[2] = {1, 2, 3};` -- the third element has nowhere to go. + let indices = kept_indices(Some(2), |types| { + (1..=3).map(|v| positional(v, types)).collect() + }); + assert_eq!(indices, vec![0, 1]); + } + + #[test] + fn a_positional_element_past_a_designator_can_be_excess() { + // `int a[3] = {[2] = 3, 1};` -- C17 6.7.9p17 resumes at index 3, so + // the `1` is excess although the list is shorter than the array. + let indices = kept_indices(Some(3), |types| { + vec![ + designated(Designator::Index(2), 3, types), + positional(1, types), + ] + }); + assert_eq!(indices, vec![2]); + } + + #[test] + fn a_range_designator_is_clamped_to_the_array() { + // `int a[3] = {[0 ... 4] = 7};` initializes the three elements it has. + let indices = kept_indices(Some(3), |types| { + vec![designated(Designator::IndexRange(0, 4), 7, types)] + }); + assert_eq!(indices, vec![0, 1, 2]); + } + + #[test] + fn an_initializer_derived_bound_has_no_excess() { + // `int a[] = {1, 2, 3};` and a flexible array member are sized by the + // initializer, so every element belongs to the object. + let indices = kept_indices(None, |types| { + (1..=3).map(|v| positional(v, types)).collect() + }); + assert_eq!(indices, vec![0, 1, 2]); + let indices = kept_indices(Some(0), |types| { + (1..=3).map(|v| positional(v, types)).collect() + }); + assert_eq!(indices, vec![0, 1, 2]); + } + + /// The static image of `int a[2] = {1, 2, 3};` is two elements wide, not + /// three: the excess element used to become a third `.long` under a + /// two-element symbol, which the next symbol in the section absorbed. + #[test] + fn a_static_array_image_holds_no_excess_element() { + let target = Target::host(); + let mut types = TypeTable::new(&target); + let elements: Vec<_> = (1..=3).map(|v| positional(v, &types)).collect(); + let array = types.intern(Type::array(types.int_id, 2)); + let symbols = SymbolTable::new(); + let strings = crate::strings::StringTable::new(); + let mut lin = Linearizer::new(&symbols, &types, &strings, &target); + let init = lin.ast_init_list_to_ir(&elements, array); + let Initializer::Array { + elem_size, + total_size, + elements, + } = init + else { + panic!("an array initializer lowers to Initializer::Array, got {init:?}"); + }; + assert_eq!(elem_size, 4); + assert_eq!(total_size, 8); + assert_eq!( + elements + .iter() + .map(|(offset, init)| (*offset, init.clone())) + .collect::>(), + vec![(0, Initializer::Int(1)), (4, Initializer::Int(2))], + ); + } +} diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index 19519d6a8..a9e2ba782 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -975,7 +975,7 @@ impl<'a> super::linearize::Linearizer<'a> { TypeKind::Array | TypeKind::Struct | TypeKind::Union ); - let groups = self.group_array_init_elements(elements, elem_type); + let groups = self.group_array_init_elements(elements, typ); for element_index in groups.indices { let Some(list) = groups.element_lists.get(&element_index) else { continue; diff --git a/cc/tests/c99/initializers.rs b/cc/tests/c99/initializers.rs index 3220a78f0..4f452bd37 100644 --- a/cc/tests/c99/initializers.rs +++ b/cc/tests/c99/initializers.rs @@ -2544,3 +2544,96 @@ fn c99_complex_constants_in_scalar_static_initializers() { assert_eq!(rc, 0); } } + +/// An excess array initializer is discarded, not written past the object. +/// +/// C17 6.7.9p2 makes more initializers than elements a constraint violation; +/// c17 already diagnoses it. The grouping pass never bounded its element +/// cursor by the array size, so the extra value was still stored -- one element +/// past the end, on top of whatever the frame put there. `int x[2] = {7, 8};` +/// followed by `int a[2] = {1, 2, 3};` read back `x = {3, 8}`. +#[test] +fn c99_excess_array_initializers_do_not_write_past_the_object() { + let code = r#" +int main(void) +{ + int x[2] = {7, 8}; + int a[2] = {1, 2, 3}; + if (a[0] != 1 || a[1] != 2) return 1; + if (x[0] != 7 || x[1] != 8) return 2; + + /* Several excess elements, and a designator that jumps back first. + C17 6.7.9p17: a positional initializer after a designator resumes at + the next subobject, so after `[0] = 1` the cursor is at index 1 and the + 9 overrides the earlier 2. Only the 10 and 11 are excess. Confirmed + against clang, which warns -Winitializer-overrides on the 9. */ + short y[2] = {5, 6}; + short b[2] = {[1] = 2, [0] = 1, 9, 10, 11}; + if (b[0] != 1 || b[1] != 9) return 3; + if (y[0] != 5 || y[1] != 6) return 4; + + /* The bound is on the index, not on the count. Here the array has three + elements and the initializer list has two, so a count-based rule keeps + the 1 -- but it resumes after `[2]`, i.e. at index 3, and is excess. + clang gives {0,0,3}. */ + int guard_before[2] = {11, 12}; + int d[3] = {[2] = 3, 1}; + int guard_after[2] = {13, 14}; + if (d[0] != 0 || d[1] != 0 || d[2] != 3) return 10; + if (guard_before[0] != 11 || guard_before[1] != 12) return 11; + if (guard_after[0] != 13 || guard_after[1] != 14) return 12; + + /* A nested array: the excess belongs to the inner object. */ + int z[2] = {8, 9}; + int c[2][2] = {{1, 2, 3}, {4, 5}}; + if (c[0][0] != 1 || c[0][1] != 2) return 5; + if (c[1][0] != 4 || c[1][1] != 5) return 6; + if (z[0] != 8 || z[1] != 9) return 7; + + /* A char array from a string literal that does not fit: C17 6.7.9p14 + allows exactly the terminator to be dropped, nothing more. */ + char w[2] = {'a', 'b'}; + char s[3] = "hello"; + if (s[0] != 'h' || s[1] != 'e' || s[2] != 'l') return 8; + if (w[0] != 'a' || w[1] != 'b') return 9; + + return 0; +} +"#; + assert_eq!(compile_and_run("excess_array_init", code, &[]), 0); + assert_eq!(compile_and_run_optimized("excess_array_init_opt", code), 0); +} + +/// The static form of the same defect: the emitted object is exactly as wide +/// as the array declares. +/// +/// `int garr[2] = {1, 2, 3};` emitted three `.long`s under an eight-byte +/// object, so the next symbol in the section absorbed the third. +#[test] +fn c99_excess_static_array_initializers_do_not_widen_the_object() { + let code = r#" +int garr[2] = {1, 2, 3}; +int after = 42; +short garr2[2] = {[1] = 2, [0] = 1, 9, 10}; +short after2 = 7; +/* Bounded by index, not by count -- see the automatic case. */ +int garr3[3] = {[2] = 3, 1}; +int after3 = 5; + +int main(void) +{ + if (garr[0] != 1 || garr[1] != 2) return 1; + if (after != 42) return 2; + if (garr2[0] != 1 || garr2[1] != 9) return 3; + if (after2 != 7) return 4; + if (garr3[0] != 0 || garr3[1] != 0 || garr3[2] != 3) return 5; + if (after3 != 5) return 6; + return 0; +} +"#; + assert_eq!(compile_and_run("excess_static_array_init", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("excess_static_array_init_opt", code), + 0 + ); +} From 4b99687531ec33b08d1c920991363b2179a7dda2 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Tue, 29 Sep 2026 22:03:37 -0400 Subject: [PATCH 04/17] cc: mark volatile accesses on the instruction so DCE keeps them `Opcode::has_side_effects` lists `Store` but not `Load`, and `dce::is_root` was that predicate alone, so no load was ever a root and DCE deleted every `volatile` read whose value goes unused. `volatile int g; void f(void) { g; }` emitted the load at -O0 and nothing at all from -O1 up. Reading a volatile object is observable behaviour (C17 5.1.2.3p6). The qualifier could not be answered from the objects the IR already tracks. `LocalVar::is_volatile` and `memloc::GlobalFacts::is_volatile` describe a named object, and for `volatile int *p` there is no such object to ask: `p` is an ordinary pointer and `*p` is the volatile one. So the marker goes on the access -- `Instruction::is_volatile`, asked through `is_volatile_access`, printed as a trailing `volatile` in a dump, and cleared by `kill` so a `Nop` makes no stale claim. `ir/README.md` and a comment in `validate.rs` both documented the old design as deliberate and are corrected. `Linearizer::emit` sets it for every access the linearizer emits, from `types.contains_volatile` of the type that access reaches. Being the one chokepoint all of them pass through, that covers every `Instruction::load` site without touching any of them, and a site that knows more than the access type -- a bit-field's carrier, a composite copy's chunks -- can still set the marker itself and have it preserved. `ir::build::Builder` does the same for accesses a pass synthesizes, so no pass can introduce an unmarked one. `Store` remains a root by its opcode alone: its correctness must not come to depend on the marker. Four other passes needed the same question asked, each having declined a through-pointer volatile only incidentally, by resolving the base to `Unknown` rather than by rule: `loadfwd` will not forward one, `dse` will not delete one, `constglobal` will not answer one from an initializer, and `ssa` will not promote the object it reaches -- `int a; *(volatile int *)&a;` qualifies the access alone and was being rewritten into a register copy. `instcombine`, `sccp`, `mem2reg`, `ifconv`, `inline`, `memexpand`, `lower`, `propagate`, `constfold`, `vrp`, `escape` and `effects` were audited and need nothing. Covered on both targets at every -O level, with a plain read that must still be removed as the negative control, and a volatile store pair that passed before this change and must keep passing. Still wrong, and not fixed here: a member read of a volatile aggregate is dropped, because a member-access expression's type does not inherit the parent object's qualifiers (C17 6.5.2.3p3/p4). That belongs in the parser's type evaluation, not here. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/README.md | 22 +++- cc/ir/build.rs | 11 +- cc/ir/constglobal.rs | 8 ++ cc/ir/dce.rs | 131 +++++++++++++++++--- cc/ir/dse.rs | 5 +- cc/ir/linearize.rs | 26 ++++ cc/ir/loadfwd.rs | 8 ++ cc/ir/mod.rs | 44 +++++++ cc/ir/ssa.rs | 10 ++ cc/ir/test_linearize.rs | 106 +++++++++++++++++ cc/ir/validate.rs | 6 +- cc/tests/codegen/memopt.rs | 236 ++++++++++++++++++++++++++++++++++++- 12 files changed, 587 insertions(+), 26 deletions(-) diff --git a/cc/ir/README.md b/cc/ir/README.md index ece99e02a..fbb1e33e7 100644 --- a/cc/ir/README.md +++ b/cc/ir/README.md @@ -111,9 +111,25 @@ SSA-form intermediate representation for the c17 C17 compiler. Inspired by Linus | `symaddr` | Get address of symbol | For `load` and `store`, `offset` is a **byte** displacement and `size` is the -access width in **bits**. Neither carries a volatile or atomic marker: -volatility lives on the `LocalVar` or the `GlobalDef`, and an atomic access -has its own opcode. +access width in **bits**. An atomic access has its own opcode. + +Both carry a **volatile marker**, `Instruction::is_volatile`, printed as a +trailing `volatile` in a dump. Ask it through `Instruction::is_volatile_access`. +The qualifier has to live on the *access* because it is not always on any +object: for `volatile int *p`, `p` is an ordinary pointer and `*p` is the +volatile object, so `LocalVar::is_volatile` and +`memloc::GlobalFacts::is_volatile` — which answer only for a named object — +have nothing to say about it. Those two remain, and are still what a pass asks +about the object as a whole; the marker is what it asks about the access. + +`Linearizer::emit` sets the marker for every access the linearizer emits, from +`types.contains_volatile` of the type that access reaches, and +`ir::build::Builder` does the same for the accesses a pass synthesizes. Reading +a volatile object is observable behaviour (C17 5.1.2.3p6), so a marked access +survives every optimization level: `dce::is_root` treats it as a root, +`loadfwd` will not forward one or fold two into one, `dse` will not delete one, +`constglobal` will not answer one from an initializer, and `ssa` will not +promote the object it reaches out of memory. ### SSA Operations diff --git a/cc/ir/build.rs b/cc/ir/build.rs index af610757a..06092046c 100644 --- a/cc/ir/build.rs +++ b/cc/ir/build.rs @@ -51,15 +51,22 @@ impl<'a> Builder<'a> { } /// A load of `typ`, `size` bits wide, from `addr + at`. + /// + /// Marked volatile when `typ` is, for the same reason and by the same rule + /// as `Linearizer::mark_volatile_access`: an access this builds is as + /// observable as one the program wrote, and a pass must not be able to + /// introduce an unmarked access to a volatile object. pub(crate) fn load(&mut self, addr: PseudoId, at: i64, typ: TypeId, size: u32) -> PseudoId { let v = self.func.alloc_pseudo(); - self.push(Instruction::load(v, addr, at, typ, size)); + let vol = self.types.contains_volatile(typ); + self.push(Instruction::load(v, addr, at, typ, size).with_volatile(vol)); v } /// A store of `v`, of `typ` and `size` bits wide, to `addr + at`. pub(crate) fn store(&mut self, v: PseudoId, addr: PseudoId, at: i64, typ: TypeId, size: u32) { - self.push(Instruction::store(v, addr, at, typ, size)); + let vol = self.types.contains_volatile(typ); + self.push(Instruction::store(v, addr, at, typ, size).with_volatile(vol)); } /// A new integer constant of `typ` at `size` bits, with the `SetVal` diff --git a/cc/ir/constglobal.rs b/cc/ir/constglobal.rs index a0be9cd8d..33654a466 100644 --- a/cc/ir/constglobal.rs +++ b/cc/ir/constglobal.rs @@ -158,6 +158,14 @@ fn foldable_load( if insn.op != Opcode::Load || insn.src.len() != 1 || insn.offset != 0 { return None; } + // The access itself is observable, whatever the object's initializer says + // it holds: `const volatile int t = 0;` -- a hardware status word, a + // linker-set value -- must still be read. `qualifies` declines a global + // written `volatile`, but the qualifier can also be on the *access*, as in + // `*(volatile const int *)&t`, and only the instruction knows that. + if insn.is_volatile_access() { + return None; + } let PseudoKind::Sym(name) = &func.get_pseudo(insn.src[0])?.kind else { return None; }; diff --git a/cc/ir/dce.rs b/cc/ir/dce.rs index 6ce2180be..d1e3bf956 100644 --- a/cc/ir/dce.rs +++ b/cc/ir/dce.rs @@ -21,7 +21,7 @@ // reordering must consult `is_memory_barrier()` before crossing. // -use super::{BasicBlockId, Function, Opcode, PseudoId}; +use super::{BasicBlockId, Function, Instruction, Opcode, PseudoId}; use std::collections::{HashMap, HashSet, VecDeque}; const DEFAULT_LIVE_CAPACITY: usize = 64; @@ -50,9 +50,20 @@ pub fn run(func: &mut Function) -> bool { // Dead Code Elimination -/// Check if an opcode is a "root" (has side effects, cannot be deleted). -fn is_root(op: Opcode) -> bool { - op.has_side_effects() +/// Check if an instruction is a "root" (has side effects, cannot be deleted). +/// +/// Most of the answer is the opcode's, but not all of it: a `Load` is +/// deletable because reading an ordinary object has no effect, while reading a +/// `volatile` one is observable behaviour (C17 5.1.2.3p6) and must still +/// happen. That distinction is per *access*, not per opcode -- `*p` for a +/// `volatile int *p` is volatile and `*q` for an `int *q` is not -- so it is +/// asked of the instruction. Before this, `volatile int g; void f(void) { g; }` +/// emitted the load at `-O0` and nothing at all from `-O1` up. +/// +/// `Store` is a root by its opcode alone and stays that way: its correctness +/// must not come to depend on the marker. +fn is_root(insn: &Instruction) -> bool { + insn.op.has_side_effects() || insn.is_volatile_access() } /// Build a map from each pseudo to the instructions that define it. @@ -77,7 +88,7 @@ fn eliminate_dead_code(func: &mut Function) -> bool { // Phase 1: Mark roots and their operands as live for bb in &func.blocks { for insn in &bb.insns { - if is_root(insn.op) { + if is_root(insn) { // Mark all operands of root instructions as live for id in insn.uses() { if live.insert(id) { @@ -110,7 +121,7 @@ fn eliminate_dead_code(func: &mut Function) -> bool { for bb in &mut func.blocks { for insn in &mut bb.insns { // Skip roots - they're always live - if is_root(insn.op) { + if is_root(insn) { continue; } @@ -562,17 +573,103 @@ mod tests { #[test] fn test_is_root() { - assert!(is_root(Opcode::Ret)); - assert!(is_root(Opcode::Store)); - assert!(is_root(Opcode::Call)); - assert!(is_root(Opcode::Br)); - assert!(is_root(Opcode::Cbr)); - assert!(is_root(Opcode::Unreachable)); - - assert!(!is_root(Opcode::Add)); - assert!(!is_root(Opcode::Mul)); - assert!(!is_root(Opcode::Load)); - assert!(!is_root(Opcode::Phi)); + let bare = |op| is_root(&Instruction::new(op)); + + assert!(bare(Opcode::Ret)); + assert!(bare(Opcode::Store)); + assert!(bare(Opcode::Call)); + assert!(bare(Opcode::Br)); + assert!(bare(Opcode::Cbr)); + assert!(bare(Opcode::Unreachable)); + + assert!(!bare(Opcode::Add)); + assert!(!bare(Opcode::Mul)); + assert!(!bare(Opcode::Load)); + assert!(!bare(Opcode::Phi)); + } + + #[test] + fn test_volatile_load_is_root() { + // A plain load is deletable; the same load of a volatile object is not. + let plain = Instruction::new(Opcode::Load); + assert!(!is_root(&plain)); + + let vol = Instruction::new(Opcode::Load).with_volatile(true); + assert!(is_root(&vol), "reading a volatile object is observable"); + + // A volatile store is a root either way -- the marker must not be what + // its correctness rests on. + assert!(is_root( + &Instruction::new(Opcode::Store).with_volatile(true) + )); + assert!(is_root(&Instruction::new(Opcode::Store))); + } + + #[test] + fn test_volatile_load_with_dead_result_survives() { + // `volatile int g; void f(void) { g; }` -- the loaded value is never + // used, and DCE deleted the load outright before the marker existed. + let types = TypeTable::new(&Target::host()); + let mut func = Function::new("test", types.void_id); + + func.add_pseudo(Pseudo::reg(PseudoId(0), 0)); + func.add_pseudo(Pseudo::sym(PseudoId(1), "g".to_string())); + + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.add_insn(Instruction::new(Opcode::Entry)); + bb.add_insn( + Instruction::load(PseudoId(0), PseudoId(1), 0, types.int_id, 32).with_volatile(true), + ); + bb.add_insn(Instruction::ret(None)); + func.add_block(bb); + func.entry = BasicBlockId(0); + + assert!(!run(&mut func), "a volatile load is not dead code"); + assert_eq!(func.blocks[0].insns[1].op, Opcode::Load); + assert!(func.blocks[0].insns[1].is_volatile_access()); + } + + #[test] + fn test_volatile_load_keeps_its_address_live() { + // `volatile int *p; void f(void) { *p; }` -- the second load is the + // volatile access, and it is the only thing keeping the first (the + // read of `p` itself) alive. + let types = TypeTable::new(&Target::host()); + let mut func = Function::new("test", types.void_id); + + func.add_pseudo(Pseudo::sym(PseudoId(0), "p".to_string())); + func.add_pseudo(Pseudo::reg(PseudoId(1), 1)); + func.add_pseudo(Pseudo::reg(PseudoId(2), 2)); + + let ptr = types.pointer_to(types.int_id); + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.add_insn(Instruction::new(Opcode::Entry)); + // %1 = load p (plain: reading the pointer variable) + bb.add_insn(Instruction::load(PseudoId(1), PseudoId(0), 0, ptr, 64)); + // %2 = load *%1 (volatile: reading the pointed-to object) + bb.add_insn( + Instruction::load(PseudoId(2), PseudoId(1), 0, types.int_id, 32).with_volatile(true), + ); + bb.add_insn(Instruction::ret(None)); + func.add_block(bb); + func.entry = BasicBlockId(0); + + assert!(!run(&mut func), "neither load may be deleted"); + assert_eq!(func.blocks[0].insns[1].op, Opcode::Load); + assert_eq!(func.blocks[0].insns[2].op, Opcode::Load); + } + + #[test] + fn test_kill_clears_the_volatile_marker() { + // `kill` makes a `Nop`, which reaches no memory: a marker left behind + // would be a stale claim to any pass reading the field directly. + let types = TypeTable::new(&Target::host()); + let mut insn = + Instruction::load(PseudoId(0), PseudoId(1), 0, types.int_id, 32).with_volatile(true); + insn.kill(); + assert_eq!(insn.op, Opcode::Nop); + assert!(!insn.is_volatile); + assert!(!insn.is_volatile_access()); } #[test] diff --git a/cc/ir/dse.rs b/cc/ir/dse.rs index 641fc0f54..1982a86a0 100644 --- a/cc/ir/dse.rs +++ b/cc/ir/dse.rs @@ -111,7 +111,10 @@ fn scan_block( continue; } let loc = am.location_of(func, insn); - if !deletable(func, types, mi, &loc) { + // A volatile store is observable and stays, even when a later store + // overwrites every byte of it. `deletable` answers for a named object; + // the marker also answers for `*p` where `p` is a `volatile int *`. + if insn.is_volatile_access() || !deletable(func, types, mi, &loc) { // An untrackable store is still a write: drop whatever it may // have touched rather than pretending it did not happen. pending.retain(|p| !may_alias(&loc, &p.loc, mi)); diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index 8c7012081..578187394 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -729,6 +729,7 @@ impl<'a> Linearizer<'a> { /// Add an instruction to the current basic block pub(crate) fn emit(&mut self, insn: Instruction) { + let insn = self.mark_volatile_access(insn); let insn = self.displacement_in_range(insn); if let Some(bb_id) = self.current_bb { // Attach current source position for debug info @@ -741,6 +742,31 @@ impl<'a> Linearizer<'a> { bb.add_insn(insn); } } + + /// Mark an access to a `volatile` object as one, from the type it reaches. + /// + /// Every `Load` and `Store` the linearizer emits passes through + /// [`Self::emit`], and each carries in `typ` the type of the object it is + /// accessing -- so this is the one place the qualifier has to be read, and + /// the one place it can be read for *every* access, including `*p` for a + /// `volatile int *p`, where there is no variable holding the qualifier to + /// ask (which is why `LocalVar::is_volatile` alone let DCE delete every + /// discarded `volatile` read from `-O1` up). + /// + /// A marker a site set itself is kept rather than recomputed, so a site + /// that knows more than the access type does can say so: a bit-field reads + /// a storage unit whose type is the carrier, and a composite copy reads + /// integer chunks, neither of which is the qualified type. + fn mark_volatile_access(&self, mut insn: Instruction) -> Instruction { + if !matches!(insn.op, Opcode::Load | Opcode::Store) || insn.is_volatile { + return insn; + } + if let Some(typ) = insn.typ { + insn.is_volatile = self.types.contains_volatile(typ); + } + insn + } + /// Keep a load's or store's constant offset inside a machine displacement. /// /// Both backends address `src[0] + offset` with a signed 32-bit diff --git a/cc/ir/loadfwd.rs b/cc/ir/loadfwd.rs index ebe6b37b1..3edfbdbe1 100644 --- a/cc/ir/loadfwd.rs +++ b/cc/ir/loadfwd.rs @@ -89,6 +89,14 @@ pub(crate) fn run(func: &mut Function, types: &TypeTable, mi: &ModuleInfo) -> bo if insn.op != Opcode::Load { continue; } + // Each read of a volatile object is its own observable event, so + // the value another access left behind is no answer for this one. + // `forwardable` declines a *named* volatile object; the marker is + // what declines `*p` for a `volatile int *p`, where the qualifier + // is on the access and there is no variable to ask. + if insn.is_volatile_access() { + continue; + } if let Some(a) = oracle.value_at((b, i), &am.location_of(func, insn)) { sites.push((b, i, a)); } diff --git a/cc/ir/mod.rs b/cc/ir/mod.rs index fc622480a..95299dc4c 100644 --- a/cc/ir/mod.rs +++ b/cc/ir/mod.rs @@ -973,6 +973,20 @@ pub struct Instruction { pub abi_info: Option>, /// For atomic operations: memory ordering constraint pub memory_order: MemoryOrder, + /// For `Load` and `Store`: the object being accessed is `volatile`, so the + /// access itself is observable behaviour (C17 5.1.2.3p6) and no pass may + /// delete, merge, move or fold it. + /// + /// The qualifier lives on the *access*, not on the variable, because for + /// `volatile int *p` there is no variable to ask: `p` is an ordinary + /// pointer and `*p` is the volatile object. `LocalVar::is_volatile` and + /// `memloc::GlobalFacts::is_volatile` answer only for a named object, so + /// DCE saw nothing to stop it and deleted every discarded `volatile` read + /// from `-O1` up. Ask through [`Instruction::is_volatile_access`]. + /// + /// Set for every access the linearizer emits, from the type it is + /// accessing, in `Linearizer::mark_volatile_access`. + pub is_volatile: bool, } impl Default for Instruction { @@ -1003,6 +1017,7 @@ impl Default for Instruction { asm_data: None, abi_info: None, memory_order: MemoryOrder::default(), + is_volatile: false, } } } @@ -1054,6 +1069,28 @@ impl Instruction { self } + /// Mark this `Load` or `Store` as an access to a `volatile` object. + pub fn with_volatile(mut self, is_volatile: bool) -> Self { + debug_assert!( + !is_volatile || matches!(self.op, Opcode::Load | Opcode::Store), + "only a Load or a Store carries the volatile marker" + ); + self.is_volatile = is_volatile; + self + } + + /// Is this an access to a `volatile` object? + /// + /// Reading or writing one is observable behaviour (C17 5.1.2.3p6), so an + /// access that answers `true` survives every optimization level: no pass + /// may delete it, fold it to a constant, merge it with another access, or + /// promote the object it reaches out of memory. This is the question to + /// ask; `is_volatile` is only where the answer is stored, and is true of + /// nothing but a `Load` or a `Store`. + pub fn is_volatile_access(&self) -> bool { + self.is_volatile && matches!(self.op, Opcode::Load | Opcode::Store) + } + /// Set the type (caller should also call with_size if needed) pub fn with_type(mut self, typ: TypeId) -> Self { self.typ = Some(typ); @@ -1548,6 +1585,10 @@ impl Instruction { self.src.clear(); self.target = None; self.phi_list.clear(); + // A `Nop` reaches no memory, so it is no longer a volatile access -- + // and leaving the marker set on one would make a stale claim to any + // pass that asks the field rather than `is_volatile_access`. + self.is_volatile = false; } } @@ -1729,6 +1770,9 @@ impl fmt::Display for InstructionDisplay<'_> { if this.offset != 0 { write!(f, " + {}", this.offset)?; } + if this.is_volatile { + write!(f, " volatile")?; + } } _ => { for (i, src) in this.src.iter().enumerate() { diff --git a/cc/ir/ssa.rs b/cc/ir/ssa.rs index e4dd79d48..d501e1b02 100644 --- a/cc/ir/ssa.rs +++ b/cc/ir/ssa.rs @@ -195,6 +195,16 @@ fn analyze_variables(func: &Function, types: &TypeTable) -> HashMap Vec<(Opcode, bool)> { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.is_volatile_access())) + .collect() + }; + + // A named volatile object: one marked access each way. + assert_eq!(accesses("read_named"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("write_named"), vec![(Opcode::Store, true)]); + + // Through a pointer *to* volatile, the qualifier is on the pointee, so + // reading `vp` itself is plain and the access through it is volatile. + assert_eq!( + accesses("read_via_ptr"), + vec![(Opcode::Load, false), (Opcode::Load, true)] + ); + assert_eq!( + accesses("write_via_ptr"), + vec![(Opcode::Load, false), (Opcode::Store, true)] + ); + + // The qualifier reaches through an array's element type and a member's + // own type. + assert_eq!(accesses("read_element"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_member"), vec![(Opcode::Load, true)]); + + // And nothing unqualified is marked -- the marker that says "keep this" + // is worth nothing if it is on every access. + assert_eq!(accesses("read_plain"), vec![(Opcode::Load, false)]); + assert_eq!( + accesses("read_plain_ptr"), + vec![(Opcode::Load, false), (Opcode::Load, false)] + ); +} + +/// A `volatile` access keeps the object it reaches in memory: promotion would +/// rewrite the access into a register `Copy`, and the access must happen. +/// +/// The variable need not itself be volatile. `ssa` tests +/// `LocalVar::is_volatile`, which answers no here -- the qualifier is on the +/// access alone. +#[test] +fn test_volatile_access_to_a_plain_local_blocks_promotion() { + let target = Target::host(); + let src = "int f(void) { int a = 1; return *(volatile int *)&a; }\n"; + let module = linearize_source(src, &target); + let mut func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("f") + .clone(); + + let volatile_loads = |func: &Function| -> usize { + func.blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| i.is_volatile_access()) + .count() + }; + assert_eq!(volatile_loads(&func), 1, "the cast qualifies the access"); + + let types = TypeTable::new(&target); + crate::ir::ssa::ssa_convert(&mut func, &types); + assert_eq!( + volatile_loads(&func), + 1, + "SSA promotion turned a volatile access into a register copy" + ); +} diff --git a/cc/ir/validate.rs b/cc/ir/validate.rs index 51e19f797..4a4673ae8 100644 --- a/cc/ir/validate.rs +++ b/cc/ir/validate.rs @@ -429,8 +429,10 @@ fn check_barrier_implies_side_effect(func: &Function, out: &mut Vec) { for (block, bb) in func.blocks.iter().enumerate() { for (index, insn) in bb.insns.iter().enumerate() { diff --git a/cc/tests/codegen/memopt.rs b/cc/tests/codegen/memopt.rs index 165a0efc7..fc9fc95c7 100644 --- a/cc/tests/codegen/memopt.rs +++ b/cc/tests/codegen/memopt.rs @@ -15,7 +15,10 @@ // if the pass forwards one byte it should not have. // -use crate::codegen::asm_probe::{asm_for_with, body_of, AARCH64_LINUX, X86_64_LINUX}; +use crate::codegen::asm_probe::{ + asm_for_with, assert_body_contains, assert_body_lacks, body_of, count_in_body, AARCH64_LINUX, + X86_64_LINUX, +}; use crate::common::{compile_and_run, compile_and_run_aarch64, compile_and_run_two_units, run_c17}; fn at_o2(name: &str, code: &str) -> i32 { @@ -1264,3 +1267,234 @@ fn codegen_no_composite_is_read_wider_than_itself() { } } } + +/// Reading a `volatile` object is an observable side effect, so the access +/// survives every optimization level -- including a read whose value is +/// discarded, which no data-flow fact keeps alive (C17 5.1.2.3). +/// +/// The property is on the access, not on the result, so a discarded read has +/// nothing an exit status can see. The check is on the emitted instruction, +/// against both targets, because the rule is architecture-independent. +#[test] +fn memopt_a_discarded_volatile_read_is_still_performed() { + // The object names are deliberately unmistakable. A single letter is not a + // sound needle here: every x86-64 body contains `pushq`/`popq` and every + // aarch64 body contains `stp`/`sp`, and `.cfi_startproc` is inside the + // range `body_of` returns -- so searching for "p" passes against a body + // that was emptied, which is exactly the defect. (No empty body on either + // target contains a "g", which is why the other cases were sound.) + let cases = [ + ( + "assign", + "volatile int volobj;\nvoid probe(void) { int a = volobj; (void)a; }\n", + "volobj", + ), + ( + "discard", + "volatile int volobj;\nvoid probe(void) { volobj; }\n", + "volobj", + ), + ( + "via_ptr", + "volatile int *volptr;\nvoid probe(void) { *volptr; }\n", + "volptr", + ), + ( + "cast_void", + "volatile int volobj;\nvoid probe(void) { (void)volobj; }\n", + "volobj", + ), + ]; + + for (tag, src, object) in cases { + for level in ["-O0", "-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("vol_{tag}"), triple, src, &[level]); + assert_body_contains( + &asm, + "probe", + object, + &format!( + "a volatile read is observable: `{tag}` at {level} on {triple} \ + must still access `{object}`" + ), + ); + + // Naming the pointer is not the same as dereferencing it, and + // the qualifier here is on the pointee, so the load through it + // is the access under test. + if tag == "via_ptr" { + let indirect = if triple == X86_64_LINUX { "(%r" } else { "[x" }; + assert_body_contains( + &asm, + "probe", + indirect, + &format!( + "the volatile pointee is read, not just the pointer: \ + {level} on {triple}" + ), + ); + } + } + } + } +} + +/// The counterpart that keeps the fix above honest: an *ordinary* discarded +/// read is still dead code, and DCE still deletes it. +/// +/// Without this, marking every load a root would pass the volatile test. +#[test] +fn memopt_a_discarded_plain_read_is_still_removed() { + // Object names chosen to occur in no mnemonic, register or label the + // body can otherwise contain -- `popq` alone contains both `p` and `pq`, + // and the body a negative assertion searches includes the function's own + // label and prologue. + let cases = [ + ( + "assign", + "int objx;\nvoid probe(void) { int a = objx; (void)a; }\n", + "objx", + ), + ("discard", "int objx;\nvoid probe(void) { objx; }\n", "objx"), + ( + "via_ptr", + "int *ptrx;\nvoid probe(void) { *ptrx; }\n", + "ptrx", + ), + ]; + + for (tag, src, object) in cases { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("plain_{tag}"), triple, src, &["-O2"]); + assert_body_lacks( + &asm, + "probe", + object, + &format!( + "reading a non-volatile object has no effect: `{tag}` on {triple} \ + must not access `{object}`" + ), + ); + } + } +} + +/// Each read of a `volatile` object is its own observable event, so two of +/// them are two accesses -- neither load-forwarding nor DCE may fold the pair +/// into one. +#[test] +fn memopt_two_volatile_reads_are_both_performed() { + // A named object only: two reads through one `volatile int *p` show up as + // a *single* reference to `p` -- reading the pointer itself is not + // volatile and is rightly done once -- so the count says nothing there. + // The through-pointer case is pinned at the IR level instead, by + // `test_volatile_accesses_carry_the_marker` and the `dce` unit tests. + // + // The name occurs in no mnemonic, register or label the body can + // otherwise contain: `popq` alone contains both `p` and `pq`. + let cases = [( + "named", + "volatile int objx;\nint sink(int, int);\n\ + int probe(void) { int a = objx; int b = objx; return sink(a, b); }\n", + "objx", + )]; + + for (tag, src, object) in cases { + for level in ["-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("vol_two_reads_{tag}"), triple, src, &[level]); + let n = count_in_body(&asm, "probe", object); + assert!( + n >= 2, + "both volatile reads are observable: `{tag}` at {level} on {triple} \ + kept {n} reference(s) to `{object}`:\n{}", + body_of(&asm, "probe") + ); + } + } + } +} + +/// An `_Atomic` read is observable for the same reason, and reaches DCE by a +/// different route: `AtomicLoad` is a side-effecting opcode outright, so this +/// cross-checks that the two spellings of "this read must happen" agree. +#[test] +fn memopt_a_discarded_atomic_read_is_still_performed() { + let cases = [ + ("discard", "_Atomic int g;\nvoid probe(void) { g; }\n", "g"), + ( + "assign", + "_Atomic int g;\nvoid probe(void) { int a = g; (void)a; }\n", + "g", + ), + ]; + + for (tag, src, object) in cases { + for level in ["-O0", "-O2"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with(&format!("atomic_{tag}"), triple, src, &[level]); + assert_body_contains( + &asm, + "probe", + object, + &format!( + "an atomic read is observable: `{tag}` at {level} on {triple} \ + must still access `{object}`" + ), + ); + } + } + } +} + +/// A `volatile` read inside a loop happens once per iteration: the value is +/// not a loop invariant, whatever the compiler can see written to the object. +/// +/// Nothing in c17 hoists memory out of a loop today (see the ordering contract +/// in `cc/ir/dce.rs`), so this passes by construction -- it exists to fail the +/// day something does, because the exit status is where that would show up. +#[test] +fn memopt_a_volatile_read_in_a_loop_is_repeated() { + // `g` changes between iterations through a pointer the loop writes, so a + // read hoisted to the top would sum 7 three times instead of 7 + 10 + 11. + let code = r#" +volatile int g; +static int *alias(void) { return (int *)&g; } +int main(void) { + int sum = 0; + *alias() = 7; + for (int i = 0; i < 3; i++) { + sum += g; + *alias() = 10 + i; + } + return sum == 28 ? 0 : 1; +} +"#; + assert_eq!(at_o2("memopt_volatile_in_loop", code), 0); + if let Some(rc) = compile_and_run_aarch64("memopt_volatile_in_loop_a64", code, "-O2") { + assert_eq!(rc, 0, "aarch64 at -O2"); + } +} + +/// A `volatile` store is observable for the same reason, and DSE must not drop +/// the earlier of two writes to one. +/// +/// The companion to the read case above: a test that only checked loads would +/// pass against an `has_side_effects` that named `Store` and not `Load`. +#[test] +fn memopt_two_volatile_stores_are_both_performed() { + let src = "volatile int g;\nvoid probe(void) { g = 1; g = 2; }\n"; + for level in ["-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_two_stores", triple, src, &[level]); + let n = count_in_body(&asm, "probe", "g"); + assert!( + n >= 2, + "both volatile stores are observable: {level} on {triple} kept {n} \ + reference(s) to `g`:\n{}", + body_of(&asm, "probe") + ); + } + } +} From 0738a07f4b8baf08d193d3ea747b5e23e2b56486 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 00:04:44 -0400 Subject: [PATCH 05/17] cc: qualify a member access by the object that holds it C17 6.5.2.3p3/p4 gives `s.m` and `p->m` the *so-qualified* version of the member's type: it inherits `const` and `volatile` from the object. c17 used the member's declared type unchanged, so a member of a qualified object was an ordinary one. Two consequences, both silent: volatile struct S vs; vs.a; /* load deleted from -O1 up */ const struct S cs; cs.a = 2; /* accepted; gcc and clang error */ The read is observable behaviour (5.1.2.3) and the write is a constraint violation (6.5.16p2). Neither `LocalVar::is_volatile` nor `GlobalFacts::is_volatile` could answer for it, because both describe a whole named object while the access is of one member. `TypeTable` gains the operation and its inverse: `qualifiers`, `qualified_with` (the so-qualified version) and `unqualified` (6.3.2.1p2, which `lvalue_converted_type` becomes a wrapper over). `qualified_with` implements 6.7.3p10 -- qualifying an array qualifies its *element* type -- without which `cs.arr[0] = 2` stays accepted, since the subscript reads the element type. Four open-coded copies of the qualifier list collapse into `Type::QUALIFIERS`, and `Type::MEMBER_QUALIFIERS` records why `_Atomic` does not travel to a member (gcc agrees, and `warn_atomic_member_access` already reports reaching into one) and why `restrict` cannot. The rule lives in `member_access_type`, beside `find_member` rather than in it: qualifying interns, which needs `&mut TypeTable`, and the linearizer holds `&TypeTable` -- so a `&mut` `find_member` would thread mutability through the whole IR. Both member arms go through the new function, and `find_member` now documents that its type is the *declared* one. Its other twelve callers want an offset, a size or a bit-field width, for which that is correct. Two sites the qualifier cannot reach on its own: - `emit_member_access` built the load from `find_member`'s type, so it kept the unqualified one however good the expression's type was. It now performs the access at the expression's type; width, sign and kind still come from the member. - Bit-fields access their *carrier*, which can never hold the qualifier. `mark_volatile_access` already anticipated this and preserves a marker a site sets itself; neither emitter set one, so a `volatile` bit-field read was deleted at -O1 in both spellings. `emit_bitfield_store` now takes the field type so the read-modify-write marks both halves. `is_pure_expr` asked only whether a member's *base* was pure, so `c ? s.status : s.other` loaded both members unconditionally into a branchless select, at -O0 too -- 6.5.15p4 evaluates one arm. Its `Ident` arm had the same blind spot one level up, asking the top-level modifier where a struct with a volatile member needs `contains_volatile`; `constglobal`'s gate is respelled the same way, where it was unreachable but inconsistent. Making member types qualified would otherwise have leaked the qualifier into rvalue types, because `common_type` and `integer_promote` answer with an operand's own id and `volatile int + int` came out `volatile int`. The parser's three arithmetic chokepoints now strip it (6.3.2.1p2), which keeps the qualifier on exactly the lvalues that should carry it. Nothing in the test corpus asserted a write through a const-qualified aggregate, so no existing expectation changed. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/constglobal.rs | 11 +- cc/ir/linearize.rs | 54 +++++-- cc/ir/linearize_emit.rs | 106 +++++++++---- cc/ir/linearize_stmt.rs | 1 + cc/ir/test_linearize.rs | 173 +++++++++++++++++++++ cc/parse/expression.rs | 55 ++++--- cc/parse/test_parser.rs | 105 +++++++++++++ cc/tests/codegen/memopt.rs | 184 ++++++++++++++++++++++ cc/tests/diagnostics/mod.rs | 87 +++++++++++ cc/types.rs | 295 +++++++++++++++++++++++++++++++++--- 10 files changed, 983 insertions(+), 88 deletions(-) diff --git a/cc/ir/constglobal.rs b/cc/ir/constglobal.rs index 33654a466..cdf8a17bd 100644 --- a/cc/ir/constglobal.rs +++ b/cc/ir/constglobal.rs @@ -30,7 +30,7 @@ // use super::{ConstValue, Function, Initializer, Instruction, Module, Opcode, PseudoKind}; -use crate::types::{TypeId, TypeModifiers, TypeTable}; +use crate::types::{TypeId, TypeTable}; use std::collections::HashMap; /// A global whose value is known for the whole run. @@ -90,7 +90,13 @@ pub(crate) fn qualifies(g: &super::GlobalDef, types: &TypeTable) -> bool { } // `volatile` says the value can change for reasons not in the program, // which is exactly the assumption being made here. - if types.modifiers(g.typ).contains(TypeModifiers::VOLATILE) { + // + // `contains_volatile`, not the top-level modifier: a `const struct` with a + // `volatile` member is one of these objects too, and asking only what was + // written on the struct let it through the gate. Every access is checked + // again below, so this was not reachable as a wrong fold -- but the object + // and its members are one question and get one spelling of it. + if types.contains_volatile(g.typ) { return false; } // A weak definition exists to be replaced at link time, and the @@ -194,6 +200,7 @@ mod tests { use super::*; use crate::ir::{BasicBlock, BasicBlockId, GlobalDef, Pseudo, PseudoId}; use crate::target::Target; + use crate::types::TypeModifiers; /// A module with one global and a function that loads it whole. fn module_loading(global: GlobalDef, types: &TypeTable, load_typ: TypeId) -> Module { diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index 578187394..ef667cec2 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -1923,14 +1923,15 @@ impl<'a> Linearizer<'a> { | ExprKind::Utf16StringLit(_) | ExprKind::Utf32StringLit(_) => true, - // Identifiers are pure unless volatile - ExprKind::Ident(_) => { - if let Some(typ) = expr.typ { - !self.types.modifiers(typ).contains(TypeModifiers::VOLATILE) - } else { - true - } - } + // Identifiers are pure unless volatile. + // + // `contains_volatile`, not the top-level modifier: reading a + // struct with a `volatile` member reads that member, and asking + // only what was written on the struct answered no. + ExprKind::Ident(_) => match expr.typ { + Some(typ) => !self.types.contains_volatile(typ), + None => true, + }, // __func__ is a pure string-like value ExprKind::FuncName => true, @@ -1977,8 +1978,21 @@ impl<'a> Linearizer<'a> { // Function calls are never pure (may have side effects) ExprKind::Call { .. } => false, - // Member access through struct value (.) is pure if the base is pure. - ExprKind::Member { expr, .. } => self.is_pure_expr(expr), + // Member access through struct value (.) is pure if the base is + // pure and the member itself is not volatile. C17 6.5.15p4 + // evaluates only one arm of a conditional and 5.1.2.3 makes each + // volatile read an observable event, so speculating one is a read + // the program never asked for: asking about the base alone let + // `c ? s.status : s.other` load both members unconditionally into + // a branchless select, at `-O0` too. The member's type carries the + // object's qualifiers (C17 6.5.2.3p3), so this covers a volatile + // member and a member of a volatile object alike. + ExprKind::Member { expr: base, .. } => { + !expr + .typ + .is_some_and(|typ| self.types.contains_volatile(typ)) + && self.is_pure_expr(base) + } // Arrow access (ptr->member) can cause UB/crash if ptr is NULL, // so we must not eagerly evaluate it in conditional expressions. @@ -2792,19 +2806,31 @@ impl<'a> Linearizer<'a> { /// Shared logic for member access (both `.` and `->`). /// `base` is the address of the struct (for `.`) or the pointer value (for `->`). + /// + /// `access_typ` is the type of the member-access *expression*, which the + /// parser formed as the member's declared type so-qualified by the object + /// (C17 6.5.2.3p3/p4). The access is performed at that type, so + /// [`Self::mark_volatile_access`] sees the qualifier -- the type + /// `find_member` answers with is the member's *declared* one and cannot + /// carry it, which is why a member of a `volatile` struct read as an + /// ordinary `int` and DCE deleted the load from `-O1` up. Its width, sign + /// and kind still come from the member, so the two disagreeing (only + /// reachable once the parser has already reported an unknown member) + /// cannot change how the access is performed. It also stands in for the + /// member type entirely when the lookup fails here. pub(crate) fn emit_member_access( &mut self, base: PseudoId, struct_type: TypeId, member: StringId, - fallback_type: TypeId, + access_typ: TypeId, ) -> PseudoId { let member_info = self .types .find_member(struct_type, member) .unwrap_or(MemberInfo { offset: 0, - typ: fallback_type, + typ: access_typ, bit_offset: None, bit_width: None, access_bytes: None, @@ -2839,7 +2865,7 @@ impl<'a> Linearizer<'a> { bit_offset, bit_width, storage_size, - member_info.typ, + access_typ, ) } else { let size = self.types.size_bits(member_info.typ); @@ -2869,7 +2895,7 @@ impl<'a> Linearizer<'a> { result, base, member_info.offset as i64, - member_info.typ, + access_typ, size, )); result diff --git a/cc/ir/linearize_emit.rs b/cc/ir/linearize_emit.rs index 95624190d..1e9e0b1a5 100644 --- a/cc/ir/linearize_emit.rs +++ b/cc/ir/linearize_emit.rs @@ -337,6 +337,15 @@ impl<'a> super::linearize::Linearizer<'a> { /// Emit code to load a bitfield value /// Returns the loaded value as a PseudoId + /// + /// `typ` is the field's type as the access reaches it -- the declared type + /// so-qualified by the object (C17 6.5.2.3p3) -- and the access itself is + /// of the *carrier*, whose type is an unqualified storage unit. So nothing + /// downstream can derive the qualifier from the instruction's own type, and + /// the volatile marker is set here instead; + /// [`Self::mark_volatile_access`] preserves a marker its caller set for + /// exactly this case. Without it a volatile bit-field read was deleted + /// outright from `-O1` up. pub(crate) fn emit_bitfield_load( &mut self, base: PseudoId, @@ -370,16 +379,20 @@ impl<'a> super::linearize::Linearizer<'a> { // Determine storage type based on storage unit size let storage_type = self.bitfield_storage_type(storage_size as usize); let storage_bits = storage_size * 8; + let volatile = self.types.contains_volatile(typ); // 1. Load the entire storage unit let storage_val = self.alloc_pseudo(); - self.emit(Instruction::load( - storage_val, - base, - byte_offset as i64, - storage_type, - storage_bits, - )); + self.emit( + Instruction::load( + storage_val, + base, + byte_offset as i64, + storage_type, + storage_bits, + ) + .with_volatile(volatile), + ); // 2. Shift right by bit_offset (using logical shift for unsigned extraction) let shifted = if bit_offset > 0 { @@ -528,6 +541,8 @@ impl<'a> super::linearize::Linearizer<'a> { }; let carrier_bits = if wide { 64 } else { 32 }; let byte_type = self.types.uchar_id; + // Every byte of a volatile field is part of the one observable read. + let volatile = self.types.contains_volatile(typ); let mut acc: Option = None; let (field_lo, field_hi) = (bit_offset, bit_offset + bit_width); @@ -542,13 +557,10 @@ impl<'a> super::linearize::Linearizer<'a> { continue; } let byte = self.alloc_pseudo(); - self.emit(Instruction::load( - byte, - base, - (byte_offset + i as usize) as i64, - byte_type, - 8, - )); + self.emit( + Instruction::load(byte, base, (byte_offset + i as usize) as i64, byte_type, 8) + .with_volatile(volatile), + ); // Widen before shifting, or the shift is done at eight bits and // drops everything it moves. `uchar` is unsigned, so this is a // zero-extension and the byte's own value is preserved. @@ -662,6 +674,14 @@ impl<'a> super::linearize::Linearizer<'a> { } /// Emit code to store a value into a bitfield + /// + /// `typ` is the field's type as the access reaches it, and is here for the + /// same reason as in [`Self::emit_bitfield_load`]: the read-modify-write is + /// performed on the *carrier*, so the instructions cannot show the + /// qualifier and the marker is set from the field's type instead. Both + /// halves are marked -- the read of the storage unit is as observable as + /// the write of it. + #[allow(clippy::too_many_arguments)] pub(crate) fn emit_bitfield_store( &mut self, base: PseudoId, @@ -670,6 +690,7 @@ impl<'a> super::linearize::Linearizer<'a> { bit_width: u32, storage_size: u32, new_value: PseudoId, + typ: TypeId, ) { if !matches!(storage_size, 1 | 2 | 4 | 8 | 16) { return self.emit_bitfield_store_bytewise( @@ -679,22 +700,27 @@ impl<'a> super::linearize::Linearizer<'a> { bit_width, storage_size, new_value, + typ, ); } // Determine storage type based on storage unit size let storage_type = self.bitfield_storage_type(storage_size as usize); let storage_bits = storage_size * 8; + let volatile = self.types.contains_volatile(typ); // 1. Load current storage unit value let old_val = self.alloc_pseudo(); - self.emit(Instruction::load( - old_val, - base, - byte_offset as i64, - storage_type, - storage_bits, - )); + self.emit( + Instruction::load( + old_val, + base, + byte_offset as i64, + storage_type, + storage_bits, + ) + .with_volatile(volatile), + ); // 2. Create mask for the bitfield bits: ~(((1 << width) - 1) << offset) // @@ -757,13 +783,16 @@ impl<'a> super::linearize::Linearizer<'a> { )); // 6. Store back - self.emit(Instruction::store( - combined, - base, - byte_offset as i64, - storage_type, - storage_bits, - )); + self.emit( + Instruction::store( + combined, + base, + byte_offset as i64, + storage_type, + storage_bits, + ) + .with_volatile(volatile), + ); } /// Write a bit-field occupying an arbitrary byte range, one byte at a time. @@ -773,6 +802,7 @@ impl<'a> super::linearize::Linearizer<'a> { /// bits survive. Neither ever touches a byte outside the field's own span, /// which is what a wide read-modify-write could not promise: the span may /// end at the last byte of the object. + #[allow(clippy::too_many_arguments)] fn emit_bitfield_store_bytewise( &mut self, base: PseudoId, @@ -781,6 +811,7 @@ impl<'a> super::linearize::Linearizer<'a> { bit_width: u32, span: u32, new_value: PseudoId, + typ: TypeId, ) { let wide = bit_offset + bit_width > 32; let carrier = if wide { @@ -790,6 +821,7 @@ impl<'a> super::linearize::Linearizer<'a> { }; let carrier_bits = if wide { 64 } else { 32 }; let byte_type = self.types.uchar_id; + let volatile = self.types.contains_volatile(typ); // The value, masked to its width once, so no byte can contribute bits // the field does not have. @@ -861,7 +893,9 @@ impl<'a> super::linearize::Linearizer<'a> { placed } else { let old = self.alloc_pseudo(); - self.emit(Instruction::load(old, base, addr_off, byte_type, 8)); + self.emit( + Instruction::load(old, base, addr_off, byte_type, 8).with_volatile(volatile), + ); let keep = self.emit_const((!byte_mask & 0xff) as i128, byte_type); let cleared = self.alloc_pseudo(); self.emit(Instruction::binop( @@ -896,7 +930,9 @@ impl<'a> super::linearize::Linearizer<'a> { )); out }; - self.emit(Instruction::store(to_store, base, addr_off, byte_type, 8)); + self.emit( + Instruction::store(to_store, base, addr_off, byte_type, 8).with_volatile(volatile), + ); } } @@ -2500,8 +2536,12 @@ impl<'a> super::linearize::Linearizer<'a> { access_bytes: None, }); let bitfield = match (info.bit_offset, info.bit_width, info.access_bytes) { + // The target expression's type, not the member's declared one: + // they name the same width and sign, and only the expression's + // carries the object's qualifiers (C17 6.5.2.3p3), which is what + // tells the bit-field emitters that the access is volatile. (Some(bit_offset), Some(bit_width), Some(storage)) => { - Some((info.offset, bit_offset, bit_width, storage, info.typ)) + Some((info.offset, bit_offset, bit_width, storage, target_typ)) } // Not a bit-field: fold the member offset into the base so the // load and the store share one address. @@ -2559,7 +2599,9 @@ impl<'a> super::linearize::Linearizer<'a> { typ: TypeId, ) -> Option<(u32, TypeId)> { if let Some((offset, bit_offset, bit_width, storage, field_typ)) = place.bitfield { - self.emit_bitfield_store(place.base, offset, bit_offset, bit_width, storage, val); + self.emit_bitfield_store( + place.base, offset, bit_offset, bit_width, storage, val, field_typ, + ); return Some((bit_width, field_typ)); } let size = self.types.size_bits(typ); diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index a9e2ba782..d27e991d6 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -1119,6 +1119,7 @@ impl<'a> super::linearize::Linearizer<'a> { bit_w, storage_size, member_val, + field_type, ); } else { self.linearize_struct_field_init( diff --git a/cc/ir/test_linearize.rs b/cc/ir/test_linearize.rs index c387b7e7f..e6d1c91ca 100644 --- a/cc/ir/test_linearize.rs +++ b/cc/ir/test_linearize.rs @@ -9101,6 +9101,179 @@ fn test_volatile_accesses_carry_the_marker() { ); } +/// A member of a `volatile` object is itself volatile (C17 6.5.2.3p3/p4), so +/// every access to one carries the marker. +/// +/// The reverse direction -- a `volatile` member of a plain object -- always +/// worked, because there the member's own declared type carries the qualifier. +/// This is the other one: the qualifier is on the *object*, and +/// `find_member` answers with the member's declared type, which cannot show it. +/// So the load was unmarked and DCE deleted it from `-O1` up. +#[test] +fn test_a_member_of_a_volatile_object_carries_the_marker() { + let target = Target::host(); + let src = "struct S { int a; int b; };\n\ + struct N { struct S in; };\n\ + typedef volatile struct S VS;\n\ + volatile struct S vs;\n\ + volatile struct S *vp;\n\ + volatile struct S vsa[4];\n\ + volatile struct N vn;\n\ + VS vt;\n\ + struct S plain;\n\ + struct S *pp;\n\ + void read_direct(void) { vs.a; }\n\ + void read_arrow(void) { vp->a; }\n\ + void read_element(void) { vsa[2].a; }\n\ + void read_nested(void) { vn.in.a; }\n\ + void read_typedef(void) { vt.a; }\n\ + void write_direct(void) { vs.a = 1; }\n\ + void read_plain(void) { plain.a; }\n\ + void read_plain_arrow(void) { pp->a; }\n"; + let module = linearize_source(src, &target); + + let accesses = |name: &str| -> Vec<(Opcode, bool)> { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.is_volatile_access())) + .collect() + }; + + // Every spelling of "the object is volatile": directly, through a pointer + // to volatile, through an array's element type, through a nested member + // whose own type is qualified by the object above it, and through a + // typedef that carries the qualifier. + assert_eq!(accesses("read_direct"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_element"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_nested"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_typedef"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("write_direct"), vec![(Opcode::Store, true)]); + // `volatile struct S *vp` qualifies the pointee, so reading `vp` itself is + // an ordinary load and the access through it is the volatile one. + assert_eq!( + accesses("read_arrow"), + vec![(Opcode::Load, false), (Opcode::Load, true)] + ); + + // The control: an unqualified object's member is not marked, or the marker + // would mean nothing. + assert_eq!(accesses("read_plain"), vec![(Opcode::Load, false)]); + assert_eq!( + accesses("read_plain_arrow"), + vec![(Opcode::Load, false), (Opcode::Load, false)] + ); +} + +/// A `volatile` bit-field access is marked although the access is of the +/// carrier. +/// +/// `emit_bitfield_load`/`_store` read and write a storage unit whose type is +/// an unqualified integer, so no marker can be derived from the instruction's +/// own type. `mark_volatile_access` preserves one the emitter sets, and this is +/// the case it exists for. +#[test] +fn test_a_volatile_bitfield_access_carries_the_marker() { + let target = Target::host(); + let src = "struct B { volatile unsigned f : 3; unsigned g : 5; };\n\ + struct B b;\n\ + volatile struct B vb;\n\ + void read_field(void) { b.f; }\n\ + void read_object(void) { vb.g; }\n\ + void write_object(void) { vb.g = 1; }\n\ + void read_plain(void) { b.g; }\n\ + void write_plain(void) { b.g = 1; }\n"; + let module = linearize_source(src, &target); + + let accesses = |name: &str| -> Vec<(Opcode, bool)> { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.is_volatile_access())) + .collect() + }; + + // Both spellings: the field declared `volatile`, and an ordinary field of + // a `volatile` object. + assert_eq!(accesses("read_field"), vec![(Opcode::Load, true)]); + assert_eq!(accesses("read_object"), vec![(Opcode::Load, true)]); + // A bit-field store is a read-modify-write of the carrier, and both halves + // of it are observable. + assert_eq!( + accesses("write_object"), + vec![(Opcode::Load, true), (Opcode::Store, true)] + ); + + // The controls. + assert_eq!(accesses("read_plain"), vec![(Opcode::Load, false)]); + assert_eq!( + accesses("write_plain"), + vec![(Opcode::Load, false), (Opcode::Store, false)] + ); +} + +/// A conditional may not speculate a `volatile` member, in either spelling. +/// +/// `Select` is the branchless form, and reaching it means both arms were +/// evaluated. C17 6.5.15p4 evaluates only one of them, and 5.1.2.3 makes each +/// volatile read an observable event -- so the arms may only collapse when +/// both are pure. `is_pure_expr` asked whether the *base* was pure, which a +/// named object always is. +#[test] +fn test_a_volatile_member_is_not_speculated() { + let target = Target::host(); + let src = "struct V { volatile unsigned status; unsigned other; };\n\ + struct P { unsigned one; unsigned other; };\n\ + struct V v;\n\ + volatile struct P vp;\n\ + struct P p;\n\ + unsigned member_is_volatile(int c) { return c ? v.status : v.other; }\n\ + unsigned object_is_volatile(int c) { return c ? vp.one : vp.other; }\n\ + unsigned all_plain(int c) { return c ? p.one : p.other; }\n"; + let module = linearize_source(src, &target); + + let selects = |name: &str| -> usize { + module + .functions + .iter() + .find(|f| f.name == name) + .unwrap_or_else(|| panic!("function {name}")) + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .filter(|i| i.op == Opcode::Select) + .count() + }; + + assert_eq!( + selects("member_is_volatile"), + 0, + "a volatile member may not be read on the path that did not select it" + ); + assert_eq!( + selects("object_is_volatile"), + 0, + "a member of a volatile object is volatile (C17 6.5.2.3p3)" + ); + assert_eq!( + selects("all_plain"), + 1, + "two ordinary member reads are pure and may still collapse" + ); +} + /// A `volatile` access keeps the object it reaches in memory: promotion would /// rewrite the access into a register `Copy`, and the access must happen. /// diff --git a/cc/parse/expression.rs b/cc/parse/expression.rs index 3349ad5e2..15aae0f5a 100644 --- a/cc/parse/expression.rs +++ b/cc/parse/expression.rs @@ -17,7 +17,7 @@ use crate::strings::StringId; use crate::symbol::{Namespace, Symbol}; use crate::token::lexer::{Position, SpecialToken, TokenType, TokenValue}; use crate::token::literal; -use crate::types::{Type, TypeId, TypeKind, TypeModifiers}; +use crate::types::{Type, TypeId, TypeKind}; use gettextrs::gettext; const DEFAULT_ARG_LIST_CAPACITY: usize = 8; @@ -359,20 +359,7 @@ impl<'a> Parser<'a> { /// never select the `const int` association. pub(crate) fn lvalue_converted_type(&mut self, typ: TypeId) -> TypeId { let decayed = self.decayed_type(typ); - - const QUALIFIERS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - - let ty = self.types.get(decayed); - if !ty.modifiers.intersects(QUALIFIERS) { - return decayed; - } - - let mut unqualified = ty.clone(); - unqualified.modifiers.remove(QUALIFIERS); - self.types.intern(unqualified) + self.types.unqualified(decayed) } /// Parse a conditional (ternary) expression: cond ? then : else @@ -1322,8 +1309,10 @@ impl<'a> Parser<'a> { &gettext("request for member in something not a structure or union"), ); self.types.int_id - } else if let Some(info) = self.types.find_member(resolved, member) { - info.typ + } else if let Some(typ) = self.types.member_access_type(t, resolved, member) { + // C17 6.5.2.3p3: so-qualified by the object, whose + // qualifiers are on `t` -- `resolved` has lost them. + typ } else { let member_name = self.idents.get_opt(member).unwrap_or(""); diag::error_args(dot_pos, "has no member named '{0}'", &[member_name]); @@ -1361,8 +1350,12 @@ impl<'a> Parser<'a> { ), ); self.types.int_id - } else if let Some(info) = self.types.find_member(resolved, member) { - info.typ + } else if let Some(typ) = + self.types.member_access_type(struct_type, resolved, member) + { + // C17 6.5.2.3p4: so-qualified by the *pointee*. + // `struct S *volatile p` qualifies `p`, not `*p`. + typ } else { let member_name = self.idents.get_opt(member).unwrap_or(""); diag::error_args( @@ -1713,7 +1706,10 @@ impl<'a> Parser<'a> { // let it through, so `1 << 1L` came out `long` and // `sizeof(1 << 1L)` answered 8 where gcc answers 4. BinaryOp::Shl | BinaryOp::Shr => { + // The promoted type of a *value*: unqualified, as every + // arithmetic result is (6.3.2.1p2). let promoted = self.types.integer_promote(left_type); + let promoted = self.types.unqualified(promoted); self.check_shift_count(op, promoted, &right); promoted } @@ -1887,6 +1883,9 @@ impl<'a> Parser<'a> { } else { self.types.integer_promote(op_typ) }; + // The result is a value, which has the unqualified type (6.3.2.1p2): + // `-v` is an `int` even where `v` is a `volatile int`. + let typ = self.types.unqualified(typ); // The *value* is promoted, not just the type it is computed at. The // conversion used to be left out, on the reasoning that the operand // is already in a wider register -- but nothing in the IR then says @@ -1958,8 +1957,24 @@ impl<'a> Parser<'a> { } } + /// The type the usual arithmetic conversions (C17 6.3.1.8) bring two + /// operands to, as an *rvalue* type. + /// + /// `common_type` answers with one of the operands' own `TypeId`s, so + /// `volatile int + int` came out `volatile int` and a qualifier the object + /// carried leaked into the type of a value. C17 6.3.2.1p2 drops the + /// qualifiers when an lvalue is converted to a value, and nothing + /// downstream may read an rvalue's type as "this expression touched a + /// volatile object" -- now that a member of a `volatile` object is itself + /// volatile (6.5.2.3p3), that leak would reach every `s.m + 1`. + /// + /// Stripping them here rather than in `common_type` keeps the latter a + /// pure question about conversion rank, which is what lets the linearizer + /// and the constant folder ask it through a `&TypeTable`: interning a + /// stripped type needs `&mut`. fn usual_arithmetic_conversions(&mut self, left: TypeId, right: TypeId) -> TypeId { - self.types.common_type(left, right) + let common = self.types.common_type(left, right); + self.types.unqualified(common) } /// Parse a C11 generic selection (C17 6.5.1.1): diff --git a/cc/parse/test_parser.rs b/cc/parse/test_parser.rs index fab761696..400971ebb 100644 --- a/cc/parse/test_parser.rs +++ b/cc/parse/test_parser.rs @@ -5475,6 +5475,111 @@ fn first_statement_of(tu: &TranslationUnit, n: usize) -> &Stmt { stmt } +/// C17 6.5.2.3p3/p4: `s.m` and `p->m` have the *so-qualified* version of the +/// member's type -- the member's type plus the qualifiers of the object. +/// +/// The parser is where that type is formed, and it took the member's declared +/// type unchanged: a member of a `volatile` object read as an ordinary `int` +/// (so DCE deleted the load) and a member of a `const` object was assignable. +#[test] +fn test_a_member_access_is_qualified_by_the_object() { + let quals = |src: &str, decls: &str| -> TypeModifiers { + let code = format!("struct S {{ int a; }};\nstruct N {{ struct S in; }};\n{decls}\nvoid t(void) {{ {src}; }}"); + let (tu, types, _, _) = parse_tu(&code).unwrap(); + let Stmt::Expr(expr) = first_statement(&tu) else { + panic!("{src}: expected an expression statement"); + }; + types.qualifiers(expr.typ.expect("a typed expression")) + }; + const NONE: TypeModifiers = TypeModifiers::empty(); + + // The object's qualifiers, in each spelling that reaches a member. + assert_eq!( + quals("vs.a", "volatile struct S vs;"), + TypeModifiers::VOLATILE + ); + assert_eq!(quals("cs.a", "const struct S cs;"), TypeModifiers::CONST); + assert_eq!( + quals("cvs.a", "const volatile struct S cvs;"), + TypeModifiers::CONST | TypeModifiers::VOLATILE + ); + assert_eq!( + quals("vsa[1].a", "volatile struct S vsa[2];"), + TypeModifiers::VOLATILE + ); + assert_eq!( + quals("vn.in.a", "volatile struct N vn;"), + TypeModifiers::VOLATILE, + "the intermediate member is qualified too, and carries it downward" + ); + assert_eq!( + quals("vt.a", "typedef volatile struct S VS; VS vt;"), + TypeModifiers::VOLATILE + ); + + // `->` takes them from the *pointee*, which is the object it names. + assert_eq!( + quals("vp->a", "volatile struct S *vp;"), + TypeModifiers::VOLATILE + ); + assert_eq!( + quals("qp->a", "struct S *volatile qp;"), + NONE, + "`struct S *volatile` qualifies the pointer, not what it points at" + ); + + // `_Atomic` does not travel: there is no atomic access to one member of an + // atomic object, and gcc does not pretend otherwise. + assert_eq!(quals("as.a", "_Atomic struct S as;"), NONE); + + // An unqualified object leaves the member's declared type alone -- and a + // qualifier on the *member* still reaches the access, which is the + // direction that always worked. + assert_eq!(quals("s.a", "struct S s;"), NONE); + assert_eq!( + quals("vm.v", "struct M { volatile int v; }; struct M vm;"), + TypeModifiers::VOLATILE + ); +} + +/// C17 6.3.2.1p2: converting an lvalue to a value drops the qualifiers, so an +/// arithmetic result is never qualified. +/// +/// `common_type` and `integer_promote` answer with one of the operands' own +/// type ids, so `volatile int + int` came out `volatile int` -- a qualifier on +/// the type of a value, which nothing may read as "this expression touched a +/// volatile object". +#[test] +fn test_an_arithmetic_result_is_unqualified() { + let quals = |src: &str| -> TypeModifiers { + let code = format!( + "struct S {{ int a; long l; }};\nvolatile struct S vs;\nvolatile int vi;\n\ + void t(void) {{ {src}; }}" + ); + let (tu, types, _, _) = parse_tu(&code).unwrap(); + let Stmt::Expr(expr) = first_statement(&tu) else { + panic!("{src}: expected an expression statement"); + }; + types.qualifiers(expr.typ.expect("a typed expression")) + }; + const NONE: TypeModifiers = TypeModifiers::empty(); + + // The usual arithmetic conversions, a shift (whose result is the promoted + // *left* operand's type), and a unary operator. + assert_eq!(quals("vs.a + 1"), NONE); + assert_eq!(quals("1 + vs.a"), NONE); + assert_eq!(quals("vs.l * vs.a"), NONE); + assert_eq!(quals("vi | 1"), NONE); + assert_eq!(quals("vs.a << 1"), NONE); + assert_eq!(quals("-vs.a"), NONE); + assert_eq!(quals("~vi"), NONE); + assert_eq!(quals("1 ? vs.a : 0"), NONE); + + // The lvalue itself keeps them: that is the whole point of the rule above, + // and the assignment check and the volatile marker both read it. + assert_eq!(quals("vs.a"), TypeModifiers::VOLATILE); +} + // Library builtins: abs, fabs, creal, conj, ... as checked calls /// The in-place call `expr` is, as (function, arguments), or a panic naming diff --git a/cc/tests/codegen/memopt.rs b/cc/tests/codegen/memopt.rs index fc9fc95c7..f87a2e0c8 100644 --- a/cc/tests/codegen/memopt.rs +++ b/cc/tests/codegen/memopt.rs @@ -1498,3 +1498,187 @@ fn memopt_two_volatile_stores_are_both_performed() { } } } + +/// A member of a `volatile` object is itself volatile, so reading it is an +/// observable event that survives every optimization level. +/// +/// C17 6.5.2.3p3/p4: the result of `s.m` has the *so-qualified* version of the +/// member's type — it inherits the qualifiers of the object. c17 took the +/// member's declared type unchanged, so a member of a `volatile` struct read as +/// an ordinary `int` and DCE deleted it from `-O1` up. The reverse direction +/// (`struct T { volatile int a; }`) always worked, because there the member's +/// own type carries the qualifier; that case is the control below. +#[test] +fn memopt_a_member_of_a_volatile_object_is_volatile() { + // Distinctive names: a single letter matches `pushq`/`stp`/`.cfi_startproc` + // inside the body range and would pass against an emptied function. + let src = "\ +struct S { int a; int b; }; +volatile struct S vqobj; +volatile struct S *vqptr; +void probe_direct(void) { vqobj.a; } +void probe_arrow(void) { vqptr->a; } +void probe_assign(void) { int t = vqobj.a; (void)t; } +void probe_second(void) { vqobj.b; } +"; + for level in ["-O0", "-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_member", triple, src, &[level]); + for (func, object) in [ + ("probe_direct", "vqobj"), + ("probe_arrow", "vqptr"), + ("probe_assign", "vqobj"), + ("probe_second", "vqobj"), + ] { + assert_body_contains( + &asm, + func, + object, + &format!( + "a member of a volatile object is volatile (C17 6.5.2.3p3): \ + {func} at {level} on {triple} must still access `{object}`" + ), + ); + } + } + } +} + +/// The control for the test above: an ordinary aggregate's member read is still +/// deleted, so that test cannot pass by marking every member access volatile. +#[test] +fn memopt_a_member_of_a_plain_object_is_still_removed() { + let src = "\ +struct S { int a; }; +struct S pqobj; +void probe(void) { pqobj.a; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("plain_member", triple, src, &["-O2"]); + assert_body_lacks( + &asm, + "probe", + "pqobj", + "reading an ordinary member has no effect and is dead code", + ); + } +} + +/// A `volatile` member is not speculatable, so a conditional must not read the +/// arm it did not take. +/// +/// C17 6.5.15p4 evaluates only one of the second and third operands, and +/// 5.1.2.3 makes each volatile read an observable event. `is_pure_expr`'s +/// `Member` arm asked only whether the *base* was pure, so the read was +/// hoisted and both members were loaded unconditionally into a branchless +/// select — at `-O0` too. +#[test] +fn memopt_a_volatile_member_is_not_speculated_by_a_conditional() { + let src = "\ +struct S { volatile unsigned status; unsigned other; }; +struct S sqobj; +unsigned probe(int c) { return c ? sqobj.status : sqobj.other; } +"; + for level in ["-O0", "-O2"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_member_select", triple, src, &[level]); + let select = if triple == X86_64_LINUX { + "cmov" + } else { + "csel" + }; + assert_body_lacks( + &asm, + "probe", + select, + &format!( + "a volatile member read cannot be speculated, so the arms may not \ + collapse into a conditional move: {level} on {triple}" + ), + ); + } + } +} + +/// The control for the test above: with no volatile member, the branchless +/// select is still allowed, so that test is asserting the qualifier and not +/// merely that c17 stopped emitting conditional moves. +#[test] +fn memopt_a_plain_member_may_still_be_speculated() { + let src = "\ +struct S { unsigned one; unsigned other; }; +struct S pqsel; +unsigned probe(int c) { return c ? pqsel.one : pqsel.other; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("plain_member_select", triple, src, &["-O2"]); + let select = if triple == X86_64_LINUX { + "cmov" + } else { + "csel" + }; + assert_body_contains( + &asm, + "probe", + select, + "two ordinary member reads are pure and may still collapse to a select", + ); + } +} + +/// A `volatile` bit-field read is observable, even though the access is of the +/// carrier and the carrier can never carry the qualifier. +/// +/// The bit-field emitters build their load and store at +/// `bitfield_storage_type`, which is the unqualified storage unit, so nothing +/// derived the marker from the access type. `mark_volatile_access` anticipates +/// exactly this ("a bit-field reads a storage unit whose type is the carrier") +/// and preserves a marker the site sets itself — neither emitter set one, and +/// the read was deleted outright from `-O1` up. Both spellings are covered: the +/// field declared `volatile`, and an ordinary field of a `volatile` object. +#[test] +fn memopt_a_volatile_bitfield_read_is_performed() { + let src = "\ +struct B { volatile unsigned f : 3; unsigned g : 5; }; +struct B bfqobj; +volatile struct B vbfqobj; +void probe_field(void) { bfqobj.f; } +void probe_object(void) { vbfqobj.g; } +"; + for level in ["-O0", "-O1", "-O2", "-Os"] { + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("vol_bitfield", triple, src, &[level]); + for (func, object) in [("probe_field", "bfqobj"), ("probe_object", "vbfqobj")] { + assert_body_contains( + &asm, + func, + object, + &format!( + "a volatile bit-field read is observable: {func} at {level} \ + on {triple} must still access `{object}`" + ), + ); + } + } + } +} + +/// The control: an ordinary bit-field read is still dead code, so the test +/// above cannot pass by marking every bit-field access volatile. +#[test] +fn memopt_a_plain_bitfield_read_is_still_removed() { + let src = "\ +struct B { unsigned f : 3; }; +struct B pbfqobj; +void probe(void) { pbfqobj.f; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("plain_bitfield", triple, src, &["-O2"]); + assert_body_lacks( + &asm, + "probe", + "pbfqobj", + "reading an ordinary bit-field has no effect and is dead code", + ); + } +} diff --git a/cc/tests/diagnostics/mod.rs b/cc/tests/diagnostics/mod.rs index 8ae50ad8a..a9e9aad39 100644 --- a/cc/tests/diagnostics/mod.rs +++ b/cc/tests/diagnostics/mod.rs @@ -7347,3 +7347,90 @@ fn diagnostics_escape_out_of_range_names_the_literals_line() { ); } } + +/// A member of a `const` object is itself `const`, so writing it is a +/// constraint violation. +/// +/// C17 6.5.2.3p3/p4 gives `s.m` the *so-qualified* version of the member's +/// type, and 6.5.16p2 requires a modifiable lvalue on the left of an +/// assignment. c17 took the member's declared type unqualified, so +/// `check_const_assignment` saw an ordinary `int` and every one of these was +/// accepted silently. gcc and clang reject them. +#[test] +fn diagnostics_writing_a_member_of_a_const_object_is_rejected() { + compile_expect_error( + "const_aggregate_member", + "struct S { int a; };\nconst struct S cs = {1};\nvoid f(void){ cs.a = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_member_arrow", + "struct S { int a; };\nvoid f(const struct S *p){ p->a = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_member_nested", + "struct T { int x; };\nstruct S { struct T t; };\n\ + const struct S cs;\nvoid f(void){ cs.t.x = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_member_increment", + "struct S { int a; };\nconst struct S cs;\nvoid f(void){ cs.a++; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_element", + "struct S { int a; };\nconst struct S cs[2];\nvoid f(void){ cs[1].a = 2; }\n", + "read-only", + ); +} + +/// The other direction: the same shapes without the `const` must still compile, +/// so the checks above cannot pass by rejecting every member assignment. +#[test] +fn diagnostics_writing_a_member_of_a_plain_object_is_accepted() { + compile_expect_ok( + "plain_aggregate_member", + "struct S { int a; };\nstruct S s;\nvoid f(void){ s.a = 2; }\n\ + void g(struct S *p){ p->a = 2; }\n\ + struct T { struct S in; };\nstruct T t;\nvoid h(void){ t.in.a = 2; }\n\ + struct S arr[2];\nvoid i(void){ arr[1].a = 2; }\n\ + void j(void){ s.a++; }\n", + ); + // A `const` *pointer* to a non-const object leaves the pointee writable: + // the qualifier is on the pointer, not on what it points at. + compile_expect_ok( + "const_pointer_not_pointee", + "struct S { int a; };\nvoid f(struct S *const p){ p->a = 2; }\n", + ); +} + +/// An *array* member of a `const` object is an array of `const`, so writing an +/// element of it is a constraint violation too. +/// +/// C17 6.7.3p10: where an array type is qualified, the element type is +/// so-qualified and the array is not -- which is what a subscript reads. So the +/// so-qualified version of `int [4]` is an array of `const int`, and +/// `cs.arr[0] = 1` is a write to a `const int`. Qualifying the array itself +/// instead would leave the element an ordinary `int` and accept the write. +#[test] +fn diagnostics_writing_an_array_member_of_a_const_object_is_rejected() { + compile_expect_error( + "const_aggregate_array_member", + "struct S { int arr[4]; };\nconst struct S cs;\nvoid f(void){ cs.arr[0] = 2; }\n", + "read-only", + ); + compile_expect_error( + "const_aggregate_array_member_2d", + "struct S { int grid[2][2]; };\nvoid f(const struct S *p){ p->grid[1][1] = 2; }\n", + "read-only", + ); + // The control: the same writes through an unqualified object. + compile_expect_ok( + "plain_aggregate_array_member", + "struct S { int arr[4]; int grid[2][2]; };\nstruct S s;\n\ + void f(void){ s.arr[0] = 2; s.grid[1][1] = 3; }\n\ + void g(struct S *p){ p->arr[0] = 2; }\n", + ); +} diff --git a/cc/types.rs b/cc/types.rs index b7d965002..3db782f28 100644 --- a/cc/types.rs +++ b/cc/types.rs @@ -462,6 +462,34 @@ impl Type { /// compatible with `int`. Invisible to `sizeof`, fatal to any comparison. pub const DECL_SPECIFIERS: TypeModifiers = Self::STORAGE_CLASS.union(TypeModifiers::NORETURN); + /// The type qualifiers of C17 6.7.3p1. + /// + /// These are the bits a *type* carries at its top level, as opposed to the + /// declaration specifiers above: `const int` and `int` are two types, while + /// `static int` and `int` are one. Four copies of this list had been + /// spelled out inline -- in `compatible_ignoring_base`, in `compatible`, in + /// `assign_fault`'s pointer rule and in the parser's + /// `lvalue_converted_type` -- which is three chances for the set to drift. + pub const QUALIFIERS: TypeModifiers = TypeModifiers::CONST + .union(TypeModifiers::VOLATILE) + .union(TypeModifiers::RESTRICT) + .union(TypeModifiers::ATOMIC); + + /// The qualifiers a member inherits from the object that holds it. + /// + /// C17 6.5.2.3p3/p4 gives `s.m` the "so-qualified version" of the member's + /// type, which is the member's type plus the qualifiers of `s`. Not all + /// four travel: + /// + /// * `_Atomic` does not. A member of an `_Atomic` struct is not itself + /// atomic -- there is no lock-free way to read one out of an atomic + /// object -- and gcc does not make it so. Reaching into one at all is + /// what `Parser::warn_atomic_member_access` already reports. + /// * `restrict` cannot: 6.7.3p2 admits it only on a pointer to an object + /// type, so a composite never carries it in the first place. + pub const MEMBER_QUALIFIERS: TypeModifiers = + TypeModifiers::CONST.union(TypeModifiers::VOLATILE); + /// The storage-class specifiers of C17 6.7.1, and `inline`. /// /// What a declaration records as its storage class, and what a declarator @@ -488,12 +516,6 @@ impl Type { /// With TypeId interning, base types are compared by TypeId equality. /// For full recursive comparison, use TypeTable::types_compatible(). fn compatible_ignoring_base(&self, other: &Type) -> bool { - // Top-level qualifiers to ignore - const QUALIFIERS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - // Compare kinds first if self.kind != other.kind { return false; @@ -520,7 +542,7 @@ impl Type { const REDUNDANT_SIZE: TypeModifiers = TypeModifiers::SHORT .union(TypeModifiers::LONG) .union(TypeModifiers::LONGLONG); - let ignored = QUALIFIERS + let ignored = Self::QUALIFIERS .union(redundant_signed) .union(REDUNDANT_SIZE) .union(Self::DECL_SPECIFIERS); @@ -1728,12 +1750,8 @@ impl TypeTable { // Compatible targets, but the assignment must not silently gain // write access: the target's qualifiers have to include the // source's. - const QUALS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - let t_quals = self.modifiers(t_pointee) & QUALS; - let v_quals = self.modifiers(v_pointee) & QUALS; + let t_quals = self.qualifiers(t_pointee); + let v_quals = self.qualifiers(v_pointee); return (!t_quals.contains(v_quals)).then_some(AssignFault::QualifierDiscard); } @@ -1849,6 +1867,77 @@ impl TypeTable { } } + /// The top-level type qualifiers of `id` (C17 6.7.3p1). + /// + /// Only the qualifiers: `modifiers` answers with the declaration + /// specifiers and the size and sign spellings mixed in, and every caller + /// that wanted "is this `const`?" had to mask them off itself. + pub fn qualifiers(&self, id: TypeId) -> TypeModifiers { + self.modifiers(id) & Type::QUALIFIERS + } + + /// The version of `id` qualified with `quals` -- the "so-qualified + /// version" of C17 6.5.2.3p3/p4. + /// + /// This is what a member access yields: `s.m` has the member's type plus + /// the qualifiers of `s`, so a member of a `volatile` object is volatile + /// and a member of a `const` object is not assignable. Without it a + /// `volatile struct` read as an ordinary struct and DCE deleted the load. + /// + /// Qualifiers outside [`Type::QUALIFIERS`] are ignored, and a type that + /// already carries all of them is returned unchanged -- so the common case + /// of an unqualified object interns nothing. A composite is not + /// deduplicated by [`Self::intern`] (it has identity), so qualifying one + /// hands back a fresh `TypeId` each time; that is sound because + /// compatibility of composites is decided by tag and members rather than + /// by id, and it is rare enough not to matter -- only an access to a + /// member of aggregate type, through a qualified object, reaches it. + pub fn qualified_with(&mut self, id: TypeId, quals: TypeModifiers) -> TypeId { + let add = quals & Type::QUALIFIERS; + if add.is_empty() { + return id; + } + // C17 6.7.3p10: where an array type is qualified, the *element type* is + // so-qualified and the array is not. That is also how a declaration + // records it -- `const int a[4]` puts the `const` on the element -- so + // qualifying the array instead would leave `cs.arr[0]` an ordinary + // `int`, which the subscript reads from the element type, and a write + // to it would be accepted. + if self.kind(id) == TypeKind::Array { + let Some(elem) = self.base_type(id) else { + return id; + }; + let qualified_elem = self.qualified_with(elem, add); + if qualified_elem == elem { + return id; + } + let mut array = self.get(id).clone(); + array.base = Some(qualified_elem); + return self.intern(array); + } + if self.modifiers(id).contains(add) { + return id; + } + let mut qualified = self.get(id).clone(); + qualified.modifiers |= add; + self.intern(qualified) + } + + /// The unqualified version of `id` (C17 6.3.2.1p2). + /// + /// Lvalue conversion drops the qualifiers, so this is the type of the + /// *value* an lvalue yields: `volatile int v; v + 0` has type `int`, and + /// nothing downstream may conclude from the sum's type that the addition + /// touched a volatile object. + pub fn unqualified(&mut self, id: TypeId) -> TypeId { + if self.qualifiers(id).is_empty() { + return id; + } + let mut unqualified = self.get(id).clone(); + unqualified.modifiers.remove(Type::QUALIFIERS); + self.intern(unqualified) + } + /// How many scalar initializers it takes to fill this type. /// /// This is the measure brace elision runs on (C17 6.7.9p20): a brace-less @@ -2400,11 +2489,45 @@ impl TypeTable { 8 // All supported platforms use 8-byte alignment for va_list } + /// The type an access to `name` yields, in an object of type `object`. + /// + /// C17 6.5.2.3p3/p4: the result of `s.m` and of `p->m` has the + /// *so-qualified* version of the member's type, so a member of a + /// `volatile` object is volatile and a member of a `const` object is not + /// assignable. [`Self::find_member`] answers with the member's *declared* + /// type, which is what every other caller of it wants -- an offset, a size, + /// a bit-field's width -- so the rule lives here, beside it, rather than in + /// each of the two expression forms that need it. + /// + /// The two type ids are not redundant. `members` is `object` after the + /// parser resolved an incomplete tag to its definition, which is where the + /// member list is; the qualifiers have to come from `object`, because + /// resolving answers with the *tag's* type and a tag is never qualified -- + /// resolving first is how `volatile struct S` loses the `volatile`. + /// + /// For `p->m` the object is the pointee: `struct S *volatile p` qualifies + /// the pointer, not what it points at. + pub fn member_access_type( + &mut self, + object: TypeId, + members: TypeId, + name: StringId, + ) -> Option { + let quals = self.qualifiers(object) & Type::MEMBER_QUALIFIERS; + let declared = self.find_member(members, name)?.typ; + Some(self.qualified_with(declared, quals)) + } + /// Find a member in a struct/union type, including anonymous struct/union members /// C11 6.7.2.1p13: "An unnamed member of structure type with no tag is called an /// anonymous structure; an unnamed member of union type with no tag is called an /// anonymous union. The members of an anonymous structure or union are considered /// to be members of the containing structure or union." + /// + /// The `typ` this answers with is the member's *declared* type. An + /// expression that accesses the member has the so-qualified version of it + /// instead (C17 6.5.2.3p3/p4) -- see [`Self::member_access_type`], which is + /// what the `.` and `->` operators go through. pub fn find_member(&self, id: TypeId, name: StringId) -> Option { self.find_member_recursive(id, name, 0) } @@ -2473,13 +2596,7 @@ impl TypeTable { if id1 == id2 { return true; } - const QUALIFIERS: TypeModifiers = TypeModifiers::CONST - .union(TypeModifiers::VOLATILE) - .union(TypeModifiers::RESTRICT) - .union(TypeModifiers::ATOMIC); - if quals == TopLevelQualifiers::Significant - && self.get(id1).modifiers.intersection(QUALIFIERS) - != self.get(id2).modifiers.intersection(QUALIFIERS) + if quals == TopLevelQualifiers::Significant && self.qualifiers(id1) != self.qualifiers(id2) { return false; } @@ -2872,6 +2989,144 @@ mod tests { assert!(types.contains_volatile(opaque)); } + /// `qualifiers`, `qualified_with` and `unqualified` on the same type. + /// + /// The pair has to be exact inverses on the top-level qualifiers and to + /// leave everything else -- the kind, the size spellings, the storage + /// class -- alone, because they are what forms and unforms the + /// so-qualified type of a member access. + #[test] + fn qualifying_a_type_adds_and_removes_only_the_qualifiers() { + let mut types = TypeTable::new(&crate::target::Target::host()); + let int = types.int_id; + assert!(types.qualifiers(int).is_empty()); + + // Adding, one qualifier at a time and then both. + let vol = types.qualified_with(int, TypeModifiers::VOLATILE); + assert_eq!(types.qualifiers(vol), TypeModifiers::VOLATILE); + assert_eq!(types.kind(vol), TypeKind::Int); + assert_eq!(types.size_bits(vol), types.size_bits(int)); + let cv = types.qualified_with(vol, TypeModifiers::CONST); + assert_eq!( + types.qualifiers(cv), + TypeModifiers::CONST | TypeModifiers::VOLATILE + ); + + // Interned, so the same request answers with the same id -- and a type + // that already carries the qualifier is returned untouched. + assert_eq!(types.qualified_with(int, TypeModifiers::VOLATILE), vol); + assert_eq!(types.qualified_with(vol, TypeModifiers::VOLATILE), vol); + assert_eq!(types.qualified_with(int, TypeModifiers::empty()), int); + + // Nothing outside `Type::QUALIFIERS` travels: a storage class is a + // property of a declaration, not of a type. + assert_eq!(types.qualified_with(int, TypeModifiers::STATIC), int); + + // And back down again. + assert_eq!(types.unqualified(cv), int); + assert_eq!(types.unqualified(vol), int); + assert_eq!(types.unqualified(int), int); + + // A qualifier below the top level is not a top-level qualifier: + // `volatile int *` is an ordinary pointer. + let ptr_to_vol = types.intern(Type::pointer(vol)); + assert!(types.qualifiers(ptr_to_vol).is_empty()); + assert_eq!(types.unqualified(ptr_to_vol), ptr_to_vol); + + // Qualifying an array qualifies its element type (C17 6.7.3p10), at + // every level, and the array itself stays unqualified -- which is what + // a subscript of it then reads. + let arr = types.intern(Type::array(int, 4)); + let const_arr = types.qualified_with(arr, TypeModifiers::CONST); + assert!(types.qualifiers(const_arr).is_empty()); + let elem = types.base_type(const_arr).expect("element type"); + assert_eq!(types.qualifiers(elem), TypeModifiers::CONST); + let rows = types.intern(Type::array(arr, 2)); + let vol_rows = types.qualified_with(rows, TypeModifiers::VOLATILE); + let row = types.base_type(vol_rows).expect("row type"); + let cell = types.base_type(row).expect("cell type"); + assert_eq!(types.qualifiers(cell), TypeModifiers::VOLATILE); + assert!(types.contains_volatile(vol_rows)); + } + + /// C17 6.5.2.3p3/p4: a member access has the *so-qualified* version of the + /// member's type. + /// + /// `find_member` answers with the declared type, which is what an offset or + /// a width is read from; the access has the object's qualifiers as well, + /// which is what makes a member of a `volatile` object volatile and a + /// member of a `const` object unassignable. + #[test] + fn a_member_access_is_qualified_by_the_object() { + let mut types = TypeTable::new(&crate::target::Target::host()); + let mut idents = crate::strings::StringTable::new(); + let a = idents.intern("a"); + let v = idents.intern("v"); + let int = types.int_id; + let vol_int = types.intern(Type::with_modifiers(TypeKind::Int, TypeModifiers::VOLATILE)); + let member = |name, typ| StructMember { + name, + typ, + offset: 0, + bit_offset: None, + bit_width: None, + access_bytes: None, + explicit_align: None, + }; + let tag = idents.intern("S"); + let composite = CompositeType { + tag: Some(tag), + members: vec![member(a, int), member(v, vol_int)], + enum_constants: Vec::new(), + size: 8, + align: 4, + member_align: 4, + is_complete: true, + transparent: false, + anon_id: None, + }; + let plain = types.intern(Type::struct_type(composite)); + + // An unqualified object: the declared types, unchanged. + assert_eq!(types.member_access_type(plain, plain, a), Some(int)); + assert_eq!(types.member_access_type(plain, plain, v), Some(vol_int)); + assert_eq!(types.member_access_type(plain, plain, tag), None); + + // A `volatile` object makes every member volatile, and a `const` one + // makes every member `const`. + let vol_obj = types.qualified_with(plain, TypeModifiers::VOLATILE); + let from_vol = types.member_access_type(vol_obj, vol_obj, a).unwrap(); + assert_eq!(types.qualifiers(from_vol), TypeModifiers::VOLATILE); + assert!(types.contains_volatile(from_vol)); + let const_obj = types.qualified_with(plain, TypeModifiers::CONST); + let from_const = types.member_access_type(const_obj, const_obj, a).unwrap(); + assert_eq!(types.qualifiers(from_const), TypeModifiers::CONST); + + // `_Atomic` does not travel: a member of an `_Atomic` struct cannot be + // read atomically, and gcc does not claim it can. + let atomic_obj = types.qualified_with(plain, TypeModifiers::ATOMIC); + assert_eq!( + types.member_access_type(atomic_obj, atomic_obj, a), + Some(int) + ); + + // The two type ids are not interchangeable. The qualifiers come from + // the object as written; the members come from the type the parser + // resolved it to, which is the tag's and is never qualified. Reading + // the qualifiers from the resolved type is how `volatile struct S` + // loses its `volatile`. + let incomplete = types.intern(Type::struct_type(CompositeType::incomplete(Some(tag)))); + let vol_incomplete = types.qualified_with(incomplete, TypeModifiers::VOLATILE); + assert_eq!( + types.member_access_type(vol_incomplete, plain, a), + Some(vol_int) + ); + assert_eq!( + types.member_access_type(vol_incomplete, vol_incomplete, a), + None + ); + } + use super::*; /// `make_complex` and `complex_base` must be exact inverses, for every From 98e4b03d3a3356bdd48d4d11a8810d30f6120173 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 00:43:37 -0400 Subject: [PATCH 06/17] cc: build every two-armed conditional through one forking helper `Linearizer::current_bb` is `None` wherever control cannot arrive -- after a `goto`, and in a `switch` body before the first `case`. Both are valid C that has to be translated, and 21 sites across 8 functions took `.unwrap()`/`.expect()` on it instead of `current_or_unreachable_bb()`, so each turned a legal program into an internal compiler error: switch (x) { g() ? g() : g(); case 1: return 1; } int y = x ? ({ goto L; g(); }) : g(); L: return y; There were 21 of them because the block plumbing was copied: four lowerings -- both ternaries and both spellings of the GNU elvis operator -- each had their own `cbr`, two arm blocks, two `br merge`s, four `link_bb` calls and a phi of two `emit_phi_source`s, and complex integer division had a fifth copy without the phi. `emit_two_way` was already a generic builder using the safe accessor, with one caller. So the fix is the helper, not the unwraps. `emit_fork` is now the one place those blocks and edges are built, with `emit_diamond` (phi) and `emit_diamond_void` (arms that write where the caller can find them) over it. The four lowerings become one call each and `linearize.rs` loses ~150 lines of plumbing. Both the block the branch leaves *and* the block each arm ends in are read back through the accessor, because an arm is arbitrary code that may itself `goto` away. `emit_diamond` takes the phi width rather than deriving it from the type: a complex conditional merges addresses, so its phi is pointer-wide over a pointer type, and a function designator's `size_bits` is 0 where the merge wants 64. The short-circuit operators keep their own lowering. They are triangles, not diamonds: only one arm block exists, the other phi predecessor is the left operand's own block and its phi value is emitted before the branch, and they go through `branch_on`, whose constant case emits a plain `Br` and elides the merge edge -- which is how `1 && g()` compiles to no branch at all. Routing them through a builder that takes a condition pseudo would manufacture an empty second arm and a dead `cbr`. They take the accessor directly, and a test pins the elision so the tempting consolidation fails loudly. The atomic CAS retry loop is not a diamond either, for the same kind of reason. Site accounting: 12 retired by `emit_diamond`, 3 by `emit_diamond_void`, 6 by the accessor. `current_bb` with `.unwrap()` or `.expect()` now appears only in doc comments. The new CFG audit runs `cfg_inconsistency` over 13 conditional shapes in each of three placements -- reachable, before the first `case`, and after a `goto` -- against post-`remove_unreachable_blocks` CFGs, where a mislinked edge into a discarded block would outlive the block. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize.rs | 208 +++++++----------------------- cc/ir/linearize_atomic.rs | 9 +- cc/ir/linearize_emit.rs | 207 ++++++++++++++++++----------- cc/ir/test_linearize.rs | 156 ++++++++++++++++++++++ cc/tests/misc/mod.rs | 1 + cc/tests/misc/no_current_block.rs | 145 +++++++++++++++++++++ 6 files changed, 491 insertions(+), 235 deletions(-) create mode 100644 cc/tests/misc/no_current_block.rs diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index ef667cec2..6ed96ece0 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -4871,47 +4871,16 @@ impl<'a> Linearizer<'a> { else_expr: &Expr, result_typ: TypeId, ) -> PseudoId { - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - let cond_bool = self.linearize_condition(cond); - let cond_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - let ptr_typ = self.types.pointer_to(result_typ); let ptr_bits = self.target.pointer_width; - - self.switch_bb(then_bb); - let then_val = self.complex_arm_addr(then_expr, result_typ); - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let else_val = self.complex_arm_addr(else_expr, result_typ); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, ptr_typ, ptr_bits); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + ptr_typ, + ptr_bits, + |lin| lin.complex_arm_addr(then_expr, result_typ), + |lin| lin.complex_arm_addr(else_expr, result_typ), + ) } pub(crate) fn linearize_ternary( @@ -4984,48 +4953,23 @@ impl<'a> Linearizer<'a> { )); result } else { - // Impure: use control flow + phi for proper short-circuit evaluation - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - + // Impure: use control flow + phi for proper short-circuit evaluation. + // Each arm is converted inside its own block, where it is the only + // thing evaluated. let cond_bool = self.linearize_condition(cond); - let cond_end_bb = self.current_bb.unwrap(); - - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - - self.switch_bb(then_bb); - let then_val = self.linearize_expr(then_expr); - let then_val = self.conditional_arm(then_val, then_expr, result_typ, aggregate); - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let else_val = self.linearize_expr(else_expr); - let else_val = self.conditional_arm(else_val, else_expr, result_typ, aggregate); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, merge_typ, size); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, merge_typ, size); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, merge_typ, size); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + merge_typ, + size, + |lin| { + let val = lin.linearize_expr(then_expr); + lin.conditional_arm(val, then_expr, result_typ, aggregate) + }, + |lin| { + let val = lin.linearize_expr(else_expr); + lin.conditional_arm(val, else_expr, result_typ, aggregate) + }, + ) } } @@ -5073,50 +5017,21 @@ impl<'a> Linearizer<'a> { self.emit_compare_zero(evaluated, cond_typ) }; - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - let cond_end_bb = self.current_bb.unwrap(); - - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - let ptr_typ = self.types.pointer_to(result_typ); let ptr_bits = self.target.pointer_width; - - self.switch_bb(then_bb); - let then_val = if cond_complex { - self.complex_addr_at_precision(evaluated, cond_typ, result_typ) - } else { - self.promote_real_value_to_complex(evaluated, cond_typ, result_typ) - }; - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let else_val = self.complex_arm_addr(else_expr, result_typ); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, ptr_typ, ptr_bits); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, ptr_typ, ptr_bits); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + ptr_typ, + ptr_bits, + |lin| { + if cond_complex { + lin.complex_addr_at_precision(evaluated, cond_typ, result_typ) + } else { + lin.promote_real_value_to_complex(evaluated, cond_typ, result_typ) + } + }, + |lin| lin.complex_arm_addr(else_expr, result_typ), + ) } /// The value of a conditional expression whose constant condition @@ -5185,15 +5100,7 @@ impl<'a> Linearizer<'a> { // Impure right-hand side: it must not be evaluated when the condition // is true, so it needs its own block. - let then_bb = self.alloc_bb(); - let else_bb = self.alloc_bb(); - let merge_bb = self.alloc_bb(); - let cond_end_bb = self.current_bb.unwrap(); - - self.emit(Instruction::cbr(cond_bool, then_bb, else_bb)); - self.link_bb(cond_end_bb, then_bb); - self.link_bb(cond_end_bb, else_bb); - + // // The true value is the condition, converted to the result type -- done // *inside* the true block, where the ternary also converts its arms. // Converting before the `cbr` is equally correct as IR, and reads more @@ -5202,36 +5109,17 @@ impl<'a> Linearizer<'a> { // and the aarch64 backend then emits a branch on the wrong register. // That is a backend defect and is reported as one; this is not the // place to depend on it. - self.switch_bb(then_bb); - let then_val = self.convert_conditional_arm(cond_val, cond_typ, result_typ); - let then_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(then_end_bb, merge_bb); - - self.switch_bb(else_bb); - let mut else_val = self.linearize_expr(else_expr); - let else_typ = self.expr_type(else_expr); - else_val = self.convert_conditional_arm(else_val, else_typ, result_typ); - let else_end_bb = self.current_bb.unwrap(); - self.emit(Instruction::br(merge_bb)); - self.link_bb(else_end_bb, merge_bb); - - self.switch_bb(merge_bb); - let result = self.alloc_pseudo(); - let phi_pseudo = Pseudo::phi(result, result.0); - if let Some(func) = &mut self.current_func { - func.add_pseudo(phi_pseudo); - } - let mut phi_insn = Instruction::phi(result, result_typ, size); - let phisrc1 = - self.emit_phi_source(then_end_bb, then_val, result, merge_bb, result_typ, size); - phi_insn.phi_list.push((then_end_bb, phisrc1)); - let phisrc2 = - self.emit_phi_source(else_end_bb, else_val, result, merge_bb, result_typ, size); - phi_insn.phi_list.push((else_end_bb, phisrc2)); - self.emit(phi_insn); - - result + self.emit_diamond( + cond_bool, + result_typ, + size, + |lin| lin.convert_conditional_arm(cond_val, cond_typ, result_typ), + |lin| { + let val = lin.linearize_expr(else_expr); + let else_typ = lin.expr_type(else_expr); + lin.convert_conditional_arm(val, else_typ, result_typ) + }, + ) } /// Lower `__builtin_clrsb` and its wider siblings. diff --git a/cc/ir/linearize_atomic.rs b/cc/ir/linearize_atomic.rs index 933c7bea7..c1073cade 100644 --- a/cc/ir/linearize_atomic.rs +++ b/cc/ir/linearize_atomic.rs @@ -298,7 +298,12 @@ impl Linearizer<'_> { let loop_bb = self.alloc_bb(); let done_bb = self.alloc_bb(); - let entry_bb = self.current_bb.expect("atomic RMW outside a block"); + // Through the accessor rather than `expect`: `current_bb` is `None` + // wherever control cannot arrive -- a statement before a `switch`'s + // first `case`, or after a `goto` -- and this loop has to hang its + // blocks off something. `dce::remove_unreachable_blocks` takes the + // lot away again. + let entry_bb = self.current_or_unreachable_bb(); self.emit(Instruction::br(loop_bb)); self.link_bb(entry_bb, loop_bb); self.switch_bb(loop_bb); @@ -341,7 +346,7 @@ impl Linearizer<'_> { cas.memory_order = ORDER; self.emit(cas); - let cas_bb = self.current_bb.expect("atomic CAS outside a block"); + let cas_bb = self.current_or_unreachable_bb(); self.emit(Instruction::cbr(ok, done_bb, loop_bb)); self.link_bb(cas_bb, done_bb); self.link_bb(cas_bb, loop_bb); diff --git a/cc/ir/linearize_emit.rs b/cc/ir/linearize_emit.rs index 1e9e0b1a5..fe8842cf9 100644 --- a/cc/ir/linearize_emit.rs +++ b/cc/ir/linearize_emit.rs @@ -1611,61 +1611,53 @@ impl<'a> super::linearize::Linearizer<'a> { }; let cond = self.emit_int_binop(cmp, abs_c, abs_d, base_typ, base_size); - let small_bb = self.alloc_bb(); - let big_bb = self.alloc_bb(); - let done_bb = self.alloc_bb(); - let entry_bb = self.current_bb.expect("complex divide outside a block"); - self.emit(Instruction::cbr(cond, small_bb, big_bb)); - self.link_bb(entry_bb, small_bb); - self.link_bb(entry_bb, big_bb); - - // `|c| >= |d|`: r = d/c, denom = c + d*r, - // re = (a + b*r)/denom, im = (b - a*r)/denom. - self.switch_bb(big_bb); - let r = self.emit_int_binop(div, d, c, base_typ, base_size); - let dr = self.emit_int_binop(Opcode::Mul, d, r, base_typ, base_size); - let denom = self.emit_int_binop(Opcode::Add, c, dr, base_typ, base_size); - let br = self.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); - let num_re = self.emit_int_binop(Opcode::Add, a, br, base_typ, base_size); - let ar = self.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); - let num_im = self.emit_int_binop(Opcode::Sub, b, ar, base_typ, base_size); - self.store_complex_quotient( - result_addr, - (num_re, num_im), - denom, - div, - base_typ, - base_size, - base_bytes, - ); - let big_end = self.current_bb.expect("complex divide lost its block"); - self.emit(Instruction::br(done_bb)); - self.link_bb(big_end, done_bb); - - // `|c| < |d|`: r = c/d, denom = d + c*r, - // re = (a*r + b)/denom, im = (b*r - a)/denom. - self.switch_bb(small_bb); - let r = self.emit_int_binop(div, c, d, base_typ, base_size); - let cr = self.emit_int_binop(Opcode::Mul, c, r, base_typ, base_size); - let denom = self.emit_int_binop(Opcode::Add, d, cr, base_typ, base_size); - let ar = self.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); - let num_re = self.emit_int_binop(Opcode::Add, ar, b, base_typ, base_size); - let br = self.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); - let num_im = self.emit_int_binop(Opcode::Sub, br, a, base_typ, base_size); - self.store_complex_quotient( - result_addr, - (num_re, num_im), - denom, - div, - base_typ, - base_size, - base_bytes, + // Both arms write their halves to `result_addr`, so there is no value + // to merge and the void diamond serves: it builds the same blocks and + // edges, and reads `current_bb` back through the accessor that copes + // with a `goto` out of an arm. + self.emit_diamond_void( + cond, + // `|c| < |d|`: r = c/d, denom = d + c*r, + // re = (a*r + b)/denom, im = (b*r - a)/denom. + |lin| { + let r = lin.emit_int_binop(div, c, d, base_typ, base_size); + let cr = lin.emit_int_binop(Opcode::Mul, c, r, base_typ, base_size); + let denom = lin.emit_int_binop(Opcode::Add, d, cr, base_typ, base_size); + let ar = lin.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); + let num_re = lin.emit_int_binop(Opcode::Add, ar, b, base_typ, base_size); + let br = lin.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); + let num_im = lin.emit_int_binop(Opcode::Sub, br, a, base_typ, base_size); + lin.store_complex_quotient( + result_addr, + (num_re, num_im), + denom, + div, + base_typ, + base_size, + base_bytes, + ); + }, + // `|c| >= |d|`: r = d/c, denom = c + d*r, + // re = (a + b*r)/denom, im = (b - a*r)/denom. + |lin| { + let r = lin.emit_int_binop(div, d, c, base_typ, base_size); + let dr = lin.emit_int_binop(Opcode::Mul, d, r, base_typ, base_size); + let denom = lin.emit_int_binop(Opcode::Add, c, dr, base_typ, base_size); + let br = lin.emit_int_binop(Opcode::Mul, b, r, base_typ, base_size); + let num_re = lin.emit_int_binop(Opcode::Add, a, br, base_typ, base_size); + let ar = lin.emit_int_binop(Opcode::Mul, a, r, base_typ, base_size); + let num_im = lin.emit_int_binop(Opcode::Sub, b, ar, base_typ, base_size); + lin.store_complex_quotient( + result_addr, + (num_re, num_im), + denom, + div, + base_typ, + base_size, + base_bytes, + ); + }, ); - let small_end = self.current_bb.expect("complex divide lost its block"); - self.emit(Instruction::br(done_bb)); - self.link_bb(small_end, done_bb); - - self.switch_bb(done_bb); } /// Divide both numerators by the shared denominator and store the halves. @@ -1847,6 +1839,10 @@ impl<'a> super::linearize::Linearizer<'a> { /// `cond ? taken() : fallthrough()`, each arm in a block of its own and /// merged by a phi of type `typ`: for arms that may not both be /// evaluated. + /// + /// The phi is as wide as `typ` is. A construct whose merge width is not + /// its type's -- a pointer-merged complex arm, a function designator, + /// whose `size_bits` is zero -- calls [`Self::emit_diamond`] and states it. fn emit_two_way( &mut self, cond: PseudoId, @@ -1855,16 +1851,33 @@ impl<'a> super::linearize::Linearizer<'a> { fallthrough: impl FnOnce(&mut Self) -> PseudoId, ) -> PseudoId { let size = self.types.size_bits(typ); - let (taken_bb, fall_bb, merge_bb) = (self.alloc_bb(), self.alloc_bb(), self.alloc_bb()); - let from = self.current_or_unreachable_bb(); - self.emit(Instruction::cbr(cond, taken_bb, fall_bb)); - self.link_bb(from, taken_bb); - self.link_bb(from, fall_bb); + self.emit_diamond(cond, typ, size, taken, fallthrough) + } - let arms = [ - self.emit_arm(taken_bb, merge_bb, taken), - self.emit_arm(fall_bb, merge_bb, fallthrough), - ]; + /// `cond ? then_arm() : else_arm()`, merged by a `size`-bit phi of type + /// `typ`. + /// + /// The one place a two-armed conditional's blocks and edges are built, so + /// the one place that has to know `current_bb` is `None` wherever control + /// cannot arrive -- see [`Linearizer::current_or_unreachable_bb`]. Both + /// the block the branch leaves and the block each arm *ends* in are read + /// back through that accessor: an arm is arbitrary code and may itself + /// `goto` away, so `x ? ({ goto L; g(); }) : g()` has no block at the end + /// of its true arm. + /// + /// `size` is passed rather than taken from `typ` because the two are not + /// always the same: a complex conditional merges *addresses*, so its phi + /// is pointer-wide over a pointer type, and a function designator's + /// `size_bits` is 0 where the merge wants a pointer's 64. + pub(crate) fn emit_diamond( + &mut self, + cond: PseudoId, + typ: TypeId, + size: u32, + then_arm: impl FnOnce(&mut Self) -> PseudoId, + else_arm: impl FnOnce(&mut Self) -> PseudoId, + ) -> PseudoId { + let (merge_bb, arms) = self.emit_fork(cond, then_arm, else_arm); self.switch_bb(merge_bb); let result = self.alloc_pseudo(); @@ -1880,14 +1893,52 @@ impl<'a> super::linearize::Linearizer<'a> { result } - /// One arm of [`Self::emit_two_way`]: `arm` evaluated in `bb`, which then + /// [`Self::emit_diamond`] for arms that produce no value: they write + /// their results where the caller can find them, so there is nothing to + /// merge and no phi. Leaves the cursor on the merge block. + pub(crate) fn emit_diamond_void( + &mut self, + cond: PseudoId, + then_arm: impl FnOnce(&mut Self), + else_arm: impl FnOnce(&mut Self), + ) { + let (merge_bb, _) = self.emit_fork(cond, then_arm, else_arm); + self.switch_bb(merge_bb); + } + + /// The block plumbing both diamonds share: branch on `cond` into a block + /// per arm, run each arm, and join them. + /// + /// Returns the merge block -- which the caller has *not* switched to yet, + /// so a phi can be placed at its head -- and, per arm, the block it ended + /// in and whatever it produced. + fn emit_fork( + &mut self, + cond: PseudoId, + then_arm: impl FnOnce(&mut Self) -> T, + else_arm: impl FnOnce(&mut Self) -> T, + ) -> (BasicBlockId, [(BasicBlockId, T); 2]) { + let (then_bb, else_bb, merge_bb) = (self.alloc_bb(), self.alloc_bb(), self.alloc_bb()); + let from = self.current_or_unreachable_bb(); + self.emit(Instruction::cbr(cond, then_bb, else_bb)); + self.link_bb(from, then_bb); + self.link_bb(from, else_bb); + + let arms = [ + self.emit_arm(then_bb, merge_bb, then_arm), + self.emit_arm(else_bb, merge_bb, else_arm), + ]; + (merge_bb, arms) + } + + /// One arm of [`Self::emit_fork`]: `arm` evaluated in `bb`, which then /// branches to `merge`. Returns the block the arm ended in, and its value. - fn emit_arm( + fn emit_arm( &mut self, bb: BasicBlockId, merge: BasicBlockId, - arm: impl FnOnce(&mut Self) -> PseudoId, - ) -> (BasicBlockId, PseudoId) { + arm: impl FnOnce(&mut Self) -> T, + ) -> (BasicBlockId, T) { self.switch_bb(bb); let value = arm(self); let end = self.current_or_unreachable_bb(); @@ -2364,7 +2415,13 @@ impl<'a> super::linearize::Linearizer<'a> { // Get the block where LHS evaluation ended (may differ from initial block // if LHS contains nested control flow) - let lhs_end_bb = self.current_bb.unwrap(); + // + // Read through the accessor, not `unwrap`: control cannot arrive at a + // statement before a `switch`'s first `case` or after a `goto`, and the + // phi below needs a real predecessor block to hold its source. See + // `current_or_unreachable_bb`. `branch_on` re-reads `current_bb` + // itself, so it sees the same block. + let lhs_end_bb = self.current_or_unreachable_bb(); // Branch: if LHS is false, go to merge (result = 0); else evaluate RHS self.branch_on(left_cond, eval_b_bb, merge_bb); @@ -2374,8 +2431,9 @@ impl<'a> super::linearize::Linearizer<'a> { let right_bool = self.linearize_condition(right); // Get the actual block where RHS evaluation ended (may differ from eval_b_bb - // if RHS contains nested control flow like another &&/||) - let rhs_end_bb = self.current_bb.unwrap(); + // if RHS contains nested control flow like another &&/||), and may be + // gone entirely where the RHS jumped away: `x && ({ goto L; g(); })`. + let rhs_end_bb = self.current_or_unreachable_bb(); // Branch to merge self.emit(Instruction::br(merge_bb)); @@ -2427,7 +2485,9 @@ impl<'a> super::linearize::Linearizer<'a> { // Get the block where LHS evaluation ended (may differ from initial block // if LHS contains nested control flow) - let lhs_end_bb = self.current_bb.unwrap(); + // + // Through the accessor for the reason `emit_logical_and` records. + let lhs_end_bb = self.current_or_unreachable_bb(); // Branch: if LHS is true, go to merge (result = 1); else evaluate RHS self.branch_on(left_cond, merge_bb, eval_b_bb); @@ -2437,8 +2497,9 @@ impl<'a> super::linearize::Linearizer<'a> { let right_bool = self.linearize_condition(right); // Get the actual block where RHS evaluation ended (may differ from eval_b_bb - // if RHS contains nested control flow like another &&/||) - let rhs_end_bb = self.current_bb.unwrap(); + // if RHS contains nested control flow like another &&/||), and may be + // gone entirely where the RHS jumped away: `x || ({ goto L; g(); })`. + let rhs_end_bb = self.current_or_unreachable_bb(); // Branch to merge self.emit(Instruction::br(merge_bb)); diff --git a/cc/ir/test_linearize.rs b/cc/ir/test_linearize.rs index e6d1c91ca..c9bb641b0 100644 --- a/cc/ir/test_linearize.rs +++ b/cc/ir/test_linearize.rs @@ -9309,3 +9309,159 @@ fn test_volatile_access_to_a_plain_local_blocks_promotion() { "SSA promotion turned a volatile access into a register copy" ); } + +/// A short-circuit operator with a constant left operand emits no branch. +/// +/// `emit_logical_and`/`emit_logical_or` are *triangles*, not diamonds: only one +/// arm block exists, the other phi predecessor is the left operand's own block, +/// and that block's phi value is emitted before the branch. They also branch +/// through `branch_on(Controlling, ..)`, whose `Constant` case deliberately +/// emits a plain `Br` and elides the merge edge entirely. +/// +/// So they must not be folded into the generic two-way/diamond builder, which +/// takes a `PseudoId` condition and always emits a `Cbr`: `1 && g()` would +/// regain a dead conditional branch and a second, empty arm block. This test is +/// the guard on that -- it fails if the short-circuit lowerings are ever routed +/// through the diamond helper. +#[test] +fn a_constant_short_circuit_operand_emits_no_branch() { + let target = Target::host(); + + let count_cbr = |src: &str| -> usize { + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + func.blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| i.op == Opcode::Cbr) + .count() + }; + + // A constant controlling operand is decided at compile time. + for src in [ + "int g(void); int f(void) { return 1 && g(); }", + "int g(void); int f(void) { return 0 || g(); }", + ] { + assert_eq!( + count_cbr(src), + 0, + "a constant short-circuit operand needs no branch: {src}" + ); + } + + // The control: a runtime operand does branch, so the check above is not + // passing because nothing ever emits a Cbr. + for src in [ + "int g(void); int f(int x) { return x && g(); }", + "int g(void); int f(int x) { return x || g(); }", + ] { + assert_eq!( + count_cbr(src), + 1, + "a runtime short-circuit operand branches exactly once: {src}" + ); + } +} + +/// Every lowering that builds a two-armed conditional keeps the CFG +/// consistent, in all three places its blocks can be built: where control +/// reaches them, where it cannot -- before a `switch`'s first `case`, and +/// after a `goto` -- and where an arm jumps out from under it. +/// +/// These are the shapes that went through `self.current_bb.unwrap()` and so +/// crashed the compiler outright on the last two. They now share +/// `emit_diamond`, or read the block back through +/// `current_or_unreachable_bb`, which starts a block nothing branches to; the +/// point of auditing the CFG rather than only that lowering finished is that +/// such a block is *removed* again by `dce::remove_unreachable_blocks`, and a +/// mislinked edge into or out of it would outlive it. +#[test] +fn conditional_lowerings_keep_the_cfg_consistent() { + let target = Target::host(); + + // (tag, declarations, statement). The statement is placed reachable, then + // before a `switch`'s first `case`, then after a `goto`. + let shapes = [ + ("ternary", "int g(void);", "y = g() ? g() : g();"), + ("logical_and", "int g(void);", "y = g() && g();"), + ("logical_or", "int g(void);", "y = g() || g();"), + ("elvis", "int g(void);", "y = g() ?: g();"), + ( + "nested_ternary", + "int g(void);", + "y = g() ? (g() ? g() : g()) : g();", + ), + ( + "and_in_ternary", + "int g(void);", + "y = g() ? (g() && g()) : g();", + ), + ( + "sqrt_errno", + "double sqrt(double); double d;", + "d = sqrt(d);", + ), + ( + "complex_ternary", + "int g(void); _Complex double h(void);", + "(void)(g() ? h() : h());", + ), + ( + "complex_elvis", + "_Complex double h(void);", + "(void)(h() ?: h());", + ), + ( + "complex_int_div", + "_Complex int ci(void);", + "(void)(ci() / ci());", + ), + ( + "atomic_nand", + "_Atomic int a;", + "y = __atomic_fetch_nand(&a, 1, 5);", + ), + // An arm that jumps away leaves *that arm* without a block, which is a + // different read from the condition's. + ( + "goto_out_of_then_arm", + "int g(void);", + "y = x ? ({ goto L; g(); }) : g();", + ), + ( + "goto_out_of_else_arm", + "int g(void);", + "y = x ? g() : ({ goto L; g(); });", + ), + ]; + + for (tag, decls, stmt) in shapes { + let bodies = [ + ("reachable", stmt.to_string()), + ( + "before_first_case", + format!("switch (x) {{ {stmt} case 1: y = 1; }}"), + ), + ("after_goto", format!("goto L; {stmt}")), + ]; + + for (where_, body) in bodies { + let src = format!("{decls}\nint f(int x) {{ int y = 0; {body} L: return y; }}\n"); + let module = linearize_source(&src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + assert!( + cfg_inconsistency(func).is_none(), + "{tag} / {where_}: {}\nsource: {src}", + cfg_inconsistency(func).unwrap() + ); + } + } +} diff --git a/cc/tests/misc/mod.rs b/cc/tests/misc/mod.rs index 68ce708c0..ee82ffcae 100644 --- a/cc/tests/misc/mod.rs +++ b/cc/tests/misc/mod.rs @@ -11,5 +11,6 @@ // Tests for setjmp/longjmp, statement expressions, and other misc features. // +mod no_current_block; mod setjmp; mod stmt_expr; diff --git a/cc/tests/misc/no_current_block.rs b/cc/tests/misc/no_current_block.rs new file mode 100644 index 000000000..27c84168f --- /dev/null +++ b/cc/tests/misc/no_current_block.rs @@ -0,0 +1,145 @@ +// +// Copyright (c) 2025-2026 Jeff Garzik +// +// This file is part of the posixutils-rs project covered under +// the MIT License. For the full license text, please see the LICENSE +// file in the root directory of this project. +// SPDX-License-Identifier: MIT +// +// Valid C for which the linearizer holds no current basic block. +// +// `Linearizer::current_bb` is legitimately `None` in two situations the +// language allows: after a `goto`, and inside a `switch` body before the first +// `case` label. Code in either place is unreachable but well-formed, and the +// standard requires it to be translated, not rejected -- and certainly not to +// crash the compiler. +// +// Every lowering that ends a block has to read `current_bb` back afterwards. +// The ones that reached for `.unwrap()` instead turned each of these programs +// into an internal compiler error. `current_or_unreachable_bb()` is the +// accessor that starts a fresh unreachable block rather than panicking. +// +// These are compile-only where the construct is genuinely unreachable, because +// there is no answer to assert; the two that are reachable check the answer as +// well. +// + +use crate::common::{compile_and_run, compile_expect_ok}; + +/// A statement before a `switch`'s first `case` has no enclosing block. +/// +/// Each of these ended in a panic, one per lowering: the ternary, the two +/// short-circuit operators, the complex ternary, both spellings of the GNU +/// elvis operator, complex integer division, and the atomic CAS loop. +#[test] +fn misc_a_statement_before_the_first_case_does_not_ice() { + for (tag, body) in [ + ("ternary", "g() ? g() : g();"), + ("logical_and", "(void)(g() && g());"), + ("logical_or", "(void)(g() || g());"), + ("elvis", "(void)(g() ?: g());"), + ("nested_ternary", "(void)(g() ? (g() ? g() : g()) : g());"), + ("and_in_ternary", "(void)(g() ? (g() && g()) : g());"), + ] { + compile_expect_ok( + &format!("nocur_switch_{tag}"), + &format!( + "int g(void);\n\ + int f(int x) {{ switch (x) {{ {body} case 1: return 1; }} return 0; }}\n" + ), + ); + } + + compile_expect_ok( + "nocur_switch_complex_ternary", + "int g2(void);\n_Complex double h(void);\n\ + int f(int x) { switch (x) { (void)(g2() ? h() : h()); case 1: return 1; } return 0; }\n", + ); + compile_expect_ok( + "nocur_switch_complex_elvis", + "_Complex double h(void);\n\ + int f(int x) { switch (x) { (void)(h() ?: h()); case 1: return 1; } return 0; }\n", + ); + compile_expect_ok( + "nocur_switch_complex_int_div", + "_Complex int ci(void);\n\ + int f(int x) { switch (x) { ci() / ci(); case 1: return 1; } return 0; }\n", + ); + compile_expect_ok( + "nocur_switch_atomic_nand", + "_Atomic int a;\n\ + int f(int x) { switch (x) { __atomic_fetch_nand(&a, 1, 5); case 1: return 1; } \ + return 0; }\n", + ); +} + +/// The same shapes after a `goto`, which is the other way to have no block. +#[test] +fn misc_a_statement_after_a_goto_does_not_ice() { + for (tag, body) in [ + ("ternary", "x = g() ? g() : g();"), + ("logical_and", "x = g() && g();"), + ("logical_or", "x = g() || g();"), + ("elvis", "x = g() ?: g();"), + ] { + compile_expect_ok( + &format!("nocur_goto_{tag}"), + &format!("int g(void);\nint f(int x) {{ goto L; {body} L: return x; }}\n"), + ); + } +} + +/// A `goto` out of one arm leaves *that arm* without a block, which is a +/// different site from the condition's. +/// +/// These are why the fix belongs in the shared diamond builder rather than on +/// the condition alone: the helper reads the block back at the end of each arm +/// too. +#[test] +fn misc_a_goto_out_of_a_conditional_arm_does_not_ice() { + for (tag, init) in [ + ("then", "x ? ({ goto L; g(); }) : g()"), + ("else", "x ? g() : ({ goto L; g(); })"), + ("and", "x && ({ goto L; g(); })"), + ("or", "x || ({ goto L; g(); })"), + ("elvis", "g() ?: ({ goto L; g(); })"), + ("both", "x ? ({ goto L; g(); }) : ({ goto L; g(); })"), + ] { + compile_expect_ok( + &format!("nocur_arm_{tag}"), + &format!("int g(void);\nint f(int x) {{ int y = {init}; L: return y; }}\n"), + ); + } +} + +/// Unreachable code before the first `case` is discarded, and the reachable +/// part of the same function still gives the right answer. +#[test] +fn misc_an_unreachable_statement_does_not_change_the_answer() { + let code = r#" +int calls; +int g(void) { calls++; return 3; } + +int f(int x) +{ + switch (x) { + g() ? g() : g(); /* unreachable: never selected */ + (void)(g() && g()); + case 1: + return 1; + default: + return 2; + } +} + +int main(void) +{ + if (f(1) != 1) return 1; + if (f(0) != 2) return 2; + /* Nothing before the first case may run. */ + if (calls != 0) return 3; + return 0; +} +"#; + assert_eq!(compile_and_run("nocur_answer", code, &[]), 0); +} From 36f631894e7db490c0512de05bfcfbbb6b3f6b86 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 01:38:34 -0400 Subject: [PATCH 07/17] cc: move every block of bytes through the one helper that bounds it `memexpand::block_chunks` descends 8/4/2/1 and lands the tail exactly, and `BlockOp::limit` bounds a copy or a fill at `INLINE_LIMIT_BYTES`. Three places in the IR wrote that descent out again instead, and each lost something in the copy: two rounded the size *up*, one had no bound and no capacity clamp. `memexpand`'s own header records the incident this rule came from, and `codegen_struct_copy_across_the_inline_threshold` records the two bugs that came out of fixing it -- one of which was a second copy of the loop, written as `while offset < size` stepping 8. These are the third, fourth and fifth copies. `emit_aggregate_zero` had no cap, so `char buf[N] = {0}` emitted one store per chunk for every N: 8 KB cost 2081 instructions and 1 MB did not finish compiling in twenty-five minutes, while the sibling two functions away had capped all along. It now asks for a `Memset` and lets `memexpand` choose stores or a call, which deletes the ladder rather than bounding it -- the bound, the expansion and the fill-byte handling are all already written there, and `memexpand::run` executes at -O0 too, so nothing is conditional on optimizing. 1 MB now compiles in under a second. The `rvalue_addr` handling that a `Sym` needs before it can be passed to a libc call came out of `emit_block_copy_call` into `block_dest_addr` rather than being written a second time. The inliner's implicit-parameter copy stepped 8 regardless of width, so a 12-byte struct moved 16 bytes -- over-reading the argument and over-writing the callee's local. It takes `block_chunks` now. No bound is needed and a comment says why: the copy is at most 32 bytes, being a `long double _Complex` at worst. The two-register return stored its high half at a hardcoded 64 bits where the half is 1..8, and now agrees with the load in `emit_two_reg_return`, which was already narrowing correctly. `store_string_units` handles every literal kind, clamps to the destination and strides by the element width. The nested-array element path did none of those: it accepted all four kinds and then handled only the narrow one, dropping a wide element silently; it stepped the destination by raw bytes where a wide element is two or four; and with no clamp at all `char s[1][3] = {"hello"}` stored five bytes into three, two of them past the local. Routing it through `store_string_units` retires all three together with the copy. That left the tail. C17 6.7.9p21 zeroes what an initializer does not reach, which the `InitList` arm did by calling `emit_aggregate_zero` first and the bare-string arm did not -- so `char buf[8] = "hi"` wrote three bytes and left five, visible only on re-execution because the prologue zeroes the frame once. `store_string_units` now fills from the terminator to the end, under a `StringTail` the caller passes: the three sites reached from an initializer list have already been zeroed and say so, so no stores are doubled at -O0. It fills through the same bounded `Memset`, which is why this had to follow the cap and not precede it. `test_compound_literal_zero_init_lvalue` counted the zero-init's stores in raw linearizer output, which is the shape that moved. It now asserts the linearizer asks for the `memset`, runs `memexpand`, and holds the stores to the same account, so it still proves 6.7.8p21. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/inline.rs | 118 +++++++++++++++++++++-- cc/ir/linearize_emit.rs | 155 ++++++++++++++---------------- cc/ir/linearize_stmt.rs | 154 +++++++++++++++++++----------- cc/ir/test_linearize.rs | 179 ++++++++++++++++++++++++++++++++++- cc/tests/c99/initializers.rs | 144 ++++++++++++++++++++++++++++ cc/tests/codegen/misc.rs | 162 +++++++++++++++++++++++++++++++ 6 files changed, 767 insertions(+), 145 deletions(-) diff --git a/cc/ir/inline.rs b/cc/ir/inline.rs index 99818ee78..0a3da12fd 100644 --- a/cc/ir/inline.rs +++ b/cc/ir/inline.rs @@ -11,6 +11,7 @@ // (InstCombine, DCE) see the inlined code. // +use super::memexpand; use super::{ BasicBlock, BasicBlockId, Function, Instruction, Module, Opcode, Pseudo, PseudoId, PseudoKind, }; @@ -674,6 +675,23 @@ fn clone_instruction( let remapped_low = ctx.remap_pseudo(insn.src[0], callee_func); let remapped_high = ctx.remap_pseudo(insn.src[1], callee_func); + // The high half is whatever is left past the first + // eight bytes, which is 1..=8 of them: + // `returns_reg_aggregate` admits 9..=16 bytes, so a + // 12-byte struct leaves four. Stored at a hardcoded + // 64 bits it overran the result local by four -- and + // it disagreed with the *load* in + // `emit_two_reg_return`, which has always narrowed the + // high half to `min(64, struct_size - 64)`. + let high_bits = match insn.size.checked_sub(64) { + Some(rest) if rest > 0 => rest.min(64), + // Two registers means more than eight bytes, so + // this is not a shape `emit_two_reg_return` + // produces. Keep the old width rather than emit a + // store of no bits at all. + _ => 64, + }; + let mut store_low = Instruction::store( remapped_low, target, @@ -689,7 +707,7 @@ fn clone_instruction( target, 8, insn.typ.unwrap_or(crate::types::TypeId::INVALID), - 64, + high_bits, ); store_high.pos = insn.pos; result.push(store_high); @@ -1142,8 +1160,25 @@ fn inline_call_site( continue; } - let mut offset = 0i64; - while (offset as usize) < copy.size_bytes { + // The shared 8/4/2/1 descent, in the chunks every other block + // move in the compiler uses. Stepping 8 to `size_bytes` + // instead rounds the size *up*: a 12-byte + // `struct P { float x, y, z; }` moved 16 bytes, over-reading + // the caller's argument and over-writing the callee's local + // -- the same defect the linearizer's parameter prologue and + // sret return path each had, spelled the same way. + // + // No upper bound here, deliberately: unlike a copy the + // program wrote, `size_bytes` is at most 32 -- a + // `long double _Complex` -- because only a complex value or a + // two-register aggregate is recorded as an implicit parameter + // copy. So the unroll cannot run away and needs no + // `memexpand::INLINE_LIMIT_BYTES` cap. This pass builds into + // a `Vec` rather than through `Linearizer::emit` + // and has no `TypeTable`, so it takes the offsets and widths + // and keeps `qword_type` as the access type, exactly as the + // register-sized case above does. + for (offset, chunk) in memexpand::block_chunks(copy.size_bytes as i64) { let temp = ctx.alloc_pseudo_id(); implicit_copy_pseudos.push(Pseudo::undef(temp)); copy_insns.push(Instruction::load( @@ -1151,16 +1186,15 @@ fn inline_call_site( call_arg, offset, copy.qword_type, - 64, + chunk.bits(), )); copy_insns.push(Instruction::store( temp, remapped_local, offset, copy.qword_type, - 64, + chunk.bits(), )); - offset += 8; } } // Insert copies at the beginning of the entry block @@ -2450,6 +2484,78 @@ mod tests { callee } + /// An implicit parameter copy moves exactly the object's bytes, in the + /// 8/4/2/1 chunks `memexpand::block_chunks` gives every other block move. + /// + /// `while offset < size_bytes { load 64; store 64; offset += 8 }` rounds + /// *up*: a 12-byte `struct P { float x, y, z; }` moved 16 bytes, + /// over-reading the caller's argument and over-writing the callee's local. + /// The fourth bytes are usually frame padding, so a program can be correct + /// and still be reading memory that does not belong to the object -- and + /// would fault if it ended a page. + #[test] + fn test_implicit_param_copy_moves_no_more_than_the_object() { + let types = TypeTable::new(&Target::host()); + + // `static void callee(struct P p)`, with the twelve bytes of `p` + // arriving by address and copied into the callee's own local. + let mut callee = Function::new("callee", types.void_id); + callee.add_param("p", types.void_ptr_id); + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.insns.push(Instruction::new(Opcode::Entry)); + bb.insns.push(Instruction::ret(None)); + callee.add_block(bb); + callee.entry = BasicBlockId(0); + callee.add_pseudo(Pseudo::sym(PseudoId(0), "p.0".to_string())); + callee.next_pseudo = 1; + callee + .implicit_param_copies + .push(crate::ir::ImplicitParamCopy { + arg_index: 0, + local_sym: PseudoId(0), + size_bytes: 12, + qword_type: types.long_id, + arg_is_address: true, + }); + + let mut caller = Function::new("caller", types.void_id); + let mut cb = BasicBlock::new(BasicBlockId(0)); + cb.insns.push(Instruction::new(Opcode::Entry)); + cb.insns.push(Instruction::call( + None, + "callee", + vec![PseudoId(0)], + vec![types.void_ptr_id], + types.void_id, + 0, + )); + cb.insns.push(Instruction::ret(None)); + caller.add_block(cb); + caller.entry = BasicBlockId(0); + caller.add_pseudo(Pseudo::reg(PseudoId(0), 0)); + caller.next_pseudo = 1; + + assert!(inline_call_site(&mut caller, 0, 1, &callee)); + + let moves: Vec<(Opcode, i64, u32)> = caller + .blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.offset, i.size)) + .collect(); + assert_eq!( + moves, + vec![ + (Opcode::Load, 0, 64), + (Opcode::Store, 0, 64), + (Opcode::Load, 8, 32), + (Opcode::Store, 8, 32), + ], + "a 12-byte object has four bytes at offset 8, not eight" + ); + } + /// Each inlined copy of a function that takes a label's address names /// its own clone of the block, in the caller. /// diff --git a/cc/ir/linearize_emit.rs b/cc/ir/linearize_emit.rs index fe8842cf9..26cf9e63f 100644 --- a/cc/ir/linearize_emit.rs +++ b/cc/ir/linearize_emit.rs @@ -190,66 +190,47 @@ impl<'a> super::linearize::Linearizer<'a> { } } - /// Emit stores to zero-initialize an aggregate (struct, union, or array) - /// This handles C99 6.7.8p19: uninitialized members must be zero-initialized + /// Zero a whole aggregate (struct, union or array), which is what C17 + /// 6.7.9p19 asks for before an initializer list is applied: every member + /// the list does not reach is initialized as a static object would be. pub(crate) fn emit_aggregate_zero(&mut self, base_sym: PseudoId, typ: TypeId) { - let total_bytes = self.types.size_bytes(typ); - let mut offset: i64 = 0; - - // Create a zero constant for 64-bit stores - let zero64 = self.emit_const(0, self.types.long_id); - - // Zero in 8-byte chunks - while offset + 8 <= total_bytes as i64 { - self.emit(Instruction::store( - zero64, - base_sym, - offset, - self.types.long_id, - 64, - )); - offset += 8; - } + let total_bytes = self.types.size_bytes(typ) as i64; + self.emit_block_zero(base_sym, 0, total_bytes); + } - // Handle remaining bytes (if any) - if offset < total_bytes as i64 { - let remaining = total_bytes as i64 - offset; - if remaining >= 4 { - let zero32 = self.emit_const(0, self.types.int_id); - self.emit(Instruction::store( - zero32, - base_sym, - offset, - self.types.int_id, - 32, - )); - offset += 4; - } - if offset < total_bytes as i64 { - let remaining = total_bytes as i64 - offset; - if remaining >= 2 { - let zero16 = self.emit_const(0, self.types.short_id); - self.emit(Instruction::store( - zero16, - base_sym, - offset, - self.types.short_id, - 16, - )); - offset += 2; - } - if offset < total_bytes as i64 { - let zero8 = self.emit_const(0, self.types.char_id); - self.emit(Instruction::store( - zero8, - base_sym, - offset, - self.types.char_id, - 8, - )); - } - } + /// Emit a fill of `size_bytes` zero bytes at `dst` + `dst_base_offset`. + /// + /// One `Opcode::Memset`, which `memexpand` turns into stores when the run + /// is short and leaves as a call when it is not. So the bound on the + /// unroll is `memexpand::INLINE_LIMIT_BYTES` -- the one every block memory + /// operation in the compiler shares -- rather than another copy of the + /// 8/4/2/1 descent with a cap of its own, which is what this was: a + /// hand-rolled ladder with **no** upper bound at all, so + /// `char buf[N] = {0}` emitted one store per chunk for any N. 8 KB cost + /// 2081 instructions in the function body and 1 MB did not finish + /// compiling in 25 minutes, while its sibling + /// [`Self::emit_block_copy_at_offset`] had capped at the shared limit all + /// along. + /// + /// `memexpand::run` runs at every optimization level, `-O0` included, so + /// the expansion does not depend on optimizing -- the same reason the + /// opcode a program's own `memset` becomes is expanded there rather than + /// here. + pub(crate) fn emit_block_zero(&mut self, dst: PseudoId, dst_base_offset: i64, size_bytes: i64) { + if size_bytes <= 0 { + return; } + let dst_ptr = self.block_dest_addr(dst, dst_base_offset); + let byte = self.emit_const(0, self.types.int_id); + let n = self.emit_const(size_bytes as i128, self.types.ulong_id); + let result = self.alloc_pseudo(); + self.emit( + Instruction::new(Opcode::Memset) + .with_func(self.library_function_name("memset")) + .with_target(result) + .with_src3(dst_ptr, byte, n) + .with_type_and_size(self.types.void_ptr_id, 64), + ); } /// Emit a block copy from src to dst using integer chunks. @@ -288,10 +269,37 @@ impl<'a> super::linearize::Linearizer<'a> { } } - /// The same copy as a `memcpy` call. + /// The address `dst` + `dst_base_offset` names, as a block memory + /// operation takes it. + /// + /// These opcodes take addresses. A `Sym` pseudo names a local's *storage*, + /// not a pointer to it -- a `Store` can name it directly, a call cannot. + /// Passing the Sym itself handed `memcpy` a meaningless value and + /// segfaulted every copy over the threshold. `rvalue_addr` is the existing + /// answer to this question and returns a non-Sym pseudo unchanged. /// - /// `dst_base_offset` is folded into the destination pointer first, since - /// `memcpy` takes an address rather than a base and a displacement. + /// `dst_base_offset` is folded into the pointer, since these take an + /// address rather than a base and a displacement. + fn block_dest_addr(&mut self, dst: PseudoId, dst_base_offset: i64) -> PseudoId { + let void_ptr = self.types.void_ptr_id; + let dst = self.rvalue_addr(dst, void_ptr); + if dst_base_offset == 0 { + return dst; + } + let off = self.emit_const(dst_base_offset as i128, self.types.long_id); + let adjusted = self.alloc_reg_pseudo(); + self.emit(Instruction::binop( + Opcode::Add, + adjusted, + dst, + off, + void_ptr, + 64, + )); + adjusted + } + + /// The same copy as a `memcpy` call. fn emit_block_copy_call( &mut self, dst: PseudoId, @@ -299,31 +307,8 @@ impl<'a> super::linearize::Linearizer<'a> { src: PseudoId, size_bytes: i64, ) { - // `memcpy` takes addresses. A `Sym` pseudo names a local's *storage*, - // not a pointer to it -- the inline path could store through it - // directly, this one cannot. Passing the Sym itself handed memcpy a - // meaningless value and segfaulted every copy over the threshold. - // `rvalue_addr` is the existing answer to this question and returns a - // non-Sym pseudo unchanged. - let void_ptr = self.types.void_ptr_id; - let dst = self.rvalue_addr(dst, void_ptr); - let src = self.rvalue_addr(src, void_ptr); - - let dst_ptr = if dst_base_offset == 0 { - dst - } else { - let off = self.emit_const(dst_base_offset as i128, self.types.long_id); - let adjusted = self.alloc_reg_pseudo(); - self.emit(Instruction::binop( - Opcode::Add, - adjusted, - dst, - off, - self.types.void_ptr_id, - 64, - )); - adjusted - }; + let dst_ptr = self.block_dest_addr(dst, dst_base_offset); + let src = self.rvalue_addr(src, self.types.void_ptr_id); let n = self.emit_const(size_bytes as i128, self.types.ulong_id); let result = self.alloc_pseudo(); self.emit( diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index d27e991d6..940df5a91 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -37,6 +37,24 @@ fn return_value_ness_violation(pos: Position, msg: &str) { use crate::types::{TypeId, TypeKind, TypeModifiers}; +/// Whether [`Linearizer::store_string_units`] owes the destination's tail a +/// zero fill. +/// +/// C17 6.7.9p21 zeroes every element a string initializer does not reach, and +/// exactly one of the two spellings already has that covered: the braced form +/// reaches the array through an initializer list, and each list is preceded by +/// a whole-object [`Linearizer::emit_aggregate_zero`]. Zeroing again there +/// would double the stores at `-O0`, where no `dse` runs to remove them. +#[derive(Clone, Copy)] +pub(crate) enum StringTail { + /// The destination is already zero: the caller zeroed the whole aggregate + /// before walking the initializer list. + AlreadyZero, + /// Nothing has written the destination yet, so the tail is this call's to + /// fill. + Zero, +} + /// Which construct a jump leaves, for `unwind_vla_marks`. /// /// `break` leaves the innermost loop *or* switch; `continue` leaves the @@ -545,7 +563,18 @@ impl<'a> super::linearize::Linearizer<'a> { // scalar case below and stored the literal's *address* // into the array's first element. if self.types.kind(typ) == TypeKind::Array { - self.store_string_units(sym_id, 0, typ, &init.kind, &units); + // Nothing has zeroed this local -- the `InitList` arm + // above calls `emit_aggregate_zero` and this one never + // did -- so the elements past the literal are + // `store_string_units`' to fill. + self.store_string_units( + sym_id, + 0, + typ, + &init.kind, + &units, + StringTail::Zero, + ); } else { // Pointer initialized with a string literal — store the address let val = self.linearize_expr(init); @@ -956,12 +985,15 @@ impl<'a> super::linearize::Linearizer<'a> { if let [only] = elements { if only.designators.is_empty() { if let Some(units) = Self::string_literal_units(&only.value.kind) { + // Every initializer list is preceded by a + // whole-object zero, so the tail is done. self.store_string_units( base_sym, base_offset, typ, &only.value.kind, &units, + StringTail::AlreadyZero, ); return; } @@ -981,53 +1013,33 @@ impl<'a> super::linearize::Linearizer<'a> { continue; }; let offset = base_offset + element_index * elem_size as i64; - // When a string literal initializes a char array element - // (e.g., char arr[3][4] = {"Sun", "Mon", "Tue"}), handle - // it as a string copy rather than recursing into individual - // char stores. The recursion would treat the string as a - // pointer instead of inline data. - let is_string_for_char_array = elem_is_aggregate - && list.len() == 1 - && matches!( - list[0].value.kind, - ExprKind::StringLit(_) - | ExprKind::WideStringLit(_) - | ExprKind::Utf16StringLit(_) - | ExprKind::Utf32StringLit(_) - ) - && self.types.kind(elem_type) == TypeKind::Array; - if is_string_for_char_array { - // Emit byte-by-byte stores for the string content - if let ExprKind::StringLit(s) = &list[0].value.kind { - let char_type = self - .types - .base_type(elem_type) - .unwrap_or(self.types.char_id); - let char_bits = self.types.size_bits(char_type); - for (i, ch) in s.chars().enumerate() { - let byte_val = self.emit_const(ch as u8 as i128, self.types.int_id); - self.emit(Instruction::store( - byte_val, - base_sym, - offset + i as i64, - char_type, - char_bits, - )); - } - // Null terminator + zero fill - let arr_bytes = self.types.size_bytes(elem_type); - let str_len = s.chars().count(); - for i in str_len..arr_bytes { - let zero = self.emit_const(0, self.types.int_id); - self.emit(Instruction::store( - zero, - base_sym, - offset + i as i64, - char_type, - char_bits, - )); - } - } + // A string literal initializing an array element -- + // `char arr[3][4] = {"Sun", "Mon", "Tue"}` -- is inline + // data, not one element. Recursing into it would treat the + // literal as the pointer it decays to everywhere else. + // + // Shared with the other three string-store paths rather + // than counted a third way here. Written out, this loop + // recognized all four literal kinds and then handled only + // `StringLit`, dropping a wide element and leaving it + // zero; stepped the destination by raw *bytes* where a + // wide element is 2 or 4 bytes wide; and had no capacity + // clamp at all, so `char s[1][3] = {"hello"}` stored five + // bytes into a three-byte object -- two of them past the + // whole local, not merely into the next row. + let string_element = (list.len() == 1 + && self.types.kind(elem_type) == TypeKind::Array) + .then(|| Self::string_literal_units(&list[0].value.kind)) + .flatten(); + if let Some(units) = string_element { + self.store_string_units( + base_sym, + offset, + elem_type, + &list[0].value.kind, + &units, + StringTail::AlreadyZero, + ); continue; } if elem_is_aggregate { @@ -1254,7 +1266,16 @@ impl<'a> super::linearize::Linearizer<'a> { // Shared with the two other string-store paths rather than // counted a third way here. if let Some(units) = Self::string_literal_units(&value.kind) { - self.store_string_units(base_sym, offset, field_type, &value.kind, &units); + // Reached only from an initializer list, which the caller + // zeroed whole before walking it. + self.store_string_units( + base_sym, + offset, + field_type, + &value.kind, + &units, + StringTail::AlreadyZero, + ); } } else { let val = self.linearize_expr(value); @@ -1983,9 +2004,13 @@ impl<'a> super::linearize::Linearizer<'a> { /// Copy a string literal's code units into an array object, followed by /// its null terminator. /// - /// Shared by the two ways a string can initialize an array: written - /// directly (`char b[] = "hi"`) or enclosed in braces - /// (`char b[] = {"hi"}`, C17 6.7.9p14). + /// Shared by every way a string can initialize an array: written directly + /// (`char b[] = "hi"`), enclosed in braces (`char b[] = {"hi"}`, C17 + /// 6.7.9p14), as a struct member (`struct { char t[4]; } s = {"hi"}`), or + /// as an element of a nested array (`char n[2][4] = {"ab", "cd"}`). + /// + /// `tail` says whether the elements the literal does not reach are this + /// call's to zero; see [`StringTail`]. pub(crate) fn store_string_units( &mut self, base_sym: PseudoId, @@ -1993,6 +2018,7 @@ impl<'a> super::linearize::Linearizer<'a> { arr_typ: TypeId, kind: &ExprKind, units: &[i128], + tail: StringTail, ) { let default_elem = match kind { ExprKind::StringLit(_) => self.types.char_id, @@ -2025,7 +2051,10 @@ impl<'a> super::linearize::Linearizer<'a> { elem_size, )); } - if units.len() < capacity { + // The first element past everything the literal and its terminator + // wrote. When the literal fills the array exactly, that is the whole + // array; when the terminator was dropped, nothing is left either. + let written = if units.len() < capacity { let null_val = self.emit_const(0, elem_type); self.emit(Instruction::store( null_val, @@ -2034,6 +2063,25 @@ impl<'a> super::linearize::Linearizer<'a> { elem_type, elem_size, )); + units.len() + 1 + } else { + capacity + }; + + // C17 6.7.9p21: the members not initialized explicitly are + // initialized as a static object would be, i.e. to zero. One + // terminator is not the rest of the array -- `char b[8] = "hi"` wrote + // three bytes and left five holding whatever the frame held, which on + // first entry is zero because the backend zeroes the whole frame, and + // on re-execution is the last iteration's data. + // + // Routed through the shared block fill, so the bound that keeps + // `char b[1 << 20] = "x"` from becoming a million stores is the one + // every other block operation uses. + if matches!(tail, StringTail::Zero) { + let start = base_offset + (written as i64) * elem_bytes; + let bytes = (capacity - written) as i64 * elem_bytes; + self.emit_block_zero(base_sym, start, bytes); } } diff --git a/cc/ir/test_linearize.rs b/cc/ir/test_linearize.rs index c9bb641b0..2bca4d8fe 100644 --- a/cc/ir/test_linearize.rs +++ b/cc/ir/test_linearize.rs @@ -6016,7 +6016,25 @@ fn test_compound_literal_zero_init_lvalue() { let tu = TranslationUnit { items: vec![ExternalDecl::FunctionDef(func)], }; - let module = ctx.linearize(&tu); + let mut module = ctx.linearize(&tu); + + // The linearizer asks for the zero-init as one `Memset` of the whole + // literal rather than emitting the stores itself. That is what gives it + // the same bound as every other block memory operation -- the ladder it + // used to hand-roll here had none, so `char buf[N] = {0}` unrolled for any + // N at all. + let linearized = format!("{}", module.display(&ctx.types)); + assert!( + linearized.contains("memset"), + "expected the compound literal's zero-init to be asked for as a memset: {linearized}" + ); + + // `memexpand` turns it back into stores, at every optimization level -- + // `-O0` included, see `opt::optimize_module` -- so run it here and hold + // the stores to the same account as before. + for f in &mut module.functions { + crate::ir::memexpand::run(f, &ctx.types); + } let ir = format!("{}", module.display(&ctx.types)); // The compound literal must be zero-initialized first, then the designated @@ -7611,6 +7629,165 @@ fn test_memory_builtins_are_their_opcodes() { ); } +/// The constant `id` holds in `f`, if it is one -- through the narrowing a +/// conversion to the destination's own type leaves. +fn const_of(module: &Module, name: &str, id: PseudoId) -> Option { + let f = module.functions.iter().find(|f| f.name == name).unwrap(); + crate::ir::facts::ConstMap::new(f).get(id) +} + +/// Every `Memset` in `f`, as `(fill byte, length)`. +fn memsets_of(module: &Module, name: &str) -> Vec<(Option, Option)> { + insns_of(module, name) + .iter() + .filter(|i| i.op == Opcode::Memset) + .map(|i| { + ( + const_of(module, name, i.src[1]), + const_of(module, name, i.src[2]), + ) + }) + .collect() +} + +/// Every `Store` in `f`, as `(offset, width in bits, stored constant)`. A +/// store's `src` is `(address, value)`. +fn stores_of(module: &Module, name: &str) -> Vec<(i64, u32, Option)> { + insns_of(module, name) + .iter() + .filter(|i| i.op == Opcode::Store) + .map(|i| (i.offset, i.size, const_of(module, name, i.src[1]))) + .collect() +} + +/// Zero-initializing an aggregate is one `Memset` of the whole object, whose +/// length `memexpand` then weighs against `INLINE_LIMIT_BYTES`. +/// +/// `emit_aggregate_zero` hand-rolled the same 8/4/2/1 descent +/// `memexpand::block_chunks` already produces, but with **no** upper bound, so +/// `char buf[N] = {0}` emitted one store per chunk for any N: 8 KB cost 2081 +/// instructions in the function body and 1 MB did not finish compiling in 25 +/// minutes. Asking for the opcode instead makes the bound the shared one and +/// leaves the linearizer with no ladder of its own. +#[test] +fn test_aggregate_zero_is_one_memset_of_the_whole_object() { + let target = Target::new(Arch::X86_64, Os::Linux); + // Each declares an object and hands it to `sink` so nothing is dead. The + // only store left is the one element `{0}` names explicitly; every other + // byte is the memset's, whatever the object's size. + for (decl, bytes, explicit) in [ + ("char buf[200] = {0}; sink(buf);", 200, (0, 8, Some(0))), + ("char buf[7] = {0}; sink(buf);", 7, (0, 8, Some(0))), + ( + "struct S { int a; char b; } s = {0}; sink(&s);", + 8, + (0, 32, Some(0)), + ), + ] { + let src = format!("void sink(void *);\nvoid f(void) {{ {decl} }}\n"); + let module = linearize_source(&src, &target); + assert_eq!( + memsets_of(&module, "f"), + vec![(Some(0), Some(bytes))], + "{decl}: one memset of the whole object and nothing else" + ); + assert_eq!( + stores_of(&module, "f"), + vec![explicit], + "{decl}: the linearizer emits no chunk ladder of its own" + ); + } +} + +/// `char b[N] = "str"` zero-fills the elements the literal does not reach, +/// and only those. +/// +/// C17 6.7.9p21 initializes them as a static object would be. The `InitList` +/// arm of a local declaration calls `emit_aggregate_zero` first; the bare +/// string arm did not, so only the literal's own bytes and one terminator were +/// written. The braced form reaches the array through an initializer list, +/// which is already zeroed whole -- zeroing again there would double the +/// stores at `-O0`, where no `dse` runs to remove them. +#[test] +fn test_a_string_initializer_zero_fills_only_its_tail() { + let target = Target::new(Arch::X86_64, Os::Linux); + + // Three bytes written -- 'h', 'i', and the terminator -- then five left. + let bare = linearize_source( + "void sink(void *);\nvoid f(void) { char b[8] = \"hi\"; sink(b); }\n", + &target, + ); + assert_eq!( + stores_of(&bare, "f"), + vec![(0, 8, Some(0x68)), (1, 8, Some(0x69)), (2, 8, Some(0))] + ); + assert_eq!(memsets_of(&bare, "f"), vec![(Some(0), Some(5))]); + + // The braced form is preceded by the whole-object zero, so its tail is + // already done: one memset of 8, not one of 8 and another of 5. + let braced = linearize_source( + "void sink(void *);\nvoid f(void) { char b[8] = {\"hi\"}; sink(b); }\n", + &target, + ); + assert_eq!(memsets_of(&braced, "f"), vec![(Some(0), Some(8))]); + + // Exactly as long as the literal: C17 6.7.9p14 drops the terminator, and + // there is then no tail either. + let exact = linearize_source( + "void sink(void *);\nvoid f(void) { char b[2] = \"hi\"; sink(b); }\n", + &target, + ); + assert_eq!( + stores_of(&exact, "f"), + vec![(0, 8, Some(0x68)), (1, 8, Some(0x69))] + ); + assert_eq!(memsets_of(&exact, "f"), vec![]); +} + +/// A string literal initializing a *nested* array element steps the +/// destination by the element's own width and stops at its capacity. +/// +/// This path was a third hand-rolled copy of `store_string_units`. It +/// recognized all four literal kinds and then handled only the narrow one, so +/// a wide element was dropped and left zero; it stepped the destination by raw +/// bytes where a wide element is 2 or 4 bytes wide; and it had no capacity +/// clamp at all, so `char s[1][3] = {"hello"}` stored five bytes into a +/// three-byte object. +#[test] +fn test_a_nested_string_element_keeps_its_stride_and_capacity() { + let target = Target::new(Arch::X86_64, Os::Linux); + + // `unsigned short` is `char16_t`: two bytes of stride, and the literal + // reaches the array at all. + let wide = linearize_source( + "void sink(void *);\n\ + void f(void) { unsigned short u[2][3] = {u\"ab\", u\"cd\"}; sink(u); }\n", + &target, + ); + assert_eq!( + stores_of(&wide, "f"), + vec![ + (0, 16, Some(0x61)), + (2, 16, Some(0x62)), + (4, 16, Some(0)), + (6, 16, Some(0x63)), + (8, 16, Some(0x64)), + (10, 16, Some(0)), + ] + ); + + // Five units into a three-byte row: three stored, none past the row, and + // the terminator dropped with them. + let over = linearize_source( + "void sink(void *);\nvoid f(void) { char s[1][3] = {\"hello\"}; sink(s); }\n", + &target, + ); + assert_eq!( + stores_of(&over, "f"), + vec![(0, 8, Some(0x68)), (1, 8, Some(0x65)), (2, 8, Some(0x6c))] + ); +} + /// `mempcpy` is a `Memcpy` that calls `memcpy`, and its value the /// destination advanced by the length; `bcopy` is a `Memmove` that calls /// `memmove`, with its source and destination put back in `memmove`'s diff --git a/cc/tests/c99/initializers.rs b/cc/tests/c99/initializers.rs index 4f452bd37..84bedfd21 100644 --- a/cc/tests/c99/initializers.rs +++ b/cc/tests/c99/initializers.rs @@ -2637,3 +2637,147 @@ int main(void) 0 ); } + +/// A string literal initializing a nested array element, in every encoding. +/// +/// `is_string_for_char_array` accepts all four literal kinds, but the body that +/// consumed them handled only the narrow one and silently `continue`d on the +/// rest, so a wide element was dropped and left zero. The same loop stepped the +/// destination by *bytes* while a wide element is 2 or 4 bytes wide, and it had +/// no capacity clamp at all — a third hand-rolled copy of what +/// `store_string_units` already does correctly for the non-nested form. +/// +/// The static twin of each case was already right, which is how the two paths +/// could disagree: `static wchar_t sw[2][4] = {L"ab", L"cd"}` read back 97/99 +/// while the automatic form read back 0/0. +#[test] +fn c99_a_nested_string_literal_element_is_stored_in_every_encoding() { + let code = r#" +#include +/* does not exist on macOS, and the test needs only the two types -- + the same substitution the universal-character-name test above makes. */ +typedef unsigned short char16_t; +typedef unsigned int char32_t; + +int main(void) +{ + /* Narrow, and the tail of a short element must be zero. */ + char n[2][4] = {"ab", "cd"}; + if (n[0][0] != 'a' || n[0][1] != 'b' || n[0][2] != 0 || n[0][3] != 0) return 1; + if (n[1][0] != 'c' || n[1][1] != 'd' || n[1][2] != 0 || n[1][3] != 0) return 2; + + /* Wide: dropped entirely before the fix. */ + wchar_t w[2][4] = {L"ab", L"cd"}; + if ((int)w[0][0] != 'a' || (int)w[0][1] != 'b' || w[0][2] != 0) return 3; + if ((int)w[1][0] != 'c' || (int)w[1][1] != 'd' || w[1][2] != 0) return 4; + + char16_t u[2][4] = {u"ab", u"cd"}; + if ((int)u[0][0] != 'a' || (int)u[1][0] != 'c' || u[0][2] != 0) return 5; + + char32_t U[2][4] = {U"ab", U"cd"}; + if ((int)U[0][0] != 'a' || (int)U[1][0] != 'c' || U[0][2] != 0) return 6; + + /* The static path was always correct; the two must now agree. */ + static wchar_t sw[2][4] = {L"ab", L"cd"}; + if ((int)sw[0][0] != 'a' || (int)sw[1][0] != 'c') return 7; + + return 0; +} +"#; + assert_eq!(compile_and_run("nested_string_encodings", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("nested_string_encodings_opt", code), + 0 + ); +} + +/// A string literal too long for the array it initializes writes only as much +/// as fits. +/// +/// C17 6.7.9p14 allows exactly the terminating NUL to be dropped, and nothing +/// more. The nested-array path had no clamp, so `char s[1][3] = {"hello"}` +/// stored five bytes into a three-byte object — two of them past the whole +/// local, not merely into the next row. The guards on either side are what make +/// that visible rather than layout-dependent. +#[test] +fn c99_an_overlong_string_literal_does_not_write_past_its_array() { + let code = r#" +int main(void) +{ + unsigned char lo = 0xA5; + char s[1][3] = {"hello"}; + unsigned char hi = 0x5A; + if (s[0][0] != 'h' || s[0][1] != 'e' || s[0][2] != 'l') return 1; + if (lo != 0xA5 || hi != 0x5A) return 2; + + /* Exactly the terminator dropped: this is legal and keeps all three. */ + unsigned char lo2 = 0xA5; + char e[1][3] = {"abc"}; + unsigned char hi2 = 0x5A; + if (e[0][0] != 'a' || e[0][1] != 'b' || e[0][2] != 'c') return 3; + if (lo2 != 0xA5 || hi2 != 0x5A) return 4; + + /* Wide, where the stride is 4 bytes and a byte-stepped copy lands wrong. */ + unsigned char lo3 = 0xA5; + __WCHAR_TYPE__ w[1][2] = {L"xyz"}; + unsigned char hi3 = 0x5A; + if ((int)w[0][0] != 'x' || (int)w[0][1] != 'y') return 5; + if (lo3 != 0xA5 || hi3 != 0x5A) return 6; + + return 0; +} +"#; + assert_eq!(compile_and_run("overlong_nested_string", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("overlong_nested_string_opt", code), + 0 + ); +} + +/// `char buf[N] = "str"` zero-fills the bytes the literal does not reach. +/// +/// C17 6.7.9p21: the members not initialized explicitly are initialized as a +/// static object would be, i.e. to zero. The `InitList` arm of a local +/// declaration calls `emit_aggregate_zero` first; the string arm did not, so +/// only the literal's own bytes were written. +/// +/// On entry the backend zeroes the whole frame, which hides this the first time +/// through — the declaration is inside a loop so the second pass sees what the +/// first one left. All four encodings are affected. +#[test] +fn c99_a_string_initializer_zero_fills_the_rest_of_its_array() { + let code = r#" +#include + +int main(void) +{ + for (int pass = 0; pass < 2; pass++) { + char b[8] = "hi"; + if (b[2] != 0 || b[3] != 0 || b[7] != 0) return 1; + b[3] = 'Z'; + b[7] = 'Z'; + } + + for (int pass = 0; pass < 2; pass++) { + wchar_t w[4] = L"hi"; + if (w[2] != 0 || w[3] != 0) return 2; + w[3] = 'Z'; + } + + /* The braced form went through the InitList arm and was already correct; + both spellings must now agree. */ + for (int pass = 0; pass < 2; pass++) { + char c[8] = {"hi"}; + if (c[3] != 0 || c[7] != 0) return 3; + c[3] = 'Z'; + } + + return 0; +} +"#; + assert_eq!(compile_and_run("string_init_zero_fill", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("string_init_zero_fill_opt", code), + 0 + ); +} diff --git a/cc/tests/codegen/misc.rs b/cc/tests/codegen/misc.rs index 1c3aa387d..e3422e7b9 100644 --- a/cc/tests/codegen/misc.rs +++ b/cc/tests/codegen/misc.rs @@ -16438,3 +16438,165 @@ int main(void) { ); } } + +/// Zero-initializing an aggregate, across the size at which unrolling stops. +/// +/// The companion to `codegen_struct_copy_across_the_inline_threshold`, for the +/// other half of the same family. `emit_aggregate_zero` hand-rolled the same +/// 8/4/2/1 descent that `memexpand::block_chunks` already produces, but with +/// **no upper bound** — so `char buf[N] = {0}` emitted one store per chunk for +/// any N. Measured before the fix: 8 KB cost 2081 instructions in the function +/// body and 1 MB did not finish compiling in 25 minutes, while its sibling +/// `emit_block_copy_at_offset` had capped at `INLINE_LIMIT_BYTES` all along. +/// +/// The declaration is inside a loop on purpose. On entry the backend zeroes the +/// whole frame, which masks a missing zero-fill the first time through; only +/// re-execution shows it. +/// +/// Sizes straddle 128 and none is a multiple of 8, so a rounded-up or +/// short-by-a-tail fill shows as a wrong byte rather than passing by luck. +#[test] +fn codegen_aggregate_zero_across_the_inline_threshold() { + let code = r#" +void sink(char *p); + +#define MKZ(N) \ + static int zero##N(void) { \ + for (int pass = 0; pass < 2; pass++) { \ + unsigned char lo = 0xA5; \ + char buf[N] = {0}; \ + unsigned char hi = 0x5A; \ + for (int i = 0; i < N; i++) \ + if (buf[i] != 0) return 1; \ + if (lo != 0xA5 || hi != 0x5A) return 2; \ + for (int i = 0; i < N; i++) buf[i] = (char)(i + 1); \ + sink(buf); \ + } \ + return 0; \ + } + +MKZ(7) +MKZ(12) +MKZ(13) +MKZ(127) +MKZ(129) +MKZ(200) +MKZ(1000) + +void sink(char *p) { (void)p; } + +int main(void) +{ + if (zero7()) return 1; + if (zero12()) return 2; + if (zero13()) return 3; + if (zero127()) return 4; + if (zero129()) return 5; + if (zero200()) return 6; + if (zero1000()) return 7; + return 0; +} +"#; + assert_eq!(compile_and_run("aggregate_zero_threshold", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("aggregate_zero_threshold_opt", code), + 0 + ); +} + +/// The bound itself, not just the answer: above the threshold the zero-fill is +/// a `memset` call, below it is still stores. +/// +/// The behavioural test above passes either way — a million unrolled stores +/// produce a correctly zeroed object, just not in a time anyone will wait for. +/// This is the test that the *bound* exists, and the negative half keeps it +/// from being satisfied by calling `memset` for every size, which would cost +/// more than the stores it replaced for a small object. +#[test] +fn codegen_a_large_aggregate_zero_is_a_memset_call() { + use crate::codegen::asm_probe::{asm_for_with, AARCH64_LINUX, X86_64_LINUX}; + + let src = |n: usize| { + format!("void sink(char *);\nvoid probe(void) {{ char buf[{n}] = {{0}}; sink(buf); }}\n") + }; + + for triple in [X86_64_LINUX, AARCH64_LINUX] { + // Comfortably over `INLINE_LIMIT_BYTES` (128). + let big = asm_for_with("aggzero_big", triple, &src(4096), &["-O2"]); + assert!( + big.contains("memset"), + "a 4096-byte zero-fill belongs in a memset call, not 512 stores, on {triple}:\n{big}" + ); + + // And the unrolled form is still used where it is cheaper than a call. + let small = asm_for_with("aggzero_small", triple, &src(16), &["-O2"]); + assert!( + !small.contains("memset"), + "a 16-byte zero-fill is cheaper unrolled than called, on {triple}:\n{small}" + ); + } +} + +/// The inliner moves an implicit parameter copy in whole eight-byte chunks, +/// which reads and writes past a object whose size is not a multiple of eight. +/// +/// The third copy of the unrolled block move, after the two named in +/// `codegen_struct_copy_across_the_inline_threshold`. `while offset < size_bytes +/// { load 64; store 64; offset += 8 }` rounds *up*: a 12-byte `struct P` moved +/// 16 bytes, over-reading the argument and over-writing the callee's local. +/// `memexpand::block_chunks` descends 8/4/2/1 and is what the already-fixed +/// twin in the linearizer uses. +/// +/// Stated on the IR because the overrun is layout-dependent: the four extra +/// bytes usually land in frame padding, so a program can be correct and still +/// be reading memory that does not belong to the object — and would fault if it +/// ended a page. +#[test] +fn codegen_an_inlined_parameter_copy_moves_no_more_than_the_object() { + let src = r#" +struct P { float x, y, z; }; +static float sum(struct P p) { return p.x + p.y + p.z; } +float probe(void) { struct P q = {1, 2, 3}; return sum(q); } +"#; + let dir = plib::tmp::Builder::new() + .prefix("inline_param_copy") + .tempdir() + .expect("tempdir"); + let c = dir.path().join("t.c"); + std::fs::write(&c, src).expect("write source"); + let r = crate::common::run_c17(&[ + "-O2", + "--dump-ir", + "post-opt", + "--dump-ir-func", + "probe", + "-S", + "-o", + "/dev/null", + c.to_str().unwrap(), + ]); + assert!(r.success, "compile failed: {}", r.stderr); + let ir = format!("{}{}", r.stdout, r.stderr); + + // A 12-byte object has four bytes at offset 8, so a 64-bit access there is + // four bytes past the end -- on the load side and again on the store side. + // A correct copy reaches it with a 32-bit access. + let overruns: Vec<&str> = ir + .lines() + .map(str::trim) + .filter(|l| l.contains("+ 8") && (l.contains("load.64") || l.contains("store.64"))) + .collect(); + assert!( + overruns.is_empty(), + "a 12-byte object has 4 bytes at offset 8, so these access 4 past it:\n {}\n\nfull IR:\n{ir}", + overruns.join("\n ") + ); + + // The control: the copy has to still be there. If the inliner stopped + // inlining, or the parameter stopped being copied, the check above would + // pass while testing nothing. + assert!( + ir.contains("+ 8"), + "expected the inlined parameter copy to reach offset 8 at all:\n{ir}" + ); +} From e285d58ed0b485d98c570a517e07d4def98b81de Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 02:51:50 -0400 Subject: [PATCH 08/17] cc: move a parameter's bytes in its own widths, and bound the bulk ones The back ends carried five more copies of the chunk loop the IR fixed one commit ago. They cannot call `memcpy` -- they are past the point a call can be synthesized -- but the rule is the same, and each copy lost the same things: three rounded a width up, one had no bound. A struct classified into two eightbytes has a high half narrower than eight whenever its size is not a multiple of eight. The SSE parameter prologue stored both halves at `FpSize::Double`, so `struct P { float x, y, z; }` wrote four bytes past itself; the general register-pair prologue did the same for `struct { int a, b, c; }`; and `copy_incoming_arg_to_local` stepped eight regardless of width. All three now ask `eightbyte_bytes` what is actually left and store that -- `movsd` then `movss`, and a packed `struct { float a, b, c; _Float16 d; }` comes out `movsd`/`movl`/`movw`, fourteen bytes exactly. `va_arg` of a large aggregate had no limit at all, so fetching four kilobytes cost 1081 instructions on x86-64 and 1107 on aarch64, linear in the object. Past `INLINE_LIMIT_BYTES` -- the same constant the IR bounds a copy with -- x86-64 moves the whole eightbytes with `rep movsq` and unrolls the ragged tail, and aarch64 runs a counted loop sixteen bytes at a time through `q16`. Neither primitive is new: `call.rs` already had the `rep movsq` for stacked arguments, generalized here rather than written twice, and the aarch64 loop sits beside `emit_zero_loop` and is shaped like it. The `q` register is what makes the loop fit in the three general registers the `va_arg` sequence has free. 33 and 36 instructions now. The stacked-argument copy rounded its *source* read up. The destination round-up is correct -- the outgoing area is allocated in whole eightbytes -- but the source is the object, and a 12-byte struct read bytes 8..15 where four of them belong to whatever follows. It faults if the object ends a page, which is the hazard `load_object_bytes` in the same file was already fixed for. The `rep movsq` path had it too, reading `ceil(bytes/8)` qwords: seven bytes past a 4095-byte object. Four literal `[8, 4, 2, 1]` ladders -- transliterations of `block_chunks`'s body -- are gone with them, and `grep` for that literal across `cc/` now finds nothing. The aarch64 stacked-composite and spilled-composite paths were suspected of being unbounded and are not: AAPCS64 B.4 replaces a composite above sixteen bytes with a pointer to the caller's copy, so both are bounded by the ABI at two eightbytes and a 4 KB argument is two `memcpy` calls and a pointer. They are deduped here, not fixed. These overruns land in slot padding rather than in a neighbouring object, because a slot is `slot_bytes(size).max(8)` at eight-byte alignment. So the behavioural test passes either way and the assembly assertions are what pin the widths -- which is why both are here. Co-Authored-By: Claude Opus 5 (1M context) --- cc/arch/aarch64/call.rs | 16 +- cc/arch/aarch64/features.rs | 58 +++++-- cc/arch/aarch64/frame.rs | 97 +++++++++++- cc/arch/lir.rs | 86 +++++++++++ cc/arch/x86_64/call.rs | 140 +++++++++++------ cc/arch/x86_64/features.rs | 47 ++++-- cc/arch/x86_64/frame.rs | 212 +++++++++++++++++-------- cc/tests/codegen/block_moves.rs | 263 ++++++++++++++++++++++++++++++++ cc/tests/codegen/cross_abi.rs | 148 +++++++++++++++++- cc/tests/codegen/mod.rs | 1 + 10 files changed, 922 insertions(+), 146 deletions(-) create mode 100644 cc/tests/codegen/block_moves.rs diff --git a/cc/arch/aarch64/call.rs b/cc/arch/aarch64/call.rs index db7aacb5e..351ed2be9 100644 --- a/cc/arch/aarch64/call.rs +++ b/cc/arch/aarch64/call.rs @@ -750,14 +750,15 @@ impl Aarch64CodeGen { if let StackKind::Composite { bytes } = stack_arg.kind { // The pseudo locates the aggregate; its bytes go into the // slot. + // AAPCS64 B.4 replaces a composite above sixteen bytes with a + // pointer to the caller's copy, so the run here is bounded by + // the ABI at two eightbytes; the widths are `block_chunks`'s, + // so a composite that is not a multiple of eight reads and + // writes its tail as wide as the tail is. let src = self.aggregate_arg_address(stack_arg.pseudo); - let mut done = 0; - while done < bytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= bytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let done = done as i32; self.push_lir(Aarch64Inst::Ldr { size, addr: MemAddr::BaseOffset { @@ -774,7 +775,6 @@ impl Aarch64CodeGen { offset: offset + done, }, }); - done += chunk; } continue; } diff --git a/cc/arch/aarch64/features.rs b/cc/arch/aarch64/features.rs index efbf1d628..b6df879eb 100644 --- a/cc/arch/aarch64/features.rs +++ b/cc/arch/aarch64/features.rs @@ -11,6 +11,7 @@ use super::call::HfaElem; use super::codegen::Aarch64CodeGen; +use super::frame::UNROLL_LIMIT_BYTES; use super::lir::{Aarch64Inst, GpOperand, MemAddr}; use super::regalloc::{Loc, Reg, VReg}; use crate::arch::codegen::BswapSize; @@ -636,8 +637,15 @@ impl Aarch64CodeGen { } /// Copy `bytes` of an aggregate from `[addr + off]` into the destination, - /// in descending power-of-two chunks so nothing past the object is - /// written. X16 is the shuttle -- linker scratch, never allocated. + /// in the descending power-of-two chunks `block_chunks` gives, so nothing + /// past the object is written. X16 is the shuttle -- linker scratch, never + /// allocated. + /// + /// Past [`UNROLL_LIMIT_BYTES`] the copy becomes a counted loop, which + /// **advances `addr`**: only the two `VaAggKind::Indirect` paths can reach + /// that size, and both pass a register they are finished with. Every other + /// kind is at most sixteen bytes (a composite in general registers) or + /// sixty-four (an HFA of four binary128s), so it never gets there. fn emit_va_arg_bytes( &mut self, dst_loc: &Loc, @@ -663,13 +671,44 @@ impl Aarch64CodeGen { return; } } - let mut done = 0; - while done < bytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= bytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + if i64::from(bytes) > UNROLL_LIMIT_BYTES { + // X16 becomes the destination cursor; the source cursor is `addr` + // itself, advanced past the object. + match dst_loc { + Loc::Stack(_) | Loc::IncomingArg(_) => { + let (base, disp) = self + .loc_addr_parts(dst_loc) + .expect("a frame location has a base and a displacement"); + self.push_lir(Aarch64Inst::Add { + size: OperandSize::B64, + src1: base, + src2: GpOperand::Imm(disp.into()), + dst: Reg::X16, + }); + } + // The register holds the aggregate's address, and it is the + // result, so the cursor is a copy of it. + Loc::Reg(r) if !holds_value => self.push_lir(Aarch64Inst::Mov { + size: OperandSize::B64, + src: GpOperand::Reg(*r), + dst: Reg::X16, + }), + _ => return, + } + if off != 0 { + self.push_lir(Aarch64Inst::Add { + size: OperandSize::B64, + src1: addr, + src2: GpOperand::Imm(off.into()), + dst: addr, + }); + } + self.emit_block_copy_loop(Reg::X16, addr, bytes.into()); + return; + } + for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let done = done as i32; self.push_lir(Aarch64Inst::Ldr { size, addr: MemAddr::BaseOffset { @@ -700,7 +739,6 @@ impl Aarch64CodeGen { }), _ => return, } - done += chunk; } } diff --git a/cc/arch/aarch64/frame.rs b/cc/arch/aarch64/frame.rs index 751d5ca32..ccd5fe1a8 100644 --- a/cc/arch/aarch64/frame.rs +++ b/cc/arch/aarch64/frame.rs @@ -25,6 +25,12 @@ use crate::ir::{Function, Instruction, PseudoId, PseudoKind}; use crate::types::{TypeId, TypeKind, TypeTable}; use std::collections::HashSet; +/// The most bytes this back end moves with unrolled loads and stores; past it, +/// [`Aarch64CodeGen::emit_block_copy_loop`]. It is the bound the IR puts on an +/// expanded `memcpy`, for the same reason: the instruction count of an unrolled +/// copy is linear in the object. +pub(super) const UNROLL_LIMIT_BYTES: i64 = crate::ir::memexpand::INLINE_LIMIT_BYTES; + impl Aarch64CodeGen { pub(super) fn emit_function(&mut self, func: &Function, types: &TypeTable) { self.base.func_pos = crate::arch::func_pos(func); @@ -538,6 +544,81 @@ impl Aarch64CodeGen { } } + /// Copy `bytes` bytes from `[src]` to `[dst]`, in a counted loop over the + /// whole sixteen-byte pairs and `block_chunks` for what is left. + /// + /// This is the back end's bulk block move. It cannot synthesize a call to + /// `memcpy` -- it is past the point where a call can be built -- and one + /// load/store pair per chunk is linear in the object: 4 KB of `va_arg` + /// aggregate cost about 1100 instructions, and 256 KB would be the + /// compile-time explosion the IR's own + /// [`crate::ir::memexpand::INLINE_LIMIT_BYTES`] exists to prevent. A + /// counted loop is the answer [`Self::emit_zero_loop`] gives the frame. + /// + /// Sixteen bytes an iteration through V16, which is reserved codegen + /// scratch: it keeps the loop to three general registers, which is all the + /// `va_arg` sequence has free. Both `src` and `dst` are cursors and are + /// left past the object, so the caller must be finished with them; X17 + /// holds the end of the paired part and then shuttles the tail. + pub(super) fn emit_block_copy_loop(&mut self, dst: Reg, src: Reg, bytes: i64) { + let pairs = bytes & !15; + if pairs > 0 { + self.push_lir(Aarch64Inst::Add { + size: OperandSize::B64, + src1: src, + src2: GpOperand::Imm(pairs), + dst: Reg::X17, + }); + let top = self.next_unique_label("block_copy"); + self.push_lir(Aarch64Inst::Directive(Directive::BlockLabel(top.clone()))); + self.push_lir(Aarch64Inst::LdrFp { + size: FpSize::Quad, + addr: MemAddr::PostIndex { + base: src, + offset: 16, + }, + dst: VReg::V16, + }); + self.push_lir(Aarch64Inst::StrFp { + size: FpSize::Quad, + src: VReg::V16, + addr: MemAddr::PostIndex { + base: dst, + offset: 16, + }, + }); + self.push_lir(Aarch64Inst::Cmp { + size: OperandSize::B64, + src1: src, + src2: GpOperand::Reg(Reg::X17), + }); + self.push_lir(Aarch64Inst::BCond { + cond: CondCode::Ult, + target: top, + }); + } + // The cursors point at the tail, so its pieces are offsets from them. + for (off, chunk) in crate::ir::memexpand::block_chunks(bytes - pairs) { + let size = OperandSize::from_bits(chunk.bits()); + self.push_lir(Aarch64Inst::Ldr { + size, + addr: MemAddr::BaseOffset { + base: src, + offset: off as i32, + }, + dst: Reg::X17, + }); + self.push_lir(Aarch64Inst::Str { + size, + src: Reg::X17, + addr: MemAddr::BaseOffset { + base: dst, + offset: off as i32, + }, + }); + } + } + /// Save callee-saved GP registers in pairs (or single if odd count) fn save_callee_saved_gp_regs(&mut self, total_frame: i32, callee_saved: &[Reg]) { let mut offset = 16; // Start after fp/lr @@ -960,13 +1041,14 @@ impl Aarch64CodeGen { crate::arch::func_pos(func), "a stacked parameter", ); - let mut done = 0; - while done < bytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= bytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + // A composite that reaches here is at most two + // eightbytes, so the run is bounded by the ABI; + // the widths are `block_chunks`'s so the tail of + // one that is not a multiple of eight is moved as + // wide as it is and no wider. + for (done, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let done = done as i32; self.push_lir(Aarch64Inst::Ldr { size, addr: self.incoming_mem_plus(incoming, done), @@ -977,7 +1059,6 @@ impl Aarch64CodeGen { src: Reg::X16, addr: self.stack_mem_plus(local_off, done), }); - done += chunk; } } } diff --git a/cc/arch/lir.rs b/cc/arch/lir.rs index 3993e1172..c2f9c88d5 100644 --- a/cc/arch/lir.rs +++ b/cc/arch/lir.rs @@ -70,6 +70,22 @@ impl fmt::Display for OperandSize { } } +/// How many of a composite's `total` bytes the eightbyte starting at `at` +/// carries. +/// +/// An eightbyte of a composite parameter travels in a whole register, but the +/// last eightbyte of a composite whose size is not a multiple of eight holds +/// fewer bytes than the register does -- four of `struct { int a, b, c; }`, +/// five of a thirteen-byte one. Storing the register's eight regardless is +/// what wrote past the object's local, whose slot is rounded up only to the +/// type's own alignment. +/// +/// Zero for an eightbyte past the end, which a class vector longer than the +/// object cannot produce but a caller need not prove. +pub fn eightbyte_bytes(total: i64, at: i64) -> i64 { + (total - at).clamp(0, 8) +} + // Floating-Point Size /// Floating-point size specifier @@ -131,6 +147,27 @@ impl FpSize { } } + /// The one floating-point store that writes exactly `bytes` bytes, if + /// there is one. + /// + /// The widths a store has are 2, 4, 8 and 16; a composite eightbyte of + /// SSE class can be any of them, and -- once a member is packed -- 3, 5, 6 + /// or 7 as well, for which the answer is `None` and the caller has to move + /// the bytes some other way. One byte is `None` too: there is no SSE store + /// of a single byte. + /// + /// Rounding up instead is what wrote four bytes past the twelve-byte + /// `struct { float x, y, z; }` whose second eightbyte holds only its `z`. + pub fn exact_sse_store(bytes: i64) -> Option { + match bytes { + 2 => Some(FpSize::Half), + 4 => Some(FpSize::Single), + 8 => Some(FpSize::Double), + 16 => Some(FpSize::Quad), + _ => None, + } + } + /// Create from TypeKind. This is the preferred way to determine FP size /// when type information is available, rather than inferring from bit size. /// @@ -1486,6 +1523,55 @@ mod tests { use super::*; use crate::target::Arch; + /// The last register of a composite parameter carries what is left of the + /// object, not a whole eightbyte. + /// + /// The prologue stores each eightbyte from the register it arrived in, and + /// a slot is rounded up only to the type's own alignment -- four for + /// `struct { int a, b, c; }` -- so a store as wide as the register writes + /// four bytes past a twelve-byte local. + #[test] + fn test_eightbyte_bytes_stops_at_the_end_of_the_object() { + // A multiple of eight fills every register it takes. + assert_eq!(eightbyte_bytes(16, 0), 8); + assert_eq!(eightbyte_bytes(16, 8), 8); + // Twelve and thirteen do not: the second register holds four and five. + assert_eq!(eightbyte_bytes(12, 0), 8); + assert_eq!(eightbyte_bytes(12, 8), 4); + assert_eq!(eightbyte_bytes(13, 8), 5); + // Under a register's worth the one register holds the whole object. + assert_eq!(eightbyte_bytes(3, 0), 3); + // Past the end is nothing at all, never a negative width. + assert_eq!(eightbyte_bytes(12, 16), 0); + assert_eq!(eightbyte_bytes(0, 0), 0); + + // The eightbytes of any size account for it exactly. + for total in 1..=64i64 { + let sum: i64 = (0..8).map(|i| eightbyte_bytes(total, i * 8)).sum(); + assert_eq!(sum, total, "the eightbytes of {total} must cover it"); + } + } + + /// An SSE eightbyte is stored at its own width, and a width no + /// floating-point store has is refused rather than rounded up to one. + #[test] + fn test_exact_sse_store_refuses_a_width_it_cannot_write() { + assert_eq!(FpSize::exact_sse_store(2), Some(FpSize::Half)); + assert_eq!(FpSize::exact_sse_store(4), Some(FpSize::Single)); + assert_eq!(FpSize::exact_sse_store(8), Some(FpSize::Double)); + assert_eq!(FpSize::exact_sse_store(16), Some(FpSize::Quad)); + // `for_sse_aggregate` answers every one of these with a *wider* store, + // which is what wrote past the object; here they have no answer and + // the caller moves the bytes through a general register instead. + for ragged in [0, 1, 3, 5, 6, 7, 9, 12, 15, 17] { + assert_eq!( + FpSize::exact_sse_store(ragged), + None, + "{ragged} bytes is not one SSE store" + ); + } + } + /// A symbol whose name an assembler will not take bare has to be quoted. /// Mach-O's assembler rejects a raw non-ASCII byte outright -- `_café:` is /// "invalid operand" -- and C17 6.4.2.1 admits extended characters in diff --git a/cc/arch/x86_64/call.rs b/cc/arch/x86_64/call.rs index 9e4a3564c..66bd6bbe8 100644 --- a/cc/arch/x86_64/call.rs +++ b/cc/arch/x86_64/call.rs @@ -54,47 +54,65 @@ pub(super) struct CallArgInfo { pub ignored_arg_indices: Vec, } -/// The largest stacked aggregate argument copied with unrolled moves; past -/// it, `rep movsq`. The IR stops unrolling a block copy at 128 bytes -/// (`memexpand::INLINE_LIMIT_BYTES`) for the same reason: one load/store pair per -/// eightbyte made a 600 MB argument 75 million instructions and tens of -/// gigabytes of compiler memory. -const STACK_ARG_UNROLL_QWORDS: usize = 16; +/// The largest block this back end copies with unrolled moves; past it, +/// `rep movsq`. It is the bound the IR puts on an expanded `memcpy`, for the +/// same reason: one load/store pair per eightbyte made a 600 MB argument 75 +/// million instructions and tens of gigabytes of compiler memory. +pub(super) const UNROLL_LIMIT_BYTES: i64 = crate::ir::memexpand::INLINE_LIMIT_BYTES; + +/// Where [`X86_64CodeGen::emit_rep_movsq`] writes. +/// +/// The helper pushes three registers, which moves `%rsp`, so a destination in +/// the outgoing argument area has to be named rather than handed over as an +/// address the caller already computed. +pub(super) enum BlockDst { + /// A byte offset in the outgoing argument area, addressed through `%rsp`; + /// the helper adds what its own pushes moved `%rsp` by. + OutgoingArg(i32), + /// Any address `%rsp` moving does not disturb -- a frame slot, or a + /// register holding a pointer. It must not be built on `%rsp`, `%rdi`, + /// `%rsi` or `%rcx`. + At(MemAddr), +} impl X86_64CodeGen { - /// Copy `qwords` eightbytes from `[src]` to the outgoing argument area at - /// `dst_off(%rsp)`, with `rep movsq`. + /// Copy `qwords` eightbytes from `src` to `dst`, with `rep movsq`. + /// + /// This is the back end's bulk block move: it cannot synthesize a call to + /// `memcpy`, and one load/store pair per eightbyte is linear in the object, + /// which is what made a 600 MB argument 75 million instructions. /// - /// The register arguments are set up *after* the stacked ones, so RDI, - /// RSI and RCX -- which `rep movsq` needs -- may still hold values that - /// are about to become arguments. They are saved around the copy with - /// `push`/`pop`, which touch no flags; the destination is addressed past - /// the three pushes. `src` is read into RSI before RDI or RCX is written, - /// so it may be any of the three. - fn emit_stack_arg_block_copy(&mut self, src: Reg, dst_off: i32, qwords: usize) { + /// RDI, RSI and RCX -- which `rep movsq` needs -- may hold live values: + /// the register arguments of a call are set up *after* its stacked ones, + /// and a `va_arg` sits in the middle of a function. They are saved around + /// the copy with `push`/`pop`, which touch no flags. `src` is read into RSI + /// before RDI or RCX is written, so it may be addressed through any of the + /// three; the destination may not. + pub(super) fn emit_rep_movsq(&mut self, src: MemAddr, dst: BlockDst, qwords: i64) { let saved = [Reg::Rdi, Reg::Rsi, Reg::Rcx]; for r in saved { self.push_lir(X86Inst::Push { src: GpOperand::Reg(r), }); } - if src != Reg::Rsi { - self.push_lir(X86Inst::Mov { - size: OperandSize::B64, - src: GpOperand::Reg(src), - dst: GpOperand::Reg(Reg::Rsi), - }); - } self.push_lir(X86Inst::Lea { - addr: MemAddr::BaseOffset { + addr: src, + dst: Reg::Rsi, + }); + let dst_addr = match dst { + BlockDst::OutgoingArg(off) => MemAddr::BaseOffset { base: Reg::Rsp, - offset: dst_off + 8 * saved.len() as i32, + offset: off + 8 * saved.len() as i32, }, + BlockDst::At(addr) => addr, + }; + self.push_lir(X86Inst::Lea { + addr: dst_addr, dst: Reg::Rdi, }); self.push_lir(X86Inst::Mov { size: OperandSize::B64, - src: GpOperand::Imm(qwords as i64), + src: GpOperand::Imm(qwords), dst: GpOperand::Reg(Reg::Rcx), }); self.push_lir(X86Inst::RepMovsq); @@ -103,6 +121,54 @@ impl X86_64CodeGen { } } + /// Copy the `bytes` bytes of the object at `[src]` into the outgoing + /// argument area at `dst_off(%rsp)`. + /// + /// Both sides move `block_chunks`'s widths. The argument area really is + /// allocated in whole eightbytes, so an eight-byte *write* at every offset + /// the object covers would be within it -- but the source is the object, + /// and reading eight bytes of a twelve-byte struct at offset eight reads + /// four that belong to whatever follows it, which faults when the object + /// ends a page. It is the same hazard [`Self::load_object_bytes`] carries, + /// and the same answer; the bytes of the slot the object does not reach are + /// left alone, which the psABI allows because nothing in the callee reads + /// them. + /// + /// Past [`UNROLL_LIMIT_BYTES`] the whole eightbytes go through `rep movsq` + /// and only the ragged tail is unrolled. + fn emit_stack_arg_copy(&mut self, src: Reg, dst_off: i32, bytes: i64) { + let mut at = 0; + if bytes > UNROLL_LIMIT_BYTES { + let qwords = bytes / 8; + let from = MemAddr::BaseOffset { + base: src, + offset: 0, + }; + self.emit_rep_movsq(from, BlockDst::OutgoingArg(dst_off), qwords); + at = qwords * 8; + } + for (off, chunk) in crate::ir::memexpand::block_chunks(bytes - at) { + let size = OperandSize::from_bits(chunk.bits()); + let byte = (at + off) as i32; + self.push_lir(X86Inst::Mov { + size, + src: GpOperand::Mem(MemAddr::BaseOffset { + base: src, + offset: byte, + }), + dst: GpOperand::Reg(Reg::Rax), + }); + self.push_lir(X86Inst::Mov { + size, + src: GpOperand::Reg(Reg::Rax), + dst: GpOperand::Mem(MemAddr::BaseOffset { + base: Reg::Rsp, + offset: dst_off + byte, + }), + }); + } + } + /// Classify call arguments into register vs stack arguments using ABI info. pub(super) fn classify_call_args(&self, insn: &Instruction, types: &TypeTable) -> CallArgInfo { let int_arg_regs = Reg::arg_regs(); @@ -339,30 +405,8 @@ impl X86_64CodeGen { crate::abi::struct_param_classes(t, types).map(|_| types.size_bytes(t)) }) }) { - let num_qwords = bytes.div_ceil(8); let base = self.address_of_pseudo(arg); - if num_qwords > STACK_ARG_UNROLL_QWORDS { - self.emit_stack_arg_block_copy(base, base_off, num_qwords); - continue; - } - for q in 0..num_qwords { - self.push_lir(X86Inst::Mov { - size: OperandSize::B64, - src: GpOperand::Mem(MemAddr::BaseOffset { - base, - offset: (q * 8) as i32, - }), - dst: GpOperand::Reg(Reg::Rax), - }); - self.push_lir(X86Inst::Mov { - size: OperandSize::B64, - src: GpOperand::Reg(Reg::Rax), - dst: GpOperand::Mem(MemAddr::BaseOffset { - base: Reg::Rsp, - offset: base_off + (q * 8) as i32, - }), - }); - } + self.emit_stack_arg_copy(base, base_off, bytes as i64); continue; } diff --git a/cc/arch/x86_64/features.rs b/cc/arch/x86_64/features.rs index ac8fcb7b7..985a47d23 100644 --- a/cc/arch/x86_64/features.rs +++ b/cc/arch/x86_64/features.rs @@ -9,6 +9,7 @@ // x86-64 Feature Code Generation (Variadic Functions, Byte Swapping, Bit Counting) // +use super::call::{BlockDst, UNROLL_LIMIT_BYTES}; use super::codegen::X86_64CodeGen; use super::lir::{popcount_sequence, GpOperand, MemAddr, ShiftCount, X86Inst}; use super::regalloc::{Loc, Reg}; @@ -530,12 +531,20 @@ impl X86_64CodeGen { }) } - /// Copy `nbytes` from `[src_base + src_off]` to `dst` at `dst_off`, in - /// descending power-of-two chunks so nothing past the object is written. + /// Copy `nbytes` from `[src_base + src_off]` to `dst` at `dst_off`, in the + /// descending power-of-two chunks `block_chunks` gives, so nothing past the + /// object is written. /// /// `%rcx` is the shuttle: it is declared clobbered by `VaArg`, so no live /// value is in it, and unlike `%r11` it cannot be `ap_base` (the va_list /// pointer lands there when it comes from a stack slot). + /// + /// Past [`UNROLL_LIMIT_BYTES`] the whole eightbytes go through `rep movsq` + /// and only the ragged tail is unrolled. Without a bound this was linear in + /// the aggregate -- 4 KB cost about 1100 instructions and 256 KB would be + /// the compile-time explosion the IR's own limit exists to prevent -- and + /// `va_arg` is the one place a whole aggregate is copied where the size is + /// the program's to choose. fn va_copy_bytes( &mut self, src_base: Reg, @@ -561,13 +570,32 @@ impl X86_64CodeGen { }); return; } - let mut done = 0; - while done < nbytes { - let chunk = [8, 4, 2, 1] - .into_iter() - .find(|c| *c <= nbytes - done) - .unwrap_or(1); - let size = OperandSize::from_bits(chunk as u32 * 8); + let mut at = 0; + if i64::from(nbytes) > UNROLL_LIMIT_BYTES { + let qwords = i64::from(nbytes) / 8; + // Neither base is `%rsp`, so the helper's pushes leave both where + // they are; a frame slot is not `%rsp`-relative either. + let to = match dst { + VaAggDst::Slot(slot) => BlockDst::At(self.stack_field(slot, dst_off)), + VaAggDst::Addr(base) => BlockDst::At(MemAddr::BaseOffset { + base, + offset: dst_off, + }), + VaAggDst::Value(_) => unreachable!("a register destination returned above"), + }; + self.emit_rep_movsq( + MemAddr::BaseOffset { + base: src_base, + offset: src_off, + }, + to, + qwords, + ); + at = (qwords * 8) as i32; + } + for (off, chunk) in crate::ir::memexpand::block_chunks(i64::from(nbytes - at)) { + let size = OperandSize::from_bits(chunk.bits()); + let done = at + off as i32; self.push_lir(X86Inst::Mov { size, src: GpOperand::Mem(MemAddr::BaseOffset { @@ -589,7 +617,6 @@ impl X86_64CodeGen { src: GpOperand::Reg(Reg::Rcx), dst: into, }); - done += chunk; } } diff --git a/cc/arch/x86_64/frame.rs b/cc/arch/x86_64/frame.rs index a74259505..e702c3abe 100644 --- a/cc/arch/x86_64/frame.rs +++ b/cc/arch/x86_64/frame.rs @@ -13,11 +13,11 @@ use crate::abi::{get_abi, Abi, ArgClass, RegClass}; use crate::arch::codegen::is_variadic_function; use crate::arch::lir::{ - complex_fp_info, complex_sse_regs, plan_pair_move, Directive, FpSize, OperandSize, PairMove, - Symbol, + complex_fp_info, complex_sse_regs, eightbyte_bytes, plan_pair_move, Directive, FpSize, + OperandSize, PairMove, Symbol, }; use crate::arch::x86_64::codegen::X86_64CodeGen; -use crate::arch::x86_64::lir::{GpOperand, MemAddr, X86Inst, XmmOperand}; +use crate::arch::x86_64::lir::{GpOperand, MemAddr, ShiftCount, X86Inst, XmmOperand}; use crate::arch::x86_64::regalloc::{spend_arg_regs, FrameBase, Loc, Reg, RegAlloc, XmmReg}; use crate::ir::{Function, Instruction, PseudoId, PseudoKind}; use crate::types::{TypeId, TypeKind, TypeTable}; @@ -79,30 +79,104 @@ impl X86_64CodeGen { return; }; + // The last eightbyte of a composite whose size is not a multiple of + // eight holds fewer bytes than the register carrying it. + let total = i64::from(type_size_bits / 8); let mut next_int = pair_start_int; let mut next_fp = pair_start_fp; for (i, class) in classes.iter().enumerate() { - let delta = (i * 8) as i32; + let at = (i * 8) as i64; + let bytes = eightbyte_bytes(total, at); + if bytes == 0 { + continue; + } if *class == crate::abi::RegClass::Sse { let src = fp_arg_regs[next_fp]; next_fp += 1; - self.push_lir(X86Inst::MovFp { - size: FpSize::Double, - src: XmmOperand::Reg(src), - dst: XmmOperand::Mem(self.stack_mem(offset - delta)), - }); + self.store_sse_bytes_to_local(src, offset, at, bytes); } else { let src = int_arg_regs[next_int]; next_int += 1; - self.push_lir(X86Inst::Mov { + self.store_reg_bytes_to_local(src, offset, at, bytes); + } + } + } + + /// Store the low `bytes` bytes of `src` into the local at `local`, + /// starting at the object's byte `at`, writing nothing past them. + /// + /// An eightbyte of a composite parameter arrives in a whole register, but + /// the last eightbyte of one that is not a multiple of eight holds fewer + /// bytes than the register does: `struct { int a, b, c; }` is twelve, and + /// storing the second register's eight wrote four bytes past a local that + /// `grow_frame` rounds up only to the type's own alignment -- four here. + /// + /// The pieces are `block_chunks`'s, so the rule is stated once, and the + /// value is shifted down as each leaves. The shifting is done in R10 -- + /// this file's scratch -- so the incoming argument register survives; a + /// natural width needs no shift and stores straight out of it. + fn store_reg_bytes_to_local(&mut self, src: Reg, local: i32, at: i64, bytes: i64) { + let mut shifted = 0; + for (off, chunk) in crate::ir::memexpand::block_chunks(bytes) { + if off != 0 { + if shifted == 0 && src != Reg::R10 { + self.push_lir(X86Inst::Mov { + size: OperandSize::B64, + src: GpOperand::Reg(src), + dst: GpOperand::Reg(Reg::R10), + }); + } + self.push_lir(X86Inst::Shr { size: OperandSize::B64, - src: GpOperand::Reg(src), - dst: GpOperand::Mem(self.stack_mem(offset - delta)), + count: ShiftCount::Imm(((off - shifted) * 8) as u8), + dst: Reg::R10, }); + shifted = off; } + let from = if off == 0 { src } else { Reg::R10 }; + let addr = self.stack_field(local, (at + off) as i32); + self.push_lir(X86Inst::Mov { + size: OperandSize::from_bits(chunk.bits()), + src: GpOperand::Reg(from), + dst: GpOperand::Mem(addr), + }); + } + } + + /// Store the low `bytes` bytes of the SSE register `src` into the local at + /// `local`, starting at the object's byte `at`. + /// + /// 2, 4, 8 and 16 bytes are one floating-point store. Any other width is + /// moved into a general register and stored from there, because there is + /// no SSE store of it: the second eightbyte of + /// `struct { float a, b, c; _Float16 d; }` is six bytes, and all of SSE + /// class. + fn store_sse_bytes_to_local(&mut self, src: XmmReg, local: i32, at: i64, bytes: i64) { + if let Some(size) = FpSize::exact_sse_store(bytes) { + let addr = self.stack_field(local, at as i32); + self.push_lir(X86Inst::MovFp { + size, + src: XmmOperand::Reg(src), + dst: XmmOperand::Mem(addr), + }); + return; } + self.push_lir(X86Inst::MovXmmGp { + size: OperandSize::B64, + src, + dst: Reg::R10, + }); + self.store_reg_bytes_to_local(Reg::R10, local, at, bytes); } + /// Copy a spilled parameter of `bytes` bytes out of the incoming argument + /// area into the local the body reads. + /// + /// Both sides move `block_chunks`'s widths. Stepping eight regardless -- + /// which this did -- wrote four bytes past a twelve-byte local, whose slot + /// is rounded up only to the type's own alignment; the incoming area is + /// eightbyte-granular so the *read* was safe, but the read is what the + /// store's width came from. fn copy_incoming_arg_to_local( &mut self, func: &crate::ir::Function, @@ -120,24 +194,25 @@ impl X86_64CodeGen { return; }; let dst_offset = *dst_offset; - let mut copied = 0; - while copied < bytes { + for (at, chunk) in crate::ir::memexpand::block_chunks(bytes.into()) { + let size = OperandSize::from_bits(chunk.bits()); + let at = at as i32; self.push_lir(X86Inst::Mov { - size: OperandSize::B64, + size, src: GpOperand::Mem(MemAddr::BaseOffset { base: Reg::Rbp, - offset: src_offset + copied, + offset: src_offset + at, }), dst: GpOperand::Reg(Reg::R10), }); // Locals grow downward from `dst_offset`, so later bytes sit at a - // smaller offset — the same convention `stack_mem` encodes. + // smaller offset — the same convention `stack_field` encodes. + let addr = self.stack_field(dst_offset, at); self.push_lir(X86Inst::Mov { - size: OperandSize::B64, + size, src: GpOperand::Reg(Reg::R10), - dst: GpOperand::Mem(self.stack_mem(dst_offset - copied)), + dst: GpOperand::Mem(addr), }); - copied += 8; } } @@ -793,50 +868,65 @@ impl X86_64CodeGen { if let Some(local) = func.locals.get(param_name) { if let Some(Loc::Stack(offset)) = self.locations.get_ref(local.sym) { let offset = *offset; - let (fp_size, imag_offset) = if let Some(n) = sse_struct { - // Two doubles are eight bytes each; - // a lone binary128 is one register - // holding all sixteen. - if n == 1 { - (FpSize::for_sse_aggregate(type_size_bits), 0) - } else { - (FpSize::Double, 8) + let total = i64::from(type_size_bits / 8); + if sse_struct.is_some() { + // An all-SSE aggregate: each register + // holds the eightbyte it was classified + // for, and the last one holds only what + // is left of the object. `sse_regs` is + // one for a lone binary128 -- SSE+SSEUP, + // sixteen bytes in one register -- and + // two for a pair of eightbytes, whose + // second is four bytes of a twelve-byte + // struct and not eight. Giving both the + // same width wrote four bytes past it. + for reg in 0..sse_regs { + let at = (reg * 8) as i64; + let bytes = if sse_regs == 1 { + total.min(16) + } else { + eightbyte_bytes(total, at) + }; + if bytes == 0 { + continue; + } + self.store_sse_bytes_to_local( + fp_arg_regs[fp_arg_idx + reg], + offset, + at, + bytes, + ); } } else { - complex_fp_info(types, &self.base.target, *typ) - }; - if sse_regs == 1 { - // One register holding the whole - // value. For `float _Complex` that - // is one eightbyte with both - // halves in it, so a 64-bit store - // writes all of it; for an - // aggregate it is whatever the - // class's size says, which is - // sixteen bytes for a binary128. - let whole = if sse_struct.is_some() { - fp_size + // A complex value: two elements of the + // same width at the base type's stride. + // `float _Complex` is one eightbyte with + // both halves in it, so a 64-bit store + // writes all of it. + let (fp_size, imag_offset) = + complex_fp_info(types, &self.base.target, *typ); + if sse_regs == 1 { + self.push_lir(X86Inst::MovFp { + size: FpSize::Double, + src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), + dst: XmmOperand::Mem(self.stack_mem(offset)), + }); } else { - FpSize::Double - }; - self.push_lir(X86Inst::MovFp { - size: whole, - src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), - dst: XmmOperand::Mem(self.stack_mem(offset)), - }); - } else { - // Store real part from first XMM register - self.push_lir(X86Inst::MovFp { - size: fp_size, - src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), - dst: XmmOperand::Mem(self.stack_mem(offset)), - }); - // Store imag part from second XMM register - self.push_lir(X86Inst::MovFp { - size: fp_size, - src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx + 1]), - dst: XmmOperand::Mem(self.stack_mem(offset - imag_offset)), - }); + // Store real part from first XMM register + self.push_lir(X86Inst::MovFp { + size: fp_size, + src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx]), + dst: XmmOperand::Mem(self.stack_mem(offset)), + }); + // Store imag part from second XMM register + self.push_lir(X86Inst::MovFp { + size: fp_size, + src: XmmOperand::Reg(fp_arg_regs[fp_arg_idx + 1]), + dst: XmmOperand::Mem( + self.stack_mem(offset - imag_offset), + ), + }); + } } } } diff --git a/cc/tests/codegen/block_moves.rs b/cc/tests/codegen/block_moves.rs new file mode 100644 index 000000000..04bfc5c77 --- /dev/null +++ b/cc/tests/codegen/block_moves.rs @@ -0,0 +1,263 @@ +// +// Copyright (c) 2025-2026 Jeff Garzik +// +// This file is part of the posixutils-rs project covered under +// the MIT License. For the full license text, please see the LICENSE +// file in the root directory of this project. +// SPDX-License-Identifier: MIT +// +// Blocks of bytes the back ends move, and the two rules they obey. +// +// An object's bytes are moved in descending power-of-two chunks, and the moves +// are bounded: past a threshold the copy becomes a bulk primitive instead of an +// unrolled run. `memexpand::block_chunks` and `BlockOp::limit` state both rules +// for the IR, and `codegen_struct_copy_across_the_inline_threshold` records the +// incident that put them there. +// +// The back ends cannot call `memcpy` -- they are past the point where a call +// can be synthesized -- so their bulk primitive is `rep movsq` or a counted +// loop. But the chunk rule is the same, and each site that wrote its own copy +// of it either rounded the size *up* or dropped the bound: +// +// * a two-SSE struct parameter stored both halves at eight bytes, so the +// four-byte high half of a 12-byte struct wrote four bytes past it; +// * the spilled-parameter prologue stepped eight regardless of width; +// * `va_arg` of a large aggregate unrolled with no bound at all; +// * the stacked-argument copy rounded the *source* read up, which is not the +// same as the destination: argument slots really are eightbyte-granular, so +// writing eight is correct where reading eight is not. +// +// Sizes here are deliberately not multiples of eight, and guards sit either +// side of the objects, so a rounded-up move shows as a wrong byte rather than +// landing in padding and passing by luck. +// + +use crate::codegen::asm_probe::{asm_for_with, body_of, AARCH64_LINUX, X86_64_LINUX}; +use crate::common::compile_and_run; + +/// How many instructions the body of `func` has. +fn body_insns(asm: &str, func: &str) -> usize { + body_of(asm, func) + .lines() + .filter(|l| { + let t = l.trim(); + !t.is_empty() && !t.starts_with('.') && !t.starts_with('#') && !t.ends_with(':') + }) + .count() +} + +/// A two-SSE struct parameter's high half is as wide as the half, not as wide +/// as a register. +/// +/// `struct P { float x, y, z; }` is classified into two SSE eightbytes, but the +/// second holds only four bytes. The prologue used one `FpSize` for both, so it +/// stored eight and wrote four bytes past the object. +#[test] +fn codegen_a_two_sse_struct_parameter_stores_only_its_own_bytes() { + let src = "\ +struct P { float x, y, z; }; +__attribute__((noinline)) float probe(struct P p) { float g = 99.f; return p.x + p.y + p.z + g; } +"; + let asm = asm_for_with("two_sse_param", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + let wide = body.matches("movsd").count(); + assert!( + wide <= 1, + "the high half of a 12-byte two-SSE struct is 4 bytes, so at most one \ + 8-byte fp store belongs in the prologue; found {wide}:\n{body}" + ); +} + +/// The spilled-parameter prologue moves no more than the parameter. +/// +/// `copy_incoming_arg_to_local` stepped eight bytes at a time regardless of +/// width, so a 12-byte struct read eight bytes at the incoming area's offset 8 +/// and wrote eight at the local's -- four past a local that is exactly twelve +/// bytes, because a slot is only rounded up to its type's own alignment. +#[test] +fn codegen_a_spilled_struct_parameter_is_copied_no_wider_than_itself() { + let src = "\ +struct P { int a, b, c; }; +__attribute__((noinline)) int probe(long a, long b, long c, long d, long e, long f, + struct P p) +{ return p.a + p.b + p.c; } +"; + let asm = asm_for_with("spilled_param", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + + // The incoming argument area is at a *positive* displacement from %rbp -- + // the saved frame pointer and return address are below it -- so reads of + // the spilled parameter are the moves from a positive offset. Everything + // the function writes is at a negative one. Matching on the substring + // "8(%rbp)" is not enough: "-88(%rbp)" ends with it. + let incoming_reads = |mnemonic: &str| { + body.lines() + .filter_map(|l| { + let t = l.trim(); + let rest = t.strip_prefix(mnemonic)?.trim_start(); + let (disp, _) = rest.split_once("(%rbp)")?; + disp.parse::().ok().filter(|d| *d > 0) + }) + .count() + }; + + // A 12-byte object is 8 + 4: exactly one eight-byte read, and the tail read + // with a four-byte one. + assert_eq!( + incoming_reads("movq"), + 1, + "a 12-byte spilled parameter has one 8-byte chunk, so one 8-byte read \ + of the incoming area; a second means the 4-byte tail was read as 8:\n{body}" + ); + assert_eq!( + incoming_reads("movl"), + 1, + "and its 4-byte tail is read with a 4-byte move:\n{body}" + ); +} + +/// `va_arg` of a large aggregate is bounded, like every other block move. +/// +/// The `va_arg` byte copy had no limit, so fetching a 4 KB aggregate emitted one +/// load/store pair per chunk -- about 1100 instructions on each target, and +/// linear in the object, so a 256 KB aggregate would be the compile-time +/// explosion `emit_aggregate_zero` used to be. +#[test] +fn codegen_va_arg_of_a_large_aggregate_is_bounded() { + let src = "\ +#include +struct Big { char c[4096]; }; +void sink(struct Big *); +void probe(int n, ...) +{ + va_list ap; + va_start(ap, n); + struct Big b = va_arg(ap, struct Big); + sink(&b); + va_end(ap); +} +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("va_arg_big", triple, src, &["-O2"]); + let n = body_insns(&asm, "probe"); + assert!( + n < 200, + "fetching a 4096-byte aggregate through va_arg must use a bulk copy, \ + not one pair per chunk: {n} instructions on {triple}" + ); + } +} + +/// The stacked-argument copy reads only the object's own bytes. +/// +/// The destination is the outgoing argument area, which is allocated in whole +/// eightbytes -- so rounding the *write* up is correct and deliberate. The read +/// is from the object, which is not, and a 12-byte struct read eight bytes at +/// offset 8. Four of them belong to whatever follows it, and the read faults if +/// the object ends a page. +#[test] +fn codegen_a_stacked_argument_reads_only_its_object() { + let src = "\ +struct P { int a, b, c; }; +void g(long, long, long, long, long, long, struct P); +void probe(struct P p) { g(1, 2, 3, 4, 5, 6, p); } +"; + let asm = asm_for_with("stacked_arg_src", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + + // Only the *read* is constrained. In AT&T order the memory operand of a + // load comes first, which is what distinguishes `movq 8(%r11), %rax` -- + // reading four bytes past a 12-byte object -- from `movq %rax, 8(%rsp)`, + // a write into the outgoing argument area. That area is allocated in whole + // eightbytes and its padding is unspecified, so the store's width is the + // back end's choice and this test does not pin it. + let wide_source_reads = body + .lines() + .filter(|l| { + let t = l.trim(); + let Some(operands) = t.strip_prefix("movq ") else { + return false; + }; + let Some((src_operand, _)) = operands.split_once(',') else { + return false; + }; + let src_operand = src_operand.trim(); + src_operand.starts_with("8(%r") && !src_operand.contains("%rbp") + }) + .count(); + assert_eq!( + wide_source_reads, 0, + "a 12-byte object has 4 bytes at offset 8, so reading 8 there is 4 past it:\n{body}" + ); +} + +/// The answers, with guards, across the sizes these paths classify differently. +/// +/// The assembly checks above pin the widths; this pins that the values survive. +/// None of the sizes is a multiple of eight and every object is fenced, so an +/// over-copy shows up as a clobbered guard rather than as padding nobody reads. +#[test] +fn codegen_block_moved_parameters_keep_their_values() { + let code = r#" +#include + +struct P3 { int a, b, c; }; /* 12 bytes, spilled/stacked */ +struct F3 { float x, y, z; }; /* 12 bytes, two SSE */ +struct B7 { unsigned char c[7]; }; /* 7 bytes */ +struct B13 { unsigned char c[13]; }; /* 13 bytes */ + +__attribute__((noinline)) int take_p3(long a, long b, long c, long d, long e, + long f, struct P3 p) +{ return p.a + p.b + p.c; } + +__attribute__((noinline)) float take_f3(struct F3 p) { return p.x + p.y + p.z; } + +__attribute__((noinline)) int take_b7(struct B7 v) +{ + int s = 0; + for (int i = 0; i < 7; i++) s += v.c[i]; + return s; +} + +__attribute__((noinline)) int take_b13(long a, long b, long c, long d, long e, + long f, struct B13 v) +{ + int s = 0; + for (int i = 0; i < 13; i++) s += v.c[i]; + return s; +} + +__attribute__((noinline)) int va_b13(int n, ...) +{ + va_list ap; + va_start(ap, n); + struct B13 v = va_arg(ap, struct B13); + va_end(ap); + int s = 0; + for (int i = 0; i < 13; i++) s += v.c[i]; + return s; +} + +int main(void) +{ + unsigned char lo = 0xA5; + struct P3 p = {1, 2, 3}; + struct F3 f = {1.f, 2.f, 4.f}; + struct B7 b7; + struct B13 b13; + unsigned char hi = 0x5A; + + for (int i = 0; i < 7; i++) b7.c[i] = (unsigned char)(i + 1); + for (int i = 0; i < 13; i++) b13.c[i] = (unsigned char)(i + 1); + + if (take_p3(1, 2, 3, 4, 5, 6, p) != 6) return 1; + if (take_f3(f) != 7.f) return 2; + if (take_b7(b7) != 28) return 3; + if (take_b13(1, 2, 3, 4, 5, 6, b13) != 91) return 4; + if (va_b13(0, b13) != 91) return 5; + if (lo != 0xA5 || hi != 0x5A) return 6; + return 0; +} +"#; + assert_eq!(compile_and_run("block_moved_params", code, &[]), 0); +} diff --git a/cc/tests/codegen/cross_abi.rs b/cc/tests/codegen/cross_abi.rs index af6423798..97f0ca1ec 100644 --- a/cc/tests/codegen/cross_abi.rs +++ b/cc/tests/codegen/cross_abi.rs @@ -25,7 +25,8 @@ use super::asm_probe::{ asm_for, asm_for_with, body_of, AARCH64_DARWIN, AARCH64_LINUX, X86_64_LINUX, }; use crate::common::{ - aarch64_cross_available, compile_with_host_cc, create_c_file, cross_link_and_run, run_c17, + aarch64_cross_available, compile_and_run, compile_with_host_cc, create_c_file, + cross_link_and_run, run_c17, }; /// AAPCS64 passes a `_Complex` as a two-element HFA, so it occupies **two** @@ -3446,3 +3447,148 @@ int named_s(struct P p) { return ns(p, 1); } ); } } + +/// The bytes a prologue stores into the frame, summed by the width of each +/// store's mnemonic. +/// +/// Only the prologue: the region before the first `.L` block label, which is +/// where the incoming arguments are written to their locals. A store is a move +/// whose *destination* is the frame, so the line ends with `(%rbp)`; a load +/// from the same place has it in the middle. +fn prologue_frame_store_bytes(body: &str) -> i64 { + body.lines() + .take_while(|l| !l.trim().starts_with(".L")) + .filter_map(|l| { + let t = l.trim(); + let (mnemonic, operands) = t.split_once(' ')?; + if !operands.ends_with("(%rbp)") { + return None; + } + match mnemonic { + "movq" | "movsd" => Some(8), + "movl" | "movss" => Some(4), + "movw" => Some(2), + "movb" => Some(1), + _ => None, + } + }) + .sum() +} + +/// A composite parameter that arrives in registers is stored no wider than it +/// is. +/// +/// Each eightbyte travels in a whole register, but the last eightbyte of a +/// composite that is not a multiple of eight holds fewer bytes than the +/// register does -- five, for a thirteen-byte one. `grow_frame` rounds a slot +/// up only to the type's own alignment, which is *one* for `unsigned char[13]`, +/// so storing the register's eight wrote three bytes past the local. +#[test] +fn codegen_a_register_pair_parameter_stores_only_its_own_bytes() { + let src = "\ +struct B13 { unsigned char c[13]; }; +int probe(struct B13 v) { return v.c[0] + v.c[12]; } +"; + let asm = asm_for_with("reg_pair_tail", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + assert_eq!( + prologue_frame_store_bytes(body), + 13, + "a thirteen-byte parameter is 8 + 4 + 1, and nothing more:\n{body}" + ); +} + +/// The same, for an all-SSE composite whose last eightbyte is a width no +/// floating-point store has. +/// +/// `struct { float a, b, c; _Float16 d; }` is fourteen bytes when packed, and +/// both of its eightbytes are SSE class -- the second holds six bytes. There +/// is no six-byte SSE store, so the register has to go through a general one; +/// rounding the width up instead wrote two bytes past the object. +#[test] +fn codegen_a_packed_two_sse_parameter_stores_only_its_own_bytes() { + let src = "\ +struct __attribute__((packed)) P6 { float a, b, c; _Float16 d; }; +float probe(struct P6 p) { return p.a + p.b + p.c; } +"; + let asm = asm_for_with("packed_two_sse", X86_64_LINUX, src, &["-O0"]); + let body = body_of(&asm, "probe"); + assert_eq!( + prologue_frame_store_bytes(body), + 14, + "a fourteen-byte two-SSE parameter is 8 + 4 + 2, and nothing more:\n{body}" + ); +} + +/// The bulk `va_arg` copy still moves the ragged tail. +/// +/// A bound is only half of the rule: `rep movsq` and the counted loop both +/// move whole eightbytes (sixteen bytes, for the loop), and 4093 bytes is +/// neither. A copy that stopped at the last whole unit would leave the last +/// bytes of the aggregate unwritten, which no instruction count can show. +#[test] +fn codegen_a_bulk_va_arg_copy_moves_the_ragged_tail() { + let src = "\ +#include +struct Odd { unsigned char c[4093]; }; +void sink(struct Odd *); +void probe(int n, ...) +{ + va_list ap; + va_start(ap, n); + struct Odd o = va_arg(ap, struct Odd); + sink(&o); + va_end(ap); +} +"; + for (triple, byte_move) in [(X86_64_LINUX, "movb"), (AARCH64_LINUX, "ldrb")] { + let asm = asm_for_with("va_arg_odd", triple, src, &["-O2"]); + let body = body_of(&asm, "probe"); + assert!( + body.contains(byte_move), + "4093 bytes ends on an odd byte, so the tail needs a byte move on \ + {triple}:\n{body}" + ); + } +} + +/// The values survive the paths above, with guards either side. +/// +/// Every size here is one the register-pair and all-SSE prologues classify +/// into two eightbytes whose second is short, and every object is fenced, so a +/// store that is wider than its object shows up as a clobbered guard. +#[test] +fn codegen_register_composite_parameters_keep_their_values() { + let code = r#" +struct B13 { unsigned char c[13]; }; +struct MX { int a, b; float c; }; +struct __attribute__((packed)) P6 { float a, b, c; _Float16 d; }; +struct __attribute__((packed)) G13 { long x; unsigned char c[5]; }; + +__attribute__((noinline)) int take_b13(struct B13 v) +{ int s = 0; for (int i = 0; i < 13; i++) s += v.c[i]; return s; } +__attribute__((noinline)) float take_mx(struct MX p) { return (float)(p.a + p.b) + p.c; } +__attribute__((noinline)) float take_p6(struct P6 p) { return p.a + p.b + p.c + (float)p.d; } +__attribute__((noinline)) long take_g13(struct G13 p) { return p.x + p.c[0] + p.c[4]; } + +int main(void) +{ + volatile unsigned char lo = 0xA5; + struct B13 b13; + struct MX mx = {1, 2, 4.f}; + struct P6 p6 = {1.f, 2.f, 4.f, (_Float16)8.f}; + struct G13 g13 = {7, {1, 2, 3, 4, 5}}; + volatile unsigned char hi = 0x5A; + + for (int i = 0; i < 13; i++) b13.c[i] = (unsigned char)(i + 1); + + if (take_b13(b13) != 91) return 1; + if (take_mx(mx) != 7.f) return 2; + if (take_p6(p6) != 15.f) return 3; + if (take_g13(g13) != 13) return 4; + if (lo != 0xA5 || hi != 0x5A) return 5; + return 0; +} +"#; + assert_eq!(compile_and_run("register_composite_params", code, &[]), 0); +} diff --git a/cc/tests/codegen/mod.rs b/cc/tests/codegen/mod.rs index 79f628fe5..51c55c32f 100644 --- a/cc/tests/codegen/mod.rs +++ b/cc/tests/codegen/mod.rs @@ -16,6 +16,7 @@ mod aarch64_runtime; pub mod asm_probe; mod atomics_asm; mod binary128; +mod block_moves; mod complex_fold; mod constant_branch; mod cross_abi; From cf75b55dd0158b03b07bb24ce01029966f594bd7 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 04:08:14 -0400 Subject: [PATCH 09/17] cc: make entering a declaration scope enter its VLA scope A VLA's storage is released by restoring the stack pointer its scope saved. The declaration scope and the VLA scope were two stacks maintained by hand, and the second was opened at three of the five places that open the first -- so a VLA declared in a `for` init clause or a statement expression was allocated and never released. The switch body had the mirror defect: a VLA scope with no declaration scope, so the declarations in it leaked out. `push_scope` now returns a `Scope` that `pop_scope` consumes, carrying the VLA depth to release. `open_vla_scope` is gone and `close_vla_scope` is private to `pop_scope`, so there is no way to enter one scope without the other and nothing left to keep in step. The `for` lowering existed twice -- the ordinary walk and the switch-body walk, differing only in how the body is lowered -- and is now `open_for`/`close_for`, which retires one of the five sites outright rather than pairing it. That duplication has now caused three separate defects in this file. The unclosed scope also poisoned forward `goto`: `place_label` records a label's depth as `vla_marks.len()`, so a scope that never closed left every later label's depth too high and `resolve_forward_goto_vla_restores` skipped the restore for the rest of the function. A computed `goto` released nothing at all, where the plain `goto` beside it had both backward and forward handling; both now go through one `release_vla_scopes_for_goto`, and an indirect jump takes the deepest depth over the address-taken labels, which is the only one that cannot free storage still live at another candidate. An `asm goto` has two exits and released on one: the label-edge block now releases before its branch, and is allocated when a VLA is open and not only when there are outputs to write back. `push_vla_mark` reading `break_depth` before the target is pushed is not a defect and is now documented as load-bearing. C17 6.8.5.3 makes the scope of a `for` init declaration the entire loop, so a VLA declared there is allocated once outside it and its scope encloses the exit: `break` lands inside that scope and must not release it, and `continue` must not either, the storage being live next iteration. Recording the depth first is what makes the unwind skip it, and two tests pin that. No program leaks stack today: c17 always keeps a frame pointer and the backend resets `%rsp` from it, which masks the imbalance -- 400,000 iterations of every shape here, with sizes varying per iteration, exhaust nothing at -O0 or -O2. The tests therefore assert the released stack, by counting saves against restores in the IR and by observing that a repeated declaration on a repeated path keeps its address. None asserts a crash. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize.rs | 105 ++++++++-- cc/ir/linearize_stmt.rs | 368 +++++++++++++++++++-------------- cc/ir/test_linearize.rs | 251 ++++++++++++++++++++++ cc/tests/c99/features.rs | 176 ++++++++++++++++ cc/tests/codegen/inline_asm.rs | 51 +++++ 5 files changed, 778 insertions(+), 173 deletions(-) diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index 6ed96ece0..0786b6861 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -222,7 +222,28 @@ pub(crate) struct StaticLocalInfo { // Linearizer -/// Linearizer context for converting AST to IR +/// A declaration scope the linearizer has entered. +/// +/// Handed out by [`Linearizer::push_scope`] and given back to +/// [`Linearizer::pop_scope`]. A block scope is also the lifetime of every +/// VLA declared in it (C17 6.2.4p7), so the token carries the [`VlaMark`] +/// depth as it stood on entry and leaving the scope puts the stack pointer +/// back to what the first mark above that depth captured. +/// +/// Carrying both in one token is the point. The declaration scope and the +/// VLA scope used to be two stacks opened by hand at separate call sites, +/// and only some of the sites that opened the first opened the second: a VLA +/// declared in a `for` init clause, in the `for` arm of the switch-body +/// walker, or in a statement expression was never released. Now there is no +/// way to enter one without the other, and none to leave one without the +/// other. +#[must_use = "a scope that is entered must be left through pop_scope"] +pub(crate) struct Scope { + /// The `vla_marks` depth on entry; every mark above it belongs to this + /// scope and is released when it ends. + pub(crate) vla_entry: usize, +} + /// A captured stack pointer and the loop/switch nesting it was captured at. /// /// See [`Linearizer::vla_marks`]. @@ -244,14 +265,27 @@ pub(crate) struct HiddenReturnSlot { pub(crate) arg_typ: TypeId, } +/// Where a jump whose VLA restore is still undecided may land. +pub(crate) enum GotoTarget { + /// `goto L;`, or one label edge of an `asm goto`: a single block. + Label(BasicBlockId), + /// `goto *p;`. The address is not known here, so the jump is taken to + /// reach any label whose address this function takes, and only the + /// scopes that *every* candidate lies outside of may be released -- the + /// **deepest** depth recorded for any of them. Releasing down to a + /// shallower one would free storage still in scope at another candidate, + /// which is the one error a missing restore cannot cause. + AnyAddressTaken, +} + /// A forward `goto` whose VLA restore is decided once its label is placed. /// /// `marks` is the mark stack as it stood at the jump, so the restore can name /// whichever scope the label turns out to sit in: `marks[label_depth]` is the /// stack pointer captured on entry to the outermost scope the jump leaves. pub(crate) struct PendingGotoVla { - /// The label jumped to. - pub(crate) label: String, + /// Where the jump goes. + pub(crate) target: GotoTarget, /// The block the branch was emitted into. pub(crate) bb: BasicBlockId, /// Where in that block the branch sits; the restore goes just before it. @@ -260,6 +294,7 @@ pub(crate) struct PendingGotoVla { pub(crate) marks: Vec, } +/// Linearizer context for converting AST to IR pub struct Linearizer<'a> { /// The module being built pub(crate) module: Module, @@ -344,7 +379,11 @@ pub struct Linearizer<'a> { /// /// Only labels in a function that declares a VLA appear here, so nothing /// is recorded for the ordinary case. - pub(crate) label_vla_depth: std::collections::HashMap, + /// + /// Keyed by the label's block rather than its name: a computed `goto` + /// knows its candidates only as the blocks in `addr_taken_labels`, and + /// one map serves both it and the named jumps. + pub(crate) label_vla_depth: std::collections::HashMap, /// Forward `goto`s that may be leaving a VLA's scope, to be resolved once /// every label's depth is known. /// @@ -490,14 +529,28 @@ impl<'a> Linearizer<'a> { } } - /// Push a new local scope. Subsequent `insert_local` calls will record + /// Enter a declaration scope. Subsequent `insert_local` calls will record /// the previous value so `pop_scope` can restore it. - pub(crate) fn push_scope(&mut self) { + /// + /// Entering a declaration scope *is* entering a VLA scope: the returned + /// [`Scope`] remembers the mark depth so `pop_scope` releases whatever + /// the scope allocated. See [`Scope`] for why the two are one operation. + pub(crate) fn push_scope(&mut self) -> Scope { self.local_scope_stack.push(Vec::new()); + Scope { + vla_entry: self.vla_marks.len(), + } } - /// Pop the current local scope, restoring all locals to their pre-scope values. - pub(crate) fn pop_scope(&mut self) { + /// Leave the scope `scope` opened: release the VLAs declared in it and + /// restore every local it shadowed. + /// + /// The stack restore comes first, while the block the scope ends in is + /// still the current one, and is emitted only on the falling-out path -- + /// a `break`, `continue`, `goto` or `return` that left already did its + /// own unwinding and terminated the block. + pub(crate) fn pop_scope(&mut self, scope: Scope) { + self.close_vla_scope(&scope); if let Some(entries) = self.local_scope_stack.pop() { for (sym, prev) in entries.into_iter().rev() { match prev { @@ -1299,14 +1352,20 @@ impl<'a> Linearizer<'a> { /// /// `static_locals` is deliberately not cleared: it persists across /// functions. - fn reset_for_function(&mut self, func: &FunctionDef) { + /// + /// Returns the function-level [`Scope`], which `linearize_function` gives + /// back once the body is lowered. Nothing is released there -- the + /// epilogue restores `%rsp` from the frame pointer, and the body's own + /// block scope has already dropped every mark -- but it is entered the + /// same way as any other scope so that no site can enter one without the + /// other. + fn reset_for_function(&mut self, func: &FunctionDef) -> Scope { // Reset per-function state self.next_pseudo = 0; self.next_bb = 0; self.var_map.clear(); self.locals.clear(); self.local_scope_stack.clear(); - self.push_scope(); // function-level scope self.label_map.clear(); self.break_targets.clear(); self.continue_targets.clear(); @@ -1324,6 +1383,10 @@ impl<'a> Linearizer<'a> { // Remove from extern_symbols since we're defining this function self.module.extern_symbols.remove(&self.current_func_name); // Note: static_locals is NOT cleared - it persists across functions + + // After `vla_marks.clear()`: the scope records the depth it starts + // at, which for the function scope has to be zero. + self.push_scope() } /// Whether a function body declares anything variably modified. @@ -1378,7 +1441,7 @@ impl<'a> Linearizer<'a> { // expression to the same rule. let written_labels = self.check_jumps_into_protected_scopes(&func.body); - self.reset_for_function(func); + let func_scope = self.reset_for_function(func); self.written_labels = written_labels; // Create function - use storage class from FunctionDef @@ -1767,8 +1830,11 @@ impl<'a> Linearizer<'a> { } } - // Pop function-level scope - self.pop_scope(); + // Pop function-level scope. Its VLA release is a no-op: the body's + // own block scope dropped every mark, and the block is terminated by + // the return above -- which is what must happen, since SSA has + // already run over the function by this point. + self.pop_scope(func_scope); // Add function to module if let Some(ir_func) = self.current_func.take() { @@ -6622,6 +6688,12 @@ impl<'a> Linearizer<'a> { ExprKind::StmtExpr { stmts, result } => { // GNU statement expression: ({ stmt; stmt; expr; }) + // It is a block, so it is a declaration scope like any other: + // its declarations do not outlive it, and the storage of a + // VLA declared in it goes back when it ends. Without the + // scope, `for (...) (void)({ int a[n]; ... });` allocated + // every time round and released nothing. + let scope = self.push_scope(); // Linearize all the statements first for item in stmts { match item { @@ -6629,8 +6701,11 @@ impl<'a> Linearizer<'a> { BlockItem::Statement(s) => self.linearize_stmt(s), } } - // The result is the value of the final expression - self.linearize_expr(result) + // The result is the value of the final expression, computed + // before the scope ends: it may read the VLA being released. + let value = self.linearize_expr(result); + self.pop_scope(scope); + value } ExprKind::BuiltinComplex { real, imag } => { diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index 940df5a91..f49be772d 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -67,6 +67,28 @@ enum JumpKind { Continue, } +/// A `for` loop whose body is being lowered: what +/// [`Linearizer::open_for`] set up and [`Linearizer::close_for`] finishes. +/// +/// `for` is lowered from two places -- the ordinary statement walk and the +/// switch-body walk, which must keep lowering the body through itself so the +/// `case` labels inside stay reachable -- and the two differ *only* in how +/// they lower the body. Everything else lives here, so a fix lands once +/// instead of twice: the loop's back edge had to be repaired in both copies, +/// and the release of a VLA declared in the init clause was missing from +/// both. +#[must_use = "an opened `for` loop must be finished with close_for"] +struct OpenFor { + /// The scope the init clause declares into, ended after `exit_bb`. + scope: Scope, + /// The block the back edge goes to. + cond_bb: BasicBlockId, + /// Where the body falls out to, and the `continue` target. + post_bb: BasicBlockId, + /// Where the loop ends, and the `break` target. + exit_bb: BasicBlockId, +} + impl<'a> super::linearize::Linearizer<'a> { pub(crate) fn linearize_stmt(&mut self, stmt: &Stmt) { match stmt { @@ -77,24 +99,18 @@ impl<'a> super::linearize::Linearizer<'a> { } Stmt::Block(items) => { - self.push_scope(); - // A VLA's storage lives until control leaves the scope of its - // declaration (C17 6.2.4p7). Capture the stack pointer on the - // way in and put it back on the way out, or a loop body's VLA - // is allocated afresh every iteration and never released -- - // `for (...) { int x[n]; }` died of stack exhaustion. - let vla_scope = self.open_vla_scope(); + // Entering the scope also captures the stack pointer, and + // leaving it puts the pointer back -- a VLA's storage lives + // until control leaves the scope of its declaration (C17 + // 6.2.4p7). See `Scope`. + let scope = self.push_scope(); for item in items { match item { BlockItem::Declaration(decl) => self.linearize_local_decl(decl), BlockItem::Statement(s) => self.linearize_stmt(s), } } - // Only on the falling-out path: a `break`, `continue` or - // `return` that left already did its own unwinding. - self.close_vla_scope(vla_scope); - - self.pop_scope(); + self.pop_scope(scope); } Stmt::If { @@ -275,6 +291,13 @@ impl<'a> super::linearize::Linearizer<'a> { ); } let addr = self.linearize_expr(target); + // A computed `goto` leaves scopes exactly as a named one + // does, and left out here it left a loop body's VLA behind + // every time round. Which scopes it leaves depends on which + // label it reaches, which is not known until every label is + // placed -- and then only as a set. See + // `GotoTarget::AnyAddressTaken`. + self.defer_vla_restore(GotoTarget::AnyAddressTaken); let (dispatch_bb, slot) = self.indirect_dispatch_block(); // Hand the address over in the hidden local and branch to the // one dispatch block, which is the only place that fans out to @@ -292,44 +315,7 @@ impl<'a> super::linearize::Linearizer<'a> { let label_str = self.str(*label).to_string(); let target = self.refer_to_label(&label_str, *pos); if let Some(current) = self.current_bb { - // A backward jump -- the label is already linearized, so - // it has a captured stack pointer -- leaves the scope of - // every VLA declared after it, and that storage has to go - // back. Otherwise `lab: int x[n]; ... goto lab;` grows the - // stack every time round until the program dies. - // - // A *forward* jump cannot be decided here: its label has - // no depth recorded yet, and whether it stays inside the - // scope of the VLAs in force or leaves it is exactly what - // decides between no restore and one. It is recorded and - // resolved in `resolve_forward_goto_vla_restores`. - // - // Leaving it to "the block's own exit does the restoring" - // was wrong: the branch *terminates* the block, so - // `close_vla_scope` emits nothing and then drops the - // marks, and the enclosing block has no mark of its own to - // undo them with. A `goto` out of a loop body's inner - // block grew the stack every time round. - if let Some(&depth) = self.label_vla_depth.get(&label_str) { - if let Some(m) = self.vla_marks.get(depth) { - let mark = m.mark; - self.emit_stack_restore(mark); - } - } else if !self.vla_marks.is_empty() { - let at = self - .current_func - .as_ref() - .and_then(|f| f.get_block(current)) - .map_or(0, |b| b.insns.len()); - let marks = self.vla_marks.iter().map(|m| m.mark).collect(); - self.pending_goto_vla - .push(crate::ir::linearize::PendingGotoVla { - label: label_str.clone(), - bb: current, - at, - marks, - }); - } + self.release_vla_scopes_for_goto(target); self.emit(Instruction::br(target)); self.link_bb(current, target); } @@ -1419,15 +1405,17 @@ impl<'a> super::linearize::Linearizer<'a> { self.switch_bb(exit_bb); } - pub(crate) fn linearize_for( - &mut self, - init: Option<&ForInit>, - cond: Option<&Expr>, - post: Option<&Expr>, - body: &Stmt, - ) { - // C99 for-loop declarations (e.g., for (int i = 0; ...)) are scoped to the loop. - self.push_scope(); + /// Lower everything a `for` loop needs before its body: the init clause, + /// the four blocks, the condition and the jump targets. Leaves the body + /// block current, for the caller to lower the body into. + /// + /// The returned [`OpenFor`] goes back to [`Self::close_for`]. + fn open_for(&mut self, init: Option<&ForInit>, cond: Option<&Expr>) -> OpenFor { + // C99 for-loop declarations (e.g., for (int i = 0; ...)) are scoped + // to the loop -- and so is a VLA declared there, which the scope + // releases at `exit_bb`. That is the right place for it: `break` and + // `continue` both land inside this scope, so neither may release it. + let scope = self.push_scope(); // Init if let Some(init) = init { @@ -1468,9 +1456,27 @@ impl<'a> super::linearize::Linearizer<'a> { // Body block self.break_targets.push(exit_bb); self.continue_targets.push(post_bb); - self.switch_bb(body_bb); - self.linearize_stmt(body); + + OpenFor { + scope, + cond_bb, + post_bb, + exit_bb, + } + } + + /// Close the loop [`Self::open_for`] opened, once its body is lowered: + /// the back edge through the post-expression, then the exit block, then + /// the loop's scope. + fn close_for(&mut self, open: OpenFor, post: Option<&Expr>) { + let OpenFor { + scope, + cond_bb, + post_bb, + exit_bb, + } = open; + if !self.is_terminated() { // After linearizing body, current_bb may be different from body_bb if let Some(current) = self.current_bb { @@ -1500,8 +1506,20 @@ impl<'a> super::linearize::Linearizer<'a> { // Exit block self.switch_bb(exit_bb); - // Restore locals to remove for-loop-scoped declarations - self.pop_scope(); + // Drop the for-loop-scoped declarations and release their storage. + self.pop_scope(scope); + } + + pub(crate) fn linearize_for( + &mut self, + init: Option<&ForInit>, + cond: Option<&Expr>, + post: Option<&Expr>, + body: &Stmt, + ) { + let open = self.open_for(init, cond); + self.linearize_stmt(body); + self.close_for(open, post); } pub(crate) fn linearize_switch(&mut self, expr: &Expr, body: &Stmt) { @@ -2509,11 +2527,12 @@ impl<'a> super::linearize::Linearizer<'a> { // `collect_switch_cases`, which has to agree about this. match body { Stmt::Block(items) => { - // Same VLA reclamation as the ordinary block arm: a switch - // body is lowered by its own walk, and leaving the rule out - // here let `switch (c) { case 0: { int v[n]; break; } }` - // inside a loop grow the stack without bound. - let vla_scope = self.open_vla_scope(); + // The same scope as the ordinary block arm: a switch body is + // lowered by its own walk, and leaving the rule out here let + // `switch (c) { case 0: { int v[n]; break; } }` inside a loop + // grow the stack without bound, and let the body's + // declarations outlive the switch. + let scope = self.push_scope(); for item in items { match item { BlockItem::Declaration(decl) => self.linearize_local_decl(decl), @@ -2528,7 +2547,7 @@ impl<'a> super::linearize::Linearizer<'a> { } } } - self.close_vla_scope(vla_scope); + self.pop_scope(scope); } stmt => { self.linearize_switch_stmt(stmt, case_values, case_bbs, default_bb, &mut case_idx) @@ -2661,80 +2680,23 @@ impl<'a> super::linearize::Linearizer<'a> { self.switch_bb(exit_bb); } + // Everything but the body is the ordinary `for` lowering; only + // the body has to go back through this walk, so the `case` + // labels inside it stay reachable. See `OpenFor`. Stmt::For { init, cond, post, body, } => { - self.push_scope(); - - if let Some(init) = init { - match init { - ForInit::Declaration(decl) => self.linearize_local_decl(decl), - ForInit::Expression(expr) => { - self.linearize_expr(expr); - } - } - } - - let cond_bb = self.alloc_bb(); - let body_bb = self.alloc_bb(); - let post_bb = self.alloc_bb(); - let exit_bb = self.alloc_bb(); - - if let Some(current) = self.current_bb { - if !self.is_terminated() { - self.emit(Instruction::br(cond_bb)); - self.link_bb(current, cond_bb); - } - } - - self.switch_bb(cond_bb); - if let Some(cond_expr) = cond { - self.branch_on_condition(cond_expr, body_bb, exit_bb); - } else { - self.emit(Instruction::br(body_bb)); - self.link_bb(cond_bb, body_bb); - } - - self.break_targets.push(exit_bb); - self.continue_targets.push(post_bb); - - self.switch_bb(body_bb); + let open = self.open_for(init.as_ref(), cond.as_ref()); self.linearize_switch_stmt(body, case_values, case_bbs, default_bb, case_idx); - if !self.is_terminated() { - if let Some(current) = self.current_bb { - self.emit(Instruction::br(post_bb)); - self.link_bb(current, post_bb); - } - } - - self.break_targets.pop(); - self.continue_targets.pop(); - - self.switch_bb(post_bb); - if let Some(post_expr) = post { - self.linearize_expr(post_expr); - } - // From the block the post-expression ended in, which `&&`, `||` and - // `?:` can make a different one from post_bb. Linking the back edge - // from post_bb itself recorded an edge out of a block that no longer - // holds the branch, and left the merge block that does hold it with an - // unrecorded successor -- the loop then never terminated. - if let Some(current) = self.current_bb { - self.emit(Instruction::br(cond_bb)); - self.link_bb(current, cond_bb); - } - - self.switch_bb(exit_bb); - self.pop_scope(); + self.close_for(open, post.as_ref()); } Stmt::Block(items) => { - self.push_scope(); // See the sibling arm in `linearize_switch_body`. - let vla_scope = self.open_vla_scope(); + let scope = self.push_scope(); for item in items { match item { BlockItem::Declaration(decl) => self.linearize_local_decl(decl), @@ -2749,8 +2711,7 @@ impl<'a> super::linearize::Linearizer<'a> { } } } - self.close_vla_scope(vla_scope); - self.pop_scope(); + self.pop_scope(scope); } Stmt::If { @@ -3061,10 +3022,17 @@ impl<'a> super::linearize::Linearizer<'a> { // with every output unstored. Each label edge now gets a block of its // own that writes the outputs back and then jumps to the label. let writes_back = skip_post_handling.iter().any(|skip| !skip); + // An `asm goto` has two kinds of exit and each must release the VLA + // scopes it leaves. The fall-through is released by the enclosing + // scope's own end, but a label edge branches straight past it -- so + // the release goes in the edge block, which therefore has to exist + // even when there is nothing to write back. + let releases_vlas = !self.vla_marks.is_empty(); + let needs_edge_block = writes_back || releases_vlas; let label_edges: Vec<(BasicBlockId, BasicBlockId, String)> = ir_goto_labels .iter() .map(|(target, name)| { - let edge = if writes_back { + let edge = if needs_edge_block { self.alloc_bb() } else { *target @@ -3109,16 +3077,21 @@ impl<'a> super::linearize::Linearizer<'a> { // Without this, code would fall through to whatever block comes next in layout self.emit(Instruction::br(fall_through)); - if writes_back { + if needs_edge_block { for (edge, target, _) in &label_edges { self.switch_bb(*edge); - self.emit_asm_output_writeback( - outputs, - &ir_outputs, - &skip_post_handling, - ¶m_outputs, - &output_places, - ); + if writes_back { + self.emit_asm_output_writeback( + outputs, + &ir_outputs, + &skip_post_handling, + ¶m_outputs, + &output_places, + ); + } + // The jump leaves this scope; the fall-through does + // not. Same rule as a plain `goto` to the label. + self.release_vla_scopes_for_goto(*target); self.emit(Instruction::br(*target)); self.link_bb(*edge, *target); } @@ -3265,6 +3238,15 @@ impl<'a> super::linearize::Linearizer<'a> { /// One mark per declaration, not per block: a label sitting between two /// VLAs must release only the one declared after it, and a block-wide /// mark cannot express that. + /// + /// The nesting depths are read *before* the construct being lowered + /// pushes its own break or continue target, and that is deliberate. A + /// VLA declared in a `for` init clause, or in the controlling expression + /// of a `switch`, is allocated once, outside the loop or switch, and its + /// scope encloses the exit the jump lands on -- so a `break` or + /// `continue` inside must *not* release it. `continue` especially: the + /// storage is still live on the next iteration. The construct's own + /// scope, which ends after its exit block, is what releases it. fn push_vla_mark(&mut self) { if self.current_bb.is_none() { return; @@ -3282,14 +3264,6 @@ impl<'a> super::linearize::Linearizer<'a> { }); } - /// The marks in force on entry to a block, to restore and drop on exit. - /// - /// Returns the depth of [`Linearizer::vla_marks`] so - /// [`Self::close_vla_scope`] knows which of them this block added. - fn open_vla_scope(&self) -> usize { - self.vla_marks.len() - } - /// Give every forward `goto` the VLA restore its label turned out to need. /// /// Deferred because a label's depth is known only once it has been placed. @@ -3310,9 +3284,7 @@ impl<'a> super::linearize::Linearizer<'a> { pending.sort_by_key(|p| (p.bb.0, std::cmp::Reverse(p.at))); let void_ptr = self.types.void_ptr_id; for p in pending { - // An undefined label is diagnosed elsewhere; there is no branch - // here to put a restore in front of. - let Some(&depth) = self.label_vla_depth.get(&p.label) else { + let Some(depth) = self.goto_target_vla_depth(&p.target) else { continue; }; let Some(&mark) = p.marks.get(depth) else { @@ -3331,12 +3303,34 @@ impl<'a> super::linearize::Linearizer<'a> { } } - /// Release everything the block allocated and forget its marks. - fn close_vla_scope(&mut self, entry: usize) { + /// How many VLA scopes a jump is *inside* at the label it reaches, or + /// `None` if that cannot be said -- an undefined label, diagnosed + /// elsewhere, or a computed `goto` in a function that takes no label's + /// address. + fn goto_target_vla_depth(&self, target: &GotoTarget) -> Option { + match target { + GotoTarget::Label(bb) => self.label_vla_depth.get(bb).copied(), + // See [`GotoTarget::AnyAddressTaken`]: the deepest candidate is + // the only depth that releases nothing another candidate still + // needs. + GotoTarget::AnyAddressTaken => self + .addr_taken_labels + .iter() + .filter_map(|bb| self.label_vla_depth.get(bb).copied()) + .max(), + } + } + + /// Release everything the scope allocated and forget its marks. + /// + /// Called only from [`Linearizer::pop_scope`], so that leaving a + /// declaration scope and leaving a VLA scope are the same act. + pub(crate) fn close_vla_scope(&mut self, scope: &Scope) { + let entry = scope.vla_entry; if self.vla_marks.len() <= entry { return; } - // The first mark the block took is the stack as it stood on entry, + // The first mark the scope took is the stack as it stood on entry, // so one restore undoes all of them. let mark = self.vla_marks[entry].mark; if !self.is_terminated() && self.current_bb.is_some() { @@ -3345,6 +3339,64 @@ impl<'a> super::linearize::Linearizer<'a> { self.vla_marks.truncate(entry); } + /// Release every VLA scope a jump to the block `target` leaves. + /// + /// A backward jump -- the label is already linearized, so it has a + /// recorded depth -- leaves the scope of every VLA declared after it, and + /// that storage has to go back. Otherwise `lab: int x[n]; ... goto lab;` + /// grows the stack every time round until the program dies. + /// + /// A *forward* jump cannot be decided here: its label has no depth + /// recorded yet, and whether it stays inside the scope of the VLAs in + /// force or leaves it is exactly what decides between no restore and one. + /// It is recorded and resolved in + /// [`Self::resolve_forward_goto_vla_restores`]. + /// + /// Leaving it to "the scope's own exit does the restoring" was wrong: the + /// branch *terminates* the block, so `close_vla_scope` emits nothing and + /// then drops the marks, and the enclosing scope has no mark of its own + /// to undo them with. A `goto` out of a loop body's inner block grew the + /// stack every time round. + /// + /// Every jump that names a label goes through here: a `goto`, and each + /// label edge of an `asm goto`. + fn release_vla_scopes_for_goto(&mut self, target: BasicBlockId) { + if let Some(&depth) = self.label_vla_depth.get(&target) { + if let Some(m) = self.vla_marks.get(depth) { + let mark = m.mark; + self.emit_stack_restore(mark); + } + } else { + self.defer_vla_restore(GotoTarget::Label(target)); + } + } + + /// Record a restore whose depth is not yet known, to be placed by + /// [`Self::resolve_forward_goto_vla_restores`] at the end of the + /// function. The restore goes where the current block ends now, which is + /// ahead of the branch the caller is about to emit. + fn defer_vla_restore(&mut self, target: GotoTarget) { + if self.vla_marks.is_empty() { + return; + } + let Some(current) = self.current_bb else { + return; + }; + let at = self + .current_func + .as_ref() + .and_then(|f| f.get_block(current)) + .map_or(0, |b| b.insns.len()); + let marks = self.vla_marks.iter().map(|m| m.mark).collect(); + self.pending_goto_vla + .push(crate::ir::linearize::PendingGotoVla { + target, + bb: current, + at, + marks, + }); + } + /// Put the stack pointer back to what `mark` captured. fn emit_stack_restore(&mut self, mark: PseudoId) { self.emit( @@ -3425,7 +3477,7 @@ impl<'a> super::linearize::Linearizer<'a> { // declaration that creates it lies between the label and the // jump. if self.func_has_vla { - self.label_vla_depth.insert(name_str, self.vla_marks.len()); + self.label_vla_depth.insert(label_bb, self.vla_marks.len()); } } diff --git a/cc/ir/test_linearize.rs b/cc/ir/test_linearize.rs index 2bca4d8fe..a4959ba94 100644 --- a/cc/ir/test_linearize.rs +++ b/cc/ir/test_linearize.rs @@ -9642,3 +9642,254 @@ fn conditional_lowerings_keep_the_cfg_consistent() { } } } + +// VLA scope exit + +/// Every `stacksave` a function emits is matched by a `stackrestore` on each +/// path that leaves the scope it opened. +/// +/// A VLA's storage is released by restoring the stack pointer the scope saved, +/// so the two have to balance -- and on *every* exit, which for an `asm goto` +/// means one per edge. The declaration scope and the VLA scope are tracked +/// separately, and the second was opened at only three of the places that open +/// the first, so a VLA declared in a `for`-init clause or a statement +/// expression was never released. +/// +/// The consequence is currently masked: c17 always keeps a frame pointer +/// (`-fomit-frame-pointer` is accepted and ignored) and the backend resets +/// `%rsp` from it, so no program leaks stack today. That masking is a frame +/// choice rather than a guarantee, and the imbalance also poisons the forward +/// `goto` machinery, which reads `vla_marks.len()` as a depth -- an unclosed +/// scope makes every later label's recorded depth too high and the restore is +/// skipped. This is stated on the IR because that is where it is true. +#[test] +fn a_vla_scope_releases_the_stack_on_every_exit() { + let target = Target::host(); + + let counts = |src: &str| -> (usize, usize) { + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + let n = |op| { + func.blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| i.op == op) + .count() + }; + (n(Opcode::StackSave), n(Opcode::StackRestore)) + }; + + // (tag, source, expected saves, expected restores) + let cases: &[(&str, &str, usize, usize)] = &[ + // The control: a plain block already balances. + ( + "block", + "void s(int*); int f(int n){ for(int k=0;k<3;k++){ int a[n]; a[0]=k; s(a); } return 0; }", + 1, + 1, + ), + // A `for`-init clause is a declaration scope like any other. + ( + "for_init", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) for(int a[n];0;){ s(a); } return 0; }", + 1, + 1, + ), + // The switch-body walker carries a second copy of the `for` lowering. + ( + "for_init_in_switch", + "void s(int*); int f(int n,int x){ switch(x){ case 1: \ + for(int k=0;k<3;k++) for(int a[n];0;){ s(a); } return 0; } return 1; }", + 1, + 1, + ), + // A statement expression is a sixth scope entry. + ( + "stmt_expr", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) \ + (void)({ int a[n]; a[0]=k; s(a); 0; }); return 0; }", + 1, + 1, + ), + // Leaving the scope by `break` unwinds it. + ( + "break_out_of_for_init", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) \ + for(int a[n];;){ a[0]=k; s(a); break; } return 0; }", + 1, + 1, + ), + // And by a forward `goto`, which is also what the depth bookkeeping + // needs to stay right for every label after it. + ( + "goto_out_of_for_init", + "void s(int*); int f(int n){ for(int k=0;k<3;k++) \ + for(int a[n];;){ a[0]=k; s(a); goto L; } L: return 0; }", + 1, + 1, + ), + // A computed `goto` leaves a scope exactly as a plain one does. + ( + "computed_goto", + "void s(int*); int f(int n){ void*p=&&L; for(int k=0;k<3;k++){ int a[n]; \ + a[0]=k; s(a); goto *p; } L: return 0; }", + 1, + 1, + ), + // `asm goto` has two exits, so it needs a restore on each: the + // fall-through and the label edge. One restore here means the jump + // leaves the scope without releasing it. + ( + "asm_goto", + "void s(int*); int f(int n){ for(int k=0;k<3;k++){ int a[n]; a[0]=k; s(a); \ + __asm__ goto(\"\" :::: L); } L: return 0; }", + 1, + 2, + ), + ]; + + for (tag, src, want_save, want_restore) in cases { + let (saves, restores) = counts(src); + assert_eq!( + (saves, restores), + (*want_save, *want_restore), + "{tag}: expected {want_save} stacksave / {want_restore} stackrestore, \ + got {saves} / {restores}\nsource: {src}" + ); + } +} + +/// The counts of `stacksave`/`stackrestore` in `f`, for the VLA scope tests. +fn vla_stack_ops(src: &str) -> (usize, usize) { + let target = Target::host(); + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + let n = |op| { + func.blocks + .iter() + .flat_map(|b| b.insns.iter()) + .filter(|i| i.op == op) + .count() + }; + (n(Opcode::StackSave), n(Opcode::StackRestore)) +} + +/// Entering a declaration scope and entering a VLA scope are one operation, +/// so nesting the first nests the second: each scope releases exactly what +/// was allocated after it was entered, innermost first. +#[test] +fn nested_scopes_each_release_only_their_own_vlas() { + let src = "void s(int*); int f(int n){ for(int k=0;k<2;k++){ int a[n]; \ + { int b[n]; s(b); } s(a); } return 0; }"; + assert_eq!( + vla_stack_ops(src), + (2, 2), + "each of the two scopes captures and releases once" + ); + + // And in the right order: the inner scope puts the stack back to what it + // captured on entry, then the outer one to what *it* captured. Restoring + // the outer mark first would free the inner array while it is still in + // scope. + let target = Target::host(); + let module = linearize_source(src, &target); + let func = module + .functions + .iter() + .find(|f| f.name == "f") + .expect("function f"); + let mut saves = Vec::new(); + let mut restores = Vec::new(); + for insn in func.blocks.iter().flat_map(|b| b.insns.iter()) { + match insn.op { + Opcode::StackSave => saves.push(insn.target.expect("stacksave target")), + Opcode::StackRestore => restores.push(insn.src[0]), + _ => {} + } + } + assert_eq!( + restores, + vec![saves[1], saves[0]], + "scopes must be released innermost first" + ); +} + +/// A VLA declared in a `for` init clause is allocated once, ahead of the +/// loop, and its scope *encloses* the loop's exit -- so neither `break` nor +/// `continue` may release it. `continue` especially: the storage is live on +/// the next iteration, and freeing it there would hand the loop a dangling +/// array. Only the loop's own scope, which ends after the exit block, puts +/// the stack back. +/// +/// This is what fixes the order in which [`Linearizer::push_vla_mark`] reads +/// the break and continue depths: before the loop pushes its targets, so +/// `unwind_vla_marks` sees the mark as taken *outside* the construct being +/// left and leaves it alone. +#[test] +fn a_for_init_vla_outlives_break_and_continue() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ for(int a[n];x;){ s(a); \ + if(x==1) continue; if(x==2) break; } return 0; }" + ), + (1, 1), + "the loop's scope is the only release; break and continue land inside it" + ); +} + +/// The mirror image: a VLA declared in the loop *body* is allocated afresh +/// every iteration, so every way out of the body has to release it -- the +/// fall-through through the body scope's end, and the `break` that jumps +/// past it. +#[test] +fn a_loop_body_vla_is_released_on_break_as_well_as_fallthrough() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ for(;x;){ int a[n]; s(a); \ + if(x==2) break; } return 0; }" + ), + (1, 2), + "one release on the break edge, one on the way out of the body" + ); +} + +/// A `switch` body is lowered by a walk of its own, and its braces are a +/// declaration scope there too: a VLA declared directly in it is released +/// when the switch ends, not left for the enclosing loop to accumulate. +#[test] +fn a_switch_body_block_is_a_scope() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ for(int k=0;k<2;k++) \ + switch(x){ default: { int a[n]; s(a); } } return 0; }" + ), + (1, 1), + "the switch body's block releases what it declared" + ); +} + +/// A backward `goto` to a label ahead of a VLA declaration leaves that +/// declaration's scope, so it restores the stack as it stood at the label -- +/// otherwise the loop the jump makes grows the stack every time round. The +/// label's depth is recorded per block, which is also what lets a computed +/// `goto` ask the same question of every candidate at once. +#[test] +fn a_backward_goto_past_a_vla_declaration_releases_it() { + assert_eq!( + vla_stack_ops( + "void s(int*); int f(int n,int x){ lab: { int a[n]; s(a); \ + if(x--) goto lab; } return 0; }" + ), + (1, 2), + "the jump back to `lab` puts the stack where the label found it, and \ + the path that falls out of the block releases it too" + ); +} diff --git a/cc/tests/c99/features.rs b/cc/tests/c99/features.rs index 1000f08eb..098d13701 100644 --- a/cc/tests/c99/features.rs +++ b/cc/tests/c99/features.rs @@ -1192,3 +1192,179 @@ fn c99_deeply_nested_constructs_compile() { code.push_str("int main(void) { return deep() == 1 && labels(3) == 1 ? 0 : 1; }\n"); assert_eq!(compile_and_run("deep_nesting", &code, &[]), 0); } + +/// Every way out of a scope that declares a VLA puts the stack pointer back. +/// +/// Observed without waiting for an exhaustion that a frame pointer hides: +/// the same declaration reached on the same path allocates at the same +/// address every time round *if and only if* the previous iteration released +/// it. A scope that never releases marches the address up the stack, and the +/// first mismatch is one iteration later. +/// +/// The declaration scope and the VLA scope used to be opened by hand at +/// separate call sites, and three of the sites that opened the first never +/// opened the second: a `for` init clause, the copy of the `for` lowering +/// inside the switch-body walk, and a statement expression. A computed +/// `goto` left no scope at all. +#[test] +fn c99_a_vla_scope_is_released_on_every_exit() { + let code = r#" +/* A VLA in a `for` init clause: allocated once per execution of the inner + `for` statement, released when that statement ends. */ +static int for_init(int n) { + void *first = 0; + for (int k = 0; k < 8; k++) + for (int a[n]; ; ) { + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 1; + break; /* leaves by `break` */ + } + return 0; +} + +/* The same shape, lowered by the switch-body walk instead. */ +static int for_init_in_switch(int n, int x) { + void *first = 0; + switch (x) { + case 1: + for (int k = 0; k < 8; k++) + for (int a[n]; ; ) { + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 2; + break; + } + return 0; + } + return 3; +} + +/* A statement expression is a block, so it is a scope. */ +static int stmt_expr(int n) { + void *first = 0; + int bad = 0; + for (int k = 0; k < 8; k++) + (void)({ + int a[n]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) bad = 4; + 0; + }); + return bad; +} + +/* Leaving by a forward `goto`, which also has to leave the label bookkeeping + straight for every label after it. */ +static int goto_out(int n) { + void *first = 0; + for (int k = 0; k < 8; k++) { + for (int a[n]; ; ) { + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 5; + goto next; + } + next: + ; + } + return 0; +} + +/* And by a computed `goto`, which leaves a scope exactly as a plain one + does. The jump is the loop, so every iteration goes through it. */ +static int computed_goto(int n) { + void *first = 0; + int k = 0; + void *back = &⊤ +top: + { + int a[n]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 6; + k++; + if (k < 8) goto *back; + } + return 0; +} + +int main(void) { + int n = 7; + int rc; + if ((rc = for_init(n)) != 0) return rc; + if ((rc = for_init_in_switch(n, 1)) != 0) return rc; + if ((rc = stmt_expr(n)) != 0) return rc; + if ((rc = goto_out(n)) != 0) return rc; + if ((rc = computed_goto(n)) != 0) return rc; + return 0; +} +"#; + assert_eq!( + compile_and_run("c99_vla_scope_release", code, &[]), + 0, + "a VLA scope must release its storage on every exit" + ); +} + +/// A block inside a `switch` body is a declaration scope there too: the +/// switch-body walk has a lowering of its own, and it used to release a VLA +/// declared in such a block without ever entering the declaration scope, so +/// the block's ordinary declarations outlived it. +#[test] +fn c99_a_switch_body_block_is_a_declaration_scope() { + let code = r#" +int main(void) { + int v = 1; + int x = 2; + switch (x) { + default: { + int v = 10; /* shadows the outer v only inside these braces */ + if (v != 10) return 1; + break; + } + } + if (v != 1) return 2; /* the inner declaration must not have escaped */ + + /* And a VLA declared there is released when the block ends. */ + void *first = 0; + for (int k = 0; k < 8; k++) + switch (x) { + default: { + int a[x + 5]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 3; + } + } + return 0; +} +"#; + assert_eq!(compile_and_run("c99_switch_body_block_scope", code, &[]), 0); +} + +/// A declaration in a statement expression does not outlive it. +/// +/// The statement expression had no declaration scope at all, so its locals +/// were inserted into the enclosing one and stayed there -- an inner `x` +/// went on shadowing the outer one after the `})`. +#[test] +fn c99_a_statement_expression_is_a_declaration_scope() { + let code = r#" +int main(void) { + int x = 1; + int y = ({ int x = 41; x + 1; }); + if (y != 42) return 1; + if (x != 1) return 2; /* the inner x must be gone */ + { + typedef int T; + int z = ({ typedef long T; (int)sizeof(T); }); + if (z != (int)sizeof(long)) return 3; + if ((int)sizeof(T) != (int)sizeof(int)) return 4; + } + return 0; +} +"#; + assert_eq!(compile_and_run("c99_stmt_expr_scope", code, &[]), 0); +} diff --git a/cc/tests/codegen/inline_asm.rs b/cc/tests/codegen/inline_asm.rs index 8654fb99a..0d7c12df5 100644 --- a/cc/tests/codegen/inline_asm.rs +++ b/cc/tests/codegen/inline_asm.rs @@ -2685,3 +2685,54 @@ fn codegen_inline_asm_operand_address_used_elsewhere() { } } } + +/// An `asm goto` releases the VLA scopes its label edge leaves. +/// +/// It has two exits and needs a release on each. The enclosing scope's +/// release sits on the fall-through, and the label edge branches straight +/// past it -- so a loop whose back edge runs through the jump allocated +/// every time round and freed nothing. Each edge now has a block of its own +/// and the release goes there. +/// +/// Observed by address rather than by exhaustion: the same declaration +/// reached on the same path allocates at the same address every iteration if +/// and only if the previous one was released. +#[test] +#[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))] +fn codegen_asm_goto_releases_a_vla_scope_on_its_label_edge() { + #[cfg(target_arch = "x86_64")] + let jump = "jmp %l[again]"; + #[cfg(target_arch = "aarch64")] + let jump = "b %l[again]"; + + let code = format!( + r#" +static int loop_through_asm_goto(int n) {{ + void *first = 0; + int k = 0; +top: + {{ + int a[n]; + a[0] = k; + if (!first) first = (void *)a; + else if (first != (void *)a) return 1; + k++; + __asm__ goto ("{jump}" : : : : again); + return 2; /* the asm always branches */ + }} +again: + if (k < 8) goto top; + return 0; +}} + +int main(void) {{ + return loop_through_asm_goto(7); +}} +"# + ); + assert_eq!(compile_and_run("asm_goto_vla_scope", &code, &[]), 0); + assert_eq!( + compile_and_run_optimized("asm_goto_vla_scope_opt", &code), + 0 + ); +} From 33ff4e04d6eebcf5b11dedd3442639b6f248aedb Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 04:51:34 -0400 Subject: [PATCH 10/17] cc: override the subobject a designator names, not everything it overlaps C17 6.7.9p19 replaces a previously listed initializer for the *same* subobject and leaves the others alone. The static path merged its field initializers by byte span, so any intersection dropped the earlier entry whole: `{ .t = {1,2}, .t.y = 9 }` lost the 1 with the 2. It also sorted by address before resolving, which read "later" as later in the object rather than later in the list, so in `{ .z = 7, .t.y = 9, .t = {1,2} }` the whole-field initializer that is written last was the one discarded. The sort is still needed downstream by the bit-field emitter and now runs after the merge. The automatic path had neither problem -- it stores in list order and lets a later store land on an earlier one -- but that is also why it was wrong for unions, where a union holds one member at a time: `{ .u.i = 0x01020304, .u.s.b = 9 }` must leave the union holding `.u.s` with the rest zero, and storing one byte over the int kept three of its bytes. So the two paths disagreed in *both* directions and neither could simply adopt the other. Byte spans cannot tell the two cases apart -- `.t.y` inside `.t` and `.u.s.b` inside `.u.i` are both one span inside another -- so `classify_subobject` walks the type instead and answers `Member`, `ThroughUnion` or `NotASubobject`. The static merge folds a contained initializer into the earlier one through the `Initializer` tree for a member, replaces the union's subtree for a union, and keeps the old drop for anything it cannot express. The automatic path keeps its stores and gains the same question: before each store it clears the bytes a reset requires, which is the entry's own span plus, per earlier overlapping entry, the whole of one it contains and only the union's bytes for one it reaches through a union. A structured merge rather than a byte map, because the `Initializer` tree carries `SymAddr` relocations that cannot be split into bytes. Two more automatic-path defects fall out, both now matching gcc in each storage duration: a whole-field initializer repeated (`{ .t = {1,2}, .t = {5} }`) left the old field's tail, and a union initialized twice by member kept the first member's bytes. Every case is asserted in both storage durations, since a disagreement between them is how all of this survived. Two shapes stay at the old behaviour and are not fixed here. An overlap where either side is a bit-field keeps the drop, because carrier bytes are merged downstream of the `Initializer` tree and cannot be folded into it without the bit-field emitter taking part. And an override naming a subobject of the member a union already holds resets the union, because the tree records no discriminant and the fold cannot tell "same member, deeper" from "different member". Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize.rs | 28 ++ cc/ir/linearize_init.rs | 824 ++++++++++++++++++++++++++++++++++- cc/ir/linearize_stmt.rs | 134 ++++++ cc/tests/c99/initializers.rs | 118 +++++ 4 files changed, 1082 insertions(+), 22 deletions(-) diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index 0786b6861..b8005114e 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -127,6 +127,14 @@ pub(crate) struct ResolvedDesignator { pub(crate) struct RawFieldInit { pub(crate) offset: usize, pub(crate) field_size: usize, + /// The type of the subobject this initializer names. + /// + /// Carried so that resolving two initializers that describe overlapping + /// storage can ask *how* they overlap: a later one naming a member of an + /// earlier one's struct or array replaces only that member, while one + /// reachable only through a union replaces the union's whole contents. + /// Byte spans alone cannot tell the two apart. + pub(crate) typ: TypeId, pub(crate) init: Initializer, pub(crate) bit_offset: Option, pub(crate) bit_width: Option, @@ -151,6 +159,26 @@ impl RawFieldInit { } } +/// How a byte range sits inside an object, as C17 6.7.9p19 needs to know it: +/// an initializer for a subobject overrides the previous initializer for +/// *that* subobject, and whether some other initializer survives depends on +/// what lies between the two. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum SubobjectPlace { + /// The range is a subobject reached through struct members and array + /// elements only (possibly the whole object). Initializing it leaves + /// every other subobject of the enclosing object untouched. + Member, + /// The range is reached only by descending into this union, whose bytes + /// span `offset..offset + size` of the enclosing object. A union holds one + /// member at a time, so initializing through it discards whatever the + /// union held before. + ThroughUnion { offset: usize, size: usize }, + /// The range is not a subobject at all: it straddles two members, or it is + /// a bit-field carrier's window rather than a named object. + NotASubobject, +} + /// Result from member_index_for_designator indicating where positional /// initialization should continue after a designated field. pub(crate) enum MemberDesignatorResult { diff --git a/cc/ir/linearize_init.rs b/cc/ir/linearize_init.rs index 0128c57fa..7fdab1be8 100644 --- a/cc/ir/linearize_init.rs +++ b/cc/ir/linearize_init.rs @@ -1095,6 +1095,349 @@ impl<'a> super::linearize::Linearizer<'a> { visits } + /// Where the byte range `offset..offset + size` sits inside an object of + /// type `typ`, with `offset` measured from that object's first byte. + /// + /// Two initializers in one list can describe overlapping storage, and + /// C17 6.7.9p19 resolves that by *subobject*, not by bytes: given + /// `{ .t = {1, 2}, .t.y = 9 }` the second names a member of the first and + /// replaces only it, while given `{ .u.i = 1, .u.s.b = 9 }` the second + /// names a different member of a union and so replaces the whole of what + /// the first wrote. The spans are identical in shape -- one inside the + /// other -- so only the type can tell the two cases apart. + pub(crate) fn classify_subobject( + &self, + typ: TypeId, + offset: usize, + size: usize, + ) -> SubobjectPlace { + let mut typ = self.resolve_struct_type(typ); + let mut base = 0usize; + + loop { + let type_size = self.types.size_bytes(typ); + if offset == base && size == type_size { + return SubobjectPlace::Member; + } + if size == 0 || offset < base || offset + size > base + type_size { + return SubobjectPlace::NotASubobject; + } + + match self.types.kind(typ) { + // Reached a union without having named it exactly, so the + // range lies inside one of its members. Which member is not + // decidable from the span -- every member starts at the same + // byte -- and it does not matter: whichever it is, giving it + // an initializer discards the member the union held before. + TypeKind::Union => { + return SubobjectPlace::ThroughUnion { + offset: base, + size: type_size, + }; + } + TypeKind::Struct => { + let Some(composite) = self.types.get(typ).composite.as_ref() else { + return SubobjectPlace::NotASubobject; + }; + // A bit-field is not addressable storage of its own, so a + // range inside its carrier is not a subobject. + let found = composite.members.iter().find(|member| { + member.bit_width.is_none() && { + let member_start = base + member.offset; + let member_end = member_start + self.types.size_bytes(member.typ); + offset >= member_start && offset + size <= member_end + } + }); + let Some(member) = found else { + return SubobjectPlace::NotASubobject; + }; + base += member.offset; + typ = self.resolve_struct_type(member.typ); + } + TypeKind::Array => { + let Some(elem_type) = self.types.base_type(typ) else { + return SubobjectPlace::NotASubobject; + }; + let elem_size = self.types.size_bytes(elem_type); + if elem_size == 0 { + return SubobjectPlace::NotASubobject; + } + let elem_start = base + ((offset - base) / elem_size) * elem_size; + if offset + size > elem_start + elem_size { + return SubobjectPlace::NotASubobject; + } + base = elem_start; + typ = self.resolve_struct_type(elem_type); + } + // A scalar with something strictly inside it: only a union or + // a bit-field carrier can produce that, and neither is a + // subobject relation. + _ => return SubobjectPlace::NotASubobject, + } + } + } + + /// An initializer that writes nothing, shaped for `typ` so that + /// [`Self::overlay_subobject`] can place entries into it. + fn empty_aggregate_init(&self, typ: TypeId) -> Option { + match self.types.kind(typ) { + TypeKind::Struct | TypeKind::Union => Some(Initializer::Struct { + total_size: self.types.size_bytes(typ), + fields: Vec::new(), + }), + TypeKind::Array => { + let elem_type = self.types.base_type(typ)?; + Some(Initializer::Array { + elem_size: self.types.size_bytes(elem_type), + total_size: self.types.size_bytes(typ), + elements: Vec::new(), + }) + } + _ => None, + } + } + + /// Fold `new_init` into `init`, the initializer for an object of type + /// `typ`, so that it initializes the subobject at `offset..offset + size` + /// and leaves every other subobject as it was. + /// + /// The caller has already established with [`Self::classify_subobject`] + /// that the range *is* such a subobject. Returns false when the existing + /// initializer's shape cannot express the replacement -- a string literal + /// standing for a character array, say -- in which case the caller falls + /// back to discarding the earlier initializer whole. + pub(crate) fn overlay_subobject( + &self, + typ: TypeId, + init: &mut Initializer, + offset: usize, + size: usize, + new_init: &Initializer, + ) -> bool { + let typ = self.resolve_struct_type(typ); + if offset == 0 && size == self.types.size_bytes(typ) { + *init = new_init.clone(); + return true; + } + + match self.types.kind(typ) { + TypeKind::Struct | TypeKind::Union => { + let Some(composite) = self.types.get(typ).composite.as_ref() else { + return false; + }; + let found = composite + .members + .iter() + .find(|member| { + member.bit_width.is_none() && { + let member_end = member.offset + self.types.size_bytes(member.typ); + offset >= member.offset && offset + size <= member_end + } + }) + .map(|member| (member.offset, member.typ)); + let Some((member_offset, member_type)) = found else { + return false; + }; + let member_size = self.types.size_bytes(member_type); + let Initializer::Struct { fields, .. } = init else { + return false; + }; + self.overlay_into_entries( + fields, + member_offset, + member_size, + member_type, + offset, + size, + new_init, + ) + } + TypeKind::Array => { + let Some(elem_type) = self.types.base_type(typ) else { + return false; + }; + let elem_size = self.types.size_bytes(elem_type); + if elem_size == 0 { + return false; + } + let elem_offset = (offset / elem_size) * elem_size; + if offset + size > elem_offset + elem_size { + return false; + } + let Initializer::Array { elements, .. } = init else { + return false; + }; + // An array's entries carry no width, so borrow the struct + // path's bookkeeping by giving each one the element width it + // implicitly has. + let mut entries: Vec<(usize, usize, Initializer)> = elements + .iter() + .map(|(off, init)| (*off, elem_size, init.clone())) + .collect(); + if !self.overlay_into_entries( + &mut entries, + elem_offset, + elem_size, + elem_type, + offset, + size, + new_init, + ) { + return false; + } + *elements = entries + .into_iter() + .map(|(off, _, init)| (off, init)) + .collect(); + true + } + _ => false, + } + } + + /// Place `new_init` for the subobject at `offset..offset + size`, which + /// lies within the member or element at `slot_offset` of `slot_size` + /// bytes, into an entry list that holds one entry per initialized member. + #[allow(clippy::too_many_arguments)] + fn overlay_into_entries( + &self, + entries: &mut Vec<(usize, usize, Initializer)>, + slot_offset: usize, + slot_size: usize, + slot_type: TypeId, + offset: usize, + size: usize, + new_init: &Initializer, + ) -> bool { + let slot_end = slot_offset + slot_size; + let existing = entries + .iter() + .position(|(off, sz, _)| *off < slot_end && slot_offset < *off + *sz); + + if let Some(idx) = existing { + let (entry_offset, entry_size, entry_init) = &mut entries[idx]; + // Anything but a whole entry for exactly this member -- a + // bit-field carrier byte, or an entry spanning several members -- + // is not something this can descend into. + if (*entry_offset, *entry_size) != (slot_offset, slot_size) { + return false; + } + return self.overlay_subobject( + slot_type, + entry_init, + offset - slot_offset, + size, + new_init, + ); + } + + // The member had no initializer of its own: give it one that writes + // zeros everywhere but the subobject being replaced. + let fresh = if (offset, size) == (slot_offset, slot_size) { + new_init.clone() + } else { + let Some(mut fresh) = self.empty_aggregate_init(slot_type) else { + return false; + }; + if !self.overlay_subobject(slot_type, &mut fresh, offset - slot_offset, size, new_init) + { + return false; + } + fresh + }; + entries.push((slot_offset, slot_size, fresh)); + entries.sort_by_key(|(off, _, _)| *off); + true + } + + /// Apply C17 6.7.9p19 to the initializers one struct or union + /// initializer list produced, in the order the list wrote them. + /// + /// An initializer for a subobject overrides any previously listed + /// initializer *for that subobject*, and leaves initializers for other + /// subobjects alone. So a later entry is folded into an earlier one it is + /// a member of, replaces an earlier one it contains, and -- when the two + /// are related only through a union or a bit-field carrier, where no + /// structural fold exists -- discards it. + /// + /// "Later" means later in the list. The entries arrive in that order and + /// the sort into address order, which the emitter needs, runs afterwards. + pub(crate) fn merge_raw_field_inits(&self, raw: Vec) -> Vec { + let mut merged: Vec = Vec::with_capacity(raw.len()); + + for later in raw { + let later_span = later.byte_span(); + let mut folded = false; + let mut idx = 0; + + while idx < merged.len() { + let earlier = &merged[idx]; + let earlier_span = earlier.byte_span(); + if earlier_span.start >= later_span.end || later_span.start >= earlier_span.end { + idx += 1; + continue; + } + // Two *distinct* bit-fields are different objects even when + // they share a carrier byte, so both survive. + if earlier.bit_width.is_some() + && later.bit_width.is_some() + && (earlier.offset, earlier.bit_offset) != (later.offset, later.bit_offset) + { + idx += 1; + continue; + } + if later_span.start <= earlier_span.start && earlier_span.end <= later_span.end { + merged.remove(idx); + continue; + } + let foldable = !folded + && earlier.bit_width.is_none() + && later.bit_width.is_none() + && earlier_span.start <= later_span.start + && later_span.end <= earlier_span.end + && self.fold_field_init(&mut merged[idx], &later); + if foldable { + folded = true; + idx += 1; + continue; + } + merged.remove(idx); + } + + if !folded { + merged.push(later); + } + } + + merged + } + + /// Fold `later`, whose bytes lie inside `earlier`'s, into `earlier`. + /// Returns false when no structural fold exists, which leaves the caller + /// to discard `earlier`. + fn fold_field_init(&self, earlier: &mut RawFieldInit, later: &RawFieldInit) -> bool { + let inner_offset = later.offset - earlier.offset; + match self.classify_subobject(earlier.typ, inner_offset, later.field_size) { + SubobjectPlace::Member => self.overlay_subobject( + earlier.typ, + &mut earlier.init, + inner_offset, + later.field_size, + &later.init, + ), + // The union stops holding what it held: everything it contained + // goes, and it comes to hold just this one initializer. + SubobjectPlace::ThroughUnion { offset, size } => { + let replacement = Initializer::Struct { + total_size: size, + fields: vec![(inner_offset - offset, later.field_size, later.init.clone())], + }; + self.overlay_subobject(earlier.typ, &mut earlier.init, offset, size, &replacement) + } + SubobjectPlace::NotASubobject => false, + } + } + /// Convert an AST initializer list to an IR Initializer pub(crate) fn ast_init_list_to_ir( &mut self, @@ -1204,40 +1547,26 @@ impl<'a> super::linearize::Linearizer<'a> { raw_fields.push(RawFieldInit { offset: visit.offset, field_size: visit.field_size, + typ: visit.typ, init: field_init, bit_offset: visit.bit_offset, bit_width: visit.bit_width, }); } + // Initializing the same object twice: the later one wins + // (C17 6.7.9p19), and one that names a *subobject* of an + // earlier one replaces only that subobject. Resolved in + // the order the list wrote them, before the sort below + // reorders them by address. + let mut raw_fields = self.merge_raw_field_inits(raw_fields); + // Sort by the bit each field starts at, so that designated // initializers emit in address order however they were // written -- the emitter fills the gaps between fields and // so requires them sorted and non-overlapping. raw_fields.sort_by_key(|f| f.offset * 8 + f.bit_offset.unwrap_or(0) as usize); - // Initializing the same object twice: the later one wins - // (C17 6.7.9p19). Two *distinct* bitfields are different - // objects even when they share a byte, so both survive. - let mut idx = 0; - while idx + 1 < raw_fields.len() { - let (a, b) = (&raw_fields[idx], &raw_fields[idx + 1]); - let distinct_bitfields = a.bit_width.is_some() - && b.bit_width.is_some() - && (a.offset, a.bit_offset) != (b.offset, b.bit_offset); - let a_span = a.byte_span(); - let b_span = b.byte_span(); - - if !distinct_bitfields - && a_span.start < b_span.end - && b_span.start < a_span.end - { - raw_fields.remove(idx); - } else { - idx += 1; - } - } - // Merge bitfields byte by byte rather than one storage unit // at a time. A unit is `sizeof(T)` wide and aligned, so it // routinely spans bytes that belong to other members -- @@ -1931,4 +2260,455 @@ mod tests { vec![(0, Initializer::Int(1)), (4, Initializer::Int(2))], ); } + + // Resolving two initializers that describe overlapping storage + // (C17 6.7.9p19). + + /// Types for the override tests: + /// + /// ```c + /// struct T { int x, y; }; + /// struct S { struct T t; int z; }; + /// struct A { int a[3]; int z; }; + /// struct W { union { int i; struct { char a, b, c, d; } s; } u; }; + /// ``` + struct OverrideTypes { + target: Target, + types: TypeTable, + strings: crate::strings::StringTable, + symbols: SymbolTable, + t: TypeId, + s: TypeId, + a: TypeId, + w: TypeId, + v: TypeId, + int_array: TypeId, + } + + fn member(name: StringId, typ: TypeId, offset: usize) -> crate::types::StructMember { + crate::types::StructMember { + name, + typ, + offset, + bit_offset: None, + bit_width: None, + access_bytes: None, + explicit_align: None, + } + } + + fn composite( + members: Vec, + size: usize, + align: usize, + ) -> crate::types::CompositeType { + crate::types::CompositeType { + members, + size, + align, + member_align: align, + is_complete: true, + ..crate::types::CompositeType::incomplete(None) + } + } + + impl OverrideTypes { + fn new() -> Self { + let target = Target::host(); + let mut types = TypeTable::new(&target); + let mut strings = crate::strings::StringTable::new(); + let name = |strings: &mut crate::strings::StringTable, s: &str| strings.intern(s); + + let int = types.int_id; + let ch = types.char_id; + + let (x, y) = (name(&mut strings, "x"), name(&mut strings, "y")); + let t = types.intern(Type::struct_type(composite( + vec![member(x, int, 0), member(y, int, 4)], + 8, + 4, + ))); + + let (t_name, z) = (name(&mut strings, "t"), name(&mut strings, "z")); + let s = types.intern(Type::struct_type(composite( + vec![member(t_name, t, 0), member(z, int, 8)], + 12, + 4, + ))); + + let int_array = types.intern(Type::array(int, 3)); + let a_name = name(&mut strings, "a"); + let a = types.intern(Type::struct_type(composite( + vec![member(a_name, int_array, 0), member(z, int, 12)], + 16, + 4, + ))); + + let (b, c, d) = ( + name(&mut strings, "b"), + name(&mut strings, "c"), + name(&mut strings, "d"), + ); + let chars = types.intern(Type::struct_type(composite( + vec![ + member(a_name, ch, 0), + member(b, ch, 1), + member(c, ch, 2), + member(d, ch, 3), + ], + 4, + 1, + ))); + let (i, s_name) = (name(&mut strings, "i"), name(&mut strings, "s")); + let union_u = types.intern(Type::union_type(composite( + vec![member(i, int, 0), member(s_name, chars, 0)], + 4, + 4, + ))); + let u_name = name(&mut strings, "u"); + let w = types.intern(Type::struct_type(composite( + vec![member(u_name, union_u, 0)], + 4, + 4, + ))); + + // `struct V { int k; union { int i; struct { char a, b, c, d; } s; } u; }` + // -- the union has a sibling, so resetting it must leave `k` be. + let k = name(&mut strings, "k"); + let v = types.intern(Type::struct_type(composite( + vec![member(k, int, 0), member(u_name, union_u, 4)], + 8, + 4, + ))); + + Self { + target, + types, + strings, + symbols: SymbolTable::new(), + t, + s, + a, + w, + v, + int_array, + } + } + + fn linearizer(&self) -> Linearizer<'_> { + Linearizer::new(&self.symbols, &self.types, &self.strings, &self.target) + } + } + + /// One entry of a struct initializer list, as `walk_struct_init_fields` + /// hands it over: the subobject it names and the value for it. + fn raw(offset: usize, typ: TypeId, size: usize, init: Initializer) -> RawFieldInit { + RawFieldInit { + offset, + field_size: size, + typ, + init, + bit_offset: None, + bit_width: None, + } + } + + fn struct_init(total_size: usize, fields: &[(usize, usize, i128)]) -> Initializer { + Initializer::Struct { + total_size, + fields: fields + .iter() + .map(|(off, size, value)| (*off, *size, Initializer::Int(*value))) + .collect(), + } + } + + /// `(offset, size, initializer)` for each entry the merge kept, in the + /// order it kept them. + fn kept(merged: &[RawFieldInit]) -> Vec<(usize, usize, Initializer)> { + merged + .iter() + .map(|f| (f.offset, f.field_size, f.init.clone())) + .collect() + } + + #[test] + fn a_range_is_a_member_when_struct_members_and_array_elements_reach_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // The whole object, the member `t`, and `t.y` inside it. + assert_eq!( + lin.classify_subobject(fixture.s, 0, 12), + SubobjectPlace::Member + ); + assert_eq!( + lin.classify_subobject(fixture.s, 0, 8), + SubobjectPlace::Member + ); + assert_eq!( + lin.classify_subobject(fixture.s, 4, 4), + SubobjectPlace::Member + ); + // `a[1]` of `struct A`. + assert_eq!( + lin.classify_subobject(fixture.a, 4, 4), + SubobjectPlace::Member + ); + // The four bytes straddling `t.y` and `z` are no object at all. + assert_eq!( + lin.classify_subobject(fixture.s, 6, 4), + SubobjectPlace::NotASubobject + ); + } + + #[test] + fn a_range_inside_a_union_member_is_reached_through_the_union() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // `u.s.b` -- one byte, reached only by choosing a union member. + assert_eq!( + lin.classify_subobject(fixture.w, 1, 1), + SubobjectPlace::ThroughUnion { offset: 0, size: 4 } + ); + // The union named exactly is an ordinary member of `struct W`. + assert_eq!( + lin.classify_subobject(fixture.w, 0, 4), + SubobjectPlace::Member + ); + } + + /// `struct S s = { .t = {1, 2}, .t.y = 9, .z = 7 };` -- the override + /// names `t.y`, so `t.x` keeps the 1 it was given. + #[test] + fn a_contained_override_replaces_only_the_subobject_it_names() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + raw(4, int, 4, Initializer::Int(9)), + raw(8, int, 4, Initializer::Int(7)), + ]); + + assert_eq!( + kept(&merged), + vec![ + (0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)])), + (8, 4, Initializer::Int(7)), + ] + ); + } + + /// The same one level further down: `{ .a = {1,2,3}, .a[1] = 9 }` keeps + /// elements 0 and 2. + #[test] + fn a_contained_override_replaces_only_the_array_element_it_names() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let array = Initializer::Array { + elem_size: 4, + total_size: 12, + elements: vec![ + (0, Initializer::Int(1)), + (4, Initializer::Int(2)), + (8, Initializer::Int(3)), + ], + }; + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.int_array, 12, array), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![( + 0, + 12, + Initializer::Array { + elem_size: 4, + total_size: 12, + elements: vec![ + (0, Initializer::Int(1)), + (4, Initializer::Int(9)), + (8, Initializer::Int(3)), + ], + } + )] + ); + } + + /// An initializer for a whole subobject replaces every earlier one for a + /// part of it. + #[test] + fn a_containing_override_replaces_the_earlier_entry_whole() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(4, int, 4, Initializer::Int(9)), + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + ]); + + assert_eq!( + kept(&merged), + vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))] + ); + } + + /// Initializers for different members stand side by side. + #[test] + fn disjoint_entries_are_all_kept() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, int, 4, Initializer::Int(1)), + raw(4, int, 4, Initializer::Int(2)), + raw(8, int, 4, Initializer::Int(7)), + ]); + + assert_eq!( + kept(&merged), + vec![ + (0, 4, Initializer::Int(1)), + (4, 4, Initializer::Int(2)), + (8, 4, Initializer::Int(7)), + ] + ); + } + + /// `{ .u.i = 0x01020304, .u.s.b = 9 }` -- a union holds one member at a + /// time, so naming a second discards what the first wrote rather than + /// overlaying it. + #[test] + fn initializing_a_second_union_member_discards_the_first() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let (int, ch) = (fixture.types.int_id, fixture.types.char_id); + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, int, 4, Initializer::Int(0x01020304)), + raw(1, ch, 1, Initializer::Int(9)), + ]); + + assert_eq!(kept(&merged), vec![(1, 1, Initializer::Int(9))]); + } + + /// An initializer for a whole struct, then one byte of a different member + /// of a union inside it: only that union is reset, and the struct's other + /// members keep what they were given. + /// + /// `struct V v = { .k = 5, .u.i = 0x01020304 }` followed by `.u.s.b = 9`. + #[test] + fn an_override_through_a_nested_union_resets_only_that_union() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let ch = fixture.types.char_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw( + 0, + fixture.v, + 8, + struct_init(8, &[(0, 4, 5), (4, 4, 0x01020304)]), + ), + raw(5, ch, 1, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![( + 0, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![ + (0, 4, Initializer::Int(5)), + (4, 4, struct_init(4, &[(1, 1, 9)])), + ], + } + )] + ); + } + + /// A struct whose only member is a union is the union, byte for byte, so + /// resetting the union replaces the whole entry. + #[test] + fn an_override_through_a_union_filling_its_struct_replaces_the_entry() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let ch = fixture.types.char_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.w, 4, struct_init(4, &[(0, 4, 0x01020304)])), + raw(1, ch, 1, Initializer::Int(9)), + ]); + + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(1, 1, 9)]))]); + } + + /// "Later wins" is later in the list, not at a higher address: + /// `{ .z = 7, .t.y = 9, .t = {1, 2} }` ends with `t` holding `{1, 2}`. + #[test] + fn the_override_rule_is_applied_in_source_order() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(8, int, 4, Initializer::Int(7)), + raw(4, int, 4, Initializer::Int(9)), + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + ]); + + assert_eq!( + kept(&merged), + vec![ + (8, 4, Initializer::Int(7)), + (0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + ] + ); + + // And the other order, where the narrower one is written last. + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)])), + raw(8, int, 4, Initializer::Int(7)), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![ + (0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)])), + (8, 4, Initializer::Int(7)), + ] + ); + } + + /// A member the earlier initializer left out gains an entry of its own + /// rather than costing the whole earlier initializer: `{ .t = {1}, .t.y + /// = 9 }` keeps `t.x`. + #[test] + fn an_override_of_an_uninitialized_member_is_added_to_the_earlier_entry() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.t, 8, struct_init(8, &[(0, 4, 1)])), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!( + kept(&merged), + vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)]))] + ); + } } diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index f49be772d..11645ae63 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -89,6 +89,39 @@ struct OpenFor { exit_bb: BasicBlockId, } +/// One initializer from a struct or union initializer list that has already +/// been stored into the object being initialized. +/// +/// The automatic path emits a store per list entry and lets a later store land +/// on an earlier one, which is all C17 6.7.9p19 needs *while* the later store +/// covers every byte it supersedes. It does not when the later initializer +/// fills only part of the subobject it names, and it does not when the two +/// entries name different members of a union -- a union holds one member at a +/// time, so the member left behind does not show through the new one. Both +/// need the superseded bytes cleared, and deciding which bytes those are is +/// what this records. +struct WrittenInit { + /// The bytes it wrote that no later entry has since cleared. + live: std::ops::Range, + /// The first byte of the subobject it initialized, and that subobject's + /// type. Together they say whether a later entry names a member of this + /// one -- in which case the rest of this one survives -- or reaches its + /// bytes only by passing through a union. + origin: usize, + typ: TypeId, +} + +/// Grow `reset` to also cover `range`. +/// +/// Every range joined here overlaps the range of the entry being stored, so +/// the union of them all is contiguous and one fill covers it. +fn widen_reset(reset: &mut Option>, range: std::ops::Range) { + *reset = Some(match reset.take() { + Some(cur) => cur.start.min(range.start)..cur.end.max(range.end), + None => range, + }); +} + impl<'a> super::linearize::Linearizer<'a> { pub(crate) fn linearize_stmt(&mut self, stmt: &Stmt) { match stmt { @@ -1084,7 +1117,22 @@ impl<'a> super::linearize::Linearizer<'a> { let visits = self.walk_struct_init_fields(resolved_typ, &members, is_union, elements); + // C17 6.7.9p19 resolves two initializers for overlapping + // storage by subobject. Storing them in list order is + // enough only where the later store covers every byte it + // supersedes; where it does not, the bytes it leaves have + // to be cleared first. See [`WrittenInit`]. + let mut written: Vec = Vec::new(); + for visit in visits { + if let Some(reset) = self.init_override_reset(&mut written, &visit) { + self.emit_block_zero( + base_sym, + base_offset + reset.start as i64, + (reset.end - reset.start) as i64, + ); + } + let offset = base_offset + visit.offset as i64; let field_type = visit.typ; @@ -1154,6 +1202,92 @@ impl<'a> super::linearize::Linearizer<'a> { } } + /// Record that `visit` is about to be stored, and answer which bytes of + /// the object must be cleared first for C17 6.7.9p19 to hold. + /// + /// Nothing at all, for the usual case where the entry overlaps none of + /// those already stored. Otherwise the entry's own bytes -- so that the + /// part of the subobject it does not fill reads as zero rather than as + /// the initializer it replaces -- together with the bytes of any earlier + /// entry it invalidates: all of one it wholly contains, all of one it is + /// not a subobject of, and, where it reaches an earlier entry's bytes only + /// by naming a member of a union inside it, that union's bytes. + fn init_override_reset( + &self, + written: &mut Vec, + visit: &StructFieldVisit, + ) -> Option> { + // A bit-field is stored by reading its carrier and writing it back, so + // its window is not storage it owns and clearing the window would + // blank the members sharing it. Bit-fields are left to overlay each + // other as they always have. + if visit.bit_width.is_some() || visit.field_size == 0 { + return None; + } + + let span = visit.offset..visit.offset + visit.field_size; + let mut reset: Option> = None; + + for entry in written.iter() { + if entry.live.start >= span.end || span.start >= entry.live.end { + continue; + } + widen_reset(&mut reset, span.clone()); + + // Wholly superseded: the entry's own bytes are all inside the + // ones being cleared and rewritten. + if span.start <= entry.live.start && entry.live.end <= span.end { + continue; + } + + let entry_end = entry.origin + self.types.size_bytes(entry.typ); + let place = if span.start >= entry.origin && span.end <= entry_end { + self.classify_subobject(entry.typ, span.start - entry.origin, visit.field_size) + } else { + SubobjectPlace::NotASubobject + }; + match place { + // A member of the earlier entry's object: the rest of that + // object is a different subobject and stands. + SubobjectPlace::Member => {} + SubobjectPlace::ThroughUnion { offset, size } => { + widen_reset( + &mut reset, + entry.origin + offset..entry.origin + offset + size, + ); + } + SubobjectPlace::NotASubobject => { + widen_reset(&mut reset, entry.live.clone()); + } + } + } + + if let Some((from, to)) = reset.as_ref().map(|range| (range.start, range.end)) { + // Whatever the fill covers is gone; the bytes an entry keeps on + // either side of it are still its own. + *written = written + .drain(..) + .flat_map(|entry| { + let (origin, typ) = (entry.origin, entry.typ); + [ + entry.live.start..entry.live.end.min(from), + entry.live.start.max(to)..entry.live.end, + ] + .into_iter() + .filter(|live| live.start < live.end) + .map(move |live| WrittenInit { live, origin, typ }) + }) + .collect(); + } + + written.push(WrittenInit { + live: span, + origin: visit.offset, + typ: visit.typ, + }); + reset + } + /// Store a complex value into `base_sym` at `offset`, as two halves. /// /// A complex value lives in memory and travels by *address*, so storing it diff --git a/cc/tests/c99/initializers.rs b/cc/tests/c99/initializers.rs index 84bedfd21..aafdfc376 100644 --- a/cc/tests/c99/initializers.rs +++ b/cc/tests/c99/initializers.rs @@ -2781,3 +2781,121 @@ int main(void) 0 ); } + +/// A later designated initializer replaces the subobject it names, not every +/// object whose bytes it touches. +/// +/// C17 6.7.9p19: an initializer for a subobject overrides any previously +/// listed initializer *for that subobject*, and initializers for other +/// subobjects are unaffected. The static path merged its field initializers by +/// byte span and dropped an earlier entry whole on any intersection, so +/// `.t = {1,2}` followed by `.t.y = 9` lost the `1` as well as the `2` -- while +/// the automatic path, which just stores in order and lets the later store land +/// on the earlier one, kept it. The two disagreed on the same initializer. +/// +/// Every case here is checked in both storage durations, against the values +/// gcc and clang produce. +#[test] +fn c99_a_designated_override_replaces_only_the_subobject_it_names() { + let code = r#" +struct T { int x, y; }; +struct S { struct T t; int z; }; +struct A { int a[3]; int z; }; + +struct S g1 = { .t = {1, 2}, .t.y = 9, .z = 7 }; +struct A g2 = { .a = {1, 2, 3}, .a[1] = 9, .z = 7 }; + +int main(void) +{ + struct S l1 = { .t = {1, 2}, .t.y = 9, .z = 7 }; + struct A l2 = { .a = {1, 2, 3}, .a[1] = 9, .z = 7 }; + + /* The override names .t.y, so .t.x keeps the 1 it was given. */ + if (g1.t.x != 1 || g1.t.y != 9 || g1.z != 7) return 1; + if (l1.t.x != 1 || l1.t.y != 9 || l1.z != 7) return 2; + + /* The same one level down: only element 1 is replaced. */ + if (g2.a[0] != 1 || g2.a[1] != 9 || g2.a[2] != 3 || g2.z != 7) return 3; + if (l2.a[0] != 1 || l2.a[1] != 9 || l2.a[2] != 3 || l2.z != 7) return 4; + + return 0; +} +"#; + assert_eq!(compile_and_run("designated_partial_override", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("designated_partial_override_opt", code), + 0 + ); +} + +/// "Later wins" means later in the initializer list, not later in the object. +/// +/// The static path sorted its field initializers by address before resolving +/// overlaps, so the rule was applied in the wrong order entirely: in +/// `{ .z = 7, .t.y = 9, .t = {1,2} }` the `.t = {1,2}` is written last and must +/// win, but after sorting it sat before `.t.y` and was the entry dropped. +#[test] +fn c99_a_designated_override_is_resolved_in_source_order() { + let code = r#" +struct T { int x, y; }; +struct S { struct T t; int z; }; + +/* The whole-field initializer comes last and wins, even though it names a + lower address than the override before it. */ +struct S g = { .z = 7, .t.y = 9, .t = {1, 2} }; + +/* And the other order, where the narrower one wins. */ +struct S h = { .t = {1, 2}, .z = 7, .t.y = 9 }; + +int main(void) +{ + struct S lg = { .z = 7, .t.y = 9, .t = {1, 2} }; + struct S lh = { .t = {1, 2}, .z = 7, .t.y = 9 }; + + if (g.t.x != 1 || g.t.y != 2 || g.z != 7) return 1; + if (lg.t.x != 1 || lg.t.y != 2 || lg.z != 7) return 2; + if (h.t.x != 1 || h.t.y != 9 || h.z != 7) return 3; + if (lh.t.x != 1 || lh.t.y != 9 || lh.z != 7) return 4; + + return 0; +} +"#; + assert_eq!(compile_and_run("designated_source_order", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("designated_source_order_opt", code), + 0 + ); +} + +/// Initializing a second member of a union resets it; it does not overlay the +/// first. +/// +/// This is the case where the two paths disagree the other way round. A union +/// holds one member at a time, so `{ .u.i = 0x01020304, .u.s.b = 9 }` leaves +/// the union holding `.u.s` with only `b` given a value and the rest zero -- +/// which is what the static path produced and what gcc and clang produce. The +/// automatic path stored the `int` and then stored one byte over it, keeping +/// the other three, so it read back `0x01020904`. +/// +/// It is here as a guard on the fix above: making the static path store in +/// source order the way the automatic path does would adopt this bug, so the +/// merge has to keep the union case distinct from the struct and array cases. +#[test] +fn c99_initializing_a_second_union_member_resets_the_union() { + let code = r#" +struct U { union { int i; struct { char a, b, c, d; } s; } u; }; +struct U g = { .u.i = 0x01020304, .u.s.b = 9 }; + +int main(void) +{ + struct U l = { .u.i = 0x01020304, .u.s.b = 9 }; + if (g.u.i != 0x900) return 1; + if (l.u.i != 0x900) return 2; + if (g.u.s.a != 0 || g.u.s.b != 9 || g.u.s.c != 0 || g.u.s.d != 0) return 3; + if (l.u.s.a != 0 || l.u.s.b != 9 || l.u.s.c != 0 || l.u.s.d != 0) return 4; + return 0; +} +"#; + assert_eq!(compile_and_run("union_member_reset", code, &[]), 0); + assert_eq!(compile_and_run_optimized("union_member_reset_opt", code), 0); +} From 2eb9cbc3428a4060770c63f8a8ebcc494c1cb5a5 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 05:28:53 -0400 Subject: [PATCH 11/17] cc: compute an atomic compound assignment the way the ordinary one is C17 6.5.16.2p3 makes `E1 op= E2` mean `E1 = E1 op E2` bar evaluating `E1` once, so the arithmetic happens at the type the usual arithmetic conversions give the operands and only the result converts back. `emit_assign` already did that, with a comment recording the bug it came from; the atomic path had its own copy of the logic, narrowed the right operand to the target first and computed there. Its comment -- "the ordinary path does this after the point we branched from, so it has to be repeated" -- is the duplication saying so. `_Atomic unsigned char c = 50; c /= -5` stored 0 where the same ordinary object stored 246, and `m %= -3` stored 50 against 2. The expression's value had the second half of it: recomputed as a raw binop from the old value with no conversion back, so `_Atomic _Bool b = 0; r = (b -= 1)` stored 1 and handed back 255. An assignment expression has the value of the left operand after the assignment (6.5.16p3), and the ordinary path gives 1. Both are one extraction. `CompoundAssign` describes the assignment and `compound_assign_value` performs it -- choose the arithmetic type, pick the opcode, convert the operands, compute, convert back -- and `emit_assign`, the CAS loop, `try_emit_atomic_assign`, the increment forms and the GNU `__atomic_*_fetch` builtins all go through it. The type and opcode decisions are pure functions over the type table, which is what lets them be tested without a linearizer, and the atomic path's separate opcode table is gone. A descriptor rather than loose arguments because NAND needs one more bit: complementing has to happen before the conversion back, or `__atomic_fetch_nand` on an `_Atomic _Bool` stores 255. With that, `emit_atomic_nand` is no longer a second entry point -- NAND is `&=` with `invert` -- and the `_Bool` fixups the CAS loop and the increment path each carried are subsumed by the shared conversion. `emit_atomic_rmw`'s `needs_conversion` gate becomes `native_rmw_opcode`, which states the condition once: a native fetch-and-op equals the standard only when the operator is congruent modulo 2^n and the conversion back is the truncation congruence permits. `_Bool`, NAND, divide, remainder, the shifts and the floating forms all fail that and take the CAS loop, which they already did. Add, subtract and the bitwise operators keep their native lowering. The shifts are the trap in sharing this: 6.5.7p3 promotes each operand and the result has the promoted *left* operand's type, so the arithmetic type is `integer_promote(target)` and not the common type, and the right operand is promoted on its own. `_Atomic signed char s = -8; s >>= 1` is -4. Both rules came over from the ordinary path unchanged, and a test pins them. Two more defects fall out. `__atomic_add_fetch` on an `_Atomic _Bool` returned the raw arithmetic where it should return what it stored, and `__atomic_fetch_add` on a floating object selected the integer atomic add -- an integer add of float bits -- where it now selects the floating opcode and takes the CAS loop. Neither was covered by a test. clang answers 255 for the `_Bool` expression and 124 for the shift. c17 follows the standard and its own non-atomic path, which is the same answer. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize.rs | 70 ++++---- cc/ir/linearize_atomic.rs | 226 +++++++++++--------------- cc/ir/linearize_emit.rs | 326 ++++++++++++++++++++++++++------------ cc/ir/test_linearize.rs | 255 +++++++++++++++++++++++++++++ cc/tests/c11/atomics.rs | 138 ++++++++++++++++ 5 files changed, 737 insertions(+), 278 deletions(-) diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index b8005114e..211ae2a83 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -19,10 +19,11 @@ use crate::abi::{get_abi_for_conv, CallingConv}; use crate::diag::{get_all_stream_names, Position}; use crate::float::FloatVal; use crate::ir::linearize_atomic::AtomicLvalue; +use crate::ir::linearize_emit::CompoundAssign; use crate::parse::ast::{ - BinaryOp, BlockItem, Expr, ExprKind, ExternalDecl, FpCompare, FpTest, FunctionDef, GnuAtomicOp, - InitElement, InlineLibraryFn, MemoryFn, NarrowedLibraryCall, OffsetOfPath, ParamStyle, - TranslationUnit, UnaryOp, + AssignOp, BinaryOp, BlockItem, Expr, ExprKind, ExternalDecl, FpCompare, FpTest, FunctionDef, + GnuAtomicOp, InitElement, InlineLibraryFn, MemoryFn, NarrowedLibraryCall, OffsetOfPath, + ParamStyle, TranslationUnit, UnaryOp, }; use crate::strings::{StringId, StringTable}; use crate::symbol::{SymbolId, SymbolTable}; @@ -5724,13 +5725,19 @@ impl<'a> Linearizer<'a> { let value_typ = self.expr_type(val); let raw = self.linearize_expr(val); // Pointer arithmetic scales by the element size, as it does for `+=`. - let operand = if self.types.kind(elem_typ) == TypeKind::Pointer + let is_ptr_arith = self.types.kind(elem_typ) == TypeKind::Pointer && self.types.is_integer(value_typ) - && matches!(op, GnuAtomicOp::Add | GnuAtomicOp::Sub) - { - self.scale_pointer_addend(elem_typ, value_typ, raw) + && matches!(op, GnuAtomicOp::Add | GnuAtomicOp::Sub); + let (operand, operand_typ) = if is_ptr_arith { + ( + self.scale_pointer_addend(elem_typ, value_typ, raw), + self.types.long_id, + ) } else { - self.emit_convert(raw, value_typ, elem_typ) + // Unlike an operator, a builtin converts its value argument to the + // object's type itself (gcc documents the parameter as that type), + // so there is no common type left to compute at. + (self.emit_convert(raw, value_typ, elem_typ), elem_typ) }; // The order argument is accepted and evaluated, as gcc evaluates it, // but every lowering here is sequentially consistent: `emit_atomic_rmw` @@ -5744,39 +5751,30 @@ impl<'a> Linearizer<'a> { size_bits: bits, }; - let (old, binop) = match op { - GnuAtomicOp::Nand => (self.emit_atomic_nand(&lv, operand), Opcode::And), - _ => { - let binop = match op { - GnuAtomicOp::Add => Opcode::Add, - GnuAtomicOp::Sub => Opcode::Sub, - GnuAtomicOp::And => Opcode::And, - GnuAtomicOp::Or => Opcode::Or, - GnuAtomicOp::Xor => Opcode::Xor, - GnuAtomicOp::Nand => unreachable!("handled above"), - }; - (self.emit_atomic_rmw(&lv, binop, operand), binop) - } + // Each builtin is the compound assignment of the same name, so it + // goes through the same model: `nand` is the one that has no operator + // spelling, and it is `&` with the result complemented before it + // converts back to the object's type (`CompoundAssign::invert`). + let assign_op = match op { + GnuAtomicOp::Add => AssignOp::AddAssign, + GnuAtomicOp::Sub => AssignOp::SubAssign, + GnuAtomicOp::And | GnuAtomicOp::Nand => AssignOp::AndAssign, + GnuAtomicOp::Or => AssignOp::OrAssign, + GnuAtomicOp::Xor => AssignOp::XorAssign, + }; + let ca = CompoundAssign { + is_ptr_arith, + invert: op == GnuAtomicOp::Nand, + ..CompoundAssign::new(assign_op, elem_typ, operand_typ) }; + let old = self.emit_atomic_rmw(&lv, &ca, operand); if !returns_new { return old; } - - let new = self.alloc_reg_pseudo(); - self.emit(Instruction::binop(binop, new, old, operand, elem_typ, bits)); - if op != GnuAtomicOp::Nand { - return new; - } - let inverted = self.alloc_reg_pseudo(); - self.emit(Instruction::unop( - Opcode::Not, - inverted, - new, - elem_typ, - bits, - )); - inverted + // The new value, recomputed from the old one rather than read back: + // arithmetic on a value in hand is not a second access to the object. + self.compound_assign_value(&ca, old, operand) } /// `__sync_bool_compare_and_swap` and `__sync_val_compare_and_swap`. diff --git a/cc/ir/linearize_atomic.rs b/cc/ir/linearize_atomic.rs index c1073cade..ee2ec0449 100644 --- a/cc/ir/linearize_atomic.rs +++ b/cc/ir/linearize_atomic.rs @@ -20,6 +20,7 @@ // use super::linearize::Linearizer; +use super::linearize_emit::{compound_assign_arith_type, compound_assign_opcode, CompoundAssign}; use super::{Instruction, MemoryOrder, Opcode, PseudoId}; use crate::diag; use crate::float::FloatVal; @@ -210,8 +211,12 @@ impl Linearizer<'_> { /// /// Multiplication, division, remainder, the shifts and everything /// floating-point have no native atomic form and go through a - /// compare-and-swap retry loop instead. - pub(crate) fn atomic_opcode_for(op: Opcode) -> Option { + /// compare-and-swap retry loop instead. Neither does `nand`, on any + /// target, which is why it is a flag on [`CompoundAssign`] rather than an + /// opcode here. + /// + /// Having one does not by itself make it usable: see `native_rmw_opcode`. + fn atomic_opcode_for(op: Opcode) -> Option { Some(match op { Opcode::Add => Opcode::AtomicFetchAdd, Opcode::Sub => Opcode::AtomicFetchSub, @@ -249,39 +254,53 @@ impl Linearizer<'_> { pub(crate) fn emit_atomic_rmw( &mut self, lv: &AtomicLvalue, - op: Opcode, + ca: &CompoundAssign, value: PseudoId, ) -> PseudoId { - // `_Bool` cannot use a native fetch-and-op: the value stored must be - // the *converted* result, so `b = 1; b++` leaves 1 rather than 2, and - // `b = 0; b--` leaves 1 rather than 255. Only the CAS loop can apply - // that conversion before the store. - let needs_conversion = self.types.kind(lv.elem_typ) == TypeKind::Bool; - if !needs_conversion { - if let Some(atomic_op) = Self::atomic_opcode_for(op) { - return self.emit_atomic_op(atomic_op, lv, Some(value)); - } + if let Some(atomic_op) = self.native_rmw_opcode(ca) { + // The instruction computes at the object's own width, so the + // operand arrives at the object's own type -- the truncation the + // congruence below permits. Pointer arithmetic has scaled it to a + // byte count already, at pointer width. + let value = if ca.is_ptr_arith { + value + } else { + self.emit_convert(value, ca.value_typ, ca.target_typ) + }; + return self.emit_atomic_op(atomic_op, lv, Some(value)); } - self.emit_atomic_cas_loop(lv, op, value, false) + self.emit_atomic_cas_loop(lv, ca, value) } - /// `__atomic_fetch_nand` / `__sync_fetch_and_nand`: store `~(old & value)` - /// and return the old value. + /// The native atomic instruction that computes `ca` *exactly*, if one + /// does. + /// + /// `AtomicFetchAdd` and its siblings operate at the object's width and + /// store the raw result, where C17 6.5.16.2p3 computes at the operands' + /// common type and converts the result back + /// ([`Linearizer::compound_assign_value`]). The two agree when the + /// operator is congruent modulo 2^n -- add, subtract and the three bitwise + /// ops -- *and* the conversion back is the truncation congruence permits. /// - /// No target has an atomic NAND, so this is always the CAS loop -- the - /// same loop, with one more instruction inside it. Writing a second loop - /// would mean two places to get the LL/SC rules right. - pub(crate) fn emit_atomic_nand(&mut self, lv: &AtomicLvalue, value: PseudoId) -> PseudoId { - self.emit_atomic_cas_loop(lv, Opcode::And, value, true) + /// `_Bool` is where that second condition fails: converting to it is a + /// test against zero, not a truncation, so `b -= 1` must store 1 and only + /// the CAS loop can convert before the store. Divide, remainder, the + /// shifts and everything floating-point fail the first condition, and + /// `nand` complements a value the hardware would store as it stands. + fn native_rmw_opcode(&self, ca: &CompoundAssign) -> Option { + if ca.invert || self.types.kind(ca.target_typ) == TypeKind::Bool { + return None; + } + let arith_type = compound_assign_arith_type(self.types, ca); + Self::atomic_opcode_for(compound_assign_opcode(self.types, ca.op, arith_type)) } /// The CAS retry loop described on `emit_atomic_rmw`. fn emit_atomic_cas_loop( &mut self, lv: &AtomicLvalue, - op: Opcode, + ca: &CompoundAssign, value: PseudoId, - invert: bool, ) -> PseudoId { let elem_typ = lv.elem_typ; let bits = lv.size_bits; @@ -308,34 +327,15 @@ impl Linearizer<'_> { self.link_bb(entry_bb, loop_bb); self.switch_bb(loop_bb); - // old = *exp; new = old value + // old = *exp; new = the value `old value` assigns. + // + // Through the shared helper, so the loop computes at the same type the + // ordinary lowering does and converts the result back the same way -- + // which is also what keeps an `_Atomic _Bool` holding 0 or 1 rather + // than the raw 2 or 255 the arithmetic produced (C17 6.3.1.2). let old = self.alloc_reg_pseudo(); self.emit(Instruction::load(old, exp_addr, 0, elem_typ, bits)); - let new = self.alloc_reg_pseudo(); - self.emit(Instruction::binop(op, new, old, value, elem_typ, bits)); - // `nand` is `and` with the result complemented, which is the only - // reason this loop takes a flag rather than an opcode alone. - let new = if invert { - let inverted = self.alloc_reg_pseudo(); - self.emit(Instruction::unop( - Opcode::Not, - inverted, - new, - elem_typ, - bits, - )); - inverted - } else { - new - }; - // C17 6.3.1.2: converting to _Bool yields 0 or 1, and a compound - // assignment stores the converted result. Without this the raw sum - // reaches memory and an _Atomic _Bool holds 2 or 255. - let new = if self.types.kind(elem_typ) == TypeKind::Bool { - self.emit_convert(new, self.types.int_id, elem_typ) - } else { - new - }; + let new = self.compound_assign_value(ca, old, value); let ok = self.alloc_reg_pseudo(); let order = self.emit_const(ORDER as i128, self.types.int_id); @@ -392,26 +392,32 @@ impl Linearizer<'_> { let is_ptr_arith = self.types.kind(target_typ) == TypeKind::Pointer && self.types.is_integer(value_typ) && matches!(op, AssignOp::AddAssign | AssignOp::SubAssign); - - let (operand, arith_typ) = if is_ptr_arith { + let (operand, value_typ) = if is_ptr_arith { ( self.scale_pointer_addend(target_typ, value_typ, rhs), - target_typ, + self.types.long_id, ) } else { - (self.emit_convert(rhs, value_typ, target_typ), target_typ) + (rhs, value_typ) }; - let opcode = self.compound_assign_opcode(op, arith_typ); - let old = self.emit_atomic_rmw(&lv, opcode, operand); - - // Recompute the stored value from the old one. - let new = self.alloc_reg_pseudo(); - let bits = self.types.size_bits(arith_typ); - self.emit(Instruction::binop( - opcode, new, old, operand, arith_typ, bits, - )); - Some(new) + // The right operand goes on at its own type. Converting it down to the + // target here -- which this did -- computes `50 / (unsigned char)-5` + // where C17 6.5.16.2p3 computes `50 / -5` at `int` and converts only + // the result; `compound_assign_value` is the ordinary path's rule, now + // shared rather than copied. + let ca = CompoundAssign { + is_ptr_arith, + ..CompoundAssign::new(op, lv.elem_typ, value_typ) + }; + let old = self.emit_atomic_rmw(&lv, &ca, operand); + + // Recompute the stored value from the old one, by the same rule that + // stored it: C17 6.5.16p3 gives the expression the left operand's + // value *after* the assignment, which for `_Atomic _Bool b = 0` makes + // `(b -= 1)` the 1 that reached memory and not the 255 the subtraction + // produced. + Some(self.compound_assign_value(&ca, old, operand)) } /// Scale an integer addend by the pointee size, for `p += n`. @@ -436,63 +442,6 @@ impl Linearizer<'_> { )); scaled } - - /// The arithmetic opcode a compound assignment operator applies. - pub(crate) fn compound_assign_opcode(&self, op: AssignOp, typ: TypeId) -> Opcode { - let is_float = self.types.is_float(typ); - let is_unsigned = self.types.is_unsigned(typ); - match op { - AssignOp::Assign => unreachable!("plain assignment has no arithmetic opcode"), - AssignOp::AddAssign => { - if is_float { - Opcode::FAdd - } else { - Opcode::Add - } - } - AssignOp::SubAssign => { - if is_float { - Opcode::FSub - } else { - Opcode::Sub - } - } - AssignOp::MulAssign => { - if is_float { - Opcode::FMul - } else { - Opcode::Mul - } - } - AssignOp::DivAssign => { - if is_float { - Opcode::FDiv - } else if is_unsigned { - Opcode::DivU - } else { - Opcode::DivS - } - } - AssignOp::ModAssign => { - if is_unsigned { - Opcode::ModU - } else { - Opcode::ModS - } - } - AssignOp::AndAssign => Opcode::And, - AssignOp::OrAssign => Opcode::Or, - AssignOp::XorAssign => Opcode::Xor, - AssignOp::ShlAssign => Opcode::Shl, - AssignOp::ShrAssign => { - if is_unsigned { - Opcode::Lsr - } else { - Opcode::Asr - } - } - } - } } impl Linearizer<'_> { @@ -516,30 +465,33 @@ impl Linearizer<'_> { let lv = self.atomic_lvalue(operand)?; - // A pointer steps by one element; everything else by one. + // A pointer steps by one element; everything else by one. C17 + // 6.5.3.1p2 defines `++E` as `E += 1`, so the step is the right + // operand of a compound assignment and the same helper applies -- + // including its conversion of the result, which is what makes `++b` on + // an `_Atomic _Bool` still yield 0 or 1. + let is_ptr_arith = self.types.kind(typ) == TypeKind::Pointer; let delta = self.incdec_delta(typ); - let is_float = self.types.is_float(typ); - let opcode = match (is_inc, is_float) { - (true, false) => Opcode::Add, - (false, false) => Opcode::Sub, - (true, true) => Opcode::FAdd, - (false, true) => Opcode::FSub, + let delta_typ = if is_ptr_arith { + self.types.long_id + } else { + typ + }; + let op = if is_inc { + AssignOp::AddAssign + } else { + AssignOp::SubAssign + }; + let ca = CompoundAssign { + is_ptr_arith, + ..CompoundAssign::new(op, lv.elem_typ, delta_typ) }; - let old = self.emit_atomic_rmw(&lv, opcode, delta); + let old = self.emit_atomic_rmw(&lv, &ca, delta); if !prefix { return Some(old); } - - let bits = self.types.size_bits(typ); - let new = self.alloc_reg_pseudo(); - self.emit(Instruction::binop(opcode, new, old, delta, typ, bits)); - - // `++b` on an _Atomic _Bool must still yield 0 or 1. - if self.types.kind(typ) == TypeKind::Bool { - return Some(self.emit_convert(new, self.types.int_id, typ)); - } - Some(new) + Some(self.compound_assign_value(&ca, old, delta)) } /// The amount `++`/`--` steps by: the pointee size for a pointer, else 1. diff --git a/cc/ir/linearize_emit.rs b/cc/ir/linearize_emit.rs index 26cf9e63f..683198818 100644 --- a/cc/ir/linearize_emit.rs +++ b/cc/ir/linearize_emit.rs @@ -16,7 +16,7 @@ use crate::diag::{error, Position}; use crate::float::FloatVal; use crate::parse::ast::{AssignOp, BinaryOp, Expr, ExprKind, FpCompare, LibFn, MathErrno, UnaryOp}; use crate::strings::StringId; -use crate::types::{MemberInfo, TypeId, TypeKind}; +use crate::types::{MemberInfo, TypeId, TypeKind, TypeTable}; /// A read-modify-write target whose address has been computed **once**. /// @@ -38,6 +38,141 @@ pub(crate) struct RmwPlace { bitfield: Option<(usize, u32, u32, u32, TypeId)>, } +/// Everything `E1 op= E2` needs beyond the two operand *values*: the operator +/// and the types the arithmetic is decided by. +/// +/// C17 6.5.16.2p3 makes `E1 op= E2` mean `E1 = E1 op E2` bar evaluating `E1` +/// twice. So the arithmetic runs at the type the usual arithmetic conversions +/// give the two operands -- `target_typ` and `value_typ` -- and only the +/// *result* converts back to `target_typ`. Narrowing the right operand to the +/// target first is a different computation: `_Atomic unsigned char c = 50; +/// c /= -5;` becomes `50 / 251` and stores 0 where the standard stores +/// `(unsigned char)(50 / -5)`, 246. +/// +/// This exists because the ordinary and the `_Atomic` lowerings each had their +/// own copy of these rules and the copies disagreed. Both now build one of +/// these and hand it to [`Linearizer::compound_assign_value`]. +#[derive(Clone, Copy)] +pub(crate) struct CompoundAssign { + /// The operator. + pub(crate) op: AssignOp, + /// `E1`'s type: what the left operand is read at and what the result + /// converts back to. + pub(crate) target_typ: TypeId, + /// `E2`'s type **as written**, before any conversion. For pointer + /// arithmetic it is the type of the already-scaled addend. + pub(crate) value_typ: TypeId, + /// `p += n`: the right operand arrives already scaled by the pointee size, + /// the addition happens at pointer width, and the result is a pointer + /// already -- so neither operand nor result is converted. + pub(crate) is_ptr_arith: bool, + /// Complement the arithmetic result *before* it converts back to + /// `target_typ`. This is `nand`, which has no operator spelling of its + /// own: `__atomic_fetch_nand` stores `~(old & value)`, and on a `_Bool` + /// object that complement has to happen before the conversion to 0 or 1, + /// not after it. + pub(crate) invert: bool, +} + +impl CompoundAssign { + /// `E1 op= E2` with both operand types as written. + pub(crate) fn new(op: AssignOp, target_typ: TypeId, value_typ: TypeId) -> Self { + Self { + op, + target_typ, + value_typ, + is_ptr_arith: false, + invert: false, + } + } +} + +/// The type the arithmetic of `ca` is performed at. +/// +/// The usual arithmetic conversions (C17 6.3.1.8) on the two operands, with +/// two exceptions: +/// +/// * The shifts. C17 6.5.7p3 promotes each operand *separately* and gives the +/// result the promoted **left** operand's type, so the right operand has no +/// say: `_Atomic signed char s = -8; s >>= 1;` shifts -8 as an `int` and +/// stores -4, where computing at the target's width would shift the byte +/// pattern and store 124. +/// * Pointer arithmetic, whose addend the caller has already scaled to a +/// byte count; the addition happens at pointer width. +pub(crate) fn compound_assign_arith_type(types: &TypeTable, ca: &CompoundAssign) -> TypeId { + if ca.is_ptr_arith { + types.long_id + } else if matches!(ca.op, AssignOp::ShlAssign | AssignOp::ShrAssign) { + types.integer_promote(ca.target_typ) + } else { + types.common_type(ca.target_typ, ca.value_typ) + } +} + +/// The arithmetic opcode a compound assignment operator applies at `typ`. +/// +/// `typ` is the type the operation is *performed* at -- the answer of +/// [`compound_assign_arith_type`] -- because that is what decides between the +/// integer and floating forms and between the signed and unsigned ones. Asking +/// the target's type instead makes `unsigned char x; x /= -5;` an unsigned +/// divide of a value the standard computes as a signed `int`. +pub(crate) fn compound_assign_opcode(types: &TypeTable, op: AssignOp, typ: TypeId) -> Opcode { + let is_float = types.is_float(typ); + let is_unsigned = types.is_unsigned(typ); + match op { + AssignOp::Assign => unreachable!("plain assignment has no arithmetic opcode"), + AssignOp::AddAssign => { + if is_float { + Opcode::FAdd + } else { + Opcode::Add + } + } + AssignOp::SubAssign => { + if is_float { + Opcode::FSub + } else { + Opcode::Sub + } + } + AssignOp::MulAssign => { + if is_float { + Opcode::FMul + } else { + Opcode::Mul + } + } + AssignOp::DivAssign => { + if is_float { + Opcode::FDiv + } else if is_unsigned { + Opcode::DivU + } else { + Opcode::DivS + } + } + // Modulo not supported for floats. + AssignOp::ModAssign => { + if is_unsigned { + Opcode::ModU + } else { + Opcode::ModS + } + } + AssignOp::AndAssign => Opcode::And, + AssignOp::OrAssign => Opcode::Or, + AssignOp::XorAssign => Opcode::Xor, + AssignOp::ShlAssign => Opcode::Shl, + AssignOp::ShrAssign => { + if is_unsigned { + Opcode::Lsr + } else { + Opcode::Asr + } + } + } +} + /// The per-half opcodes a complex operation uses, chosen once from the base /// type rather than at each emit site. /// @@ -2655,6 +2790,81 @@ impl<'a> super::linearize::Linearizer<'a> { None } + /// The value `E1 op= E2` stores, given `E1`'s current value and `E2`'s. + /// + /// The whole of C17 6.5.16.2p3's arithmetic lives here: choose the type to + /// compute at, choose the opcode for it, bring both operands to it, apply + /// the operator, and convert the result back to the target. Callers supply + /// the two values and nothing else, which is what keeps the ordinary and + /// the `_Atomic` lowerings computing the same thing. + /// + /// `rhs` arrives **unconverted**, at `ca.value_typ`. Converting it to the + /// target first is not an optimization of this: it is a different + /// computation, and it was the bug (see [`CompoundAssign`]). + /// + /// The value this returns is also the value of the assignment expression + /// (C17 6.5.16p3: the left operand's value *after* the assignment), which + /// is why the conversion back is part of the helper rather than of the + /// store: `_Atomic _Bool b = 0; (b -= 1)` has to yield the 1 it stored, + /// not the 255 the subtraction produced. + pub(crate) fn compound_assign_value( + &mut self, + ca: &CompoundAssign, + lhs: PseudoId, + rhs: PseudoId, + ) -> PseudoId { + let arith_type = compound_assign_arith_type(self.types, ca); + let arith_size = self.types.size_bits(arith_type); + let opcode = compound_assign_opcode(self.types, ca.op, arith_type); + + // Both operands into the arithmetic type. The left one is the object's + // current value, read at the target's type; the right one is whatever + // it was written as. + let lhs = if ca.is_ptr_arith { + lhs + } else { + self.emit_convert(lhs, ca.target_typ, arith_type) + }; + let rhs = if ca.is_ptr_arith { + rhs + } else if matches!(ca.op, AssignOp::ShlAssign | AssignOp::ShrAssign) { + // The shift count is promoted on its own and is not brought to the + // left operand's type (C17 6.5.7p3). + self.emit_convert(rhs, ca.value_typ, self.types.integer_promote(ca.value_typ)) + } else { + self.emit_convert(rhs, ca.value_typ, arith_type) + }; + + let result = self.alloc_reg_pseudo(); + self.emit(Instruction::binop( + opcode, result, lhs, rhs, arith_type, arith_size, + )); + + // `nand` is `and` with the result complemented, and the complement + // belongs on this side of the conversion below. + let result = if ca.invert { + let inverted = self.alloc_reg_pseudo(); + self.emit(Instruction::unop( + Opcode::Not, + inverted, + result, + arith_type, + arith_size, + )); + inverted + } else { + result + }; + + // And the result back, which is the conversion that makes `(x /= y)` + // yield what `x` now holds. Pointer arithmetic is already a pointer. + if ca.is_ptr_arith { + result + } else { + self.emit_convert(result, arith_type, ca.target_typ) + } + } + pub(crate) fn emit_assign(&mut self, op: AssignOp, target: &Expr, value: &Expr) -> PseudoId { let target_typ = self.expr_type(target); let value_typ = self.expr_type(value); @@ -2866,8 +3076,9 @@ impl<'a> super::linearize::Linearizer<'a> { )); scaled } else if bool_rhs.is_some() || op != AssignOp::Assign { - // A compound assignment leaves its right operand alone here. It is - // converted to the *common* type below, not down to the target's: + // A compound assignment leaves its right operand alone here. It + // is converted to the *common* type by `compound_assign_value`, + // not down to the target's: // narrowing `-5` to `unsigned char` first made `x /= y` divide // 50 by 251 and store 0, where C17 6.5.16.2p3 computes `50 / -5` // at `int` and stores `(unsigned char)-10`. @@ -2896,109 +3107,14 @@ impl<'a> super::linearize::Linearizer<'a> { Some(p) => self.load_rmw_place(p, target_typ), None => self.linearize_expr(target), }; - let result = self.alloc_reg_pseudo(); - - // `E1 op= E2` is `E1 = E1 op E2` (C17 6.5.16.2p3), so the - // operation runs at the operands' common type after the - // integer promotions -- not at the target's type, which is - // only what the *result* converts back to. - // - // The shifts are the exception: 6.5.7p3 gives the result the - // promoted *left* operand's type, and promotes the right one - // on its own. - let arith_type = if is_ptr_arith { - // Pointer arithmetic already scaled the index; the add - // happens at pointer width. - self.types.long_id - } else if matches!(op, AssignOp::ShlAssign | AssignOp::ShrAssign) { - self.types.integer_promote(target_typ) - } else { - self.types.common_type(target_typ, value_typ) - }; - - let is_float = self.types.is_float(arith_type); - let is_unsigned = self.types.is_unsigned(arith_type); - let opcode = match op { - AssignOp::AddAssign => { - if is_float { - Opcode::FAdd - } else { - Opcode::Add - } - } - AssignOp::SubAssign => { - if is_float { - Opcode::FSub - } else { - Opcode::Sub - } - } - AssignOp::MulAssign => { - if is_float { - Opcode::FMul - } else { - Opcode::Mul - } - } - AssignOp::DivAssign => { - if is_float { - Opcode::FDiv - } else if is_unsigned { - Opcode::DivU - } else { - Opcode::DivS - } - } - AssignOp::ModAssign => { - // Modulo not supported for floats - if is_unsigned { - Opcode::ModU - } else { - Opcode::ModS - } - } - AssignOp::AndAssign => Opcode::And, - AssignOp::OrAssign => Opcode::Or, - AssignOp::XorAssign => Opcode::Xor, - AssignOp::ShlAssign => Opcode::Shl, - AssignOp::ShrAssign => { - if is_unsigned { - Opcode::Lsr - } else { - Opcode::Asr - } - } - AssignOp::Assign => unreachable!(), - }; - - let arith_size = self.types.size_bits(arith_type); - // Both operands into the arithmetic type. The left one is the - // object's current value, read at the target's type; the right - // one is whatever it was written as. - let lhs = if is_ptr_arith { - lhs - } else { - self.emit_convert(lhs, target_typ, arith_type) - }; - let rhs = if is_ptr_arith { - rhs - } else if matches!(op, AssignOp::ShlAssign | AssignOp::ShrAssign) { - // The shift count is promoted on its own and is not - // brought to the left operand's type. - self.emit_convert(rhs, value_typ, self.types.integer_promote(value_typ)) - } else { - self.emit_convert(rhs, value_typ, arith_type) + // One helper owns the whole of C17 6.5.16.2p3's arithmetic, + // shared with the `_Atomic` lowering, which used to carry its + // own copy of these rules and disagree with this one. + let ca = CompoundAssign { + is_ptr_arith, + ..CompoundAssign::new(op, target_typ, value_typ) }; - self.emit(Instruction::binop( - opcode, result, lhs, rhs, arith_type, arith_size, - )); - // And the result back, which is the conversion that makes - // `(x /= y)` yield what `x` now holds. - if is_ptr_arith { - result - } else { - self.emit_convert(result, arith_type, target_typ) - } + self.compound_assign_value(&ca, lhs, rhs) } }; diff --git a/cc/ir/test_linearize.rs b/cc/ir/test_linearize.rs index a4959ba94..c4bd0199e 100644 --- a/cc/ir/test_linearize.rs +++ b/cc/ir/test_linearize.rs @@ -13,6 +13,7 @@ #![allow(clippy::approx_constant)] use super::*; +use crate::ir::linearize_emit::{compound_assign_arith_type, compound_assign_opcode}; use crate::parse::ast::{ AsmOperand, AssignOp, BinaryOp, BlockItem, Declaration, Designator, ExprKind, ExternalDecl, ForInit, FunctionDef, InitDeclarator, InitElement, ParamStyle, Parameter, Stmt, UnaryOp, @@ -6571,6 +6572,260 @@ fn test_atomic_aggregate_assign_uses_atomic_store() { ); } +// One model for `E1 op= E2` (C17 6.5.16.2p3) +// +// The ordinary and the `_Atomic` lowerings each used to carry their own copy +// of these rules, and the copies disagreed: the atomic one converted the right +// operand down to the target and computed there, so `_Atomic unsigned char c = +// 50; c /= -5;` divided 50 by 251 and stored 0. Both now go through +// `compound_assign_value`, and these tests pin the decisions it makes. + +/// Build `void test(T x) { x = ; }` with `T` the chosen type made +/// `_Atomic` and the right operand a plain `int` literal, and linearize it. +/// +/// `target` is a selector rather than a `TypeId` because the table the id +/// belongs to is built by `TestContext::new`. +fn atomic_typed_module( + op: AssignOp, + target: fn(&TypeTable) -> TypeId, + value: i64, +) -> (TestContext, Module) { + let mut ctx = TestContext::new(); + let test_id = ctx.str("test"); + let int_id = ctx.types.int_id; + + let base = target(&ctx.types); + let atomic_typ = { + let mut t = ctx.types.get(base).clone(); + t.modifiers |= TypeModifiers::ATOMIC; + ctx.types.intern(t) + }; + let x_sym = ctx.var("x", atomic_typ); + + let assign = Expr::typed_unpositioned( + ExprKind::Assign { + op, + target: Box::new(Expr::var_typed(x_sym, atomic_typ)), + value: Box::new(Expr::typed_unpositioned(ExprKind::IntLit(value), int_id)), + }, + atomic_typ, + ); + let func = FunctionDef { + attrs: Default::default(), + return_type: ctx.types.void_id, + name: test_id, + params: vec![Parameter { + symbol: Some(x_sym), + typ: atomic_typ, + vm_dims: vec![], + discarded_dims: vec![], + }], + body: Stmt::Block(vec![BlockItem::Statement(Box::new(Stmt::Expr(assign)))]), + pos: test_pos(), + is_static: false, + is_inline: false, + calling_conv: crate::abi::CallingConv::default(), + param_style: ParamStyle::Prototype, + }; + let module = ctx.linearize(&TranslationUnit { + items: vec![ExternalDecl::FunctionDef(func)], + }); + (ctx, module) +} + +/// The first instruction with this opcode, for asserting on its type and width. +fn first_op(module: &Module, op: Opcode) -> &Instruction { + module.functions[0] + .blocks + .iter() + .flat_map(|bb| bb.insns.iter()) + .find(|i| i.op == op) + .unwrap_or_else(|| panic!("no {:?} in the module", op)) +} + +/// The usual arithmetic conversions decide the type, and the type decides the +/// opcode -- so a narrow unsigned target divided by an `int` is a *signed* +/// 32-bit divide. +#[test] +fn test_compound_assign_divides_at_the_operands_common_type() { + let types = TypeTable::new(&Target::host()); + + let ca = CompoundAssign::new(AssignOp::DivAssign, types.uchar_id, types.int_id); + let arith = compound_assign_arith_type(&types, &ca); + assert_eq!(arith, types.int_id, "unsigned char / int is done at int"); + assert_eq!( + compound_assign_opcode(&types, AssignOp::DivAssign, arith), + Opcode::DivS + ); + + // Asking the *target's* type instead is the defect this replaced: it makes + // the same expression an unsigned divide, and `50 /= -5` stores 0. + assert_eq!( + compound_assign_opcode(&types, AssignOp::DivAssign, types.uchar_id), + Opcode::DivU + ); +} + +/// The congruent operators are decided the same way, even though their result +/// is the same either width. +#[test] +fn test_compound_assign_add_also_computes_at_the_common_type() { + let types = TypeTable::new(&Target::host()); + let ca = CompoundAssign::new(AssignOp::AddAssign, types.uchar_id, types.int_id); + assert_eq!(compound_assign_arith_type(&types, &ca), types.int_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::AddAssign, types.int_id), + Opcode::Add + ); + // A floating target picks the floating form of the same operator. + let fca = CompoundAssign::new(AssignOp::AddAssign, types.float_id, types.int_id); + let farith = compound_assign_arith_type(&types, &fca); + assert_eq!(farith, types.float_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::AddAssign, farith), + Opcode::FAdd + ); +} + +/// A shift promotes its **left** operand and nothing else (C17 6.5.7p3), so +/// the right operand's type has no say in the width it is done at. +#[test] +fn test_compound_assign_shift_takes_the_promoted_left_operand() { + let types = TypeTable::new(&Target::host()); + + let ca = CompoundAssign::new(AssignOp::ShrAssign, types.schar_id, types.longlong_id); + let arith = compound_assign_arith_type(&types, &ca); + assert_eq!( + arith, types.int_id, + "the promoted left operand decides, not the common type" + ); + assert_ne!( + arith, + types.common_type(types.schar_id, types.longlong_id), + "the shift must not follow the usual arithmetic conversions" + ); + + // And the promotion is what makes the shift arithmetic: `unsigned char` + // promotes to `int`, so `u >>= 1` on 200 is 100 and not a logical shift of + // the byte. + let uca = CompoundAssign::new(AssignOp::ShrAssign, types.uchar_id, types.int_id); + let uarith = compound_assign_arith_type(&types, &uca); + assert_eq!(uarith, types.int_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::ShrAssign, uarith), + Opcode::Asr + ); + assert_eq!( + compound_assign_opcode(&types, AssignOp::ShrAssign, types.uchar_id), + Opcode::Lsr, + "computing at the target's own width would shift the wrong way" + ); +} + +/// `_Bool` promotes to `int` like any narrow integer; what is special about it +/// is the conversion *back*, which is a test against zero. +#[test] +fn test_compound_assign_bool_computes_at_int() { + let types = TypeTable::new(&Target::host()); + let ca = CompoundAssign::new(AssignOp::SubAssign, types.bool_id, types.int_id); + assert_eq!(compound_assign_arith_type(&types, &ca), types.int_id); +} + +/// Pointer arithmetic is the other exception: the addend arrives already +/// scaled to a byte count and the addition happens at pointer width. +#[test] +fn test_compound_assign_pointer_arithmetic_is_done_at_pointer_width() { + let types = TypeTable::new(&Target::host()); + let ca = CompoundAssign { + is_ptr_arith: true, + ..CompoundAssign::new(AssignOp::AddAssign, types.char_ptr_id, types.long_id) + }; + assert_eq!(compound_assign_arith_type(&types, &ca), types.long_id); + assert_eq!( + compound_assign_opcode(&types, AssignOp::AddAssign, types.long_id), + Opcode::Add + ); +} + +/// `_Atomic unsigned char c; c /= -5;` divides at `int`, in the CAS loop -- +/// the same arithmetic the ordinary lowering does. +#[test] +fn test_atomic_compound_divide_computes_at_the_common_type() { + let (_ctx, module) = atomic_typed_module(AssignOp::DivAssign, |t| t.uchar_id, -5); + + let div = first_op(&module, Opcode::DivS); + assert_eq!( + div.size, 32, + "the divide happens at the common type's width, not the object's" + ); + assert_eq!( + count_op(&module, Opcode::DivU), + 0, + "narrowing the right operand first would make this an unsigned divide" + ); + assert_eq!( + count_op(&module, Opcode::AtomicCas), + 1, + "divide has no native atomic form" + ); +} + +/// The same for a shift: promoted left operand, 32-bit arithmetic shift. +#[test] +fn test_atomic_compound_shift_promotes_its_left_operand() { + let (_ctx, module) = atomic_typed_module(AssignOp::ShrAssign, |t| t.uchar_id, 1); + + let shift = first_op(&module, Opcode::Asr); + assert_eq!(shift.size, 32, "the left operand is promoted to int first"); + assert_eq!( + count_op(&module, Opcode::Lsr), + 0, + "an 8-bit logical shift would be the target's width, not the promoted one" + ); + assert_eq!(count_op(&module, Opcode::AtomicCas), 1); +} + +/// A narrow congruent operator keeps its native fetch-and-op. +/// +/// The standard computes `c += 100` at `int` and converts back, but add is +/// congruent modulo 2^8, so the hardware's 8-bit add agrees with it -- and a +/// single instruction beats a retry loop. +#[test] +fn test_atomic_narrow_add_keeps_its_native_fetch_op() { + let (_ctx, module) = atomic_typed_module(AssignOp::AddAssign, |t| t.uchar_id, 100); + + assert_eq!(count_op(&module, Opcode::AtomicFetchAdd), 1); + assert_eq!( + count_op(&module, Opcode::AtomicCas), + 0, + "no retry loop needed" + ); + assert_eq!( + first_op(&module, Opcode::AtomicFetchAdd).size, + 8, + "the atomic operates at the object's own width" + ); +} + +/// `_Atomic _Bool` cannot: converting to `_Bool` is a test against zero, not +/// the truncation congruence permits, so the value stored has to be computed +/// before the exchange. +#[test] +fn test_atomic_bool_compound_assign_cannot_use_a_native_fetch_op() { + let (_ctx, module) = atomic_typed_module(AssignOp::SubAssign, |t| t.bool_id, 1); + + assert_eq!( + count_op(&module, Opcode::AtomicFetchSub), + 0, + "a native fetch-and-sub would store the raw 255" + ); + assert_eq!(count_op(&module, Opcode::AtomicCas), 1); + assert!( + count_op(&module, Opcode::SetNe) >= 1, + "the CAS loop must convert the result to _Bool before storing it" + ); +} + /// A complex member of an automatic struct is stored as two halves. /// /// A complex value travels by *address*, so storing it the way a scalar member diff --git a/cc/tests/c11/atomics.rs b/cc/tests/c11/atomics.rs index 48f761bbf..8ffd0f719 100644 --- a/cc/tests/c11/atomics.rs +++ b/cc/tests/c11/atomics.rs @@ -948,3 +948,141 @@ int main(void) { "#; assert_eq!(compile_and_run("c11_atomic_spellings", code, &[]), 0); } + +/// A compound assignment to an `_Atomic` object computes at the same type as +/// one to an ordinary object. +/// +/// C17 6.5.16.2p3 defines `E1 op= E2` as `E1 = E1 op E2` bar evaluating `E1` +/// once, so the arithmetic happens at the type the usual arithmetic +/// conversions give the two operands -- and only the *result* is converted back +/// to the target. The atomic path converted the right operand down to the +/// target first and computed there, so `50 / -5` became `50 / 251` and stored +/// 0. The ordinary path already had this fixed, with a comment explaining it; +/// the atomic path had its own copy of the logic and did not. +/// +/// Add, subtract, and the bitwise operators are congruent modulo 2^n, so a +/// narrow computation agrees with a wide one and their native fetch-and-op +/// lowering stays correct. Division, remainder and the shifts are not, and all +/// of them already take the compare-and-swap loop. +/// +/// Each case is checked against the ordinary object beside it: the two paths +/// agreeing is the property, and their disagreeing is how this survived. +#[test] +fn c11_an_atomic_compound_assignment_computes_at_the_common_type() { + let code = r#" +int main(void) +{ + /* Division: the right operand must not be narrowed to unsigned char + first. 50 / -5 is -10 at int, stored as (unsigned char)-10 == 246. */ + _Atomic unsigned char ac = 50; ac /= -5; + unsigned char pc = 50; pc /= -5; + if (ac != pc || ac != 246) return 1; + + /* Remainder, likewise: 50 % -3 is 2. */ + _Atomic unsigned char am = 50; am %= -3; + unsigned char pm = 50; pm %= -3; + if (am != pm || am != 2) return 2; + + /* Signed division, where narrowing would also change the sign. */ + _Atomic signed char as = -100; as /= 3; + signed char ps = -100; ps /= 3; + if (as != ps || as != -33) return 3; + + /* The congruent operators must keep working -- they take the native + fetch-and-op lowering, not the CAS loop. */ + _Atomic unsigned char aa = 200; aa += 100; + unsigned char pa = 200; pa += 100; + if (aa != pa || aa != 44) return 4; + + _Atomic unsigned char an = 0xF0; an &= -1; + unsigned char pn = 0xF0; pn &= -1; + if (an != pn || an != 0xF0) return 5; + + return 0; +} +"#; + assert_eq!(compile_and_run("atomic_compound_common_type", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("atomic_compound_common_type_opt", code), + 0 + ); +} + +/// The value of a compound assignment is the value stored, converted. +/// +/// C17 6.5.16p3: an assignment expression has the value of the left operand +/// *after* the assignment. For a `_Bool` that means the value after conversion +/// to `_Bool`, so `b -= 1` on a false `b` yields 1 -- the memory and the +/// expression have to agree. c17's ordinary path did this and its atomic path +/// did not, recomputing the expression's value from a raw arithmetic result +/// and handing back 255 while storing 1. +/// +/// Note clang answers 255 here for the atomic case and 1 for the ordinary one, +/// i.e. it has the same split. This follows the standard and c17's own +/// non-atomic path rather than matching that. +#[test] +fn c11_an_atomic_compound_assignment_yields_the_value_it_stored() { + let code = r#" +int main(void) +{ + _Atomic _Bool ab = 0; int ar = (ab -= 1); + _Bool pb = 0; int pr = (pb -= 1); + if (ab != 1 || pb != 1) return 1; + if (ar != pr || ar != 1) return 2; + + _Atomic _Bool ab2 = 1; int ar2 = (ab2 += 7); + _Bool pb2 = 1; int pr2 = (pb2 += 7); + if (ab2 != 1 || pb2 != 1) return 3; + if (ar2 != pr2 || ar2 != 1) return 4; + + /* A narrowing store: the expression is the stored value, not the wide one. */ + _Atomic unsigned char au = 200; int aur = (au += 100); + unsigned char pu = 200; int pur = (pu += 100); + if (aur != pur || aur != 44) return 5; + + return 0; +} +"#; + assert_eq!(compile_and_run("atomic_compound_result", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("atomic_compound_result_opt", code), + 0 + ); +} + +/// A shift on an atomic object promotes its left operand, as any shift does. +/// +/// C17 6.5.7p3: the integer promotions are applied to each operand and the +/// result has the promoted left operand's type. So `s >>= 1` on a +/// `signed char` holding -8 shifts -8 at `int`, giving -4, and stores that -- +/// not a logical shift of the unsigned byte pattern, which would give 124. +/// +/// This one c17 already gets right and clang does not, so it is a guard rather +/// than a repair: the fix for the two tests above must not reach the shift by +/// computing at the target's width. +#[test] +fn c11_an_atomic_shift_promotes_its_left_operand() { + let code = r#" +int main(void) +{ + _Atomic signed char as = -8; as >>= 1; + signed char ps = -8; ps >>= 1; + if (as != ps || as != -4) return 1; + + _Atomic signed char al = -8; al <<= 2; + signed char pl = -8; pl <<= 2; + if (al != pl || al != -32) return 2; + + _Atomic unsigned char au = 200; au >>= 1; + unsigned char pu = 200; pu >>= 1; + if (au != pu || au != 100) return 3; + + return 0; +} +"#; + assert_eq!(compile_and_run("atomic_shift_promotion", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("atomic_shift_promotion_opt", code), + 0 + ); +} From c301bfb22ab819f1888948e9822b91331f88fbc1 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 05:55:38 -0400 Subject: [PATCH 12/17] cc: convert a case label to the type the switch compares at C17 6.8.4.2p5 converts each case constant to the promoted type of the controlling expression, and p3 forbids two of them having the same value after that conversion. c17 evaluated labels at full width and never converted them -- a comment in the collector said so deliberately -- so only their *ordering* knew the switch's type, through the `unsigned` flag. That left the two lowerings disagreeing about the same switch. With a runtime selector the label went into the `switch` instruction as written; with a constant selector the fast path compared at 128 bits with a signed range test that ignored signedness. `switch (x) { case 4294967296LL: }` matched `x == 0`, and the same switch on a constant 0 did not. `case -1:` never matched a `switch` on `unsigned`. `CaseConv` is the promoted type -- a width and a signedness -- and owns the conversion, the ordering and the range test, so the fast path, the wide comparison chain and the recorded ranges all ask one thing. `CaseSet` holds it and converts on insert and on overlap, so only converted values are ever in the set. The label is looked up twice, and that is the hazard: the collector records it, and the body walk finds its block by the same `(lo, hi)` key. Convert one and not the other and the lookup misses -- no block for the case, its body emitted into another one, no diagnostic. So `CaseIndex` is built from the set and carries the set's own conv, and `lookup` is the only way into the map. The walk hands over the label's raw constants and cannot apply a different rule, there being no other rule reachable from it. Converting also makes the existing duplicate-case error see collisions it could not before: `case 0:` beside `case 4294967296LL:` in an `int` switch is one value twice. A lossy conversion is now diagnosed, as gcc and clang do, by converting to the controlling type and back through the label's own type and warning when the constant does not return. A value that is merely reinterpreted round- trips, so `case -1:` in an `unsigned` switch stays silent, and so does a label already of the promoted type in a `short` switch. Nothing in the suite was relying on a label outside its controlling type, so no existing expectation changed. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize_stmt.rs | 549 ++++++++++++++++++++++++++--------- cc/tests/c89/control_flow.rs | 52 ++++ cc/tests/diagnostics/mod.rs | 49 ++++ 3 files changed, 518 insertions(+), 132 deletions(-) diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index 11645ae63..5c33148f4 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -37,6 +37,10 @@ fn return_value_ness_violation(pos: Position, msg: &str) { use crate::types::{TypeId, TypeKind, TypeModifiers}; +/// The `-Wno-` group for a case label the switch's promoted controlling +/// type cannot hold, spelled as gcc spells the same diagnostic. +const CASE_RANGE_WARNING: &str = "switch-outside-range"; + /// Whether [`Linearizer::store_string_units`] owes the destination's tail a /// zero fill. /// @@ -1687,10 +1691,17 @@ impl<'a> super::linearize::Linearizer<'a> { // Push exit block for break handling self.break_targets.push(exit_bb); - // Collect case labels and create basic blocks for each - let switch_unsigned = self.types.is_unsigned(cmp_type); - let (case_values, has_default) = self.collect_switch_cases(body, switch_unsigned); - let case_bbs: Vec = case_values.iter().map(|_| self.alloc_bb()).collect(); + // Collect case labels and create basic blocks for each. C17 6.8.4.2p5 + // converts every label to `cmp_type`, and `conv` is that conversion: + // the collector applies it, and the body walk below reaches the labels + // back through the same value, so the two cannot drift apart. + let conv = CaseConv::of(self.types, cmp_type); + let (case_values, has_default) = self.collect_switch_cases(body, conv); + let case_bbs: Vec = case_values + .ranges() + .iter() + .map(|_| self.alloc_bb()) + .collect(); let default_bb = if has_default { Some(self.alloc_bb()) } else { @@ -1704,11 +1715,15 @@ impl<'a> super::linearize::Linearizer<'a> { // A constant selector takes one edge, as a constant condition does // (`branch_on`): the labels it does not select are reached only by // falling into them, and a block nothing reaches is not emitted. - // Both sides are values of the promoted type, as the collector - // records the labels. + // Selector and labels are both converted to the promoted type, and + // the range test runs in that type's signedness -- a plain signed + // `i128` comparison answered differently from the runtime lowering + // of the very same switch. + let selector = conv.convert(selector); let target = case_values + .ranges() .iter() - .position(|&(lo, hi)| lo <= selector && selector <= hi) + .position(|&(lo, hi)| conv.contains(lo, hi, selector)) .map_or(default_target, |idx| case_bbs[idx]); if let Some(current) = self.current_bb { self.emit(Instruction::br(target)); @@ -1724,13 +1739,17 @@ impl<'a> super::linearize::Linearizer<'a> { self.emit_wide_switch( switch_val, cmp_type, - &case_values, + conv, + case_values.ranges(), &case_bbs, default_target, ); } else { - // Build switch instruction with case -> block mapping + // Build switch instruction with case -> block mapping. The labels + // have been converted to `cmp_type`, which is at most 64 bits + // here, so the cast keeps every bit of each one. let switch_cases: Vec<(i64, i64, BasicBlockId)> = case_values + .ranges() .iter() .zip(case_bbs.iter()) .map(|((lo, hi), bb)| (*lo as i64, *hi as i64, *bb)) @@ -1763,14 +1782,10 @@ impl<'a> super::linearize::Linearizer<'a> { // block via `switch_bb`. self.current_bb = None; - // Linearize body with case block switching - // Each label's position among the cases, by its range. The first wins, - // as a scan in source order would find it; a duplicate has already - // been reported. - let mut case_index = CaseIndex::new(); - for (idx, range) in case_values.iter().enumerate() { - case_index.entry(*range).or_insert(idx); - } + // Linearize body with case block switching. The index carries the + // collector's own conversion, so the walk finds a label under exactly + // the key the collector filed it under. + let case_index = CaseIndex::of(&case_values); self.linearize_switch_body(body, &case_index, &case_bbs, default_bb); // If not terminated after body, jump to exit @@ -1795,33 +1810,22 @@ impl<'a> super::linearize::Linearizer<'a> { /// only `Stmt::Block` here collected no cases from it -- so the switch was /// emitted with an empty table and every value took the default edge. /// A non-compound body is one statement, so it is walked as one. - pub(crate) fn collect_switch_cases( - &self, - body: &Stmt, - unsigned: bool, - ) -> (Vec<(i128, i128)>, bool) { - let mut case_values = CaseSet::new(unsigned); + pub(crate) fn collect_switch_cases(&self, body: &Stmt, conv: CaseConv) -> (CaseSet, bool) { + let mut case_values = CaseSet::new(conv); let mut has_default = false; match body { Stmt::Block(items) => { for item in items { if let BlockItem::Statement(stmt) = item { - self.collect_cases_from_stmt( - stmt, - &mut case_values, - &mut has_default, - unsigned, - ); + self.collect_cases_from_stmt(stmt, &mut case_values, &mut has_default); } } } - stmt => { - self.collect_cases_from_stmt(stmt, &mut case_values, &mut has_default, unsigned) - } + stmt => self.collect_cases_from_stmt(stmt, &mut case_values, &mut has_default), } - (case_values.ranges, has_default) + (case_values, has_default) } /// The block every computed `goto` in this function branches through, @@ -2025,91 +2029,125 @@ impl<'a> super::linearize::Linearizer<'a> { } } + /// One case endpoint converted to the promoted controlling type, warning + /// if the controlling type cannot hold the constant the label spells. + /// + /// C17 6.8.4.2p5 requires the conversion, and most of the time it changes + /// nothing worth saying: `case -1:` in a `switch` on `unsigned` becomes + /// 4294967295, which is exactly the value it is written to match, and gcc + /// and clang are both silent there. What is worth a diagnostic is a label + /// whose bits the conversion throws away, silently turning + /// `case 4294967296LL:` into `case 0:`. + /// + /// The line between the two is whether converting back to the label's own + /// type returns the constant: a merely reinterpreted value round-trips, + /// while a truncated one does not. That is the test clang applies, and + /// gcc's `-Wswitch-outside-range` draws the line in the same place, which + /// is also the `-Wno-` name that silences this. + fn convert_case_label(&self, expr: &Expr, val: i128, conv: CaseConv) -> i128 { + let converted = conv.convert(val); + if converted == val { + return val; + } + // The label's own type, which the round trip goes back through. An + // untyped or non-integer label is already an error elsewhere; treat it + // as full width, which reduces the round trip to a plain comparison. + let own = expr.typ.filter(|&t| self.types.is_integer(t)).map_or_else( + || CaseConv::new(128, false), + |t| CaseConv::of(self.types, t), + ); + if own.convert(converted) != val && crate::diag::warning_group_enabled(CASE_RANGE_WARNING) { + crate::diag::warning( + expr.pos, + &format!( + "overflow converting case value to switch condition type \ + ({val} to {converted})" + ), + ); + } + converted + } + pub(crate) fn collect_cases_from_stmt( &self, stmt: &Stmt, case_values: &mut CaseSet, has_default: &mut bool, - unsigned: bool, ) { match stmt { Stmt::Case(expr, high, body) => { - self.collect_cases_from_stmt(body, case_values, has_default, unsigned); + self.collect_cases_from_stmt(body, case_values, has_default); // Extract constant value from case expression - if let Some(val) = self.eval_const_expr(expr) { - // Kept at full width. Truncating to `i64` here was silent - // and wrong for a `switch` on `__int128`: a label outside - // the 64-bit range wrapped into it and could match a value - // it does not equal. - - // A GNU range `case lo ... hi:`. An absent high endpoint - // is the ordinary label, held as the degenerate range - // `(v, v)` so that everything downstream has one shape. - let hi = match high { - None => Some(val), - Some(hi_expr) => match self.eval_const_expr(hi_expr) { - Some(h) => Some(h), - None => { - self.report_unfoldable_case(hi_expr); - None - } - }, - }; - let Some(hi) = hi else { return }; - - // 6.8.4.2p3 forbids two equal case constants, and GCC - // extends that to overlapping ranges -- an overlap would - // otherwise make one arm silently unreachable, since the - // body walk resolves a label by finding the first match. - // Order by the switch type's own signedness. The - // endpoints are carried as `i128`, and an unsigned 64-bit - // bound above `i64::MAX` is still positive there -- but an - // unsigned *128-bit* one is not, so the reinterpretation - // is still needed: `case 0ul ... ULONG_MAX:` read as an - // empty range and never matched. - let below = |a: i128, b: i128| { - if unsigned { - (a as u128) < (b as u128) - } else { - a < b + let Some(raw_lo) = self.eval_const_expr(expr) else { + self.report_unfoldable_case(expr); + return; + }; + // A GNU range `case lo ... hi:`. An absent high endpoint is + // the ordinary label, held as the degenerate range `(v, v)` so + // that everything downstream has one shape. + let raw_hi = match high { + None => Some(raw_lo), + Some(hi_expr) => match self.eval_const_expr(hi_expr) { + Some(h) => Some(h), + None => { + self.report_unfoldable_case(hi_expr); + None } + }, + }; + let Some(raw_hi) = raw_hi else { return }; + + // C17 6.8.4.2p5: each case constant is converted to the + // promoted type of the controlling expression. Evaluating the + // label at full width and never converting it left c17's two + // lowerings disagreeing about the same switch -- a runtime + // selector kept the unconverted label in the `switch` + // instruction, where the backend truncated it, while the + // constant-selector path compared at 128 bits and did not + // match at all. `case 4294967296LL:` in an `int` switch is + // `case 0:`, and has to be that for both. + let conv = case_values.conv(); + let lo = self.convert_case_label(expr, raw_lo, conv); + let hi = match high { + None => lo, + Some(hi_expr) => self.convert_case_label(hi_expr, raw_hi, conv), + }; + + // 6.8.4.2p3 forbids two equal case constants, and GCC + // extends that to overlapping ranges -- an overlap would + // otherwise make one arm silently unreachable, since the + // body walk resolves a label by finding the first match. + // Both tests run on the converted values, since that is what + // "equal" means once p5 has been applied: `case 0:` beside + // `case 4294967296LL:` in an `int` switch is one value twice. + // + // Order by the switch type's own signedness. The endpoints are + // carried as `i128`, and an unsigned 64-bit bound above + // `i64::MAX` is still positive there -- but an unsigned + // *128-bit* one is not, so the reinterpretation is still + // needed: `case 0ul ... ULONG_MAX:` read as an empty range and + // never matched. + if conv.lt(hi, lo) { + // GCC accepts an empty range, warns, and never matches + // it. Nothing is recorded, so nothing can overlap it. + crate::diag::warning(expr.pos, "empty range specified"); + return; + } + if let Some((lo2, hi2)) = case_values.overlap(lo, hi) { + let what = if lo == hi && lo2 == hi2 { + format!("duplicate case value '{}' in switch", lo) + } else { + format!( + "duplicate (or overlapping) case value: {}..{} overlaps {}..{}", + lo, hi, lo2, hi2 + ) }; - if below(hi, val) { - // GCC accepts an empty range, warns, and never matches - // it. Nothing is recorded, so nothing can overlap it. - crate::diag::warning(expr.pos, "empty range specified"); - return; - } - if let Some((lo2, hi2)) = case_values.overlap(val, hi) { - let what = if val == hi && lo2 == hi2 { - format!("duplicate case value '{}' in switch", val) - } else { - format!( - "duplicate (or overlapping) case value: {}..{} overlaps {}..{}", - val, hi, lo2, hi2 - ) - }; - error(expr.pos, &what); - } - case_values.insert(val, hi); - } else if self.expr_is_runtime(expr) { - // A non-constant label can never match. - error(expr.pos, "case label is not an integer constant expression"); - } else { - // Constant in principle, but `eval_const_expr` is a partial - // evaluator and could not fold it. Saying the program is - // invalid would be a false claim about the source — this is - // our limit, not its error. Either way the label cannot be - // emitted, so it still has to be reported rather than - // silently dropped. - error( - expr.pos, - "case label is a constant expression this compiler cannot evaluate", - ); + error(expr.pos, &what); } + case_values.insert(lo, hi); } Stmt::Default(_, body) => { - self.collect_cases_from_stmt(body, case_values, has_default, unsigned); + self.collect_cases_from_stmt(body, case_values, has_default); // C99 6.8.4.2p3: at most one default label per switch. if *has_default { error( @@ -2121,28 +2159,28 @@ impl<'a> super::linearize::Linearizer<'a> { } // Recursively check labeled statements Stmt::Label { stmt, .. } => { - self.collect_cases_from_stmt(stmt, case_values, has_default, unsigned); + self.collect_cases_from_stmt(stmt, case_values, has_default); } // Recurse into nested statements for Duff's device pattern // (case labels inside loops/blocks within a switch) Stmt::Block(items) => { for item in items { if let BlockItem::Statement(s) = item { - self.collect_cases_from_stmt(s, case_values, has_default, unsigned); + self.collect_cases_from_stmt(s, case_values, has_default); } } } Stmt::DoWhile { body, .. } | Stmt::While { body, .. } | Stmt::For { body, .. } => { - self.collect_cases_from_stmt(body, case_values, has_default, unsigned); + self.collect_cases_from_stmt(body, case_values, has_default); } Stmt::If { then_stmt, else_stmt, .. } => { - self.collect_cases_from_stmt(then_stmt, case_values, has_default, unsigned); + self.collect_cases_from_stmt(then_stmt, case_values, has_default); if let Some(e) = else_stmt { - self.collect_cases_from_stmt(e, case_values, has_default, unsigned); + self.collect_cases_from_stmt(e, case_values, has_default); } } // Stop at inner switch — its case labels belong to it @@ -2606,20 +2644,27 @@ impl<'a> super::linearize::Linearizer<'a> { &mut self, switch_val: PseudoId, cmp_type: TypeId, + conv: CaseConv, case_values: &[(i128, i128)], case_bbs: &[BasicBlockId], default_target: BasicBlockId, ) { let size = self.types.size_bits(cmp_type); - let unsigned = self.types.is_unsigned(cmp_type); - // `>=` and `<=` for a range, in the controlling type's own signedness. - let (ge, le) = if unsigned { + // `>=` and `<=` for a range, in the controlling type's own signedness + // -- the same `conv` that converted the labels, so the comparison and + // the constants it compares are describing one type. + let (ge, le) = if conv.unsigned() { (Opcode::SetAe, Opcode::SetBe) } else { (Opcode::SetGe, Opcode::SetLe) }; for (&(lo, hi), &case_bb) in case_values.iter().zip(case_bbs.iter()) { + debug_assert_eq!( + (conv.convert(lo), conv.convert(hi)), + (lo, hi), + "a case label reaches lowering already converted to the controlling type" + ); let Some(from) = self.current_bb else { return }; let next = self.alloc_bb(); let cond = if lo == hi { @@ -2699,20 +2744,18 @@ impl<'a> super::linearize::Linearizer<'a> { ) { match stmt { Stmt::Case(expr, high, body) => { - // Find the matching case block. A label is identified by its - // whole range, so that `case 1 ... 3:` and a later `case 1:` - // could not resolve to the same block -- the overlap check - // rejects that pair anyway, but matching on the low endpoint - // alone would have made the two indistinguishable here. - if let Some(val) = self.eval_const_expr(expr) { - // Matched at full width, as the collector records them. - let lo = val; + // Find the matching case block. The endpoints are the label's + // raw constants; `CaseIndex::lookup` converts them to the + // promoted controlling type with the very conversion the + // collector used, which is what keeps this lookup from missing + // and dropping the case body into the wrong block. + if let Some(lo) = self.eval_const_expr(expr) { let hi = match high { None => Some(lo), Some(hi_expr) => self.eval_const_expr(hi_expr), }; let Some(hi) = hi else { return }; - if let Some(&idx) = case_values.get(&(lo, hi)) { + if let Some(idx) = case_values.lookup(lo, hi) { let case_bb = case_bbs[idx]; // Fall through from previous case if not terminated @@ -4107,8 +4150,121 @@ impl AddrWalk<'_> { } } -/// Each case range's position among a switch's labels. -pub(crate) type CaseIndex = std::collections::HashMap<(i128, i128), usize>; +/// The promoted type of a switch's controlling expression: the width and the +/// signedness in which C17 6.8.4.2 says every case label lives. +/// +/// p5 converts each case constant to that type and p3 forbids two of them +/// being equal *after* the conversion, so a label's converted value is the +/// only one the rest of the switch path may see. Two places have to agree +/// about it -- the collector that records a label's range, and the body walk +/// that looks that same range back up to find the block it was given. If they +/// disagreed the lookup would simply miss, leaving the case body emitted into +/// the wrong block with nothing diagnosed. +/// +/// One value carries the whole conversion so they cannot disagree: +/// [`CaseSet`] owns the `CaseConv`, [`CaseSet::insert`] and +/// [`CaseSet::overlap`] convert what they are handed, [`CaseIndex::of`] copies +/// the conversion out of the set it indexes, and [`CaseIndex::lookup`] -- the +/// only way into the map -- converts too. Conversion is idempotent, so a +/// caller that has already converted for its own reasons stays in step. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub(crate) struct CaseConv { + /// Width of the promoted controlling type, in bits. + bits: u32, + /// Whether that type is unsigned. + unsigned: bool, +} + +impl CaseConv { + pub(crate) fn new(bits: u32, unsigned: bool) -> Self { + Self { bits, unsigned } + } + + /// The conversion a `switch` whose promoted controlling type is `typ` + /// applies to its labels. + pub(crate) fn of(types: &TypeTable, typ: TypeId) -> Self { + Self::new(types.size_bits(typ), types.is_unsigned(typ)) + } + + pub(crate) fn unsigned(self) -> bool { + self.unsigned + } + + /// `v` converted to this type, per C17 6.3.1.3: its low `bits` bits, read + /// back with the type's own signedness. + /// + /// The result is carried the way the whole switch path carries a value -- + /// an `i128` holding the type's bit pattern -- so a 128-bit unsigned label + /// above `i128::MAX` stays negative here and is ordered by [`Self::lt`] + /// rather than by Rust's signed `<`. + pub(crate) fn convert(self, v: i128) -> i128 { + if self.bits == 0 || self.bits >= 128 { + return v; + } + let shift = 128 - self.bits; + let truncated = ((v as u128) << shift) >> shift; + if self.unsigned { + truncated as i128 + } else { + ((truncated << shift) as i128) >> shift + } + } + + /// `a < b` in this type's signedness. + pub(crate) fn lt(self, a: i128, b: i128) -> bool { + if self.unsigned { + (a as u128) < (b as u128) + } else { + a < b + } + } + + /// Whether the converted range `lo..=hi` holds the converted value `v`. + /// + /// The constant-selector lowering picks its one edge with this, and has to + /// ask in the switch's signedness: a plain `i128` test read `case -1:` in + /// a `switch` on `unsigned` as a huge lower bound and never selected it. + pub(crate) fn contains(self, lo: i128, hi: i128, v: i128) -> bool { + !self.lt(v, lo) && !self.lt(hi, v) + } +} + +/// Each case range's position among a switch's labels, keyed by the range as +/// the controlling type sees it. +pub(crate) struct CaseIndex { + conv: CaseConv, + by_range: std::collections::HashMap<(i128, i128), usize>, +} + +impl CaseIndex { + /// Index the labels `set` collected, carrying `set`'s own conversion so + /// that a lookup converts exactly as the insert did. + /// + /// The first label of a repeated range wins, as a scan in source order + /// would find it; a duplicate has already been reported. + pub(crate) fn of(set: &CaseSet) -> Self { + let mut by_range = std::collections::HashMap::new(); + for (idx, range) in set.ranges().iter().enumerate() { + by_range.entry(*range).or_insert(idx); + } + Self { + conv: set.conv(), + by_range, + } + } + + /// The position of the label written `lo ... hi`, whose endpoints are the + /// raw constants as the label spells them. + /// + /// A label is identified by its whole range, so that `case 1 ... 3:` and a + /// later `case 1:` cannot resolve to the same block -- the overlap check + /// rejects that pair anyway, but matching on the low endpoint alone would + /// make the two indistinguishable here. + pub(crate) fn lookup(&self, lo: i128, hi: i128) -> Option { + let key = (self.conv.convert(lo), self.conv.convert(hi)); + self.by_range.get(&key).copied() + } +} /// A switch's case ranges, in source order, with an index that finds an /// overlap in logarithmic time. @@ -4117,43 +4273,56 @@ pub(crate) type CaseIndex = std::collections::HashMap<(i128, i128), usize>; /// quadratic in its case count: 70,000 labels took five seconds to compile /// and gcc's `limits-caselabels` eleven. pub(crate) struct CaseSet { - /// The ranges `(lo, hi)`, in the order the labels were written. + /// The ranges `(lo, hi)`, converted to the controlling type, in the order + /// the labels were written. ranges: Vec<(i128, i128)>, /// Each range by its low end, as an order-preserving key, to its high end. by_lo: std::collections::BTreeMap, - unsigned: bool, + /// What every endpoint entering the set is converted by. + conv: CaseConv, } impl CaseSet { - fn new(unsigned: bool) -> Self { + fn new(conv: CaseConv) -> Self { Self { ranges: Vec::new(), by_lo: std::collections::BTreeMap::new(), - unsigned, + conv, } } + pub(crate) fn conv(&self) -> CaseConv { + self.conv + } + + /// The ranges, converted, in source order. Parallel to the case blocks. + pub(crate) fn ranges(&self) -> &[(i128, i128)] { + &self.ranges + } + /// `v` as a signed key ordered the way the switch's type orders it: an /// unsigned value has its top bit flipped, which maps unsigned order onto /// signed order. fn key(&self, v: i128) -> i128 { - if self.unsigned { + if self.conv.unsigned { v ^ i128::MIN } else { v } } - /// An earlier range sharing a value with `lo..=hi`, if any. + /// An earlier range sharing a value with `lo..=hi`, if any, as converted. /// /// The ranges recorded are disjoint -- an overlap is an error -- so the /// only candidate is the one starting last at or before `hi`. fn overlap(&self, lo: i128, hi: i128) -> Option<(i128, i128)> { + let (lo, hi) = (self.conv.convert(lo), self.conv.convert(hi)); let (_, &(hi_key, lo2, hi2)) = self.by_lo.range(..=self.key(hi)).next_back()?; (hi_key >= self.key(lo)).then_some((lo2, hi2)) } fn insert(&mut self, lo: i128, hi: i128) { + let (lo, hi) = (self.conv.convert(lo), self.conv.convert(hi)); self.ranges.push((lo, hi)); let (lo_key, hi_key) = (self.key(lo), self.key(hi)); self.by_lo.insert(lo_key, (hi_key, lo, hi)); @@ -4162,11 +4331,127 @@ impl CaseSet { #[cfg(test)] mod case_set_tests { - use super::CaseSet; + use super::{CaseConv, CaseIndex, CaseSet}; + + /// A 128-bit conversion is the identity, which is what the ranges below + /// want: they are about ordering, not about width. + fn wide(unsigned: bool) -> CaseConv { + CaseConv::new(128, unsigned) + } + + /// C17 6.8.4.2p5 converts a case constant to the promoted controlling + /// type: the low bits, read back with that type's signedness. + #[test] + fn convert_takes_the_low_bits_with_the_types_signedness() { + let int = CaseConv::new(32, false); + let uint = CaseConv::new(32, true); + + // In range: unchanged either way. + assert_eq!(int.convert(7), 7); + assert_eq!(uint.convert(7), 7); + + // 2^32 is zero in 32 bits -- the label that silently became `case 0:`. + assert_eq!(int.convert(4294967296), 0); + assert_eq!(uint.convert(4294967296), 0); + + // -1 keeps its value as `int` and is the largest `unsigned int`. + assert_eq!(int.convert(-1), -1); + assert_eq!(uint.convert(-1), 4294967295); + + // The boundary of the signed range wraps the way C says. + assert_eq!(int.convert(2147483648), -2147483648); + assert_eq!(uint.convert(2147483648), 2147483648); + + // Narrower and wider types, and the 128-bit identity. + assert_eq!(CaseConv::new(8, false).convert(255), -1); + assert_eq!(CaseConv::new(8, true).convert(-1), 255); + assert_eq!(CaseConv::new(64, true).convert(-1), u64::MAX as i128); + assert_eq!(CaseConv::new(128, true).convert(-1), -1); + assert_eq!(CaseConv::new(128, false).convert(i128::MIN), i128::MIN); + } + + /// Converting is idempotent, which is what lets the collector convert for + /// its own diagnostics and still hand the set and the index raw or + /// converted endpoints interchangeably. + #[test] + fn convert_is_idempotent() { + for conv in [ + CaseConv::new(8, false), + CaseConv::new(16, true), + CaseConv::new(32, false), + CaseConv::new(64, true), + CaseConv::new(128, true), + ] { + for v in [0, 1, -1, 255, 4294967296, i128::MIN, i128::MAX] { + let once = conv.convert(v); + assert_eq!(conv.convert(once), once, "{conv:?} {v}"); + } + } + } + + /// The constant-selector lowering asks in the switch's own signedness. + #[test] + fn contains_tests_the_range_in_the_switch_signedness() { + let uint = CaseConv::new(32, true); + let big = uint.convert(-1); // 4294967295 + assert!(uint.contains(big, big, big)); + assert!(!uint.contains(big, big, 0)); + assert!(uint.contains(0, big, 5)); + + let int = CaseConv::new(32, false); + assert!(int.contains(-1, -1, -1)); + assert!(int.contains(-5, 5, 0)); + assert!(!int.contains(-5, 5, 6)); + // A signed test would read the unsigned bound as below zero. + assert!(!int.contains(0, 10, big)); + } + + /// The two-site invariant: what the collector inserts is exactly what the + /// body walk finds, even though the walk looks the label up by the + /// constant as written rather than as converted. + #[test] + fn the_index_finds_a_label_by_its_unconverted_constant() { + let conv = CaseConv::new(32, false); + let mut set = CaseSet::new(conv); + set.insert(0, 0); + set.insert(-1, -1); + set.insert(70000, 70005); + // Stored converted, and 2^32+3 is 3 in an `int` switch. + set.insert(4294967299, 4294967299); + assert_eq!(set.ranges(), [(0, 0), (-1, -1), (70000, 70005), (3, 3)]); + + let index = CaseIndex::of(&set); + assert_eq!(index.lookup(0, 0), Some(0)); + assert_eq!(index.lookup(-1, -1), Some(1)); + assert_eq!(index.lookup(70000, 70005), Some(2)); + // Looked up as written, found as converted. + assert_eq!(index.lookup(4294967299, 4294967299), Some(3)); + assert_eq!(index.lookup(3, 3), Some(3)); + assert_eq!(index.lookup(9, 9), None); + + // And a label the controlling type sees as negative. + let uconv = CaseConv::new(32, true); + let mut uset = CaseSet::new(uconv); + uset.insert(-1, -1); + assert_eq!(uset.ranges(), [(4294967295, 4294967295)]); + let uindex = CaseIndex::of(&uset); + assert_eq!(uindex.lookup(-1, -1), Some(0)); + assert_eq!(uindex.lookup(4294967295, 4294967295), Some(0)); + } + + /// Two labels that differ before the conversion collide after it, which is + /// the duplicate C17 6.8.4.2p3 forbids. + #[test] + fn overlap_sees_the_converted_values() { + let mut set = CaseSet::new(CaseConv::new(32, false)); + set.insert(0, 0); + assert_eq!(set.overlap(4294967296, 4294967296), Some((0, 0))); + assert_eq!(set.overlap(1, 1), None); + } #[test] fn overlap_finds_the_range_sharing_a_value() { - let mut set = CaseSet::new(false); + let mut set = CaseSet::new(wide(false)); set.insert(-10, -5); set.insert(0, 0); set.insert(10, 20); @@ -4186,7 +4471,7 @@ mod case_set_tests { #[test] fn overlap_orders_by_the_switch_type_signedness() { let big = u128::MAX as i128; // -1 as i128, the largest unsigned value - let mut set = CaseSet::new(true); + let mut set = CaseSet::new(wide(true)); set.insert(1, 5); set.insert(big - 10, big); assert_eq!(set.overlap(big - 3, big - 3), Some((big - 10, big))); diff --git a/cc/tests/c89/control_flow.rs b/cc/tests/c89/control_flow.rs index b310d96d2..b3c5e5500 100644 --- a/cc/tests/c89/control_flow.rs +++ b/cc/tests/c89/control_flow.rs @@ -1060,3 +1060,55 @@ int main(void) 0 ); } + +/// A `case` label is converted to the promoted type of the controlling +/// expression, and matching happens after that conversion. +/// +/// C17 6.8.4.2p5: the constant expression of each `case` is converted to the +/// promoted type of the controlling expression. c17 evaluated labels at full +/// width and never converted them, so a label outside the controlling type +/// matched or missed depending on which lowering saw it -- a runtime selector +/// kept the label in the `switch` instruction, while the constant-selector fast +/// path compared at 128 bits with a signed test that ignored the switch's +/// signedness. The same switch answered differently depending on whether its +/// selector was a constant. +#[test] +fn c89_a_case_label_is_converted_to_the_controlling_type() { + let code = r#" +/* 4294967296 is 2^32: zero when converted to int. */ +int runtime_sel(int x) { switch (x) { case 4294967296LL: return 1; default: return 2; } } +int const_sel(void) { switch (0) { case 4294967296LL: return 1; default: return 2; } } + +/* -1 converted to unsigned int is 4294967295. */ +int unsigned_runtime(unsigned x) { switch (x) { case -1: return 1; default: return 2; } } +int unsigned_const(void) { switch (4294967295u) { case -1: return 1; default: return 2; } } + +/* A label that converts without changing value still behaves. */ +int plain(int x) { switch (x) { case -1: return 1; case 7: return 3; default: return 2; } } + +/* Short controlling expression: promoted to int, so the label is too. */ +int shorty(short x) { switch (x) { case 65536 + 5: return 1; case 5: return 3; default: return 2; } } + +int main(void) +{ + /* Both lowerings must agree, and both must match. */ + if (runtime_sel(0) != 1) return 1; + if (const_sel() != 1) return 2; + + if (unsigned_runtime(4294967295u) != 1) return 3; + if (unsigned_const() != 1) return 4; + + if (plain(-1) != 1 || plain(7) != 3 || plain(0) != 2) return 5; + + /* 65541 converts to short's promoted int unchanged, so it cannot match 5. */ + if (shorty(5) != 3) return 6; + + return 0; +} +"#; + assert_eq!(compile_and_run("case_label_conversion", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("case_label_conversion_opt", code), + 0 + ); +} diff --git a/cc/tests/diagnostics/mod.rs b/cc/tests/diagnostics/mod.rs index a9e9aad39..25dff57a5 100644 --- a/cc/tests/diagnostics/mod.rs +++ b/cc/tests/diagnostics/mod.rs @@ -7434,3 +7434,52 @@ fn diagnostics_writing_an_array_member_of_a_const_object_is_rejected() { void g(struct S *p){ p->arr[0] = 2; }\n", ); } + +/// A `case` label whose conversion to the controlling type changes its value +/// is diagnosed, and two labels that become equal are a constraint violation. +/// +/// C17 6.8.4.2p5 converts each label to the promoted type of the controlling +/// expression, and p3 forbids two labels in one switch having the same value +/// *after* that conversion. c17 kept labels at full width, so it diagnosed +/// neither: `case 4294967296LL` in an `int` switch silently became `case 0`, +/// and sitting beside a real `case 0` it was silently accepted. +/// +/// gcc and clang both warn on the value-changing conversion and reject the +/// collision. +#[test] +fn diagnostics_a_case_label_outside_the_controlling_type_is_diagnosed() { + compile_expect_warning( + "case_label_overflow", + "int f(int x){ switch(x){ case 4294967296LL: return 1; default: return 2; } }\n", + "case", + ); + compile_expect_error( + "case_label_duplicate_after_conversion", + "int f(int x){ switch(x){ case 0: return 1; case 4294967296LL: return 2; } return 0; }\n", + "duplicate", + ); +} + +/// The other direction: a conversion that preserves the value is silent, so +/// the check above cannot pass by warning about every label. +/// +/// `case -1` in a `switch` on `unsigned` converts to 4294967295 and genuinely +/// matches it -- the conversion is value-changing in representation but well +/// defined and intended, which is why gcc and clang say nothing here either. +#[test] +fn diagnostics_a_case_label_inside_the_controlling_type_is_silent() { + compile_expect_no_diagnostic( + "case_label_in_range", + "int f(int x){ switch(x){ case -1: return 1; case 7: return 3; default: return 2; } }\n", + "case", + ); + compile_expect_no_diagnostic( + "case_label_negative_in_unsigned", + "int f(unsigned x){ switch(x){ case -1: return 1; default: return 2; } }\n", + "case", + ); + compile_expect_ok( + "case_labels_distinct_after_conversion", + "int f(int x){ switch(x){ case 0: return 1; case 1: return 2; default: return 3; } }\n", + ); +} From c54f27e13691504bdc8f1cd923d3124758a7a027 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 06:25:14 -0400 Subject: [PATCH 13/17] cc: ask one question about whether a division traps `range.rs` states the rule: c17 does not assume undefined behaviour away, because that makes -O0 and -O2 disagree about a program which really does divide by zero. `eval_divmod` honoured it for a zero divisor and then folded the other trap the same instruction raises. On x86-64 `idiv` raises #DE for a zero divisor and for `INT_MIN / -1`, whose quotient is not representable, and the remainder form traps identically because it is the same instruction. So `int a = INT_MIN, b = -1; a / b` faulted at -O0 and printed -2147483648 at -O2, `a % b` printed 0, and `0 / x` with `x` zero at run time printed 0 where -O0 faulted. The rule was stated three different ways, which is how it drifted: `eval_divmod` knew both values and asked about one of them, `range` knows sets and asks whether one contains zero, and `instcombine`'s algebraic arms knew one operand and did not reason about the other at all. It is now one predicate over optional operands -- unknown meaning "may trap" -- in `constfold.rs`, which is where the README already says these rules live. `eval_divmod` asks it and folds nothing when the answer is yes, which covers `instcombine`'s constant folding, `sccp` and `vrp` together, all three reaching it through `eval_binop`. `simplify_div` and `simplify_mod` ask it before their algebraic arms; that makes `0 / x` and `0 % x` unreachable, so those arms are deleted rather than left behind a new guard -- `0 / c` for a known `c` is ordinary folding, and an unknown divisor is what the predicate refuses. `x / 1` and `x % 1` cannot trap and still fold. `range`'s `contains(0)` refusal is unchanged and now says it is the same question lifted from a value to a set. gcc and clang fold all of these. This is a deliberate divergence, and the predicate says so where someone would otherwise assume a bug-for-bug match was intended. `Lsr` at 128 bits folded as an *arithmetic* shift, being correct at narrower widths only because the width-narrowing helper had already cleared the high half, and at 128 it returns the value unchanged. It shifts the unsigned view now, so it is the logical shift at every width. This is hardening rather than a repair: `arch::mapping` runs before the optimizer and either expands a 128-bit shift with a constant count into 64-bit halves or rewrites the count into a `pair64` the constant map cannot answer, so no foldable `lsr.128` reaches the optimizer by either route. A test pins the arithmetic anyway, and the old code fails it. The tests assert that the division is still emitted rather than that the program faults: the property is that it survives to run. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/constfold.rs | 318 ++++++++++++++++++++++++++++- cc/ir/instcombine.rs | 34 ++- cc/ir/range.rs | 16 +- cc/tests/codegen/mod.rs | 1 + cc/tests/codegen/trapping_folds.rs | 134 ++++++++++++ 5 files changed, 485 insertions(+), 18 deletions(-) create mode 100644 cc/tests/codegen/trapping_folds.rs diff --git a/cc/ir/constfold.rs b/cc/ir/constfold.rs index a83d0ad73..a466a00db 100644 --- a/cc/ir/constfold.rs +++ b/cc/ir/constfold.rs @@ -339,20 +339,91 @@ pub(crate) fn eval_binop(insn: &Instruction, a: i128, b: i128) -> Option { } } +/// The most negative value at `size` bits, read as signed: the one dividend +/// whose quotient by -1 is not representable. +/// +/// `size == 0` and `size >= 128` both answer `i128::MIN`, matching +/// [`at_width`], which leaves a value alone at those widths. +fn signed_min_at(size: u32) -> i128 { + if size == 0 || size >= 128 { + i128::MIN + } else { + -1i128 << (size - 1) + } +} + +/// Whether `op` -- one of `DivS`/`DivU`/`ModS`/`ModU` -- computed at `size` +/// bits over these operands may raise a hardware trap. `None` is an operand +/// that is not known, which may be anything, and so may trap. +/// +/// This is the one place the rule lives, because the three passes that fold a +/// division each know a different amount about the operands and the rule drifted +/// apart between them. `eval_divmod` knows both values; `instcombine`'s +/// algebraic arms know one and nothing about the other; `range::udiv`/`umod` +/// know sets rather than values, and answer this question of a set by asking +/// whether zero is in it (the signed trap has no unsigned counterpart). +/// +/// Two operand pairs trap, and they are the two `idiv` raises #DE for: +/// +/// * a zero divisor, in either signedness; and +/// * the single signed overflow, `INT_MIN / -1`, whose quotient is not +/// representable. The remainder form traps with it, because on x86-64 it is +/// the same instruction, computing the same quotient. +/// +/// c17 does not assume this undefined behaviour away. Folding either one turns +/// a program that faults at `-O0` into one that prints an answer at `-O2`, and +/// the two disagreeing about the same source is worse than either answer. +/// gcc and clang both fold these, treating the undefined behaviour as licence; +/// this is a deliberate divergence from both rather than a bug-for-bug match. +/// +/// The operands are raw: the narrowing this needs is applied here, and +/// narrowing is idempotent, so a caller that has already read them at their own +/// width may pass those instead. +pub(crate) fn divmod_may_trap(op: Opcode, size: u32, a: Option, b: Option) -> bool { + let signed = match op { + Opcode::DivS | Opcode::ModS => true, + Opcode::DivU | Opcode::ModU => false, + // Nothing else in this IR traps on its operands. + _ => return false, + }; + let size = size.max(1); + + // An unknown divisor may be zero. + let Some(b) = b.map(|v| at_width(v, size, signed)) else { + return true; + }; + if b == 0 { + return true; + } + if !signed || b != -1 { + return false; + } + // Divisor -1: the trap turns on whether the dividend is the most negative + // value, so an unknown dividend may trap. + match a.map(|v| at_width(v, size, true)) { + Some(a) => a == signed_min_at(size), + None => true, + } +} + /// Division and remainder. /// /// Read at the operand's own width, in the signedness the opcode implies. /// Division is not congruent modulo 2^n the way add/sub/mul are: it reads the /// whole value and its sign, so `(int)0xFFFFFFFFu` arriving as 4294967295 /// rather than -1 answered 2147483647 where C says 0. +/// +/// An operation that traps is not folded: see [`divmod_may_trap`]. That single +/// refusal covers `instcombine`, `sccp` and `vrp` at once, since all three +/// route their constant folding through [`eval_binop`]. fn eval_divmod(insn: &Instruction, a: i128, b: i128) -> Option { let signed = matches!(insn.op, Opcode::DivS | Opcode::ModS); let size = insn.size.max(1); - let a = at_width(a, size, signed); - let b = at_width(b, size, signed); - if b == 0 { + if divmod_may_trap(insn.op, size, Some(a), Some(b)) { return None; } + let a = at_width(a, size, signed); + let b = at_width(b, size, signed); let folded = match (insn.op, signed) { (Opcode::DivS, _) => a.wrapping_div(b), (Opcode::DivU, _) => (a as u128).wrapping_div(b as u128) as i128, @@ -376,7 +447,22 @@ fn eval_shift(insn: &Instruction, a: i128, b: i128) -> Option { // The result is truncated back, so an overflowing shift wraps at the // operand width rather than growing into the i128. Opcode::Shl => at_width(at_width(a, size, true).wrapping_shl(b as u32), size, true), - Opcode::Lsr => at_width(a, size, false).wrapping_shr(b as u32), + // Shifted in the unsigned view, which is what makes it the logical + // shift. Doing it on the `i128` instead only agrees below 128 bits, + // where `at_width` has already cleared the high half: at 128 bits + // `at_width` hands the value back unchanged and `i128::wrapping_shr` + // is the arithmetic shift, so a negative operand would shift in ones. + // Unreachable as the passes stand: `arch::mapping` runs before the + // optimizer, and it leaves no 128-bit shift whose count this can read + // -- a literally constant count expands the shift into 64-bit halves, + // and any other count is rewritten into a `Pair64` the constant map + // cannot answer. Written correctly anyway, so that the arm does not + // depend on that pass ordering for its answer. + Opcode::Lsr => at_width( + (at_width(a, size, false) as u128).wrapping_shr(b as u32) as i128, + size, + false, + ), Opcode::Asr => at_width(a, size, true).wrapping_shr(b as u32), _ => return None, }) @@ -589,4 +675,228 @@ mod tests { } assert_eq!(fcmp_against_constant(Opcode::SetGt, inf, false), None); } + + const DIVMOD: [Opcode; 4] = [Opcode::DivS, Opcode::DivU, Opcode::ModS, Opcode::ModU]; + + fn is_signed(op: Opcode) -> bool { + matches!(op, Opcode::DivS | Opcode::ModS) + } + + /// A zero divisor traps in either signedness, at every width, whatever the + /// dividend is -- including when the dividend is itself unknown. + #[test] + fn divmod_may_trap_refuses_every_zero_divisor() { + for op in DIVMOD { + for size in [8, 16, 32, 64, 128] { + for a in [None, Some(0), Some(1), Some(-1), Some(i128::MAX)] { + assert!( + divmod_may_trap(op, size, a, Some(0)), + "{op:?}.{size} {a:?} / 0" + ); + } + // The zero may also arrive unnarrowed: 2^size reads as zero at + // `size` bits, and it is the narrowed value that divides. + if size < 128 { + assert!( + divmod_may_trap(op, size, Some(1), Some(1i128 << size)), + "{op:?}.{size} 1 / 2^{size}" + ); + } + } + } + } + + /// An operand that is not known may be anything, so it may be the zero + /// divisor, or the `INT_MIN` dividend that overflows against -1. + #[test] + fn divmod_may_trap_refuses_an_unknown_operand() { + for op in DIVMOD { + // An unknown divisor, whatever the dividend. + assert!(divmod_may_trap(op, 32, Some(7), None), "{op:?} 7 / x"); + assert!(divmod_may_trap(op, 32, Some(0), None), "{op:?} 0 / x"); + assert!(divmod_may_trap(op, 32, None, None), "{op:?} y / x"); + + // An unknown dividend only matters against -1, and only when the + // opcode is a signed one -- for `DivU`/`ModU`, -1 at 32 bits is + // 4294967295, an ordinary divisor. + assert_eq!( + divmod_may_trap(op, 32, None, Some(-1)), + is_signed(op), + "{op:?} x / -1" + ); + assert!(!divmod_may_trap(op, 32, None, Some(3)), "{op:?} x / 3"); + } + } + + /// The one signed overflow, at every width it exists at, and only for the + /// signed opcodes. + #[test] + fn divmod_may_trap_refuses_the_signed_overflow() { + for size in [8, 16, 32, 64, 128] { + let min = signed_min_at(size); + for op in DIVMOD { + // Only the signed opcodes: the unsigned reading of the same + // bits is a large positive dividend divided by a larger one, + // which is 0 and cannot trap. + assert_eq!( + divmod_may_trap(op, size, Some(min), Some(-1)), + is_signed(op), + "{op:?}.{size} MIN / -1" + ); + } + } + assert_eq!(signed_min_at(8), -128); + assert_eq!(signed_min_at(32), -2147483648); + assert_eq!(signed_min_at(64), i64::MIN as i128); + assert_eq!(signed_min_at(128), i128::MIN); + } + + /// The dividend is read at its own width first, so `INT_MIN` written as + /// the unsigned pattern 2147483648 is still the overflowing dividend, and + /// a 64-bit `INT_MIN` divided at 32 bits is not. + #[test] + fn divmod_may_trap_reads_the_dividend_at_its_width() { + assert!(divmod_may_trap( + Opcode::DivS, + 32, + Some(2147483648), + Some(-1) + )); + assert!(divmod_may_trap( + Opcode::DivS, + 32, + Some(-2147483648), + Some(-1) + )); + // -2^31 at 64 bits is an ordinary negative number, not `LONG_MIN`. + assert!(!divmod_may_trap( + Opcode::DivS, + 64, + Some(-2147483648), + Some(-1) + )); + // The divisor, likewise: 4294967295 at 32 bits signed is -1. + assert!(divmod_may_trap( + Opcode::DivS, + 32, + Some(-2147483648), + Some(4294967295) + )); + } + + /// The safe neighbours of both traps still fold, which is what stops the + /// rule from becoming "never fold a division". + #[test] + fn divmod_may_trap_allows_the_safe_neighbours() { + for op in DIVMOD { + for (a, b) in [ + (12, 4), + (13, 4), + (0, 4), // 0 / c, with the divisor known + (-2147483647, -1), // one above the overflowing dividend + (-2147483648, 1), // the dividend, but not against -1 + (-2147483648, -2), + (1, -1), + (i128::from(i32::MAX), -1), + ] { + assert!( + !divmod_may_trap(op, 32, Some(a), Some(b)), + "{op:?} {a} / {b} cannot trap" + ); + } + // `x / 1` and `x % 1` fold with the dividend unknown. + assert!(!divmod_may_trap(op, 32, None, Some(1)), "{op:?} x / 1"); + } + } + + /// Nothing else in this IR trips the divide trap, so nothing else is asked + /// to answer for it. + #[test] + fn divmod_may_trap_answers_only_for_division() { + for op in [Opcode::Add, Opcode::Mul, Opcode::Shl, Opcode::Lsr] { + assert!(!divmod_may_trap(op, 32, Some(1), Some(0)), "{op:?}"); + assert!(!divmod_may_trap(op, 32, None, None), "{op:?}"); + } + } + + fn binop_at(op: Opcode, size: u32) -> Instruction { + Instruction::new(op).with_size(size) + } + + /// `eval_binop` is the one door `instcombine`, `sccp` and `vrp` fold + /// through, so the refusal has to be visible from there. + #[test] + fn eval_binop_does_not_fold_a_trapping_division() { + for (op, size) in [ + (Opcode::DivS, 32), + (Opcode::ModS, 32), + (Opcode::DivS, 64), + (Opcode::ModS, 64), + ] { + let insn = binop_at(op, size); + assert_eq!(eval_binop(&insn, 1, 0), None, "{op:?}.{size} 1 / 0"); + assert_eq!( + eval_binop(&insn, signed_min_at(size), -1), + None, + "{op:?}.{size} MIN / -1" + ); + } + for op in [Opcode::DivU, Opcode::ModU] { + assert_eq!(eval_binop(&binop_at(op, 32), 1, 0), None, "{op:?} 1 / 0"); + } + } + + /// And still gives the answers it gave, at the truncation C requires. + #[test] + fn eval_binop_still_folds_a_safe_division() { + for (op, a, b, want) in [ + (Opcode::DivS, 12, 4, 3), + (Opcode::ModS, 13, 4, 1), + (Opcode::DivS, -13, 4, -3), + (Opcode::ModS, -13, 4, -1), + (Opcode::DivS, 13, -4, -3), + (Opcode::ModS, 13, -4, 1), + (Opcode::DivS, -2147483647, -1, 2147483647), + (Opcode::DivS, -2147483648, 1, -2147483648), + (Opcode::ModS, -2147483648, 1, 0), + (Opcode::DivU, 4294967295, 5, 858993459), + (Opcode::ModU, 4294967295, 5, 0), + ] { + assert_eq!( + eval_binop(&binop_at(op, 32), a, b), + Some(want), + "{op:?} {a} op {b}" + ); + } + } + + /// `Lsr` is the logical shift at every width, the 128-bit one included, + /// where the value is not narrowed first and an `i128` shift would be the + /// arithmetic one. No pass reaches this today (`arch::mapping` decomposes + /// every 128-bit shift before the optimizer runs), so this pins the + /// evaluation rather than a pass's behaviour. + #[test] + fn eval_shift_lsr_is_logical_at_every_width() { + assert_eq!( + eval_binop(&binop_at(Opcode::Lsr, 128), -1, 4), + Some((u128::MAX >> 4) as i128), + "-1 >>u 4 at 128 bits fills with zeros" + ); + assert_eq!( + eval_binop(&binop_at(Opcode::Lsr, 128), i128::MIN, 127), + Some(1), + "the sign bit shifts down to bit 0" + ); + assert_eq!( + eval_binop(&binop_at(Opcode::Asr, 128), -1, 4), + Some(-1), + "`Asr` is still the arithmetic shift" + ); + // The narrower widths, which were already right, are unchanged. + assert_eq!(eval_binop(&binop_at(Opcode::Lsr, 32), -1, 28), Some(15)); + assert_eq!(eval_binop(&binop_at(Opcode::Asr, 32), -1, 28), Some(-1)); + // An out-of-range count is undefined and is not folded, at 128 as + // anywhere else. + assert_eq!(eval_binop(&binop_at(Opcode::Lsr, 128), -1, 128), None); + } } diff --git a/cc/ir/instcombine.rs b/cc/ir/instcombine.rs index 7fa52de89..7efcc352f 100644 --- a/cc/ir/instcombine.rs +++ b/cc/ir/instcombine.rs @@ -25,9 +25,9 @@ // use super::constfold::{ - at_width, cmp_operand_width, eval_binop, eval_fbinop, eval_fcvt, eval_fcvtf, eval_fternop, - eval_funop, eval_unop, fcmp_decided, fcmp_mask, fcmp_outcome, get_cmp_info, mirror_mask, - possible_against, result_type_of, FCMP_ALL, + at_width, cmp_operand_width, divmod_may_trap, eval_binop, eval_fbinop, eval_fcvt, eval_fcvtf, + eval_fternop, eval_funop, eval_unop, fcmp_decided, fcmp_mask, fcmp_outcome, get_cmp_info, + mirror_mask, possible_against, result_type_of, FCMP_ALL, }; use super::facts::{CmpDomain, CmpFacts, ConstMap, Relation}; use super::{ConstValue, Function, Instruction, Opcode, PseudoId}; @@ -401,16 +401,23 @@ fn simplify_div(insn: &Instruction, consts: &ConstMap) -> Simplification { let val1 = consts.get(src1).map(|v| at_width(v, size, signed)); let val2 = consts.get(src2).map(|v| at_width(v, size, signed)); + // A division that can trap is left to trap: `constfold::divmod_may_trap` + // states the rule and both of the operand pairs it covers. This is the + // whole of what stands between an unknown divisor and a folded answer -- + // `0 / x` used to fold to zero here without ever asking what `x` was, and + // a divisor of -1 is only safe once the dividend is known not to be the + // most negative value. + if divmod_may_trap(insn.op, size, val1, val2) { + return Simplification::None; + } + match (val1, val2) { - // Constant folding: a / b -> (a / b) (avoid div by zero) + // Constant folding: a / b -> (a / b) (Some(a), Some(b)) => fold_with(insn, a, b), // Algebraic: x / 1 -> x (None, Some(1)) => Simplification::CopyFrom(src1), - // Algebraic: 0 / x -> 0 - (Some(0), None) => fold_to_zero(), - _ => Simplification::None, } } @@ -430,13 +437,18 @@ fn simplify_mod(insn: &Instruction, consts: &ConstMap) -> Simplification { let val1 = consts.get(src1).map(|v| at_width(v, size, signed)); let val2 = consts.get(src2).map(|v| at_width(v, size, signed)); + // The remainder traps exactly where the division does -- on x86-64 it is + // the same `idiv`, computing the same quotient -- so it asks the same + // question. `0 % x` folded to zero here for an unknown `x`, and + // `INT_MIN % -1` folded to zero through `fold_with`. + if divmod_may_trap(insn.op, size, val1, val2) { + return Simplification::None; + } + match (val1, val2) { - // Constant folding: a % b -> (a % b) (avoid mod by zero) + // Constant folding: a % b -> (a % b) (Some(a), Some(b)) => fold_with(insn, a, b), - // Algebraic: 0 % x -> 0 - (Some(0), None) => fold_to_zero(), - // Algebraic: x % 1 -> 0 (None, Some(1)) => fold_to_zero(), diff --git a/cc/ir/range.rs b/cc/ir/range.rs index 28c806973..5edf42617 100644 --- a/cc/ir/range.rs +++ b/cc/ir/range.rs @@ -558,9 +558,16 @@ impl Range { /// Unsigned division. A divisor range containing zero answers `Full`: /// c17 does not assume undefined behaviour away, and - /// `constfold::eval_divmod` already refuses a zero divisor rather than - /// inventing a result. Disagreeing here would make `-O2` and `-O0` - /// differ on a program that really does divide by zero. + /// `constfold::divmod_may_trap` -- the one statement of which operand + /// pairs trap -- refuses a zero divisor rather than inventing a result. + /// Disagreeing here would make `-O2` and `-O0` differ on a program that + /// really does divide by zero. + /// + /// `contains(0)` is that predicate asked of a set rather than a value: + /// "may any divisor in this range trap?". For the unsigned forms that is + /// the whole of it, because the other trapping pair, `INT_MIN / -1`, is a + /// signed overflow with no unsigned counterpart -- and `vrp` gives the + /// signed opcodes `Range::full` rather than asking here at all. pub(crate) fn udiv(&self, other: &Range) -> Range { if let Some(r) = self.binary_guard(other) { return r; @@ -576,6 +583,9 @@ impl Range { Range::inclusive(self.width, amin / bmax, amax / bmin) } + /// Unsigned remainder, refusing a divisor range containing zero for the + /// reason [`Range::udiv`] gives: the remainder form is the same trapping + /// instruction. pub(crate) fn umod(&self, other: &Range) -> Range { if let Some(r) = self.binary_guard(other) { return r; diff --git a/cc/tests/codegen/mod.rs b/cc/tests/codegen/mod.rs index 51c55c32f..e727a6169 100644 --- a/cc/tests/codegen/mod.rs +++ b/cc/tests/codegen/mod.rs @@ -32,3 +32,4 @@ mod scaling; mod sections; mod stacked_args; mod tls_models; +mod trapping_folds; diff --git a/cc/tests/codegen/trapping_folds.rs b/cc/tests/codegen/trapping_folds.rs new file mode 100644 index 000000000..d4f0b700e --- /dev/null +++ b/cc/tests/codegen/trapping_folds.rs @@ -0,0 +1,134 @@ +// +// Copyright (c) 2025-2026 Jeff Garzik +// +// This file is part of the posixutils-rs project covered under +// the MIT License. For the full license text, please see the LICENSE +// file in the root directory of this project. +// SPDX-License-Identifier: MIT +// +// Operations that trap are not folded away. +// +// `cc/ir/range.rs` states the rule this file guards: c17 does not assume +// undefined behaviour away, because doing so makes `-O0` and `-O2` disagree +// about a program that really does divide by zero. `eval_divmod` already +// refuses a zero divisor for exactly that reason -- and then folded the other +// trap the same instruction raises. +// +// On x86-64 `idiv` raises #DE for a zero divisor *and* for the one signed +// overflow, `INT_MIN / -1`, whose quotient is not representable; the remainder +// form traps identically, because it is the same instruction. So folding +// `INT_MIN / -1` to `INT_MIN`, or `0 / x` to `0` without knowing `x`, turns a +// program that faults at `-O0` into one that prints an answer at `-O2`. +// +// gcc and clang both fold these -- they treat the undefined behaviour as +// licence. c17's stated policy is the opposite, so these tests are written +// against the policy rather than against another compiler. +// +// They assert on the emitted instruction rather than on a fault: the point is +// that the division survives to run, and a test that asserts a crash is worse +// at saying so. +// + +use crate::codegen::asm_probe::{asm_for_with, body_of, AARCH64_LINUX, X86_64_LINUX}; +use crate::common::compile_and_run; + +/// How many divide instructions the body of `func` contains. +fn divisions(asm: &str, func: &str) -> usize { + body_of(asm, func) + .lines() + .filter(|l| { + let t = l.trim(); + t.starts_with("idiv") + || t.starts_with("div") + || t.starts_with("sdiv") + || t.starts_with("udiv") + }) + .count() +} + +/// The two traps `idiv` raises are both left alone. +#[test] +fn codegen_a_trapping_division_is_not_folded() { + let src = "\ +int ovf_div(void) { int a = -2147483647 - 1, b = -1; return a / b; } +int ovf_mod(void) { int a = -2147483647 - 1, b = -1; return a % b; } +long ovf_div_long(void) { long a = -9223372036854775807L - 1, b = -1; return a / b; } +int zero_div(int x) { return 0 / x; } +int zero_mod(int x) { return 0 % x; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("trap_fold", triple, src, &["-O2"]); + for func in ["ovf_div", "ovf_mod", "ovf_div_long", "zero_div", "zero_mod"] { + assert!( + divisions(&asm, func) >= 1, + "{func} traps at -O0 and must still divide at -O2 on {triple}, \ + not be folded to an answer:\n{}", + body_of(&asm, func) + ); + } + } +} + +/// The control: a division that cannot trap is still folded. +/// +/// Without this the test above is satisfied by never folding a division at +/// all, which would cost every program that divides by a constant. +/// +/// `x / 8` is deliberately not here: with `x` unknown there is nothing to fold, +/// and c17 has no divide-by-constant strength reduction, so it emits a division +/// for reasons that have nothing to do with trapping. +#[test] +fn codegen_a_safe_division_is_still_folded() { + let src = "\ +int by_one(int x) { return x / 1; } +int mod_one(int x) { return x % 1; } +int both_known(void) { return 12 / 4; } +int both_known_mod(void) { return 13 % 4; } +int neg_but_safe(void) { int a = -2147483647, b = -1; return a / b; } +"; + for triple in [X86_64_LINUX, AARCH64_LINUX] { + let asm = asm_for_with("safe_fold", triple, src, &["-O2"]); + for func in [ + "by_one", + "mod_one", + "both_known", + "both_known_mod", + "neg_but_safe", + ] { + assert_eq!( + divisions(&asm, func), + 0, + "{func} cannot trap, so it is still folded on {triple}:\n{}", + body_of(&asm, func) + ); + } + } +} + +/// The answers the folder does give are unchanged. +#[test] +fn codegen_constant_division_still_computes_the_right_answer() { + let code = r#" +int main(void) +{ + if (12 / 4 != 3) return 1; + if (13 % 4 != 1) return 2; + if (-13 / 4 != -3) return 3; + if (-13 % 4 != -1) return 4; + if (13 / -4 != -3) return 5; + if (13 % -4 != 1) return 6; + + /* The largest magnitudes that do not overflow. */ + if ((-2147483647 - 1) / 1 != -2147483647 - 1) return 7; + if (-2147483647 / -1 != 2147483647) return 8; + if ((-2147483647 - 1) % -1 != 0 && 0) return 9; + + unsigned u = 4294967295u; + if (u / 5u != 858993459u) return 10; + if (u % 5u != 0u) return 11; + + return 0; +} +"#; + assert_eq!(compile_and_run("const_division_answers", code, &[]), 0); +} From 18ee7ed07f66731d4d845c960db065bba101b42a Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 07:23:30 -0400 Subject: [PATCH 14/17] cc: give the caller the bytes of an inlined aggregate return, not its address Two places decide whether a register-returned aggregate's `Ret` carries the value or the address of the callee's copy, and they disagreed. `emit_two_reg_return` emits the address form for three classes -- x87, an HFA, and one SSE register at sixteen bytes -- while `returns_addr_aggregate`, which sets the flag telling the inliner not to splice across that boundary, named only the first two. So `struct { __float128 a; }` emitted an address and was reported as carrying a value, and the inliner fed the `symaddr` straight into the phi: %16 = symaddr.64 %11(@mk_inline0_r.2) %17 = phisrc.128 %16 The caller then stored those eight bytes into a sixteen-byte slot and read the other eight from wherever the frame happened to leave them, so inlining changed the answer: `leaq -96(%rbp), %rax; movq %r10, -64(%rbp); movq -56(%rbp), %rax` where the un-inlined call had `movups %xmm0, -64(%rbp)`. It is the two-register path's neighbour, and both are the same idea -- the callee returns an aggregate in registers, so the caller's result local receives its bytes -- so the inliner gains that third branch and copies through `block_chunks` rather than phi-ing anything. `aggregate_ret_is_address` is now the one statement of which classes those are, including the size bound that keeps an eight-byte `struct { float a, float b; }` on the ordinary value path, and both sites ask it. Nothing else at sixteen bytes in one register was affected: a union of `__float128`, a nested struct holding one, a sixteen-byte float vector, `__int128`, `struct { double, double }`, `struct { float[4] }`, `struct { double[3] }`, `struct { long double }` and `struct { float, float }` all splice their value. `_Decimal128` is not implemented. The shape cannot be built for Darwin, where `__float128` is rejected, which is why nothing covered it. The test compiles for an explicit target, reads the optimized IR, and asserts no phi source is a pseudo that a `symaddr` defined -- a phi source is a value and an address is not. Its companion keeps the callee inlined, so the fix cannot be to stop inlining the shape. Inlining is still refused for x87 and HFA returns. Lifting it was measured: sixteen-byte HFAs splice correctly, but a twenty-four or thirty-two byte one still phis its address, because `returns_reg_aggregate` is capped at 128 bits at the definition site and those returns carry no classification for the new branch to read. Attaching it to every register-returned aggregate is a separate change, and the comment at the refusal records that. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/inline.rs | 209 ++++++++++++++++++++++++++++++++-- cc/ir/linearize.rs | 102 ++++++++--------- cc/ir/mod.rs | 108 ++++++++++++++++++ cc/tests/codegen/cross_abi.rs | 100 ++++++++++++++++ 4 files changed, 457 insertions(+), 62 deletions(-) diff --git a/cc/ir/inline.rs b/cc/ir/inline.rs index 0a3da12fd..56caea403 100644 --- a/cc/ir/inline.rs +++ b/cc/ir/inline.rs @@ -449,9 +449,11 @@ struct InlineContext { ret_typ: Option, /// Size (in bits) captured from the first cloned `Ret`. ret_size: u32, - /// Pseudos allocated as PhiSource targets in the cloned Ret blocks. - /// Added to the caller alongside other inlined pseudos. - phisrc_pseudos: Vec, + /// Pseudos allocated while lowering a cloned `Ret`: the PhiSource target + /// of a value return, and the temporaries that carry an aggregate handed + /// back by address into the result local. Added to the caller alongside + /// the other inlined pseudos. + ret_pseudos: Vec, /// Value pseudos created while cloning -- a constant's value lives on its /// pseudo, not on the instruction, so resolving /// `__builtin_va_arg_pack_len()` makes one. Added to the caller with the @@ -511,7 +513,7 @@ impl InlineContext { ret_arms: Vec::new(), ret_typ: None, ret_size: 0, - phisrc_pseudos: Vec::new(), + ret_pseudos: Vec::new(), const_pseudos: Vec::new(), } } @@ -659,10 +661,12 @@ fn clone_instruction( // matching Phi is materialized by `inline_call_site` after all blocks // are cloned. // - // For a two-register struct return, both halves are stored to the - // result local's memory (a Sym pseudo). The Sym itself remains - // single-defined (it is the local's address); the stores are - // side-effecting writes to memory and do not violate SSA. + // An aggregate returned in registers takes neither path: the call's + // result slot is a local, and the returned bytes are written into it. + // Two-register form, both halves are stored there; address form + // (`returns_aggregate_address`), the bytes are copied there. The Sym + // itself remains single-defined (it is the local's address); the + // stores are side-effecting writes to memory and do not violate SSA. Opcode::Ret => { let mut result = Vec::new(); @@ -711,6 +715,61 @@ fn clone_instruction( ); store_high.pos = insn.pos; result.push(store_high); + } else if insn.returns_aggregate_address() { + // The callee hands back the *address* of the + // aggregate -- one SSE register holding sixteen + // bytes, an x87 aggregate, an HFA -- because no pair + // of general registers can carry it (see + // `aggregate_ret_is_address`). A call leaves the + // value in the result local and the caller reads it + // from there, so the spliced return has to put the + // bytes there itself. + // + // Asked of the `Ret`'s own classification, which is + // what `returns_two_regs` just above asks and all + // this pass has: it carries no `TypeTable` and cannot + // classify anything itself. Only the one-SSE shape + // reaches here today -- `Function::ret_is_address` + // still keeps an x87 aggregate and an HFA out of the + // inliner entirely, for the reason recorded where it + // is set. + // + // Phi-ing the source instead handed the caller a + // pointer where the value belonged: + // + // %16 = symaddr.64 %11(@mk_inline0_r.2) + // %17 = phisrc.128 %16 + // + // -- a 64-bit address as a 128-bit value. Inlining + // changed the answer, and only for this shape: the + // two-register return stores its halves just above, + // and an aggregate of eight bytes or less never + // reaches `emit_two_reg_return` at all, so its `Ret` + // already carries a loaded value. + // + // `insn.size` is the aggregate's own width, in the + // 8/4/2/1 chunks every other block move in the + // compiler uses, so a size that is not a multiple of + // eight is copied exactly rather than rounded up past + // either object. No bound is needed: a register + // return is at most a four-element HFA, thirty-two + // bytes. The access type stays the `Ret`'s own, as + // the two-register stores above keep theirs -- this + // pass has no `TypeTable` to name a qword with. + let src_addr = ctx.remap_pseudo(*ret_val, callee_func); + let typ = insn.typ.unwrap_or(crate::types::TypeId::INVALID); + for (offset, chunk) in memexpand::block_chunks(i64::from(insn.size / 8)) { + let temp = ctx.alloc_pseudo_id(); + ctx.ret_pseudos.push(Pseudo::undef(temp)); + let mut load = + Instruction::load(temp, src_addr, offset, typ, chunk.bits()); + load.pos = insn.pos; + result.push(load); + let mut store = + Instruction::store(temp, target, offset, typ, chunk.bits()); + store.pos = insn.pos; + result.push(store); + } } else { // Single-value return: emit PhiSource in the predecessor // and record the arm for `inline_call_site` to assemble @@ -724,7 +783,7 @@ fn clone_instruction( before cloning a Ret", ); let phisrc_target = ctx.alloc_pseudo_id(); - ctx.phisrc_pseudos + ctx.ret_pseudos .push(Pseudo::phi(phisrc_target, phisrc_target.0)); let typ = insn.typ.unwrap_or(crate::types::TypeId::INVALID); @@ -1357,8 +1416,9 @@ fn inline_call_site( for pseudo in std::mem::take(&mut ctx.const_pseudos) { caller.replace_pseudo(pseudo); } - // Add PhiSource target pseudos generated for the return-value Phi. - for pseudo in std::mem::take(&mut ctx.phisrc_pseudos) { + // Add the pseudos the cloned returns allocated: PhiSource targets, and + // the temporaries of an aggregate copied into the result local. + for pseudo in std::mem::take(&mut ctx.ret_pseudos) { if !caller.has_pseudo(pseudo.id) { caller.add_pseudo(pseudo); } @@ -2556,6 +2616,133 @@ mod tests { ); } + /// `static struct Q mk(void) { struct Q r; ...; return r; }`, whose `Ret` + /// hands the aggregate back by *address* under `ret` -- the shape + /// `emit_two_reg_return` emits for a class no pair of general registers + /// can carry. + fn aggregate_address_ret_callee( + types: &TypeTable, + ret: crate::abi::ArgClass, + size_bits: u32, + ) -> Function { + let mut callee = Function::new("mk", types.void_id); + callee.is_static = true; + let mut bb = BasicBlock::new(BasicBlockId(0)); + bb.insns.push(Instruction::new(Opcode::Entry)); + // The address of the callee's own local, which is what the `Ret` + // carries: `%1 = symaddr %0(@r)`. + bb.insns.push(Instruction::sym_addr( + PseudoId(1), + PseudoId(0), + types.long_id, + )); + let mut ret_insn = Instruction::ret_typed(Some(PseudoId(1)), types.long_id, size_bits); + ret_insn.abi_info = Some(Box::new(crate::ir::CallAbiInfo::new(vec![], ret))); + bb.insns.push(ret_insn); + callee.add_block(bb); + callee.entry = BasicBlockId(0); + callee.add_pseudo(Pseudo::sym(PseudoId(0), "r.0".to_string())); + callee.add_pseudo(Pseudo::reg(PseudoId(1), 1)); + callee.next_pseudo = 2; + callee + } + + /// An inlined return of an aggregate handed back by address copies the + /// aggregate's *bytes* into the call's result local, and phis nothing. + /// + /// A call leaves the value in that local and the caller reads it from + /// there. Feeding the `Ret`'s source into the continuation phi instead + /// gave the caller the address of the callee's own copy -- for + /// `struct { __float128 a; }`, one SSE register and sixteen bytes, the + /// post-opt IR read `%17 = phisrc.128 %16` where `%16` was a + /// `symaddr.64`. The caller then stored a pointer into the first eight + /// bytes of a sixteen-byte slot and read the other eight uninitialized, + /// so inlining changed the answer. + /// + /// The bytes move in `memexpand::block_chunks`, so a width that is not a + /// multiple of eight -- a three-`float` HFA is twelve bytes -- lands + /// exactly rather than reaching past either object. + #[test] + fn test_inlined_aggregate_address_return_copies_the_value() { + let types = TypeTable::new(&Target::host()); + let cases = [ + ( + "struct { __float128 a; }: one SSE register, sixteen bytes", + crate::abi::ArgClass::Direct { + classes: vec![crate::abi::RegClass::Sse], + size_bits: 128, + }, + 128u32, + vec![ + (Opcode::Load, 0, 64), + (Opcode::Store, 0, 64), + (Opcode::Load, 8, 64), + (Opcode::Store, 8, 64), + ], + ), + ( + "struct { float a, b, c; } as an HFA: twelve bytes", + crate::abi::ArgClass::Hfa { + base: crate::abi::HfaBase::Float32, + count: 3, + }, + 96, + vec![ + (Opcode::Load, 0, 64), + (Opcode::Store, 0, 64), + (Opcode::Load, 8, 32), + (Opcode::Store, 8, 32), + ], + ), + ]; + + for (what, ret_class, size_bits, want) in cases { + let callee = aggregate_address_ret_callee(&types, ret_class, size_bits); + + // `struct Q v = mk();` -- the result local the backend would have + // written the returned registers into is the call's target. + let mut caller = Function::new("caller", types.void_id); + let mut cb = BasicBlock::new(BasicBlockId(0)); + cb.insns.push(Instruction::new(Opcode::Entry)); + cb.insns.push(Instruction::call( + Some(PseudoId(0)), + "mk", + vec![], + vec![], + types.long_id, + size_bits, + )); + cb.insns.push(Instruction::ret(None)); + caller.add_block(cb); + caller.entry = BasicBlockId(0); + caller.add_pseudo(Pseudo::sym(PseudoId(0), "__2reg_0".to_string())); + caller.next_pseudo = 1; + + assert!(inline_call_site(&mut caller, 0, 1, &callee), "{what}"); + + let insns: Vec<&Instruction> = + caller.blocks.iter().flat_map(|b| b.insns.iter()).collect(); + let moves: Vec<(Opcode, i64, u32)> = insns + .iter() + .filter(|i| matches!(i.op, Opcode::Load | Opcode::Store)) + .map(|i| (i.op, i.offset, i.size)) + .collect(); + assert_eq!(moves, want, "{what}"); + + // Every store writes the result local, and nothing phis the + // address: an address is not a value. + for store in insns.iter().filter(|i| i.op == Opcode::Store) { + assert_eq!(store.src.first(), Some(&PseudoId(0)), "{what}"); + } + assert!( + !insns + .iter() + .any(|i| matches!(i.op, Opcode::Phi | Opcode::PhiSource)), + "{what}: the returned aggregate is not phi-ed" + ); + } + } + /// Each inlined copy of a function that takes a label's address names /// its own clone of the block, in the caller. /// diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index 211ae2a83..c847df071 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -1572,32 +1572,48 @@ impl<'a> Linearizer<'a> { } // A `Ret` that carries an address; a call's result slot holds the - // value. The inliner has to know not to splice across that boundary. - // An aggregate returned in st(0) has exactly the same shape as a - // complex one, and missing it is a miscompile visible only at -O. + // value. Whoever consumes that return has to read the bytes out of the + // storage it names, and the inliner has to know which of the two it is + // splicing. `aggregate_ret_is_address` is the one place that answers + // it -- the same function `emit_two_reg_return` asks before emitting + // the address form, so the shape and the question about the shape + // cannot drift apart. They had: spelled out a second time here as + // "x87 or HFA", this missed a sixteen-byte aggregate returned in one + // SSE register, and before that it was gated behind the + // *two-register* path's 128-bit cap, which missed every HFA past + // sixteen bytes -- four `double`s is thirty-two bytes and still comes + // back in d0-d3. // - // Asked of the ABI classification directly rather than through - // `returns_reg_aggregate`, which is the *two-register* return path and - // so stops at 128 bits. An HFA comes back in registers at any size -- - // four `double`s is thirty-two bytes and still returns in d0-d3 -- so - // gating on that cap made every HFA past 128 bits report that its `Ret` - // carried a value. The inliner then spliced the body in and phi-ed the - // address as if it were the aggregate, and the caller read the pointer's - // own storage as the struct's bytes. The call-site half of this - // decision already has no size bound; the two had drifted. - let returns_addr_aggregate = (ret_kind == TypeKind::Struct || ret_kind == TypeKind::Union) - // An aggregate that fits in one register comes back *as* a value, - // so its `Ret` carries one; only past 64 bits is an address handed - // back. Dropping this bound along with the 128-bit cap refused to - // inline every HFA, including `struct { float x, y; }`, which was - // correct before and is the common aarch64 shape. - && struct_size_bits > 64 - && !returns_large_struct - && matches!( + // Classified only for an aggregate that is not going through the + // hidden pointer, which is the only shape the question is about. + let ret_class = ((ret_kind == TypeKind::Struct || ret_kind == TypeKind::Union) + && !returns_large_struct) + .then(|| { get_abi_for_conv(self.current_calling_conv, self.target) - .classify_return(func.return_type, self.types), - crate::abi::ArgClass::X87 { .. } | crate::abi::ArgClass::Hfa { .. } - ); + .classify_return(func.return_type, self.types) + }); + // Which of those the inliner is kept away from -- a narrower question + // than the shape, and no longer the same one. `clone_instruction`'s + // `Ret` arm now copies an aggregate handed back by address into the + // call's result local, which is where a call leaves it, so the + // one-SSE-register shape is spliced correctly instead of refused. + // + // An x87 aggregate and an HFA travel by address for the same reason + // and that copy would move them just as well, but they are not ready + // to be let through: the copy reads the `Ret`'s own ABI + // classification, and only `emit_two_reg_return` attaches one -- + // which `returns_reg_aggregate` above stops calling past 128 bits. So + // a three- or four-`double` HFA returns an address under no + // classification at all, and lifting this refusal has it phi-ed again + // (`%45 = phisrc.192 %44`, where `%44` is a `symaddr.64`). Letting + // those in means carrying the classification onto every + // register-returned aggregate's `Ret` first, which is its own change + // -- and `codegen_aarch64_hfa_returning_function_is_not_inlined` + // pins this refusal until then. + let returns_addr_aggregate = ret_class.as_ref().is_some_and(|class| { + super::aggregate_ret_is_address(class, struct_size_bits) + && !matches!(class, crate::abi::ArgClass::Direct { .. }) + }); ir_func.ret_is_address = self.types.is_complex(func.return_type) || returns_addr_aggregate; // Add parameters @@ -1908,33 +1924,17 @@ impl<'a> Linearizer<'a> { let abi = get_abi_for_conv(self.current_calling_conv, self.target); let ret_class = abi.classify_return(ret_type, self.types); - // An aggregate that is nothing but a `long double` comes back in - // st(0), exactly as the bare scalar does, so the `Ret` carries the - // value's *address* and the backend loads it onto the FPU stack. - // Splitting it across RAX and RDX left the caller reading a slot - // nobody had written. - // A single SSE register carrying sixteen bytes -- an aggregate whose - // sole content is a `__float128` -- is the same shape: the register - // holds the whole value, so the `Ret` carries its address and the - // backend moves all sixteen bytes at once. Splitting it into two - // general registers handed the caller half a value in the wrong place. - let one_sse_reg = matches!( - ret_class, - crate::abi::ArgClass::Direct { ref classes, .. } - if classes.len() == 1 && classes[0] == crate::abi::RegClass::Sse - ); - // A one-element HFA is the aarch64 spelling of the same thing: one V - // register holds the whole value. Splitting it into two general - // registers was survivable on its own -- the backend put the halves - // back together -- but the *inliner* then spliced a two-source `Ret` - // into a caller expecting one value, and the top half came out zero. - // Any HFA, not just a one-element one: the two-element form has the - // same hazard. Its `Ret` carried the halves as two general registers, - // and splicing that into a caller expecting one value dropped the - // second -- an inlined `struct { double a, b; }` return came back with - // its second half zeroed. - let one_hfa_reg = matches!(ret_class, crate::abi::ArgClass::Hfa { .. }); - if matches!(ret_class, crate::abi::ArgClass::X87 { .. }) || one_sse_reg || one_hfa_reg { + // The three classes no pair of general registers can carry: an x87 + // aggregate, an HFA, and sixteen bytes in one SSE register. Each hands + // the value back by address, and splitting any of them into RAX/RDX + // was a miscompile of its own -- an x87 aggregate left the caller + // reading a slot nobody had written, a `__float128` one handed a + // gcc-compiled caller half a value in the wrong place, and an HFA's + // two-source `Ret` spliced into a caller expecting one value dropped + // its second half. `aggregate_ret_is_address` is where that list + // lives, because the inliner has to ask the same question of the + // `Ret` this emits. + if super::aggregate_ret_is_address(&ret_class, struct_size) { let mut ret_insn = Instruction::ret_typed(Some(src_addr), ret_type, struct_size); ret_insn.abi_info = Some(Box::new(CallAbiInfo::new(vec![], ret_class))); self.emit(ret_insn); diff --git a/cc/ir/mod.rs b/cc/ir/mod.rs index 95299dc4c..b94ea5fc3 100644 --- a/cc/ir/mod.rs +++ b/cc/ir/mod.rs @@ -78,6 +78,44 @@ impl CallAbiInfo { } } +/// Whether an aggregate of `size_bits` returned in class `ret` is handed back +/// by *address*: the `Ret` carries a pointer to the value's storage rather +/// than the value, and whoever consumes the return has to read the bytes out. +/// +/// Three classifications answer yes, and they are exactly the three that no +/// pair of general registers can carry: +/// +/// * `X87` -- an aggregate that is nothing but a `long double` comes back in +/// st(0), which is loaded from memory because nothing else holds 80 bits. +/// * `Hfa` -- AAPCS64 returns a homogeneous floating-point aggregate in one +/// V register per element, at any size: four `double`s is thirty-two bytes +/// and still comes back in `d0`-`d3`. +/// * one `Sse` -- a single SSE register holding all sixteen bytes, which on +/// x86-64 is an aggregate whose sole content is a `__float128`: SSE+SSEUP +/// is *one* register, and splitting it into RAX/RDX hands a gcc-compiled +/// caller half a value in the wrong place. +/// +/// The size bound is part of the rule, not a caller's business: an aggregate +/// that fits in one register comes back *as* a value, so `struct { float a, +/// b; }` is `Direct { classes: [Sse] }` and yet carries its value. Only past +/// 64 bits is an address handed back. +/// +/// One function because the answer is asked in three places -- the return +/// emitter, the flag that tells the inliner what it is splicing, and the +/// inliner's own `Ret` lowering -- and two spellings of it had already +/// drifted: the flag omitted the one-SSE case, so a `struct { __float128 a; }` +/// return reported a value-carrying `Ret`, the inliner spliced the body in and +/// phi-ed the callee's local *address* as though it were the aggregate. +pub fn aggregate_ret_is_address(ret: &ArgClass, size_bits: u32) -> bool { + use crate::abi::RegClass; + size_bits > 64 + && match ret { + ArgClass::X87 { .. } | ArgClass::Hfa { .. } => true, + ArgClass::Direct { classes, .. } => classes.as_slice() == [RegClass::Sse], + _ => false, + } +} + // Instruction Reference - for def-use chains /// Reference to an instruction by (basic block id, instruction index) @@ -1579,6 +1617,21 @@ impl Instruction { .unwrap_or(false) } + /// True when this `Ret` hands back an aggregate by *address*: its source + /// is a pointer to the value's storage, not the value. + /// + /// Asked of the `Ret`'s own ABI classification, which is the only place + /// the answer is recorded -- `Instruction::size` is the aggregate's width, + /// so [`aggregate_ret_is_address`] can apply its own size bound without a + /// `TypeTable`. Only [`crate::ir::Linearizer::emit_two_reg_return`] ever + /// puts `abi_info` on a `Ret`, and only for a struct or union, so no + /// scalar reaches this. + pub fn returns_aggregate_address(&self) -> bool { + self.abi_info + .as_ref() + .is_some_and(|ai| aggregate_ret_is_address(&ai.ret, self.size)) + } + /// Convert this instruction to a no-op, clearing all operands. pub fn kill(&mut self) { self.op = Opcode::Nop; @@ -3584,6 +3637,61 @@ mod tests { assert!(insn.returns_two_regs()); } + /// The three classes whose `Ret` hands back an aggregate's *address*, and + /// the size bound that is part of the rule. + /// + /// `Direct { classes: [Sse] }` is the discriminating row: at sixteen bytes + /// it is one SSE register holding a whole `__float128`, so the `Ret` names + /// the storage; at eight it is `struct { float a, b; }`, which comes back + /// *as* a value and never reaches `emit_two_reg_return` at all. Answering + /// the first one "no" is what made the inliner phi an address as though it + /// were the aggregate. + #[test] + fn test_aggregate_ret_is_address() { + let sse = |n: usize, bits: u32| ArgClass::Direct { + classes: vec![RegClass::Sse; n], + size_bits: bits, + }; + assert!(aggregate_ret_is_address(&sse(1, 128), 128), "one SSE, 16B"); + assert!(!aggregate_ret_is_address(&sse(1, 64), 64), "one SSE, 8B"); + assert!( + !aggregate_ret_is_address(&sse(2, 128), 128), + "two SSE registers carry the halves, not an address" + ); + assert!( + !aggregate_ret_is_address( + &ArgClass::Direct { + classes: vec![RegClass::Integer, RegClass::Integer], + size_bits: 128, + }, + 128 + ), + "__int128 comes back in RAX/RDX" + ); + assert!(aggregate_ret_is_address( + &ArgClass::X87 { size_bits: 80 }, + 128 + )); + assert!(aggregate_ret_is_address( + &ArgClass::Hfa { + base: crate::abi::HfaBase::Float64, + count: 4, + }, + 256 + )); + assert!( + !aggregate_ret_is_address( + &ArgClass::Indirect { + align: 8, + size_bytes: 64, + }, + 512 + ), + "the hidden pointer is not this" + ); + assert!(!aggregate_ret_is_address(&ArgClass::Ignore, 0)); + } + // Function::create_reg_pseudo #[test] diff --git a/cc/tests/codegen/cross_abi.rs b/cc/tests/codegen/cross_abi.rs index 97f0ca1ec..44690cbe4 100644 --- a/cc/tests/codegen/cross_abi.rs +++ b/cc/tests/codegen/cross_abi.rs @@ -3592,3 +3592,103 @@ int main(void) "#; assert_eq!(compile_and_run("register_composite_params", code, &[]), 0); } + +/// The optimized IR of `src` for `target`, with inlining left on. +fn post_opt_ir_inlined(prefix: &str, src: &str, target: &str, func: &str) -> String { + let dir = plib::tmp::Builder::new() + .prefix(prefix) + .tempdir() + .expect("tempdir"); + let c = dir.path().join("t.c"); + std::fs::write(&c, src).expect("write source"); + let r = run_c17(&[ + "--target", + target, + "-O2", + "--dump-ir", + "post-opt", + "--dump-ir-func", + func, + "-S", + "-o", + "/dev/null", + c.to_str().unwrap(), + ]); + assert!(r.success, "compile failed: {}", r.stderr); + format!("{}{}", r.stdout, r.stderr) +} + +/// An aggregate returned in registers is spliced into its caller as its value, +/// not as the address of the callee's copy. +/// +/// The inliner replaces a `Ret` with a phi of the returned value. For a +/// register-returned aggregate it has to read that value out of the callee's +/// result local first. It did for the two-register case and for a one-register +/// aggregate of eight bytes, but a *sixteen*-byte aggregate returned in one SSE +/// register -- `struct { __float128 a; }` -- had its `symaddr` fed straight +/// into the phi, so the caller received the address where the value belonged: +/// +/// leaq -96(%rbp), %rax ; the callee's result local +/// movq %r10, -64(%rbp) ; stored into eight bytes of a sixteen-byte slot +/// movq -56(%rbp), %rax ; the other eight read uninitialized +/// +/// Inlining therefore changed the answer. Compiling for an explicit target is +/// what makes this testable at all: `__float128` is rejected on Darwin, so the +/// shape cannot be built for the host, and no test covered it. +#[test] +fn codegen_an_inlined_register_aggregate_return_is_a_value() { + let src = "\ +struct Q { __float128 a; }; +static struct Q mk(__float128 x) { struct Q r = {x}; return r; } +__float128 probe(__float128 x) { struct Q v = mk(x); return v.a; } +"; + let ir = post_opt_ir_inlined("inl_sse_ret", src, X86_64_LINUX, "probe"); + + // Every pseudo that holds an address rather than a value. + let addresses: Vec<&str> = ir + .lines() + .filter_map(|l| { + let t = l.trim(); + let (target, rest) = t.split_once(" = ")?; + rest.starts_with("symaddr").then_some(target) + }) + .collect(); + + // A phi source carries the returned value, so none of them may be one. + for line in ir.lines().map(str::trim).filter(|l| l.contains("phisrc")) { + for addr in &addresses { + assert!( + !line + .split_whitespace() + .any(|w| w.trim_end_matches(',') == *addr), + "the inlined return hands the caller {addr}, which is an address, \ + where the aggregate's value belongs:\n {line}\n\nfull IR:\n{ir}" + ); + } + } +} + +/// The control: the shapes that already worked must keep working, so the check +/// above cannot pass by the inliner declining to inline. +#[test] +fn codegen_inlined_aggregate_returns_still_inline() { + let src = "\ +struct F2 { float a, b; }; +struct F4 { float a, b, c, d; }; +static struct F2 mk2(float x) { struct F2 r = {x, x + 1}; return r; } +static struct F4 mk4(float x) { struct F4 r = {x, x+1, x+2, x+3}; return r; } +float probe2(float x) { struct F2 v = mk2(x); return v.a + v.b; } +float probe4(float x) { struct F4 v = mk4(x); return v.a + v.d; } +"; + for func in ["probe2", "probe4"] { + let ir = post_opt_ir_inlined("inl_agg_ok", src, X86_64_LINUX, func); + assert!( + ir.contains("_inline"), + "{func}'s callee must still be inlined, or the check above is vacuous:\n{ir}" + ); + assert!( + ir.lines().any(|l| l.contains("load")), + "{func} must read the returned aggregate's value:\n{ir}" + ); + } +} From 0a33e60cda0c17d8ef4fbe2581ec28f86528f9e9 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 08:29:46 -0400 Subject: [PATCH 15/17] cc: override a bit-field, and a member of the union already held The two shapes 33ff4e04 left at the old behaviour, both of which disagreed with gcc and one of which still had the static and automatic paths disagreeing with each other -- the defect class that commit set out to end. A bit-field's bytes are merged after the `Initializer` tree is built, so the static merge had nothing to descend into and dropped the earlier entry whole: `{ .t = {1,2}, .t.a = 3 }` lost `b`, while the automatic path overlaid the carrier and kept it. Nothing needs un-lowering, as it turns out -- by the time an entry reaches the merge a nested struct's bit-fields are already one byte apiece in its initializer, so the carrier *is* those bytes. `classify_subobject`'s refusal for a carrier window becomes an answer, `BitfieldCarrier`, naming the struct that declares the field, and the fold masks the field's bits out of the byte and the new ones in. The classifier and its bit-field form now share one descent, differing only in whether the range is an object or a span of bits: for bits an exact byte match is no reason to stop, since two fields can share one byte. `bitfield_carrier_bytes` is the one definition of that arithmetic, and the emitter calls it rather than repeating it. A union holds one member, so an override naming a member *of that member* keeps the rest of it: `{ .u = {1,2}, .u.p.y = 9 }` leaves `p.x` at 1. c17 reset the union, because the tree records no discriminant and the merge could not tell "the same member, deeper" from "a different member". It is threaded now rather than guessed: the designator records the member it crosses, an initializer-list walk records the member an earlier entry gave a value to, and a union is descended only where the two agree. Both paths take the discriminant from the same walk, and the reset case -- a *different* member named -- is unchanged, including for a union whose members have identical layouts, where no shape guess could have told them apart. Two more, found while probing and of the same kind: a whole-struct initializer after a bit-field one did not supersede it, and a bit-field initializer displacing a union did not reset it. So the paths agree by construction and not by coincidence: they share the classifier, the discriminant walk and the field walk, and differ only in whether they fold a tree or clear bytes before storing. Not fixed, and not this: an eight-byte local's 32-bit store at offset 0 is widened to 64 in the x86-64 store lowering, because the guard that spares a struct asks for more than 64 bits, so `struct S { struct P t; } l = { .t = {1,2}, .t.x = 3 }` reads back y as 0 at -O0. The tests here use objects wider than eight bytes deliberately, so that they test 6.7.9p19 and not that. Co-Authored-By: Claude Opus 5 (1M context) --- cc/ir/linearize.rs | 113 ++++ cc/ir/linearize_init.rs | 1083 ++++++++++++++++++++++++++++------ cc/ir/linearize_stmt.rs | 86 ++- cc/tests/c99/initializers.rs | 192 ++++++ 4 files changed, 1289 insertions(+), 185 deletions(-) diff --git a/cc/ir/linearize.rs b/cc/ir/linearize.rs index c847df071..7dbaab6c7 100644 --- a/cc/ir/linearize.rs +++ b/cc/ir/linearize.rs @@ -123,6 +123,99 @@ pub(crate) struct ResolvedDesignator { pub(crate) bit_offset: Option, pub(crate) bit_width: Option, pub(crate) access_bytes: Option, + /// The member this designator chain named in each union it passed + /// through. See [`UnionMembers`]. + pub(crate) unions: UnionMembers, +} + +/// Which member a union came to hold, for each union an initializer list +/// reached. +/// +/// C17 6.7.9p19 makes a later initializer override the earlier one for the +/// *same* subobject, and a union has only one subobject at a time: whether +/// `.u.p.y = 9` overrides part of what `.u = {1, 2}` wrote or replaces all of +/// it turns on whether the union still holds `p`. Neither the byte offset nor +/// the lowered [`Initializer`] can say -- every member of a union begins at +/// the same byte, and `Initializer` is what the emitter consumes and has no +/// room for a discriminant -- so the choice is recorded beside it. +/// +/// Each entry is the byte offset of a union within the object whose +/// initializer list produced it, that union's type, and the index of the +/// member in question. The type is part of the key because a union declared +/// directly inside another begins at the same byte as it does. +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub(crate) struct UnionMembers(Vec<(usize, TypeId, usize)>); + +impl UnionMembers { + /// Note that the union at `offset` holds `member`, replacing whatever it + /// was last said to hold. + pub(crate) fn record(&mut self, offset: usize, typ: TypeId, member: usize) { + match self.0.iter_mut().find(|e| (e.0, e.1) == (offset, typ)) { + Some(entry) => entry.2 = member, + None => self.0.push((offset, typ, member)), + } + } + + fn member_at(&self, offset: usize, typ: TypeId) -> Option { + self.0 + .iter() + .find(|e| (e.0, e.1) == (offset, typ)) + .map(|e| e.2) + } + + /// Take on everything `other` says, which is later and so decisive. + pub(crate) fn absorb(&mut self, other: &UnionMembers) { + for &(offset, typ, member) in &other.0 { + self.record(offset, typ, member); + } + } + + /// Forget every union starting inside `range`, whose contents some later + /// initializer has just discarded. + pub(crate) fn clear_range(&mut self, range: std::ops::Range) { + self.0.retain(|e| !range.contains(&e.0)); + } +} + +/// What the two initializers being merged say about the unions between them, +/// and how far into the earlier one's object the merge has descended. +/// +/// A union is descended into only where the two agree: `held` is the member +/// the earlier initializer gave a value to, `named` the member the later +/// one's designator reached through, and anything else means the union comes +/// to hold something different and everything it held goes. Both are keyed by +/// byte offset within the object whose initializer list holds both entries, +/// which is what `base` counts from. +#[derive(Clone, Copy, Default)] +pub(crate) struct UnionFold<'a> { + held: Option<&'a UnionMembers>, + named: Option<&'a UnionMembers>, + base: usize, +} + +impl<'a> UnionFold<'a> { + pub(crate) fn new(held: &'a UnionMembers, named: &'a UnionMembers, base: usize) -> Self { + Self { + held: Some(held), + named: Some(named), + base, + } + } + + /// The same view, `offset` bytes further into the object. + pub(crate) fn inside(self, offset: usize) -> Self { + Self { + base: self.base + offset, + ..self + } + } + + /// The member both sides agree the union of type `typ` at the current + /// offset holds, if they do. + pub(crate) fn agreed(&self, typ: TypeId) -> Option { + let held = self.held?.member_at(self.base, typ)?; + (self.named?.member_at(self.base, typ)? == held).then_some(held) + } } pub(crate) struct RawFieldInit { @@ -139,6 +232,10 @@ pub(crate) struct RawFieldInit { pub(crate) init: Initializer, pub(crate) bit_offset: Option, pub(crate) bit_width: Option, + /// The member each union inside this subobject came to hold. + pub(crate) held: UnionMembers, + /// The member each union this entry's designator passed through named. + pub(crate) named: UnionMembers, } impl RawFieldInit { @@ -175,6 +272,15 @@ pub(crate) enum SubobjectPlace { /// member at a time, so initializing through it discards whatever the /// union held before. ThroughUnion { offset: usize, size: usize }, + /// The range is the storage of a bit-field declared by the struct + /// spanning `offset..offset + size` of the enclosing object. A bit-field + /// is not addressable storage of its own -- it shares a carrier with its + /// neighbours -- so an initializer for one replaces its bits and leaves + /// theirs, which is done in that struct's own initializer rather than by + /// replacing a subobject. + /// + /// Only asked for, and only ever answered, for a bit-field's own bits. + BitfieldCarrier { offset: usize, size: usize }, /// The range is not a subobject at all: it straddles two members, or it is /// a bit-field carrier's window rather than a named object. NotASubobject, @@ -231,6 +337,13 @@ pub(crate) struct StructFieldVisit { pub(crate) bit_offset: Option, pub(crate) bit_width: Option, pub(crate) access_bytes: Option, + /// Index, in the member list walked, of the member this visit initializes + /// or lies inside. For a union that is the member it comes to hold, which + /// its byte offset cannot say. + pub(crate) member_index: Option, + /// The member this visit's designator chain named in each union it passed + /// through, by byte offset within the object being initialized. + pub(crate) unions: UnionMembers, } pub(crate) enum StructFieldVisitKind { diff --git a/cc/ir/linearize_init.rs b/cc/ir/linearize_init.rs index 7fdab1be8..2dc202919 100644 --- a/cc/ir/linearize_init.rs +++ b/cc/ir/linearize_init.rs @@ -48,6 +48,92 @@ pub(crate) fn is_const_object_type(types: &TypeTable, typ: TypeId) -> bool { /// converts to an integer type outside its range, as it does in gcc. const OVERFLOW_WARNING: &str = "overflow"; +/// The bytes a bit-field's own bits occupy, as `(byte offset from the field's +/// own offset, the bits `value` puts there, the mask of the bits the field +/// owns)`. +/// +/// Only the bytes the field's bits reach. Its access span is wider and +/// generally starts earlier -- `unsigned a:1` after a `char` sits at bit 8 of +/// a span based at byte 0 -- and writing the whole span would blank the +/// members sharing it. +fn bitfield_carrier_bytes( + bit_offset: u32, + bit_width: u32, + value: i128, +) -> impl Iterator { + // A field with no bits, or one no carrier could hold, occupies no byte. + // Otherwise the mask comes of shifting `u128::MAX` down rather than + // `1 << width` up, for the reason `bitfield_value_mask` records: the + // latter overflows at the carrier's own width. + let fits = bit_width > 0 && u64::from(bit_offset) + u64::from(bit_width) <= 128; + let (shift, width_mask) = if fits { + (bit_offset, u128::MAX >> (128 - bit_width)) + } else { + (0, 0) + }; + let placed = ((value as u128) & width_mask) << shift; + let owned = width_mask << shift; + let bytes = if fits { + (bit_offset / 8) as usize..((bit_offset + bit_width - 1) / 8) as usize + 1 + } else { + 0..0 + }; + bytes.map(move |byte| { + let shift = byte * 8; + ( + byte, + ((placed >> shift) & 0xff) as u8, + ((owned >> shift) & 0xff) as u8, + ) + }) +} + +/// The bytes a bit-field member's own bits occupy, measured from the first +/// byte of the struct that declares it. `None` for anything but a bit-field. +fn bitfield_byte_span(member: &crate::types::StructMember) -> Option> { + let (bit_offset, bit_width) = (member.bit_offset?, member.bit_width?); + let start = member.offset + (bit_offset / 8) as usize; + let end = member.offset + (bit_offset + bit_width).div_ceil(8) as usize; + Some(start..end.max(start + 1)) +} + +/// Give the carrier byte at `offset` of a struct initializer the bits `bits` +/// in place of the ones `mask` marks, leaving the neighbouring bit-fields +/// that share the byte as they are. +/// +/// False when the byte is not a carrier byte the lowering wrote -- an entry +/// for an ordinary member covering it, or something other than an integer -- +/// which leaves the caller to discard the earlier initializer whole. +fn replace_carrier_bits( + fields: &mut Vec<(usize, usize, Initializer)>, + offset: usize, + bits: u8, + mask: u8, +) -> bool { + match fields + .iter() + .position(|(off, size, _)| *off <= offset && offset < *off + *size) + { + Some(index) => { + if (fields[index].0, fields[index].1) != (offset, 1) { + return false; + } + let Initializer::Int(held) = fields[index].2 else { + return false; + }; + fields[index].2 = Initializer::Int(i128::from((held as u8 & !mask) | bits)); + true + } + None => { + // The byte holds no bit-field the earlier initializer gave a + // value to, so the bits around this field's are zero. + fields.push((offset, 1, Initializer::Int(i128::from(bits)))); + fields.sort_by_key(|(off, _, _)| *off); + true + } + } +} + /// Name an expression the way a C programmer would, for a diagnostic. /// /// The alternative these messages used was `{:?}` on the AST node, which @@ -1015,15 +1101,25 @@ impl<'a> super::linearize::Linearizer<'a> { if element.designators.is_empty() { // Positional: check anonymous struct continuation, then next member let mut member = None; + let mut member_index = None; if anon_cont.is_some() { + // A continuation is filling the anonymous aggregate at + // `outer_idx`, so that is the member this element lands in. + member_index = anon_cont.as_ref().map(|cont| cont.outer_idx); member = self.get_anon_continuation_member( &mut anon_cont, members, &mut current_field_idx, ); + if member.is_none() { + member_index = None; + } } if member.is_none() { - member = self.next_positional_member(members, is_union, &mut current_field_idx); + let picked = + self.next_positional_member(members, is_union, &mut current_field_idx); + member_index = picked.as_ref().map(|(_, index)| *index); + member = picked.map(|(member, _)| member); } let Some(member) = member else { elem_idx += 1; @@ -1048,6 +1144,8 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset: member.bit_offset, bit_width: member.bit_width, access_bytes: member.access_bytes, + member_index, + unions: UnionMembers::default(), }); continue; } @@ -1060,19 +1158,23 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset, bit_width, access_bytes, + unions, }) = resolved else { elem_idx += 1; continue; }; + let mut member_index = None; if let Some(Designator::Field(name)) = element.designators.first() { if let Some(result) = self.member_index_for_designator(members, *name) { match result { MemberDesignatorResult::Direct(next_idx) => { + member_index = Some(next_idx - 1); current_field_idx = next_idx; anon_cont = None; } MemberDesignatorResult::Anonymous { outer_idx, levels } => { + member_index = Some(outer_idx); current_field_idx = outer_idx; anon_cont = Some(AnonContinuation { outer_idx, levels }); } @@ -1088,6 +1190,8 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset, bit_width, access_bytes, + member_index, + unions, }); elem_idx += 1; } @@ -1095,6 +1199,124 @@ impl<'a> super::linearize::Linearizer<'a> { visits } + /// The member each union inside the subobject a visit initializes comes + /// to hold, by byte offset within the object being initialized. + /// + /// Answered from the initializer list rather than from the lowered + /// `Initializer`, so that the static and the automatic path get the same + /// answer from the same walk. Both need it to resolve a later initializer + /// naming something inside a union against this one: only where the union + /// still holds the same member does the rest of that member survive. + pub(crate) fn held_union_members( + &self, + typ: TypeId, + kind: &StructFieldVisitKind, + base: usize, + ) -> UnionMembers { + let mut held = UnionMembers::default(); + // The walk costs a second pass over the initializer, so it is only + // made where there is a union to find. + if self.type_holds_union(typ) { + self.record_held_unions_of(typ, kind, base, &mut held); + } + held + } + + /// Whether an object of type `typ` has a union anywhere inside it, itself + /// included. Pointers are not looked through, so this terminates on the + /// self-referential types C allows. + fn type_holds_union(&self, typ: TypeId) -> bool { + let typ = self.resolve_struct_type(typ); + match self.types.kind(typ) { + TypeKind::Union => true, + TypeKind::Struct => self + .types + .get(typ) + .composite + .as_ref() + .is_some_and(|composite| { + composite + .members + .iter() + .any(|member| self.type_holds_union(member.typ)) + }), + TypeKind::Array => self + .types + .base_type(typ) + .is_some_and(|elem| self.type_holds_union(elem)), + _ => false, + } + } + + fn record_held_unions( + &self, + typ: TypeId, + elements: &[InitElement], + base: usize, + held: &mut UnionMembers, + ) { + let typ = self.resolve_struct_type(typ); + match self.types.kind(typ) { + TypeKind::Struct | TypeKind::Union => { + let Some(composite) = self.types.get(typ).composite.as_ref() else { + return; + }; + let members = composite.members.clone(); + let is_union = self.types.kind(typ) == TypeKind::Union; + let visits = self.walk_struct_init_fields(typ, &members, is_union, elements); + if is_union { + // Each element of a union's initializer list replaces what + // the one before it put there, so the union ends up + // holding whichever member the last of them named. + if let Some(index) = visits.last().and_then(|visit| visit.member_index) { + held.record(base, typ, index); + } + } + for visit in &visits { + self.record_held_unions_of(visit.typ, &visit.kind, base + visit.offset, held); + } + } + TypeKind::Array => { + let Some(elem_type) = self.types.base_type(typ) else { + return; + }; + let elem_size = self.types.size_bytes(elem_type); + let groups = self.group_array_init_elements(elements, typ); + for index in groups.indices { + let Some(list) = groups.element_lists.get(&index) else { + continue; + }; + self.record_held_unions( + elem_type, + list, + base + index as usize * elem_size, + held, + ); + } + } + _ => {} + } + } + + fn record_held_unions_of( + &self, + typ: TypeId, + kind: &StructFieldVisitKind, + base: usize, + held: &mut UnionMembers, + ) { + match kind { + StructFieldVisitKind::BraceElision(elements) => { + self.record_held_unions(typ, elements, base, held); + } + StructFieldVisitKind::Expr(expr) => { + if let ExprKind::InitList { elements } = &expr.kind { + self.record_held_unions(typ, elements, base, held); + } + } + } + } + /// Where the byte range `offset..offset + size` sits inside an object of /// type `typ`, with `offset` measured from that object's first byte. /// @@ -1105,18 +1327,53 @@ impl<'a> super::linearize::Linearizer<'a> { /// names a different member of a union and so replaces the whole of what /// the first wrote. The spans are identical in shape -- one inside the /// other -- so only the type can tell the two cases apart. + /// + /// `unions` says which member each union on the way holds; a union it has + /// nothing to say about is answered [`SubobjectPlace::ThroughUnion`], + /// which discards its contents. pub(crate) fn classify_subobject( &self, typ: TypeId, offset: usize, size: usize, + unions: UnionFold<'_>, + ) -> SubobjectPlace { + self.classify_range(typ, offset, size, false, unions) + } + + /// The same question for the bytes a *bit-field's* own bits occupy. + /// + /// A bit-field is not addressable storage, so it is never a subobject of + /// its own and the answer wanted is the struct that declares it, whose + /// initializer holds the carrier the field shares with its neighbours. + /// An exact byte match is no reason to stop early: `struct { unsigned + /// char a : 4, b : 4; }` is one byte wide, and replacing it whole would + /// take `b` with it. + pub(crate) fn classify_bitfield_carrier( + &self, + typ: TypeId, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> SubobjectPlace { + self.classify_range(typ, offset, size, true, unions) + } + + fn classify_range( + &self, + typ: TypeId, + offset: usize, + size: usize, + bitfield: bool, + unions: UnionFold<'_>, ) -> SubobjectPlace { let mut typ = self.resolve_struct_type(typ); let mut base = 0usize; + let mut unions = unions; loop { let type_size = self.types.size_bytes(typ); - if offset == base && size == type_size { + if !bitfield && offset == base && size == type_size { return SubobjectPlace::Member; } if size == 0 || offset < base || offset + size > base + type_size { @@ -1125,15 +1382,22 @@ impl<'a> super::linearize::Linearizer<'a> { match self.types.kind(typ) { // Reached a union without having named it exactly, so the - // range lies inside one of its members. Which member is not - // decidable from the span -- every member starts at the same - // byte -- and it does not matter: whichever it is, giving it - // an initializer discards the member the union held before. + // range lies inside one of its members. Which member the union + // holds is not decidable from the span -- every member starts + // at the same byte -- so unless both initializers agree on it, + // giving this range an initializer discards what the union + // held before. TypeKind::Union => { - return SubobjectPlace::ThroughUnion { - offset: base, - size: type_size, + let Some(member) = self.agreed_union_member(typ, base, offset, size, unions) + else { + return SubobjectPlace::ThroughUnion { + offset: base, + size: type_size, + }; }; + base += member.0; + unions = unions.inside(member.0); + typ = self.resolve_struct_type(member.1); } TypeKind::Struct => { let Some(composite) = self.types.get(typ).composite.as_ref() else { @@ -1149,9 +1413,25 @@ impl<'a> super::linearize::Linearizer<'a> { } }); let Some(member) = found else { - return SubobjectPlace::NotASubobject; + // No ordinary member holds these bytes. For a + // bit-field's bits that is expected, and this is the + // struct that declares it. + let declares = bitfield + && composite.members.iter().any(|member| { + bitfield_byte_span(member) + .is_some_and(|span| self.holds(base, &span, offset, size)) + }); + return if declares { + SubobjectPlace::BitfieldCarrier { + offset: base, + size: type_size, + } + } else { + SubobjectPlace::NotASubobject + }; }; base += member.offset; + unions = unions.inside(member.offset); typ = self.resolve_struct_type(member.typ); } TypeKind::Array => { @@ -1166,6 +1446,7 @@ impl<'a> super::linearize::Linearizer<'a> { if offset + size > elem_start + elem_size { return SubobjectPlace::NotASubobject; } + unions = unions.inside(elem_start - base); base = elem_start; typ = self.resolve_struct_type(elem_type); } @@ -1177,8 +1458,45 @@ impl<'a> super::linearize::Linearizer<'a> { } } + /// Whether the member span `span`, measured from a struct beginning at + /// `base`, holds all of `offset..offset + size`. + fn holds( + &self, + base: usize, + span: &std::ops::Range, + offset: usize, + size: usize, + ) -> bool { + offset >= base + span.start && offset + size <= base + span.end + } + + /// The member a union of type `typ`, beginning at `base`, is agreed to + /// hold -- as `(its offset, its type)` -- when that member holds all of + /// `offset..offset + size`. + fn agreed_union_member( + &self, + typ: TypeId, + base: usize, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> Option<(usize, TypeId)> { + let index = unions.agreed(typ)?; + let member = self + .types + .get(typ) + .composite + .as_ref()? + .members + .get(index) + .filter(|member| member.bit_width.is_none())?; + let span = member.offset..member.offset + self.types.size_bytes(member.typ); + self.holds(base, &span, offset, size) + .then_some((member.offset, member.typ)) + } + /// An initializer that writes nothing, shaped for `typ` so that - /// [`Self::overlay_subobject`] can place entries into it. + /// [`Self::subobject_init_mut`] can place entries into it. fn empty_aggregate_init(&self, typ: TypeId) -> Option { match self.types.kind(typ) { TypeKind::Struct | TypeKind::Union => Some(Initializer::Struct { @@ -1213,141 +1531,162 @@ impl<'a> super::linearize::Linearizer<'a> { offset: usize, size: usize, new_init: &Initializer, + unions: UnionFold<'_>, ) -> bool { + match self.subobject_init_mut(typ, init, offset, size, unions) { + Some(slot) => { + *slot = new_init.clone(); + true + } + None => false, + } + } + + /// The initializer for the subobject at `offset..offset + size` of an + /// object of type `typ` initialized by `init`, to be read or replaced in + /// place. A subobject the initializer left out gains an empty one, so + /// that what is written there leaves the rest of its parent alone. + /// + /// `None` when the existing initializer's shape cannot express the + /// subobject -- a string literal standing for a character array, an entry + /// for a bit-field carrier rather than for a member -- in which case the + /// caller discards the earlier initializer whole. It may by then have + /// gained an empty initializer for a member on the way down, which is + /// harmless precisely because the whole of it is discarded. + fn subobject_init_mut<'i>( + &self, + typ: TypeId, + init: &'i mut Initializer, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> Option<&'i mut Initializer> { let typ = self.resolve_struct_type(typ); if offset == 0 && size == self.types.size_bytes(typ) { - *init = new_init.clone(); - return true; + return Some(init); } match self.types.kind(typ) { TypeKind::Struct | TypeKind::Union => { - let Some(composite) = self.types.get(typ).composite.as_ref() else { - return false; - }; - let found = composite - .members - .iter() - .find(|member| { - member.bit_width.is_none() && { - let member_end = member.offset + self.types.size_bytes(member.typ); - offset >= member.offset && offset + size <= member_end - } - }) - .map(|member| (member.offset, member.typ)); - let Some((member_offset, member_type)) = found else { - return false; + let (slot_offset, slot_type) = if self.types.kind(typ) == TypeKind::Union { + // Every member of a union begins at the same byte, so the + // range cannot say which one it is in. Only the member the + // two initializers agree the union holds will do. + self.agreed_union_member(typ, 0, offset, size, unions)? + } else { + self.types + .get(typ) + .composite + .as_ref()? + .members + .iter() + .find(|member| { + member.bit_width.is_none() && { + let span = member.offset + ..member.offset + self.types.size_bytes(member.typ); + self.holds(0, &span, offset, size) + } + }) + .map(|member| (member.offset, member.typ))? }; - let member_size = self.types.size_bytes(member_type); + let slot_size = self.types.size_bytes(slot_type); let Initializer::Struct { fields, .. } = init else { - return false; + return None; }; - self.overlay_into_entries( - fields, - member_offset, - member_size, - member_type, - offset, + let slot = Self::entry_init_mut(fields, slot_offset, slot_size, || { + self.empty_aggregate_init(slot_type) + })?; + self.subobject_init_mut( + slot_type, + slot, + offset - slot_offset, size, - new_init, + unions.inside(slot_offset), ) } TypeKind::Array => { - let Some(elem_type) = self.types.base_type(typ) else { - return false; - }; + let elem_type = self.types.base_type(typ)?; let elem_size = self.types.size_bytes(elem_type); if elem_size == 0 { - return false; + return None; } let elem_offset = (offset / elem_size) * elem_size; if offset + size > elem_offset + elem_size { - return false; + return None; } let Initializer::Array { elements, .. } = init else { - return false; + return None; }; - // An array's entries carry no width, so borrow the struct - // path's bookkeeping by giving each one the element width it - // implicitly has. - let mut entries: Vec<(usize, usize, Initializer)> = elements - .iter() - .map(|(off, init)| (*off, elem_size, init.clone())) - .collect(); - if !self.overlay_into_entries( - &mut entries, - elem_offset, - elem_size, + // An array's entries carry no width, so give each one the + // element width it implicitly has and borrow the struct + // path's bookkeeping. + let slot = Self::element_init_mut(elements, elem_offset, || { + self.empty_aggregate_init(elem_type) + })?; + self.subobject_init_mut( elem_type, - offset, + slot, + offset - elem_offset, size, - new_init, - ) { - return false; - } - *elements = entries - .into_iter() - .map(|(off, _, init)| (off, init)) - .collect(); - true + unions.inside(elem_offset), + ) } - _ => false, + _ => None, } } - /// Place `new_init` for the subobject at `offset..offset + size`, which - /// lies within the member or element at `slot_offset` of `slot_size` - /// bytes, into an entry list that holds one entry per initialized member. - #[allow(clippy::too_many_arguments)] - fn overlay_into_entries( - &self, - entries: &mut Vec<(usize, usize, Initializer)>, + /// The entry in a struct initializer's field list for the member at + /// `slot_offset` of `slot_size` bytes, added as an empty initializer if + /// the list has none. + /// + /// `None` when an entry overlaps the member without being exactly it -- a + /// bit-field carrier byte, or an entry spanning several members -- which + /// is not something a subobject can be placed into. + fn entry_init_mut( + fields: &mut Vec<(usize, usize, Initializer)>, slot_offset: usize, slot_size: usize, - slot_type: TypeId, - offset: usize, - size: usize, - new_init: &Initializer, - ) -> bool { + empty: impl FnOnce() -> Option, + ) -> Option<&mut Initializer> { let slot_end = slot_offset + slot_size; - let existing = entries + let index = match fields .iter() - .position(|(off, sz, _)| *off < slot_end && slot_offset < *off + *sz); - - if let Some(idx) = existing { - let (entry_offset, entry_size, entry_init) = &mut entries[idx]; - // Anything but a whole entry for exactly this member -- a - // bit-field carrier byte, or an entry spanning several members -- - // is not something this can descend into. - if (*entry_offset, *entry_size) != (slot_offset, slot_size) { - return false; + .position(|(off, sz, _)| *off < slot_end && slot_offset < *off + *sz) + { + Some(index) => { + if (fields[index].0, fields[index].1) != (slot_offset, slot_size) { + return None; + } + index } - return self.overlay_subobject( - slot_type, - entry_init, - offset - slot_offset, - size, - new_init, - ); - } + None => { + // A scalar member has no aggregate to make empty; whatever is + // put here is about to be replaced outright, since nothing + // lies inside a scalar to descend to. + fields.push((slot_offset, slot_size, empty().unwrap_or_default())); + fields.sort_by_key(|(off, _, _)| *off); + fields.iter().position(|(off, _, _)| *off == slot_offset)? + } + }; + Some(&mut fields[index].2) + } - // The member had no initializer of its own: give it one that writes - // zeros everywhere but the subobject being replaced. - let fresh = if (offset, size) == (slot_offset, slot_size) { - new_init.clone() - } else { - let Some(mut fresh) = self.empty_aggregate_init(slot_type) else { - return false; - }; - if !self.overlay_subobject(slot_type, &mut fresh, offset - slot_offset, size, new_init) - { - return false; + /// The same for an array initializer's element list, whose entries are + /// one element wide by construction. + fn element_init_mut( + elements: &mut Vec<(usize, Initializer)>, + elem_offset: usize, + empty: impl FnOnce() -> Option, + ) -> Option<&mut Initializer> { + let index = match elements.iter().position(|(off, _)| *off == elem_offset) { + Some(index) => index, + None => { + elements.push((elem_offset, empty().unwrap_or_default())); + elements.sort_by_key(|(off, _)| *off); + elements.iter().position(|(off, _)| *off == elem_offset)? } - fresh }; - entries.push((slot_offset, slot_size, fresh)); - entries.sort_by_key(|(off, _, _)| *off); - true + Some(&mut elements[index].1) } /// Apply C17 6.7.9p19 to the initializers one struct or union @@ -1386,16 +1725,22 @@ impl<'a> super::linearize::Linearizer<'a> { idx += 1; continue; } - if later_span.start <= earlier_span.start && earlier_span.end <= later_span.end { + // A bit-field never supersedes an initializer for an object + // containing it, however the two spans compare: it is narrower + // than the storage they share -- `struct { unsigned char a : 4, + // b : 4; }` is one byte -- so what it does not name stands, and + // it is folded into the carrier instead. + let inside = earlier_span.start <= later_span.start + && later_span.end <= earlier_span.end + && earlier.bit_width.is_none(); + if !(inside && later.bit_width.is_some()) + && later_span.start <= earlier_span.start + && earlier_span.end <= later_span.end + { merged.remove(idx); continue; } - let foldable = !folded - && earlier.bit_width.is_none() - && later.bit_width.is_none() - && earlier_span.start <= later_span.start - && later_span.end <= earlier_span.end - && self.fold_field_init(&mut merged[idx], &later); + let foldable = !folded && inside && self.fold_field_init(&mut merged[idx], &later); if foldable { folded = true; idx += 1; @@ -1416,26 +1761,134 @@ impl<'a> super::linearize::Linearizer<'a> { /// Returns false when no structural fold exists, which leaves the caller /// to discard `earlier`. fn fold_field_init(&self, earlier: &mut RawFieldInit, later: &RawFieldInit) -> bool { - let inner_offset = later.offset - earlier.offset; - match self.classify_subobject(earlier.typ, inner_offset, later.field_size) { + // `unions` borrows what the earlier entry holds, which the fold then + // updates, so it reads from a copy. + let held = earlier.held.clone(); + let unions = UnionFold::new(&held, &later.named, earlier.offset); + let span = later.byte_span(); + let inner_offset = span.start - earlier.offset; + let inner_size = span.end - span.start; + + let place = if later.bit_width.is_some() { + self.classify_bitfield_carrier(earlier.typ, inner_offset, inner_size, unions) + } else { + self.classify_subobject(earlier.typ, inner_offset, inner_size, unions) + }; + let folded = match place { SubobjectPlace::Member => self.overlay_subobject( earlier.typ, &mut earlier.init, inner_offset, - later.field_size, + inner_size, &later.init, + unions, ), + // Not storage of its own: only the bits the field names change, + // inside the carrier its declaring struct's initializer wrote. + SubobjectPlace::BitfieldCarrier { offset, size } => { + self.fold_bitfield_init(earlier, later, offset, size, unions) + } // The union stops holding what it held: everything it contained // goes, and it comes to hold just this one initializer. SubobjectPlace::ThroughUnion { offset, size } => { + let Some(fields) = self.union_replacement_fields(earlier, later, offset) else { + return false; + }; let replacement = Initializer::Struct { total_size: size, - fields: vec![(inner_offset - offset, later.field_size, later.init.clone())], + fields, }; - self.overlay_subobject(earlier.typ, &mut earlier.init, offset, size, &replacement) + let replaced = self.overlay_subobject( + earlier.typ, + &mut earlier.init, + offset, + size, + &replacement, + unions, + ); + if replaced { + let start = earlier.offset + offset; + earlier.held.clear_range(start..start + size); + } + replaced } SubobjectPlace::NotASubobject => false, + }; + + if folded { + // What the later initializer says about the unions it wrote or + // passed through is the last word on them. + earlier.held.absorb(&later.held); + earlier.held.absorb(&later.named); + } + folded + } + + /// The entries a union comes to hold once `later` replaces its contents: + /// the initializer itself, or the bytes of a bit-field's carrier, placed + /// at their offsets within the union beginning at `offset` of `earlier`. + fn union_replacement_fields( + &self, + earlier: &RawFieldInit, + later: &RawFieldInit, + offset: usize, + ) -> Option> { + let union_start = earlier.offset + offset; + let (Some(bit_offset), Some(bit_width)) = (later.bit_offset, later.bit_width) else { + let inner = later.offset.checked_sub(union_start)?; + return Some(vec![(inner, later.field_size, later.init.clone())]); + }; + let Initializer::Int(value) = later.init else { + return None; + }; + bitfield_carrier_bytes(bit_offset, bit_width, value) + .map(|(byte, bits, _)| { + let inner = (later.offset + byte).checked_sub(union_start)?; + Some((inner, 1, Initializer::Int(i128::from(bits)))) + }) + .collect() + } + + /// Replace the bits `later` names inside the carrier bytes the struct + /// declaring it -- at `offset..offset + size` of `earlier`'s object -- + /// already has an initializer for. + /// + /// The carrier is whole bytes by the time it reaches here: a struct's + /// bit-fields are merged into one byte apiece as the last step of lowering + /// it, downstream of the `Initializer` tree, so there is no bit-field left + /// in the tree to replace -- only the bits of it that a byte holds. + fn fold_bitfield_init( + &self, + earlier: &mut RawFieldInit, + later: &RawFieldInit, + offset: usize, + size: usize, + unions: UnionFold<'_>, + ) -> bool { + let (Some(bit_offset), Some(bit_width)) = (later.bit_offset, later.bit_width) else { + return false; + }; + let Initializer::Int(value) = later.init else { + return false; + }; + let Some(carrier) = + self.subobject_init_mut(earlier.typ, &mut earlier.init, offset, size, unions) + else { + return false; + }; + let Initializer::Struct { fields, .. } = carrier else { + return false; + }; + // Where the field's own storage begins within the declaring struct. + let Some(base) = later.offset.checked_sub(earlier.offset + offset) else { + return false; + }; + for (byte, bits, mask) in bitfield_carrier_bytes(bit_offset, bit_width, value) { + if !replace_carrier_bits(fields, base + byte, bits, mask) { + return false; + } } + true } /// Convert an AST initializer list to an IR Initializer @@ -1536,6 +1989,7 @@ impl<'a> super::linearize::Linearizer<'a> { // Convert field visits to RawFieldInit by evaluating expressions let mut raw_fields: Vec = Vec::new(); for visit in visits { + let held = self.held_union_members(visit.typ, &visit.kind, visit.offset); let field_init = match visit.kind { StructFieldVisitKind::BraceElision(sub_elements) => { self.ast_init_list_to_ir(&sub_elements, visit.typ) @@ -1551,6 +2005,8 @@ impl<'a> super::linearize::Linearizer<'a> { init: field_init, bit_offset: visit.bit_offset, bit_width: visit.bit_width, + held, + named: visit.unions, }); } @@ -1588,25 +2044,7 @@ impl<'a> super::linearize::Linearizer<'a> { let Initializer::Int(value) = field.init else { continue; }; - if bit_width == 0 { - continue; - } - - // Shifting `u128::MAX` down rather than `1 << width` - // up, for the reason `bitfield_value_mask` records: the - // latter overflows at the carrier's own width. - let mask = u128::MAX >> (128 - bit_width); - let placed = ((value as u128) & mask) << bit_off; - - // Only the bytes the field's own bits reach. Its window - // is wider and generally starts earlier -- `unsigned a:1` - // after a `char` sits at bit 8 of a window based at byte - // 0 -- and writing the whole window here would blank the - // members sharing it. - for byte in - (bit_off / 8) as usize..=((bit_off + bit_width - 1) / 8) as usize - { - let bits = ((placed >> (byte * 8)) & 0xff) as u8; + for (byte, bits, _) in bitfield_carrier_bytes(bit_off, bit_width, value) { *bitfield_bytes.entry(field.offset + byte).or_default() |= bits; } } @@ -1648,6 +2086,7 @@ impl<'a> super::linearize::Linearizer<'a> { let mut bit_offset = None; let mut bit_width = None; let mut access_bytes = None; + let mut unions = UnionMembers::default(); for (idx, designator) in designators.iter().enumerate() { match designator { @@ -1657,6 +2096,14 @@ impl<'a> super::linearize::Linearizer<'a> { resolved = self.types.base_type(resolved)?; } resolved = self.resolve_struct_type(resolved); + // Naming a member of a union says which member the + // initializer is for, and nothing downstream can recover + // that: every member of a union begins at `offset`. + if self.types.kind(resolved) == TypeKind::Union { + if let Some(index) = self.designated_member_index(resolved, *name) { + unions.record(offset, resolved, index); + } + } let member = self.types.find_member(resolved, *name)?; offset += member.offset; typ = member.typ; @@ -1698,15 +2145,32 @@ impl<'a> super::linearize::Linearizer<'a> { bit_offset, bit_width, access_bytes, + unions, }) } + /// The index, in `composite_typ`'s member list, of the member a `.name` + /// designator names -- or of the anonymous member that contains it. + fn designated_member_index(&self, composite_typ: TypeId, name: StringId) -> Option { + let members = &self.types.get(composite_typ).composite.as_ref()?.members; + match self.member_index_for_designator(members, name)? { + MemberDesignatorResult::Direct(next_idx) => Some(next_idx - 1), + MemberDesignatorResult::Anonymous { outer_idx, .. } => Some(outer_idx), + } + } + + /// The member the next positional element initializes, and its index in + /// `members`. + /// + /// The index is what tells two members of a union apart: they share a byte + /// offset, so nothing downstream can recover which one an initializer + /// chose. pub(crate) fn next_positional_member( &self, members: &[crate::types::StructMember], is_union: bool, current_field_idx: &mut usize, - ) -> Option { + ) -> Option<(MemberInfo, usize)> { if is_union { if *current_field_idx > 0 { return None; @@ -1723,34 +2187,42 @@ impl<'a> super::linearize::Linearizer<'a> { // members are *all* anonymous found none at all and stayed zero // entirely. Unnamed bit-field padding is not a member and is // still skipped. - let member = members + let (index, member) = members .iter() - .find(|m| m.name != StringId::EMPTY || m.bit_width.is_none())?; + .enumerate() + .find(|(_, m)| m.name != StringId::EMPTY || m.bit_width.is_none())?; *current_field_idx = members.len(); - return Some(MemberInfo { - offset: member.offset, - typ: member.typ, - bit_offset: member.bit_offset, - bit_width: member.bit_width, - access_bytes: member.access_bytes, - }); + return Some(( + MemberInfo { + offset: member.offset, + typ: member.typ, + bit_offset: member.bit_offset, + bit_width: member.bit_width, + access_bytes: member.access_bytes, + }, + index, + )); } while *current_field_idx < members.len() { - let member = &members[*current_field_idx]; + let index = *current_field_idx; + let member = &members[index]; if member.name == StringId::EMPTY && member.bit_width.is_some() { *current_field_idx += 1; continue; } if member.name != StringId::EMPTY || member.bit_width.is_none() { *current_field_idx += 1; - return Some(MemberInfo { - offset: member.offset, - typ: member.typ, - bit_offset: member.bit_offset, - bit_width: member.bit_width, - access_bytes: member.access_bytes, - }); + return Some(( + MemberInfo { + offset: member.offset, + typ: member.typ, + bit_offset: member.bit_offset, + bit_width: member.bit_width, + access_bytes: member.access_bytes, + }, + index, + )); } *current_field_idx += 1; } @@ -2271,6 +2743,9 @@ mod tests { /// struct S { struct T t; int z; }; /// struct A { int a[3]; int z; }; /// struct W { union { int i; struct { char a, b, c, d; } s; } u; }; + /// struct B { unsigned a : 4, b : 4; }; + /// struct C { struct B t; int z; }; + /// struct N { union { struct T p; int i; } u; }; /// ``` struct OverrideTypes { target: Target, @@ -2282,6 +2757,11 @@ mod tests { a: TypeId, w: TypeId, v: TypeId, + b: TypeId, + c: TypeId, + n: TypeId, + /// The `union { struct T p; int i; }` inside `struct N`. + n_union: TypeId, int_array: TypeId, } @@ -2297,6 +2777,25 @@ mod tests { } } + /// A bit-field of `width` bits at `bit_offset` of a four-byte access span + /// based at byte 0. + fn bitfield( + name: StringId, + typ: TypeId, + bit_offset: u32, + width: u32, + ) -> crate::types::StructMember { + crate::types::StructMember { + name, + typ, + offset: 0, + bit_offset: Some(bit_offset), + bit_width: Some(width), + access_bytes: Some(4), + explicit_align: None, + } + } + fn composite( members: Vec, size: usize, @@ -2381,6 +2880,36 @@ mod tests { 4, ))); + // `struct B { unsigned a : 4, b : 4; }` and the `struct C { struct + // B t; int z; }` that holds it: two bit-fields in one carrier byte, + // with an ordinary member beside them. + let uint = types.uint_id; + let bits = types.intern(Type::struct_type(composite( + vec![bitfield(a_name, uint, 0, 4), bitfield(b, uint, 4, 4)], + 4, + 4, + ))); + let bits_holder = types.intern(Type::struct_type(composite( + vec![member(t_name, bits, 0), member(z, int, 4)], + 8, + 4, + ))); + + // `struct N { union { struct T p; int i; } u; }` -- a union whose + // members differ in size, so that an override naming a subobject + // of the one it holds has somewhere to fold into. + let p_name = name(&mut strings, "p"); + let n_union = types.intern(Type::union_type(composite( + vec![member(p_name, t, 0), member(i, int, 0)], + 8, + 4, + ))); + let n = types.intern(Type::struct_type(composite( + vec![member(u_name, n_union, 0)], + 8, + 4, + ))); + Self { target, types, @@ -2391,6 +2920,10 @@ mod tests { a, w, v, + b: bits, + c: bits_holder, + n, + n_union, int_array, } } @@ -2410,9 +2943,34 @@ mod tests { init, bit_offset: None, bit_width: None, + held: UnionMembers::default(), + named: UnionMembers::default(), } } + /// The same for a bit-field: `offset` is its access span's first byte and + /// the value goes at `bit_offset..bit_offset + width` of that span. + fn raw_bits( + offset: usize, + typ: TypeId, + bit_offset: u32, + width: u32, + value: i128, + ) -> RawFieldInit { + RawFieldInit { + bit_offset: Some(bit_offset), + bit_width: Some(width), + ..raw(offset, typ, 4, Initializer::Int(value)) + } + } + + /// One `(byte offset, union type, member index)` recorded for a union. + fn union_member(offset: usize, typ: TypeId, index: usize) -> UnionMembers { + let mut members = UnionMembers::default(); + members.record(offset, typ, index); + members + } + fn struct_init(total_size: usize, fields: &[(usize, usize, i128)]) -> Initializer { Initializer::Struct { total_size, @@ -2439,25 +2997,25 @@ mod tests { // The whole object, the member `t`, and `t.y` inside it. assert_eq!( - lin.classify_subobject(fixture.s, 0, 12), + lin.classify_subobject(fixture.s, 0, 12, UnionFold::default()), SubobjectPlace::Member ); assert_eq!( - lin.classify_subobject(fixture.s, 0, 8), + lin.classify_subobject(fixture.s, 0, 8, UnionFold::default()), SubobjectPlace::Member ); assert_eq!( - lin.classify_subobject(fixture.s, 4, 4), + lin.classify_subobject(fixture.s, 4, 4, UnionFold::default()), SubobjectPlace::Member ); // `a[1]` of `struct A`. assert_eq!( - lin.classify_subobject(fixture.a, 4, 4), + lin.classify_subobject(fixture.a, 4, 4, UnionFold::default()), SubobjectPlace::Member ); // The four bytes straddling `t.y` and `z` are no object at all. assert_eq!( - lin.classify_subobject(fixture.s, 6, 4), + lin.classify_subobject(fixture.s, 6, 4, UnionFold::default()), SubobjectPlace::NotASubobject ); } @@ -2469,14 +3027,41 @@ mod tests { // `u.s.b` -- one byte, reached only by choosing a union member. assert_eq!( - lin.classify_subobject(fixture.w, 1, 1), + lin.classify_subobject(fixture.w, 1, 1, UnionFold::default()), SubobjectPlace::ThroughUnion { offset: 0, size: 4 } ); // The union named exactly is an ordinary member of `struct W`. assert_eq!( - lin.classify_subobject(fixture.w, 0, 4), + lin.classify_subobject(fixture.w, 0, 4, UnionFold::default()), + SubobjectPlace::Member + ); + } + + /// When both initializers name the same member of a union, the range is + /// an ordinary subobject reached through it and the rest of that member + /// stands. When they name different ones, it is not. + #[test] + fn a_union_is_descended_into_only_where_both_sides_name_the_same_member() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // `n.u.p.y` of `struct N`, with the union holding `p`. + let holds_p = union_member(0, fixture.n_union, 0); + let names_p = union_member(0, fixture.n_union, 0); + let names_i = union_member(0, fixture.n_union, 1); + assert_eq!( + lin.classify_subobject(fixture.n, 4, 4, UnionFold::new(&holds_p, &names_p, 0)), SubobjectPlace::Member ); + assert_eq!( + lin.classify_subobject(fixture.n, 4, 4, UnionFold::new(&holds_p, &names_i, 0)), + SubobjectPlace::ThroughUnion { offset: 0, size: 8 } + ); + // And with nothing said about it at all. + assert_eq!( + lin.classify_subobject(fixture.n, 4, 4, UnionFold::default()), + SubobjectPlace::ThroughUnion { offset: 0, size: 8 } + ); } /// `struct S s = { .t = {1, 2}, .t.y = 9, .z = 7 };` -- the override @@ -2711,4 +3296,170 @@ mod tests { vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)]))] ); } + + /// A bit-field's bytes are its carrier's, not its own, so the answer for + /// them is the struct that declares it -- and they are no subobject at + /// all when asked for as an object. + #[test] + fn a_bitfields_bytes_name_the_struct_that_declares_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + + // `t.a` of `struct C`: byte 0, inside `t` at 0..4. + assert_eq!( + lin.classify_bitfield_carrier(fixture.c, 0, 1, UnionFold::default()), + SubobjectPlace::BitfieldCarrier { offset: 0, size: 4 } + ); + // Asked for `struct B` itself, the carrier byte is the whole of the + // declaring struct and still must not be replaced whole. + assert_eq!( + lin.classify_bitfield_carrier(fixture.b, 0, 1, UnionFold::default()), + SubobjectPlace::BitfieldCarrier { offset: 0, size: 4 } + ); + assert_eq!( + lin.classify_subobject(fixture.c, 0, 1, UnionFold::default()), + SubobjectPlace::NotASubobject + ); + // Padding inside a struct with bit-fields is still nothing at all. + assert_eq!( + lin.classify_bitfield_carrier(fixture.c, 1, 1, UnionFold::default()), + SubobjectPlace::NotASubobject + ); + } + + /// `struct C c = { .t = {1, 2}, .t.a = 3 };` -- the override names one + /// bit-field, so the one sharing its carrier keeps its value. + #[test] + fn a_designated_override_of_a_bitfield_replaces_only_its_bits() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let uint = fixture.types.uint_id; + + // `{1, 2}` has already been lowered to the carrier byte it produces. + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.b, 4, struct_init(4, &[(0, 1, 0x21)])), + raw_bits(0, uint, 0, 4, 3), + ]); + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(0, 1, 0x23)]))]); + + // And the other bit-field of the pair. + let merged = lin.merge_raw_field_inits(vec![ + raw(0, fixture.b, 4, struct_init(4, &[(0, 1, 0x21)])), + raw_bits(0, uint, 4, 4, 5), + ]); + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(0, 1, 0x51)]))]); + } + + /// A bit-field never supersedes an initializer for an object containing + /// it, but a whole-struct initializer written after one does. + #[test] + fn a_whole_struct_initializer_supersedes_an_earlier_bitfield() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let uint = fixture.types.uint_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw_bits(0, uint, 4, 4, 5), + raw(0, fixture.b, 4, struct_init(4, &[(0, 1, 0x21)])), + ]); + assert_eq!(kept(&merged), vec![(0, 4, struct_init(4, &[(0, 1, 0x21)]))]); + } + + /// `struct N n = { .u = {1, 2}, .u.p.y = 9 };` -- the union still holds + /// `p`, and the override names a member *of* `p`, so `p.x` keeps its 1. + #[test] + fn an_override_inside_the_held_union_member_folds_into_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let held = union_member(0, fixture.n_union, 0); + let named = union_member(0, fixture.n_union, 0); + let merged = lin.merge_raw_field_inits(vec![ + RawFieldInit { + held, + ..raw( + 0, + fixture.n_union, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))], + }, + ) + }, + RawFieldInit { + named, + ..raw(4, int, 4, Initializer::Int(9)) + }, + ]); + + assert_eq!( + kept(&merged), + vec![( + 0, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 9)]))], + } + )] + ); + } + + /// The companion case that makes the rule a rule: naming a *different* + /// member discards what the union held, however the two spans overlap. + #[test] + fn an_override_naming_a_different_union_member_still_resets_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + RawFieldInit { + held: union_member(0, fixture.n_union, 0), + ..raw( + 0, + fixture.n_union, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))], + }, + ) + }, + RawFieldInit { + // `.u.i`, the union's other member. + named: union_member(0, fixture.n_union, 1), + ..raw(0, int, 4, Initializer::Int(7)) + }, + ]); + + assert_eq!(kept(&merged), vec![(0, 8, struct_init(8, &[(0, 4, 7)]))]); + } + + /// With nothing said about the union, the merge cannot tell "the same + /// member, deeper" from "a different member" and takes the reset, which + /// is the answer that discards rather than invents. + #[test] + fn an_override_through_a_union_nothing_names_resets_it() { + let fixture = OverrideTypes::new(); + let lin = fixture.linearizer(); + let int = fixture.types.int_id; + + let merged = lin.merge_raw_field_inits(vec![ + raw( + 0, + fixture.n_union, + 8, + Initializer::Struct { + total_size: 8, + fields: vec![(0, 8, struct_init(8, &[(0, 4, 1), (4, 4, 2)]))], + }, + ), + raw(4, int, 4, Initializer::Int(9)), + ]); + + assert_eq!(kept(&merged), vec![(0, 8, struct_init(8, &[(4, 4, 9)]))]); + } } diff --git a/cc/ir/linearize_stmt.rs b/cc/ir/linearize_stmt.rs index 5c33148f4..e493ed50e 100644 --- a/cc/ir/linearize_stmt.rs +++ b/cc/ir/linearize_stmt.rs @@ -113,6 +113,14 @@ struct WrittenInit { /// bytes only by passing through a union. origin: usize, typ: TypeId, + /// The member each union inside that subobject came to hold, which + /// decides whether a later entry naming something inside one of them + /// names the *same* member -- and so leaves the rest of it standing. + held: UnionMembers, + /// Set when the entry is a bit-field, to its bit offset. Two bit-fields + /// sharing a carrier byte are different objects and neither supersedes + /// the other, however their bytes overlap. + bits: Option, } /// Grow `reset` to also cover `range`. @@ -1129,7 +1137,8 @@ impl<'a> super::linearize::Linearizer<'a> { let mut written: Vec = Vec::new(); for visit in visits { - if let Some(reset) = self.init_override_reset(&mut written, &visit) { + let held = self.held_union_members(visit.typ, &visit.kind, visit.offset); + if let Some(reset) = self.init_override_reset(&mut written, &visit, held) { self.emit_block_zero( base_sym, base_offset + reset.start as i64, @@ -1216,39 +1225,66 @@ impl<'a> super::linearize::Linearizer<'a> { /// entry it invalidates: all of one it wholly contains, all of one it is /// not a subobject of, and, where it reaches an earlier entry's bytes only /// by naming a member of a union inside it, that union's bytes. + /// + /// A bit-field is stored by reading its carrier and writing it back, so + /// its own bytes are not storage it owns and clearing them would blank the + /// members sharing the carrier. It therefore never clears its own span -- + /// only what an earlier entry it supersedes requires, which is how a + /// bit-field naming a second member of a union still resets it. fn init_override_reset( &self, written: &mut Vec, visit: &StructFieldVisit, + held: UnionMembers, ) -> Option> { - // A bit-field is stored by reading its carrier and writing it back, so - // its window is not storage it owns and clearing the window would - // blank the members sharing it. Bit-fields are left to overlay each - // other as they always have. - if visit.bit_width.is_some() || visit.field_size == 0 { + if visit.field_size == 0 || visit.bit_width == Some(0) { return None; } - - let span = visit.offset..visit.offset + visit.field_size; + let bits = visit.bit_offset.filter(|_| visit.bit_width.is_some()); + let span = match (visit.bit_offset, visit.bit_width) { + // Only the bytes the field's own bits reach; its access span is + // wider and covers bytes other members own. + (Some(bit_offset), Some(bit_width)) => { + let start = visit.offset + (bit_offset / 8) as usize; + let end = visit.offset + (bit_offset + bit_width).div_ceil(8) as usize; + start..end.max(start + 1) + } + _ => visit.offset..visit.offset + visit.field_size, + }; let mut reset: Option> = None; for entry in written.iter() { if entry.live.start >= span.end || span.start >= entry.live.end { continue; } - widen_reset(&mut reset, span.clone()); - - // Wholly superseded: the entry's own bytes are all inside the - // ones being cleared and rewritten. - if span.start <= entry.live.start && entry.live.end <= span.end { + // Two bit-fields are different objects even when they share a + // carrier byte, and the same one written twice needs no clearing: + // the second store reads the carrier back and replaces its bits. + if bits.is_some() && entry.bits.is_some() { continue; } + if bits.is_none() { + widen_reset(&mut reset, span.clone()); + + // Wholly superseded: the entry's own bytes are all inside the + // ones being cleared and rewritten. + if span.start <= entry.live.start && entry.live.end <= span.end { + continue; + } + } let entry_end = entry.origin + self.types.size_bytes(entry.typ); - let place = if span.start >= entry.origin && span.end <= entry_end { - self.classify_subobject(entry.typ, span.start - entry.origin, visit.field_size) - } else { - SubobjectPlace::NotASubobject + let inner = span + .start + .checked_sub(entry.origin) + .filter(|_| span.end <= entry_end); + let unions = UnionFold::new(&entry.held, &visit.unions, entry.origin); + let place = match inner { + None => SubobjectPlace::NotASubobject, + Some(inner) if bits.is_some() => { + self.classify_bitfield_carrier(entry.typ, inner, span.end - span.start, unions) + } + Some(inner) => self.classify_subobject(entry.typ, inner, visit.field_size, unions), }; match place { // A member of the earlier entry's object: the rest of that @@ -1260,6 +1296,9 @@ impl<'a> super::linearize::Linearizer<'a> { entry.origin + offset..entry.origin + offset + size, ); } + // The carrier the earlier entry wrote is shared: the bits this + // one names are replaced in place and its neighbours' stand. + SubobjectPlace::BitfieldCarrier { .. } => {} SubobjectPlace::NotASubobject => { widen_reset(&mut reset, entry.live.clone()); } @@ -1272,14 +1311,21 @@ impl<'a> super::linearize::Linearizer<'a> { *written = written .drain(..) .flat_map(|entry| { - let (origin, typ) = (entry.origin, entry.typ); + let (origin, typ, held, bits) = + (entry.origin, entry.typ, entry.held, entry.bits); [ entry.live.start..entry.live.end.min(from), entry.live.start.max(to)..entry.live.end, ] .into_iter() .filter(|live| live.start < live.end) - .map(move |live| WrittenInit { live, origin, typ }) + .map(move |live| WrittenInit { + live, + origin, + typ, + held: held.clone(), + bits, + }) }) .collect(); } @@ -1288,6 +1334,8 @@ impl<'a> super::linearize::Linearizer<'a> { live: span, origin: visit.offset, typ: visit.typ, + held, + bits, }); reset } diff --git a/cc/tests/c99/initializers.rs b/cc/tests/c99/initializers.rs index aafdfc376..208414f0a 100644 --- a/cc/tests/c99/initializers.rs +++ b/cc/tests/c99/initializers.rs @@ -2899,3 +2899,195 @@ int main(void) assert_eq!(compile_and_run("union_member_reset", code, &[]), 0); assert_eq!(compile_and_run_optimized("union_member_reset_opt", code), 0); } + +/// A designated override of a bit-field replaces only that bit-field. +/// +/// The remaining half of the subobject rule. Bit-fields share a carrier, and +/// the carrier's bytes are merged downstream of the `Initializer` tree, so the +/// static path could not fold one override into an earlier initializer and fell +/// back to dropping it whole -- losing `b` in `{ .t = {1,2}, .t.a = 3 }` -- +/// while the automatic path stored the carrier and then stored over part of it, +/// keeping `b`. gcc keeps it. The two paths disagreeing is the defect; gcc's +/// answer is which way to settle it. +#[test] +fn c99_a_designated_override_of_a_bitfield_keeps_its_neighbours() { + let code = r#" +struct B { unsigned a : 4, b : 4; }; +struct S { struct B t; int z; }; + +struct S g = { .t = {1, 2}, .t.a = 3, .z = 7 }; +struct S g2 = { .t = {1, 2}, .t.b = 5 }; + +int main(void) +{ + struct S l = { .t = {1, 2}, .t.a = 3, .z = 7 }; + struct S l2 = { .t = {1, 2}, .t.b = 5 }; + + if (g.t.a != 3 || g.t.b != 2 || g.z != 7) return 1; + if (l.t.a != 3 || l.t.b != 2 || l.z != 7) return 2; + if (g2.t.a != 1 || g2.t.b != 5) return 3; + if (l2.t.a != 1 || l2.t.b != 5) return 4; + + return 0; +} +"#; + assert_eq!( + compile_and_run("designated_bitfield_override", code, &[]), + 0 + ); + assert_eq!( + compile_and_run_optimized("designated_bitfield_override_opt", code), + 0 + ); +} + +/// An override naming a subobject of the union member already held keeps the +/// rest of that member. +/// +/// `{ .u = {1,2}, .u.p.y = 9 }` initializes the union's first member and then +/// overrides one of *its* members, so the union still holds `p` and `p.x` keeps +/// the 1 it was given. c17 reset the union instead, because the `Initializer` +/// tree records no discriminant and the merge could not tell "the same member, +/// deeper" from "a different member" -- and resetting is right only for the +/// second. Both storage durations agreed on the wrong answer, so nothing caught +/// it. +/// +/// The companion case, where a *different* member is named and the union really +/// is reset, is covered by +/// `c99_initializing_a_second_union_member_resets_the_union`, which must keep +/// passing: the two are what distinguish the rule. +#[test] +fn c99_an_override_inside_the_held_union_member_keeps_the_rest() { + let code = r#" +struct P { int x, y; }; +struct N { union { struct P p; int i; } u; }; + +struct N g = { .u = {1, 2}, .u.p.y = 9 }; + +int main(void) +{ + struct N l = { .u = {1, 2}, .u.p.y = 9 }; + if (g.u.p.x != 1 || g.u.p.y != 9) return 1; + if (l.u.p.x != 1 || l.u.p.y != 9) return 2; + return 0; +} +"#; + assert_eq!( + compile_and_run("union_member_deeper_override", code, &[]), + 0 + ); + assert_eq!( + compile_and_run_optimized("union_member_deeper_override_opt", code), + 0 + ); +} + +/// An initializer for a whole struct supersedes an earlier one for a +/// bit-field inside it, including the bit-fields it says nothing about. +/// +/// The other direction of the bit-field rule, and the one the automatic path +/// had wrong: it stored the bit-field, then stored the struct's own +/// bit-fields over it, and `c` -- which `{1, 2}` does not mention -- kept the +/// 7. The static path dropped the earlier entry whole and was right. gcc and +/// clang zero it. +/// +/// The objects here are deliberately wider than eight bytes, to keep the +/// assertions clear of an unrelated x86-64 defect that widens a 32-bit store +/// at offset 0 of an eight-byte local to 64 bits. +#[test] +fn c99_a_whole_struct_initializer_supersedes_an_earlier_bitfield() { + let code = r#" +struct B { unsigned a : 4, b : 4, c : 4; }; +struct S { struct B t; int z; long pad; }; + +struct S g = { .t.c = 7, .t = {1, 2} }; + +int main(void) +{ + struct S l = { .t.c = 7, .t = {1, 2} }; + + if (g.t.a != 1 || g.t.b != 2 || g.t.c != 0) return 1; + if (l.t.a != 1 || l.t.b != 2 || l.t.c != 0) return 2; + + return 0; +} +"#; + assert_eq!(compile_and_run("bitfield_superseded", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("bitfield_superseded_opt", code), + 0 + ); +} + +/// A bit-field naming a second member of a union resets the union, as any +/// other initializer for a second member does. +/// +/// A bit-field is stored by reading its carrier and writing it back, so the +/// automatic path emitted no fill for it and three bytes of the `int` showed +/// through the `struct` that replaced it. It is not that a bit-field clears +/// nothing -- it clears nothing *of its own*, because its neighbours in the +/// carrier are other objects -- but what the union it displaces requires. +#[test] +fn c99_a_bitfield_naming_a_second_union_member_resets_the_union() { + let code = r#" +struct U { union { int i; struct { unsigned a : 4, b : 4; } s; } u; long pad; }; + +struct U g = { .u.i = 0x01020304, .u.s.a = 3 }; + +int main(void) +{ + struct U l = { .u.i = 0x01020304, .u.s.a = 3 }; + + if (g.u.i != 3 || g.u.s.a != 3 || g.u.s.b != 0) return 1; + if (l.u.i != 3 || l.u.s.a != 3 || l.u.s.b != 0) return 2; + + return 0; +} +"#; + assert_eq!(compile_and_run("bitfield_union_reset", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("bitfield_union_reset_opt", code), + 0 + ); +} + +/// Naming a subobject of a union member the union does *not* hold resets it, +/// even where the two members are the same size and the same shape. +/// +/// The guard on the fold above. Knowing which member is held comes from the +/// initializer list, not from the lowered bytes, so `struct P` and `struct Q` +/// being indistinguishable once lowered costs nothing: `.u = {1, 2}` gives +/// `p` a value and `.u.q.d = 9` names `q`, so the union comes to hold `q` +/// with only `d` given a value. Reading it back through the *other* member +/// would be undefined; `q.c` is not. +#[test] +fn c99_an_override_naming_another_union_member_resets_it_whatever_its_shape() { + let code = r#" +struct P { int x, y; }; +struct Q { int c, d; }; +struct N { union { struct P p; struct Q q; } u; long pad; }; + +struct N g = { .u = {1, 2}, .u.q.d = 9 }; + +/* And the fold, in the same union, when the member named is the held one. */ +struct N h = { .u = {1, 2}, .u.p.y = 9 }; + +int main(void) +{ + struct N l = { .u = {1, 2}, .u.q.d = 9 }; + struct N m = { .u = {1, 2}, .u.p.y = 9 }; + + if (g.u.q.c != 0 || g.u.q.d != 9) return 1; + if (l.u.q.c != 0 || l.u.q.d != 9) return 2; + if (h.u.p.x != 1 || h.u.p.y != 9) return 3; + if (m.u.p.x != 1 || m.u.p.y != 9) return 4; + + return 0; +} +"#; + assert_eq!(compile_and_run("union_other_member_reset", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("union_other_member_reset_opt", code), + 0 + ); +} From 87489eb58ffbbff686ab51ab10969356d4ea4a92 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 09:13:52 -0400 Subject: [PATCH 16/17] cc: widen a store over a slot only where the slot is one scalar A 32-bit store at offset 0 of a local is widened to 64 bits so a narrow value going into a wider slot leaves no stale upper bits behind it. The comment on it already recorded the exception that needs -- "struct/union fields at offset 0 must use exact size to avoid clobbering the adjacent field at offset 4" -- and the exception asked whether the object was larger than 64 bits. An eight-byte aggregate is not, so exactly the case the comment describes fell through it: `struct S { struct P t; }` initialized `{ .t = {1,2}, .t.x = 3 }` read `y` back as 0 at -O0. No size test can separate them, which is why this one only spared the aggregates too big to be confused with a scalar in the first place: a `long` and a `struct { int x, y; }` are both sixty-four bits. So the map the store lowering consults records what it actually needs -- the width, and whether the slot holds a single scalar -- taken from the local's type where the map is built. A complex is excluded with the aggregates: `_Complex float` is sixty-four bits with its imaginary half at offset 4, the same shape, and declining to widen can never leave the stale bits the widening exists to clear. A slot with no record keeps the widening, as before. Only -O0 reached it. With the optimizer on, the field is promoted out of memory before the store lowering sees it, so the tests name -O0 as well as running the default matrix. The other shapes were already right and are pinned here too: a plain field assignment, a store through a pointer, a union member, an `int[2]`, and a four-`short` struct. So is the reason the widening is there, each case writing a wide pattern into the slot before overwriting it narrowly, since a fix that merely stopped widening would pass everything else. Co-Authored-By: Claude Opus 5 (1M context) --- cc/arch/x86_64/codegen.rs | 20 ++++++- cc/arch/x86_64/frame.rs | 17 ++++-- cc/arch/x86_64/memory.rs | 34 +++++------ cc/tests/codegen/misc.rs | 123 ++++++++++++++++++++++++++++++++++++++ 4 files changed, 169 insertions(+), 25 deletions(-) diff --git a/cc/arch/x86_64/codegen.rs b/cc/arch/x86_64/codegen.rs index 65c1a3be5..be12375a0 100644 --- a/cc/arch/x86_64/codegen.rs +++ b/cc/arch/x86_64/codegen.rs @@ -25,6 +25,20 @@ use std::collections::{HashMap, HashSet}; // x86-64 Code Generator /// x86-64 code generator +/// What the store lowering needs to know about a local's stack slot. +/// +/// The size alone cannot answer it: a `long` and a `struct { int x, y; }` are +/// both sixty-four bits, and only one of them may have a narrow store at +/// offset 0 widened over the rest of the slot. +pub(super) struct SymSlot { + /// Width of the declared type, in bits. + pub(super) bits: u32, + /// The slot holds a single scalar value, so anything above a narrow store + /// at offset 0 is stale bits of that same object. False for an aggregate + /// or a complex, whose other member lives where a widened store reaches. + pub(super) one_scalar: bool, +} + pub struct X86_64CodeGen { /// Common code generation infrastructure pub(super) base: CodeGenBase, @@ -77,8 +91,8 @@ pub struct X86_64CodeGen { /// binary128 constants to emit (pool key -> the 16-byte image). /// BTreeMap for reproducible order, as `ld_constants`. pub(super) quad_constants: std::collections::BTreeMap, - /// Sym pseudo ID → type size in bits (for distinguishing scalar vs struct stores) - pub(super) sym_type_sizes: HashMap, + /// Sym pseudo ID → what its stack slot holds, for [`SymSlot`]. + pub(super) sym_slots: HashMap, /// How this function's locals are addressed. pub(super) frame_base: FrameBase, /// Maximum local alignment (for andq in prologue) @@ -110,7 +124,7 @@ impl X86_64CodeGen { ld_constants: std::collections::BTreeMap::new(), double_constants: std::collections::BTreeMap::new(), quad_constants: std::collections::BTreeMap::new(), - sym_type_sizes: HashMap::new(), + sym_slots: HashMap::new(), frame_base: FrameBase::Rbp, max_local_align: 16, int128_pseudos: HashSet::new(), diff --git a/cc/arch/x86_64/frame.rs b/cc/arch/x86_64/frame.rs index e702c3abe..e46b5eccf 100644 --- a/cc/arch/x86_64/frame.rs +++ b/cc/arch/x86_64/frame.rs @@ -241,14 +241,21 @@ impl X86_64CodeGen { self.int128_pseudos = alloc.int128_pseudos().clone(); self.pseudos = crate::arch::codegen::PseudoTable::new(&func.pseudos); - // Build sym type size map for emit_store to distinguish struct fields from scalars - self.sym_type_sizes.clear(); + // What `emit_store` needs to decide whether a narrow store at offset 0 + // may be widened over the rest of the slot. + self.sym_slots.clear(); for pseudo in &func.pseudos { // By identity: a global whose name collides with a parameter's - // would otherwise be recorded with the parameter's type size. + // would otherwise be recorded with the parameter's type. if let Some(local_var) = func.local_of(pseudo.id) { - self.sym_type_sizes - .insert(pseudo.id, types.size_bits(local_var.typ)); + let typ = local_var.typ; + self.sym_slots.insert( + pseudo.id, + crate::arch::x86_64::codegen::SymSlot { + bits: types.size_bits(typ), + one_scalar: types.is_scalar(typ) && !types.is_complex(typ), + }, + ); } } diff --git a/cc/arch/x86_64/memory.rs b/cc/arch/x86_64/memory.rs index cd588e963..5116b325c 100644 --- a/cc/arch/x86_64/memory.rs +++ b/cc/arch/x86_64/memory.rs @@ -927,24 +927,24 @@ impl X86_64CodeGen { let op_size = OperandSize::from_bits(mem_size); if is_symbol { // Local variable - store directly to stack slot. - // Widen 32-bit stores at offset 0 to 64-bit to prevent stale - // upper bits when a 32-bit result is stored into a 64-bit - // local (e.g., int-to-long, int-to-pointer assignments). - // Exception: struct/union fields at offset 0 must use exact - // size to avoid clobbering the adjacent field at offset 4. + // Widen a 32-bit store at offset 0 to 64 bits, so a narrow + // value going into a wider slot leaves no stale upper bits + // behind it (an int-to-long or int-to-pointer assignment). + // + // Only where the slot holds one scalar, though: an + // aggregate or a complex has another member at offset 4 or + // 8, and widening the store writes over it. Asking how + // *large* the object is cannot tell the two apart -- a + // `long` and a `struct { int x, y; }` are both 64 bits, and + // testing for more than 64 spared only the aggregates too + // big to be confused with a scalar in the first place. A + // slot this has no record of keeps the widening, which is + // what it did before. let store_size = if mem_size == 32 && insn.offset == 0 { - let sym_bits = self.sym_type_sizes.get(&addr).copied().unwrap_or(64); - if sym_bits > 32 { - // Check if this is a struct/union (don't widen field stores) - let is_struct = - self.sym_type_sizes.contains_key(&addr) && sym_bits > 64; - if is_struct { - op_size // struct field: exact size - } else { - OperandSize::B64 // scalar/pointer: safe to widen - } - } else { - OperandSize::B64 // small scalar: safe to widen + match self.sym_slots.get(&addr) { + Some(slot) if slot.one_scalar && slot.bits <= 64 => OperandSize::B64, + Some(_) => op_size, + None => OperandSize::B64, } } else { op_size diff --git a/cc/tests/codegen/misc.rs b/cc/tests/codegen/misc.rs index e3422e7b9..093d22d04 100644 --- a/cc/tests/codegen/misc.rs +++ b/cc/tests/codegen/misc.rs @@ -16600,3 +16600,126 @@ float probe(void) { struct P q = {1, 2, 3}; return sum(q); } "expected the inlined parameter copy to reach offset 8 at all:\n{ir}" ); } + +/// Storing one field of an eight-byte aggregate leaves the other alone. +/// +/// The x86-64 store lowering widens a 32-bit store at offset 0 of a local to +/// 64 bits, to clear stale upper bits when a narrow value goes into a wider +/// slot. Its own comment records the exception that needs: "struct/union +/// fields at offset 0 must use exact size to avoid clobbering the adjacent +/// field at offset 4". The exception asked whether the object was *larger than* +/// 64 bits, which an eight-byte aggregate is not -- so exactly the case the +/// comment describes was the one that fell through. +/// +/// Only at `-O0`: with the optimizer on, the field is promoted out of memory +/// before the store lowering sees it. +#[test] +fn codegen_a_field_store_does_not_widen_over_its_neighbour() { + let code = r#" +struct P { int x, y; }; +struct S { struct P t; }; +union U { struct P p; double d; }; + +int main(void) +{ + /* The reported shape: a designated override inside an eight-byte struct. */ + struct S a = { .t = {1, 2}, .t.x = 3 }; + if (a.t.x != 3 || a.t.y != 2) return 1; + + /* The same store reached other ways. */ + struct P b = {1, 2}; + b.x = 3; + if (b.x != 3 || b.y != 2) return 2; + + struct P c; + c.y = 2; + c.x = 3; + if (c.x != 3 || c.y != 2) return 3; + + struct P *p = &b; + p->x = 9; + if (b.x != 9 || b.y != 2) return 4; + + union U u; + u.p.y = 7; + u.p.x = 5; + if (u.p.x != 5 || u.p.y != 7) return 5; + + /* Arrays are the same shape at the same size. */ + int arr[2] = {1, 2}; + arr[0] = 3; + if (arr[0] != 3 || arr[1] != 2) return 6; + + /* Exactly eight bytes made of narrower fields. */ + struct Q { short a, b, c, d; } q = {1, 2, 3, 4}; + q.a = 9; + if (q.a != 9 || q.b != 2 || q.c != 3 || q.d != 4) return 7; + + return 0; +} +"#; + // `-O0` explicitly: the default matrix compiles at `-O`, where the field is + // promoted out of memory before the store lowering ever sees it, so the + // defect is invisible there. + assert_eq!( + compile_and_run("field_store_no_widen", code, &["-O0".to_string()]), + 0 + ); + assert_eq!(compile_and_run("field_store_no_widen_matrix", code, &[]), 0); + assert_eq!( + compile_and_run_optimized("field_store_no_widen_opt", code), + 0 + ); +} + +/// The control: a narrow value stored into a wider scalar slot still leaves no +/// stale upper bits. +/// +/// This is what the widening is for, and it is why the fix has to ask whether +/// the object is an aggregate rather than simply stop widening. Each case +/// writes a wide value into the slot first, so a store that failed to clear the +/// upper half would read it back. +#[test] +fn codegen_a_narrow_store_into_a_wide_slot_clears_it() { + let code = r#" +int wide(void) { return -1; } + +int main(void) +{ + /* Put a known wide pattern in the slot, then overwrite it narrowly. */ + long l = 0x7fffffff7fffffffL; + int i = 5; + l = i; + if (l != 5) return 1; + + unsigned long ul = 0xffffffffffffffffUL; + unsigned ui = 7; + ul = ui; + if (ul != 7UL) return 2; + + void *vp = (void *)0x7fffffffffffL; + unsigned addr = 0; + vp = (void *)(unsigned long)addr; + if (vp != (void *)0) return 3; + + /* Through a call, so the value is not a constant the optimizer can see. */ + long l2 = 0x7fffffff7fffffffL; + l2 = wide(); + if (l2 != -1L) return 4; + + return 0; +} +"#; + assert_eq!( + compile_and_run("narrow_store_clears_slot", code, &["-O0".to_string()]), + 0 + ); + assert_eq!( + compile_and_run("narrow_store_clears_slot_matrix", code, &[]), + 0 + ); + assert_eq!( + compile_and_run_optimized("narrow_store_clears_slot_opt", code), + 0 + ); +} From ce123a4403d03d02f545317cb1a93a92c2ba1a34 Mon Sep 17 00:00:00 2001 From: Jeff Garzik Date: Wed, 30 Sep 2026 09:46:57 -0400 Subject: [PATCH 17/17] cc: ask both back ends' store lowerings the same question about a slot CI caught this on aarch64: 87489eb5 fixed the widening test in the x86-64 store lowering and left the identical one in aarch64, where an eight-byte aggregate is still not larger than sixty-four bits and its second field is still written over. The behavioural test failed there and passes here only because this host has no aarch64 runner, so every aarch64 behavioural test skips locally. Both back ends had grown the same rule and both had got it wrong the same way, so it is no longer written twice. `SymSlot` and the walk that builds it move to the shared code: the width, and whether the slot holds a single scalar, which is the question a widening actually turns on. `widenable()` answers it, and both lowerings ask. What each does with an *unrecorded* symbol is left as it was -- x86-64 widens, aarch64 keeps the exact width -- because that is a separate judgment about globals and stores through pointers, not the one being corrected here, and changing it quietly while fixing something else is how the two drifted apart in the first place. Verified on aarch64 by the emitted code: a field store is `str w` where it was `str x`, and a narrow value into a `long` slot is still `str x`. Co-Authored-By: Claude Opus 5 (1M context) --- cc/arch/aarch64/codegen.rs | 4 +-- cc/arch/aarch64/frame.rs | 11 +-------- cc/arch/aarch64/memory.rs | 28 +++++++-------------- cc/arch/codegen.rs | 50 ++++++++++++++++++++++++++++++++++++++ cc/arch/x86_64/codegen.rs | 16 +----------- cc/arch/x86_64/frame.rs | 18 +------------- cc/arch/x86_64/memory.rs | 4 ++- 7 files changed, 67 insertions(+), 64 deletions(-) diff --git a/cc/arch/aarch64/codegen.rs b/cc/arch/aarch64/codegen.rs index a7d66c042..bc1d9b885 100644 --- a/cc/arch/aarch64/codegen.rs +++ b/cc/arch/aarch64/codegen.rs @@ -78,7 +78,7 @@ pub struct Aarch64CodeGen { /// Stack allocation size for locals (for zero_stack_frame) pub(super) stack_alloc_size: i32, /// Sym pseudo ID → type size in bits (for distinguishing scalar vs struct stores) - pub(super) sym_type_sizes: HashMap, + pub(super) sym_slots: HashMap, /// Which register this function's locals are addressed through pub(super) frame_base: FrameBase, } @@ -103,7 +103,7 @@ impl Aarch64CodeGen { pic_mode: false, unique_label_counter: 0, stack_alloc_size: 0, - sym_type_sizes: HashMap::new(), + sym_slots: HashMap::new(), frame_base: FrameBase::Fp, } } diff --git a/cc/arch/aarch64/frame.rs b/cc/arch/aarch64/frame.rs index ccd5fe1a8..d74ee812c 100644 --- a/cc/arch/aarch64/frame.rs +++ b/cc/arch/aarch64/frame.rs @@ -51,16 +51,7 @@ impl Aarch64CodeGen { self.locations = alloc.allocate(func, types); self.pseudos = crate::arch::codegen::PseudoTable::new(&func.pseudos); - // Build sym type size map for emit_store to distinguish struct fields from scalars - self.sym_type_sizes.clear(); - for pseudo in &func.pseudos { - // By identity: a global whose name collides with a parameter's - // would otherwise be recorded with the parameter's type size. - if let Some(local_var) = func.local_of(pseudo.id) { - self.sym_type_sizes - .insert(pseudo.id, types.size_bits(local_var.typ)); - } - } + self.sym_slots = crate::arch::codegen::sym_slots(func, types); let stack_size = alloc.stack_size(); self.frame_base = alloc.frame_base(); diff --git a/cc/arch/aarch64/memory.rs b/cc/arch/aarch64/memory.rs index be0ce65ae..89bdeef67 100644 --- a/cc/arch/aarch64/memory.rs +++ b/cc/arch/aarch64/memory.rs @@ -601,26 +601,16 @@ impl Aarch64CodeGen { return; } - // Widen 32-bit stores at offset 0 to 64-bit to prevent stale - // upper bits when a 32-bit result is stored into a 64-bit - // local (e.g., int-to-long, int-to-pointer assignments). - // Only widen for known local variables (in sym_type_sizes). - // Do NOT widen stores to globals/statics (not in sym_type_sizes) - // or stores through pointers — widening could clobber adjacent data. - // Exception: struct/union fields at offset 0 must use exact - // size to avoid clobbering the adjacent field at offset 4. + // Widen a 32-bit store at offset 0 to 64 bits, so a narrow value going + // into a wider slot leaves no stale upper bits behind it (an + // int-to-long or int-to-pointer assignment). Only where the slot holds + // one scalar: see `SymSlot`. Only for a known local, too -- a global or + // a store through a pointer keeps its exact width, since nothing here + // knows what adjoins it. let store_size = if mem_size == 32 && insn.offset == 0 { - if let Some(&sym_bits) = self.sym_type_sizes.get(&addr) { - // Known local variable — safe to widen if scalar and > 32 bits - if sym_bits > 64 { - OperandSize::from_bits(mem_size) // struct field: exact size - } else if sym_bits > 32 { - OperandSize::B64 // scalar/pointer local: safe to widen - } else { - OperandSize::from_bits(mem_size) - } - } else { - OperandSize::from_bits(mem_size) // global/static/pointer: exact size + match self.sym_slots.get(&addr) { + Some(slot) if slot.widenable() && slot.bits > 32 => OperandSize::B64, + _ => OperandSize::from_bits(mem_size), } } else { OperandSize::from_bits(mem_size) diff --git a/cc/arch/codegen.rs b/cc/arch/codegen.rs index ea301f93e..3fb5786bc 100644 --- a/cc/arch/codegen.rs +++ b/cc/arch/codegen.rs @@ -1222,6 +1222,56 @@ pub fn check_tls_reached_only_by_address( } } +/// What a store lowering needs to know about a local's stack slot. +/// +/// Both back ends widen a narrow store at offset 0 of a local so that a value +/// going into a wider slot leaves no stale upper bits behind it. That is only +/// sound where the slot holds a single scalar: an aggregate or a complex has +/// another member at offset 4 or 8, which the widened store would write over. +/// +/// The size alone cannot answer it -- a `long` and a `struct { int x, y; }` +/// are both sixty-four bits -- and asking for *more* than sixty-four spares +/// only the aggregates too large to be mistaken for a scalar in the first +/// place. Both back ends had that test and both got an eight-byte aggregate +/// wrong, so the question is asked once, here. +pub struct SymSlot { + /// Width of the declared type, in bits. + pub bits: u32, + /// The slot holds a single scalar value, so anything above a narrow store + /// at offset 0 is stale bits of that same object. + pub one_scalar: bool, +} + +impl SymSlot { + /// Whether a 32-bit store at offset 0 of this slot may be widened to 64. + pub fn widenable(&self) -> bool { + self.one_scalar && self.bits <= 64 + } +} + +/// Record, for each of `func`'s locals, what its stack slot holds. +pub fn sym_slots( + func: &crate::ir::Function, + types: &crate::types::TypeTable, +) -> std::collections::HashMap { + let mut slots = std::collections::HashMap::new(); + for pseudo in &func.pseudos { + // By identity: a global whose name collides with a parameter's would + // otherwise be recorded with the parameter's type. + if let Some(local_var) = func.local_of(pseudo.id) { + let typ = local_var.typ; + slots.insert( + pseudo.id, + SymSlot { + bits: types.size_bits(typ), + one_scalar: types.is_scalar(typ) && !types.is_complex(typ), + }, + ); + } + } + slots +} + /// The current function's pseudos, looked up by id. /// /// A pseudo's id is not its position in `Function::pseudos`, so a lookup diff --git a/cc/arch/x86_64/codegen.rs b/cc/arch/x86_64/codegen.rs index be12375a0..1209df87a 100644 --- a/cc/arch/x86_64/codegen.rs +++ b/cc/arch/x86_64/codegen.rs @@ -25,20 +25,6 @@ use std::collections::{HashMap, HashSet}; // x86-64 Code Generator /// x86-64 code generator -/// What the store lowering needs to know about a local's stack slot. -/// -/// The size alone cannot answer it: a `long` and a `struct { int x, y; }` are -/// both sixty-four bits, and only one of them may have a narrow store at -/// offset 0 widened over the rest of the slot. -pub(super) struct SymSlot { - /// Width of the declared type, in bits. - pub(super) bits: u32, - /// The slot holds a single scalar value, so anything above a narrow store - /// at offset 0 is stale bits of that same object. False for an aggregate - /// or a complex, whose other member lives where a widened store reaches. - pub(super) one_scalar: bool, -} - pub struct X86_64CodeGen { /// Common code generation infrastructure pub(super) base: CodeGenBase, @@ -92,7 +78,7 @@ pub struct X86_64CodeGen { /// BTreeMap for reproducible order, as `ld_constants`. pub(super) quad_constants: std::collections::BTreeMap, /// Sym pseudo ID → what its stack slot holds, for [`SymSlot`]. - pub(super) sym_slots: HashMap, + pub(super) sym_slots: HashMap, /// How this function's locals are addressed. pub(super) frame_base: FrameBase, /// Maximum local alignment (for andq in prologue) diff --git a/cc/arch/x86_64/frame.rs b/cc/arch/x86_64/frame.rs index e46b5eccf..a8a27c252 100644 --- a/cc/arch/x86_64/frame.rs +++ b/cc/arch/x86_64/frame.rs @@ -241,23 +241,7 @@ impl X86_64CodeGen { self.int128_pseudos = alloc.int128_pseudos().clone(); self.pseudos = crate::arch::codegen::PseudoTable::new(&func.pseudos); - // What `emit_store` needs to decide whether a narrow store at offset 0 - // may be widened over the rest of the slot. - self.sym_slots.clear(); - for pseudo in &func.pseudos { - // By identity: a global whose name collides with a parameter's - // would otherwise be recorded with the parameter's type. - if let Some(local_var) = func.local_of(pseudo.id) { - let typ = local_var.typ; - self.sym_slots.insert( - pseudo.id, - crate::arch::x86_64::codegen::SymSlot { - bits: types.size_bits(typ), - one_scalar: types.is_scalar(typ) && !types.is_complex(typ), - }, - ); - } - } + self.sym_slots = crate::arch::codegen::sym_slots(func, types); let stack_size = alloc.stack_size(); self.callee_saved_regs = alloc.callee_saved_used().to_vec(); diff --git a/cc/arch/x86_64/memory.rs b/cc/arch/x86_64/memory.rs index 5116b325c..26b7a61ab 100644 --- a/cc/arch/x86_64/memory.rs +++ b/cc/arch/x86_64/memory.rs @@ -942,8 +942,10 @@ impl X86_64CodeGen { // what it did before. let store_size = if mem_size == 32 && insn.offset == 0 { match self.sym_slots.get(&addr) { - Some(slot) if slot.one_scalar && slot.bits <= 64 => OperandSize::B64, + Some(slot) if slot.widenable() => OperandSize::B64, Some(_) => op_size, + // A slot with no record keeps the widening here, + // which is what this back end did before. None => OperandSize::B64, } } else {