From edd7655050e8ad4b70e5982d7e0484e64aeaf5ce Mon Sep 17 00:00:00 2001 From: Noah Gift Date: Wed, 16 Sep 2026 09:30:52 +0200 Subject: [PATCH 1/7] test(PMAT-3346): the measured Qwen3.5-0.8B inventory the dense path cannot reproduce MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit QE2E-INV-001 could not be judged because nothing in the tree held a MEASURED Qwen3.5 tensor inventory to judge against. This adds one: the 320 tensors of ~/models/Qwen3.5-0.8B-Q4_K_M.gguf (sha256 bd258782...dc517), read straight from the GGUF header rather than from a model card. It already falsifies the current arithmetic. Dense/GQA accounting applied to that file gives 644,400,128 against a measured 752,393,024 — short by 107,992,896, 14.4% of the model, because 18 of the 24 layers are Gated DeltaNet and no term here counts their conv, gate, state or output projections. Two shapes in the file are also not what dense accounting predicts, and both are pinned: attn_q is [1024, 4096] = 2 * num_heads * head_dim (the q projection emits the attention output gate alongside the query; attn_output [2048, 1024] confirms num_heads * head_dim = 2048), and attn_q_norm/attn_k_norm are present at head_dim. The file is also TIED — it has no output.weight. Refs #3346 Pmat-Ticket: PMAT-3346 Co-Authored-By: Claude Opus 5 (1M context) --- .../src/format/model_arithmetic_tests.rs | 137 ++++++++++++++++++ docs/roadmaps/roadmap.yaml | 16 ++ 2 files changed, 153 insertions(+) diff --git a/crates/aprender-core/src/format/model_arithmetic_tests.rs b/crates/aprender-core/src/format/model_arithmetic_tests.rs index f80d810730..27ca64e222 100644 --- a/crates/aprender-core/src/format/model_arithmetic_tests.rs +++ b/crates/aprender-core/src/format/model_arithmetic_tests.rs @@ -55,6 +55,143 @@ fn qwen35_constraints() -> ModelConstraints { } } +// --------------------------------------------------------------------------- +// Ground truth: a REAL Qwen3.5 file (#3346) +// --------------------------------------------------------------------------- +// +// Every number below was read out of `~/models/Qwen3.5-0.8B-Q4_K_M.gguf` +// (sha256 `bd258782e35f7f458f8aced1adc053e6e92e89bc735ba3be89d38a06121dc517`, +// GGUF v3, 320 tensors) by parsing the file header directly on 2026-09-16 — +// not from a model card, a memory, or the family descriptor. The descriptor +// `contracts/model-families/qwen3_5.yaml` declares only the 9b and 27b +// variants, and no Qwen3.5-9B file is on this box, so the 0.8B file is the +// only Qwen3.5 whose true tensor inventory can be MEASURED here. It is the +// oracle for the shape arithmetic: if the config-derived count and this +// inventory disagree, the arithmetic is wrong. +// +// Shapes are GGUF `ne` order (`[in, out]` for a 2-D weight); only the element +// COUNT matters for a parameter total, so the order is reproduced verbatim +// rather than transposed. + +/// One Gated DeltaNet layer of `Qwen3.5-0.8B-Q4_K_M.gguf` (measured: `blk.0`, +/// identical for the 18 layers whose index is not `interval-1 mod interval`). +const QWEN35_0_8B_GDN_LAYER: &[(&str, &[usize])] = &[ + ("attn_gate.weight", &[1024, 2048]), + ("attn_norm.weight", &[1024]), + ("attn_qkv.weight", &[1024, 6144]), + ("ffn_down.weight", &[3584, 1024]), + ("ffn_gate.weight", &[1024, 3584]), + ("ffn_up.weight", &[1024, 3584]), + ("post_attention_norm.weight", &[1024]), + ("ssm_a", &[16]), + ("ssm_alpha.weight", &[1024, 16]), + ("ssm_beta.weight", &[1024, 16]), + ("ssm_conv1d.weight", &[4, 6144]), + ("ssm_dt.bias", &[16]), + ("ssm_norm.weight", &[128]), + ("ssm_out.weight", &[2048, 1024]), +]; + +/// One full-attention layer of the same file (measured: `blk.3`, identical for +/// the 6 layers at indices 3, 7, 11, 15, 19, 23 — `full_attention_interval` 4). +/// +/// Two shapes here are NOT what dense/GQA accounting predicts, and both are +/// facts of the file: `attn_q` is `[1024, 4096]` = `2 * num_heads * head_dim` +/// (Qwen3.5 gates the attention output, so the q projection emits the gate +/// alongside the query — `attn_output` is `[2048, 1024]`, confirming +/// `num_heads * head_dim` = 2048), and `attn_q_norm`/`attn_k_norm` are present +/// at `head_dim`. +const QWEN35_0_8B_ATTENTION_LAYER: &[(&str, &[usize])] = &[ + ("attn_k.weight", &[1024, 512]), + ("attn_k_norm.weight", &[256]), + ("attn_norm.weight", &[1024]), + ("attn_output.weight", &[2048, 1024]), + ("attn_q.weight", &[1024, 4096]), + ("attn_q_norm.weight", &[256]), + ("attn_v.weight", &[1024, 512]), + ("ffn_down.weight", &[3584, 1024]), + ("ffn_gate.weight", &[1024, 3584]), + ("ffn_up.weight", &[1024, 3584]), + ("post_attention_norm.weight", &[1024]), +]; + +/// The file's non-layer tensors. There is no `output.weight`: the 0.8B TIES +/// its unembedding to the embedding, unlike the 9b descriptor. +const QWEN35_0_8B_GLOBAL: &[(&str, &[usize])] = &[ + ("output_norm.weight", &[1024]), + ("token_embd.weight", &[1024, 248_320]), +]; + +/// Number of GDN and full-attention layers in the measured file (24 blocks, +/// `qwen35.full_attention_interval` = 4). +const QWEN35_0_8B_GDN_LAYERS: u64 = 18; +const QWEN35_0_8B_ATTENTION_LAYERS: u64 = 6; + +/// Total elements of a measured tensor list. +fn tensor_elements(tensors: &[(&str, &[usize])]) -> u64 { + tensors + .iter() + .map(|(_, dims)| u64::try_from(dims.iter().product::()).unwrap_or(u64::MAX)) + .sum() +} + +/// `Qwen3.5-0.8B` as the GGUF's own metadata keys describe it: `block_count` +/// 24, `embedding_length` 1024, `feed_forward_length` 3584, +/// `attention.head_count` 8, `attention.head_count_kv` 2, `attention.key_length` +/// 256, `rope.freq_base` 1e7, `attention.layer_norm_rms_epsilon` 1e-6, and a +/// 248320-token vocabulary (`token_embd.weight` is `[1024, 248320]`). +fn qwen35_0_8b_size() -> ModelSizeConfig { + ModelSizeConfig { + parameters: "0.8B".to_string(), + hidden_dim: 1024, + num_layers: 24, + num_heads: 8, + num_kv_heads: 2, + intermediate_dim: 3584, + vocab_size: 248_320, + max_position_embeddings: 262_144, + head_dim: 256, + rope_theta: 10_000_000.0, + norm_eps: 1e-6, + } +} + +/// The measured file total: `752,393,024` parameters. +const QWEN35_0_8B_MEASURED_TOTAL: u64 = 752_393_024; + +#[test] +fn qwen35_0_8b_measured_inventory_sums_to_the_file_total() { + // Tensor COUNT: 14 per GDN layer, 11 per attention layer, 2 global = 320, + // which is what the GGUF header declares (`n_tensors`). + let counted = QWEN35_0_8B_GDN_LAYER.len() * 18 + QWEN35_0_8B_ATTENTION_LAYER.len() * 6 + 2; + assert_eq!(counted, 320, "GGUF header declares 320 tensors"); + + assert_eq!(tensor_elements(QWEN35_0_8B_GDN_LAYER), 21_555_360); + assert_eq!(tensor_elements(QWEN35_0_8B_ATTENTION_LAYER), 18_352_640); + assert_eq!(tensor_elements(QWEN35_0_8B_GLOBAL), 254_280_704); + + let total = tensor_elements(QWEN35_0_8B_GLOBAL) + + QWEN35_0_8B_GDN_LAYERS * tensor_elements(QWEN35_0_8B_GDN_LAYER) + + QWEN35_0_8B_ATTENTION_LAYERS * tensor_elements(QWEN35_0_8B_ATTENTION_LAYER); + assert_eq!(total, QWEN35_0_8B_MEASURED_TOTAL); +} + +/// The defect of #3346, as a number rather than a claim: dense/GQA accounting +/// applied to a hybrid family under-counts a REAL file by 107,992,896 +/// parameters — 14.4% of the model. Three quarters of the layers are Gated +/// DeltaNet, and none of their conv/gate/state tensors have a term here. +#[test] +fn dense_accounting_cannot_reproduce_the_measured_qwen35_0_8b_file() { + let size = qwen35_0_8b_size(); + let mut constraints = qwen35_constraints(); + constraints.tied_embeddings = true; // measured: the file has no output.weight + let layers = uniform_layers(&size, &constraints); + let p = model_parameter_count(&size, &constraints, &layers); + + assert_eq!(p.total, 644_400_128); + assert_eq!(QWEN35_0_8B_MEASURED_TOTAL - p.total, 107_992_896); +} + // --------------------------------------------------------------------------- // model_parameter_count // --------------------------------------------------------------------------- diff --git a/docs/roadmaps/roadmap.yaml b/docs/roadmaps/roadmap.yaml index 3a90bd6297..7a960ee709 100644 --- a/docs/roadmaps/roadmap.yaml +++ b/docs/roadmaps/roadmap.yaml @@ -18124,6 +18124,22 @@ roadmap: estimated_effort: null labels: [] notes: null +- id: PMAT-3346 + github_issue: 3346 + item_type: task + title: ModelConstraints carries the gated-DeltaNet shape keys + status: planned + priority: medium + assigned_to: null + created: 2026-09-16T07:21:39Z + updated: 2026-09-16T07:21:39Z + spec: null + acceptance_criteria: [] + phases: [] + subtasks: [] + estimated_effort: null + labels: [] + notes: null - id: PMAT-3347 github_issue: 3347 item_type: task From c6af4a0cca1cf2715a3d61df04add1daa98d8dfa Mon Sep 17 00:00:00 2001 From: Noah Gift Date: Wed, 16 Sep 2026 09:43:16 +0200 Subject: [PATCH 2/7] fix(PMAT-3346): ModelConstraints carries the gated-DeltaNet shape, so a hybrid model can be counted MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit contracts/model-families/qwen3_5.yaml declares inner_size, state_size, conv_kernel, group_count and full_attention_interval under constraints:, and ModelConstraints carried none of them. A Gated DeltaNet layer's parameters live entirely in those dimensions, so every consumer of the descriptor counted Qwen3.5 as if three quarters of its layers did not exist. Carried through as ModelConstraints::deltanet: Option — the runtime YAML loader (parsing.rs) and the compiled-in registry (build_parsing.rs + build_codegen.rs) both populate it, and FALSIFY-MF-QWEN35-010 pins that the declared values survive the trip and that no other family acquires a shape it never declared. qwen3_5.yaml is the only descriptor with these keys, so every other family keeps byte-identical accounting. model_arithmetic gains gated_deltanet_layer_params (one term per GGUF tensor: attn_qkv, attn_gate, ssm_conv1d, ssm_alpha/beta, ssm_a, ssm_dt.bias, ssm_norm, ssm_out) and hybrid_layers (the interleaved schedule). attention_layer_params gained two terms the real file has and dense accounting did not model: the gated q projection (2*n_h*d_k) and the q/k norm vectors. Falsified against a real model, not against itself: fed the 0.8B configuration, the equation now reproduces the 320-tensor inventory of Qwen3.5-0.8B-Q4_K_M.gguf EXACTLY — 752,393,024, both layer kinds matching tensor for tensor. QE2E-INV-001 is still NOT asserted, and no range was widened to make it pass. The 9b descriptor now yields 8,344,907,136, up from 8,208,519,168 but still 0.655B below [9.0B, 9.2B]. The remaining gap looks like descriptor drift rather than missing arithmetic: 9b keeps inner_size 2048 — the value the 0.8B uses at hidden_dim 1024 — while quadrupling hidden_dim, and its group_count 8 fails 8 * 128 == 2048, a consistency the measured 0.8B satisfies at 16 * 128. Only a real Qwen3.5-9B file can settle it; none is on this box. Refs #3346, #3347 Pmat-Ticket: PMAT-3346 Co-Authored-By: Claude Opus 5 (1M context) --- .../commands/oracle_compute_param_memory.rs | 1 + .../oracle_cross_validation_architecture.rs | 1 + crates/apr-cli/src/commands/tests.rs | 3 + crates/aprender-core/build_codegen.rs | 15 ++ crates/aprender-core/build_parsing.rs | 29 +++ .../src/format/model_arithmetic.rs | 150 ++++++++++++-- .../src/format/model_arithmetic_tests.rs | 189 +++++++++++++++--- .../aprender-core/src/format/model_family.rs | 47 +++++ .../format/model_family_contract_falsify.rs | 50 +++++ .../src/format/model_family_loader.rs | 4 +- .../src/format/model_family_tests.rs | 1 + crates/aprender-core/src/format/parsing.rs | 18 ++ 12 files changed, 460 insertions(+), 48 deletions(-) diff --git a/crates/apr-cli/src/commands/oracle_compute_param_memory.rs b/crates/apr-cli/src/commands/oracle_compute_param_memory.rs index ec47f463ea..5179d689c4 100644 --- a/crates/apr-cli/src/commands/oracle_compute_param_memory.rs +++ b/crates/apr-cli/src/commands/oracle_compute_param_memory.rs @@ -275,6 +275,7 @@ positional_encoding: PositionalEncoding::Absolute, mlp_type: MlpType::GeluMlp, qk_norm: false, + deltanet: None, }, tensor_template: TensorTemplate { embedding: "wte.weight".to_string(), diff --git a/crates/apr-cli/src/commands/oracle_cross_validation_architecture.rs b/crates/apr-cli/src/commands/oracle_cross_validation_architecture.rs index a558db5feb..0c91135949 100644 --- a/crates/apr-cli/src/commands/oracle_cross_validation_architecture.rs +++ b/crates/apr-cli/src/commands/oracle_cross_validation_architecture.rs @@ -273,6 +273,7 @@ positional_encoding: PositionalEncoding::Rope, mlp_type: MlpType::GeluMlp, qk_norm: false, + deltanet: None, }; let params = compute_param_count(&size, &constraints); assert!(params > 0, "Even minimal model should have params"); diff --git a/crates/apr-cli/src/commands/tests.rs b/crates/apr-cli/src/commands/tests.rs index b35741ead8..cc54f89b2b 100644 --- a/crates/apr-cli/src/commands/tests.rs +++ b/crates/apr-cli/src/commands/tests.rs @@ -44,6 +44,7 @@ positional_encoding: PositionalEncoding::Rope, mlp_type: MlpType::SwiGlu, qk_norm: false, + deltanet: None, }, tensor_template: TensorTemplate { embedding: "embed.weight".to_string(), @@ -160,6 +161,7 @@ positional_encoding: aprender::format::model_family::PositionalEncoding::Rope, mlp_type: aprender::format::model_family::MlpType::SwiGlu, qk_norm: false, + deltanet: None, }, tensor_template: aprender::format::model_family::TensorTemplate { embedding: String::new(), @@ -301,6 +303,7 @@ positional_encoding: PositionalEncoding::Rope, mlp_type: MlpType::SwiGlu, qk_norm: false, + deltanet: None, } } diff --git a/crates/aprender-core/build_codegen.rs b/crates/aprender-core/build_codegen.rs index 16c506f1c9..114fc9a3d5 100644 --- a/crates/aprender-core/build_codegen.rs +++ b/crates/aprender-core/build_codegen.rs @@ -204,6 +204,7 @@ fn generate_family_registration(f: &FamilyData) -> String { \x20 positional_encoding: PositionalEncoding::from_str_contract(\"{}\").unwrap_or(PositionalEncoding::Rope),\n\ \x20 mlp_type: MlpType::from_str_contract(\"{}\").unwrap_or(MlpType::SwiGlu),\n\ \x20 qk_norm: {},\n\ + \x20 deltanet: {},\n\ \x20 }},\n\ \x20 tensor_template: TensorTemplate {{\n\ \x20 embedding: \"{}\".to_string(),\n\ @@ -243,6 +244,7 @@ fn generate_family_registration(f: &FamilyData) -> String { f.constraints.position, f.constraints.mlp, f.constraints.qk_norm, + deltanet_expr(f), f.embedding_tensor, f.lm_head_tensor .as_ref() @@ -490,3 +492,16 @@ fn generate_algebraic_proofs(f: &FamilyData) -> String { out.push('\n'); out } + +/// #3346: render a family's Gated DeltaNet shape as a Rust expression, so the +/// compiled-in registry carries the same keys the runtime YAML parser does. +/// A family that declares none renders `None` and keeps dense accounting. +fn deltanet_expr(f: &FamilyData) -> String { + match &f.constraints.deltanet { + None => "None".to_string(), + Some(d) => format!( + "Some(DeltaNetShape {{ inner_size: {}, state_size: {}, conv_kernel: {}, group_count: {}, full_attention_interval: {} }})", + d.inner_size, d.state_size, d.conv_kernel, d.group_count, d.full_attention_interval + ), + } +} diff --git a/crates/aprender-core/build_parsing.rs b/crates/aprender-core/build_parsing.rs index c0cedbca33..724f327396 100644 --- a/crates/aprender-core/build_parsing.rs +++ b/crates/aprender-core/build_parsing.rs @@ -52,6 +52,34 @@ struct ConstraintsData { position: String, mlp: String, qk_norm: bool, + /// #3346: Gated DeltaNet shape keys. `None` unless the descriptor declares + /// both `inner_size` and `state_size`. + deltanet: Option, +} + +/// #3346: the `inner_size`/`state_size`/`conv_kernel`/`group_count` + +/// `full_attention_interval` block of a hybrid family's `constraints:`. +struct DeltaNetData { + inner_size: usize, + state_size: usize, + conv_kernel: usize, + group_count: usize, + full_attention_interval: usize, +} + +/// #3346: read the Gated DeltaNet shape out of a `constraints:` section. +/// Both `inner_size` and `state_size` are required — they are what make the +/// block a DeltaNet mixer — so every other family yields `None`. +fn parse_deltanet_data(section: &str) -> Option { + let inner_size = get_usize(section, "inner_size")?; + let state_size = get_usize(section, "state_size")?; + Some(DeltaNetData { + inner_size, + state_size, + conv_kernel: get_usize(section, "conv_kernel").unwrap_or(0), + group_count: get_usize(section, "group_count").unwrap_or(0), + full_attention_interval: get_usize(section, "full_attention_interval").unwrap_or(0), + }) } // ============================================================================ @@ -144,6 +172,7 @@ fn parse_family_yaml(content: &str, path: &Path) -> FamilyData { position: c_str("positional_encoding", "rope"), mlp: c_str("mlp_type", "swiglu"), qk_norm: get_bool(&constraints_section, "qk_norm").unwrap_or(false), + deltanet: parse_deltanet_data(&constraints_section), }; // Parse tensor_template diff --git a/crates/aprender-core/src/format/model_arithmetic.rs b/crates/aprender-core/src/format/model_arithmetic.rs index f49f54719c..1441bbb5f9 100644 --- a/crates/aprender-core/src/format/model_arithmetic.rs +++ b/crates/aprender-core/src/format/model_arithmetic.rs @@ -43,7 +43,9 @@ //! does not settle, so it is left unbound and `QE2E-BND-005` stays false. use crate::format::layout_contract::block_sizes; -use crate::format::model_family::{MlpType, ModelConstraints, ModelSizeConfig}; +use crate::format::model_family::{ + AttentionType, DeltaNetShape, MlpType, ModelConstraints, ModelSizeConfig, +}; // ============================================================================ // Equation: model_parameter_count @@ -96,13 +98,16 @@ pub struct ParameterBreakdown { /// `d_attn`, `d_ffn`, `d_norm` for one ordinary (softmax-attention) decoder /// layer of a family described by `size` + `constraints`. /// -/// - `d_attn` = `d*(n_h*d_k) + 2*d*(n_kv*d_k) + (n_h*d_k)*d` (Q, K, V, O), -/// plus the four bias vectors when `constraints.has_bias`. +/// - `d_attn` = `d*q_out + 2*d*(n_kv*d_k) + (n_h*d_k)*d` (Q, K, V, O), plus the +/// four bias vectors when `constraints.has_bias`, plus `2*d_k` of q/k norm +/// weights when `constraints.qk_norm`. `q_out` is `n_h*d_k`, or twice that +/// for a gated-attention family (see the comment on the `q_out` binding). /// - `d_ffn` = `3*d*d_ff` for a gated MLP (SwiGLU/GeGLU), else `2*d*d_ff`. /// - `d_norm` = `2*d` (input norm + post-attention norm). /// -/// This is the dense/GQA accounting. It is NOT the Gated DeltaNet accounting: -/// see [`model_parameter_count`] for what that needs. +/// This is the softmax-attention layer. It is NOT the Gated DeltaNet +/// accounting: that is [`gated_deltanet_layer_params`], and +/// [`hybrid_layers`] interleaves the two. #[must_use] pub fn attention_layer_params( size: &ModelSizeConfig, @@ -113,8 +118,20 @@ pub fn attention_layer_params( let q_dim = (size.num_heads as u64).saturating_mul(d_k); let kv_dim = (size.num_kv_heads as u64).saturating_mul(d_k); + // A gated-attention family emits the output gate from the q projection, so + // that matrix is 2*n_h*d_k wide rather than n_h*d_k. MEASURED in + // Qwen3.5-0.8B-Q4_K_M.gguf: attn_q is [1024, 4096] while attn_output is + // [2048, 1024], so o_proj still sees n_h*d_k = 2048 and only q is doubled. + let q_out = if matches!( + constraints.attention_type, + AttentionType::HybridGatedDeltaNet + ) { + q_dim.saturating_mul(2) + } else { + q_dim + }; let projections = d - .saturating_mul(q_dim) + .saturating_mul(q_out) .saturating_add(d.saturating_mul(kv_dim).saturating_mul(2)) .saturating_add(q_dim.saturating_mul(d)); let biases = if constraints.has_bias { @@ -124,21 +141,115 @@ pub fn attention_layer_params( } else { 0 }; + // Per-head q/k RMSNorm weights (attn_q_norm/attn_k_norm, one d_k vector each). + let qk_norms = if constraints.qk_norm { + d_k.saturating_mul(2) + } else { + 0 + }; - let d_ff = size.intermediate_dim as u64; + LayerParams { + d_attn: projections.saturating_add(biases).saturating_add(qk_norms), + d_ffn: ffn_params(size, constraints), + d_norm: d.saturating_mul(2), + } +} + +/// `d_ffn` for one layer: `3*d*d_ff` for a gated MLP (SwiGLU/GeGLU — gate, up +/// and down), else `2*d*d_ff`. Both layer kinds of a hybrid model share it. +fn ffn_params(size: &ModelSizeConfig, constraints: &ModelConstraints) -> u64 { let matrices = if matches!(constraints.mlp_type, MlpType::SwiGlu | MlpType::GatedMlp) { 3 } else { 2 }; + (size.hidden_dim as u64) + .saturating_mul(size.intermediate_dim as u64) + .saturating_mul(matrices) +} + +/// `d_attn`, `d_ffn`, `d_norm` for one **Gated DeltaNet** layer — the `d_attn` +/// the dense formula cannot express, because none of its dimensions are +/// `n_h * d_k`. +/// +/// Every term is one tensor of `Qwen3.5-0.8B-Q4_K_M.gguf`, named here as the +/// file names it (`i` = `inner_size`, `s` = `state_size`, `k` = `conv_kernel`, +/// `h` = `group_count`): +/// +/// | Tensor | Shape | Parameters | +/// |--------|-------|------------| +/// | `attn_qkv.weight` | `[d, 3*i]` | `3*d*i` | +/// | `attn_gate.weight` | `[d, i]` | `d*i` | +/// | `ssm_conv1d.weight` | `[k, 3*i]` | `3*k*i` | +/// | `ssm_alpha.weight`, `ssm_beta.weight` | `[d, h]` each | `2*d*h` | +/// | `ssm_a`, `ssm_dt.bias` | `[h]` each | `2*h` | +/// | `ssm_norm.weight` | `[s]` | `s` | +/// | `ssm_out.weight` | `[i, d]` | `i*d` | +/// +/// `d_norm` is `2*d` (`attn_norm` + `post_attention_norm`) and `d_ffn` is the +/// same SwiGLU block as an attention layer — a DeltaNet layer differs only in +/// how it mixes tokens. +#[must_use] +pub fn gated_deltanet_layer_params( + size: &ModelSizeConfig, + constraints: &ModelConstraints, + shape: &DeltaNetShape, +) -> LayerParams { + let d = size.hidden_dim as u64; + let inner = shape.inner_size as u64; + let heads = shape.group_count as u64; + let qkv = d.saturating_mul(inner).saturating_mul(3); + let gate = d.saturating_mul(inner); + let conv = (shape.conv_kernel as u64) + .saturating_mul(inner) + .saturating_mul(3); + let alpha_beta = d.saturating_mul(heads).saturating_mul(2); + let per_head = heads.saturating_mul(2); + let out = inner.saturating_mul(d); + + let d_attn = qkv + .saturating_add(gate) + .saturating_add(conv) + .saturating_add(alpha_beta) + .saturating_add(per_head) + .saturating_add(shape.state_size as u64) + .saturating_add(out); LayerParams { - d_attn: projections.saturating_add(biases), - d_ffn: d.saturating_mul(d_ff).saturating_mul(matrices), + d_attn, + d_ffn: ffn_params(size, constraints), d_norm: d.saturating_mul(2), } } +/// The per-layer input for a HYBRID model: Gated DeltaNet layers with a +/// softmax-attention layer every `full_attention_interval`-th position, the +/// last of each group. +/// +/// Falls back to [`uniform_layers`] for any family that declares no DeltaNet +/// shape (every family but `qwen3_5`) or declares no schedule, so the answer +/// for a dense family is byte-for-byte what it was before #3346. +#[must_use] +pub fn hybrid_layers(size: &ModelSizeConfig, constraints: &ModelConstraints) -> Vec { + let Some(shape) = constraints.deltanet else { + return uniform_layers(size, constraints); + }; + if shape.full_attention_interval == 0 { + return uniform_layers(size, constraints); + } + let attention = attention_layer_params(size, constraints); + let deltanet = gated_deltanet_layer_params(size, constraints, &shape); + (0..size.num_layers) + .map(|i| { + if (i + 1) % shape.full_attention_interval == 0 { + attention + } else { + deltanet + } + }) + .collect() +} + /// `L` copies of [`attention_layer_params`] — the per-layer input for a /// homogeneous (non-hybrid) model of `size.num_layers` layers. #[must_use] @@ -166,14 +277,19 @@ pub fn uniform_layers(size: &ModelSizeConfig, constraints: &ModelConstraints) -> /// /// # What this does NOT discharge /// -/// `QE2E-INV-001` wants `P(Qwen3.5-9B) ∈ [9.0B, 9.2B]`. Feeding this function -/// [`uniform_layers`] for the 9B variant yields ≈8.21B, because Qwen3.5 is -/// `hybrid_gated_deltanet`: three of every four layers are Gated DeltaNet, whose -/// `d_attn` covers conv, gate and state projections sized by `inner_size`, -/// `state_size`, `conv_kernel` and `group_count`. Those four keys exist in -/// `contracts/model-families/qwen3_5.yaml` but NOT in [`ModelConstraints`], so -/// the GDN `d_attn` cannot be derived from the config type as it stands. The -/// equation is implemented; the 9B invariant is not verified. +/// `QE2E-INV-001` wants `P(Qwen3.5-9B) ∈ [9.0B, 9.2B]`, and it is still NOT +/// discharged — but for a different reason than before #3346. +/// +/// The arithmetic is now verified against a real file: fed the configuration of +/// `Qwen3.5-0.8B-Q4_K_M.gguf`, [`hybrid_layers`] + this function reproduce that +/// file's 320-tensor inventory EXACTLY (752,393,024 parameters). The Gated +/// DeltaNet shape reaches it through [`ModelConstraints::deltanet`]. +/// +/// Applying the same, now-falsified, arithmetic to the 9b variant of +/// `contracts/model-families/qwen3_5.yaml` gives **8,344,907,136** — 0.655B +/// below the range. The remaining gap is in the DESCRIPTOR, not here, and is +/// not something this function may paper over: see +/// `model_arithmetic_tests.rs::qwen35_9b_hybrid_layers_still_fall_short_of_the_invariant_range`. #[must_use] pub fn model_parameter_count( size: &ModelSizeConfig, diff --git a/crates/aprender-core/src/format/model_arithmetic_tests.rs b/crates/aprender-core/src/format/model_arithmetic_tests.rs index 27ca64e222..7b8df3e938 100644 --- a/crates/aprender-core/src/format/model_arithmetic_tests.rs +++ b/crates/aprender-core/src/format/model_arithmetic_tests.rs @@ -4,7 +4,9 @@ //! the formula changes) and, where the contract states one, one property. use super::*; -use crate::format::model_family::{Activation, AttentionType, NormType, PositionalEncoding}; +use crate::format::model_family::{ + Activation, AttentionType, DeltaNetShape, NormType, PositionalEncoding, +}; /// A four-dimension toy model whose parameter count is small enough to verify /// by hand: V=10, d=4, L=2, n_h=2, n_kv=1, d_k=2, d_ff=8. @@ -52,6 +54,13 @@ fn qwen35_constraints() -> ModelConstraints { positional_encoding: PositionalEncoding::Rope, mlp_type: MlpType::SwiGlu, qk_norm: true, + deltanet: Some(DeltaNetShape { + inner_size: 2048, + state_size: 128, + conv_kernel: 4, + group_count: 8, + full_attention_interval: 4, + }), } } @@ -176,20 +185,100 @@ fn qwen35_0_8b_measured_inventory_sums_to_the_file_total() { assert_eq!(total, QWEN35_0_8B_MEASURED_TOTAL); } -/// The defect of #3346, as a number rather than a claim: dense/GQA accounting -/// applied to a hybrid family under-counts a REAL file by 107,992,896 -/// parameters — 14.4% of the model. Three quarters of the layers are Gated -/// DeltaNet, and none of their conv/gate/state tensors have a term here. +/// The defect of #3346, as a number rather than a claim: one layer kind applied +/// to all 24 layers cannot reproduce a REAL hybrid file. +/// +/// Before #3346 this shortfall was 107,992,896 (14.4%) against the then-dense +/// `attention_layer_params`. The dense baseline is gone — that function now +/// models the gated q projection and the q/k norms the file actually has — so +/// what is left to measure is the mixer itself: 18 of the 24 layers are Gated +/// DeltaNet, and at these dimensions a DeltaNet layer is LARGER than an +/// attention layer (21,555,360 against 18,352,640), so uniform accounting is +/// short by 57,648,960. +/// +/// The sign is not universal, which is the point: at the 9b descriptor's +/// dimensions the same comparison inverts (see +/// `qwen35_9b_uniform_layers_are_the_wrong_model_for_a_hybrid_family`), because +/// that descriptor keeps `inner_size: 2048` while quadrupling `hidden_dim`. +/// Only the real schedule gets a hybrid model right. #[test] -fn dense_accounting_cannot_reproduce_the_measured_qwen35_0_8b_file() { +fn uniform_accounting_cannot_reproduce_the_measured_qwen35_0_8b_file() { let size = qwen35_0_8b_size(); - let mut constraints = qwen35_constraints(); - constraints.tied_embeddings = true; // measured: the file has no output.weight - let layers = uniform_layers(&size, &constraints); + let constraints = qwen35_0_8b_constraints(); + let p = model_parameter_count(&size, &constraints, &uniform_layers(&size, &constraints)); + + assert_eq!(p.total, 694_744_064); + assert_eq!(QWEN35_0_8B_MEASURED_TOTAL - p.total, 57_648_960); + assert!( + tensor_elements(QWEN35_0_8B_GDN_LAYER) > tensor_elements(QWEN35_0_8B_ATTENTION_LAYER), + "at 0.8B dims the DeltaNet layer is the bigger of the two" + ); +} + +/// The same constraints, but as the MEASURED 0.8B file declares them: its +/// `ssm.group_count` is 16 (not the 9b descriptor's 8), it ties its +/// unembedding, and its full-attention layers carry q/k norms. +fn qwen35_0_8b_constraints() -> ModelConstraints { + ModelConstraints { + tied_embeddings: true, + deltanet: Some(DeltaNetShape { + inner_size: 2048, + state_size: 128, + conv_kernel: 4, + group_count: 16, + full_attention_interval: 4, + }), + ..qwen35_constraints() + } +} + +/// #3346 acceptance. Fed the 0.8B configuration, the equation must reproduce +/// the measured file EXACTLY — not approximately, and not after tuning a +/// constant. Both layer kinds are checked separately so a failure names which +/// block is mis-shaped rather than only that the total drifted. +#[test] +fn qwen35_0_8b_config_derived_count_equals_the_measured_gguf_inventory() { + let size = qwen35_0_8b_size(); + let constraints = qwen35_0_8b_constraints(); + let shape = constraints + .deltanet + .expect("the 0.8B constraints declare a DeltaNet shape"); + + assert_eq!( + gated_deltanet_layer_params(&size, &constraints, &shape).total(), + tensor_elements(QWEN35_0_8B_GDN_LAYER), + "Gated DeltaNet layer" + ); + assert_eq!( + attention_layer_params(&size, &constraints).total(), + tensor_elements(QWEN35_0_8B_ATTENTION_LAYER), + "full-attention layer" + ); + + let layers = hybrid_layers(&size, &constraints); + assert_eq!(layers.len(), 24); let p = model_parameter_count(&size, &constraints, &layers); + assert_eq!(p.total, QWEN35_0_8B_MEASURED_TOTAL); +} - assert_eq!(p.total, 644_400_128); - assert_eq!(QWEN35_0_8B_MEASURED_TOTAL - p.total, 107_992_896); +/// The hybrid schedule is measured, not assumed: in the real file the layers +/// carrying `attn_q`/`attn_k`/`attn_v` are exactly 3, 7, 11, 15, 19, 23 — every +/// `full_attention_interval`-th layer, counting the LAST of each group. +#[test] +fn the_hybrid_schedule_puts_full_attention_last_in_each_group() { + let size = qwen35_0_8b_size(); + let constraints = qwen35_0_8b_constraints(); + let shape = constraints.deltanet.expect("declared"); + let attn = attention_layer_params(&size, &constraints); + let full: Vec = hybrid_layers(&size, &constraints) + .iter() + .enumerate() + .filter(|(_, l)| **l == attn) + .map(|(i, _)| i) + .collect(); + + assert_eq!(full, vec![3, 7, 11, 15, 19, 23]); + assert_eq!(shape.full_attention_interval, 4); } // --------------------------------------------------------------------------- @@ -199,13 +288,17 @@ fn dense_accounting_cannot_reproduce_the_measured_qwen35_0_8b_file() { #[test] fn worked_example_attention_layer_params() { let p = attention_layer_params(&toy_size(), &qwen35_constraints()); - // d_attn = d*q_dim + 2*d*kv_dim + q_dim*d = 4*4 + 2*4*2 + 4*4 = 48 - assert_eq!(p.d_attn, 48); + // Qwen3.5 gates the attention output, so q_out = 2*q_dim (MEASURED: + // attn_q [1024, 4096] against attn_output [2048, 1024] in the 0.8B file), + // and qk_norm adds one d_k vector each for attn_q_norm/attn_k_norm. + // d_attn = d*q_out + 2*d*kv_dim + q_dim*d + 2*d_k + // = 4*8 + 2*4*2 + 4*4 + 2*2 = 68 + assert_eq!(p.d_attn, 68); // d_ffn = 3*d*d_ff = 3*4*8 = 96 assert_eq!(p.d_ffn, 96); // d_norm = 2*d = 8 assert_eq!(p.d_norm, 8); - assert_eq!(p.total(), 152); + assert_eq!(p.total(), 172); } #[test] @@ -216,10 +309,10 @@ fn worked_example_model_parameter_count() { let p = model_parameter_count(&size, &constraints, &layers); assert_eq!(p.embedding, 40); // V*d = 10*4 - assert_eq!(p.layers, 304); // L*(48+96+8) = 2*152 + assert_eq!(p.layers, 344); // L*(68+96+8) = 2*172 assert_eq!(p.final_norm, 4); // d_final = d assert_eq!(p.unembedding, 40); // untied lm_head = V*d - assert_eq!(p.total, 388); // P = 40 + 304 + 4 + 40 + assert_eq!(p.total, 428); // P = 40 + 344 + 4 + 40 } #[test] @@ -231,7 +324,7 @@ fn tied_embeddings_drop_the_trailing_v_times_d() { let p = model_parameter_count(&size, &constraints, &layers); assert_eq!(p.unembedding, 0); - assert_eq!(p.total, 348); // 388 - 40 + assert_eq!(p.total, 388); // 428 - 40 } #[test] @@ -241,29 +334,67 @@ fn bias_adds_exactly_the_four_projection_bias_vectors() { constraints.has_bias = true; let p = attention_layer_params(&size, &constraints); // q_dim + 2*kv_dim + d = 4 + 4 + 4 = 12 - assert_eq!(p.d_attn, 48 + 12); + assert_eq!(p.d_attn, 68 + 12); } -/// QE2E-INV-001 wants `P(Qwen3.5-9B) ∈ [9.0B, 9.2B]`. Dense/GQA accounting -/// gives 8.21B, because three of every four Qwen3.5 layers are Gated DeltaNet -/// and their `d_attn` is sized by `inner_size`/`state_size`/`conv_kernel`/ -/// `group_count`, none of which exist in `ModelConstraints`. This test pins the -/// number so the gap is a measured fact, not a claim. +/// The premise this test used to carry — "the four DeltaNet keys do not exist +/// in `ModelConstraints`" — stopped being true in #3346, so it states what is +/// true now: [`uniform_layers`] is the WRONG model for a hybrid family in +/// either direction. Pretending all 32 layers run softmax attention OVER-counts +/// the mixer (8.745B) exactly as pretending they are all dense under-counted it +/// before; only [`hybrid_layers`] describes the architecture. #[test] -fn qwen35_9b_uniform_layers_do_not_reach_the_invariant_range() { +fn qwen35_9b_uniform_layers_are_the_wrong_model_for_a_hybrid_family() { let size = qwen35_9b_size(); let constraints = qwen35_constraints(); - let layers = uniform_layers(&size, &constraints); + let uniform = model_parameter_count(&size, &constraints, &uniform_layers(&size, &constraints)); + let hybrid = model_parameter_count(&size, &constraints, &hybrid_layers(&size, &constraints)); + + assert_eq!(uniform.embedding, 1_017_118_720); + assert_eq!(uniform.unembedding, 1_017_118_720); + assert_eq!(uniform.total, 8_745_406_464); + assert!( + uniform.total > hybrid.total, + "a full-attention layer is bigger than a DeltaNet layer at 9B dims" + ); +} + +/// QE2E-INV-001 wants `P(Qwen3.5-9B) ∈ [9.0B, 9.2B]`. With the DeltaNet shape +/// carried and the arithmetic falsified against a real file, the 9b descriptor +/// yields **8,344,907,136** — still 0.655B short. The obligation therefore +/// stays UNPROVED, and this test exists to keep it that way: the honest move is +/// to pin the number, not to widen the range until it passes. +/// +/// What is unaccounted for is in the descriptor, and it is visible in the +/// numbers it declares. `inner_size: 2048` is the value the 0.8B file uses at +/// `hidden_dim` 1024, i.e. `2*d`; at the 9b's `hidden_dim` 4096 the same 2048 +/// makes the mixer NARROWER than the residual stream it mixes, and it is also +/// inconsistent with the 9b's own `group_count: 8` (8 * 128 != 2048, whereas +/// the measured 0.8B satisfies 16 * 128 == 2048). Either the 9b `inner_size` +/// and `group_count` were copied from the small variant, or the range was +/// copied from a model card. Only a real Qwen3.5-9B file can tell them apart, +/// and no such file is on this box — see #3346. +#[test] +fn qwen35_9b_hybrid_layers_still_fall_short_of_the_invariant_range() { + let size = qwen35_9b_size(); + let constraints = qwen35_constraints(); + let layers = hybrid_layers(&size, &constraints); let p = model_parameter_count(&size, &constraints, &layers); - assert_eq!(p.embedding, 1_017_118_720); - assert_eq!(p.unembedding, 1_017_118_720); - assert_eq!(p.total, 8_208_519_168); + assert_eq!(layers.len(), 32); + assert_eq!(p.total, 8_344_907_136); assert!( p.total < 9_000_000_000, - "QE2E-INV-001 is NOT discharged by dense accounting: got {}", + "QE2E-INV-001 is still NOT discharged: got {}", p.total ); + + // The descriptor's own self-inconsistency, as a fact rather than a remark. + let shape = constraints.deltanet.expect("9b declares a DeltaNet shape"); + assert!( + !shape.heads_span_the_mixer(), + "9b declares group_count * state_size != inner_size" + ); } /// A hybrid model is a slice of DIFFERENT per-layer costs — the reason the diff --git a/crates/aprender-core/src/format/model_family.rs b/crates/aprender-core/src/format/model_family.rs index ccea26f952..f35192896c 100644 --- a/crates/aprender-core/src/format/model_family.rs +++ b/crates/aprender-core/src/format/model_family.rs @@ -247,6 +247,50 @@ pub struct ModelSizeConfig { pub norm_eps: f64, } +/// #3346: the Gated DeltaNet shape of a hybrid family. +/// +/// `contracts/model-families/qwen3_5.yaml` declares `inner_size`, +/// `state_size`, `conv_kernel`, `group_count` and `full_attention_interval` +/// under `constraints:`, and until #3346 [`ModelConstraints`] carried none of +/// them. A DeltaNet layer's parameters live entirely in these dimensions — the +/// conv, the in/out projections and the gates — so a type that drops them +/// cannot count a hybrid model: dense accounting under-counted the real +/// `Qwen3.5-0.8B-Q4_K_M.gguf` by 14.4%. +/// +/// Tensor names below are the GGUF names of that measured file. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct DeltaNetShape { + /// Width of the DeltaNet mixer — `attn_gate`/`ssm_out` are `inner_size` + /// wide and `attn_qkv`/`ssm_conv1d` are `3 * inner_size` wide. + pub inner_size: usize, + /// Per-head state width; `ssm_norm` is a vector of this length. + pub state_size: usize, + /// Depthwise conv width: `ssm_conv1d` is `[conv_kernel, 3 * inner_size]`. + pub conv_kernel: usize, + /// Number of DeltaNet heads — the length of `ssm_a` and `ssm_dt.bias`, and + /// the output width of `ssm_alpha`/`ssm_beta`. + pub group_count: usize, + /// Period of the hybrid schedule: every `full_attention_interval`-th layer + /// runs softmax attention instead, the LAST of each group (measured: layers + /// 3, 7, 11, 15, 19, 23 of 24 at interval 4). + pub full_attention_interval: usize, +} + +impl DeltaNetShape { + /// `group_count * state_size`, which must equal `inner_size` for the + /// declared shape to describe one mixer. + /// + /// This is a CHECK, not a repair: the measured 0.8B satisfies it + /// (16 * 128 = 2048) and the 9b descriptor does not (8 * 128 != 2048). + #[must_use] + pub const fn heads_span_the_mixer(&self) -> bool { + match self.group_count.checked_mul(self.state_size) { + Some(span) => span == self.inner_size, + None => false, + } + } +} + /// Architectural constraints for a model family. #[derive(Debug, Clone)] pub struct ModelConstraints { @@ -259,6 +303,9 @@ pub struct ModelConstraints { pub mlp_type: MlpType, /// GH-280: Whether Q and K projections have per-head RMSNorm (e.g., Qwen3) pub qk_norm: bool, + /// #3346: Gated DeltaNet shape, for hybrid families that declare one. + /// `None` for every family whose descriptor has no `inner_size`. + pub deltanet: Option, } /// Tensor name template for a model family. diff --git a/crates/aprender-core/src/format/model_family_contract_falsify.rs b/crates/aprender-core/src/format/model_family_contract_falsify.rs index d801f98337..a9bb70b243 100644 --- a/crates/aprender-core/src/format/model_family_contract_falsify.rs +++ b/crates/aprender-core/src/format/model_family_contract_falsify.rs @@ -869,6 +869,56 @@ mod contract_falsification { ); } + // ======================================================================== + // FALSIFY-MF-QWEN35-010: the Gated DeltaNet shape keys survive the loader + // + // Prediction: loading contracts/model-families/qwen3_5.yaml yields + // constraints.deltanet = Some(inner 2048, state 128, conv 4, + // group 8, interval 4) — the values the descriptor declares — + // and EVERY other family yields None. + // If fails: the keys are declared in YAML and dropped on the way into + // ModelConstraints, which is the #3346 defect. That drop made the + // arithmetic under-count a real Qwen3.5 file by 14.4%, because a + // DeltaNet layer's parameters live entirely in these dimensions. + // ======================================================================== + #[test] + fn falsify_mf_qwen35_010_deltanet_shape_reaches_constraints() { + let families = load_all_families(); + let qwen35 = families + .iter() + .find(|(name, _)| name == "qwen3_5") + .expect("FALSIFIED: qwen3_5 family not found"); + + let shape = qwen35 + .1 + .constraints + .deltanet + .expect("FALSIFIED QWEN35-010: qwen3_5 declares inner_size/state_size, got None"); + + assert_eq!( + shape, + DeltaNetShape { + inner_size: 2048, + state_size: 128, + conv_kernel: 4, + group_count: 8, + full_attention_interval: 4, + }, + "FALSIFIED QWEN35-010: loader did not reproduce the declared shape" + ); + + // No other family declares a DeltaNet mixer, so none may acquire one: + // a false Some() here would change that family's parameter accounting. + for (name, config) in &families { + if name != "qwen3_5" { + assert!( + config.constraints.deltanet.is_none(), + "FALSIFIED QWEN35-010: {name} acquired a DeltaNet shape it never declared" + ); + } + } + } + // ======================================================================== // FALSIFY-MF-QWEN35-007: Qwen3.5 architecture class registered // diff --git a/crates/aprender-core/src/format/model_family_loader.rs b/crates/aprender-core/src/format/model_family_loader.rs index 76e1dc996b..0d792337fe 100644 --- a/crates/aprender-core/src/format/model_family_loader.rs +++ b/crates/aprender-core/src/format/model_family_loader.rs @@ -13,8 +13,8 @@ use std::path::Path; use crate::error::{AprenderError, Result}; use crate::format::model_family::{ - Activation, AttentionType, CertificationConfig, ChatTemplateConfig, DynModelFamily, - FamilyRegistry, GgufFusionRule, GgufTensorTemplate, MlpType, ModelConstraints, + Activation, AttentionType, CertificationConfig, ChatTemplateConfig, DeltaNetShape, + DynModelFamily, FamilyRegistry, GgufFusionRule, GgufTensorTemplate, MlpType, ModelConstraints, ModelFamilyConfig, ModelSizeConfig, NormType, PositionalEncoding, ShapeTemplate, TensorTemplate, }; diff --git a/crates/aprender-core/src/format/model_family_tests.rs b/crates/aprender-core/src/format/model_family_tests.rs index 327719fefc..95992f411f 100644 --- a/crates/aprender-core/src/format/model_family_tests.rs +++ b/crates/aprender-core/src/format/model_family_tests.rs @@ -122,6 +122,7 @@ mod tests { positional_encoding: PositionalEncoding::Rope, mlp_type: MlpType::SwiGlu, qk_norm: false, + deltanet: None, }, tensor_template: TensorTemplate { embedding: "model.embed_tokens.weight".to_string(), diff --git a/crates/aprender-core/src/format/parsing.rs b/crates/aprender-core/src/format/parsing.rs index 6ae3911bef..cdfca2f99a 100644 --- a/crates/aprender-core/src/format/parsing.rs +++ b/crates/aprender-core/src/format/parsing.rs @@ -73,6 +73,24 @@ fn parse_constraints(yaml: &YamlValue) -> Result { )?, mlp_type: MlpType::from_str_contract(yaml.get_str("mlp_type").unwrap_or("gelu_mlp"))?, qk_norm: yaml.get_bool("qk_norm").unwrap_or(false), + deltanet: parse_deltanet_shape(yaml), + }) +} + +/// #3346: the Gated DeltaNet shape keys of a `constraints:` block. +/// +/// `inner_size` and `state_size` are what make the block a DeltaNet mixer, so +/// both are required; a descriptor declaring neither (every family but +/// `qwen3_5`) yields `None` and the dense accounting it had before. +fn parse_deltanet_shape(yaml: &YamlValue) -> Option { + let inner_size = yaml.get_usize("inner_size")?; + let state_size = yaml.get_usize("state_size")?; + Some(DeltaNetShape { + inner_size, + state_size, + conv_kernel: yaml.get_usize("conv_kernel").unwrap_or(0), + group_count: yaml.get_usize("group_count").unwrap_or(0), + full_attention_interval: yaml.get_usize("full_attention_interval").unwrap_or(0), }) } From 6acc27be6a776da0d6b02499bfa8b36c1c4a7ad0 Mon Sep 17 00:00:00 2001 From: Noah Gift Date: Wed, 16 Sep 2026 09:48:09 +0200 Subject: [PATCH 3/7] contracts(binding): the QE2E-INV-001 note blamed a gap that is now closed The note said the obligation was undischarged because ModelConstraints does not carry the DeltaNet shape keys. It does now, and the 0.8B count reproduces the real GGUF exactly. What actually blocks the obligation is descriptor drift at the 9b variant. Pmat-Ticket: PMAT-3346 Co-Authored-By: Claude Opus 5 (1M context) --- contracts/binding.yaml | 2 +- docs/roadmaps/roadmap.yaml | 16 ---------------- 2 files changed, 1 insertion(+), 17 deletions(-) diff --git a/contracts/binding.yaml b/contracts/binding.yaml index c4a9520ec8..09a3e91119 100644 --- a/contracts/binding.yaml +++ b/contracts/binding.yaml @@ -764,7 +764,7 @@ bindings: notes: "P = V*d + L*(d_attn + d_ffn + d_norm) + d_final + V*d. Per-layer terms come in as data\ \ (&[LayerParams]) so a hybrid family can mix layer kinds. QE2E-INV-001 (P(9B) in [9.0B, 9.2B])\ \ is NOT discharged: dense/GQA accounting gives 8.21B and the Gated DeltaNet d_attn needs\ - \ inner_size/state_size/conv_kernel/group_count, which ModelConstraints does not carry (#3347)." + \ inner_size/state_size/conv_kernel/group_count, which the 9b descriptor disagrees with itself (inner_size 2048 at hidden_dim 4096, and group_count 8 fails group_count*state_size == inner_size), so P(9B) computes to 8.345B; settling it needs a real Qwen3.5-9B GGUF (#3346) - contract: qwen35-e2e-verification-v1.yaml equation: flops_per_token module_path: aprender::format::model_arithmetic diff --git a/docs/roadmaps/roadmap.yaml b/docs/roadmaps/roadmap.yaml index 7a960ee709..3a90bd6297 100644 --- a/docs/roadmaps/roadmap.yaml +++ b/docs/roadmaps/roadmap.yaml @@ -18124,22 +18124,6 @@ roadmap: estimated_effort: null labels: [] notes: null -- id: PMAT-3346 - github_issue: 3346 - item_type: task - title: ModelConstraints carries the gated-DeltaNet shape keys - status: planned - priority: medium - assigned_to: null - created: 2026-09-16T07:21:39Z - updated: 2026-09-16T07:21:39Z - spec: null - acceptance_criteria: [] - phases: [] - subtasks: [] - estimated_effort: null - labels: [] - notes: null - id: PMAT-3347 github_issue: 3347 item_type: task From da1ee1506cd006dbc3154196445f05a2835e8501 Mon Sep 17 00:00:00 2001 From: Noah Gift Date: Mon, 21 Sep 2026 00:37:11 +0200 Subject: [PATCH 4/7] PMAT-3346 (adoption): the QE2E-INV-001 note lost its closing quote, and pv extract shrank the graph by 356 triples instead of refusing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit d3cc76f1f rewrote the note on the QE2E-INV-001 binding and dropped the trailing `"` — contracts/binding.yaml stopped being valid YAML at line 764 (`found unexpected end of stream`). Nothing in the PR noticed because `pv extract contracts` does not refuse a binding registry that will not parse: it emitted a graph with 15,244 triples where main has 15,600 — every bound symbol AFTER the broken entry (prune::run, distill::run, harness_ir::*, ptx_explain::run, …) silently gone — and `--check` would have agreed with itself. Found while regenerating the derivative for this adoption, by the drop, not by any gate. One character. With it, binding.yaml parses (156 entries, same as main) and the extraction is byte-identical to main's committed contracts.nt, so this PR owes no graph change after all. Refs #3346, #3350 Co-Authored-By: Claude Opus 5 (1M context) --- contracts/binding.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/contracts/binding.yaml b/contracts/binding.yaml index 09a3e91119..1391a732f4 100644 --- a/contracts/binding.yaml +++ b/contracts/binding.yaml @@ -764,7 +764,7 @@ bindings: notes: "P = V*d + L*(d_attn + d_ffn + d_norm) + d_final + V*d. Per-layer terms come in as data\ \ (&[LayerParams]) so a hybrid family can mix layer kinds. QE2E-INV-001 (P(9B) in [9.0B, 9.2B])\ \ is NOT discharged: dense/GQA accounting gives 8.21B and the Gated DeltaNet d_attn needs\ - \ inner_size/state_size/conv_kernel/group_count, which the 9b descriptor disagrees with itself (inner_size 2048 at hidden_dim 4096, and group_count 8 fails group_count*state_size == inner_size), so P(9B) computes to 8.345B; settling it needs a real Qwen3.5-9B GGUF (#3346) + \ inner_size/state_size/conv_kernel/group_count, which the 9b descriptor disagrees with itself (inner_size 2048 at hidden_dim 4096, and group_count 8 fails group_count*state_size == inner_size), so P(9B) computes to 8.345B; settling it needs a real Qwen3.5-9B GGUF (#3346)." - contract: qwen35-e2e-verification-v1.yaml equation: flops_per_token module_path: aprender::format::model_arithmetic From 1ffdccf2500116a4dd39af6a66a697f7bb841136 Mon Sep 17 00:00:00 2001 From: Noah Gift Date: Mon, 21 Sep 2026 00:54:39 +0200 Subject: [PATCH 5/7] PMAT-3346: the roadmap row, so the AD-04 quorum for #3350 has a ticket to judge against Acceptance transcribed from issue #3346 as this PR answers it (the type that reads the descriptor was wrong, not the range or the descriptor), with the 9B range instantiation explicitly out of scope until a real 9B GGUF exists. Refs #3346 Co-Authored-By: Claude Opus 5 (1M context) --- docs/roadmaps/entries/PMAT-3346.yaml | 17 +++++++++++++++++ docs/roadmaps/roadmap.yaml | 17 +++++++++++++++++ 2 files changed, 34 insertions(+) create mode 100644 docs/roadmaps/entries/PMAT-3346.yaml diff --git a/docs/roadmaps/entries/PMAT-3346.yaml b/docs/roadmaps/entries/PMAT-3346.yaml new file mode 100644 index 0000000000..71ee6e8fc0 --- /dev/null +++ b/docs/roadmaps/entries/PMAT-3346.yaml @@ -0,0 +1,17 @@ +- id: PMAT-3346 + github_issue: 3346 + item_type: task + title: 'QE2E-INV-001: ModelConstraints dropped the gated-DeltaNet shape, so 18 of every 24 Qwen3.5 layers were counted as if their tensors did not exist (#3346)' + status: in_progress + priority: high + assigned_to: null + created: 2026-09-20T22:54:24Z + updated: 2026-09-20T22:54:24Z + spec: null + acceptance_criteria: [] + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'ACCEPTANCE, from issue #3346 ("decide which side is wrong, with a measurement") as PR #3350 answers it — the answer is neither the range nor the descriptor but the TYPE that reads the descriptor: (1) ModelConstraints carries inner_size, state_size, conv_kernel, group_count and full_attention_interval from contracts/model-families/qwen3_5.yaml, which it previously dropped; (2) the config-derived model_parameter_count equals the measured GGUF tensor sum of a REAL file — ~/models/Qwen3.5-0.8B-Q4_K_M.gguf, 320 tensors, 752,393,024 parameters, delta 0 — with the gated-DeltaNet layer (21,555,360) and the full-attention layer (18,352,640) each matching the measured blk; (3) the two shapes that contradict dense accounting are modelled from the tensors, not assumed: attn_q is [d, 2*n_h*d_k] (the q projection emits the output gate) and the DeltaNet mixer projections/conv/state norm are counted for the 18-of-24 linear layers; (4) the 9B range instantiation is NOT asserted in this PR — the repo 9b descriptor disagrees with itself (inner_size 2048 at hidden 4096; group_count 8 fails group_count*state_size == inner_size) so P(9B) computes to 8.345B, and settling it needs a real Qwen3.5-9B GGUF, recorded in the QE2E-INV-001 binding note; (5) the note edit leaves contracts/binding.yaml valid YAML (156 entries, same as main) and the tracked contracts.nt equals a fresh extraction. Out of scope: acquiring the 9B GGUF; widening the range.' diff --git a/docs/roadmaps/roadmap.yaml b/docs/roadmaps/roadmap.yaml index 3a90bd6297..aee1ca18c2 100644 --- a/docs/roadmaps/roadmap.yaml +++ b/docs/roadmaps/roadmap.yaml @@ -18124,6 +18124,23 @@ roadmap: estimated_effort: null labels: [] notes: null +- id: PMAT-3346 + github_issue: 3346 + item_type: task + title: 'QE2E-INV-001: ModelConstraints dropped the gated-DeltaNet shape, so 18 of every 24 Qwen3.5 layers were counted as if their tensors did not exist (#3346)' + status: in_progress + priority: high + assigned_to: null + created: 2026-09-20T22:54:24Z + updated: 2026-09-20T22:54:24Z + spec: null + acceptance_criteria: [] + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'ACCEPTANCE, from issue #3346 ("decide which side is wrong, with a measurement") as PR #3350 answers it — the answer is neither the range nor the descriptor but the TYPE that reads the descriptor: (1) ModelConstraints carries inner_size, state_size, conv_kernel, group_count and full_attention_interval from contracts/model-families/qwen3_5.yaml, which it previously dropped; (2) the config-derived model_parameter_count equals the measured GGUF tensor sum of a REAL file — ~/models/Qwen3.5-0.8B-Q4_K_M.gguf, 320 tensors, 752,393,024 parameters, delta 0 — with the gated-DeltaNet layer (21,555,360) and the full-attention layer (18,352,640) each matching the measured blk; (3) the two shapes that contradict dense accounting are modelled from the tensors, not assumed: attn_q is [d, 2*n_h*d_k] (the q projection emits the output gate) and the DeltaNet mixer projections/conv/state norm are counted for the 18-of-24 linear layers; (4) the 9B range instantiation is NOT asserted in this PR — the repo 9b descriptor disagrees with itself (inner_size 2048 at hidden 4096; group_count 8 fails group_count*state_size == inner_size) so P(9B) computes to 8.345B, and settling it needs a real Qwen3.5-9B GGUF, recorded in the QE2E-INV-001 binding note; (5) the note edit leaves contracts/binding.yaml valid YAML (156 entries, same as main) and the tracked contracts.nt equals a fresh extraction. Out of scope: acquiring the 9B GGUF; widening the range.' - id: PMAT-3347 github_issue: 3347 item_type: task From e060c7e2c22bc221b5e454a9b677d09d2b27c2e2 Mon Sep 17 00:00:00 2001 From: Noah Gift Date: Mon, 21 Sep 2026 01:21:10 +0200 Subject: [PATCH 6/7] PMAT-3346 (adoption): a bias vector is as wide as the projection it biases MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by the AD-04 quorum on #3350 (lane 1, gemini-3.1-pro-high, cited model_arithmetic.rs:144): the projection term used q_out for a gated family's q matrix (2*n_h*d_k — attn_q emits the output gate, MEASURED in Qwen3.5-0.8B) while the bias term still used q_dim. A bias narrower than its projection is not a model. One token: q_dim -> q_out in the bias sum. Why a delta-0 measurement did not catch it: no shipped family exercises the case. Qwen3.5 has no attention bias; Qwen2.5 has biases but is not gated, so q_out == q_dim there. The test that pinned 12 (a q_dim bias under a q_out matrix) now asserts 16 and says why, and a second test holds the other polarity — a non-gated family with biases is unchanged at 12. 115 model_arithmetic + model_family tests pass; oracle 218; clippy clean. Refs #3346, #3350 Co-Authored-By: Claude Opus 5 (1M context) --- .../src/format/model_arithmetic.rs | 8 +++++- .../src/format/model_arithmetic_tests.rs | 26 +++++++++++++++++-- 2 files changed, 31 insertions(+), 3 deletions(-) diff --git a/crates/aprender-core/src/format/model_arithmetic.rs b/crates/aprender-core/src/format/model_arithmetic.rs index 1441bbb5f9..3c97ae1675 100644 --- a/crates/aprender-core/src/format/model_arithmetic.rs +++ b/crates/aprender-core/src/format/model_arithmetic.rs @@ -134,8 +134,14 @@ pub fn attention_layer_params( .saturating_mul(q_out) .saturating_add(d.saturating_mul(kv_dim).saturating_mul(2)) .saturating_add(q_dim.saturating_mul(d)); + // A bias vector is as wide as the projection it biases, so the q bias is + // q_out — doubled with the matrix for a gated family. No shipped family + // exercises this today (Qwen3.5 has no attention bias; Qwen2.5 has biases + // but is not gated, so q_out == q_dim), which is exactly how the original + // `q_dim` here survived a delta-0 measurement: found by the AD-04 quorum + // on #3350, not by any model. let biases = if constraints.has_bias { - q_dim + q_out .saturating_add(kv_dim.saturating_mul(2)) .saturating_add(d) } else { diff --git a/crates/aprender-core/src/format/model_arithmetic_tests.rs b/crates/aprender-core/src/format/model_arithmetic_tests.rs index 7b8df3e938..941925e46b 100644 --- a/crates/aprender-core/src/format/model_arithmetic_tests.rs +++ b/crates/aprender-core/src/format/model_arithmetic_tests.rs @@ -333,8 +333,30 @@ fn bias_adds_exactly_the_four_projection_bias_vectors() { let mut constraints = qwen35_constraints(); constraints.has_bias = true; let p = attention_layer_params(&size, &constraints); - // q_dim + 2*kv_dim + d = 4 + 4 + 4 = 12 - assert_eq!(p.d_attn, 68 + 12); + // qwen35_constraints() is a GATED family, so the q projection is q_out = + // 2*q_dim wide and its bias vector is too: q_out + 2*kv_dim + d + // = 8 + 4 + 4 = 16. This test asserted 12 (a q_dim-wide bias under a + // q_out-wide matrix) until the quorum on #3350 read the two lines against + // each other; a bias narrower than its projection is not a model. + assert_eq!(p.d_attn, 68 + 16); +} + +#[test] +fn a_non_gated_family_with_biases_still_counts_a_q_dim_wide_q_bias() { + // The other polarity: where q_out == q_dim (every non-gated family), the + // fix above must change nothing. + let size = toy_size(); + let mut constraints = qwen35_constraints(); + constraints.attention_type = AttentionType::Gqa; + constraints.has_bias = true; + let p = attention_layer_params(&size, &constraints); + let without = { + let mut c = constraints.clone(); + c.has_bias = false; + attention_layer_params(&size, &c) + }; + // q_dim + 2*kv_dim + d = 4 + 4 + 4 = 12, on top of the un-doubled matrix. + assert_eq!(p.d_attn - without.d_attn, 12); } /// The premise this test used to carry — "the four DeltaNet keys do not exist From 52cd6ce164269f5be69541244d470f1a6d902c37 Mon Sep 17 00:00:00 2001 From: Noah Gift Date: Mon, 21 Sep 2026 01:40:10 +0200 Subject: [PATCH 7/7] PMAT-3346: quorum verdict 3/3 on e060c7e2c (AD-04) Round 0 was 1 FAIL / 1 no-verdict / 1 PASS and the FAIL was real (the bias width, fixed in e060c7e2c). Round 1 on the fixed head: 3/3 PASS, gemini-3.1-pro-high / pro-low / 3.6-flash-high, each measured, no dissent. Refs #3346, #3350 Co-Authored-By: Claude Opus 5 (1M context) --- docs/audits/quorum-PMAT-3346.json | 211 ++++++++++++++++++++++++++++++ 1 file changed, 211 insertions(+) create mode 100644 docs/audits/quorum-PMAT-3346.json diff --git a/docs/audits/quorum-PMAT-3346.json b/docs/audits/quorum-PMAT-3346.json new file mode 100644 index 0000000000..a890916309 --- /dev/null +++ b/docs/audits/quorum-PMAT-3346.json @@ -0,0 +1,211 @@ +{ + "ticket": "PMAT-3346", + "base": "main", + "base_resolved": "origin/main", + "base_note": "local main differs from origin/main by 127 commit(s); judged against origin/main", + "head": "e060c7e2c22bc221b5e454a9b677d09d2b27c2e2", + "diff_sha256": "a9a91b0c1f95b9e18fb6a00b117963990981c632db99f3a4b240f13a321a81ab", + "width": 3, + "executor": "agy", + "prompt_mode": "inline", + "prompt_bytes": 48102, + "author": { + "model": "claude-opus-5", + "family": "claude", + "source": "flag" + }, + "agreed": true, + "lanes": [ + { + "lane": 1, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "I have reviewed the diff against the requirements specified in PMAT-3346. I applied the patch and ran `cargo test -p aprender-core` on the patched workspace. The tests passed, confirming that the arithmetic models both the Gated DeltaNet layer (21,555,360) and full-attention layer (18,352,640) correctly to hit the exact 752,393,024 parameter mark of Qwen3.5-0.8B-Q4_K_M.gguf. The 9B config accurately computes to 8.345B, appropriately documenting the gap and NOT forcefully discharging the constraint. The `binding.yaml` remains valid and unmodified where it matters (156 items). No gates were weakened and all receipt claims are backed by the diff. Therefore, the verdict is PASS.", + "findings": [ + { + "claim": "ModelConstraints carries inner_size, state_size, conv_kernel, group_count, full_attention_interval", + "file": "crates/aprender-core/src/format/model_family.rs", + "grounding": "cited", + "line": 250 + }, + { + "claim": "The config-derived model_parameter_count equals the measured GGUF tensor sum of a REAL file (320 tensors, 752,393,024 parameters, delta 0) with both gated-DeltaNet and full-attention layers matching exactly.", + "command": "cd pmat-3477/aprender && cargo test -p aprender-core", + "file": "crates/aprender-core/src/format/model_arithmetic_tests.rs", + "grounding": "measured", + "line": 242, + "output": "running 280 tests... test qwen35_0_8b_config_derived_count_equals_the_measured_gguf_inventory ... ok" + }, + { + "claim": "The two shapes that contradict dense accounting are modelled from the tensors, not assumed: attn_q is [d, 2*n_h*d_k] (q projection emits output gate) and the DeltaNet mixer projections are counted for 18-of-24 linear layers.", + "file": "crates/aprender-core/src/format/model_arithmetic.rs", + "grounding": "cited", + "line": 121 + }, + { + "claim": "The 9B range instantiation is NOT asserted in this PR; the descriptor disagrees with itself so P(9B) computes to 8.345B, and settling it needs a real Qwen3.5-9B GGUF.", + "command": "cd pmat-3477/aprender && cargo test -p aprender-core", + "file": "crates/aprender-core/src/format/model_arithmetic_tests.rs", + "grounding": "measured", + "line": 389, + "output": "running 280 tests... test qwen35_9b_hybrid_layers_still_fall_short_of_the_invariant_range ... ok" + }, + { + "claim": "The note edit leaves contracts/binding.yaml valid YAML with 156 entries.", + "command": "cat pmat-3477/aprender/contracts/binding.yaml | grep \"^-\" | wc -l", + "file": "contracts/binding.yaml", + "grounding": "measured", + "line": 764, + "output": "156" + } + ], + "raw_bytes": 7208, + "err_bytes": 100, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "grounding_check": "parity", + "model": "gemini-3.1-pro-high", + "model_measured": "gemini-3.1-pro-high", + "model_source": "measured", + "family": "gemini" + }, + { + "lane": 2, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "The diff perfectly implements the requested ticket criteria. The ModelConstraints now carry the delta net shape. The tests verify the GGUF arithmetic perfectly against the given Qwen3.5-0.8B parameters. The 9B range instantiation is left unverified and noted in bindings.yaml. The diff is fully backed by code logic and tests.", + "findings": [], + "raw_bytes": 3680, + "err_bytes": 100, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "grounding_check": "parity", + "model": "gemini-3.1-pro-low", + "model_measured": "gemini-3.1-pro-low", + "model_source": "measured", + "family": "gemini" + }, + { + "lane": 3, + "status": "SUCCESS", + "verdict": "PASS", + "summary": "The diff correctly resolves PMAT-3346 by adding DeltaNetShape to ModelConstraints, updating build codegen and runtime YAML parsing to preserve the Gated DeltaNet shape parameters, and modeling per-layer parameter arithmetic for hybrid Gated DeltaNet architectures. Parameter accounting was empirically verified against the 320-tensor inventory of Qwen3.5-0.8B-Q4_K_M.gguf (752,393,024 parameters, delta 0). All unit tests and contract falsification tests pass. No gates were weakened, no tests assert opposite behavior, and no out-of-scope changes were made.", + "findings": [ + { + "claim": "ModelConstraints now carries DeltaNetShape (inner_size, state_size, conv_kernel, group_count, full_attention_interval) and config-derived parameter count matches the measured 752,393,024 parameters of Qwen3.5-0.8B-Q4_K_M.gguf with 0 delta. All 29 model_arithmetic unit tests pass.", + "command": "CARGO_TARGET_DIR=/tmp/target_3346 cargo test --lib -p aprender-core model_arithmetic", + "file": "crates/aprender-core/src/format/model_arithmetic_tests.rs", + "fix": "Verified passing test suite", + "grounding": "measured", + "line": 198, + "output": "test result: ok. 29 passed; 0 failed; 0 ignored; 0 measured; 14245 filtered out; finished in 0.01s" + }, + { + "claim": "Contract falsification test falsify_mf_qwen35_010_deltanet_shape_reaches_constraints verifies qwen3_5 loads deltanet shape into ModelConstraints while all other families receive None.", + "command": "CARGO_TARGET_DIR=/tmp/target_3346 cargo test --lib -p aprender-core falsify_mf_qwen35_010_deltanet_shape_reaches_constraints", + "file": "crates/aprender-core/src/format/model_family_contract_falsify.rs", + "fix": "Verified passing contract test", + "grounding": "measured", + "line": 885, + "output": "test format::model_family_loader::contract_falsification::falsify_mf_qwen35_010_deltanet_shape_reaches_constraints ... ok\ntest result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 14273 filtered out; finished in 0.01s" + } + ], + "raw_bytes": 6032, + "err_bytes": 0, + "envelope_status": "SUCCESS", + "verdict_source": "structured_output", + "grounding_check": "parity", + "model": "gemini-3.6-flash-high", + "model_measured": "gemini-3.6-flash-high", + "model_source": "measured", + "family": "gemini" + } + ], + "dissent": [], + "dedup": [ + { + "file": "contracts/binding.yaml", + "line": 764, + "lanes_agreeing": [ + 1 + ], + "claims": [ + "The note edit leaves contracts/binding.yaml valid YAML with 156 entries." + ] + }, + { + "file": "crates/aprender-core/src/format/model_arithmetic.rs", + "line": 121, + "lanes_agreeing": [ + 1 + ], + "claims": [ + "The two shapes that contradict dense accounting are modelled from the tensors, not assumed: attn_q is [d, 2*n_h*d_k] (q projection emits output gate) and the DeltaNet mixer projections are counted for 18-of-24 linear layers." + ] + }, + { + "file": "crates/aprender-core/src/format/model_arithmetic_tests.rs", + "line": 198, + "lanes_agreeing": [ + 3 + ], + "claims": [ + "ModelConstraints now carries DeltaNetShape (inner_size, state_size, conv_kernel, group_count, full_attention_interval) and config-derived parameter count matches the measured 752,393,024 parameters of Qwen3.5-0.8B-Q4_K_M.gguf with 0 delta. All 29 model_arithmetic unit tests pass." + ] + }, + { + "file": "crates/aprender-core/src/format/model_arithmetic_tests.rs", + "line": 242, + "lanes_agreeing": [ + 1 + ], + "claims": [ + "The config-derived model_parameter_count equals the measured GGUF tensor sum of a REAL file (320 tensors, 752,393,024 parameters, delta 0) with both gated-DeltaNet and full-attention layers matching exactly." + ] + }, + { + "file": "crates/aprender-core/src/format/model_arithmetic_tests.rs", + "line": 389, + "lanes_agreeing": [ + 1 + ], + "claims": [ + "The 9B range instantiation is NOT asserted in this PR; the descriptor disagrees with itself so P(9B) computes to 8.345B, and settling it needs a real Qwen3.5-9B GGUF." + ] + }, + { + "file": "crates/aprender-core/src/format/model_family.rs", + "line": 250, + "lanes_agreeing": [ + 1 + ], + "claims": [ + "ModelConstraints carries inner_size, state_size, conv_kernel, group_count, full_attention_interval" + ] + }, + { + "file": "crates/aprender-core/src/format/model_family_contract_falsify.rs", + "line": 885, + "lanes_agreeing": [ + 3 + ], + "claims": [ + "Contract falsification test falsify_mf_qwen35_010_deltanet_shape_reaches_constraints verifies qwen3_5 loads deltanet shape into ModelConstraints while all other families receive None." + ] + } + ], + "uncovered": [], + "coverage_source": "lanes", + "partial": false, + "partial_reasons": [], + "auto_merge": { + "checked": true, + "was_armed": false, + "disarmed": false, + "note": "auto-merge not armed" + }, + "lint": { + "ok": true, + "output": "receipt complete: kind=artifact lanes=3 author=claude-opus-5/claude" + } +}