Skip to content
Closed
2 changes: 1 addition & 1 deletion contracts/binding.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -764,7 +764,7 @@ bindings:
notes: "P = V*d + L*(d_attn + d_ffn + d_norm) + d_final + V*d. Per-layer terms come in as data\
\ (&[LayerParams]) so a hybrid family can mix layer kinds. QE2E-INV-001 (P(9B) in [9.0B, 9.2B])\
\ is NOT discharged: dense/GQA accounting gives 8.21B and the Gated DeltaNet d_attn needs\
\ inner_size/state_size/conv_kernel/group_count, which ModelConstraints does not carry (#3347)."
\ inner_size/state_size/conv_kernel/group_count, which the 9b descriptor disagrees with itself (inner_size 2048 at hidden_dim 4096, and group_count 8 fails group_count*state_size == inner_size), so P(9B) computes to 8.345B; settling it needs a real Qwen3.5-9B GGUF (#3346)."
- contract: qwen35-e2e-verification-v1.yaml
equation: flops_per_token
module_path: aprender::format::model_arithmetic
Expand Down
1 change: 1 addition & 0 deletions crates/apr-cli/src/commands/oracle_compute_param_memory.rs
Original file line number Diff line number Diff line change
Expand Up @@ -275,6 +275,7 @@
positional_encoding: PositionalEncoding::Absolute,
mlp_type: MlpType::GeluMlp,
qk_norm: false,
deltanet: None,
},
tensor_template: TensorTemplate {
embedding: "wte.weight".to_string(),
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -273,6 +273,7 @@
positional_encoding: PositionalEncoding::Rope,
mlp_type: MlpType::GeluMlp,
qk_norm: false,
deltanet: None,
};
let params = compute_param_count(&size, &constraints);
assert!(params > 0, "Even minimal model should have params");
Expand Down
3 changes: 3 additions & 0 deletions crates/apr-cli/src/commands/tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,7 @@
positional_encoding: PositionalEncoding::Rope,
mlp_type: MlpType::SwiGlu,
qk_norm: false,
deltanet: None,
},
tensor_template: TensorTemplate {
embedding: "embed.weight".to_string(),
Expand Down Expand Up @@ -160,6 +161,7 @@
positional_encoding: aprender::format::model_family::PositionalEncoding::Rope,
mlp_type: aprender::format::model_family::MlpType::SwiGlu,
qk_norm: false,
deltanet: None,
},
tensor_template: aprender::format::model_family::TensorTemplate {
embedding: String::new(),
Expand Down Expand Up @@ -301,6 +303,7 @@
positional_encoding: PositionalEncoding::Rope,
mlp_type: MlpType::SwiGlu,
qk_norm: false,
deltanet: None,
}
}

Expand Down
15 changes: 15 additions & 0 deletions crates/aprender-core/build_codegen.rs
Original file line number Diff line number Diff line change
Expand Up @@ -204,6 +204,7 @@ fn generate_family_registration(f: &FamilyData) -> String {
\x20 positional_encoding: PositionalEncoding::from_str_contract(\"{}\").unwrap_or(PositionalEncoding::Rope),\n\
\x20 mlp_type: MlpType::from_str_contract(\"{}\").unwrap_or(MlpType::SwiGlu),\n\
\x20 qk_norm: {},\n\
\x20 deltanet: {},\n\
\x20 }},\n\
\x20 tensor_template: TensorTemplate {{\n\
\x20 embedding: \"{}\".to_string(),\n\
Expand Down Expand Up @@ -243,6 +244,7 @@ fn generate_family_registration(f: &FamilyData) -> String {
f.constraints.position,
f.constraints.mlp,
f.constraints.qk_norm,
deltanet_expr(f),
f.embedding_tensor,
f.lm_head_tensor
.as_ref()
Expand Down Expand Up @@ -490,3 +492,16 @@ fn generate_algebraic_proofs(f: &FamilyData) -> String {
out.push('\n');
out
}

/// #3346: render a family's Gated DeltaNet shape as a Rust expression, so the
/// compiled-in registry carries the same keys the runtime YAML parser does.
/// A family that declares none renders `None` and keeps dense accounting.
fn deltanet_expr(f: &FamilyData) -> String {
match &f.constraints.deltanet {
None => "None".to_string(),
Some(d) => format!(
"Some(DeltaNetShape {{ inner_size: {}, state_size: {}, conv_kernel: {}, group_count: {}, full_attention_interval: {} }})",
d.inner_size, d.state_size, d.conv_kernel, d.group_count, d.full_attention_interval
),
}
}
29 changes: 29 additions & 0 deletions crates/aprender-core/build_parsing.rs
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,34 @@ struct ConstraintsData {
position: String,
mlp: String,
qk_norm: bool,
/// #3346: Gated DeltaNet shape keys. `None` unless the descriptor declares
/// both `inner_size` and `state_size`.
deltanet: Option<DeltaNetData>,
}

/// #3346: the `inner_size`/`state_size`/`conv_kernel`/`group_count` +
/// `full_attention_interval` block of a hybrid family's `constraints:`.
struct DeltaNetData {
inner_size: usize,
state_size: usize,
conv_kernel: usize,
group_count: usize,
full_attention_interval: usize,
}

/// #3346: read the Gated DeltaNet shape out of a `constraints:` section.
/// Both `inner_size` and `state_size` are required — they are what make the
/// block a DeltaNet mixer — so every other family yields `None`.
fn parse_deltanet_data(section: &str) -> Option<DeltaNetData> {
let inner_size = get_usize(section, "inner_size")?;
let state_size = get_usize(section, "state_size")?;
Some(DeltaNetData {
inner_size,
state_size,
conv_kernel: get_usize(section, "conv_kernel").unwrap_or(0),
group_count: get_usize(section, "group_count").unwrap_or(0),
full_attention_interval: get_usize(section, "full_attention_interval").unwrap_or(0),
})
}

// ============================================================================
Expand Down Expand Up @@ -144,6 +172,7 @@ fn parse_family_yaml(content: &str, path: &Path) -> FamilyData {
position: c_str("positional_encoding", "rope"),
mlp: c_str("mlp_type", "swiglu"),
qk_norm: get_bool(&constraints_section, "qk_norm").unwrap_or(false),
deltanet: parse_deltanet_data(&constraints_section),
};

// Parse tensor_template
Expand Down
158 changes: 140 additions & 18 deletions crates/aprender-core/src/format/model_arithmetic.rs
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,9 @@
//! does not settle, so it is left unbound and `QE2E-BND-005` stays false.

use crate::format::layout_contract::block_sizes;
use crate::format::model_family::{MlpType, ModelConstraints, ModelSizeConfig};
use crate::format::model_family::{
AttentionType, DeltaNetShape, MlpType, ModelConstraints, ModelSizeConfig,
};

// ============================================================================
// Equation: model_parameter_count
Expand Down Expand Up @@ -96,13 +98,16 @@ pub struct ParameterBreakdown {
/// `d_attn`, `d_ffn`, `d_norm` for one ordinary (softmax-attention) decoder
/// layer of a family described by `size` + `constraints`.
///
/// - `d_attn` = `d*(n_h*d_k) + 2*d*(n_kv*d_k) + (n_h*d_k)*d` (Q, K, V, O),
/// plus the four bias vectors when `constraints.has_bias`.
/// - `d_attn` = `d*q_out + 2*d*(n_kv*d_k) + (n_h*d_k)*d` (Q, K, V, O), plus the
/// four bias vectors when `constraints.has_bias`, plus `2*d_k` of q/k norm
/// weights when `constraints.qk_norm`. `q_out` is `n_h*d_k`, or twice that
/// for a gated-attention family (see the comment on the `q_out` binding).
/// - `d_ffn` = `3*d*d_ff` for a gated MLP (SwiGLU/GeGLU), else `2*d*d_ff`.
/// - `d_norm` = `2*d` (input norm + post-attention norm).
///
/// This is the dense/GQA accounting. It is NOT the Gated DeltaNet accounting:
/// see [`model_parameter_count`] for what that needs.
/// This is the softmax-attention layer. It is NOT the Gated DeltaNet
/// accounting: that is [`gated_deltanet_layer_params`], and
/// [`hybrid_layers`] interleaves the two.
#[must_use]
pub fn attention_layer_params(
size: &ModelSizeConfig,
Expand All @@ -113,32 +118,144 @@ pub fn attention_layer_params(
let q_dim = (size.num_heads as u64).saturating_mul(d_k);
let kv_dim = (size.num_kv_heads as u64).saturating_mul(d_k);

// A gated-attention family emits the output gate from the q projection, so
// that matrix is 2*n_h*d_k wide rather than n_h*d_k. MEASURED in
// Qwen3.5-0.8B-Q4_K_M.gguf: attn_q is [1024, 4096] while attn_output is
// [2048, 1024], so o_proj still sees n_h*d_k = 2048 and only q is doubled.
let q_out = if matches!(
constraints.attention_type,
AttentionType::HybridGatedDeltaNet
) {
q_dim.saturating_mul(2)
} else {
q_dim
};
let projections = d
.saturating_mul(q_dim)
.saturating_mul(q_out)
.saturating_add(d.saturating_mul(kv_dim).saturating_mul(2))
.saturating_add(q_dim.saturating_mul(d));
// A bias vector is as wide as the projection it biases, so the q bias is
// q_out — doubled with the matrix for a gated family. No shipped family
// exercises this today (Qwen3.5 has no attention bias; Qwen2.5 has biases
// but is not gated, so q_out == q_dim), which is exactly how the original
// `q_dim` here survived a delta-0 measurement: found by the AD-04 quorum
// on #3350, not by any model.
let biases = if constraints.has_bias {
q_dim
q_out
.saturating_add(kv_dim.saturating_mul(2))
.saturating_add(d)
} else {
0
};
// Per-head q/k RMSNorm weights (attn_q_norm/attn_k_norm, one d_k vector each).
let qk_norms = if constraints.qk_norm {
d_k.saturating_mul(2)
} else {
0
};

let d_ff = size.intermediate_dim as u64;
LayerParams {
d_attn: projections.saturating_add(biases).saturating_add(qk_norms),
d_ffn: ffn_params(size, constraints),
d_norm: d.saturating_mul(2),
}
}

/// `d_ffn` for one layer: `3*d*d_ff` for a gated MLP (SwiGLU/GeGLU — gate, up
/// and down), else `2*d*d_ff`. Both layer kinds of a hybrid model share it.
fn ffn_params(size: &ModelSizeConfig, constraints: &ModelConstraints) -> u64 {
let matrices = if matches!(constraints.mlp_type, MlpType::SwiGlu | MlpType::GatedMlp) {
3
} else {
2
};
(size.hidden_dim as u64)
.saturating_mul(size.intermediate_dim as u64)
.saturating_mul(matrices)
}

/// `d_attn`, `d_ffn`, `d_norm` for one **Gated DeltaNet** layer — the `d_attn`
/// the dense formula cannot express, because none of its dimensions are
/// `n_h * d_k`.
///
/// Every term is one tensor of `Qwen3.5-0.8B-Q4_K_M.gguf`, named here as the
/// file names it (`i` = `inner_size`, `s` = `state_size`, `k` = `conv_kernel`,
/// `h` = `group_count`):
///
/// | Tensor | Shape | Parameters |
/// |--------|-------|------------|
/// | `attn_qkv.weight` | `[d, 3*i]` | `3*d*i` |
/// | `attn_gate.weight` | `[d, i]` | `d*i` |
/// | `ssm_conv1d.weight` | `[k, 3*i]` | `3*k*i` |
/// | `ssm_alpha.weight`, `ssm_beta.weight` | `[d, h]` each | `2*d*h` |
/// | `ssm_a`, `ssm_dt.bias` | `[h]` each | `2*h` |
/// | `ssm_norm.weight` | `[s]` | `s` |
/// | `ssm_out.weight` | `[i, d]` | `i*d` |
///
/// `d_norm` is `2*d` (`attn_norm` + `post_attention_norm`) and `d_ffn` is the
/// same SwiGLU block as an attention layer — a DeltaNet layer differs only in
/// how it mixes tokens.
#[must_use]
pub fn gated_deltanet_layer_params(
size: &ModelSizeConfig,
constraints: &ModelConstraints,
shape: &DeltaNetShape,
) -> LayerParams {
let d = size.hidden_dim as u64;
let inner = shape.inner_size as u64;
let heads = shape.group_count as u64;
let qkv = d.saturating_mul(inner).saturating_mul(3);
let gate = d.saturating_mul(inner);
let conv = (shape.conv_kernel as u64)
.saturating_mul(inner)
.saturating_mul(3);
let alpha_beta = d.saturating_mul(heads).saturating_mul(2);
let per_head = heads.saturating_mul(2);
let out = inner.saturating_mul(d);

let d_attn = qkv
.saturating_add(gate)
.saturating_add(conv)
.saturating_add(alpha_beta)
.saturating_add(per_head)
.saturating_add(shape.state_size as u64)
.saturating_add(out);

LayerParams {
d_attn: projections.saturating_add(biases),
d_ffn: d.saturating_mul(d_ff).saturating_mul(matrices),
d_attn,
d_ffn: ffn_params(size, constraints),
d_norm: d.saturating_mul(2),
}
}

/// The per-layer input for a HYBRID model: Gated DeltaNet layers with a
/// softmax-attention layer every `full_attention_interval`-th position, the
/// last of each group.
///
/// Falls back to [`uniform_layers`] for any family that declares no DeltaNet
/// shape (every family but `qwen3_5`) or declares no schedule, so the answer
/// for a dense family is byte-for-byte what it was before #3346.
#[must_use]
pub fn hybrid_layers(size: &ModelSizeConfig, constraints: &ModelConstraints) -> Vec<LayerParams> {
let Some(shape) = constraints.deltanet else {
return uniform_layers(size, constraints);
};
if shape.full_attention_interval == 0 {
return uniform_layers(size, constraints);
}
let attention = attention_layer_params(size, constraints);
let deltanet = gated_deltanet_layer_params(size, constraints, &shape);
(0..size.num_layers)
.map(|i| {
if (i + 1) % shape.full_attention_interval == 0 {
attention
} else {
deltanet
}
})
.collect()
}

/// `L` copies of [`attention_layer_params`] — the per-layer input for a
/// homogeneous (non-hybrid) model of `size.num_layers` layers.
#[must_use]
Expand Down Expand Up @@ -166,14 +283,19 @@ pub fn uniform_layers(size: &ModelSizeConfig, constraints: &ModelConstraints) ->
///
/// # What this does NOT discharge
///
/// `QE2E-INV-001` wants `P(Qwen3.5-9B) ∈ [9.0B, 9.2B]`. Feeding this function
/// [`uniform_layers`] for the 9B variant yields ≈8.21B, because Qwen3.5 is
/// `hybrid_gated_deltanet`: three of every four layers are Gated DeltaNet, whose
/// `d_attn` covers conv, gate and state projections sized by `inner_size`,
/// `state_size`, `conv_kernel` and `group_count`. Those four keys exist in
/// `contracts/model-families/qwen3_5.yaml` but NOT in [`ModelConstraints`], so
/// the GDN `d_attn` cannot be derived from the config type as it stands. The
/// equation is implemented; the 9B invariant is not verified.
/// `QE2E-INV-001` wants `P(Qwen3.5-9B) ∈ [9.0B, 9.2B]`, and it is still NOT
/// discharged — but for a different reason than before #3346.
///
/// The arithmetic is now verified against a real file: fed the configuration of
/// `Qwen3.5-0.8B-Q4_K_M.gguf`, [`hybrid_layers`] + this function reproduce that
/// file's 320-tensor inventory EXACTLY (752,393,024 parameters). The Gated
/// DeltaNet shape reaches it through [`ModelConstraints::deltanet`].
///
/// Applying the same, now-falsified, arithmetic to the 9b variant of
/// `contracts/model-families/qwen3_5.yaml` gives **8,344,907,136** — 0.655B
/// below the range. The remaining gap is in the DESCRIPTOR, not here, and is
/// not something this function may paper over: see
/// `model_arithmetic_tests.rs::qwen35_9b_hybrid_layers_still_fall_short_of_the_invariant_range`.
#[must_use]
pub fn model_parameter_count(
size: &ModelSizeConfig,
Expand Down
Loading
Loading